Files
datascape/search.go
T
luxickandClaude Opus 5.5 4365fa2567 Render the folder tree on the server
The sidebar tree is part of the page (chain to the current page
expanded, active row marked), so it no longer paints late and the
sessionStorage snapshot is gone; only the rail's scroll position is
kept. Folders are native <details> that lazy-load their children as
HTML via htmx. The tree picker hosts the same server-rendered rows with
radio inputs, so ?tree returns HTML and the JSON listing is removed.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
2026-09-30 16:51:20 +02:00

635 lines
18 KiB
Go

package main
import (
"encoding/json"
"html/template"
"io/fs"
"log"
"net/http"
"net/url"
"path/filepath"
"sort"
"strings"
"sync"
"time"
"unicode"
"unicode/utf8"
)
// searchSectionCap bounds how many rows each results section renders. The
// section header still reports the true total so the user knows more exist.
const searchSectionCap = 10
// exactNameScore is the score scoreName assigns to a whole-name exact match.
// handleSearch peels these into the promoted "Exact Match" section.
const exactNameScore = 1000
type searchResult struct {
Name string
URL string
Path string
Score int
// Meta is the formatted "size · date" line for file results; empty for
// page results.
Meta string
}
type searchPageData struct {
Title string
EditMode bool
Query string
// Exact holds page(s) whose name equals the query exactly, promoted to
// their own section above the fuzzy matches.
Exact []searchResult
Pages []searchResult
Files []searchResult
// PageTotal/FileTotal are the true match totals; Pages/Files are capped at
// searchSectionCap for rendering.
PageTotal int
FileTotal int
IndexBuiltAt time.Time
Tree template.HTML // sidebar folder tree
renderTimer
}
// indexEntry is the shared scoreable core of both folder and file index
// entries: a forward-slash relative path plus its pre-tokenized basename so
// the per-query scoring loop avoids redoing the lowercasing and tokenization
// on every request.
type indexEntry struct {
Path string
NameLower string
NameTokens []string
}
// folderEntry is a single indexed directory. It is exactly an indexEntry; the
// alias keeps the existing folder-index code readable while letting files and
// folders share the scoring loop.
type folderEntry = indexEntry
// fileEntry is a single indexed file: its scoreable core plus the size/modtime
// captured during the walk so results can show listing-parity metadata without
// a second stat.
type fileEntry struct {
indexEntry
Size int64
ModTime time.Time
}
// folderIndex holds the in-memory directory index used by search. Writers
// always replace the entries slice wholesale so a reader that snapshots the
// header under RLock can score without holding the lock.
var folderIndex struct {
sync.RWMutex
entries []folderEntry
builtAt time.Time
buildMu sync.Mutex
ready chan struct{}
}
// fileIndex mirrors folderIndex for files. It is held separately so file
// volume can't perturb the page index, and shares folderIndex.buildMu since
// both are populated by the same single-pass walk. It is refreshed only by the
// full rebuild (startup / ticker / manual), never by the incremental folder
// hooks — file freshness on disk lags until the next rebuild.
var fileIndex struct {
sync.RWMutex
entries []fileEntry
builtAt time.Time
ready chan struct{}
}
func init() {
folderIndex.ready = make(chan struct{})
fileIndex.ready = make(chan struct{})
}
// handleSearch renders the search results page for the query in
// r.URL.Query().Get("q"). Only invoked when path is "/" and "q" is present.
func (h *handler) handleSearch(w http.ResponseWriter, r *http.Request) {
query := strings.TrimSpace(r.URL.Query().Get("q"))
pages, builtAt := searchWiki(query)
files := searchFiles(query)
// searchWiki sorts by score desc, so exact-name matches (score 1000) are
// at the front; peel them into their own section and drop them from the
// fuzzy page list so they aren't shown twice.
var exact []searchResult
for len(pages) > 0 && pages[0].Score == exactNameScore {
exact = append(exact, pages[0])
pages = pages[1:]
}
title := "Search"
if query != "" {
title = "Search: " + query
}
data := searchPageData{
Title: title,
Query: query,
Exact: exact,
Pages: capResults(pages),
Files: capResults(files),
PageTotal: len(pages),
FileTotal: len(files),
IndexBuiltAt: builtAt,
Tree: sidebarTree(h.root, "/"),
}
w.Header().Set("Content-Type", "text/html; charset=utf-8")
data.renderTimer = renderTimer{requestStart(r)}
if err := searchTmpl.ExecuteTemplate(w, "layout", data); err != nil {
log.Printf("search template error: %v", err)
}
}
// capResults truncates a section to searchSectionCap rows for rendering.
func capResults(results []searchResult) []searchResult {
if len(results) > searchSectionCap {
return results[:searchSectionCap]
}
return results
}
// scoredEntry pairs a matched index entry with its score before it is turned
// into a UI searchResult.
type scoredEntry[T any] struct {
entry T
score int
}
// scoreEntries scores every entry against query, drops non-matches, and returns
// the survivors sorted by score (desc), then path depth (asc), then basename
// (asc). core extracts the shared scoreable fields so the folder and file
// indexes reuse one loop. fuzzy toggles the levenshtein fallback (pages yes,
// files no). Returns nil for an empty/tokenless query.
func scoreEntries[T any](entries []T, query string, fuzzy bool, core func(T) indexEntry) []scoredEntry[T] {
qLower := strings.ToLower(query)
qTokens := tokenize(qLower)
if len(qTokens) == 0 {
return nil
}
var out []scoredEntry[T]
for _, e := range entries {
c := core(e)
score := scoreName(c.NameLower, c.NameTokens, qLower, qTokens, fuzzy)
if score == 0 {
continue
}
out = append(out, scoredEntry[T]{entry: e, score: score})
}
sort.SliceStable(out, func(i, j int) bool {
if out[i].score != out[j].score {
return out[i].score > out[j].score
}
ci, cj := core(out[i].entry), core(out[j].entry)
di, dj := strings.Count(ci.Path, "/"), strings.Count(cj.Path, "/")
if di != dj {
return di < dj
}
return ci.NameLower < cj.NameLower
})
return out
}
// searchWiki scores the cached folder index against query. Blocks on the
// initial build so the very first request after startup serves correct
// results rather than an empty list. Returns the snapshot's builtAt so the
// UI can show how fresh the index is.
func searchWiki(query string) ([]searchResult, time.Time) {
<-folderIndex.ready
folderIndex.RLock()
entries := folderIndex.entries
builtAt := folderIndex.builtAt
folderIndex.RUnlock()
if query == "" {
return nil, builtAt
}
scored := scoreEntries(entries, query, true, func(e folderEntry) indexEntry { return e })
results := make([]searchResult, 0, len(scored))
for _, s := range scored {
results = append(results, searchResult{
Name: filepath.Base(s.entry.Path),
URL: "/" + s.entry.Path + "/",
Path: s.entry.Path,
Score: s.score,
})
}
return results, builtAt
}
// searchFiles scores the cached file index against query, matching on filename
// only. Blocks on the initial build so the first request after startup doesn't
// serve an empty Files section while the walk is still running.
func searchFiles(query string) []searchResult {
<-fileIndex.ready
fileIndex.RLock()
entries := fileIndex.entries
fileIndex.RUnlock()
if query == "" {
return nil
}
scored := scoreEntries(entries, query, false, func(e fileEntry) indexEntry { return e.indexEntry })
results := make([]searchResult, 0, len(scored))
for _, s := range scored {
p := s.entry.Path
results = append(results, searchResult{
Name: filepath.Base(p),
URL: fileURL(p),
Path: p,
Score: s.score,
Meta: formatSize(s.entry.Size) + " · " + s.entry.ModTime.Format("2006-01-02"),
})
}
return results
}
// fileURL builds the browser URL for a file's forward-slash relative path,
// percent-escaping each segment so spaces/umlauts/&c. survive round-tripping
// through the companion's wikiPathFromHref decode.
func fileURL(relPath string) string {
parts := strings.Split(relPath, "/")
for i, p := range parts {
parts[i] = url.PathEscape(p)
}
return "/" + strings.Join(parts, "/")
}
// scoreName ranks how well nameLower matches the query. Whole-name exact
// match dominates; otherwise score is the sum of each token's best match
// against the words in the name. nameTokens is precomputed by the index. fuzzy
// enables the levenshtein near-match fallback; the file index passes false so
// large file volumes don't pay the edit-distance cost per query.
func scoreName(nameLower string, nameTokens []string, qLower string, qTokens []string, fuzzy bool) int {
if nameLower == qLower {
return exactNameScore
}
score := 0
for _, qt := range qTokens {
best := 0
for _, w := range nameTokens {
switch {
case w == qt:
if best < 100 {
best = 100
}
case strings.HasPrefix(w, qt):
if best < 50 {
best = 50
}
case strings.Contains(w, qt):
if best < 20 {
best = 20
}
case fuzzy && withinTwoEdits(w, qt):
if best < 5 {
best = 5
}
}
}
score += best
}
return score
}
// handleSearchSuggest serves the JSON typeahead for the header dropdown and
// the editor's link picker. Caps results at 5; reports total so the UI can
// surface a "show all" footer when more matches exist. Empty/whitespace query
// is a no-op (200 with empty results), not a 400 — every keystroke fires this.
func (h *handler) handleSearchSuggest(w http.ResponseWriter, r *http.Request) {
if !h.checkAuth(w, r) {
return
}
query := strings.TrimSpace(r.URL.Query().Get("q"))
type suggestResult struct {
Name string `json:"name"`
Path string `json:"path"`
URL string `json:"url"`
}
type suggestResp struct {
Query string `json:"query"`
Results []suggestResult `json:"results"`
Total int `json:"total"`
}
resp := suggestResp{Query: query, Results: []suggestResult{}}
if query != "" {
all, _ := searchWiki(query)
resp.Total = len(all)
limit := 5
if len(all) < limit {
limit = len(all)
}
for i := 0; i < limit; i++ {
resp.Results = append(resp.Results, suggestResult{
Name: all[i].Name,
Path: all[i].Path,
URL: all[i].URL,
})
}
}
w.Header().Set("Content-Type", "application/json; charset=utf-8")
if err := json.NewEncoder(w).Encode(resp); err != nil {
log.Printf("search suggest encode error: %v", err)
}
}
// handleReindex rebuilds the folder index synchronously and returns 204.
// The frontend reloads the page on success. Serialized via buildMu so a
// double-click waits rather than running two walks in parallel.
func (h *handler) handleReindex(w http.ResponseWriter, r *http.Request) {
if !h.checkAuth(w, r) {
return
}
if r.Method != http.MethodPost {
http.Error(w, "method not allowed", http.StatusMethodNotAllowed)
return
}
rebuildFolderIndex(h.root)
w.WriteHeader(http.StatusNoContent)
}
// buildIndexes walks root once and returns fresh folder and file entries. The
// single pass avoids a second full traversal on the ARMv7 NAS. Hidden
// directories (`.git`, `.thumbs`, …) are pruned and hidden files skipped; the
// root itself and every `index.md` (page content, not a browsable file) are
// excluded.
func buildIndexes(root string) ([]folderEntry, []fileEntry) {
walkRoot := resolveWalkRoot(root)
var folders []folderEntry
var files []fileEntry
_ = filepath.WalkDir(walkRoot, func(fsPath string, d fs.DirEntry, err error) error {
if err != nil {
return nil
}
if skip, walkErr := hiddenSkip(fsPath, walkRoot, d); skip {
return walkErr
}
rel, relErr := filepath.Rel(walkRoot, fsPath)
if relErr != nil {
return nil
}
relSlash := filepath.ToSlash(rel)
if d.IsDir() {
if fsPath != walkRoot {
folders = append(folders, newFolderEntry(relSlash))
}
return nil
}
if d.Name() == "index.md" {
return nil
}
info, infoErr := d.Info()
if infoErr != nil {
return nil
}
files = append(files, newFileEntry(relSlash, info.Size(), info.ModTime()))
return nil
})
return folders, files
}
// newFolderEntry builds a folderEntry from a forward-slash relative path,
// computing the lowercased basename and its tokens once so search scoring
// doesn't have to redo it per query.
func newFolderEntry(relPath string) folderEntry {
return newIndexEntry(relPath)
}
// newFileEntry builds a fileEntry, capturing the walk-time size/modtime so
// results show listing-parity metadata without a second stat.
func newFileEntry(relPath string, size int64, modTime time.Time) fileEntry {
return fileEntry{
indexEntry: newIndexEntry(relPath),
Size: size,
ModTime: modTime,
}
}
// newIndexEntry precomputes the lowercased basename and its tokens for the
// per-query scoring loop.
func newIndexEntry(relPath string) indexEntry {
name := relPath
if i := strings.LastIndex(relPath, "/"); i >= 0 {
name = relPath[i+1:]
}
nameLower := strings.ToLower(name)
return indexEntry{
Path: relPath,
NameLower: nameLower,
NameTokens: tokenize(nameLower),
}
}
// rebuildFolderIndex walks root once and atomically replaces both the folder
// and file indexes. buildMu serializes overlapping rebuilds (manual + ticker +
// startup) so the WalkDir cost is paid once even under contention.
func rebuildFolderIndex(root string) {
folderIndex.buildMu.Lock()
defer folderIndex.buildMu.Unlock()
folders, files := buildIndexes(root)
now := time.Now()
folderIndex.Lock()
folderIndex.entries = folders
folderIndex.builtAt = now
folderIndex.Unlock()
fileIndex.Lock()
fileIndex.entries = files
fileIndex.builtAt = now
fileIndex.Unlock()
}
// folderIndexAdd appends relPath as a new entry. No-op for empty/root paths.
func folderIndexAdd(relPath string) {
relPath = strings.Trim(relPath, "/")
if relPath == "" {
return
}
folderIndex.Lock()
folderIndex.entries = append(folderIndex.entries, newFolderEntry(relPath))
folderIndex.Unlock()
}
// folderIndexRemoveSubtree drops the entry at relPath plus every descendant.
// Replaces the slice rather than mutating in place so any in-flight search
// reader keeps a valid snapshot.
func folderIndexRemoveSubtree(relPath string) {
relPath = strings.Trim(relPath, "/")
if relPath == "" {
return
}
prefix := relPath + "/"
folderIndex.Lock()
defer folderIndex.Unlock()
old := folderIndex.entries
out := make([]folderEntry, 0, len(old))
for _, e := range old {
if e.Path == relPath || strings.HasPrefix(e.Path, prefix) {
continue
}
out = append(out, e)
}
folderIndex.entries = out
}
// folderIndexRenameSubtree rewrites the path prefix for every entry under
// oldRel. The renamed root entry's basename may have changed so its
// NameLower/NameTokens are recomputed; descendants keep their basenames.
func folderIndexRenameSubtree(oldRel, newRel string) {
oldRel = strings.Trim(oldRel, "/")
newRel = strings.Trim(newRel, "/")
if oldRel == "" || newRel == "" {
return
}
oldPrefix := oldRel + "/"
folderIndex.Lock()
defer folderIndex.Unlock()
old := folderIndex.entries
out := make([]folderEntry, len(old))
for i, e := range old {
switch {
case e.Path == oldRel:
out[i] = newFolderEntry(newRel)
case strings.HasPrefix(e.Path, oldPrefix):
out[i] = folderEntry{
Path: newRel + "/" + strings.TrimPrefix(e.Path, oldPrefix),
NameLower: e.NameLower,
NameTokens: e.NameTokens,
}
default:
out[i] = e
}
}
folderIndex.entries = out
}
// folderIndexMergeSubtree updates the index after a merge move: it drops the
// source root entry (that folder is gone) and rewrites every descendant's
// prefix to live under newRel. Unlike folderIndexRenameSubtree it does not add
// a newRel entry, since the destination folder already exists in the index.
func folderIndexMergeSubtree(oldRel, newRel string) {
oldRel = strings.Trim(oldRel, "/")
newRel = strings.Trim(newRel, "/")
if oldRel == "" || newRel == "" {
return
}
oldPrefix := oldRel + "/"
folderIndex.Lock()
defer folderIndex.Unlock()
old := folderIndex.entries
out := make([]folderEntry, 0, len(old))
for _, e := range old {
switch {
case e.Path == oldRel:
continue
case strings.HasPrefix(e.Path, oldPrefix):
out = append(out, folderEntry{
Path: newRel + "/" + strings.TrimPrefix(e.Path, oldPrefix),
NameLower: e.NameLower,
NameTokens: e.NameTokens,
})
default:
out = append(out, e)
}
}
folderIndex.entries = out
}
// resolveWalkRoot resolves symlinks so WalkDir descends into the real tree
// even when the configured wiki root is itself a symlink (as on the NAS).
func resolveWalkRoot(root string) string {
if r, err := filepath.EvalSymlinks(root); err == nil {
return r
}
return root
}
// hiddenSkip handles dotfile/dot-dir entries during a WalkDir. It returns
// (skipped, walkErr): skipped=true means the caller should `return walkErr`
// to either prune the subtree (hidden dir) or move past the entry (hidden
// file). When skipped=false the entry should be processed normally.
func hiddenSkip(fsPath, walkRoot string, d fs.DirEntry) (bool, error) {
if !strings.HasPrefix(d.Name(), ".") {
return false, nil
}
if d.IsDir() && fsPath != walkRoot {
return true, filepath.SkipDir
}
return true, nil
}
// tokenize splits s into lowercase word tokens, breaking on any rune that is
// not a letter or digit. Unicode-aware so umlauts etc. survive intact.
func tokenize(s string) []string {
var tokens []string
var b strings.Builder
for _, r := range s {
if unicode.IsLetter(r) || unicode.IsDigit(r) {
b.WriteRune(unicode.ToLower(r))
continue
}
if b.Len() > 0 {
tokens = append(tokens, b.String())
b.Reset()
}
}
if b.Len() > 0 {
tokens = append(tokens, b.String())
}
return tokens
}
// withinTwoEdits reports whether a and b are at most 2 edits apart. Lengths
// differing by more than 2 rule that out without running levenshtein, which
// skips nearly every word pair in the per-query scoring loop.
func withinTwoEdits(a, b string) bool {
d := utf8.RuneCountInString(a) - utf8.RuneCountInString(b)
if d < -2 || d > 2 {
return false
}
return levenshtein(a, b) <= 2
}
// levenshtein returns the edit distance between a and b. Operates on runes so
// multi-byte characters count as one edit.
func levenshtein(a, b string) int {
ar, br := []rune(a), []rune(b)
if len(ar) == 0 {
return len(br)
}
if len(br) == 0 {
return len(ar)
}
prev := make([]int, len(br)+1)
curr := make([]int, len(br)+1)
for j := range prev {
prev[j] = j
}
for i := 1; i <= len(ar); i++ {
curr[0] = i
for j := 1; j <= len(br); j++ {
cost := 1
if ar[i-1] == br[j-1] {
cost = 0
}
del := prev[j] + 1
ins := curr[j-1] + 1
sub := prev[j-1] + cost
curr[j] = min3(del, ins, sub)
}
prev, curr = curr, prev
}
return prev[len(br)]
}
func min3(a, b, c int) int {
m := a
if b < m {
m = b
}
if c < m {
m = c
}
return m
}