package endpoint import ( "encoding/json" "net/http" "os" "path/filepath" "regexp" "sort" "strconv" "strings" "sync" "time" "unicode/utf8" "git.coffeylabs.org/coffey-labs/hotdog-cms/internal/site" ) // Search answered here instead of in the browser: the visitor downloads ten // results instead of the whole index. It reads the same search.json the build // writes for in-browser search, so a site can offer either, or both. // // Scoring: a hit in the title outranks one in the terms (tags), which // outranks hits in the body (counted, up to five), and a page matching every // word of the query earns a bonus. type searchEntry struct { URL string `json:"u"` Title string `json:"t"` Date string `json:"d,omitempty"` Terms [][2]string `json:"g,omitempty"` Summary string `json:"s,omitempty"` Text string `json:"x"` lowTitle, lowTerms, lowText string } type searchIndex struct { mod time.Time entries []*searchEntry } type searcher struct { mu sync.Mutex cache map[string]*searchIndex // by file } func newSearcher() *searcher { return &searcher{cache: map[string]*searchIndex{}} } var collectionRe = regexp.MustCompile(`^[a-z0-9][a-z0-9_-]{0,63}$`) // index loads a collection's search.json, again only when the file changes. func (sr *searcher) index(file string) ([]*searchEntry, error) { fi, err := os.Stat(file) if err != nil { return nil, err } sr.mu.Lock() defer sr.mu.Unlock() if ix, ok := sr.cache[file]; ok && ix.mod.Equal(fi.ModTime()) { return ix.entries, nil } raw, err := os.ReadFile(file) if err != nil { return nil, err } var entries []*searchEntry if err := json.Unmarshal(raw, &entries); err != nil { return nil, err } for _, e := range entries { e.lowTitle = strings.ToLower(e.Title) var t []string for _, g := range e.Terms { t = append(t, g[0]) } e.lowTerms = strings.ToLower(strings.Join(t, " ")) e.lowText = strings.ToLower(e.Text) } sr.cache[file] = &searchIndex{mod: fi.ModTime(), entries: entries} return entries, nil } var querySplit = regexp.MustCompile(`[^\p{L}\p{N}+#._-]+`) // maxTerms bounds the work one search does: each term is looked for in // every page. const maxTerms = 8 func queryTerms(q string) []string { var out []string seen := map[string]bool{} for _, t := range querySplit.Split(strings.ToLower(strings.TrimSpace(q)), -1) { if utf8.RuneCountInString(t) >= 2 && !seen[t] { seen[t] = true out = append(out, t) if len(out) == maxTerms { break } } } return out } type searchResult struct { URL string `json:"u"` Title string `json:"t"` Date string `json:"d,omitempty"` Terms [][2]string `json:"g,omitempty"` Snippet string `json:"snippet"` score int } func search(entries []*searchEntry, q string, limit int) []searchResult { terms := queryTerms(q) if len(terms) == 0 { return nil } var out []searchResult for _, e := range entries { score, matched := 0, 0 for _, t := range terms { hit := 0 if strings.Contains(e.lowTitle, t) { hit += 12 } if strings.Contains(e.lowTerms, t) { hit += 6 } if n := strings.Count(e.lowText, t); n > 0 { hit += min(5, n) } if hit > 0 { matched++ score += hit } } if matched == 0 { continue } if matched == len(terms) { score += 20 } out = append(out, searchResult{URL: e.URL, Title: e.Title, Date: e.Date, Terms: e.Terms, Snippet: e.Text, score: score}) } sort.SliceStable(out, func(i, j int) bool { if out[i].score != out[j].score { return out[i].score > out[j].score } return out[i].Date > out[j].Date }) if len(out) > limit { out = out[:limit] } // Snippets are the costly part (each renders the page's Markdown), so // only the results kept get one. for i := range out { out[i].Snippet = snippet(out[i].Snippet, terms) } return out } // snippet is a window of plain text around the first word of the query found. func snippet(md string, terms []string) string { plain := []rune(site.PlainMarkdown(md)) lower := []rune(strings.ToLower(string(plain))) at := -1 for _, t := range terms { if i := strings.Index(string(lower), t); i >= 0 { at = utf8.RuneCountInString(string(lower)[:i]) break } } if at < 0 { if len(plain) > 200 { return string(plain[:200]) + "…" } return string(plain) } start := max(0, at-66) end := min(len(plain), start+200) out := strings.TrimSpace(string(plain[start:end])) + "…" if start > 0 { out = "…" + out } return out } // handleSearch answers GET /_hotdog/search/?q=...&page=N as JSON. func (s *Server) handleSearch(w http.ResponseWriter, r *http.Request, rt *siteRuntime, collection string) { if rt.cfg.Public == "" || !collectionRe.MatchString(collection) { http.NotFound(w, r) return } if !s.searchLimit.allow(limitKey(s.clientIP(r)), s.now()) { http.Error(w, "too many searches; wait a minute", http.StatusTooManyRequests) return } entries, err := s.search.index(filepath.Join(rt.cfg.Public, collection, "search.json")) if err != nil { http.Error(w, "no search for this collection", http.StatusNotFound) return } q := r.URL.Query().Get("q") if r := []rune(q); len(r) > 120 { q = string(r[:120]) } per := 12 if n, err := strconv.Atoi(r.URL.Query().Get("per_page")); err == nil && n > 0 && n <= 50 { per = n } results := search(entries, q, 40) pages := max(1, (len(results)+per-1)/per) page := 1 if n, err := strconv.Atoi(r.URL.Query().Get("page")); err == nil && n > 1 { page = min(n, pages) } from := min(len(results), (page-1)*per) to := min(len(results), page*per) writeJSON(w, http.StatusOK, map[string]any{ "query": q, "terms": queryTerms(q), "total": len(results), "page": page, "pages": pages, "results": results[from:to], }) }