224 lines
5.7 KiB
Go
224 lines
5.7 KiB
Go
package endpoint
|
|
|
|
import (
|
|
"encoding/json"
|
|
"net/http"
|
|
"os"
|
|
"path/filepath"
|
|
"regexp"
|
|
"sort"
|
|
"strconv"
|
|
"strings"
|
|
"sync"
|
|
"time"
|
|
"unicode/utf8"
|
|
|
|
"git.coffeylabs.org/coffey-labs/hotdog-cms/internal/site"
|
|
)
|
|
|
|
// Search answered here instead of in the browser: the visitor downloads ten
|
|
// results instead of the whole index. It reads the same search.json the build
|
|
// writes for in-browser search, so a site can offer either, or both.
|
|
//
|
|
// Scoring: a hit in the title outranks one in the terms (tags), which
|
|
// outranks hits in the body (counted, up to five), and a page matching every
|
|
// word of the query earns a bonus.
|
|
|
|
type searchEntry struct {
|
|
URL string `json:"u"`
|
|
Title string `json:"t"`
|
|
Date string `json:"d,omitempty"`
|
|
Terms [][2]string `json:"g,omitempty"`
|
|
Summary string `json:"s,omitempty"`
|
|
Text string `json:"x"`
|
|
|
|
lowTitle, lowTerms, lowText string
|
|
}
|
|
|
|
type searchIndex struct {
|
|
mod time.Time
|
|
entries []*searchEntry
|
|
}
|
|
|
|
type searcher struct {
|
|
mu sync.Mutex
|
|
cache map[string]*searchIndex // by file
|
|
}
|
|
|
|
func newSearcher() *searcher { return &searcher{cache: map[string]*searchIndex{}} }
|
|
|
|
var collectionRe = regexp.MustCompile(`^[a-z0-9][a-z0-9_-]{0,63}$`)
|
|
|
|
// index loads a collection's search.json, again only when the file changes.
|
|
func (sr *searcher) index(file string) ([]*searchEntry, error) {
|
|
fi, err := os.Stat(file)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
sr.mu.Lock()
|
|
defer sr.mu.Unlock()
|
|
if ix, ok := sr.cache[file]; ok && ix.mod.Equal(fi.ModTime()) {
|
|
return ix.entries, nil
|
|
}
|
|
raw, err := os.ReadFile(file)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
var entries []*searchEntry
|
|
if err := json.Unmarshal(raw, &entries); err != nil {
|
|
return nil, err
|
|
}
|
|
for _, e := range entries {
|
|
e.lowTitle = strings.ToLower(e.Title)
|
|
var t []string
|
|
for _, g := range e.Terms {
|
|
t = append(t, g[0])
|
|
}
|
|
e.lowTerms = strings.ToLower(strings.Join(t, " "))
|
|
e.lowText = strings.ToLower(e.Text)
|
|
}
|
|
sr.cache[file] = &searchIndex{mod: fi.ModTime(), entries: entries}
|
|
return entries, nil
|
|
}
|
|
|
|
var querySplit = regexp.MustCompile(`[^\p{L}\p{N}+#._-]+`)
|
|
|
|
// maxTerms bounds the work one search does: each term is looked for in
|
|
// every page.
|
|
const maxTerms = 8
|
|
|
|
func queryTerms(q string) []string {
|
|
var out []string
|
|
seen := map[string]bool{}
|
|
for _, t := range querySplit.Split(strings.ToLower(strings.TrimSpace(q)), -1) {
|
|
if utf8.RuneCountInString(t) >= 2 && !seen[t] {
|
|
seen[t] = true
|
|
out = append(out, t)
|
|
if len(out) == maxTerms {
|
|
break
|
|
}
|
|
}
|
|
}
|
|
return out
|
|
}
|
|
|
|
type searchResult struct {
|
|
URL string `json:"u"`
|
|
Title string `json:"t"`
|
|
Date string `json:"d,omitempty"`
|
|
Terms [][2]string `json:"g,omitempty"`
|
|
Snippet string `json:"snippet"`
|
|
score int
|
|
}
|
|
|
|
func search(entries []*searchEntry, q string, limit int) []searchResult {
|
|
terms := queryTerms(q)
|
|
if len(terms) == 0 {
|
|
return nil
|
|
}
|
|
var out []searchResult
|
|
for _, e := range entries {
|
|
score, matched := 0, 0
|
|
for _, t := range terms {
|
|
hit := 0
|
|
if strings.Contains(e.lowTitle, t) {
|
|
hit += 12
|
|
}
|
|
if strings.Contains(e.lowTerms, t) {
|
|
hit += 6
|
|
}
|
|
if n := strings.Count(e.lowText, t); n > 0 {
|
|
hit += min(5, n)
|
|
}
|
|
if hit > 0 {
|
|
matched++
|
|
score += hit
|
|
}
|
|
}
|
|
if matched == 0 {
|
|
continue
|
|
}
|
|
if matched == len(terms) {
|
|
score += 20
|
|
}
|
|
out = append(out, searchResult{URL: e.URL, Title: e.Title, Date: e.Date, Terms: e.Terms, Snippet: e.Text, score: score})
|
|
}
|
|
sort.SliceStable(out, func(i, j int) bool {
|
|
if out[i].score != out[j].score {
|
|
return out[i].score > out[j].score
|
|
}
|
|
return out[i].Date > out[j].Date
|
|
})
|
|
if len(out) > limit {
|
|
out = out[:limit]
|
|
}
|
|
// Snippets are the costly part (each renders the page's Markdown), so
|
|
// only the results kept get one.
|
|
for i := range out {
|
|
out[i].Snippet = snippet(out[i].Snippet, terms)
|
|
}
|
|
return out
|
|
}
|
|
|
|
// snippet is a window of plain text around the first word of the query found.
|
|
func snippet(md string, terms []string) string {
|
|
plain := []rune(site.PlainMarkdown(md))
|
|
lower := []rune(strings.ToLower(string(plain)))
|
|
at := -1
|
|
for _, t := range terms {
|
|
if i := strings.Index(string(lower), t); i >= 0 {
|
|
at = utf8.RuneCountInString(string(lower)[:i])
|
|
break
|
|
}
|
|
}
|
|
if at < 0 {
|
|
if len(plain) > 200 {
|
|
return string(plain[:200]) + "…"
|
|
}
|
|
return string(plain)
|
|
}
|
|
start := max(0, at-66)
|
|
end := min(len(plain), start+200)
|
|
out := strings.TrimSpace(string(plain[start:end])) + "…"
|
|
if start > 0 {
|
|
out = "…" + out
|
|
}
|
|
return out
|
|
}
|
|
|
|
// handleSearch answers GET /_hotdog/search/<collection>?q=...&page=N as JSON.
|
|
func (s *Server) handleSearch(w http.ResponseWriter, r *http.Request, rt *siteRuntime, collection string) {
|
|
if rt.cfg.Public == "" || !collectionRe.MatchString(collection) {
|
|
http.NotFound(w, r)
|
|
return
|
|
}
|
|
if !s.searchLimit.allow(limitKey(s.clientIP(r)), s.now()) {
|
|
http.Error(w, "too many searches; wait a minute", http.StatusTooManyRequests)
|
|
return
|
|
}
|
|
entries, err := s.search.index(filepath.Join(rt.cfg.Public, collection, "search.json"))
|
|
if err != nil {
|
|
http.Error(w, "no search for this collection", http.StatusNotFound)
|
|
return
|
|
}
|
|
q := r.URL.Query().Get("q")
|
|
if r := []rune(q); len(r) > 120 {
|
|
q = string(r[:120])
|
|
}
|
|
per := 12
|
|
if n, err := strconv.Atoi(r.URL.Query().Get("per_page")); err == nil && n > 0 && n <= 50 {
|
|
per = n
|
|
}
|
|
results := search(entries, q, 40)
|
|
pages := max(1, (len(results)+per-1)/per)
|
|
page := 1
|
|
if n, err := strconv.Atoi(r.URL.Query().Get("page")); err == nil && n > 1 {
|
|
page = min(n, pages)
|
|
}
|
|
from := min(len(results), (page-1)*per)
|
|
to := min(len(results), page*per)
|
|
writeJSON(w, http.StatusOK, map[string]any{
|
|
"query": q, "terms": queryTerms(q), "total": len(results), "page": page, "pages": pages, "results": results[from:to],
|
|
})
|
|
}
|