package web
import (
"html"
"html/template"
"regexp"
"strings"
"time"
"unicode/utf8"
"github.com/microcosm-cc/bluemonday"
"github.com/gramanas/blogspace/internal/store"
)
const (
searchPerPage = 20
maxSearchRunes = 100 // longer queries are cut; nobody types more on purpose
maxSearchWords = 8 // more ".*" joins only make the regex scan dearer, never the answer better
snippetContext = 80 // runes kept on each side of the match
searchDeadline = 5 * time.Second // the scan is unindexed; past this it is abuse or a blog too big for it
)
// searchPattern turns what the reader typed into the regular expression both
// Postgres (~*) and Go run: each word literal, the spaces between them "anything
// in between", so "go tem" finds "Go templates". Blank → "".
func searchPattern(q string) string {
words := strings.Fields(q)
if len(words) > maxSearchWords {
words = words[:maxSearchWords]
}
for i, w := range words {
words[i] = regexp.QuoteMeta(w)
}
return strings.Join(words, ".*")
}
// searchHit is one result: the post plus the excerpt the list shows.
type searchHit struct {
store.Post
Snippet template.HTML
}
var stripTags = bluemonday.StrictPolicy()
// snippetSource is the text a post's snippet is cut from: the Markdown as
// written, or an HTML post with its tags stripped so the excerpt reads as
// prose. (Postgres still matches against the source, so a query can hit a
// tag or attribute name in an HTML post; the snippet then shows the text
// nearest to it.)
func snippetSource(p *store.Post) string {
if p.Format == store.FormatHTML {
return html.UnescapeString(stripTags.Sanitize(p.BodyMD))
}
return p.BodyMD
}
// searchSnippet is a short piece of the body around the first match, with the
// match marked; when only the title matched it is the body's start.
func searchSnippet(body string, re *regexp.Regexp) template.HTML {
text := strings.Join(strings.Fields(body), " ")
loc := re.FindStringIndex(text)
if loc == nil {
loc = []int{0, 0}
}
before, match, after := text[:loc[0]], text[loc[0]:loc[1]], text[loc[1]:]
for n := utf8.RuneCountInString(before); n > snippetContext; n-- { // rune by rune: never cut inside one
_, size := utf8.DecodeRuneInString(before)
before = before[size:]
}
if len(before) < loc[0] {
before = "…" + before
}
if utf8.RuneCountInString(after) > snippetContext {
i := 0
for n := 0; n < snippetContext; n++ {
_, size := utf8.DecodeRuneInString(after[i:])
i += size
}
after = after[:i] + "…"
}
out := template.HTMLEscapeString(before)
if match != "" {
out += "" + template.HTMLEscapeString(match) + ""
}
return template.HTML(out + template.HTMLEscapeString(after))
}