diff options
| author | grm <grm@eyesin.space> | 2026-09-18 13:50:35 +0300 |
|---|---|---|
| committer | grm <grm@eyesin.space> | 2026-09-18 13:50:42 +0300 |
| commit | c3026c34b042cc044cddfc5674d5f5ad69bb845d (patch) | |
| tree | 7111e8fc310b193c510110e813befb00695a7597 | |
| parent | c47397ac1e2ceafafe2be3cdec86366dd396ed6f (diff) | |
| download | blogspace-c3026c34b042cc044cddfc5674d5f5ad69bb845d.tar.gz blogspace-c3026c34b042cc044cddfc5674d5f5ad69bb845d.tar.bz2 blogspace-c3026c34b042cc044cddfc5674d5f5ad69bb845d.zip | |
Security: Throttle search, cap its words and give the query a deadline
/search runs an unindexed regular-expression scan over every published
post, built from up to fifty ".*"-joined words, for anyone who asks —
the cheapest way for a bot to keep Postgres busy. Queries are now cut
at eight words (more never improve the answer), each address gets
thirty searches and then thirty a minute, and the statement is
cancelled after five seconds; a timeout reads as no results and is
logged, rather than a 500.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01Sd8UPWrvyYCLj97JexNw3A
| -rw-r--r-- | AGENTS.md | 8 | ||||
| -rw-r--r-- | internal/web/handlers_blog.go | 14 | ||||
| -rw-r--r-- | internal/web/search.go | 10 | ||||
| -rw-r--r-- | internal/web/search_test.go | 11 | ||||
| -rw-r--r-- | internal/web/server.go | 5 |
5 files changed, 36 insertions, 12 deletions
@@ -143,7 +143,7 @@ internal/web/ server.go (host router, middleware, render helpers) - **Hardening** (`web/ratelimit.go`): a per-key token bucket throttles the anonymous endpoints worth abusing — `loginLimit` on `POST /webadmin`, keyed by client address *and* by lowercased username (10 at once, then 10 a - minute each), answered with 429 by `s.throttle` and a log line; failed + minute each), `searchLimit` on `GET /search` per address — answered with 429 by `s.throttle` and a log line; failed logins are logged with the username and address, and the login body is capped at `maxLoginBody` (64 KB) since it is the one POST outside `guardPOST`. `s.clientIP` is the peer address, or the last @@ -481,7 +481,11 @@ internal/web/ server.go (host router, middleware, render helpers) are literal, case does not matter (`~*` / `(?is)`) and spaces mean "anything in between", in order. `SearchPublishedPosts` matches it against `title || '\n' || body_md` of published posts from every page, 20 per page - (`searchPerPage`); queries are cut at 100 runes. Results are title, date + (`searchPerPage`); queries are cut at 100 runes and 8 words + (`maxSearchWords`). The scan is unindexed, so the handler is throttled + per client address (`searchLimit`, 30 then 30 a minute) and the query + runs under `searchDeadline` (5 s): a timeout is logged and shown as no + results, not a 500. Results are title, date and `searchSnippet` (the Markdown around the first match, escaped, the match in `<mark>`; the body's start when only the title matched). A blank or unmatched query is a normal 200, not a 404; the pager is the page's diff --git a/internal/web/handlers_blog.go b/internal/web/handlers_blog.go index f138c58..55cec2e 100644 --- a/internal/web/handlers_blog.go +++ b/internal/web/handlers_blog.go @@ -1,6 +1,7 @@ package web import ( + "context" "encoding/xml" "errors" "net/http" @@ -159,6 +160,9 @@ func (s *Server) handleBlogTag(w http.ResponseWriter, r *http.Request) { // from every page of the blog. A blank query shows just the box; no matches // is an ordinary result, not a 404. func (s *Server) handleBlogSearch(w http.ResponseWriter, r *http.Request) { + if !s.throttle(w, r, s.searchLimit, s.clientIP(r)) { + return + } q := strings.TrimSpace(r.URL.Query().Get("q")) if utf8.RuneCountInString(q) > maxSearchRunes { q = string([]rune(q)[:maxSearchRunes]) @@ -172,7 +176,15 @@ func (s *Server) handleBlogSearch(w http.ResponseWriter, r *http.Request) { pattern := searchPattern(q) if pattern != "" { n := pageNum(r) - posts, total, err := blogStore(r).SearchPublishedPosts(r.Context(), pattern, searchPerPage, (n-1)*searchPerPage) + // Bounded: pgx cancels the statement when the context ends, and the + // reader sees "nothing found" rather than an error page. + ctx, cancel := context.WithTimeout(r.Context(), searchDeadline) + defer cancel() + posts, total, err := blogStore(r).SearchPublishedPosts(ctx, pattern, searchPerPage, (n-1)*searchPerPage) + if errors.Is(err, context.DeadlineExceeded) || errors.Is(ctx.Err(), context.DeadlineExceeded) { + logf("search %q on %s timed out", q, currentBlog(r).Subdomain) + posts, total, err = nil, 0, nil + } if err != nil { s.serverError(w, err) return diff --git a/internal/web/search.go b/internal/web/search.go index 9af980d..e0c3533 100644 --- a/internal/web/search.go +++ b/internal/web/search.go @@ -5,6 +5,7 @@ import ( "html/template" "regexp" "strings" + "time" "unicode/utf8" "github.com/microcosm-cc/bluemonday" @@ -14,8 +15,10 @@ import ( const ( searchPerPage = 20 - maxSearchRunes = 100 // longer queries are cut; nobody types more on purpose - snippetContext = 80 // runes kept on each side of the match + maxSearchRunes = 100 // longer queries are cut; nobody types more on purpose + maxSearchWords = 8 // more ".*" joins only make the regex scan dearer, never the answer better + snippetContext = 80 // runes kept on each side of the match + searchDeadline = 5 * time.Second // the scan is unindexed; past this it is abuse or a blog too big for it ) // searchPattern turns what the reader typed into the regular expression both @@ -23,6 +26,9 @@ const ( // in between", so "go tem" finds "Go templates". Blank → "". func searchPattern(q string) string { words := strings.Fields(q) + if len(words) > maxSearchWords { + words = words[:maxSearchWords] + } for i, w := range words { words[i] = regexp.QuoteMeta(w) } diff --git a/internal/web/search_test.go b/internal/web/search_test.go index 6ced39b..7ade83f 100644 --- a/internal/web/search_test.go +++ b/internal/web/search_test.go @@ -10,11 +10,12 @@ import ( func TestSearchPattern(t *testing.T) { for q, want := range map[string]string{ - "go tem": "go.*tem", - " a.b (c) ": `a\.b.*\(c\)`, - "one": "one", - " ": "", - "": "", + "go tem": "go.*tem", + " a.b (c) ": `a\.b.*\(c\)`, + "one": "one", + " ": "", + "": "", + "a b c d e f g h i j": "a.*b.*c.*d.*e.*f.*g.*h", // maxSearchWords } { if got := searchPattern(q); got != want { t.Errorf("searchPattern(%q) = %q, want %q", q, got, want) diff --git a/internal/web/server.go b/internal/web/server.go index 6009310..f02dd68 100644 --- a/internal/web/server.go +++ b/internal/web/server.go @@ -26,14 +26,15 @@ type Server struct { root http.Handler blog http.Handler // Anonymous endpoints worth abusing get a token bucket each (ratelimit.go). - loginLimit *limiter // per client address and per username: 10 guesses, then 10 a minute + loginLimit *limiter // per client address and per username: 10 guesses, then 10 a minute + searchLimit *limiter // per client address: the search is an unindexed regex scan } // logf is log.Printf, a variable so tests can silence it. var logf = log.Printf func NewServer(cfg *config.Config, st *store.Store) *Server { - s := &Server{cfg: cfg, st: st, tpl: newTemplates(cfg.Dev), loginLimit: newLimiter(10, 10)} + s := &Server{cfg: cfg, st: st, tpl: newTemplates(cfg.Dev), loginLimit: newLimiter(10, 10), searchLimit: newLimiter(30, 30)} s.root = s.rootRoutes() s.blog = s.subdomainRoutes() return s |
