diff options
Diffstat (limited to 'internal/web/filetype.go')
| -rw-r--r-- | internal/web/filetype.go | 196 |
1 files changed, 196 insertions, 0 deletions
diff --git a/internal/web/filetype.go b/internal/web/filetype.go new file mode 100644 index 0000000..9c0f590 --- /dev/null +++ b/internal/web/filetype.go @@ -0,0 +1,196 @@ +package web + +import ( + "mime" + "net/http" + "net/url" + "path" + "strconv" + "strings" + "unicode" +) + +// Uploads: what a file is and how it may be served. +// +// The root domain carries the session cookie and serves every blog's files +// (/b/{sub}/media previews, the root blog's own /media), so an uploaded HTML, +// SVG or XML page must never render there. The rules below make that a +// property of the stored content type: the sniffer is trusted first, a +// filename extension may only refine a generic sniff to a type on an +// allowlist, and anything not on the inline list is a download. + +// maxUploadFiles is how many files one Files-page request may carry; the body +// cap of that route is this many upload limits. +const maxUploadFiles = 10 + +// maxUploadMB caps the superadmin's per-blog override: substring() takes int4 +// offsets, and the whole upload sits in memory while it is stored. +const maxUploadMB = 1024 + +var fileKinds = []string{"image", "document", "audio", "video", "archive", "other"} + +// fileKindNames are the tab labels (translated where used, like moduleNames). +var fileKindNames = map[string]string{"image": "Images", "document": "Documents", "audio": "Audio", "video": "Video", "archive": "Archives", "other": "Other"} + +var imageTypes = map[string]bool{"image/png": true, "image/jpeg": true, "image/gif": true, "image/webp": true, "image/x-icon": true, "image/avif": true, "image/bmp": true} + +var documentTypes = map[string]bool{ + "application/pdf": true, "text/plain": true, "text/csv": true, "application/rtf": true, "application/epub+zip": true, + "application/msword": true, "application/vnd.ms-excel": true, "application/vnd.ms-powerpoint": true, + "application/vnd.openxmlformats-officedocument.wordprocessingml.document": true, + "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet": true, + "application/vnd.openxmlformats-officedocument.presentationml.presentation": true, + "application/vnd.oasis.opendocument.text": true, + "application/vnd.oasis.opendocument.spreadsheet": true, + "application/vnd.oasis.opendocument.presentation": true, +} + +var archiveTypes = map[string]bool{ + "application/zip": true, "application/gzip": true, "application/x-gzip": true, "application/x-7z-compressed": true, + "application/x-tar": true, "application/x-bzip2": true, "application/x-xz": true, "application/x-rar-compressed": true, "application/vnd.rar": true, +} + +var mediaTypes = map[string]bool{ + "audio/mpeg": true, "audio/mp4": true, "audio/ogg": true, "audio/flac": true, "audio/wav": true, "audio/wave": true, "audio/x-wav": true, "audio/webm": true, "audio/aac": true, + "video/mp4": true, "video/ogg": true, "video/webm": true, "video/x-matroska": true, "video/quicktime": true, +} + +// extAllowed is what an extension may turn a generic sniff into: nothing a +// browser would run. Images are deliberately absent — a real image sniffs. +func extAllowed(ct string) bool { + return documentTypes[ct] || archiveTypes[ct] || mediaTypes[ct] +} + +// fileType decides what an upload is from its first bytes and its name. The +// client's declared type is never consulted, and the result is always one of +// the known types above or application/octet-stream, so a stored content +// type is safe to serve by construction. +func fileType(head []byte, filename string) (contentType, kind string) { + ct := mediaType(http.DetectContentType(head)) + ext := strings.ToLower(path.Ext(filename)) + // The sniffer only knows containers for these. + switch { + case ct == "application/ogg": + ct = "audio/ogg" + if ext == ".ogv" { + ct = "video/ogg" + } + case ct == "video/mp4" && ext == ".m4a": + ct = "audio/mp4" + case ct == "image/vnd.microsoft.icon": + ct = "image/x-icon" + } + if ct == "text/plain" || ct == "application/octet-stream" { + switch e := mediaType(mime.TypeByExtension(ext)); { + case e == "": // unknown extension: the bytes are all we have + case extAllowed(e): + ct = e + case ct == "text/plain" && strings.HasPrefix(e, "text/") && !scriptTypes[e]: + // .md, .log, .ini…: text is text (the tables differ between machines, so the name only confirms) + default: // a name we would not serve inline (.html, .svg, .js, .exe…), whatever the bytes look like + ct = "application/octet-stream" + } + } + if !imageTypes[ct] && !extAllowed(ct) && !strings.HasPrefix(ct, "audio/") && !strings.HasPrefix(ct, "video/") { + ct = "application/octet-stream" // sniffed HTML/XML, fonts, and everything else we do not name + } + return ct, kindOf(ct) +} + +// scriptTypes are text types a browser would execute or interpret as markup. +var scriptTypes = map[string]bool{"text/html": true, "text/javascript": true, "text/xml": true, "text/css": true} + +// mediaType drops the parameters ("; charset=utf-8") from a content type. +func mediaType(ct string) string { + if ct == "" { + return "" + } + mt, _, err := mime.ParseMediaType(ct) + if err != nil { + return "" + } + return mt +} + +// kindOf buckets a content type for the Files page tabs. +func kindOf(ct string) string { + switch { + case imageTypes[ct]: + return "image" + case documentTypes[ct]: + return "document" + case strings.HasPrefix(ct, "audio/"): + return "audio" + case strings.HasPrefix(ct, "video/"): + return "video" + case archiveTypes[ct]: + return "archive" + } + return "other" +} + +// inlineOK says whether a browser may render the type in place. +func inlineOK(ct string) bool { + return imageTypes[ct] || ct == "application/pdf" || ct == "text/plain" || strings.HasPrefix(ct, "audio/") || strings.HasPrefix(ct, "video/") +} + +// servedAs is the Content-Type and disposition /media answers with. +func servedAs(ct string, download bool) (ctype, disposition string) { + if !inlineOK(ct) { + return "application/octet-stream", "attachment" + } + if ct == "text/plain" { + ct = "text/plain; charset=utf-8" + } + if download { + return ct, "attachment" + } + return ct, "inline" +} + +// contentDisposition carries the filename in both the plain form (ASCII only, +// for old browsers) and the RFC 5987 one (Greek names survive). +func contentDisposition(disposition, name string) string { + ascii := strings.Map(func(r rune) rune { + if r < 0x20 || r > 0x7e || r == '"' || r == '\\' { + return '_' + } + return r + }, name) + return disposition + `; filename="` + ascii + `"; filename*=utf-8''` + url.PathEscape(name) +} + +// cleanFilename keeps only the base name a browser sent (Windows paths +// included), without control characters or quotes, at most 120 runes. +func cleanFilename(name string) string { + name = path.Base(strings.ReplaceAll(name, `\`, "/")) + name = strings.Map(func(r rune) rune { + if unicode.IsControl(r) || r == '"' || r == '\'' { + return -1 + } + return r + }, name) + name = strings.TrimSpace(name) + if r := []rune(name); len(r) > 120 { + name = string(r[:120]) + } + if name == "" || name == "." || name == "/" || name == ".." { + return "file" + } + return name +} + +// humanSize prints a byte count the way the dashboard shows it. +func humanSize(n int64) string { + if n < 1<<20 { + return strconv.FormatInt(max(1, n>>10), 10) + " KB" + } + return strings.TrimSuffix(strconv.FormatFloat(float64(n)/(1<<20), 'f', 1, 64), ".0") + " MB" +} + +// pageBounds clamps a 1-based page number to the list and gives the SQL offset. +func pageBounds(total, per, n int) (offset, page, last int) { + last = max(1, (total+per-1)/per) + page = min(max(1, n), last) + return (page - 1) * per, page, last +} |
