Files
hotdog-cms/internal/check/rules.go
T

311 lines
10 KiB
Go

package check
import (
"bufio"
"encoding/xml"
"io"
"io/fs"
"os"
"path/filepath"
"regexp"
"sort"
"strings"
"time"
"git.coffeylabs.org/coffey-labs/hotdog-cms/internal/media"
"git.coffeylabs.org/coffey-labs/hotdog-cms/internal/site"
)
// Hosts whose scripts track visitors. Loading one on page view means the
// tracking starts before anyone could agree to it. A site that asks first
// loads it from its consent code instead, which this check doesn't see.
var trackers = []string{
"googletagmanager.com", "google-analytics.com", "doubleclick.net", "googleadservices.com",
"connect.facebook.net", "facebook.net", "analytics.tiktok.com", "snap.licdn.com",
"static.hotjar.com", "script.hotjar.com", "clarity.ms", "cdn.segment.com",
"static.cloudflareinsights.com", "js.hs-scripts.com", "matomo.cloud", "cdn.amplitude.com",
"mc.yandex.ru", "bat.bing.com",
}
func isTracker(h string) bool {
for _, t := range trackers {
if h == t || strings.HasSuffix(h, "."+t) {
return true
}
}
return false
}
var paginated = regexp.MustCompile(`/page/\d+/$`)
func (c *checker) checkPage(pg *page) {
strict := c.cfg.Check.CSP == "strict"
inlineLevel := Warning
if strict {
inlineLevel = Error
}
// Tracking and third parties.
seen := map[string]bool{}
for _, ref := range append(append(append([]string{}, pg.scripts...), pg.frames...), pg.styles...) {
h := c.host(ref)
if h == "" || seen[h] {
continue
}
seen[h] = true
switch {
case isTracker(h):
c.add(Error, "tracker-before-consent", pg.file, "%s loads on page view, before any consent; load it from your consent code instead", h)
case !c.allow[h]:
c.add(Warning, "third-party", pg.file, "loads from %s; list it in check.allow_third_party if that's intended, and in the privacy notice", h)
}
}
if pg.ungated {
c.add(Warning, "analytics-ungated", pg.file, "visit statistics start on page view, without asking (analytics.gated: false in site.yaml)")
}
for _, js := range pg.inlineJS {
if strings.Contains(js, "gtag(") || strings.Contains(js, "dataLayer") || strings.Contains(js, "fbq(") || strings.Contains(js, "_paq") {
c.add(Error, "tracker-before-consent", pg.file, "inline tracking code runs on page view: %s", js)
}
}
// What a strict Content-Security-Policy would refuse.
if len(pg.inlineJS) > 0 {
c.add(inlineLevel, "inline-script", pg.file, "%d inline script(s), first: %s", len(pg.inlineJS), pg.inlineJS[0])
}
if pg.inlineStyles > 0 {
c.add(inlineLevel, "inline-style", pg.file, "%d inline style(s) (style attributes or <style>)", pg.inlineStyles)
}
for _, h := range pg.handlers {
c.add(Error, "inline-handler", pg.file, "event handler attribute %s; attach it from a script file", h)
}
for _, h := range pg.scriptLinks {
c.add(Error, "script-link", pg.file, "a link that runs code (%.60s); link to a page instead", h)
}
// A redirect page has nothing else to check: it isn't meant to be read
// or found. (The security rules above still apply to it.)
if pg.redirect {
return
}
for _, u := range pg.insecure {
c.add(Error, "mixed-content", pg.file, "loads %s over plain http", u)
}
if pg.imgNoAlt > 0 {
c.add(Warning, "img-alt", pg.file, "%d image(s) without an alt attribute (use alt=\"\" for decoration)", pg.imgNoAlt)
}
// Broken links.
for _, ref := range pg.links {
target, ok := c.internal(ref, pg.url)
if !ok {
continue
}
if !exists(c.out, target) {
c.add(Error, "broken-link", pg.file, "links to %s, which is not in the site", ref)
}
}
// Cloudflare's Email Address Obfuscation.
if c.cfg.Check.Cloudflare {
for _, e := range uniq(pg.exposed) {
c.add(Warning, "cloudflare-email", pg.file, "%s is outside <!--email_off--> markers; Cloudflare will rewrite it", e)
}
}
// Search and social metadata, for pages meant to be found.
if pg.noindex || pg.static || strings.HasSuffix(pg.file, "/404.html") {
return
}
if strings.TrimSpace(pg.title) == "" {
c.add(Error, "missing-title", pg.file, "no <title>")
}
if pg.description == "" {
c.add(Warning, "missing-description", pg.file, "no meta description")
} else if n := len([]rune(pg.description)); n > 300 {
c.add(Warning, "description-length", pg.file, "meta description is %d characters; search results show about 160", n)
}
if pg.canonical == "" {
c.add(Warning, "missing-canonical", pg.file, "no canonical link")
}
if pg.ogImage == "" {
c.add(Warning, "social-image", pg.file, "no og:image; links to this page will preview without a picture")
} else {
if !pg.ogImageAlt {
c.add(Warning, "social-image-alt", pg.file, "og:image has no og:image:alt")
}
if target, ok := c.internal(pg.ogImage, pg.url); ok && !exists(c.out, target) {
c.add(Error, "social-image", pg.file, "og:image %s is not in the site", pg.ogImage)
}
}
}
func uniq(xs []string) []string {
seen := map[string]bool{}
var out []string
for _, x := range xs {
if !seen[x] {
seen[x] = true
out = append(out, x)
}
}
return out
}
func (c *checker) sitewide(pages []*page) {
byURL := map[string]*page{}
titles := map[string][]string{}
descs := map[string][]string{}
thirdParty := false
for _, pg := range pages {
byURL[pg.url] = pg
if pg.redirect {
continue
}
for _, ref := range append(append([]string{}, pg.scripts...), pg.frames...) {
if c.host(ref) != "" {
thirdParty = true
}
}
if pg.noindex || pg.static || paginated.MatchString(pg.url) || strings.HasSuffix(pg.file, "/404.html") {
continue
}
if t := strings.TrimSpace(pg.title); t != "" {
titles[t] = append(titles[t], pg.file)
}
if pg.description != "" {
descs[pg.description] = append(descs[pg.description], pg.file)
}
}
dup := func(rule, what string, m map[string][]string) {
keys := make([]string, 0, len(m))
for k := range m {
keys = append(keys, k)
}
sort.Strings(keys)
for _, k := range keys {
if files := m[k]; len(files) > 1 {
sort.Strings(files)
c.add(Warning, rule, files[0], "same %s as %d other page(s), e.g. %s", what, len(files)-1, files[1])
}
}
}
dup("duplicate-title", "title", titles)
dup("duplicate-description", "description", descs)
// The sitemap lists only pages meant to be indexed, and robots.txt points to it.
sitemap := filepath.Join(c.out, "sitemap.xml")
if raw, err := os.ReadFile(sitemap); err == nil {
var sm struct {
URLs []struct {
Loc string `xml:"loc"`
} `xml:"url"`
}
if xml.Unmarshal(raw, &sm) == nil {
for _, u := range sm.URLs {
path := strings.TrimPrefix(u.Loc, strings.TrimRight(c.cfg.URL, "/"))
if pg := byURL[path]; pg != nil && (pg.noindex || pg.redirect) {
c.add(Error, "noindex-in-sitemap", "public/sitemap.xml", "lists %s, which asks not to be indexed", path)
}
}
}
robots, err := os.ReadFile(filepath.Join(c.out, "robots.txt"))
if err != nil {
c.add(Warning, "robots-sitemap", "public/robots.txt", "no robots.txt, so crawlers aren't told where the sitemap is")
} else if !strings.Contains(strings.ToLower(string(robots)), "sitemap:") {
c.add(Warning, "robots-sitemap", "public/robots.txt", "has no Sitemap: line")
}
}
// security.txt (RFC 9116): present, and not past its Expires date.
sec := filepath.Join(c.out, ".well-known", "security.txt")
if f, err := os.Open(sec); err != nil {
c.add(Warning, "security-txt", "public/.well-known/security.txt", "missing; it tells researchers how to report a vulnerability")
} else {
var expires time.Time
sc := bufio.NewScanner(f)
for sc.Scan() {
if v, ok := strings.CutPrefix(strings.TrimSpace(sc.Text()), "Expires:"); ok {
expires, _ = time.Parse(time.RFC3339, strings.TrimSpace(v))
}
}
f.Close()
switch {
case expires.IsZero():
c.add(Error, "security-txt", "public/.well-known/security.txt", "no valid Expires: field (required)")
case expires.Before(c.now):
c.add(Error, "security-txt", "public/.well-known/security.txt", "expired on %s", expires.Format("2006-01-02"))
case expires.Before(c.now.AddDate(0, 0, 30)):
c.add(Warning, "security-txt", "public/.well-known/security.txt", "expires on %s, within 30 days", expires.Format("2006-01-02"))
}
}
// A site that loads anything from elsewhere owes its visitors a privacy
// notice, and the notice owes them a way to reach a person.
if thirdParty {
pp := c.cfg.Check.PrivacyPage
if pp == "" {
pp = "/privacy/"
}
pg := byURL[pp]
switch {
case pg == nil || pg.redirect:
c.add(Error, "privacy-page", pp, "the site loads third-party code but has no privacy notice at %s (check.privacy_page)", pp)
case len(pg.mailtos) == 0:
c.add(Warning, "privacy-contact", pg.file, "the privacy notice has no email address to write to")
}
}
}
// Published pictures: a photo straight off a phone carries the place it was
// taken, which on a home page is the author's address. Pictures uploaded
// through the editor are re-encoded and never have it; this catches the ones
// committed by hand.
func (c *checker) images() {
_ = filepath.WalkDir(c.out, func(p string, d fs.DirEntry, err error) error {
if err != nil || d.IsDir() {
return nil
}
ext := strings.ToLower(filepath.Ext(p))
if ext != ".jpg" && ext != ".jpeg" && ext != ".png" && ext != ".webp" && ext != ".gif" {
return nil
}
rel, _ := filepath.Rel(c.out, p)
file := "public/" + filepath.ToSlash(rel)
info, err := d.Info()
if err == nil && info.Size() > imageHeavy {
c.add(Warning, "image-weight", file, "is %d KB; pictures over %d KB slow pages down (the editor resizes uploads)", info.Size()>>10, imageHeavy>>10)
}
if ext != ".jpg" && ext != ".jpeg" {
return nil
}
f, err := os.Open(p)
if err != nil {
return nil
}
head := make([]byte, 256<<10) // EXIF sits at the start
n, _ := io.ReadFull(f, head)
f.Close()
if media.ReadJPEGMeta(head[:n]).GPS {
c.add(Error, "image-location", file, "carries the GPS location where it was taken; upload it through the editor, or remove it (exiftool -gps:all= FILE)")
}
return nil
})
}
const imageHeavy = 500 << 10
// The look: every pair the theme says must stay readable, with the site's
// choices applied, in light and in dark mode.
func (c *checker) look() {
lf, err := site.LoadLook(c.siteDir)
if err != nil || lf == nil {
return // a broken look.yaml fails the build first
}
vals, _ := lf.Values(c.cfg.Look)
for _, r := range lf.Contrasts(vals) {
for _, m := range []struct {
mode string
ratio float64
}{{"light", r.Light}, {"dark", r.Dark}} {
if m.ratio < site.MinContrast {
c.add(Error, "look-contrast", "site.yaml", "%s on %s is %.2f:1 in %s mode; text needs at least %.1f:1", r.Fore, r.Back, m.ratio, m.mode, site.MinContrast)
}
}
}
}