thermograph/frontend/server/internal/contentdata/loader_test.go
Emi Griffith e377c4de03 content: write the entities literally in pages.yaml titles and descriptions
pages.yaml's title/description are interpolated as plain strings into <title>
and <meta name="description">, so html/template escapes them. Writing an HTML
entity in the YAML therefore escapes it a second time and the user sees the
source. Live on main right now:

  <meta name="description" content="... a &amp;plusmn;7-day seasonal window ...">
  <title>Weather &amp;amp; climate glossary: ...</title>

which renders as a literal "&plusmn;7-day" in the SERP snippet and "&amp;" in
the browser tab, on the about and glossary pages. The Python->Go rewrite did not
change this: the entities live in the shared data file and both engines
autoescape identically.

glossary.yaml is deliberately NOT touched. Its `body` is typed template.HTML and
rendered raw (glossary_term.html.tmpl), so the entities and <b> tags there are
correct and would break if "fixed".

The regression test runs against the real content dir, which the frontend
Dockerfile already copies into the builder stage, so it gates the image rather
than only local runs.
2026-07-25 01:35:59 -07:00

147 lines
4.6 KiB
Go

package contentdata
import (
"os"
"path/filepath"
"regexp"
"testing"
)
func writeContent(t *testing.T, name, body string) string {
t.Helper()
dir := t.TempDir()
if err := os.WriteFile(filepath.Join(dir, name), []byte(body), 0o644); err != nil {
t.Fatal(err)
}
return dir
}
func TestLoadGlossaryOrderAndLookup(t *testing.T) {
dir := writeContent(t, "glossary.yaml", `
terms:
- slug: b-term
term: B
short: short b
body: |-
Body <b>b</b> with {base} link.
- slug: a-term
term: A
short: short a
body: body a
`)
g, err := LoadGlossary(dir)
if err != nil {
t.Fatalf("LoadGlossary: %v", err)
}
// File order, not alphabetical — the index page renders in this order.
if len(g.Terms) != 2 || g.Terms[0].Slug != "b-term" || g.Terms[1].Slug != "a-term" {
t.Errorf("order wrong: %+v", g.Terms)
}
e, ok := g.Get("b-term")
if !ok || e.Term != "B" || e.Body != "Body <b>b</b> with {base} link." {
t.Errorf("Get(b-term) = %+v, %v", e, ok)
}
if _, ok := g.Get("missing"); ok {
t.Error("Get(missing) should report not-found")
}
}
func TestLoadGlossaryFailsLoud(t *testing.T) {
cases := map[string]string{
"empty terms": "terms: []\n",
"missing field": "terms:\n- slug: x\n term: X\n short: s\n", // no body
"missing slug": "terms:\n- term: X\n short: s\n body: b\n",
"duplicate slug": "terms:\n- {slug: x, term: X, short: s, body: b}\n- {slug: x, term: Y, short: s, body: b}\n",
"empty required": "terms:\n- {slug: x, term: X, short: '', body: b}\n", // falsy value fails, like the Python's `not entry.get(f)`
}
for name, body := range cases {
if _, err := LoadGlossary(writeContent(t, "glossary.yaml", body)); err == nil {
t.Errorf("%s: expected an error", name)
}
}
}
func TestLoadPages(t *testing.T) {
dir := writeContent(t, "pages.yaml", `
pages:
about:
title: "About | X"
description: >-
Folded
description.
`)
p, err := LoadPages(dir)
if err != nil {
t.Fatalf("LoadPages: %v", err)
}
if p["about"].Title != "About | X" || p["about"].Description != "Folded description." {
t.Errorf("pages = %+v", p)
}
if _, err := LoadPages(writeContent(t, "pages.yaml", "pages: {}\n")); err == nil {
t.Error("empty pages mapping should fail loud")
}
if _, err := LoadPages(writeContent(t, "pages.yaml", "pages:\n about: {title: T}\n")); err == nil {
t.Error("missing description should fail loud")
}
}
// Smoke-check against the real committed content files (frontend/content/),
// when present — the same files the Python loader validated at import.
func TestLoadRealContentFiles(t *testing.T) {
dir := filepath.Join("..", "..", "..", "content")
if _, err := os.Stat(filepath.Join(dir, "glossary.yaml")); err != nil {
t.Skipf("real content dir not available: %v", err)
}
g, err := LoadGlossary(dir)
if err != nil {
t.Fatalf("LoadGlossary(real): %v", err)
}
if len(g.Terms) == 0 {
t.Error("real glossary should have terms")
}
p, err := LoadPages(dir)
if err != nil {
t.Fatalf("LoadPages(real): %v", err)
}
for _, key := range []string{"about", "privacy", "hub", "glossary_index"} {
if _, ok := p[key]; !ok {
t.Errorf("real pages.yaml missing key %q (content.py reads it)", key)
}
}
}
// pages.yaml's title/description are interpolated as plain strings into
// <title> and <meta name="description"> (base.html.tmpl), so html/template
// escapes them. An HTML entity written in the YAML is therefore escaped a
// second time and the user sees the literal source: "&plusmn;7-day" in the
// SERP snippet, "Weather &amp; climate glossary" in the browser tab. Both
// shipped live until this test existed.
//
// glossary.yaml's `body` is deliberately NOT checked: it is typed
// template.HTML (see glossary_term.html.tmpl) and rendered raw, so entities
// and <b> tags there are correct and must stay.
func TestRealPagesHaveNoDoubleEscapedEntities(t *testing.T) {
dir := filepath.Join("..", "..", "..", "content")
if _, err := os.Stat(filepath.Join(dir, "pages.yaml")); err != nil {
t.Skipf("real content dir not available: %v", err)
}
pages, err := LoadPages(dir)
if err != nil {
t.Fatalf("LoadPages(real): %v", err)
}
// Named (&amp;), decimal (&#177;) and hex (&#xB1;) entity forms.
entity := regexp.MustCompile(`&([a-zA-Z][a-zA-Z0-9]*|#[0-9]+|#[xX][0-9a-fA-F]+);`)
for key, p := range pages {
for field, value := range map[string]string{
"title": p.Title, "description": p.Description,
} {
if m := entity.FindString(value); m != "" {
t.Errorf("pages.yaml[%s].%s contains the HTML entity %q; "+
"write the character literally (this field is escaped at "+
"render time, so the entity reaches the user as source)",
key, field, m)
}
}
}
}