"""Tests for the crawlable SEO content pages: the city set, robots/sitemap, and that the server-rendered pages contain the stats in the HTML (with the weather layer faked, like test_api.py).""" import math import re import pytest from fastapi.testclient import TestClient import app as appmod import cities import climate import grading # BASE defaults to /thermograph in tests (THERMOGRAPH_BASE unset). B = "/thermograph" SLUG = "london-england-gb" # present in cities.json @pytest.fixture def client(monkeypatch, history, recent): monkeypatch.setattr(climate, "load_cached_history", lambda cell: history.clone()) monkeypatch.setattr(climate, "get_recent_forecast", lambda cell: recent.clone()) monkeypatch.setattr(climate, "get_history", lambda cell: (history.clone(), {"cached": True})) return TestClient(appmod.app) def test_city_set_slugs_unique_and_lookup(): slugs = cities.all_slugs() assert len(slugs) == len(set(slugs)) >= 100 assert cities.get(SLUG) is not None assert cities.get("does-not-exist") is None def test_robots_txt(client): r = client.get(f"{B}/robots.txt") assert r.status_code == 200 assert "Sitemap:" in r.text and "/sitemap.xml" in r.text assert "Disallow: /thermograph/api/" in r.text def test_sitemap_lists_city_urls(client): r = client.get(f"{B}/sitemap.xml") assert r.status_code == 200 assert "" in r.text assert f"/climate/{SLUG}/july" in r.text assert f"/climate/{SLUG}/records" in r.text def test_city_page_renders_stats_in_html(client): r = client.get(f"{B}/climate/{SLUG}") assert r.status_code == 200 b = r.text assert "London" in b and "climate" in b.lower() assert 'rel="canonical"' in b and f"/climate/{SLUG}" in b assert '"@type":"Dataset"' in b assert "average temperatures by month" in b.lower() # a real number rendered in the normals/today text (London is GB, so °C) assert re.search(r"-?\d+°C", b) def _jsonld_breadcrumbs(html: str) -> list[dict]: import json, re m = re.search(r'', html, re.S) assert m, "no JSON-LD block on page" graph = json.loads(m.group(1))["@graph"] return [n for n in graph if n["@type"] == "BreadcrumbList"] @pytest.mark.parametrize("path", [f"/climate/{SLUG}", f"/climate/{SLUG}/records"]) def test_breadcrumb_jsonld_items_valid_for_google(client, path): # Google: every ListItem except the last must carry an ``item`` URL, so # unlinked intermediate crumbs (the country) must not appear in JSON-LD. (bc,) = _jsonld_breadcrumbs(client.get(B + path).text) els = bc["itemListElement"] assert [e["position"] for e in els] == list(range(1, len(els) + 1)) for e in els[:-1]: assert e["item"].startswith("http"), f"crumb {e['name']!r} missing item" def test_city_404(client): assert client.get(f"{B}/climate/nope-not-a-city").status_code == 404 def test_curated_events_have_valid_slugs(): import city_events slugs = set(cities.all_slugs()) assert len(city_events.EVENTS) >= 20 assert all(s in slugs for s in city_events.EVENTS), \ [s for s in city_events.EVENTS if s not in slugs] def test_curated_event_renders(client): b = client.get(f"{B}/climate/new-orleans-louisiana-us").text assert "Notable weather in New Orleans" in b assert "Hurricane Katrina" in b def test_uncurated_city_has_no_event_section(client): b = client.get(f"{B}/climate/shanghai-cn").text assert "city-event" not in b # falls back to the flavor blurb def test_city_travel_cta_prefills_compare(client): b = client.get(f"{B}/climate/{SLUG}").text assert "Thinking of visiting" in b assert "/compare#loc=" in b # the city is pre-loaded on the compare page def test_city_blurb_renders_with_attribution(client, monkeypatch): monkeypatch.setattr(cities, "flavor", lambda slug: {"extract": "Testville is a lovely place to test.", "url": "https://en.wikipedia.org/wiki/Testville", "title": "Testville"}) b = client.get(f"{B}/climate/{SLUG}").text assert "Testville is a lovely place to test." in b assert "via Wikipedia" in b # CC BY-SA attribution link def test_month_and_records(client): assert client.get(f"{B}/climate/{SLUG}/july").status_code == 200 assert client.get(f"{B}/climate/{SLUG}/records").status_code == 200 assert client.get(f"{B}/climate/{SLUG}/notamonth").status_code == 404 def test_month_records_cover_every_metric_high_and_low(client): # The month page's records show a record high AND low for each metric present, not # just warmest/coldest temperature — one card per metric with both extremes. (The # test fixture carries tmax/tmin/precip; real cities add feels/humidity/wind/gust.) b = client.get(f"{B}/climate/{SLUG}/july").text assert "July records" in b for label in ("High", "Low", "Precip"): assert f'class="rc-metric">{label}<' in b # Every card carries a labelled high and low row (temp metrics use Highest/Lowest). n = b.count('rc-row rc-high') assert n >= 3 and b.count('rc-row rc-low') == n assert "Highest" in b and "Lowest" in b # Precip's extremes are the wettest/driest whole month by total accumulation. assert "Wettest" in b and "Driest" in b def test_records_page_has_monthly_and_seasonal(client): b = client.get(f"{B}/climate/{SLUG}/records").text # Monthly records: all 12 months, each linking to its month page. assert "Records by month" in b for month in ("January", "July", "December"): assert month in b assert f"/climate/{SLUG}/january" in b # Seasonal records: labelled for London's Northern Hemisphere, with month spans. assert "Records by season" in b assert "Northern Hemisphere" in b assert 'Winter (Dec–Feb)' in b assert 'Summer (Jun–Aug)' in b # All-time table still present, plus structured data and real values. assert "All-time records" in b assert '"@type":"Dataset"' in b assert re.search(r"-?\d+°C", b) # The Dataset structured data still advertises the month/season coverage. assert "monthly" in b.lower() and "seasonal" in b.lower() def test_records_seasons_flip_in_southern_hemisphere(client): # São Paulo is south of the equator, so Dec–Feb is summer, not winter. b = client.get(f"{B}/climate/sao-paulo-br/records").text assert "Southern Hemisphere" in b assert 'Summer (Dec–Feb)' in b assert 'Winter (Jun–Aug)' in b def test_climate_pages_are_colour_coded(client): # Temperatures wear the diverging cold→hot palette (a heat map, not a plain grid), # and the city page carries the monthly temperature-range strip. city = client.get(f"{B}/climate/{SLUG}").text assert "t-cell t-" in city # tinted temperature cells in the normals table assert "range-fill" in city and "--c1:var(--" in city # the monthly range strip's gradient bars recs = client.get(f"{B}/climate/{SLUG}/records").text assert "t-inline t-" in recs # tinted record chips month = client.get(f"{B}/climate/{SLUG}/july").text assert "t-inline t-" in month # tinted hero numbers + records def test_records_show_both_extremes_per_metric(client): # Each metric column shows BOTH its record warmest (▲) and coldest (▼) — so the # daytime high carries its record-low high, and the overnight low its record-high low. b = client.get(f"{B}/climate/{SLUG}/records").text assert "Daytime high" in b and "Overnight low" in b assert "▲" in b and "▼" in b # 12 months × 2 metric columns × 2 extremes = 48 chips just in the monthly table. assert b.count('class="rec-ext"') >= 48 def test_hub_glossary_about(client): assert client.get(f"{B}/climate").status_code == 200 assert client.get(f"{B}/glossary").status_code == 200 assert client.get(f"{B}/glossary/percentile").status_code == 200 assert client.get(f"{B}/glossary/not-a-term").status_code == 404 assert client.get(f"{B}/about").status_code == 200 def test_title_name_is_short_and_unique(): # Titles lead with the bare city name so the payload ("… in July") survives # SERP truncation — display_name()'s region+country is what pushed it off the # end. The exception is a name that recurs: those must still be told apart, # and every rendered title must stay unique across the whole city set. all_cities = cities.all_cities() titles = [cities.title_name(c) for c in all_cities] assert len(titles) == len(set(titles)), "two cities share a title" # The overwhelming majority pay no disambiguation cost at all. assert sum(1 for t in titles if "," not in t) > 0.9 * len(titles) by_slug = {c["slug"]: cities.title_name(c) for c in all_cities} assert by_slug["seattle-washington-us"] == "Seattle" # Same name, different countries -> country code is enough. assert by_slug["london-england-gb"] == "London, GB" assert by_slug["london-ontario-ca"] == "London, CA" # Same name AND same country -> the country code would collide, so fall back # to admin1. Without this, both Columbuses would render "Columbus, US". assert by_slug["columbus-ohio-us"] == "Columbus, Ohio" assert by_slug["columbus-georgia-us"] == "Columbus, Georgia" def test_page_titles_lead_with_city_and_payload(client): for path, must_start, payload in ( (f"{B}/climate/{SLUG}", "London, GB climate", "how unusual it is now"), (f"{B}/climate/{SLUG}/july", "London, GB in July", "normal weather"), (f"{B}/climate/{SLUG}/records", "London, GB weather records", "hottest & coldest"), ): b = client.get(path).text title = re.search(r"(.*?)", b, re.S).group(1) title = title.replace("&", "&") assert title.startswith(must_start), title assert payload in title, title # The region+country the old template spent its width on is gone. assert "England, United Kingdom" not in title # Whatever else happens, the city and the page's subject are inside the # first ~40 chars, so a SERP cut can no longer remove the payload. assert len(must_start) <= 40 # og:title mirrors via base.html.j2's self.title(). og = re.search(r'<meta property="og:title" content="(.*?)"', b, re.S).group(1) assert og == re.search(r"<title>(.*?)", b, re.S).group(1) def test_meta_descriptions_lead_with_real_numbers(client): for path in (f"{B}/climate/{SLUG}", f"{B}/climate/{SLUG}/july", f"{B}/climate/{SLUG}/records"): b = client.get(path).text desc = re.search(r']*)>(-?\d+)°([CF]?)') def _temp_spans(html): """(fahrenheit_attr, shown_number, shown_letter) for every rendered temp span. Reading the spans rather than the raw document matters: base.html.j2 mentions "°F/°C" in an HTML comment, so a bare `"°F" in html` assertion passes on a page whose every temperature is Celsius. """ return [(float(f), int(n), letter) for f, _attrs, n, letter in _TEMP_SPAN.findall(html)] def test_climate_pages_carry_convertible_temps(client): # Every temperature is a client-convertible span; the old inline "(NN°C)" dual # form is gone (the toggle now supplies the other unit). for path in (f"{B}/climate/{SLUG}", f"{B}/climate/{SLUG}/july", f"{B}/climate/{SLUG}/records"): b = client.get(path).text spans = _temp_spans(b) assert spans, f"no temperature spans on {path}" assert re.search(r"\d+°F \(-?\d+°C\)", b) is None # no leftover dual-unit strings def test_units_default_to_the_citys_country(client): # The search snippet shows whatever the server rendered, and impressions skew # to °C countries — so a UK city page must say °C in the HTML itself, with no # JS and no stored preference. data-temp-f stays Fahrenheit either way: it is # what climate.js converts from. gb = client.get(f"{B}/climate/{SLUG}").text # London, GB us = client.get(f"{B}/climate/seattle-washington-us").text gb_spans, us_spans = _temp_spans(gb), _temp_spans(us) assert gb_spans and us_spans assert {letter for _f, _n, letter in gb_spans if letter} == {"C"} assert {letter for _f, _n, letter in us_spans if letter} == {"F"} # The attribute is still Fahrenheit, and really does differ from the Celsius # text — this is what breaks if someone "helpfully" converts the attribute too. assert any(math.floor(f + 0.5) != n for f, n, _l in gb_spans), "data-temp-f was converted" # On a °F page the attribute and the text agree by definition. Rounded half-up, # not with Python's round(): the page renders the number JS would. assert all(math.floor(f + 0.5) == n for f, n, _l in us_spans) assert 'data-unit-default="C"' in gb assert 'data-unit-default="F"' in us _PRECIP_SPAN = re.compile(r'([\d.]+) (in|cm)') def test_precip_and_wind_follow_the_citys_unit(client): # One toggle drives every measure: a °C page shows cm and km/h, not °C mixed # with inches and mph. The data-* attribute stays imperial in both, so # climate.js converts from the same source of truth it does for temperature. gb = client.get(f"{B}/climate/{SLUG}/records").text # London, GB us = client.get(f"{B}/climate/seattle-washington-us/records").text gb_p, us_p = _PRECIP_SPAN.findall(gb), _PRECIP_SPAN.findall(us) assert gb_p and us_p assert {u for _raw, _shown, u in gb_p} == {"cm"} assert {u for _raw, _shown, u in us_p} == {"in"} # The conversion is real, not just a relabel. for raw, shown, _u in gb_p: assert abs(float(shown) - float(raw) * 2.54) < 0.005, (raw, shown) # On an imperial page the attribute and the text agree. for raw, shown, _u in us_p: assert abs(float(shown) - float(raw)) < 0.005, (raw, shown) def test_wind_and_humidity_render_per_unit_system(): # The synthetic history carries only tmax/tmin/precip, so wind and humidity # never reach a rendered page in tests — exercise the renderers directly # rather than assert on a page that structurally cannot show them. import content with content._unit_scope("F"): assert content._wind_text(10) == "10 mph" assert '10 mph' == content._wind(10) assert content._fmt("humid", 7.75) == "7.8 g/m³" with content._unit_scope("C"): # 10 mph -> 16.09 km/h, rounded half-up like JS. assert content._wind_text(10) == "16 km/h" # data-wind-mph stays imperial: it is what climate.js converts from. assert 'data-wind-mph="10.0"' in content._wind(10) # Absolute humidity has no imperial equivalent anyone would recognise, so # it reads the same in both systems. assert content._fmt("humid", 7.75) == "7.8 g/m³" def test_precip_precision_differs_by_unit(): # cm and inches are only 2.54x apart, so cm keeps two decimals — at one # decimal every light-rain day would collapse to 0.0 or 0.1 cm. import content with content._unit_scope("F"): assert content._precip_text(0.04) == "0.04 in" with content._unit_scope("C"): assert content._precip_text(0.04) == "0.10 cm" assert content._precip_text(1.81) == "4.60 cm" def test_measure_conversion_constants_match_the_frontend(): # Same drift risk as the country list: two hand-maintained copies. import os import content units_js = os.path.join(os.path.dirname(os.path.dirname(os.path.abspath(__file__))), os.pardir, "frontend", "units.js") with open(units_js, encoding="utf-8") as f: src = f.read() cm = float(re.search(r"CM_PER_IN\s*=\s*([\d.]+)", src).group(1)) kmh = float(re.search(r"KMH_PER_MPH\s*=\s*([\d.]+)", src).group(1)) assert cm == content.CM_PER_IN assert kmh == content.KMH_PER_MPH def test_unit_default_absent_without_a_city(client): # No city -> no attribute -> units.js keeps today's locale behaviour. Pinning # these to "F" would silently regress the interactive pages. for path in (f"{B}/", f"{B}/climate", f"{B}/about", f"{B}/glossary"): assert "data-unit-default" not in client.get(path).text, path def test_unit_does_not_leak_between_requests(client): # The page handlers are sync `def`, so Starlette runs them on a recycled # threadpool thread: a unit left set would bleed into the next request served # by that thread. Interleave and re-check rather than trusting the reset. assert 'data-unit-default="C"' in client.get(f"{B}/climate/{SLUG}").text about = client.get(f"{B}/about").text assert "data-unit-default" not in about # about.html.j2 renders an illustrative temperature; it must stay °F. assert {l for _f, _n, l in _temp_spans(about) if l} == {"F"} us = client.get(f"{B}/climate/seattle-washington-us").text assert {l for _f, _n, l in _temp_spans(us) if l} == {"F"} gb_again = client.get(f"{B}/climate/{SLUG}").text assert {l for _f, _n, l in _temp_spans(gb_again) if l} == {"C"} def test_f_country_list_matches_the_frontend(): # Two hand-maintained copies of the same 14 codes; nothing else stops them # drifting apart, and drift means the server and the client disagree about # what a first-time visitor should see. import os import content units_js = os.path.join(os.path.dirname(os.path.dirname(os.path.abspath(__file__))), os.pardir, "frontend", "units.js") with open(units_js, encoding="utf-8") as f: src = f.read() literal = re.search(r"F_REGIONS = new Set\(\[(.*?)\]\)", src, re.S).group(1) assert set(re.findall(r'"([A-Z]{2})"', literal)) == set(content.F_COUNTRIES) def test_celsius_rounding_matches_js(): # Python's round() goes to even, JS's Math.round goes up. Where they disagree # the server prints one number and climate.js repaints another — a flicker for # the visitor whose unit already matched. import content assert content._round_half_up(0.5) == 1 # round(0.5) would give 0 assert content._round_half_up(1.5) == 2 assert content._round_half_up(2.5) == 3 # round(2.5) would give 2 assert content._round_half_up(-0.5) == 0 # Math.round(-0.5) === 0 # 31.1°F -> -0.5°C exactly: the case that actually shows up on a cold page. assert content._c(31.1) == 0 def test_city_page_etag_is_stable(client): # The unit comes from the URL's city, never from a request header, so the # response stays a pure function of the URL — no Vary, and the ETag still works. r1 = client.get(f"{B}/climate/{SLUG}") r2 = client.get(f"{B}/climate/{SLUG}") assert r1.headers["etag"] == r2.headers["etag"] r3 = client.get(f"{B}/climate/{SLUG}", headers={"If-None-Match": r1.headers["etag"]}) assert r3.status_code == 304 def test_seo_pages_load_unit_and_account_scripts(client): # Header parity with the interactive pages: the °F/°C toggle, account/bell, and # the temperature converter are all loaded. b = client.get(f"{B}/climate/{SLUG}").text for src in (f"{B}/units.js", f"{B}/account.js", f"{B}/climate.js"): assert f'src="{src}"' in b def test_indexnow_key_file_served(client): import indexnow k = indexnow.key() r = client.get(f"{B}/{k}.txt") assert r.status_code == 200 assert r.text.strip() == k assert r.headers["content-type"].startswith("text/plain") def test_sitemap_lastmod_is_stable(client): import datetime import re body = client.get(f"{B}/sitemap.xml").text mods = set(re.findall(r"([^<]+)", body)) assert len(mods) == 1 # one build date, not a per-request churn datetime.date.fromisoformat(next(iter(mods))) # a valid ISO date def test_indexnow_submit_all_builds_payload(monkeypatch): import indexnow calls = [] class _Resp: status_code = 200 text = "ok" monkeypatch.setattr(indexnow.httpx, "post", lambda url, json=None, **kw: (calls.append((url, json)), _Resp())[1]) res = indexnow.submit_all("https://thermograph.org") assert res["submitted"] == res["total"] > 100 first = calls[0][1] assert first["host"] == "thermograph.org" assert first["keyLocation"] == f"https://thermograph.org/{indexnow.key()}.txt" assert first["urlList"][0] == "https://thermograph.org/" assert all(len(c[1]["urlList"]) <= 10000 for c in calls) # batched under the IndexNow cap def test_indexnow_if_changed_state(monkeypatch, tmp_path): import indexnow monkeypatch.setattr(indexnow, "_STATE_PATH", str(tmp_path / "state.txt")) sig = indexnow.url_signature() assert len(sig) == 40 and sig == indexnow.url_signature() # stable sha1 of the URL set assert indexnow._read_state() == "" # first deploy → would submit indexnow._write_state(sig) assert indexnow._read_state() == sig # unchanged → deploy would skip def test_search_verification_meta(client, monkeypatch): monkeypatch.setenv("THERMOGRAPH_GOOGLE_VERIFY", "gtok123") monkeypatch.setenv("THERMOGRAPH_BING_VERIFY", "btok456") seo = client.get(f"{B}/climate/{SLUG}").text # Jinja SEO page home = client.get(f"{B}/").text # static homepage via _page() for b in (seo, home): assert '' in b assert '' in b # The brand lockup is the site-wide "go home" affordance. The static tool pages # and the Jinja pages build their headers separately, so this is easy to add to # one and forget on the other — assert every page carries it. BRAND_PAGES = ["/", "/calendar", "/day", "/compare", "/legend", "/alerts", "/about", "/privacy", "/climate", "/glossary"] @pytest.mark.parametrize("path", BRAND_PAGES) def test_brand_lockup_links_home(client, path): html = client.get(f"{B}{path}").text brand = html.split('
', 1)[1].split("", 1)[0] head = brand.split("")[0] if "")[0] # The link must wrap the whole lockup, so clicking the mark works too, not # just the word. assert "