268 lines
11 KiB
Python
268 lines
11 KiB
Python
|
|
"""Typo-tolerant place-name index for search suggestions.
|
|||
|
|
|
|||
|
|
Open-Meteo's geocoder (climate.geocode) only matches exact spellings — one
|
|||
|
|
mistyped letter and the query returns nothing. This module keeps a local index
|
|||
|
|
of world places (a GeoNames cities dump, downloaded once into data/geonames/
|
|||
|
|
and reused across restarts) so /suggest can answer instantly and tolerate a
|
|||
|
|
single-letter typo — a substituted, missing, or extra letter, or two adjacent
|
|||
|
|
letters swapped — anywhere in the query, including the first character.
|
|||
|
|
|
|||
|
|
It also builds a vocabulary of place-name tokens so a typo'd word in a
|
|||
|
|
multi-word query can be respelled against words the index knows ("pest
|
|||
|
|
seattle" → "west seattle"); the endpoint verifies those candidates against the
|
|||
|
|
upstream geocoder, which covers neighbourhood-level places the cities dump
|
|||
|
|
lacks.
|
|||
|
|
|
|||
|
|
The index loads in a background thread (start_loading()). Until it's ready —
|
|||
|
|
or forever, if the download fails — search()/corrections() return None/empty
|
|||
|
|
and /suggest degrades to the plain upstream geocoder, so the app never needs
|
|||
|
|
this data to boot or serve.
|
|||
|
|
"""
|
|||
|
|
import bisect
|
|||
|
|
import heapq
|
|||
|
|
import os
|
|||
|
|
import threading
|
|||
|
|
import time
|
|||
|
|
import unicodedata
|
|||
|
|
import zipfile
|
|||
|
|
|
|||
|
|
import httpx
|
|||
|
|
|
|||
|
|
import audit
|
|||
|
|
|
|||
|
|
GEO_DIR = os.path.join(os.path.dirname(__file__), "..", "data", "geonames")
|
|||
|
|
GEONAMES_URL = "https://download.geonames.org/export/dump/"
|
|||
|
|
|
|||
|
|
# Which GeoNames cities dump to index. cities1000 (~170k places, population
|
|||
|
|
# ≥ 1000) is small enough to hold in memory and big enough to cover the tiny
|
|||
|
|
# vacation towns people actually search for; set THERMOGRAPH_CITIES=cities5000
|
|||
|
|
# (or cities15000) to trade coverage for a lighter footprint.
|
|||
|
|
CITIES = os.environ.get("THERMOGRAPH_CITIES", "cities1000")
|
|||
|
|
|
|||
|
|
# Entry tuple layout: (norm_name, name, admin1, country, country_code, lat, lon, pop)
|
|||
|
|
_POP = 7
|
|||
|
|
|
|||
|
|
_load_lock = threading.Lock()
|
|||
|
|
_load_started = False
|
|||
|
|
|
|||
|
|
# Set once, atomically, by the loader thread: (entries, names, order, vocab)
|
|||
|
|
# where `entries` is population-descending (so fuzzy scans can stop at the
|
|||
|
|
# first matches found), `names`/`order` are the normalized names sorted
|
|||
|
|
# alphabetically with their entry indices (for prefix bisection), and `vocab`
|
|||
|
|
# maps each place-name token to the population of the biggest place using it.
|
|||
|
|
_data = None
|
|||
|
|
|
|||
|
|
|
|||
|
|
def start_loading() -> None:
|
|||
|
|
"""Kick off the background index load, once per process (idempotent)."""
|
|||
|
|
global _load_started
|
|||
|
|
with _load_lock:
|
|||
|
|
if _load_started:
|
|||
|
|
return
|
|||
|
|
_load_started = True
|
|||
|
|
threading.Thread(target=_load, name="places-index", daemon=True).start()
|
|||
|
|
|
|||
|
|
|
|||
|
|
def ready() -> bool:
|
|||
|
|
return _data is not None
|
|||
|
|
|
|||
|
|
|
|||
|
|
def _fetch(name: str) -> str:
|
|||
|
|
"""Path to a GeoNames dump file, downloading into data/geonames/ if absent.
|
|||
|
|
Cached forever — cities don't move; delete the folder to force a refresh."""
|
|||
|
|
path = os.path.join(GEO_DIR, name)
|
|||
|
|
if os.path.exists(path) and os.path.getsize(path) > 0:
|
|||
|
|
return path
|
|||
|
|
r = httpx.get(GEONAMES_URL + name, timeout=120, follow_redirects=True)
|
|||
|
|
r.raise_for_status()
|
|||
|
|
tmp = path + ".part"
|
|||
|
|
with open(tmp, "wb") as f:
|
|||
|
|
f.write(r.content)
|
|||
|
|
os.replace(tmp, path) # atomic: never leave a truncated file behind
|
|||
|
|
return path
|
|||
|
|
|
|||
|
|
|
|||
|
|
def _load() -> None:
|
|||
|
|
t0 = time.monotonic()
|
|||
|
|
try:
|
|||
|
|
os.makedirs(GEO_DIR, exist_ok=True)
|
|||
|
|
# Admin-division and country display names ("US.WA" → Washington).
|
|||
|
|
admin1 = {}
|
|||
|
|
with open(_fetch("admin1CodesASCII.txt"), encoding="utf-8") as f:
|
|||
|
|
for line in f:
|
|||
|
|
cols = line.rstrip("\n").split("\t")
|
|||
|
|
if len(cols) >= 2:
|
|||
|
|
admin1[cols[0]] = cols[1]
|
|||
|
|
countries = {}
|
|||
|
|
with open(_fetch("countryInfo.txt"), encoding="utf-8") as f:
|
|||
|
|
for line in f:
|
|||
|
|
if line.startswith("#"):
|
|||
|
|
continue
|
|||
|
|
cols = line.split("\t")
|
|||
|
|
if len(cols) >= 5:
|
|||
|
|
countries[cols[0]] = cols[4]
|
|||
|
|
entries = []
|
|||
|
|
with zipfile.ZipFile(_fetch(f"{CITIES}.zip")) as z, z.open(f"{CITIES}.txt") as f:
|
|||
|
|
for raw in f:
|
|||
|
|
cols = raw.decode("utf-8").rstrip("\n").split("\t")
|
|||
|
|
if len(cols) < 15:
|
|||
|
|
continue
|
|||
|
|
name, ascii_name, cc, a1 = cols[1], cols[2], cols[8], cols[10]
|
|||
|
|
norm = _norm(ascii_name or name)
|
|||
|
|
if len(norm) < 2:
|
|||
|
|
continue
|
|||
|
|
try:
|
|||
|
|
lat, lon, pop = float(cols[4]), float(cols[5]), int(cols[14] or 0)
|
|||
|
|
except ValueError:
|
|||
|
|
continue
|
|||
|
|
entries.append((norm, name, admin1.get(f"{cc}.{a1}"),
|
|||
|
|
countries.get(cc), cc, lat, lon, pop))
|
|||
|
|
global _data
|
|||
|
|
_data = _build(entries)
|
|||
|
|
except Exception as e: # noqa: BLE001 - suggestions degrade to the upstream geocoder
|
|||
|
|
audit.log_event("error", {"phase": "places_load", "error": repr(e),
|
|||
|
|
"seconds": round(time.monotonic() - t0, 1)})
|
|||
|
|
|
|||
|
|
|
|||
|
|
def _build(entries: list) -> tuple:
|
|||
|
|
entries.sort(key=lambda e: e[_POP], reverse=True)
|
|||
|
|
order = sorted(range(len(entries)), key=lambda i: entries[i][0])
|
|||
|
|
names = [entries[i][0] for i in order]
|
|||
|
|
vocab = {}
|
|||
|
|
for e in entries:
|
|||
|
|
for tok in e[0].split():
|
|||
|
|
if len(tok) >= 3 and vocab.get(tok, -1) < e[_POP]:
|
|||
|
|
vocab[tok] = e[_POP]
|
|||
|
|
return (entries, names, order, vocab)
|
|||
|
|
|
|||
|
|
|
|||
|
|
def _norm(s: str) -> str:
|
|||
|
|
"""Lowercased, accent-stripped, punctuation-flattened matching key, so
|
|||
|
|
"Coeur d'alene" finds Cœur d'Alene and "winston salem" Winston-Salem."""
|
|||
|
|
s = unicodedata.normalize("NFKD", s)
|
|||
|
|
s = "".join(c for c in s if not unicodedata.combining(c))
|
|||
|
|
for ch in ("'", "’", "."):
|
|||
|
|
s = s.replace(ch, "")
|
|||
|
|
for ch in ("-", ",", "/"):
|
|||
|
|
s = s.replace(ch, " ")
|
|||
|
|
return " ".join(s.casefold().split())
|
|||
|
|
|
|||
|
|
|
|||
|
|
def _prefix_edit1(q: str, w: str) -> bool:
|
|||
|
|
"""True when `q` is within one edit — a substituted, extra, or missing
|
|||
|
|
letter, or an adjacent swap — of some prefix of `w`. Prefix matching (not
|
|||
|
|
whole-name) so "chicgo" already suggests Chicago mid-typing."""
|
|||
|
|
n = len(q)
|
|||
|
|
i = 0
|
|||
|
|
m = min(n, len(w))
|
|||
|
|
while i < m and q[i] == w[i]:
|
|||
|
|
i += 1
|
|||
|
|
if i == n:
|
|||
|
|
return True # exact prefix, zero edits
|
|||
|
|
if w.startswith(q[i + 1:], i + 1):
|
|||
|
|
return True # one letter substituted
|
|||
|
|
if w.startswith(q[i + 1:], i):
|
|||
|
|
return True # one extra letter typed
|
|||
|
|
if w.startswith(q[i:], i + 1):
|
|||
|
|
return True # one letter missed
|
|||
|
|
return (i + 1 < n and i + 1 < len(w) and q[i] == w[i + 1] and q[i + 1] == w[i]
|
|||
|
|
and w.startswith(q[i + 2:], i + 2)) # adjacent letters swapped
|
|||
|
|
|
|||
|
|
|
|||
|
|
def _within1(a: str, b: str) -> bool:
|
|||
|
|
"""Whole-token Damerau-Levenshtein distance ≤ 1 (used for the vocabulary)."""
|
|||
|
|
la, lb = len(a), len(b)
|
|||
|
|
if abs(la - lb) > 1:
|
|||
|
|
return False
|
|||
|
|
i = 0
|
|||
|
|
m = min(la, lb)
|
|||
|
|
while i < m and a[i] == b[i]:
|
|||
|
|
i += 1
|
|||
|
|
if la == lb:
|
|||
|
|
if i == la:
|
|||
|
|
return True
|
|||
|
|
if a[i + 1:] == b[i + 1:]:
|
|||
|
|
return True # substitution
|
|||
|
|
return (i + 1 < la and a[i] == b[i + 1] and a[i + 1] == b[i]
|
|||
|
|
and a[i + 2:] == b[i + 2:]) # transposition
|
|||
|
|
s, l = (a, b) if la < lb else (b, a)
|
|||
|
|
return s[i:] == l[i + 1:] # insertion/deletion
|
|||
|
|
|
|||
|
|
|
|||
|
|
def _result(e: tuple, match: str) -> dict:
|
|||
|
|
return {"name": e[1], "admin1": e[2], "country": e[3], "country_code": e[4],
|
|||
|
|
"lat": e[5], "lon": e[6], "population": e[_POP], "match": match}
|
|||
|
|
|
|||
|
|
|
|||
|
|
def search(q: str, limit: int = 5) -> list[dict] | None:
|
|||
|
|
"""Top `limit` places matching `q` as a (possibly typo'd) name prefix.
|
|||
|
|
|
|||
|
|
Exact-prefix matches come first, population-descending; remaining slots are
|
|||
|
|
filled with names one edit away (only for queries of 4+ characters — with
|
|||
|
|
fewer, "one letter off" matches everything). Returns None while the index
|
|||
|
|
isn't loaded so the caller can fall back to the upstream geocoder.
|
|||
|
|
"""
|
|||
|
|
data = _data
|
|||
|
|
if data is None:
|
|||
|
|
return None
|
|||
|
|
entries, names, order, _ = data
|
|||
|
|
qn = _norm(q)
|
|||
|
|
if len(qn) < 2:
|
|||
|
|
return []
|
|||
|
|
lo = bisect.bisect_left(names, qn)
|
|||
|
|
hi = bisect.bisect_left(names, qn + "\uffff", lo)
|
|||
|
|
top = heapq.nlargest(limit, range(lo, hi), key=lambda i: entries[order[i]][_POP])
|
|||
|
|
out = [_result(entries[order[i]], "prefix") for i in top]
|
|||
|
|
if len(out) < limit and len(qn) >= 4:
|
|||
|
|
need = limit - len(out)
|
|||
|
|
# A single edit can only touch one of the first two characters, so a
|
|||
|
|
# candidate's first two must overlap the query's — a cheap filter that
|
|||
|
|
# rejects ~85% of the index before the real per-name check. Scanning in
|
|||
|
|
# population order means we can stop at the first `need` hits.
|
|||
|
|
q0, q1 = qn[0], qn[1]
|
|||
|
|
for e in entries:
|
|||
|
|
w = e[0]
|
|||
|
|
if w[0] != q0 and w[0] != q1 and (len(w) < 2 or (w[1] != q0 and w[1] != q1)):
|
|||
|
|
continue
|
|||
|
|
if w.startswith(qn):
|
|||
|
|
continue # already counted in the prefix tier
|
|||
|
|
if _prefix_edit1(qn, w):
|
|||
|
|
out.append(_result(e, "fuzzy"))
|
|||
|
|
need -= 1
|
|||
|
|
if need == 0:
|
|||
|
|
break
|
|||
|
|
return out
|
|||
|
|
|
|||
|
|
|
|||
|
|
def corrections(q: str, max_phrases: int = 3) -> list[str]:
|
|||
|
|
"""Candidate respellings of `q` with one token replaced by a known
|
|||
|
|
place-name token a single edit away — "pest seattle" → "west seattle".
|
|||
|
|
|
|||
|
|
Ranked by how likely the swap is: tokens the vocabulary has never seen get
|
|||
|
|
corrected first, and replacements are ordered by the population of the
|
|||
|
|
biggest place using them ("west" over "wesh"). Callers verify candidates by
|
|||
|
|
actually searching, so a wrong guess only costs one lookup.
|
|||
|
|
"""
|
|||
|
|
data = _data
|
|||
|
|
if data is None:
|
|||
|
|
return []
|
|||
|
|
vocab = data[3]
|
|||
|
|
toks = _norm(q).split()
|
|||
|
|
cands: dict[str, tuple] = {}
|
|||
|
|
for i, t in enumerate(toks):
|
|||
|
|
if len(t) < 3:
|
|||
|
|
continue # respelling 1-2 letter tokens is noise
|
|||
|
|
unknown = t not in vocab
|
|||
|
|
t0, t1 = t[0], t[1]
|
|||
|
|
for v, vpop in vocab.items():
|
|||
|
|
if v[0] != t0 and v[0] != t1 and (len(v) < 2 or (v[1] != t0 and v[1] != t1)):
|
|||
|
|
continue # same first-two-chars filter as search()
|
|||
|
|
if abs(len(v) - len(t)) > 1 or v == t or not _within1(t, v):
|
|||
|
|
continue
|
|||
|
|
phrase = " ".join(toks[:i] + [v] + toks[i + 1:])
|
|||
|
|
score = (unknown, vpop)
|
|||
|
|
if cands.get(phrase, (False, -1)) < score:
|
|||
|
|
cands[phrase] = score
|
|||
|
|
ranked = sorted(cands, key=cands.get, reverse=True)
|
|||
|
|
return ranked[:max_phrases]
|