import pytest import places def _entry(name, admin1, country, cc, lat, lon, pop): return (places._norm(name), name, admin1, country, cc, lat, lon, pop) @pytest.fixture def index(monkeypatch): entries = [ _entry("Seattle", "Washington", "United States", "US", 47.60, -122.33, 750_000), _entry("West Seattle", "Washington", "United States", "US", 47.57, -122.39, 30_000), _entry("SeaTac", "Washington", "United States", "US", 47.44, -122.29, 31_000), _entry("Portland", "Oregon", "United States", "US", 45.52, -122.68, 650_000), _entry("Paris", None, "France", "FR", 48.86, 2.35, 2_100_000), # The index normalizes GeoNames' ASCII name; the display name keeps accents. (places._norm("Coeur d'Alene"), "Cœur d'Alene", "Idaho", "United States", "US", 47.68, -116.78, 56_000), ] monkeypatch.setattr(places, "_data", places._build(entries)) return places # ---- normalization ------------------------------------------------------------- def test_norm_strips_accents_and_punctuation(): assert places._norm("Coeur d'Alene") == "coeur dalene" assert places._norm("Winston-Salem") == "winston salem" assert places._norm(" MÜNCHEN ") == "munchen" assert places._norm("São Paulo") == "sao paulo" # ---- one-edit matchers ---------------------------------------------------------- @pytest.mark.parametrize("q,w,hit", [ ("chic", "chicago", True), # plain prefix ("chicgo", "chicago", True), # missing letter ("chhicago", "chicago", True), # extra letter ("chixago", "chicago", True), # substituted letter ("cihcago", "chicago", True), # adjacent swap ("chizzgo", "chicago", False), # two edits ("boston", "chicago", False), ]) def test_prefix_edit1(q, w, hit): assert places._prefix_edit1(q, w) is hit @pytest.mark.parametrize("a,b,hit", [ ("west", "west", True), ("pest", "west", True), # substitution ("wst", "west", True), # deletion ("wesst", "west", True), # insertion ("ewst", "west", True), # transposition ("east", "west", False), # two substitutions ("we", "west", False), # length differs by 2 ]) def test_within1(a, b, hit): assert places._within1(a, b) is hit # ---- search ---------------------------------------------------------------------- def test_search_returns_none_until_loaded(monkeypatch): monkeypatch.setattr(places, "_data", None) assert places.search("seattle") is None def test_search_prefix_matches_by_population(index): out = places.search("sea", 5) assert [r["name"] for r in out] == ["Seattle", "SeaTac"] assert all(r["match"] == "prefix" for r in out) def test_search_tolerates_one_typo(index): out = places.search("seatle", 5) assert out[0]["name"] == "Seattle" assert out[0]["match"] == "fuzzy" def test_search_no_fuzzy_for_short_queries(index): # 3 letters: "one letter off" would match half the index — prefix only. assert all(r["match"] == "prefix" for r in places.search("sea", 5)) assert places.search("xea", 5) == [] def test_search_normalizes_the_query(index): out = places.search("coeur dalene", 5) assert out and out[0]["name"] == "Cœur d'Alene" def test_search_result_shape(index): r = places.search("paris", 1)[0] assert r == {"name": "Paris", "admin1": None, "country": "France", "country_code": "FR", "lat": 48.86, "lon": 2.35, "population": 2_100_000, "match": "prefix"} # ---- corrections ------------------------------------------------------------------ def test_corrections_respell_a_typod_token(index): assert "west seattle" in places.corrections("pest seattle") def test_corrections_empty_until_loaded(monkeypatch): monkeypatch.setattr(places, "_data", None) assert places.corrections("pest seattle") == [] def test_corrections_skip_short_tokens(index): assert places.corrections("st paris") == [] # 1-2 letter tokens are noise