thermograph/tests/test_places.py

113 lines
3.9 KiB
Python
Raw Normal View History

import pytest
import places
def _entry(name, admin1, country, cc, lat, lon, pop):
return (places._norm(name), name, admin1, country, cc, lat, lon, pop)
@pytest.fixture
def index(monkeypatch):
entries = [
_entry("Seattle", "Washington", "United States", "US", 47.60, -122.33, 750_000),
_entry("West Seattle", "Washington", "United States", "US", 47.57, -122.39, 30_000),
_entry("SeaTac", "Washington", "United States", "US", 47.44, -122.29, 31_000),
_entry("Portland", "Oregon", "United States", "US", 45.52, -122.68, 650_000),
_entry("Paris", None, "France", "FR", 48.86, 2.35, 2_100_000),
# The index normalizes GeoNames' ASCII name; the display name keeps accents.
(places._norm("Coeur d'Alene"), "Cœur d'Alene", "Idaho", "United States",
"US", 47.68, -116.78, 56_000),
]
monkeypatch.setattr(places, "_data", places._build(entries))
return places
# ---- normalization -------------------------------------------------------------
def test_norm_strips_accents_and_punctuation():
assert places._norm("Coeur d'Alene") == "coeur dalene"
assert places._norm("Winston-Salem") == "winston salem"
assert places._norm(" MÜNCHEN ") == "munchen"
assert places._norm("São Paulo") == "sao paulo"
# ---- one-edit matchers ----------------------------------------------------------
@pytest.mark.parametrize("q,w,hit", [
("chic", "chicago", True), # plain prefix
("chicgo", "chicago", True), # missing letter
("chhicago", "chicago", True), # extra letter
("chixago", "chicago", True), # substituted letter
("cihcago", "chicago", True), # adjacent swap
("chizzgo", "chicago", False), # two edits
("boston", "chicago", False),
])
def test_prefix_edit1(q, w, hit):
assert places._prefix_edit1(q, w) is hit
@pytest.mark.parametrize("a,b,hit", [
("west", "west", True),
("pest", "west", True), # substitution
("wst", "west", True), # deletion
("wesst", "west", True), # insertion
("ewst", "west", True), # transposition
("east", "west", False), # two substitutions
("we", "west", False), # length differs by 2
])
def test_within1(a, b, hit):
assert places._within1(a, b) is hit
# ---- search ----------------------------------------------------------------------
def test_search_returns_none_until_loaded(monkeypatch):
monkeypatch.setattr(places, "_data", None)
assert places.search("seattle") is None
def test_search_prefix_matches_by_population(index):
out = places.search("sea", 5)
assert [r["name"] for r in out] == ["Seattle", "SeaTac"]
assert all(r["match"] == "prefix" for r in out)
def test_search_tolerates_one_typo(index):
out = places.search("seatle", 5)
assert out[0]["name"] == "Seattle"
assert out[0]["match"] == "fuzzy"
def test_search_no_fuzzy_for_short_queries(index):
# 3 letters: "one letter off" would match half the index — prefix only.
assert all(r["match"] == "prefix" for r in places.search("sea", 5))
assert places.search("xea", 5) == []
def test_search_normalizes_the_query(index):
out = places.search("coeur dalene", 5)
assert out and out[0]["name"] == "Cœur d'Alene"
def test_search_result_shape(index):
r = places.search("paris", 1)[0]
assert r == {"name": "Paris", "admin1": None, "country": "France",
"country_code": "FR", "lat": 48.86, "lon": 2.35,
"population": 2_100_000, "match": "prefix"}
# ---- corrections ------------------------------------------------------------------
def test_corrections_respell_a_typod_token(index):
assert "west seattle" in places.corrections("pest seattle")
def test_corrections_empty_until_loaded(monkeypatch):
monkeypatch.setattr(places, "_data", None)
assert places.corrections("pest seattle") == []
def test_corrections_skip_short_tokens(index):
assert places.corrections("st paris") == [] # 1-2 letter tokens are noise