Files
bggpipe/tests/test_resolve.py
T
Eric Wagoner 65d4cdd5ec Re-audit round 2: 5 blind reviewers, 17 fixes, +12 tests
The re-run confirmed round 1 held and then caught second-order bugs in
its own fixes plus two long-standing ones everyone missed. TUI decisions
after a mid-session reload were counted but never written (rows are now
re-adopted into the fresh list on every save, preferring undecided slots
on duplicate keys); row_ix was computed by equality so duplicate rows
shared an ordinal (identity now, merges included, veto sends it); upload
job keys collided for two same-version copies (completions are counted
per key, so --limit or an interrupt can no longer strand the second
copy); diff consumes collids on exact-version matches (a vetoed
same-version second copy was silently swallowed) and splits mismatches:
report-only disagreement while an unclaimed copy exists, second-copy add
only when every copy is claimed.

Also: XML responses are validated and written atomically before caching
(a torn or truncated 200 body can never poison a re-run), JSON artifacts
write atomically, thing/search parsers refuse missing ids like the
collection parser, empty game names are refused by the upload queue, a
never-rendering version picker fails retryably instead of terminally,
the systemic-failure abort compares exception types, blocked same-title
entries defer as a group so positional pairing can't misalign,
truncation heads pick the earliest separator, diff messages tell the
truth when a token exists without a username, and the shared-constant
sweep now actually covers every module (statuses, search types, marker
names, client_for, ports). pydantic declared as a direct dependency.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-08-02 14:34:44 -04:00

634 lines
21 KiB
Python

"""Resolve-stage tests, replayed from tests/fixtures/bgg_cache — never online.
The transport below raises on any network attempt, proving every response
comes from the committed fixture cache.
"""
from __future__ import annotations
import csv
import json
import shutil
from pathlib import Path
import httpx
import pytest
from bggpipe.bgg_client import BGGClient, cache_key
from bggpipe.config import Config
from bggpipe.models import GameVersion
from bggpipe.normalize import normalize_title
from bggpipe.resolve import (
Candidate,
MatchRow,
TitleEntry,
_dominant,
_score_version,
_truncation_heads,
dedupe_matches,
load_titles,
read_matches,
resolve_entry,
run_resolve,
write_matches,
)
FIXTURES = Path(__file__).parent / "fixtures" / "bgg_cache"
TITLES_JSON = Path(__file__).parent / "data" / "titles.json"
def _no_network(request: httpx.Request) -> httpx.Response:
raise AssertionError(f"test hit the network: {request.url}")
@pytest.fixture
def client() -> BGGClient:
return BGGClient(
cache_dir=FIXTURES,
transport=httpx.MockTransport(_no_network),
sleep=lambda s: pytest.fail("slept during fixture replay"),
)
@pytest.fixture
def rows_by_title(client) -> dict[str, object]:
entries = load_titles(TITLES_JSON)
return {e.title_raw: resolve_entry(client, e) for e in entries}
def test_catan_auto_matches_base_game(rows_by_title):
row = rows_by_title["Catan"]
assert (row.match_status, row.bgg_id, row.bgg_name) == ("auto", 13, "CATAN")
assert row.version_status == "version_unknown" # no cues -> never guess
def test_wingspan_auto_with_version_from_cues(rows_by_title):
row = rows_by_title["Wingspan"]
assert (row.match_status, row.bgg_id) == ("auto", 266192)
# cues: Stonemaier + 2019 + English -> the first edition, not the
# 2020 printing (score 5 vs 3) and not the German edition
assert row.version_status == "version_auto"
assert row.version_id == 465063
assert row.version_name == "English first edition"
def test_expansion_matches_expansion_not_base(rows_by_title):
row = rows_by_title["Wingspan: European Expansion"]
assert (row.match_status, row.bgg_id) == ("auto", 290448)
assert row.type == "boardgameexpansion"
def test_wingspan_europe_spine_never_matches_base(rows_by_title):
"""The spec's canonical trap: a truncated spine must not auto-match."""
row = rows_by_title["Wingspan Europe"]
assert row.match_status == "unmatched"
assert row.bgg_id is None
def test_accented_title_resolves(rows_by_title):
row = rows_by_title["Café International"]
assert (row.match_status, row.bgg_id) == ("auto", 373)
def test_same_name_editions_stay_ambiguous(rows_by_title):
row = rows_by_title["Citadels"]
assert row.match_status == "ambiguous"
assert {c.bgg_id for c in row.candidates} == {478, 205398}
# stats were fetched for the review UI
assert all(c.owned is not None for c in row.candidates)
def test_nonsense_lands_unmatched(rows_by_title):
row = rows_by_title["Blorvath: Quest of the Zzyzx"]
assert row.match_status == "unmatched"
def test_run_resolve_writes_csv_and_is_idempotent(client, tmp_path):
data_dir = tmp_path / "data"
data_dir.mkdir()
shutil.copy(TITLES_JSON, data_dir / "titles.json")
cfg = Config(data_dir=data_dir)
first = run_resolve(cfg, client=client)
assert len(first) == 7
with cfg.matches_path.open(newline="") as f:
rows = list(csv.DictReader(f))
assert len(rows) == 7
by_title = {r["title_raw"]: r for r in rows}
assert by_title["Catan"]["match_status"] == "auto"
assert by_title["Citadels"]["match_status"] == "ambiguous"
candidates = json.loads(by_title["Citadels"]["candidates_json"])
assert len(candidates) == 2
assert by_title["Wingspan"]["version_id"] == "465063"
# second run: everything already in matches.csv is skipped, file unchanged
before = cfg.matches_path.read_text()
second = run_resolve(cfg, client=client)
assert second == []
assert cfg.matches_path.read_text() == before
# -- pure-logic unit tests (no fixtures) --------------------------------
def _cand(bgg_id, exact=True, owned=None, type_="boardgame"):
return Candidate(
bgg_id=bgg_id,
name="X",
year=2000,
type=type_,
exact=exact,
fuzzy=100.0,
owned=owned,
)
def test_dominant_famous_game_beats_obscure_duplicate():
top = [_cand(1, owned=50000), _cand(2, owned=300)]
assert _dominant(top).bgg_id == 1
def test_dominant_refuses_mixed_types():
top = [_cand(1, owned=50000), _cand(2, owned=300, type_="boardgameexpansion")]
assert _dominant(top) is None
def test_dominant_refuses_close_call():
top = [_cand(1, owned=50000), _cand(2, owned=20000)]
assert _dominant(top) is None
def test_score_version_all_cues():
entry = TitleEntry(
title_raw="Wingspan",
title_normalized="wingspan",
publisher_hint="Stonemaier",
year_hint=2019,
language_hint="english",
)
version = GameVersion(
version_id=1,
name="English first edition",
year=2019,
publishers=("Stonemaier Games",),
languages=("English",),
)
assert _score_version(entry, version) == 5
def test_score_version_no_overlap():
entry = TitleEntry(
title_raw="Wingspan",
title_normalized="wingspan",
publisher_hint="Feuerland",
year_hint=2021,
language_hint="german",
)
version = GameVersion(
version_id=1,
name="English first edition",
year=2019,
publishers=("Stonemaier Games",),
languages=("English",),
)
assert _score_version(entry, version) == 0
# -- progressive title truncation (long transcribed box titles) ---------
CIV_TITLE = (
"CIVILIZATION Game of the Heroic Age - The Dawn of History 8000 BC to 250 BC"
)
AH_HINT = "The Avalon Hill Game Company, Baltimore, Maryland"
def _entry(title, **kw):
return TitleEntry(title_raw=title, title_normalized=normalize_title(title), **kw)
def test_truncation_heads_order_and_dedupe():
heads = _truncation_heads(CIV_TITLE)
assert heads[0] == "CIVILIZATION Game of the Heroic Age" # before separator
assert heads[1] == "CIVILIZATION" # before the "Game ..." descriptor
assert len(heads) <= 3
def test_truncation_heads_game_word_without_separator():
assert _truncation_heads("SORCERER The Game of Magical Conflict")[0] == "SORCERER"
def test_truncation_heads_two_word_fallback():
heads = _truncation_heads(
"STARFORCE ALPHA CENTAURI Interstellar Conflict in the 25th Century"
)
assert heads[-1] == "STARFORCE ALPHA"
def test_short_titles_get_no_heads():
assert _truncation_heads("Catan") == []
assert _truncation_heads("Herbaceous") == []
def test_civilization_resolves_via_truncation(client):
entry = _entry(
CIV_TITLE,
publisher_hint=AH_HINT,
edition_hint="Bookcase Game",
language_hint="English",
)
row = resolve_entry(client, entry)
assert (row.match_status, row.bgg_id) == ("auto", 71)
# publisher cue picks the Avalon Hill version, not Hartland/Gibsons
assert row.version_status == "version_auto"
assert row.version_id == 71001
def test_advanced_civilization_resolves_as_expansion(client):
entry = _entry(
"ADVANCED CIVILIZATION Game Expansion of the Heroic Age - "
"Featuring New Civilization, Commodity, and Calamity Cards",
publisher_hint=AH_HINT,
edition_hint="Bookcase Game",
language_hint="English",
)
row = resolve_entry(client, entry)
assert (row.match_status, row.bgg_id, row.type) == (
"auto",
177,
"boardgameexpansion",
)
assert row.version_status == "version_auto"
def test_sorcerer_publisher_tiebreak_beats_owned_dominance(client):
"""Head search finds TWO games named 'Sorcerer' (SPI 1975 vs White
Wizard 2019, which has far more owners) — the SPI publisher cue must
pick the SPI game, and the edition cue the Designer's Edition."""
entry = _entry(
"SORCERER The Game of Magical Conflict",
publisher_hint="Simulations Publications Incorporated (SPI)",
edition_hint="Designer's Edition",
language_hint="English",
)
row = resolve_entry(client, entry)
assert (row.match_status, row.bgg_id) == ("auto", 3585)
assert row.version_status == "version_auto"
assert row.version_name == "SPI Designer's Edition"
def test_starforce_two_word_head_matches_full_title(client):
entry = _entry(
"STARFORCE ALPHA CENTAURI Interstellar Conflict in the 25th Century",
publisher_hint="Simulations Publications Incorporated (SPI)",
edition_hint="Designer's Edition",
language_hint="English",
)
row = resolve_entry(client, entry)
assert (row.match_status, row.bgg_id) == ("auto", 2524)
assert row.version_status == "version_auto"
def test_wrong_year_hint_never_drives_a_version(client):
"""Flat Top's box says 1942 (the theme, not the print year): version
scoring must not pick any version off the back of it."""
entry = _entry("FLAT TOP", year_hint=1942, language_hint="English")
row = resolve_entry(client, entry)
assert (row.match_status, row.bgg_id) == ("auto", 2529)
assert row.version_status == "version_unknown"
assert row.version_id is None
# -- graceful degradation without a BGG token ---------------------------
def test_run_resolve_saves_progress_when_token_missing(tmp_path):
"""Cached titles resolve; uncached ones wait for the token instead of
crashing the run and losing everything."""
import shutil as _shutil
partial_cache = tmp_path / "cache"
partial_cache.mkdir()
for f in FIXTURES.glob("search_query=Catan-*"):
_shutil.copy(f, partial_cache / f.name)
data_dir = tmp_path / "data"
data_dir.mkdir()
(data_dir / "titles.json").write_text(
json.dumps(
[
{"title_raw": "Catan", "source_photos": ["a.jpg"]},
{"title_raw": "Wingspan", "source_photos": ["a.jpg"]},
]
)
)
cfg = Config(data_dir=data_dir)
unauthorized = BGGClient(
cache_dir=partial_cache,
transport=httpx.MockTransport(
lambda req: httpx.Response(401, text="Unauthorized")
),
)
rows = run_resolve(cfg, client=unauthorized)
assert [r.title_raw for r in rows] == ["Catan"] # cached one made it
with cfg.matches_path.open(newline="") as f:
saved = list(csv.DictReader(f))
assert len(saved) == 1 # blocked title left for a future run
# future run (fixtures now "recorded"): picks up only the blocked title
full = BGGClient(
cache_dir=FIXTURES,
transport=httpx.MockTransport(
lambda req: (_ for _ in ()).throw(AssertionError("network"))
),
)
rows2 = run_resolve(cfg, client=full)
assert [r.title_raw for r in rows2] == ["Wingspan"]
# -- post-resolve dedupe ------------------------------------------------
def _mrow(
title,
bgg_id,
photos,
name="Joking Hazard",
vstatus="version_unknown",
vid="",
status="auto",
):
return {
"title_raw": title,
"bgg_id": bgg_id,
"bgg_name": name,
"year": "2016",
"type": "boardgame",
"match_status": status,
"version_id": vid,
"version_name": "",
"version_status": vstatus,
"candidates_json": "[]",
"version_candidates_json": "[]",
"source_photos": photos,
"merged_into": "",
}
def _tentry(title, photos, **cues):
return TitleEntry(
title_raw=title,
title_normalized=normalize_title(title),
publisher_hint=cues.get("publisher_hint", ""),
edition_hint=cues.get("edition_hint", ""),
year_hint=cues.get("year_hint"),
language_hint=cues.get("language_hint", ""),
source_photos=tuple(photos),
)
def test_dedupe_merges_typo_read_into_canonical():
rows = [
_mrow("Jokin Ha...", "193621", "a.jpg"),
_mrow("Joking Hazard", "193621", "b.jpg;c.jpg"),
_mrow("Catan", "13", "d.jpg", name="CATAN"),
]
events = dedupe_matches(rows, [])
(event,) = events
assert event.loser_title == "Jokin Ha..."
assert event.survivor_title == "Joking Hazard" # name-matching read survives
by_title = {r["title_raw"]: r for r in rows}
assert by_title["Jokin Ha..."]["match_status"] == "merged"
assert by_title["Jokin Ha..."]["merged_into"] == "Joking Hazard"
assert by_title["Joking Hazard"]["match_status"] == "auto" # untouched
assert by_title["Catan"]["match_status"] == "auto"
assert len(rows) == 3 # nothing disappears
def test_dedupe_respects_conflicting_edition_cues():
rows = [
_mrow("Cosmic Encounter", "40529", "a.jpg", name="Cosmic Encounter"),
_mrow("COSMIC ENCOUNTER", "40529", "b.jpg", name="Cosmic Encounter"),
]
titles = [
_tentry("Cosmic Encounter", ["a.jpg"], edition_hint="42nd Anniversary Edition"),
_tentry("COSMIC ENCOUNTER", ["b.jpg"], edition_hint="Eon 1977 edition"),
]
assert dedupe_matches(rows, titles) == []
assert all(r["match_status"] == "auto" for r in rows)
def test_dedupe_versions_must_agree():
# different confident versions: two physical editions, never merged
rows = [
_mrow("Wingspan", "266192", "a.jpg", vstatus="version_auto", vid="465063"),
_mrow("WINGSPAN", "266192", "b.jpg", vstatus="version_auto", vid="521212"),
]
assert dedupe_matches(rows, []) == []
# same confident version: same box seen twice
rows2 = [
_mrow("Wingspan", "266192", "a.jpg", vstatus="version_auto", vid="465063"),
_mrow("WINGSPAN", "266192", "b.jpg", vstatus="version_auto", vid="465063"),
]
assert len(dedupe_matches(rows2, [])) == 1
# confident version vs unknown: conservative, no merge
rows3 = [
_mrow("Wingspan", "266192", "a.jpg", vstatus="version_auto", vid="465063"),
_mrow("WINGSPAN", "266192", "b.jpg"),
]
assert dedupe_matches(rows3, []) == []
def test_dedupe_is_idempotent_and_skips_merged():
rows = [
_mrow("Jokin Ha...", "193621", "a.jpg"),
_mrow("Joking Hazard", "193621", "b.jpg"),
]
assert len(dedupe_matches(rows, [])) == 1
assert dedupe_matches(rows, []) == [] # second pass: nothing new
def test_run_resolve_dedupes_and_keeps_all_rows(tmp_path):
cache = tmp_path / "cache"
cache.mkdir()
wingspan_xml = (
'<items total="1"><item type="boardgame" id="266192">'
'<name type="primary" value="Wingspan"/><yearpublished value="2019"/>'
"</item></items>"
)
for query in ("Wingspan", "WINGSPAN!"):
key = cache_key(
"search", {"query": query, "type": "boardgame,boardgameexpansion"}
)
(cache / key).write_text(wingspan_xml)
data_dir = tmp_path / "data"
data_dir.mkdir()
(data_dir / "titles.json").write_text(
json.dumps(
[
{"title_raw": "Wingspan", "source_photos": ["a.jpg"]},
{"title_raw": "WINGSPAN!", "source_photos": ["b.jpg"]}, # variant read
]
)
)
cfg = Config(data_dir=data_dir)
client = BGGClient(cache_dir=cache, transport=httpx.MockTransport(_no_network))
run_resolve(cfg, client=client)
from bggpipe.resolve import read_matches as _rm
saved = {r["title_raw"]: r for r in _rm(cfg.matches_path)}
assert len(saved) == 2 # no row disappeared
assert saved["Wingspan"]["match_status"] == "auto"
assert saved["WINGSPAN!"]["match_status"] == "merged"
assert saved["WINGSPAN!"]["merged_into"] == "Wingspan"
# -- audit-fix regressions ----------------------------------------------
def test_run_resolve_force_rebuilds_from_scratch(client, tmp_path):
data_dir = tmp_path / "data"
data_dir.mkdir()
shutil.copy(TITLES_JSON, data_dir / "titles.json")
cfg = Config(data_dir=data_dir)
run_resolve(cfg, client=client)
# poison one row: force must throw it away and re-resolve everything
rows = read_matches(cfg.matches_path)
rows[0]["match_status"] = "rejected"
write_matches(cfg.matches_path, rows)
forced = run_resolve(cfg, force=True, client=client)
assert len(forced) == 7 # every title re-resolved, none skipped
fresh = {r["title_raw"]: r for r in read_matches(cfg.matches_path)}
assert fresh["Catan"]["match_status"] == "auto"
def test_new_photo_of_resolved_game_updates_row_instead_of_duplicating(
client, tmp_path
):
data_dir = tmp_path / "data"
data_dir.mkdir()
shutil.copy(TITLES_JSON, data_dir / "titles.json")
cfg = Config(data_dir=data_dir)
run_resolve(cfg, client=client)
n_rows = len(read_matches(cfg.matches_path))
# extract sees Catan again on a reshoot photo: the entry's photo set
# grows, its (title, photos) key changes
titles = json.loads((data_dir / "titles.json").read_text())
for entry in titles:
if entry["title_raw"] == "Catan":
entry["source_photos"] = sorted([*entry["source_photos"], "reshoot.jpg"])
(data_dir / "titles.json").write_text(json.dumps(titles))
assert run_resolve(cfg, client=client) == [] # nothing re-resolved
rows = read_matches(cfg.matches_path)
assert len(rows) == n_rows # and no duplicate row appended
catan = next(r for r in rows if r["title_raw"] == "Catan")
assert "reshoot.jpg" in catan["source_photos"] # provenance followed
def test_dedupe_never_overturns_a_human_veto():
a = _mrow("CATAN", "13", "p1.jpg")
b = _mrow("Catan", "13", "p2.jpg", status="approved")
b["dedupe_veto"] = "1" # review said: genuinely two copies
events = dedupe_matches([a, b], [])
assert events == []
assert b["match_status"] == "approved"
def test_publisher_pick_refuses_multiple_same_publisher_candidates():
from bggpipe.resolve import _publisher_pick
entry = TitleEntry(
title_raw="Sorcerer", title_normalized="sorcerer", publisher_hint="SPI"
)
cands = [
_cand(1, exact=True),
_cand(2, exact=True),
]
for c in cands:
c.publishers = ["Simulations Publications, Inc. (SPI)"]
assert _publisher_pick(entry, cands) is None
def test_publisher_pick_refuses_mixed_base_and_expansion():
from bggpipe.resolve import _publisher_pick
entry = TitleEntry(
title_raw="Wingspan", title_normalized="wingspan", publisher_hint="Stonemaier"
)
base = _cand(1, exact=True, type_="boardgame")
expansion = _cand(2, exact=True, type_="boardgameexpansion")
for c in (base, expansion):
c.publishers = ["Stonemaier Games"]
assert _publisher_pick(entry, [base, expansion]) is None
def test_resolve_version_handles_unknown_id():
from bggpipe.resolve import resolve_version
class EmptyThings:
def things(self, ids, **kwargs):
return []
entry = TitleEntry(title_raw="X", title_normalized="x", publisher_hint="Someone")
row = MatchRow(title_raw="X", bgg_id=999999)
resolve_version(EmptyThings(), entry, row) # must not raise
assert row.version_status == "version_unknown"
def test_empty_normalized_title_never_matches(client):
# 风声 normalizes to "" — empty-vs-empty must not count as exact
from bggpipe.resolve import _plausible_candidates
entry = TitleEntry(title_raw="风声", title_normalized="")
# any cached query works; candidates must be rejected regardless of name
assert _plausible_candidates(client, entry, "Catan") == []
def test_blocked_same_title_entry_defers_the_whole_group(tmp_path):
# entry1 of a two-edition title is blocked (no token); entry2 must NOT
# resolve, or its row would occupy entry1's pairing slot next run
import httpx as _httpx
data_dir = tmp_path / "data"
data_dir.mkdir()
(data_dir / "titles.json").write_text(
json.dumps(
[
{
"title_raw": "Catan",
"edition_hint": "3rd edition",
"source_photos": ["a.jpg"],
},
{
"title_raw": "Catan",
"edition_hint": "5th edition",
"source_photos": ["b.jpg"],
},
]
)
)
cfg = Config(data_dir=data_dir)
blocked_client = BGGClient(
cache_dir=tmp_path / "empty_cache",
transport=_httpx.MockTransport(
lambda req: _httpx.Response(401, text="Unauthorized")
),
sleep=lambda s: None,
)
run_resolve(cfg, client=blocked_client)
assert read_matches(cfg.matches_path) == [] # both deferred, none misplaced
def test_truncation_separator_chosen_by_position():
heads = _truncation_heads("Blorvath: Quest of the Zzyzx - 2nd Edition")
assert heads[0] == "Blorvath"
assert "Blorvath: Quest" not in heads # the comment's guarantee, now true