38e20f2c30
Correctness: review vetoes persist via a dedupe_veto column (resolve re-runs no longer overturn humans); diff emits second copies whose confident version matches no owned copy (spec: pairs own only on both ids) and fetches the live collection with refresh; resolve pairs titles.json entries to rows by title so a reshoot photo updates provenance instead of duplicating rows; version lookups survive empty /thing results; publisher tie-break now honors the mixed base/expansion veto and refuses multi-candidate picks; empty-normalized (non-Latin) titles never count as exact. Upload: LoginError aborts a run instead of logging N bogus failures (and 3 identical consecutive failures abort as systemic); Cloudflare interstitials are detected; added-without-version gets its own logged status that verify understands; same-game updates run one per pass so the name-targeted row edit can't overwrite a fresh version; absent diff outputs fail loudly; pagination clicks are paced. Web review: a lock serializes freshen/decide (threadpool race dropped decisions); failed saves roll memory back and always alert the browser (non-JSON 500s included); session warnings reach the page instead of a StringIO; state-load failures and dead servers show banners instead of a blank page; duplicate (title, photos) rows are addressable by ordinal. Consistency: shared CONFIDENT_VERSION_STATUSES, client_for(), Config paths for every artifact, one review-port constant, named matching thresholds, strict collection-id parsing, error-doc responses never cached, unknown config keys warn, extract reports dropped vision entries, fixture generators share escaping + marker text. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
240 lines
7.7 KiB
Python
240 lines
7.7 KiB
Python
"""Diff-stage tests: pure compute_diff cases plus the real 2018 collection
|
|
snapshots as parsing fixtures. No network anywhere."""
|
|
|
|
from __future__ import annotations
|
|
|
|
from pathlib import Path
|
|
|
|
from bggpipe.config import Config
|
|
from bggpipe.diff import compute_diff, load_snapshot_collection
|
|
from bggpipe.models import CollectionItem
|
|
|
|
FIXTURES = Path(__file__).parent / "fixtures"
|
|
|
|
|
|
def _item(object_id, coll_id, name="Game", version_id=None, own=True):
|
|
return CollectionItem(
|
|
object_id=object_id,
|
|
coll_id=coll_id,
|
|
name=name,
|
|
subtype="boardgame",
|
|
own=own,
|
|
year=None,
|
|
version_id=version_id,
|
|
)
|
|
|
|
|
|
def _match(
|
|
title, bgg_id="", status="auto", vstatus="version_unknown", vid="", vname=""
|
|
):
|
|
return {
|
|
"title_raw": title,
|
|
"bgg_id": bgg_id,
|
|
"bgg_name": title.title(),
|
|
"year": "2000",
|
|
"type": "boardgame",
|
|
"match_status": status,
|
|
"version_id": vid,
|
|
"version_name": vname,
|
|
"version_status": vstatus,
|
|
"candidates_json": "[]",
|
|
"version_candidates_json": "[]",
|
|
"source_photos": "x.jpg",
|
|
}
|
|
|
|
|
|
# -- snapshot loading ---------------------------------------------------
|
|
|
|
|
|
def test_load_real_snapshots_merges_and_dedupes():
|
|
collection = load_snapshot_collection(FIXTURES)
|
|
# 79 base + 3 expansions, but all 3 expansion collids also appear in
|
|
# the base file -> 79 unique physical copies
|
|
assert len(collection) == 79
|
|
assert len({c.coll_id for c in collection}) == 79
|
|
by_id = {c.object_id: c for c in collection}
|
|
assert by_id[207830].name == "5-Minute Dungeon"
|
|
assert by_id[177].name == "Advanced Civilization"
|
|
# hand-entered in 2018: every entry is version-less (parsed, not assumed)
|
|
assert all(c.version_id is None for c in collection)
|
|
|
|
|
|
# -- compute_diff -------------------------------------------------------
|
|
|
|
|
|
def test_new_game_goes_to_add_with_version():
|
|
result = compute_diff(
|
|
[
|
|
_match(
|
|
"Cat Crimes",
|
|
"235096",
|
|
vstatus="version_auto",
|
|
vid="360982",
|
|
vname="ThinkFun edition",
|
|
)
|
|
],
|
|
[_item(13, 1)],
|
|
)
|
|
(row,) = result.to_add
|
|
assert row["bgg_id"] == "235096"
|
|
assert row["version_id"] == "360982"
|
|
assert not result.to_update
|
|
|
|
|
|
def test_owned_versionless_plus_confident_version_goes_to_update():
|
|
result = compute_diff(
|
|
[
|
|
_match(
|
|
"Britannia",
|
|
"240",
|
|
vstatus="version_auto",
|
|
vid="55555",
|
|
vname="AH English edition",
|
|
)
|
|
],
|
|
[_item(240, 900001, name="Britannia")],
|
|
)
|
|
assert result.already_owned == ["Britannia"]
|
|
(row,) = result.to_update
|
|
assert row == {
|
|
"collid": "900001",
|
|
"bgg_id": "240",
|
|
"bgg_name": "Britannia",
|
|
"version_id": "55555",
|
|
"version_name": "AH English edition",
|
|
}
|
|
assert not result.to_add
|
|
|
|
|
|
def test_owned_with_matching_version_is_just_owned():
|
|
result = compute_diff(
|
|
[_match("Wingspan", "266192", vstatus="version_auto", vid="465063")],
|
|
[_item(266192, 5, version_id=465063)],
|
|
)
|
|
assert result.already_owned == ["Wingspan"]
|
|
assert not result.to_update and not result.to_add
|
|
|
|
|
|
def test_confident_version_matching_no_copy_is_a_second_copy_to_add():
|
|
# Spec: a (bgg_id, version_id) pair is owned only if a collection item
|
|
# matches BOTH. All copies carry different versions -> this is an
|
|
# additional physical copy; existing entries are never edited.
|
|
result = compute_diff(
|
|
[
|
|
_match(
|
|
"Wingspan",
|
|
"266192",
|
|
vstatus="version_auto",
|
|
vid="521212",
|
|
vname="fourth printing",
|
|
)
|
|
],
|
|
[_item(266192, 5, version_id=465063)],
|
|
)
|
|
assert not result.to_update # additive only: never edit a set version
|
|
assert [r["version_id"] for r in result.to_add] == ["521212"]
|
|
assert "fourth printing" in result.second_copies[0]
|
|
assert result.already_owned == []
|
|
|
|
|
|
def test_version_unknown_owned_by_bare_id():
|
|
result = compute_diff(
|
|
[_match("Catan", "13")],
|
|
[_item(13, 1, name="Catan")],
|
|
)
|
|
assert result.already_owned == ["Catan"]
|
|
assert not result.to_add and not result.to_update
|
|
|
|
|
|
def test_two_photo_editions_consume_distinct_collids():
|
|
matches = [
|
|
_match("Cosmic A", "39", vstatus="version_auto", vid="111"),
|
|
_match("Cosmic B", "39", vstatus="version_auto", vid="222"),
|
|
]
|
|
collection = [_item(39, 701), _item(39, 702)]
|
|
result = compute_diff(matches, collection)
|
|
assert {r["collid"] for r in result.to_update} == {"701", "702"}
|
|
assert {r["version_id"] for r in result.to_update} == {"111", "222"}
|
|
|
|
|
|
def test_pending_rejected_and_unseen_are_reported():
|
|
matches = [
|
|
_match("Mystery Spine", status="ambiguous"),
|
|
_match("Junk", status="rejected"),
|
|
_match("Catan", "13"),
|
|
]
|
|
collection = [_item(13, 1, name="Catan"), _item(9209, 2, name="Ticket to Ride")]
|
|
result = compute_diff(matches, collection)
|
|
assert result.pending == ["Mystery Spine"]
|
|
assert result.rejected == 1
|
|
assert [c.object_id for c in result.unseen] == [9209] # informational
|
|
assert result.recognized == 1
|
|
|
|
|
|
def test_merged_rows_are_skipped_but_photos_carry_to_survivor():
|
|
matches = [
|
|
_match("Joking Hazard", "193621"),
|
|
{
|
|
**_match("Jokin Ha...", "193621", status="merged"),
|
|
"merged_into": "Joking Hazard",
|
|
"source_photos": "other.jpg",
|
|
},
|
|
]
|
|
result = compute_diff(matches, []) # empty collection -> to_add
|
|
assert result.merged == 1
|
|
assert result.pending == [] # merged is not "needs review"
|
|
(row,) = result.to_add
|
|
assert row["title_raw"] == "Joking Hazard"
|
|
assert row["source_photos"] == "other.jpg;x.jpg" # combined
|
|
|
|
|
|
def test_versionless_copies_exhaust_then_second_copy_becomes_add():
|
|
# two confident-version matches, ONE versionless copy: the first consumes
|
|
# it (to_update), the second is an additional physical copy (to_add)
|
|
result = compute_diff(
|
|
[
|
|
_match("Sorcerer", "39", vstatus="version_auto", vid="111", vname="1st"),
|
|
_match("Sorcerer", "39", vstatus="version_auto", vid="222", vname="2nd"),
|
|
],
|
|
[_item(39, 701)],
|
|
)
|
|
assert [u["version_id"] for u in result.to_update] == ["111"]
|
|
assert [a["version_id"] for a in result.to_add] == ["222"]
|
|
assert len(result.second_copies) == 1
|
|
|
|
|
|
def test_run_diff_outputs_feed_upload_unchanged(tmp_path, monkeypatch):
|
|
# the cross-stage contract: whatever run_diff writes, run_upload must
|
|
# read — a column rename on either side has to fail HERE
|
|
import shutil
|
|
|
|
from bggpipe.diff import run_diff
|
|
from bggpipe.resolve import write_matches
|
|
from bggpipe.upload import run_upload
|
|
|
|
monkeypatch.delenv("BGG_API_TOKEN", raising=False)
|
|
cfg = Config(data_dir=tmp_path)
|
|
fixtures = Path(__file__).parent / "fixtures"
|
|
for name in (
|
|
"collection_snapshot_base.xml",
|
|
"collection_snapshot_expansions.xml",
|
|
):
|
|
shutil.copy(fixtures / name, tmp_path / name)
|
|
write_matches(
|
|
cfg.matches_path,
|
|
[
|
|
_match("Wingspan", "266192", status="auto"), # not in the snapshots
|
|
_match("5 MINUTE DUNGEON", "207830", status="auto"), # owned
|
|
],
|
|
)
|
|
result = run_diff(cfg)
|
|
assert [r["bgg_id"] for r in result.to_add] == ["266192"]
|
|
|
|
from test_upload import FakeUploader
|
|
|
|
fake = FakeUploader()
|
|
run_upload(cfg, uploader=fake, sleep=lambda s: None, now=lambda: "t")
|
|
assert [(j.action, j.bgg_id, j.name) for j in fake.calls] == [
|
|
("add", "266192", "Wingspan")
|
|
]
|