Skip to content

Commit 77c2d54

Browse files
committed
feat(verify): support Wikipedia image backfill across device categories
1 parent 5addd1f commit 77c2d54

2 files changed

Lines changed: 173 additions & 14 deletions

File tree

‎app/verify/wikipedia_image_backfill.py‎

Lines changed: 48 additions & 14 deletions
Original file line numberDiff line numberDiff line change
@@ -1,7 +1,9 @@
11
"""Backfill freely licensed Commons photos from already cited Wikipedia articles.
22
33
Run a dry sample first, then use --apply --offset/--limit for sequential batches.
4-
The append-only decision cache is shared between dry runs and apply runs.
4+
Select data/<category> with --category (default: smartphone).
5+
The append-only decision cache is shared between categories, dry runs and apply
6+
runs; repository-relative paths keep each category's decisions distinct.
57
"""
68

79
from __future__ import annotations
@@ -25,6 +27,7 @@
2527
)
2628

2729
VERSION = 7
30+
CATEGORIES = ("smartphone", "laptop", "monitor", "tablet", "watch", "pda")
2831
BAD_IMAGE = re.compile(
2932
r"(?<![A-Za-z0-9])(?:logo|logotype|wordmark|icon|emblem|flag|symbol|diagram|chart|screenshot|placeholder|render|advertisement|battery|headquarters?|building|campus|series|lineup|packaging|시리즈)(?![A-Za-z0-9])",
3033
re.I,
@@ -38,7 +41,10 @@
3841

3942

4043
def article_url(record: dict[str, Any]) -> str | None:
41-
for url in record.get("source_urls") or []:
44+
sources = record.get("source_urls") or []
45+
if not isinstance(sources, list):
46+
return None
47+
for url in sources:
4248
if not isinstance(url, str):
4349
continue
4450
parsed = urlparse(url)
@@ -53,17 +59,25 @@ def article_url(record: dict[str, Any]) -> str | None:
5359

5460

5561
def eligible(
56-
root: Path, *, include_missing_key: bool = False
62+
root: Path, *, category: str = "smartphone", include_missing_key: bool = False
5763
) -> list[tuple[Path, dict[str, Any], str]]:
64+
if category not in CATEGORIES:
65+
raise ValueError(f"unsupported category: {category}")
5866
rows = []
59-
for path in sorted((root / "data" / "smartphone").rglob("*.json")):
67+
for path in sorted((root / "data" / category).rglob("*.json")):
6068
try:
6169
record = json.loads(path.read_text(encoding="utf-8-sig"))
62-
except (ValueError, OSError):
70+
except (ValueError, OSError) as exc:
71+
print(f"Skipping {path}: unreadable record ({exc})", flush=True)
72+
continue
73+
if not isinstance(record, dict):
74+
print(f"Skipping {path}: record must be a JSON object", flush=True)
6375
continue
64-
if isinstance(record, dict) and (
65-
("image_url" in record and record["image_url"] is None)
66-
or (include_missing_key and "image_url" not in record)
76+
if record.get("source_urls") is not None and not isinstance(record["source_urls"], list):
77+
print(f"Skipping {path}: source_urls must be a list", flush=True)
78+
continue
79+
if ("image_url" in record and record["image_url"] is None) or (
80+
include_missing_key and "image_url" not in record
6781
):
6882
url = article_url(record)
6983
if url:
@@ -269,10 +283,15 @@ def inspect(url: str, fetcher: CommonsFetcher, name: str = "") -> dict[str, str]
269283

270284

271285
def write_image(path: Path, result: dict[str, str]) -> None:
272-
text = path.read_bytes().decode("utf-8")
286+
original = path.read_bytes()
287+
text = original.decode("utf-8-sig")
273288
record = json.loads(text)
289+
if not isinstance(record, dict):
290+
raise ValueError(f"record must be a JSON object in {path}")
274291
if record.get("image_url") is not None:
275292
return
293+
if "image_license" in record or "image_attribution" in record:
294+
raise ValueError(f"existing image metadata in {path}")
276295
newline = "\r\n" if "\r\n" in text else "\n"
277296
replacement = (
278297
'"image_url": ' + json.dumps(result["image_url"], ensure_ascii=False) + ",\n"
@@ -287,20 +306,20 @@ def write_image(path: Path, result: dict[str, str]) -> None:
287306
if count != 1:
288307
raise ValueError(f"missing null image_url in {path}")
289308
else:
290-
if "image_license" in record or "image_attribution" in record:
291-
raise ValueError(f"existing image metadata in {path}")
292309
match = re.match(r'\{(?P<newline>\r?\n)(?P<indent>[ \t]+)(?=")', text)
293310
if match is None:
294311
raise ValueError(f"cannot insert image fields in {path}")
295312
indent = match.group("indent")
296313
fields = replacement.replace(newline + " ", newline + indent)
297314
updated = text[: match.end()] + fields + "," + newline + indent + text[match.end() :]
298-
path.write_bytes(updated.encode("utf-8"))
315+
encoding = "utf-8-sig" if original.startswith(b"\xef\xbb\xbf") else "utf-8"
316+
path.write_bytes(updated.encode(encoding))
299317

300318

301319
def run(
302320
root: Path,
303321
*,
322+
category: str = "smartphone",
304323
offset: int = 0,
305324
limit: int | None = None,
306325
apply: bool = False,
@@ -310,14 +329,23 @@ def run(
310329
) -> list[dict[str, Any]]:
311330
cache_path = cache_path or root / "data" / "_verify" / "state" / "wikipedia_image_cache.jsonl"
312331
cache = load_decisions(cache_path)
313-
rows = eligible(root, include_missing_key=include_missing_key)[
332+
rows = eligible(root, category=category, include_missing_key=include_missing_key)[
314333
offset : None if limit is None else offset + limit
315334
]
316335
fetcher = CommonsFetcher(sleep_s)
317336
results = []
318337
for index, (path, record, article) in enumerate(rows, 1):
319338
rel = path.relative_to(root).as_posix()
320339
decision = cache.get(rel)
340+
if not isinstance(record.get("name"), str) or not record["name"].strip():
341+
decision = {
342+
"path": rel,
343+
"reason": "invalid_record",
344+
"error": "name must be a nonempty string",
345+
}
346+
results.append(decision)
347+
print(f"Skipping {rel}: {decision['error']}", flush=True)
348+
continue
321349
if (
322350
decision is not None
323351
and decision.get("version") == 6
@@ -354,11 +382,15 @@ def run(
354382
if decision["reason"] != "error":
355383
append_cache(decision, cache_path)
356384
if apply and decision["reason"] == "accepted":
357-
write_image(path, decision)
385+
try:
386+
write_image(path, decision)
387+
except (ValueError, OSError) as exc:
388+
decision = dict(decision, reason="invalid_record", error=str(exc))
358389
results.append(decision)
359390
message = (
360391
f"[{index}/{len(rows)}] {decision['reason']}: "
361392
f"{record.get('name')} ({decision.get('file', '')})"
393+
+ (f": {decision['error']}" if decision.get("error") else "")
362394
)
363395
print(message.encode("ascii", "backslashreplace").decode("ascii"), flush=True)
364396
return results
@@ -367,6 +399,7 @@ def run(
367399
def main() -> None:
368400
parser = argparse.ArgumentParser(description=__doc__)
369401
parser.add_argument("--data-root", type=Path, required=True)
402+
parser.add_argument("--category", choices=CATEGORIES, default="smartphone")
370403
parser.add_argument("--offset", type=int, default=0)
371404
parser.add_argument("--limit", type=int)
372405
parser.add_argument("--sleep", type=float, default=1.0)
@@ -375,6 +408,7 @@ def main() -> None:
375408
args = parser.parse_args()
376409
results = run(
377410
args.data_root,
411+
category=args.category,
378412
offset=args.offset,
379413
limit=args.limit,
380414
apply=args.apply,

‎tests/verify/test_wikipedia_image_backfill.py‎

Lines changed: 125 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -3,11 +3,16 @@
33
from __future__ import annotations
44

55
import json
6+
import sys
67
import tempfile
78
from pathlib import Path
89

10+
import pytest
11+
12+
from app.verify import wikipedia_image_backfill as backfill
913
from app.verify.wikipedia_image_backfill import (
1014
BAD_IMAGE,
15+
CATEGORIES,
1116
GROUP_IMAGE,
1217
NON_PHONE_MODEL,
1318
article_url,
@@ -86,6 +91,126 @@ def test_rejects_render_and_nonfree() -> None:
8691
assert license_name(meta("CC-BY-SA-4.0", terms="Non-free media")) is None
8792

8893

94+
@pytest.mark.parametrize("category", CATEGORIES)
95+
@pytest.mark.parametrize("missing_key", [False, True])
96+
def test_category_scan_and_apply(
97+
tmp_path: Path, monkeypatch: pytest.MonkeyPatch, category: str, missing_key: bool
98+
) -> None:
99+
record = {
100+
"name": "Example X123",
101+
"image_url": None,
102+
"source_urls": ["https://en.wikipedia.org/wiki/Example_X123"],
103+
"specification": {"untouched": True},
104+
}
105+
if missing_key:
106+
record.pop("image_url")
107+
for folder in CATEGORIES:
108+
directory = tmp_path / "data" / folder
109+
directory.mkdir(parents=True)
110+
(directory / "example.json").write_text(json.dumps(record, indent=2), encoding="utf-8")
111+
assert [
112+
path.parent.name
113+
for path, _, _ in eligible(tmp_path, category=category, include_missing_key=missing_key)
114+
] == [category]
115+
monkeypatch.setattr(
116+
backfill, "CommonsFetcher", lambda _: FakeFetcher("Example_X123.jpg", meta("CC-BY-4.0"))
117+
)
118+
results = backfill.run(tmp_path, category=category, apply=True, include_missing_key=missing_key)
119+
assert [row["reason"] for row in results] == ["accepted"]
120+
for folder in CATEGORIES:
121+
updated = json.loads((tmp_path / "data" / folder / "example.json").read_text())
122+
assert updated["specification"] == record["specification"]
123+
assert (updated.get("image_url") is not None) == (folder == category)
124+
cache_path = tmp_path / "data" / "_verify" / "state" / "wikipedia_image_cache.jsonl"
125+
assert json.loads(cache_path.read_text())["path"] == f"data/{category}/example.json"
126+
127+
128+
def test_malformed_records_skip_with_reason(tmp_path: Path, capsys: pytest.CaptureFixture) -> None:
129+
directory = tmp_path / "data" / "laptop"
130+
directory.mkdir(parents=True)
131+
for name, record in {
132+
"array": [],
133+
"sources": {"image_url": None, "source_urls": 42},
134+
}.items():
135+
(directory / f"{name}.json").write_text(json.dumps(record), encoding="utf-8")
136+
(directory / "broken.json").write_text("{", encoding="utf-8")
137+
assert eligible(tmp_path, category="laptop") == []
138+
output = capsys.readouterr().out
139+
assert "record must be a JSON object" in output
140+
assert "source_urls must be a list" in output
141+
assert "unreadable record" in output
142+
assert article_url({"source_urls": 42}) is None
143+
with pytest.raises(ValueError, match="unsupported category"):
144+
eligible(tmp_path, category="../outside")
145+
146+
147+
def test_apply_skips_incompatible_record_and_continues(
148+
tmp_path: Path, monkeypatch: pytest.MonkeyPatch
149+
) -> None:
150+
directory = tmp_path / "data" / "watch"
151+
directory.mkdir(parents=True)
152+
record = {
153+
"name": "Example X123",
154+
"image_url": None,
155+
"source_urls": ["https://en.wikipedia.org/wiki/Example_X123"],
156+
}
157+
incompatible = directory / "a.json"
158+
before = json.dumps(dict(record, image_license=None))
159+
incompatible.write_text(before, encoding="utf-8")
160+
(directory / "b.json").write_text(json.dumps(record), encoding="utf-8")
161+
(directory / "c.json").write_text(json.dumps(dict(record, name=None)), encoding="utf-8")
162+
monkeypatch.setattr(
163+
backfill, "CommonsFetcher", lambda _: FakeFetcher("Example_X123.jpg", meta("CC-BY-4.0"))
164+
)
165+
results = backfill.run(tmp_path, category="watch", apply=True)
166+
assert [row["reason"] for row in results] == ["invalid_record", "accepted", "invalid_record"]
167+
assert "existing image metadata" in results[0]["error"]
168+
assert "name must be a nonempty string" in results[2]["error"]
169+
assert incompatible.read_text() == before
170+
171+
172+
@pytest.mark.parametrize("category", [None, "pda"])
173+
def test_cli_selects_category(monkeypatch: pytest.MonkeyPatch, category: str | None) -> None:
174+
calls = []
175+
monkeypatch.setattr(backfill, "run", lambda *args, **kwargs: calls.append(kwargs) or [])
176+
argv = ["backfill", "--data-root", ".", "--apply"]
177+
if category:
178+
argv += ["--category", category]
179+
monkeypatch.setattr(sys, "argv", argv)
180+
backfill.main()
181+
assert calls[0]["category"] == (category or "smartphone")
182+
assert calls[0]["apply"] is True
183+
184+
185+
def test_shared_cache_replays_without_crossing_categories(
186+
tmp_path: Path, monkeypatch: pytest.MonkeyPatch
187+
) -> None:
188+
fetcher = FakeFetcher("Example_X123.jpg", meta("CC-BY-4.0"))
189+
monkeypatch.setattr(backfill, "CommonsFetcher", lambda _: fetcher)
190+
for category in ("laptop", "pda"):
191+
path = tmp_path / "data" / category / "example.json"
192+
path.parent.mkdir(parents=True)
193+
path.write_text(
194+
json.dumps(
195+
{
196+
"name": "Example X123",
197+
"image_url": None,
198+
"source_urls": ["https://en.wikipedia.org/wiki/Example_X123"],
199+
}
200+
),
201+
encoding="utf-8",
202+
)
203+
backfill.run(tmp_path, category=category)
204+
assert fetcher.calls == 4
205+
cache = tmp_path / "data" / "_verify" / "state" / "wikipedia_image_cache.jsonl"
206+
before = cache.read_bytes()
207+
backfill.run(tmp_path, category="laptop", apply=True)
208+
assert fetcher.calls == 4
209+
assert cache.read_bytes() == before
210+
assert json.loads((tmp_path / "data" / "laptop" / "example.json").read_text())["image_url"]
211+
assert json.loads((tmp_path / "data" / "pda" / "example.json").read_text())["image_url"] is None
212+
213+
89214
def test_filename_must_name_device() -> None:
90215
assert filename_matches_model("HONOR Magic6 Pro", "Honor_Magic_6_Pro.jpg")
91216
assert not filename_matches_model("HONOR Magic5 Pro", "Honor_headquarter.jpg")

0 commit comments

Comments
 (0)