11"""Backfill freely licensed Commons photos from already cited Wikipedia articles.
22
33Run a dry sample first, then use --apply --offset/--limit for sequential batches.
4- The append-only decision cache is shared between dry runs and apply runs.
4+ Select data/<category> with --category (default: smartphone).
5+ The append-only decision cache is shared between categories, dry runs and apply
6+ runs; repository-relative paths keep each category's decisions distinct.
57"""
68
79from __future__ import annotations
2527)
2628
2729VERSION = 7
30+ CATEGORIES = ("smartphone" , "laptop" , "monitor" , "tablet" , "watch" , "pda" )
2831BAD_IMAGE = re .compile (
2932 r"(?<![A-Za-z0-9])(?:logo|logotype|wordmark|icon|emblem|flag|symbol|diagram|chart|screenshot|placeholder|render|advertisement|battery|headquarters?|building|campus|series|lineup|packaging|시리즈)(?![A-Za-z0-9])" ,
3033 re .I ,
3841
3942
4043def article_url (record : dict [str , Any ]) -> str | None :
41- for url in record .get ("source_urls" ) or []:
44+ sources = record .get ("source_urls" ) or []
45+ if not isinstance (sources , list ):
46+ return None
47+ for url in sources :
4248 if not isinstance (url , str ):
4349 continue
4450 parsed = urlparse (url )
@@ -53,17 +59,25 @@ def article_url(record: dict[str, Any]) -> str | None:
5359
5460
5561def eligible (
56- root : Path , * , include_missing_key : bool = False
62+ root : Path , * , category : str = "smartphone" , include_missing_key : bool = False
5763) -> list [tuple [Path , dict [str , Any ], str ]]:
64+ if category not in CATEGORIES :
65+ raise ValueError (f"unsupported category: { category } " )
5866 rows = []
59- for path in sorted ((root / "data" / "smartphone" ).rglob ("*.json" )):
67+ for path in sorted ((root / "data" / category ).rglob ("*.json" )):
6068 try :
6169 record = json .loads (path .read_text (encoding = "utf-8-sig" ))
62- except (ValueError , OSError ):
70+ except (ValueError , OSError ) as exc :
71+ print (f"Skipping { path } : unreadable record ({ exc } )" , flush = True )
72+ continue
73+ if not isinstance (record , dict ):
74+ print (f"Skipping { path } : record must be a JSON object" , flush = True )
6375 continue
64- if isinstance (record , dict ) and (
65- ("image_url" in record and record ["image_url" ] is None )
66- or (include_missing_key and "image_url" not in record )
76+ if record .get ("source_urls" ) is not None and not isinstance (record ["source_urls" ], list ):
77+ print (f"Skipping { path } : source_urls must be a list" , flush = True )
78+ continue
79+ if ("image_url" in record and record ["image_url" ] is None ) or (
80+ include_missing_key and "image_url" not in record
6781 ):
6882 url = article_url (record )
6983 if url :
@@ -269,10 +283,15 @@ def inspect(url: str, fetcher: CommonsFetcher, name: str = "") -> dict[str, str]
269283
270284
271285def write_image (path : Path , result : dict [str , str ]) -> None :
272- text = path .read_bytes ().decode ("utf-8" )
286+ original = path .read_bytes ()
287+ text = original .decode ("utf-8-sig" )
273288 record = json .loads (text )
289+ if not isinstance (record , dict ):
290+ raise ValueError (f"record must be a JSON object in { path } " )
274291 if record .get ("image_url" ) is not None :
275292 return
293+ if "image_license" in record or "image_attribution" in record :
294+ raise ValueError (f"existing image metadata in { path } " )
276295 newline = "\r \n " if "\r \n " in text else "\n "
277296 replacement = (
278297 '"image_url": ' + json .dumps (result ["image_url" ], ensure_ascii = False ) + ",\n "
@@ -287,20 +306,20 @@ def write_image(path: Path, result: dict[str, str]) -> None:
287306 if count != 1 :
288307 raise ValueError (f"missing null image_url in { path } " )
289308 else :
290- if "image_license" in record or "image_attribution" in record :
291- raise ValueError (f"existing image metadata in { path } " )
292309 match = re .match (r'\{(?P<newline>\r?\n)(?P<indent>[ \t]+)(?=")' , text )
293310 if match is None :
294311 raise ValueError (f"cannot insert image fields in { path } " )
295312 indent = match .group ("indent" )
296313 fields = replacement .replace (newline + " " , newline + indent )
297314 updated = text [: match .end ()] + fields + "," + newline + indent + text [match .end () :]
298- path .write_bytes (updated .encode ("utf-8" ))
315+ encoding = "utf-8-sig" if original .startswith (b"\xef \xbb \xbf " ) else "utf-8"
316+ path .write_bytes (updated .encode (encoding ))
299317
300318
301319def run (
302320 root : Path ,
303321 * ,
322+ category : str = "smartphone" ,
304323 offset : int = 0 ,
305324 limit : int | None = None ,
306325 apply : bool = False ,
@@ -310,14 +329,23 @@ def run(
310329) -> list [dict [str , Any ]]:
311330 cache_path = cache_path or root / "data" / "_verify" / "state" / "wikipedia_image_cache.jsonl"
312331 cache = load_decisions (cache_path )
313- rows = eligible (root , include_missing_key = include_missing_key )[
332+ rows = eligible (root , category = category , include_missing_key = include_missing_key )[
314333 offset : None if limit is None else offset + limit
315334 ]
316335 fetcher = CommonsFetcher (sleep_s )
317336 results = []
318337 for index , (path , record , article ) in enumerate (rows , 1 ):
319338 rel = path .relative_to (root ).as_posix ()
320339 decision = cache .get (rel )
340+ if not isinstance (record .get ("name" ), str ) or not record ["name" ].strip ():
341+ decision = {
342+ "path" : rel ,
343+ "reason" : "invalid_record" ,
344+ "error" : "name must be a nonempty string" ,
345+ }
346+ results .append (decision )
347+ print (f"Skipping { rel } : { decision ['error' ]} " , flush = True )
348+ continue
321349 if (
322350 decision is not None
323351 and decision .get ("version" ) == 6
@@ -354,11 +382,15 @@ def run(
354382 if decision ["reason" ] != "error" :
355383 append_cache (decision , cache_path )
356384 if apply and decision ["reason" ] == "accepted" :
357- write_image (path , decision )
385+ try :
386+ write_image (path , decision )
387+ except (ValueError , OSError ) as exc :
388+ decision = dict (decision , reason = "invalid_record" , error = str (exc ))
358389 results .append (decision )
359390 message = (
360391 f"[{ index } /{ len (rows )} ] { decision ['reason' ]} : "
361392 f"{ record .get ('name' )} ({ decision .get ('file' , '' )} )"
393+ + (f": { decision ['error' ]} " if decision .get ("error" ) else "" )
362394 )
363395 print (message .encode ("ascii" , "backslashreplace" ).decode ("ascii" ), flush = True )
364396 return results
@@ -367,6 +399,7 @@ def run(
367399def main () -> None :
368400 parser = argparse .ArgumentParser (description = __doc__ )
369401 parser .add_argument ("--data-root" , type = Path , required = True )
402+ parser .add_argument ("--category" , choices = CATEGORIES , default = "smartphone" )
370403 parser .add_argument ("--offset" , type = int , default = 0 )
371404 parser .add_argument ("--limit" , type = int )
372405 parser .add_argument ("--sleep" , type = float , default = 1.0 )
@@ -375,6 +408,7 @@ def main() -> None:
375408 args = parser .parse_args ()
376409 results = run (
377410 args .data_root ,
411+ category = args .category ,
378412 offset = args .offset ,
379413 limit = args .limit ,
380414 apply = args .apply ,
0 commit comments