|
| 1 | +"""Read-only, reproducible CPU ingest audit; never writes to the dataset. |
| 2 | +
|
| 3 | +Example:: |
| 4 | +
|
| 5 | + python -m app.ingest.cpu_audit --page List_of_AMD_Ryzen_processors \ |
| 6 | + --page List_of_AMD_Opteron_processors --data-root ../TechAPI/data \ |
| 7 | + --output amd-cpu-dry-run.json |
| 8 | +
|
| 9 | +``--html-dir`` replays previously downloaded HTML instead of fetching pages. |
| 10 | +The JSON includes every proposed record and every unresolved unique model. |
| 11 | +""" |
| 12 | + |
| 13 | +from __future__ import annotations |
| 14 | + |
| 15 | +import argparse |
| 16 | +import hashlib |
| 17 | +import json |
| 18 | +from collections import Counter |
| 19 | +from datetime import UTC, datetime |
| 20 | +from pathlib import Path |
| 21 | + |
| 22 | +from app.coverage.sources.wikipedia import fetch_wikipedia_html |
| 23 | +from app.coverage.sources.wikipedia_cpu import WikipediaCpu |
| 24 | + |
| 25 | +from .pipeline import run |
| 26 | +from .sources.base import IngestCandidate |
| 27 | +from .sources.wikipedia_cpu import PAGES, WikipediaCpuIngest |
| 28 | + |
| 29 | + |
| 30 | +def _entry(candidate: IngestCandidate) -> dict[str, object]: |
| 31 | + return { |
| 32 | + "output_path": candidate.output_path.as_posix(), |
| 33 | + "record": candidate.record, |
| 34 | + "missing_fields": list(candidate.missing_fields), |
| 35 | + } |
| 36 | + |
| 37 | + |
| 38 | +def _coverage_audit( |
| 39 | + html_by_page: dict[str, str], |
| 40 | + candidates: list[IngestCandidate], |
| 41 | + ready: set[str], |
| 42 | + existing: set[str], |
| 43 | +) -> dict[str, object]: |
| 44 | + points = { |
| 45 | + point.slug: point |
| 46 | + for page, html in html_by_page.items() |
| 47 | + for point in WikipediaCpu._extract(html, "amd", page) |
| 48 | + } |
| 49 | + entries = [] |
| 50 | + for slug, point in sorted(points.items()): |
| 51 | + matches = [ |
| 52 | + c |
| 53 | + for c in candidates |
| 54 | + if c.source_url == point.url and (c.slug == slug or c.slug.endswith("-" + slug)) |
| 55 | + ] |
| 56 | + if not any(c.source_url == point.url for c in candidates): |
| 57 | + status = "outside_requested_pages" |
| 58 | + elif any(c.slug in ready for c in matches): |
| 59 | + status = "ready_to_add" |
| 60 | + elif any(c.slug in existing for c in matches): |
| 61 | + status = "already_curated" |
| 62 | + elif matches: |
| 63 | + status = "missing_required_specs" |
| 64 | + else: |
| 65 | + status = "non_model_or_unparsed_cell" |
| 66 | + entries.append( |
| 67 | + { |
| 68 | + "coverage_slug": slug, |
| 69 | + "source_url": point.url, |
| 70 | + "status": status, |
| 71 | + "candidate_slugs": sorted({c.slug for c in matches}), |
| 72 | + "missing_fields": sorted({field for c in matches for field in c.missing_fields}), |
| 73 | + } |
| 74 | + ) |
| 75 | + return { |
| 76 | + "total": len(points), |
| 77 | + "counts": dict(Counter(entry["status"] for entry in entries)), |
| 78 | + "entries": entries, |
| 79 | + } |
| 80 | + |
| 81 | + |
| 82 | +def main(argv: list[str] | None = None) -> int: |
| 83 | + parser = argparse.ArgumentParser(description=__doc__) |
| 84 | + parser.add_argument("--page", action="append", required=True, choices=[p[1] for p in PAGES]) |
| 85 | + parser.add_argument("--data-root", required=True, type=Path) |
| 86 | + parser.add_argument("--output", required=True, type=Path) |
| 87 | + parser.add_argument("--html-dir", type=Path) |
| 88 | + parser.add_argument( |
| 89 | + "--coverage-page", |
| 90 | + action="append", |
| 91 | + default=[], |
| 92 | + help="Optional AMD pages whose raw coverage entries should be reconciled.", |
| 93 | + ) |
| 94 | + args = parser.parse_args(argv) |
| 95 | + if not args.data_root.is_dir(): |
| 96 | + parser.error("--data-root must be an existing TechAPI data directory") |
| 97 | + candidates: list[IngestCandidate] = [] |
| 98 | + sources = [] |
| 99 | + html_by_page: dict[str, str] = {} |
| 100 | + for manufacturer, page, family in PAGES: |
| 101 | + if page not in args.page: |
| 102 | + continue |
| 103 | + html = ( |
| 104 | + (args.html_dir / f"{page}.html").read_text(encoding="utf-8") |
| 105 | + if args.html_dir |
| 106 | + else fetch_wikipedia_html(page) |
| 107 | + ) |
| 108 | + sources.append( |
| 109 | + { |
| 110 | + "url": f"https://en.wikipedia.org/wiki/{page}", |
| 111 | + "html_sha256": hashlib.sha256(html.encode()).hexdigest(), |
| 112 | + } |
| 113 | + ) |
| 114 | + html_by_page[page] = html |
| 115 | + candidates.extend(WikipediaCpuIngest._extract(html, manufacturer, page, family)) |
| 116 | + result = run(candidates, data_root=args.data_root, dry_run=True) |
| 117 | + existing = {c.slug: c for c in result.skipped_existing} |
| 118 | + incomplete = {c.slug: c for c in result.skipped_incomplete if c.slug not in existing} |
| 119 | + missing = Counter(field for c in incomplete.values() for field in c.missing_fields) |
| 120 | + snapshot = hashlib.sha256() |
| 121 | + manufacturers = {c.manufacturer for c in candidates} |
| 122 | + curated_paths = sorted( |
| 123 | + path for maker in manufacturers for path in (args.data_root / "cpu" / maker).rglob("*.json") |
| 124 | + ) |
| 125 | + for path in curated_paths: |
| 126 | + snapshot.update(path.relative_to(args.data_root).as_posix().encode()) |
| 127 | + snapshot.update(b"\0") |
| 128 | + snapshot.update(path.read_bytes()) |
| 129 | + payload = { |
| 130 | + "generated_at": datetime.now(UTC).isoformat(), |
| 131 | + "dry_run": True, |
| 132 | + "include_drafts": False, |
| 133 | + "sources": sources, |
| 134 | + "curated_cpu_snapshot": { |
| 135 | + "records": len(curated_paths), |
| 136 | + "sha256": snapshot.hexdigest(), |
| 137 | + }, |
| 138 | + "counts": { |
| 139 | + "candidate_rows": len(candidates), |
| 140 | + "unique_models": len({c.slug for c in candidates}), |
| 141 | + "would_add": len(result.written), |
| 142 | + "already_existing": len(existing), |
| 143 | + "incomplete": len(incomplete), |
| 144 | + "missing_fields": dict(missing), |
| 145 | + }, |
| 146 | + "would_add": [_entry(c) for c in result.written], |
| 147 | + "already_existing": sorted(existing), |
| 148 | + "incomplete": [ |
| 149 | + {"slug": c.slug, "missing_fields": list(c.missing_fields), "source_url": c.source_url} |
| 150 | + for c in incomplete.values() |
| 151 | + ], |
| 152 | + } |
| 153 | + if args.coverage_page: |
| 154 | + for page in args.coverage_page: |
| 155 | + if page not in html_by_page: |
| 156 | + html_by_page[page] = ( |
| 157 | + (args.html_dir / f"{page}.html").read_text(encoding="utf-8") |
| 158 | + if args.html_dir |
| 159 | + else fetch_wikipedia_html(page) |
| 160 | + ) |
| 161 | + payload["coverage_reconciliation"] = _coverage_audit( |
| 162 | + {page: html_by_page[page] for page in args.coverage_page}, |
| 163 | + candidates, |
| 164 | + {c.slug for c in result.written}, |
| 165 | + set(existing), |
| 166 | + ) |
| 167 | + args.output.parent.mkdir(parents=True, exist_ok=True) |
| 168 | + args.output.write_text( |
| 169 | + json.dumps(payload, indent=2, ensure_ascii=False) + "\n", encoding="utf-8" |
| 170 | + ) |
| 171 | + print(json.dumps(payload["counts"])) |
| 172 | + return 0 |
| 173 | + |
| 174 | + |
| 175 | +if __name__ == "__main__": |
| 176 | + raise SystemExit(main()) |
0 commit comments