Skip to content

Commit 864053d

Browse files
author
Seungpyo1007
committed
feat(verify): gate new-device candidates before they enter device_catalog
python -m app.verify candidates <rows.jsonl> --out <decisions.jsonl> runs Play new-device rows through staged checks (junk, seller, label, dup, in-batch) against the dataset and marks each accept or hold with reasons. Brand-level shares (non-mobile Console rows, rows naming another brand) catch TV vendors and ODMs that per-row rules miss; ids are matched across all brands including variant.model_numbers. Refs #98
1 parent 56a2be4 commit 864053d

3 files changed

Lines changed: 406 additions & 1 deletion

File tree

‎app/verify/catalog_candidates.py‎

Lines changed: 294 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,294 @@
1+
"""Gate new-device candidates (Google Play rows) before they become ``device_catalog`` entries.
2+
3+
A candidate is ``{brand, name, models, codenames}`` plus optional Play Console fields
4+
(``form_factor``, ``soc_raw``, ``screen``, ``sdk_min``). Each one runs through staged checks
5+
and comes out ``accept`` or ``hold`` with the reasons. Nothing is dropped: held rows stay in
6+
the review queue (ADR-016).
7+
8+
Stages, cheapest first:
9+
10+
* ``junk`` — not a phone/tablet/watch: TV/panel SoCs, 4K screens, TV/box/POS names and
11+
codenames, and *vendor mix* — a brand whose Play Console rows are mostly non-mobile
12+
(Hisense, Landi…) cannot vouch for a row that has no Console row of its own.
13+
* ``seller`` — the Play brand is the manufacturer or distributor, not the seller. Measured
14+
per brand: if many of its rows name *another* known brand (Foxconn ships "Kogan…",
15+
Brightstar ships "Alcatel…"), the brand is an ODM and its rows are held. A row that
16+
names exactly one sibling brand (TCL under alcatel, OnePlus under oppo) is refiled.
17+
* ``label`` — the name is not a marketing name: market suffix (``WP33_Pro_EEA`` → the clean
18+
name when ``models`` has it), codename echo (``acer_A12P2``), non-Latin script, and for
19+
major brands a bare model code.
20+
* ``dup`` — the device already exists: exact model number / codename in any record,
21+
top level or ``variant``, any brand; region-normalised ids; the name's tokens all inside
22+
one of the brand's slugs; major brands also by numeric stem and by age (old rows map to
23+
pre-id imports the id join cannot see).
24+
* ``batch`` — two candidates share a normalised model number; the first one wins.
25+
"""
26+
27+
from __future__ import annotations
28+
29+
import re
30+
from collections import Counter, defaultdict
31+
from collections.abc import Iterable
32+
from dataclasses import dataclass, field
33+
from typing import Any
34+
35+
from .common import Record
36+
37+
MAJOR = frozenset({
38+
"samsung", "xiaomi", "huawei", "honor", "oppo", "vivo", "realme", "oneplus", "motorola",
39+
"lg", "sony", "zte", "lenovo", "alcatel", "tcl", "nokia", "hmd", "asus", "google", "htc",
40+
"tecno", "infinix", "itel", "meizu", "sharp", "kyocera", "acer", "blackview", "doogee",
41+
"umidigi", "oukitel", "nubia", "redmi", "poco", "iqoo", "panasonic",
42+
})
43+
# Brands that share devices or file each other's devices on Play.
44+
SIBLINGS: dict[str, frozenset[str]] = {
45+
"alcatel": frozenset({"tcl"}), "tcl": frozenset({"alcatel"}),
46+
"huawei": frozenset({"honor"}), "honor": frozenset({"huawei"}),
47+
"xiaomi": frozenset({"redmi", "poco", "blackshark"}), "redmi": frozenset({"xiaomi"}),
48+
"poco": frozenset({"xiaomi"}), "vivo": frozenset({"iqoo", "jovi"}),
49+
"iqoo": frozenset({"vivo"}), "jovi": frozenset({"vivo"}),
50+
"oppo": frozenset({"realme", "oneplus"}), "realme": frozenset({"oppo"}),
51+
"oneplus": frozenset({"oppo"}), "zte": frozenset({"nubia", "redmagic"}),
52+
"nubia": frozenset({"zte", "redmagic"}), "nokia": frozenset({"hmd"}),
53+
"hmd": frozenset({"nokia"}), "meizu": frozenset({"lynkco"}),
54+
}
55+
JUNK_VENDORS = frozenset({
56+
"pax", "datecs", "imin", "zkteco", "avocor", "kaon", "kaonmedia", "czur", "via-tech",
57+
"horizon", "vios", "prestigio-solutions", "lango", "verifone", "kandao", "idemia", "fbc",
58+
"jimi", "i3-technologies", "i3connect", "onescreen", "clevertouch", "maxhub", "benq",
59+
"innocn", "zaikai", "apolosign", "landi", "newland", "changhong", "dangbei", "gobox",
60+
"linxdot", "nautilus", "micropos", "telpo", "sunmi", "urovo", "castles",
61+
})
62+
ODM_VENDORS = frozenset({
63+
"foxconn", "compal", "anydata", "abocom", "brightstar", "dbm-maroc", "cellon", "enspert",
64+
"hon-hai-precision-industry-co-ltd", "wingtech", "huaqin", "longcheer", "tinno", "coosea",
65+
"lechpol",
66+
})
67+
MOBILE_FORM_FACTORS = frozenset({"Phone", "Tablet", "Wearable"})
68+
JUNK_SOC = re.compile(r"RK3588|Amlogic|AMLA311|A311D|MT8195|MStar|Realtek RTD|MSD6|MT96\d\d", re.I)
69+
JUNK_NAME = re.compile(
70+
r"^(LED\d|LE\d\d|LCD|LC-|HITV|DV\d|IFPD|TH-\d)|4K|\btv\b|\bbox\b|\bstb\b|set.?top|dongle|"
71+
r"translat|dvd|intelliboard|whiteboard|display|signage|kiosk|\bpos\b|terminal|scanner|"
72+
r"printer|projector|treadmill|\bbike\b|miner|\bvr\b|headset|mirage|vidaa|theater|"
73+
r"\bcar\b|automotive|dashcam|\bops\b|meeting|walkman|chromebook|laptop",
74+
re.I,
75+
)
76+
JUNK_CODENAME = re.compile(r"^(rk3588.*|oversea_v|ifpd.*|.*_tv|atv.*|.*stb.*|.*dongle.*)$", re.I)
77+
MARKET = re.compile(r"(?i)[_ ](EEA|EU|ROW|NEU|RU|TUR|GL|US|LATAM|IN|GLOBAL)(?=_|$)")
78+
REGION_ID = re.compile(r"(?i)[_\- ](EEA|ROW|US|EU|RU|TR|UK|IN|GLOBAL|ARG|NEU|LATAM)$")
79+
NOISE = frozenset({
80+
"dual", "sim", "ds", "global", "version", "edition", "the", "mobile", "smartphone",
81+
"phone", "wifi", "wi", "fi", "lte", "4g", "5g", "3g", "nfc", "plus",
82+
})
83+
ODM_SHARE = 0.2 # share of a brand's rows naming another brand that marks it an ODM
84+
MOBILE_SHARE = 0.8 # share of a brand's Console rows that must be mobile to vouch for the rest
85+
OLD_SDK = 25 # Android 7.1: major-brand rows this old predate stored model numbers
86+
87+
88+
def compact(s: str | None) -> str:
89+
return re.sub(r"[^A-Z0-9]", "", (s or "").upper())
90+
91+
92+
def words(s: str | None) -> list[str]:
93+
return [w for w in re.split(r"[\W_]+", (s or "").lower().replace("+", " plus ")) if w]
94+
95+
96+
@dataclass
97+
class Candidate:
98+
brand: str
99+
name: str
100+
models: list[str]
101+
codenames: list[str]
102+
form_factor: str | None = None
103+
soc_raw: str | None = None
104+
screen: str | None = None
105+
sdk_min: int | None = None
106+
extra: dict[str, Any] = field(default_factory=dict)
107+
108+
@classmethod
109+
def from_row(cls, row: dict[str, Any]) -> Candidate:
110+
known = {"brand", "name", "models", "codenames", "all_models", "all_codenames",
111+
"form_factor", "soc_raw", "screen", "sdk_min"}
112+
return cls(
113+
brand=str(row["brand"]), name=str(row["name"]),
114+
models=list(row.get("all_models") or row.get("models") or []),
115+
codenames=list(row.get("all_codenames") or row.get("codenames") or []),
116+
form_factor=row.get("form_factor") or None, soc_raw=row.get("soc_raw"),
117+
screen=row.get("screen"), sdk_min=row.get("sdk_min"),
118+
extra={k: v for k, v in row.items() if k not in known},
119+
)
120+
121+
122+
@dataclass
123+
class Decision:
124+
candidate: Candidate
125+
accept: bool
126+
reasons: list[str]
127+
128+
def as_row(self) -> dict[str, Any]:
129+
c = self.candidate
130+
return {**c.extra, "brand": c.brand, "name": c.name, "all_models": c.models,
131+
"all_codenames": c.codenames, "form_factor": c.form_factor,
132+
"soc_raw": c.soc_raw, "screen": c.screen, "sdk_min": c.sdk_min,
133+
"decision": "accept" if self.accept else "hold", "reasons": self.reasons}
134+
135+
136+
class Index:
137+
"""What the dataset already holds, keyed the ways a Play row can collide with it."""
138+
139+
def __init__(self, records: Iterable[Record], brand_slugs: Iterable[str]) -> None:
140+
self.brands = set(brand_slugs)
141+
self.brand_words = {b for b in self.brands if len(b) >= 4 and "-" not in b}
142+
self.ids: dict[str, str] = {} # compact id -> path, any brand
143+
self.brand_ids: dict[str, set[str]] = defaultdict(set) # region-stripped
144+
self.slugs: dict[str, set[str]] = defaultdict(set)
145+
for r in records:
146+
brand = str(r.data.get("brand") or "")
147+
for slug in (r.slug, r.data.get("base_model_slug")):
148+
if isinstance(slug, str) and slug:
149+
self.slugs[brand].add(slug)
150+
raw_variant = r.data.get("variant")
151+
variant: dict[str, Any] = raw_variant if isinstance(raw_variant, dict) else {}
152+
for key in ("model_numbers", "codenames"):
153+
for src in (r.data.get(key), variant.get(key)):
154+
for i in src if isinstance(src, list) else []:
155+
if isinstance(i, str) and i.strip():
156+
self.ids.setdefault(compact(i), r.path)
157+
self.brand_ids[brand].add(compact(REGION_ID.sub("", i.strip())))
158+
159+
def family(self, brand: str) -> set[str]:
160+
return {brand} | set(SIBLINGS.get(brand, ()))
161+
162+
163+
def brand_profiles(rows: Iterable[Candidate], index: Index) -> dict[str, dict[str, float]]:
164+
"""Per-brand shares measured on *all* Play rows of that brand (not only candidates)."""
165+
total: Counter[str] = Counter()
166+
foreign: Counter[str] = Counter()
167+
console: Counter[str] = Counter()
168+
mobile: Counter[str] = Counter()
169+
for c in rows:
170+
total[c.brand] += 1
171+
named = {w for x in [c.name, *c.models] for w in words(x) if w in index.brand_words}
172+
if named - index.family(c.brand) - set(c.brand.split("-")):
173+
foreign[c.brand] += 1
174+
if c.form_factor is not None:
175+
console[c.brand] += 1
176+
mobile[c.brand] += c.form_factor in MOBILE_FORM_FACTORS
177+
return {
178+
b: {"foreign": foreign[b] / total[b],
179+
"mobile": (mobile[b] / console[b]) if console[b] else 1.0,
180+
"console_rows": float(console[b])}
181+
for b in total
182+
}
183+
184+
185+
def _clean_name(c: Candidate) -> bool:
186+
"""Swap a market-suffixed label for the clean name ``models`` carries; False if none."""
187+
if not MARKET.search(c.name):
188+
return True
189+
clean = MARKET.sub("", c.name).replace("_", " ").strip()
190+
hit = [m for m in c.models if compact(m) == compact(clean)]
191+
if hit:
192+
c.name = hit[0]
193+
return True
194+
return False
195+
196+
197+
def _refile(c: Candidate, index: Index) -> str | None:
198+
"""Return the sibling brand a row really belongs to (TCL under alcatel), if exactly one."""
199+
named = {w for w in words(c.name) if w in index.brands} - {c.brand}
200+
sib = named & set(SIBLINGS.get(c.brand, ()))
201+
return next(iter(sib)) if len(sib) == 1 and named == sib else None
202+
203+
204+
def _junk(c: Candidate, prof: dict[str, float]) -> str | None:
205+
if c.brand in JUNK_VENDORS:
206+
return "junk:vendor"
207+
if c.form_factor is not None and c.form_factor not in MOBILE_FORM_FACTORS:
208+
return "junk:form_factor"
209+
if JUNK_SOC.search(c.soc_raw or "") or c.screen in ("2160x3840", "3840x2160"):
210+
return "junk:soc_or_screen"
211+
if any(JUNK_NAME.search(x) for x in [c.name, *c.models]):
212+
return "junk:name"
213+
if any(JUNK_CODENAME.match(x) for x in c.codenames):
214+
return "junk:codename"
215+
if c.form_factor is None and prof["mobile"] < MOBILE_SHARE:
216+
return "junk:vendor_mix"
217+
return None
218+
219+
220+
def _seller(c: Candidate, prof: dict[str, float], index: Index) -> str | None:
221+
if c.brand in ODM_VENDORS:
222+
return "seller:odm_vendor"
223+
if prof["foreign"] >= ODM_SHARE:
224+
return "seller:odm_share"
225+
named = {w for x in [c.name, *c.models] for w in words(x) if w in index.brand_words}
226+
if named - index.family(c.brand) - set(c.brand.split("-")):
227+
return "seller:names_other_brand"
228+
return None
229+
230+
231+
def _label(c: Candidate) -> str | None:
232+
if not _clean_name(c):
233+
return "label:market_suffix"
234+
if re.search(r"[^\x00-\x7f]", c.name):
235+
return "label:non_latin"
236+
if any(x.lower() in (f"{c.brand}_{c.name}".lower(), c.name.lower())
237+
and not re.search(r"[a-z].*\s", c.name) and c.brand in MAJOR for x in c.codenames):
238+
return "label:codename"
239+
code = r"(?i)(\w+ )?[A-Z]{0,4}[_-]?\d{3,5}[A-Za-z0-9_-]*"
240+
if c.brand in MAJOR and re.fullmatch(code, c.name):
241+
return "label:model_code"
242+
if " / " in c.name or re.search(r"(?i)[a-z]{3,}_[a-z0-9]+_", c.name):
243+
return "label:internal"
244+
return None
245+
246+
247+
def _dup(c: Candidate, index: Index) -> str | None:
248+
fam = index.family(c.brand)
249+
for x in [*c.models, *c.codenames]:
250+
k = compact(x)
251+
if len(k) >= 4 and re.search(r"\d", k) and k in index.ids:
252+
return "dup:id"
253+
if any(compact(REGION_ID.sub("", x.strip())) in index.brand_ids[b] for b in fam):
254+
return "dup:region_id"
255+
toks = [t for t in words(re.sub(r"\([^)]*\)", " ", c.name))
256+
if t not in NOISE and t not in set(c.brand.split("-"))]
257+
if toks:
258+
for b in fam:
259+
for slug in index.slugs.get(b, ()):
260+
parts = set(slug.split("-"))
261+
if all(t in parts for t in toks):
262+
return "dup:name_in_slug"
263+
if c.brand in MAJOR:
264+
if c.sdk_min is not None and c.sdk_min <= OLD_SDK:
265+
return "dup:old_major"
266+
stems = {re.sub(r"[A-Z]+$", "", t) for x in [c.name, *c.models]
267+
for t in re.split(r"[\W_]+", x.upper()) if len(t) >= 3 and re.search(r"\d", t)}
268+
stems = {s for s in stems if len(s) >= 3}
269+
for b in fam:
270+
for slug in index.slugs.get(b, ()):
271+
if stems & {re.sub(r"[A-Z]+$", "", t.upper()) for t in slug.split("-")}:
272+
return "dup:numeric_stem"
273+
return None
274+
275+
276+
def gate(candidates: list[Candidate], all_rows: list[Candidate], index: Index) -> list[Decision]:
277+
"""Run every stage; a candidate is accepted only if no stage objects."""
278+
profiles = brand_profiles(all_rows, index)
279+
seen: set[str] = set()
280+
out: list[Decision] = []
281+
for c in candidates:
282+
moved = _refile(c, index)
283+
if moved:
284+
c.extra["refiled_from"], c.brand = c.brand, moved
285+
prof = profiles.get(c.brand, {"foreign": 0.0, "mobile": 1.0, "console_rows": 0.0})
286+
reasons = [r for r in (_junk(c, prof), _seller(c, prof, index), _label(c), _dup(c, index))
287+
if r]
288+
keys = {compact(REGION_ID.sub("", m.strip())) for m in c.models if len(compact(m)) >= 4}
289+
if not reasons and keys & seen:
290+
reasons.append("batch:same_model")
291+
if not reasons:
292+
seen |= keys
293+
out.append(Decision(c, not reasons, reasons))
294+
return out

‎app/verify/cli.py‎

Lines changed: 29 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -22,7 +22,7 @@
2222

2323
from app.validate import DATA_DIR
2424

25-
from . import crossref, http_check, ledger, offline, promote, wikidata
25+
from . import catalog_candidates, crossref, http_check, ledger, offline, promote, wikidata
2626
from .common import (
2727
CATEGORIES,
2828
SCORES_PATH,
@@ -366,6 +366,28 @@ def cmd_report(args: argparse.Namespace) -> int:
366366
return 0
367367

368368

369+
def cmd_candidates(args: argparse.Namespace) -> int:
370+
"""Gate Play new-device candidates against the dataset (see catalog_candidates)."""
371+
rows = [json.loads(line) for line in args.input.read_text(encoding="utf-8").splitlines()
372+
if line.strip()]
373+
all_rows = [catalog_candidates.Candidate.from_row(r) for r in rows]
374+
cands = [catalog_candidates.Candidate.from_row(r) for r in rows
375+
if r.get("bucket", args.bucket) == args.bucket]
376+
cats = ("brand", "smartphone", "tablet", "watch", "pda", "device_catalog")
377+
records = load_all(cats)
378+
index = catalog_candidates.Index(
379+
(r for c in cats[1:] for r in records[c]), (r.slug for r in records["brand"] if r.slug)
380+
)
381+
decisions = catalog_candidates.gate(cands, all_rows, index)
382+
with args.out.open("w", encoding="utf-8") as f:
383+
for d in decisions:
384+
f.write(json.dumps(d.as_row(), ensure_ascii=False) + "\n")
385+
tally = Counter(d.reasons[0].split(":")[0] if d.reasons else "accept" for d in decisions)
386+
for key, n in tally.most_common():
387+
print(f" {n:>7} {key}")
388+
return 0
389+
390+
369391
def _ranked_unverified(
370392
records: dict[str, list[Record]], soc_release: dict[str, str], now_year: int,
371393
categories: tuple[str, ...],
@@ -812,6 +834,12 @@ def build_parser() -> argparse.ArgumentParser:
812834
pr.add_argument("--max", type=int, default=40, help="cap changed records for network tiers")
813835
pr.set_defaults(func=cmd_pr)
814836

837+
ca = sub.add_parser("candidates", help="gate new-device candidates for device_catalog")
838+
ca.add_argument("input", type=Path, help="JSONL of Play rows (all rows: brand stats use them)")
839+
ca.add_argument("--out", type=Path, required=True, help="JSONL with decision + reasons")
840+
ca.add_argument("--bucket", default="new_catalog", help="rows with this bucket are candidates")
841+
ca.set_defaults(func=cmd_candidates)
842+
815843
return p
816844

817845

0 commit comments

Comments
 (0)