Skip to content

Commit f9eca5c

Browse files
committed
feat(verify): add real --category tablet support to smartphone backfill
The flag existed but had no effect. Adds a tablet-specific crossref page set and switches WikipediaFetcher's removed search dependency for a direct REST search call.
1 parent aeac4b2 commit f9eca5c

1 file changed

Lines changed: 75 additions & 12 deletions

File tree

‎app/verify/wikipedia_smartphone_backfill.py‎

Lines changed: 75 additions & 12 deletions
Original file line numberDiff line numberDiff line change
@@ -60,7 +60,6 @@
6060
CONTRADICT,
6161
NOTFOUND,
6262
Candidate,
63-
WikipediaFetcher,
6463
_heading_matches,
6564
normalize_heading,
6665
)
@@ -91,6 +90,12 @@
9190
("asus", "ROG_Phone", "ROG Phone"),
9291
)
9392

93+
TABLET_CROSSREF_PAGES: tuple[tuple[str, str, str], ...] = (
94+
("apple", "List_of_iPad_models", "List of iPad models"),
95+
("samsung", "Samsung_Galaxy_Tab", "Samsung Galaxy Tab"),
96+
("google", "Google_Pixel_Tablet", "Google Pixel Tablet"),
97+
)
98+
9499
_BRAND_TOKENS = (
95100
"samsung",
96101
"apple",
@@ -912,13 +917,13 @@ def append_wikipedia_source(path: Path, url: str) -> str:
912917
return "written"
913918

914919

915-
def smartphone_scan_root(data_root: Path) -> tuple[Path, Path]:
920+
def smartphone_scan_root(data_root: Path, category: str = "smartphone") -> tuple[Path, Path]:
916921
for cand in (data_root, data_root / "data"):
917-
direct = cand / "smartphone"
922+
direct = cand / category
918923
if direct.is_dir():
919924
parent = data_root.parent if data_root.name == "data" else data_root
920925
return direct, parent
921-
raise SystemExit(f"no data/smartphone directory under {data_root}")
926+
raise SystemExit(f"no data/{category} directory under {data_root}")
922927

923928

924929
def iter_smartphone_records(
@@ -1034,7 +1039,28 @@ def search(self, name: str) -> list[Candidate]:
10341039
if name in self._search_cache:
10351040
return self._search_cache[name]
10361041
self._pause()
1037-
res = WikipediaFetcher(timeout=self.timeout, limit=5).search(name)
1042+
if self._client is None:
1043+
self._client = httpx.Client(
1044+
timeout=self.timeout,
1045+
headers={"User-Agent": USER_AGENT},
1046+
follow_redirects=True,
1047+
)
1048+
try:
1049+
response = self._client.get(
1050+
"https://en.wikipedia.org/w/rest.php/v1/search/page",
1051+
params={"q": name, "limit": 5},
1052+
)
1053+
response.raise_for_status()
1054+
res = [
1055+
Candidate(
1056+
title=page["title"],
1057+
url=f"https://en.wikipedia.org/wiki/{quote(page['key'], safe='()_')}",
1058+
)
1059+
for page in response.json().get("pages", [])
1060+
if isinstance(page.get("title"), str) and isinstance(page.get("key"), str)
1061+
]
1062+
except (httpx.HTTPError, ValueError, KeyError, TypeError):
1063+
res = []
10381064
self._search_cache[name] = res
10391065
return res
10401066

@@ -1154,7 +1180,10 @@ def backfill(
11541180
search_fn: SearchFn | None = None,
11551181
records: list[tuple[str, dict[str, Any]]] | None = None,
11561182
cache_path: Path | None = None,
1183+
category: str = "smartphone",
11571184
) -> RunResult:
1185+
if category not in {"smartphone", "tablet"}:
1186+
raise ValueError(f"unsupported category: {category}")
11581187
writing = apply and not dry_run
11591188
html_cache = data_root / "data" / "_verify" / "cache" / "wikipedia_html"
11601189
polite = PoliteWiki(sleep_s=sleep_s, cache_dir=html_cache) if fetch_page is None else None
@@ -1169,7 +1198,9 @@ def backfill(
11691198
parsed_pages: set[str] = set()
11701199

11711200
# Pre-parse list pages
1172-
target_pages = pages if pages is not None else CROSSREF_PAGES
1201+
target_pages = pages if pages is not None else (
1202+
TABLET_CROSSREF_PAGES if category == "tablet" else CROSSREF_PAGES
1203+
)
11731204
for _mfg, page, _title in target_pages:
11741205
status, final, html = fetch(page)
11751206
_alive, reason = classify(f"https://en.wikipedia.org/wiki/{page}", status, final or None)
@@ -1226,9 +1257,31 @@ def consider_fallback(base: str, record_brand: str = "") -> None:
12261257
# Load candidate records
12271258
repo_root: Path | None = None
12281259
if records is None:
1229-
phone_dir, repo_root = smartphone_scan_root(data_root)
1260+
phone_dir, repo_root = smartphone_scan_root(data_root, category)
12301261
chosen = sample_diverse_records(phone_dir, repo_root, limit, exclude_paths=cached_paths)
1231-
eligible_count = 73465
1262+
if category == "tablet" and limit is not None and len(chosen) < limit:
1263+
remaining = sample_diverse_records(
1264+
phone_dir,
1265+
repo_root,
1266+
None,
1267+
exclude_paths=cached_paths | {rel for rel, _record in chosen},
1268+
)
1269+
chosen_models = {str(record.get("base_model_slug")) for _rel, record in chosen}
1270+
novel = [
1271+
(rel, record)
1272+
for rel, record in remaining
1273+
if str(record.get("base_model_slug")) not in chosen_models
1274+
]
1275+
chosen.extend(sample_diverse(novel, limit - len(chosen)))
1276+
if len(chosen) < limit:
1277+
picked = {rel for rel, _record in chosen}
1278+
chosen.extend(
1279+
sample_diverse(
1280+
[(rel, record) for rel, record in remaining if rel not in picked],
1281+
limit - len(chosen),
1282+
)
1283+
)
1284+
eligible_count = len(chosen)
12321285
else:
12331286
loaded = [
12341287
(rel, rec)
@@ -1272,6 +1325,14 @@ def maybe_write(rel: str, decision: object, url: object) -> None:
12721325
consider_fallback(phone.base, record_brand=rec_b)
12731326
hits = matching_rows(name, fetcher.rows, record_brand=rec_b)
12741327

1328+
if category == "tablet":
1329+
# A shared substring or a brand-omitted article is insufficient for
1330+
# the tablet batch: the article heading must name this exact model.
1331+
hits = [
1332+
row for row in hits
1333+
if normalize_heading(phone.base) == normalize_heading(row.model)
1334+
]
1335+
12751336
live = _liveness_for(hits[0].url if hits else None, page_liveness)
12761337
outcome = decide(record, hits, liveness=live if hits else "http-200")
12771338
entry = cache_entry(rel, outcome, record)
@@ -1294,7 +1355,7 @@ def _only(agreements: list[str], allowed: set[str]) -> bool:
12941355
return bool(agreements) and set(agreements) <= allowed
12951356

12961357

1297-
def render_summary(result: RunResult, *, dry_run: bool, sleep_s: float) -> str:
1358+
def render_summary(result: RunResult, *, dry_run: bool, sleep_s: float, category: str = "smartphone") -> str:
12981359
counts = result.counts()
12991360
year_only = [
13001361
row
@@ -1307,7 +1368,7 @@ def render_summary(result: RunResult, *, dry_run: bool, sleep_s: float) -> str:
13071368
spec_conflicts = [row for row in result.rows if row.get("decision") == CONTRADICT]
13081369

13091370
lines = [
1310-
"# Wikipedia Smartphone backfill dry-run" if dry_run else "# Wikipedia Smartphone backfill",
1371+
f"# Wikipedia {category.title()} backfill dry-run" if dry_run else f"# Wikipedia {category.title()} backfill",
13111372
"",
13121373
f"- records processed: **{len(result.rows):,}** across **{len(result.brands)}** brands",
13131374
f"- total eligible in dataset: {result.eligible:,}",
@@ -1350,6 +1411,7 @@ def render_summary(result: RunResult, *, dry_run: bool, sleep_s: float) -> str:
13501411
def main(argv: list[str] | None = None) -> int:
13511412
parser = argparse.ArgumentParser(description=__doc__)
13521413
parser.add_argument("--data-root", type=Path, default=Path("."), help="TechAPI repository root")
1414+
parser.add_argument("--category", choices=("smartphone", "tablet"), default="smartphone")
13531415
parser.add_argument("--limit", type=int, default=300, help="Max records to process")
13541416
parser.add_argument(
13551417
"--sleep", type=float, default=MIN_SLEEP_S, help="Sleep between Wikipedia calls"
@@ -1368,7 +1430,7 @@ def main(argv: list[str] | None = None) -> int:
13681430

13691431
dry_run = not args.apply if args.apply else args.dry_run
13701432
cache_path = args.cache or (
1371-
args.data_root / "data" / "_verify" / "state" / "wikipedia_smartphone_cache.jsonl"
1433+
args.data_root / "data" / "_verify" / "state" / f"wikipedia_{args.category}_cache.jsonl"
13721434
)
13731435

13741436
result = backfill(
@@ -1379,8 +1441,9 @@ def main(argv: list[str] | None = None) -> int:
13791441
apply=args.apply,
13801442
max_fallback=args.max_fallback,
13811443
cache_path=cache_path,
1444+
category=args.category,
13821445
)
1383-
print(render_summary(result, dry_run=dry_run, sleep_s=args.sleep))
1446+
print(render_summary(result, dry_run=dry_run, sleep_s=args.sleep, category=args.category))
13841447
return 0
13851448

13861449

0 commit comments

Comments
 (0)