6060 CONTRADICT ,
6161 NOTFOUND ,
6262 Candidate ,
63- WikipediaFetcher ,
6463 _heading_matches ,
6564 normalize_heading ,
6665)
9190 ("asus" , "ROG_Phone" , "ROG Phone" ),
9291)
9392
93+ TABLET_CROSSREF_PAGES : tuple [tuple [str , str , str ], ...] = (
94+ ("apple" , "List_of_iPad_models" , "List of iPad models" ),
95+ ("samsung" , "Samsung_Galaxy_Tab" , "Samsung Galaxy Tab" ),
96+ ("google" , "Google_Pixel_Tablet" , "Google Pixel Tablet" ),
97+ )
98+
9499_BRAND_TOKENS = (
95100 "samsung" ,
96101 "apple" ,
@@ -912,13 +917,13 @@ def append_wikipedia_source(path: Path, url: str) -> str:
912917 return "written"
913918
914919
915- def smartphone_scan_root (data_root : Path ) -> tuple [Path , Path ]:
920+ def smartphone_scan_root (data_root : Path , category : str = "smartphone" ) -> tuple [Path , Path ]:
916921 for cand in (data_root , data_root / "data" ):
917- direct = cand / "smartphone"
922+ direct = cand / category
918923 if direct .is_dir ():
919924 parent = data_root .parent if data_root .name == "data" else data_root
920925 return direct , parent
921- raise SystemExit (f"no data/smartphone directory under { data_root } " )
926+ raise SystemExit (f"no data/{ category } directory under { data_root } " )
922927
923928
924929def iter_smartphone_records (
@@ -1034,7 +1039,28 @@ def search(self, name: str) -> list[Candidate]:
10341039 if name in self ._search_cache :
10351040 return self ._search_cache [name ]
10361041 self ._pause ()
1037- res = WikipediaFetcher (timeout = self .timeout , limit = 5 ).search (name )
1042+ if self ._client is None :
1043+ self ._client = httpx .Client (
1044+ timeout = self .timeout ,
1045+ headers = {"User-Agent" : USER_AGENT },
1046+ follow_redirects = True ,
1047+ )
1048+ try :
1049+ response = self ._client .get (
1050+ "https://en.wikipedia.org/w/rest.php/v1/search/page" ,
1051+ params = {"q" : name , "limit" : 5 },
1052+ )
1053+ response .raise_for_status ()
1054+ res = [
1055+ Candidate (
1056+ title = page ["title" ],
1057+ url = f"https://en.wikipedia.org/wiki/{ quote (page ['key' ], safe = '()_' )} " ,
1058+ )
1059+ for page in response .json ().get ("pages" , [])
1060+ if isinstance (page .get ("title" ), str ) and isinstance (page .get ("key" ), str )
1061+ ]
1062+ except (httpx .HTTPError , ValueError , KeyError , TypeError ):
1063+ res = []
10381064 self ._search_cache [name ] = res
10391065 return res
10401066
@@ -1154,7 +1180,10 @@ def backfill(
11541180 search_fn : SearchFn | None = None ,
11551181 records : list [tuple [str , dict [str , Any ]]] | None = None ,
11561182 cache_path : Path | None = None ,
1183+ category : str = "smartphone" ,
11571184) -> RunResult :
1185+ if category not in {"smartphone" , "tablet" }:
1186+ raise ValueError (f"unsupported category: { category } " )
11581187 writing = apply and not dry_run
11591188 html_cache = data_root / "data" / "_verify" / "cache" / "wikipedia_html"
11601189 polite = PoliteWiki (sleep_s = sleep_s , cache_dir = html_cache ) if fetch_page is None else None
@@ -1169,7 +1198,9 @@ def backfill(
11691198 parsed_pages : set [str ] = set ()
11701199
11711200 # Pre-parse list pages
1172- target_pages = pages if pages is not None else CROSSREF_PAGES
1201+ target_pages = pages if pages is not None else (
1202+ TABLET_CROSSREF_PAGES if category == "tablet" else CROSSREF_PAGES
1203+ )
11731204 for _mfg , page , _title in target_pages :
11741205 status , final , html = fetch (page )
11751206 _alive , reason = classify (f"https://en.wikipedia.org/wiki/{ page } " , status , final or None )
@@ -1226,9 +1257,31 @@ def consider_fallback(base: str, record_brand: str = "") -> None:
12261257 # Load candidate records
12271258 repo_root : Path | None = None
12281259 if records is None :
1229- phone_dir , repo_root = smartphone_scan_root (data_root )
1260+ phone_dir , repo_root = smartphone_scan_root (data_root , category )
12301261 chosen = sample_diverse_records (phone_dir , repo_root , limit , exclude_paths = cached_paths )
1231- eligible_count = 73465
1262+ if category == "tablet" and limit is not None and len (chosen ) < limit :
1263+ remaining = sample_diverse_records (
1264+ phone_dir ,
1265+ repo_root ,
1266+ None ,
1267+ exclude_paths = cached_paths | {rel for rel , _record in chosen },
1268+ )
1269+ chosen_models = {str (record .get ("base_model_slug" )) for _rel , record in chosen }
1270+ novel = [
1271+ (rel , record )
1272+ for rel , record in remaining
1273+ if str (record .get ("base_model_slug" )) not in chosen_models
1274+ ]
1275+ chosen .extend (sample_diverse (novel , limit - len (chosen )))
1276+ if len (chosen ) < limit :
1277+ picked = {rel for rel , _record in chosen }
1278+ chosen .extend (
1279+ sample_diverse (
1280+ [(rel , record ) for rel , record in remaining if rel not in picked ],
1281+ limit - len (chosen ),
1282+ )
1283+ )
1284+ eligible_count = len (chosen )
12321285 else :
12331286 loaded = [
12341287 (rel , rec )
@@ -1272,6 +1325,14 @@ def maybe_write(rel: str, decision: object, url: object) -> None:
12721325 consider_fallback (phone .base , record_brand = rec_b )
12731326 hits = matching_rows (name , fetcher .rows , record_brand = rec_b )
12741327
1328+ if category == "tablet" :
1329+ # A shared substring or a brand-omitted article is insufficient for
1330+ # the tablet batch: the article heading must name this exact model.
1331+ hits = [
1332+ row for row in hits
1333+ if normalize_heading (phone .base ) == normalize_heading (row .model )
1334+ ]
1335+
12751336 live = _liveness_for (hits [0 ].url if hits else None , page_liveness )
12761337 outcome = decide (record , hits , liveness = live if hits else "http-200" )
12771338 entry = cache_entry (rel , outcome , record )
@@ -1294,7 +1355,7 @@ def _only(agreements: list[str], allowed: set[str]) -> bool:
12941355 return bool (agreements ) and set (agreements ) <= allowed
12951356
12961357
1297- def render_summary (result : RunResult , * , dry_run : bool , sleep_s : float ) -> str :
1358+ def render_summary (result : RunResult , * , dry_run : bool , sleep_s : float , category : str = "smartphone" ) -> str :
12981359 counts = result .counts ()
12991360 year_only = [
13001361 row
@@ -1307,7 +1368,7 @@ def render_summary(result: RunResult, *, dry_run: bool, sleep_s: float) -> str:
13071368 spec_conflicts = [row for row in result .rows if row .get ("decision" ) == CONTRADICT ]
13081369
13091370 lines = [
1310- "# Wikipedia Smartphone backfill dry-run" if dry_run else "# Wikipedia Smartphone backfill" ,
1371+ f "# Wikipedia { category . title () } backfill dry-run" if dry_run else f "# Wikipedia { category . title () } backfill" ,
13111372 "" ,
13121373 f"- records processed: **{ len (result .rows ):,} ** across **{ len (result .brands )} ** brands" ,
13131374 f"- total eligible in dataset: { result .eligible :,} " ,
@@ -1350,6 +1411,7 @@ def render_summary(result: RunResult, *, dry_run: bool, sleep_s: float) -> str:
13501411def main (argv : list [str ] | None = None ) -> int :
13511412 parser = argparse .ArgumentParser (description = __doc__ )
13521413 parser .add_argument ("--data-root" , type = Path , default = Path ("." ), help = "TechAPI repository root" )
1414+ parser .add_argument ("--category" , choices = ("smartphone" , "tablet" ), default = "smartphone" )
13531415 parser .add_argument ("--limit" , type = int , default = 300 , help = "Max records to process" )
13541416 parser .add_argument (
13551417 "--sleep" , type = float , default = MIN_SLEEP_S , help = "Sleep between Wikipedia calls"
@@ -1368,7 +1430,7 @@ def main(argv: list[str] | None = None) -> int:
13681430
13691431 dry_run = not args .apply if args .apply else args .dry_run
13701432 cache_path = args .cache or (
1371- args .data_root / "data" / "_verify" / "state" / "wikipedia_smartphone_cache .jsonl"
1433+ args .data_root / "data" / "_verify" / "state" / f"wikipedia_ { args . category } _cache .jsonl"
13721434 )
13731435
13741436 result = backfill (
@@ -1379,8 +1441,9 @@ def main(argv: list[str] | None = None) -> int:
13791441 apply = args .apply ,
13801442 max_fallback = args .max_fallback ,
13811443 cache_path = cache_path ,
1444+ category = args .category ,
13821445 )
1383- print (render_summary (result , dry_run = dry_run , sleep_s = args .sleep ))
1446+ print (render_summary (result , dry_run = dry_run , sleep_s = args .sleep , category = args . category ))
13841447 return 0
13851448
13861449
0 commit comments