Files
Scrapyard/scrapers/scraper_cpu.py
2026-04-13 10:36:12 +02:00

759 lines
28 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
"""
CPU enricher (Playwright / Firefox)
======================================
Looks up detailed CPU specs for every cpu_full value found in the laptops
table and stores results in the cpu_specs table.
Sources:
Intel intel.com product spec pages, found via Google search (Playwright)
Apple built-in static data (M1, M1 Pro, M2)
AMD currently skipped (spec pages are bot-protected without a solution)
Usage (from main.py web trigger, or standalone):
from scrapers.scraper_cpu import run
conn = init_db(DB_PATH)
run(conn, reset=False)
"""
import re
import os
import time
import random
import sqlite3
from typing import Optional
from datetime import datetime
from urllib.parse import quote as url_quote
from bs4 import BeautifulSoup
from .shared import log, SCHEMA
# ── Constants ─────────────────────────────────────────────────────────────────
INTEL_BASE = "https://www.intel.com"
GOOGLE_SEARCH = "https://www.google.com/search?q={query}&hl=en"
CPU_DELAY_MIN = 2.0
CPU_DELAY_MAX = 5.0
# Persistent Firefox profile for Playwright cookies/cache survive between
# runs so the browser looks like a returning user, not a fresh automation session.
BROWSER_DATA_DIR = os.path.normpath(
os.path.join(os.path.dirname(os.path.abspath(__file__)), "..", "browser_data")
)
# ── Playwright (deferred import) ──────────────────────────────────────────────
def _import_playwright():
from playwright.sync_api import sync_playwright, BrowserContext
return sync_playwright, BrowserContext
# ── cpu_specs INSERT helpers ──────────────────────────────────────────────────
_CPU_COLS = [
"cpu_key", "cpu_model_raw", "vendor", "full_name", "brand_name",
"codename", "lithography", "launch_date",
"total_cores", "performance_cores", "efficient_cores", "total_threads",
"base_freq_ghz", "max_turbo_ghz",
"l2_cache", "l3_cache",
"tdp_w", "max_tdp_w",
"pcie_version", "pcie_lanes",
"max_ram_gb", "ram_types", "ram_speeds_mhz", "ecc_support",
"igpu_name", "igpu_base_mhz", "igpu_boost_mhz",
"source_url", "lookup_status", "scraped_at",
]
CPU_SPECS_INSERT = """
INSERT OR REPLACE INTO cpu_specs (
cpu_key, cpu_model_raw, vendor, full_name, brand_name,
codename, lithography, launch_date,
total_cores, performance_cores, efficient_cores, total_threads,
base_freq_ghz, max_turbo_ghz,
l2_cache, l3_cache,
tdp_w, max_tdp_w,
pcie_version, pcie_lanes,
max_ram_gb, ram_types, ram_speeds_mhz, ecc_support,
igpu_name, igpu_base_mhz, igpu_boost_mhz,
source_url, lookup_status, scraped_at
) VALUES (
:cpu_key, :cpu_model_raw, :vendor, :full_name, :brand_name,
:codename, :lithography, :launch_date,
:total_cores, :performance_cores, :efficient_cores, :total_threads,
:base_freq_ghz, :max_turbo_ghz,
:l2_cache, :l3_cache,
:tdp_w, :max_tdp_w,
:pcie_version, :pcie_lanes,
:max_ram_gb, :ram_types, :ram_speeds_mhz, :ecc_support,
:igpu_name, :igpu_base_mhz, :igpu_boost_mhz,
:source_url, :lookup_status, :scraped_at
)
"""
def _safe_cpu_row(data: dict) -> dict:
"""Ensure every expected cpu_specs column key exists (default None)."""
return {col: data.get(col) for col in _CPU_COLS}
# ── CPU string normalisation ──────────────────────────────────────────────────
def normalize_cpu_key(cpu_raw: str) -> str:
"""Return a stable lowercase dedup key from any CPU model string."""
s = cpu_raw.strip()
s = re.sub(r'\s+\(?\d+(?:[.,]\d+)?\s*GHz.*', '', s, flags=re.I).strip()
s = re.sub(r'\s+\d+-Core\b.*', '', s, flags=re.I).strip()
s = re.sub(r'\s+\d+P\s+[A-Z]\d+\b.*', '', s, flags=re.I).strip()
for prefix_pattern in (
r'^Intel\s+Core\s+',
r'^Intel\s+',
r'^AMD\s+',
):
s = re.sub(prefix_pattern, '', s, flags=re.I).strip()
s = re.sub(r'\s+', '-', s)
return s.lower()
def detect_vendor(cpu_raw: str) -> str:
if re.search(r'\bApple\b', cpu_raw, re.I) or re.search(r'\bM[123]\b', cpu_raw):
return "Apple"
if re.search(r'\bAMD\b|\bRyzen\b|\bAthlon\b', cpu_raw, re.I):
return "AMD"
return "Intel"
def _intel_search_term(cpu_raw: str) -> str:
# Safety net: truncate any marketing copy that starts after a colon.
s = cpu_raw.split(":")[0].strip()
# Strip trailing "-processor" / "-chip" labels.
s = re.sub(r'[-\s]*(processor|cpu|chip)$', '', s, flags=re.I).strip()
# Drop speed/turbo info — only the model number is needed for the search.
s = re.sub(r'\s+\(?\d+(?:[.,]\d+)?\s*GHz.*', '', s, flags=re.I).strip()
s = re.sub(r'\s+\d+P\s+[A-Z]\d+\b.*', '', s, flags=re.I).strip()
s = re.sub(r'^Intel\s+', '', s, flags=re.I).strip()
return s
# ── Playwright (Firefox) for JS-heavy / bot-protected pages ──────────────────
_pw_instance = None
_pw_context = None # type: ignore[assignment]
def _get_pw_context():
"""Lazy-init a persistent Firefox context stored in browser_data/."""
global _pw_instance, _pw_context
if _pw_context is None:
sync_playwright, _BrowserContext = _import_playwright()
os.makedirs(BROWSER_DATA_DIR, exist_ok=True)
_pw_instance = sync_playwright().__enter__()
_pw_context = _pw_instance.firefox.launch_persistent_context(
user_data_dir=BROWSER_DATA_DIR,
headless=False,
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64; rv:136.0) Gecko/20100101 Firefox/136.0",
viewport={"width": 1280, "height": 900},
locale="en-US",
timezone_id="America/New_York",
firefox_user_prefs={
"dom.webdriver.enabled": False,
"useAutomationExtension": False,
"privacy.resistFingerprinting": False,
"media.navigator.enabled": True,
"geo.enabled": False,
"network.http.referer.sendRefererHeader": 2,
},
)
return _pw_context
def playwright_get(url: str, retries: int = 3) -> Optional[str]:
"""Fetch a page using Firefox (Playwright) with retry and back-off logic."""
for attempt in range(1, retries + 1):
page = None
try:
ctx = _get_pw_context()
page = ctx.new_page()
response = page.goto(url, wait_until="networkidle", timeout=30_000)
status = response.status if response else 200
if status == 200:
return page.content()
if status == 429:
wait = 60 + random.uniform(10, 30)
log.warning(f" Rate-limited (429), sleeping {wait:.0f}s (attempt {attempt}/{retries})")
time.sleep(wait)
elif status == 502:
wait = 180 + random.uniform(0, 20) if attempt == 1 else random.uniform(2, 3)
log.warning(f" HTTP 502 backing off {wait:.0f}s (attempt {attempt}/{retries})")
time.sleep(wait)
elif status in (404, 410):
log.info(f" Not found ({status}): {url}")
return None
else:
log.warning(f" HTTP {status} (attempt {attempt}/{retries}): {url}")
time.sleep(5 * attempt)
except Exception as exc:
log.warning(f" Playwright error attempt {attempt}/{retries} for {url}: {exc}")
time.sleep(10 * attempt)
finally:
if page:
try:
page.close()
except Exception:
pass
log.error(f" Gave up after {retries} attempts: {url}")
return None
def close_playwright() -> None:
global _pw_instance, _pw_context
if _pw_context:
try:
_pw_context.close()
except Exception:
pass
if _pw_instance:
try:
_pw_instance.__exit__(None, None, None)
except Exception:
pass
_pw_instance = _pw_context = None
def _cpu_sleep():
time.sleep(random.uniform(CPU_DELAY_MIN, CPU_DELAY_MAX))
# ── Shared parsing utilities ──────────────────────────────────────────────────
def _parse_freq_ghz(s: str) -> Optional[float]:
m = re.search(r'(\d+(?:[.,]\d+)?)\s*GHz', s, re.I)
return float(m.group(1).replace(',', '.')) if m else None
def _parse_int(s: str) -> Optional[int]:
m = re.search(r'(\d+)', s.replace(',', ''))
return int(m.group(1)) if m else None
def _spec_lookup(specs: dict, pattern: str) -> str:
rx = re.compile(pattern, re.I)
for k, v in specs.items():
if rx.search(k):
return v
return ""
# ── Intel (intel.com product spec pages) ─────────────────────────────────────
def _intel_find_product_url(html: str) -> Optional[str]:
"""Scan a rendered page for the first Intel product spec link."""
soup = BeautifulSoup(html, "lxml")
def _clean(href: str) -> str:
m = re.search(r'/url\?q=(https://www\.intel\.com[^&]+)', href)
if m:
href = m.group(1)
if href.startswith("http"):
return href
return INTEL_BASE + href
for a in soup.find_all("a", href=True):
href = a["href"]
if "/products/sku/" in href and "specifications" in href:
return _clean(href)
for a in soup.find_all("a", href=True):
href = a["href"]
if "/products/sku/" in href and ".html" in href:
cleaned = _clean(href)
base = re.sub(r'/(?:ordering|compatible|downloads|support)\.html$', '', cleaned)
if not base.endswith("specifications.html"):
base = re.sub(r'\.html$', '/specifications.html', base)
return base
for a in soup.find_all("a", href=True):
href = a["href"]
if "/products/sku/" in href:
cleaned = _clean(href)
if not cleaned.endswith("specifications.html"):
cleaned = cleaned.rstrip("/") + "/specifications.html"
return cleaned
return None
def _intel_parse_specs(html: str, url: str) -> dict:
"""Parse an intel.com /products/sku/.../specifications.html page."""
data: dict = {"source_url": url}
soup = BeautifulSoup(html, "lxml")
h1 = soup.select_one("h1")
if h1:
data["full_name"] = h1.get_text(strip=True)
specs: dict[str, str] = {}
for row in soup.select(
".tech-section-row, .specs-list-item, li.tech-section-item, "
".blade-content li, [class*='spec-row']"
):
lbl = row.select_one(".tech-label, .label, span.label")
val = row.select_one(".tech-data, .value, span.value")
if lbl and val:
k = lbl.get_text(strip=True).rstrip(":")
v = val.get_text(" ", strip=True)
if k and v and len(k) < 80:
specs[k] = v
if len(specs) < 5:
for dl in soup.select("dl"):
dts = dl.select("dt")
dds = dl.select("dd")
for dt, dd in zip(dts, dds):
k = dt.get_text(strip=True).rstrip(":")
v = dd.get_text(" ", strip=True)
if k and v and len(k) < 80:
specs[k] = v
if len(specs) < 5:
_INTEL_KNOWN_LABELS = [
"Processor Number", "Product Collection", "Code Name", "Lithography",
"Launch Date", "Vertical Segment", "Marketing Status",
"Total Cores", "Total Threads", "Performance-cores", "Efficient-cores",
"Processor Base Frequency", "Max Turbo Frequency",
"Intel Thermal Velocity Boost Frequency",
"Cache", "TDP", "Configurable TDP-down", "Max Turbo Power",
"Bus Speed",
"Max Memory Size", "Memory Types", "Max # of Memory Channels",
"Max Memory Bandwidth", "ECC Memory Supported",
"GPU Name", "Graphics Base Frequency", "Graphics Max Dynamic Frequency",
"Graphics Burst Frequency",
"PCI Express Revision", "Max # of PCI Express Lanes",
"PCI Express Configurations",
]
label_rx = re.compile(
r'(' + '|'.join(re.escape(l) for l in _INTEL_KNOWN_LABELS) + r')',
re.I
)
for section in soup.select("section, div[class*='section'], div[class*='specs']"):
text = section.get_text(" ", strip=True)
parts = label_rx.split(text)
for j in range(1, len(parts) - 1, 2):
k = parts[j].strip()
v = parts[j + 1].strip().split(" ")[0].strip()
if k and v and len(v) < 200:
specs.setdefault(k, v)
if len(specs) < 5:
for table in soup.select("table"):
for tr in table.select("tr"):
cells = tr.select("td, th")
if len(cells) >= 2:
k = cells[0].get_text(strip=True).rstrip(":")
v = cells[1].get_text(" ", strip=True)
if k and v and len(k) < 80:
specs.setdefault(k, v)
if specs:
_ark_map(specs, data)
return data
def _ark_map(specs: dict, data: dict) -> None:
"""Map flat Intel ARK spec key→value pairs to our cpu_specs columns."""
s = lambda pattern: _spec_lookup(specs, pattern)
cn = s(r'code.?name|product.?collection')
if cn:
data["codename"] = cn
lit = s(r'lithograph|process.?technolog|manufactur')
if lit:
data["lithography"] = lit
ld = s(r'launch.?date|announced|release')
if ld:
data["launch_date"] = ld
pn = s(r'processor.?number|brand.?name')
if pn:
data.setdefault("brand_name", pn)
tc = s(r'#\s*of\s*cores|total\s*cores')
if tc:
data["total_cores"] = _parse_int(tc)
pc = s(r'performance.?core|p.?core')
if pc:
data["performance_cores"] = _parse_int(pc)
ec = s(r'efficient.?core|e.?core')
if ec:
data["efficient_cores"] = _parse_int(ec)
tt = s(r'#\s*of\s*threads|total\s*threads')
if tt:
data["total_threads"] = _parse_int(tt)
base = s(r'processor\s+base\s+freq|base\s+freq|base\s+clock')
if base:
data["base_freq_ghz"] = _parse_freq_ghz(base)
turbo = s(r'max\s*turbo\s*freq|max\s*boost|turbo\s*freq')
if turbo:
data["max_turbo_ghz"] = _parse_freq_ghz(turbo)
l3 = s(r'l3\s*cache|total.*l3|smart\s*cache|^cache$')
if l3:
data["l3_cache"] = l3
l2 = s(r'l2\s*cache|total.*l2')
if l2:
data["l2_cache"] = l2
tdp = s(r'^TDP$|^PBP$|base\s*power|thermal\s*design\s*power')
if tdp:
m = re.search(r'(\d+)', tdp)
if m:
data["tdp_w"] = int(m.group(1))
max_tdp = s(r'max\s*turbo\s*power|^MTP$|maximum\s*turbo\s*power|configurable\s*tdp.?up')
if max_tdp:
m = re.search(r'(\d+)', max_tdp)
if m:
data["max_tdp_w"] = int(m.group(1))
pcie_ver = s(r'pci\s*express\s*revision|pcie\s*version|pci\s*express\s*version')
if pcie_ver:
m = re.search(r'(\d+(?:\.\d+)?)', pcie_ver)
if m:
data["pcie_version"] = m.group(1)
pcie_l = s(r'max.+pci.+lanes|pci.+express.+lanes')
if pcie_l:
m = re.search(r'(\d+)', pcie_l)
if m:
data["pcie_lanes"] = int(m.group(1))
max_mem = s(r'max\s*memory\s*size|maximum\s*memory')
if max_mem:
m = re.search(r'(\d+)\s*GB', max_mem, re.I)
if m:
data["max_ram_gb"] = int(m.group(1))
ram_t = s(r'memory\s*types?\b|supported\s*memory')
if ram_t:
data["ram_types"] = ram_t
ram_spd = s(r'memory\s*speed|memory.*frequency')
if ram_spd:
data["ram_speeds_mhz"] = ram_spd
ecc = s(r'ecc\s*memory|ecc\s*support')
if ecc:
data["ecc_support"] = 1 if re.search(r'yes|supported|✓', ecc, re.I) else 0
igpu = s(r'processor\s*graphics?\b|integrated\s*graphics?')
if igpu:
data["igpu_name"] = igpu
igpu_base = s(r'graphics\s*base\s*freq|gpu\s*base\s*freq')
if igpu_base:
m = re.search(r'(\d+)\s*MHz', igpu_base, re.I)
if m:
data["igpu_base_mhz"] = int(m.group(1))
igpu_boost = s(r'graphics.*dynamic.*freq|graphics.*max.*freq|gpu.*max.*freq')
if igpu_boost:
m = re.search(r'(\d+[.,]\d*)\s*GHz', igpu_boost, re.I)
if m:
data["igpu_boost_mhz"] = int(float(m.group(1).replace(',', '.')) * 1000)
else:
m2 = re.search(r'(\d+)\s*MHz', igpu_boost, re.I)
if m2:
data["igpu_boost_mhz"] = int(m2.group(1))
def _is_captcha_page(page) -> bool:
"""Detect if the current page is a Google CAPTCHA / unusual-traffic page."""
try:
url = page.url
if "sorry/" in url or "sorry.google.com" in url:
return True
content = page.content()
if re.search(r'unusual.{0,40}traffic|g-recaptcha|recaptcha|captcha', content, re.I):
return True
except Exception:
pass
return False
def _google_find_intel_url(term: str) -> Optional[str]:
"""Search Google for the Intel product spec page URL via Playwright.
If Google returns a CAPTCHA or 429, the browser window (already non-headless)
stays open and we wait up to 5 minutes for the user to solve it manually.
"""
# Strip parenthetical annotations like "(4 kärnor / 8 trådar)" that some
# retailers append to CPU names — they break the Google site: search.
clean_term = re.sub(r'\s*\([^)]*\)', '', term).strip()
query = url_quote(f"Intel {clean_term} site:intel.com/content/www/us/en/products/sku")
search_url = GOOGLE_SEARCH.format(query=query)
log.info(f" → Google: {search_url}")
page = None
html = None
try:
ctx = _get_pw_context()
page = ctx.new_page()
response = page.goto(search_url, wait_until="networkidle", timeout=30_000)
status = response.status if response else 200
# Google sometimes returns 429 directly; more often it redirects to
# /sorry/... with a 200 status that contains a CAPTCHA form.
if status == 429 or _is_captcha_page(page):
log.warning("=" * 60)
log.warning(" ⚠ Google CAPTCHA / rate-limit detected!")
log.warning(" Search term: '%s'", term)
log.warning(" The browser window is open — please solve the CAPTCHA manually.")
log.warning(" Waiting up to 5 minutes for you to reach the search results…")
log.warning("=" * 60)
try:
page.wait_for_selector("#search a, #rso a", timeout=300_000)
log.info(" ✓ CAPTCHA solved — resuming.")
except Exception:
log.error(
" Timed out (5 min) waiting for CAPTCHA solution for '%s'. Giving up.", term
)
return None
elif status not in (200, 301, 302):
log.warning(f" Google returned HTTP {status}")
return None
else:
# Normal page — just wait a moment for JS results to settle
try:
page.wait_for_selector("#search a, #rso a", timeout=10_000)
except Exception:
pass
html = page.content()
except Exception as exc:
log.warning(f" Google search error for '{term}': {exc}")
return None
finally:
if page:
try:
page.close()
except Exception:
pass
return _intel_find_product_url(html)
def lookup_intel(cpu_raw: str) -> dict:
"""Fetch Intel product specs from intel.com via Google → Playwright."""
term = _intel_search_term(cpu_raw)
log.info(f" → Intel lookup via Google: '{term}'")
product_url = _google_find_intel_url(term)
_cpu_sleep()
if not product_url:
log.warning(f" No intel.com product URL found for '{term}'")
return {"lookup_status": "failed"}
if "/products/sku/" in product_url and not product_url.endswith("specifications.html"):
product_url = re.sub(
r'/(?:ordering|compatible|downloads|support)(\.html)$', '', product_url
)
if not product_url.endswith(".html"):
product_url += "/specifications.html"
else:
product_url = product_url.replace(".html", "/specifications.html")
log.info(f" → Intel product: {product_url}")
html = playwright_get(product_url)
_cpu_sleep()
if not html:
return {"lookup_status": "failed"}
result = _intel_parse_specs(html, product_url)
result["lookup_status"] = (
"ok" if result.get("total_cores") or result.get("tdp_w") or result.get("full_name")
else "failed"
)
return result
# ── Apple Silicon (static data) ───────────────────────────────────────────────
_APPLE_STATIC: dict[str, dict] = {
"apple-m1": {
"full_name": "Apple M1", "vendor": "Apple",
"codename": "Firestorm / Icestorm (M1)", "lithography": "5nm (TSMC N5)",
"launch_date": "2020-11", "total_cores": 8, "performance_cores": 4,
"efficient_cores": 4, "total_threads": 8, "base_freq_ghz": 3.2,
"max_turbo_ghz": 3.2, "l2_cache": "12 MB (P-core), 4 MB (E-core)",
"l3_cache": "N/A (Unified Memory Architecture)", "tdp_w": 15, "max_tdp_w": 20,
"pcie_version": "4.0", "pcie_lanes": None, "max_ram_gb": 16,
"ram_types": "LPDDR4X (Unified Memory)", "ram_speeds_mhz": "4266",
"ecc_support": 0, "igpu_name": "Apple M1 8-core GPU",
"igpu_base_mhz": None, "igpu_boost_mhz": None,
"source_url": "https://www.apple.com/newsroom/2020/11/apple-unleashes-m1/",
"lookup_status": "ok",
},
"apple-m1-pro": {
"full_name": "Apple M1 Pro", "vendor": "Apple",
"codename": "Avalanche / Blizzard (M1 Pro)", "lithography": "5nm (TSMC N5P)",
"launch_date": "2021-10", "total_cores": 10, "performance_cores": 8,
"efficient_cores": 2, "total_threads": 10, "base_freq_ghz": 3.22,
"max_turbo_ghz": 3.22, "l2_cache": "28 MB (P-core), 4 MB (E-core)",
"l3_cache": "N/A (Unified Memory Architecture)", "tdp_w": 30, "max_tdp_w": 60,
"pcie_version": "4.0", "pcie_lanes": None, "max_ram_gb": 32,
"ram_types": "LPDDR5 (Unified Memory)", "ram_speeds_mhz": "6400",
"ecc_support": 0, "igpu_name": "Apple M1 Pro 16-core GPU",
"igpu_base_mhz": None, "igpu_boost_mhz": None,
"source_url": "https://www.apple.com/newsroom/2021/10/apple-unveils-m1-pro-and-m1-max-supercharged-for-pros/",
"lookup_status": "ok",
},
"apple-m2": {
"full_name": "Apple M2", "vendor": "Apple",
"codename": "Everest / Sawtooth (M2)", "lithography": "5nm 2nd gen (TSMC N5P)",
"launch_date": "2022-06", "total_cores": 8, "performance_cores": 4,
"efficient_cores": 4, "total_threads": 8, "base_freq_ghz": 3.49,
"max_turbo_ghz": 3.49, "l2_cache": "16 MB (P-core), 4 MB (E-core)",
"l3_cache": "N/A (Unified Memory Architecture)", "tdp_w": 15, "max_tdp_w": 20,
"pcie_version": "4.0", "pcie_lanes": None, "max_ram_gb": 24,
"ram_types": "LPDDR5 (Unified Memory)", "ram_speeds_mhz": "6400",
"ecc_support": 0, "igpu_name": "Apple M2 10-core GPU",
"igpu_base_mhz": None, "igpu_boost_mhz": None,
"source_url": "https://www.apple.com/newsroom/2022/06/apple-unveils-m2-with-breakthrough-performance-and-capabilities/",
"lookup_status": "ok",
},
}
def lookup_apple(cpu_key: str) -> dict:
result = _APPLE_STATIC.get(cpu_key)
if result:
log.info(f" → Apple: using static data for '{cpu_key}'")
return dict(result)
log.warning(f" No static data for Apple key '{cpu_key}'")
return {"lookup_status": "failed"}
# ── Entry point ───────────────────────────────────────────────────────────────
def run(conn: sqlite3.Connection, reset: bool = False) -> None:
"""Look up detailed specs for all CPU models found in the laptops table."""
if reset:
conn.execute("DROP TABLE IF EXISTS cpu_specs")
conn.executescript(SCHEMA)
conn.commit()
log.info("cpu_specs table reset.")
log.info("═══ CPU Lookup starting ═══")
rows = conn.execute(
"SELECT DISTINCT cpu_full FROM laptops "
"WHERE cpu_full IS NOT NULL AND cpu_full != '' "
"ORDER BY cpu_full"
).fetchall()
if not rows:
log.warning("No cpu_full values found in laptops table. "
"Has the main scraper finished Phase 2 yet?")
return
key_to_raws: dict[str, list[str]] = {}
for row in rows:
raw = row["cpu_full"].strip()
key = normalize_cpu_key(raw)
key_to_raws.setdefault(key, []).append(raw)
log.info(
f"Found {len(rows)} distinct cpu_full strings "
f"{len(key_to_raws)} unique CPU keys after normalisation"
)
for key, raws in sorted(key_to_raws.items()):
if len(raws) > 1:
log.info(f" Merged {len(raws)} strings → '{key}': {raws}")
done_keys = {
r["cpu_key"]
for r in conn.execute(
"SELECT cpu_key FROM cpu_specs WHERE lookup_status = 'ok'"
).fetchall()
}
todo = {k: v for k, v in key_to_raws.items() if k not in done_keys}
log.info(f"{len(done_keys)} already done, {len(todo)} to look up")
if not todo:
log.info("Nothing to do all CPUs already in cpu_specs.")
close_playwright()
return
total = len(todo)
done_ok = 0
done_fail = 0
for i, (cpu_key, raws) in enumerate(sorted(todo.items()), 1):
model_raw = max(raws, key=len)
vendor = detect_vendor(model_raw)
log.info(f"[{i}/{total}] {cpu_key} vendor={vendor}")
log.info(f" raw: '{model_raw}'")
try:
if vendor == "Apple":
result = lookup_apple(cpu_key)
elif vendor == "AMD":
log.info(f" Skipping AMD CPU (not supported): {cpu_key}")
result = {"lookup_status": "skipped"}
else:
result = lookup_intel(model_raw)
result["cpu_key"] = cpu_key
result["cpu_model_raw"] = model_raw
result.setdefault("vendor", vendor)
result["scraped_at"] = datetime.now().isoformat()
conn.execute(CPU_SPECS_INSERT, _safe_cpu_row(result))
conn.commit()
status = result.get("lookup_status", "?")
if status == "ok":
done_ok += 1
log.info(f"{result.get('full_name', cpu_key)}")
else:
done_fail += 1
log.warning(f" ✗ lookup_status={status} for {cpu_key}")
except Exception as exc:
log.error(f" Exception processing {cpu_key}: {exc}", exc_info=True)
try:
conn.execute(
"""INSERT OR REPLACE INTO cpu_specs
(cpu_key, cpu_model_raw, vendor, scraped_at, lookup_status)
VALUES (?, ?, ?, ?, 'failed')""",
(cpu_key, model_raw, vendor, datetime.now().isoformat()),
)
conn.commit()
except Exception:
pass
done_fail += 1
if vendor != "Apple":
time.sleep(random.uniform(1.0, 2.0))
log.info(
f"═══ CPU Lookup done. ✓ {done_ok} ok ✗ {done_fail} failed "
f"(total processed: {total}) ═══"
)
summary = conn.execute(
"SELECT vendor, lookup_status, COUNT(*) as n "
"FROM cpu_specs GROUP BY vendor, lookup_status ORDER BY vendor, lookup_status"
).fetchall()
log.info("Summary:")
for row in summary:
log.info(f" {row['vendor']:8s} {row['lookup_status']:8s} {row['n']} entries")
close_playwright()