759 lines
28 KiB
Python
759 lines
28 KiB
Python
#!/usr/bin/env python3
|
||
"""
|
||
CPU enricher (Playwright / Firefox)
|
||
======================================
|
||
Looks up detailed CPU specs for every cpu_full value found in the laptops
|
||
table and stores results in the cpu_specs table.
|
||
|
||
Sources:
|
||
Intel – intel.com product spec pages, found via Google search (Playwright)
|
||
Apple – built-in static data (M1, M1 Pro, M2)
|
||
AMD – currently skipped (spec pages are bot-protected without a solution)
|
||
|
||
Usage (from main.py web trigger, or standalone):
|
||
from scrapers.scraper_cpu import run
|
||
conn = init_db(DB_PATH)
|
||
run(conn, reset=False)
|
||
"""
|
||
|
||
import re
|
||
import os
|
||
import time
|
||
import random
|
||
import sqlite3
|
||
from typing import Optional
|
||
from datetime import datetime
|
||
from urllib.parse import quote as url_quote
|
||
|
||
from bs4 import BeautifulSoup
|
||
|
||
from .shared import log, SCHEMA
|
||
|
||
# ── Constants ─────────────────────────────────────────────────────────────────
|
||
|
||
INTEL_BASE = "https://www.intel.com"
|
||
GOOGLE_SEARCH = "https://www.google.com/search?q={query}&hl=en"
|
||
|
||
CPU_DELAY_MIN = 2.0
|
||
CPU_DELAY_MAX = 5.0
|
||
|
||
# Persistent Firefox profile for Playwright – cookies/cache survive between
|
||
# runs so the browser looks like a returning user, not a fresh automation session.
|
||
BROWSER_DATA_DIR = os.path.normpath(
|
||
os.path.join(os.path.dirname(os.path.abspath(__file__)), "..", "browser_data")
|
||
)
|
||
|
||
# ── Playwright (deferred import) ──────────────────────────────────────────────
|
||
|
||
|
||
def _import_playwright():
|
||
from playwright.sync_api import sync_playwright, BrowserContext
|
||
return sync_playwright, BrowserContext
|
||
|
||
|
||
# ── cpu_specs INSERT helpers ──────────────────────────────────────────────────
|
||
|
||
_CPU_COLS = [
|
||
"cpu_key", "cpu_model_raw", "vendor", "full_name", "brand_name",
|
||
"codename", "lithography", "launch_date",
|
||
"total_cores", "performance_cores", "efficient_cores", "total_threads",
|
||
"base_freq_ghz", "max_turbo_ghz",
|
||
"l2_cache", "l3_cache",
|
||
"tdp_w", "max_tdp_w",
|
||
"pcie_version", "pcie_lanes",
|
||
"max_ram_gb", "ram_types", "ram_speeds_mhz", "ecc_support",
|
||
"igpu_name", "igpu_base_mhz", "igpu_boost_mhz",
|
||
"source_url", "lookup_status", "scraped_at",
|
||
]
|
||
|
||
CPU_SPECS_INSERT = """
|
||
INSERT OR REPLACE INTO cpu_specs (
|
||
cpu_key, cpu_model_raw, vendor, full_name, brand_name,
|
||
codename, lithography, launch_date,
|
||
total_cores, performance_cores, efficient_cores, total_threads,
|
||
base_freq_ghz, max_turbo_ghz,
|
||
l2_cache, l3_cache,
|
||
tdp_w, max_tdp_w,
|
||
pcie_version, pcie_lanes,
|
||
max_ram_gb, ram_types, ram_speeds_mhz, ecc_support,
|
||
igpu_name, igpu_base_mhz, igpu_boost_mhz,
|
||
source_url, lookup_status, scraped_at
|
||
) VALUES (
|
||
:cpu_key, :cpu_model_raw, :vendor, :full_name, :brand_name,
|
||
:codename, :lithography, :launch_date,
|
||
:total_cores, :performance_cores, :efficient_cores, :total_threads,
|
||
:base_freq_ghz, :max_turbo_ghz,
|
||
:l2_cache, :l3_cache,
|
||
:tdp_w, :max_tdp_w,
|
||
:pcie_version, :pcie_lanes,
|
||
:max_ram_gb, :ram_types, :ram_speeds_mhz, :ecc_support,
|
||
:igpu_name, :igpu_base_mhz, :igpu_boost_mhz,
|
||
:source_url, :lookup_status, :scraped_at
|
||
)
|
||
"""
|
||
|
||
|
||
def _safe_cpu_row(data: dict) -> dict:
|
||
"""Ensure every expected cpu_specs column key exists (default None)."""
|
||
return {col: data.get(col) for col in _CPU_COLS}
|
||
|
||
|
||
# ── CPU string normalisation ──────────────────────────────────────────────────
|
||
|
||
def normalize_cpu_key(cpu_raw: str) -> str:
|
||
"""Return a stable lowercase dedup key from any CPU model string."""
|
||
s = cpu_raw.strip()
|
||
s = re.sub(r'\s+\(?\d+(?:[.,]\d+)?\s*GHz.*', '', s, flags=re.I).strip()
|
||
s = re.sub(r'\s+\d+-Core\b.*', '', s, flags=re.I).strip()
|
||
s = re.sub(r'\s+\d+P\s+[A-Z]\d+\b.*', '', s, flags=re.I).strip()
|
||
for prefix_pattern in (
|
||
r'^Intel\s+Core\s+',
|
||
r'^Intel\s+',
|
||
r'^AMD\s+',
|
||
):
|
||
s = re.sub(prefix_pattern, '', s, flags=re.I).strip()
|
||
s = re.sub(r'\s+', '-', s)
|
||
return s.lower()
|
||
|
||
|
||
def detect_vendor(cpu_raw: str) -> str:
|
||
if re.search(r'\bApple\b', cpu_raw, re.I) or re.search(r'\bM[123]\b', cpu_raw):
|
||
return "Apple"
|
||
if re.search(r'\bAMD\b|\bRyzen\b|\bAthlon\b', cpu_raw, re.I):
|
||
return "AMD"
|
||
return "Intel"
|
||
|
||
|
||
def _intel_search_term(cpu_raw: str) -> str:
|
||
# Safety net: truncate any marketing copy that starts after a colon.
|
||
s = cpu_raw.split(":")[0].strip()
|
||
# Strip trailing "-processor" / "-chip" labels.
|
||
s = re.sub(r'[-\s]*(processor|cpu|chip)$', '', s, flags=re.I).strip()
|
||
# Drop speed/turbo info — only the model number is needed for the search.
|
||
s = re.sub(r'\s+\(?\d+(?:[.,]\d+)?\s*GHz.*', '', s, flags=re.I).strip()
|
||
s = re.sub(r'\s+\d+P\s+[A-Z]\d+\b.*', '', s, flags=re.I).strip()
|
||
s = re.sub(r'^Intel\s+', '', s, flags=re.I).strip()
|
||
return s
|
||
|
||
|
||
# ── Playwright (Firefox) for JS-heavy / bot-protected pages ──────────────────
|
||
|
||
_pw_instance = None
|
||
_pw_context = None # type: ignore[assignment]
|
||
|
||
|
||
def _get_pw_context():
|
||
"""Lazy-init a persistent Firefox context stored in browser_data/."""
|
||
global _pw_instance, _pw_context
|
||
if _pw_context is None:
|
||
sync_playwright, _BrowserContext = _import_playwright()
|
||
os.makedirs(BROWSER_DATA_DIR, exist_ok=True)
|
||
_pw_instance = sync_playwright().__enter__()
|
||
_pw_context = _pw_instance.firefox.launch_persistent_context(
|
||
user_data_dir=BROWSER_DATA_DIR,
|
||
headless=False,
|
||
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64; rv:136.0) Gecko/20100101 Firefox/136.0",
|
||
viewport={"width": 1280, "height": 900},
|
||
locale="en-US",
|
||
timezone_id="America/New_York",
|
||
firefox_user_prefs={
|
||
"dom.webdriver.enabled": False,
|
||
"useAutomationExtension": False,
|
||
"privacy.resistFingerprinting": False,
|
||
"media.navigator.enabled": True,
|
||
"geo.enabled": False,
|
||
"network.http.referer.sendRefererHeader": 2,
|
||
},
|
||
)
|
||
return _pw_context
|
||
|
||
|
||
def playwright_get(url: str, retries: int = 3) -> Optional[str]:
|
||
"""Fetch a page using Firefox (Playwright) with retry and back-off logic."""
|
||
for attempt in range(1, retries + 1):
|
||
page = None
|
||
try:
|
||
ctx = _get_pw_context()
|
||
page = ctx.new_page()
|
||
response = page.goto(url, wait_until="networkidle", timeout=30_000)
|
||
status = response.status if response else 200
|
||
|
||
if status == 200:
|
||
return page.content()
|
||
|
||
if status == 429:
|
||
wait = 60 + random.uniform(10, 30)
|
||
log.warning(f" Rate-limited (429), sleeping {wait:.0f}s (attempt {attempt}/{retries})")
|
||
time.sleep(wait)
|
||
elif status == 502:
|
||
wait = 180 + random.uniform(0, 20) if attempt == 1 else random.uniform(2, 3)
|
||
log.warning(f" HTTP 502 – backing off {wait:.0f}s (attempt {attempt}/{retries})")
|
||
time.sleep(wait)
|
||
elif status in (404, 410):
|
||
log.info(f" Not found ({status}): {url}")
|
||
return None
|
||
else:
|
||
log.warning(f" HTTP {status} (attempt {attempt}/{retries}): {url}")
|
||
time.sleep(5 * attempt)
|
||
|
||
except Exception as exc:
|
||
log.warning(f" Playwright error attempt {attempt}/{retries} for {url}: {exc}")
|
||
time.sleep(10 * attempt)
|
||
finally:
|
||
if page:
|
||
try:
|
||
page.close()
|
||
except Exception:
|
||
pass
|
||
|
||
log.error(f" Gave up after {retries} attempts: {url}")
|
||
return None
|
||
|
||
|
||
def close_playwright() -> None:
|
||
global _pw_instance, _pw_context
|
||
if _pw_context:
|
||
try:
|
||
_pw_context.close()
|
||
except Exception:
|
||
pass
|
||
if _pw_instance:
|
||
try:
|
||
_pw_instance.__exit__(None, None, None)
|
||
except Exception:
|
||
pass
|
||
_pw_instance = _pw_context = None
|
||
|
||
|
||
def _cpu_sleep():
|
||
time.sleep(random.uniform(CPU_DELAY_MIN, CPU_DELAY_MAX))
|
||
|
||
|
||
# ── Shared parsing utilities ──────────────────────────────────────────────────
|
||
|
||
def _parse_freq_ghz(s: str) -> Optional[float]:
|
||
m = re.search(r'(\d+(?:[.,]\d+)?)\s*GHz', s, re.I)
|
||
return float(m.group(1).replace(',', '.')) if m else None
|
||
|
||
|
||
def _parse_int(s: str) -> Optional[int]:
|
||
m = re.search(r'(\d+)', s.replace(',', ''))
|
||
return int(m.group(1)) if m else None
|
||
|
||
|
||
def _spec_lookup(specs: dict, pattern: str) -> str:
|
||
rx = re.compile(pattern, re.I)
|
||
for k, v in specs.items():
|
||
if rx.search(k):
|
||
return v
|
||
return ""
|
||
|
||
|
||
# ── Intel (intel.com product spec pages) ─────────────────────────────────────
|
||
|
||
def _intel_find_product_url(html: str) -> Optional[str]:
|
||
"""Scan a rendered page for the first Intel product spec link."""
|
||
soup = BeautifulSoup(html, "lxml")
|
||
|
||
def _clean(href: str) -> str:
|
||
m = re.search(r'/url\?q=(https://www\.intel\.com[^&]+)', href)
|
||
if m:
|
||
href = m.group(1)
|
||
if href.startswith("http"):
|
||
return href
|
||
return INTEL_BASE + href
|
||
|
||
for a in soup.find_all("a", href=True):
|
||
href = a["href"]
|
||
if "/products/sku/" in href and "specifications" in href:
|
||
return _clean(href)
|
||
|
||
for a in soup.find_all("a", href=True):
|
||
href = a["href"]
|
||
if "/products/sku/" in href and ".html" in href:
|
||
cleaned = _clean(href)
|
||
base = re.sub(r'/(?:ordering|compatible|downloads|support)\.html$', '', cleaned)
|
||
if not base.endswith("specifications.html"):
|
||
base = re.sub(r'\.html$', '/specifications.html', base)
|
||
return base
|
||
|
||
for a in soup.find_all("a", href=True):
|
||
href = a["href"]
|
||
if "/products/sku/" in href:
|
||
cleaned = _clean(href)
|
||
if not cleaned.endswith("specifications.html"):
|
||
cleaned = cleaned.rstrip("/") + "/specifications.html"
|
||
return cleaned
|
||
|
||
return None
|
||
|
||
|
||
def _intel_parse_specs(html: str, url: str) -> dict:
|
||
"""Parse an intel.com /products/sku/.../specifications.html page."""
|
||
data: dict = {"source_url": url}
|
||
soup = BeautifulSoup(html, "lxml")
|
||
|
||
h1 = soup.select_one("h1")
|
||
if h1:
|
||
data["full_name"] = h1.get_text(strip=True)
|
||
|
||
specs: dict[str, str] = {}
|
||
|
||
for row in soup.select(
|
||
".tech-section-row, .specs-list-item, li.tech-section-item, "
|
||
".blade-content li, [class*='spec-row']"
|
||
):
|
||
lbl = row.select_one(".tech-label, .label, span.label")
|
||
val = row.select_one(".tech-data, .value, span.value")
|
||
if lbl and val:
|
||
k = lbl.get_text(strip=True).rstrip(":")
|
||
v = val.get_text(" ", strip=True)
|
||
if k and v and len(k) < 80:
|
||
specs[k] = v
|
||
|
||
if len(specs) < 5:
|
||
for dl in soup.select("dl"):
|
||
dts = dl.select("dt")
|
||
dds = dl.select("dd")
|
||
for dt, dd in zip(dts, dds):
|
||
k = dt.get_text(strip=True).rstrip(":")
|
||
v = dd.get_text(" ", strip=True)
|
||
if k and v and len(k) < 80:
|
||
specs[k] = v
|
||
|
||
if len(specs) < 5:
|
||
_INTEL_KNOWN_LABELS = [
|
||
"Processor Number", "Product Collection", "Code Name", "Lithography",
|
||
"Launch Date", "Vertical Segment", "Marketing Status",
|
||
"Total Cores", "Total Threads", "Performance-cores", "Efficient-cores",
|
||
"Processor Base Frequency", "Max Turbo Frequency",
|
||
"Intel Thermal Velocity Boost Frequency",
|
||
"Cache", "TDP", "Configurable TDP-down", "Max Turbo Power",
|
||
"Bus Speed",
|
||
"Max Memory Size", "Memory Types", "Max # of Memory Channels",
|
||
"Max Memory Bandwidth", "ECC Memory Supported",
|
||
"GPU Name", "Graphics Base Frequency", "Graphics Max Dynamic Frequency",
|
||
"Graphics Burst Frequency",
|
||
"PCI Express Revision", "Max # of PCI Express Lanes",
|
||
"PCI Express Configurations",
|
||
]
|
||
label_rx = re.compile(
|
||
r'(' + '|'.join(re.escape(l) for l in _INTEL_KNOWN_LABELS) + r')',
|
||
re.I
|
||
)
|
||
for section in soup.select("section, div[class*='section'], div[class*='specs']"):
|
||
text = section.get_text(" ", strip=True)
|
||
parts = label_rx.split(text)
|
||
for j in range(1, len(parts) - 1, 2):
|
||
k = parts[j].strip()
|
||
v = parts[j + 1].strip().split(" ")[0].strip()
|
||
if k and v and len(v) < 200:
|
||
specs.setdefault(k, v)
|
||
|
||
if len(specs) < 5:
|
||
for table in soup.select("table"):
|
||
for tr in table.select("tr"):
|
||
cells = tr.select("td, th")
|
||
if len(cells) >= 2:
|
||
k = cells[0].get_text(strip=True).rstrip(":")
|
||
v = cells[1].get_text(" ", strip=True)
|
||
if k and v and len(k) < 80:
|
||
specs.setdefault(k, v)
|
||
|
||
if specs:
|
||
_ark_map(specs, data)
|
||
|
||
return data
|
||
|
||
|
||
def _ark_map(specs: dict, data: dict) -> None:
|
||
"""Map flat Intel ARK spec key→value pairs to our cpu_specs columns."""
|
||
s = lambda pattern: _spec_lookup(specs, pattern)
|
||
|
||
cn = s(r'code.?name|product.?collection')
|
||
if cn:
|
||
data["codename"] = cn
|
||
lit = s(r'lithograph|process.?technolog|manufactur')
|
||
if lit:
|
||
data["lithography"] = lit
|
||
ld = s(r'launch.?date|announced|release')
|
||
if ld:
|
||
data["launch_date"] = ld
|
||
pn = s(r'processor.?number|brand.?name')
|
||
if pn:
|
||
data.setdefault("brand_name", pn)
|
||
|
||
tc = s(r'#\s*of\s*cores|total\s*cores')
|
||
if tc:
|
||
data["total_cores"] = _parse_int(tc)
|
||
pc = s(r'performance.?core|p.?core')
|
||
if pc:
|
||
data["performance_cores"] = _parse_int(pc)
|
||
ec = s(r'efficient.?core|e.?core')
|
||
if ec:
|
||
data["efficient_cores"] = _parse_int(ec)
|
||
tt = s(r'#\s*of\s*threads|total\s*threads')
|
||
if tt:
|
||
data["total_threads"] = _parse_int(tt)
|
||
|
||
base = s(r'processor\s+base\s+freq|base\s+freq|base\s+clock')
|
||
if base:
|
||
data["base_freq_ghz"] = _parse_freq_ghz(base)
|
||
turbo = s(r'max\s*turbo\s*freq|max\s*boost|turbo\s*freq')
|
||
if turbo:
|
||
data["max_turbo_ghz"] = _parse_freq_ghz(turbo)
|
||
|
||
l3 = s(r'l3\s*cache|total.*l3|smart\s*cache|^cache$')
|
||
if l3:
|
||
data["l3_cache"] = l3
|
||
l2 = s(r'l2\s*cache|total.*l2')
|
||
if l2:
|
||
data["l2_cache"] = l2
|
||
|
||
tdp = s(r'^TDP$|^PBP$|base\s*power|thermal\s*design\s*power')
|
||
if tdp:
|
||
m = re.search(r'(\d+)', tdp)
|
||
if m:
|
||
data["tdp_w"] = int(m.group(1))
|
||
max_tdp = s(r'max\s*turbo\s*power|^MTP$|maximum\s*turbo\s*power|configurable\s*tdp.?up')
|
||
if max_tdp:
|
||
m = re.search(r'(\d+)', max_tdp)
|
||
if m:
|
||
data["max_tdp_w"] = int(m.group(1))
|
||
|
||
pcie_ver = s(r'pci\s*express\s*revision|pcie\s*version|pci\s*express\s*version')
|
||
if pcie_ver:
|
||
m = re.search(r'(\d+(?:\.\d+)?)', pcie_ver)
|
||
if m:
|
||
data["pcie_version"] = m.group(1)
|
||
pcie_l = s(r'max.+pci.+lanes|pci.+express.+lanes')
|
||
if pcie_l:
|
||
m = re.search(r'(\d+)', pcie_l)
|
||
if m:
|
||
data["pcie_lanes"] = int(m.group(1))
|
||
|
||
max_mem = s(r'max\s*memory\s*size|maximum\s*memory')
|
||
if max_mem:
|
||
m = re.search(r'(\d+)\s*GB', max_mem, re.I)
|
||
if m:
|
||
data["max_ram_gb"] = int(m.group(1))
|
||
ram_t = s(r'memory\s*types?\b|supported\s*memory')
|
||
if ram_t:
|
||
data["ram_types"] = ram_t
|
||
ram_spd = s(r'memory\s*speed|memory.*frequency')
|
||
if ram_spd:
|
||
data["ram_speeds_mhz"] = ram_spd
|
||
ecc = s(r'ecc\s*memory|ecc\s*support')
|
||
if ecc:
|
||
data["ecc_support"] = 1 if re.search(r'yes|supported|✓', ecc, re.I) else 0
|
||
|
||
igpu = s(r'processor\s*graphics?\b|integrated\s*graphics?')
|
||
if igpu:
|
||
data["igpu_name"] = igpu
|
||
igpu_base = s(r'graphics\s*base\s*freq|gpu\s*base\s*freq')
|
||
if igpu_base:
|
||
m = re.search(r'(\d+)\s*MHz', igpu_base, re.I)
|
||
if m:
|
||
data["igpu_base_mhz"] = int(m.group(1))
|
||
igpu_boost = s(r'graphics.*dynamic.*freq|graphics.*max.*freq|gpu.*max.*freq')
|
||
if igpu_boost:
|
||
m = re.search(r'(\d+[.,]\d*)\s*GHz', igpu_boost, re.I)
|
||
if m:
|
||
data["igpu_boost_mhz"] = int(float(m.group(1).replace(',', '.')) * 1000)
|
||
else:
|
||
m2 = re.search(r'(\d+)\s*MHz', igpu_boost, re.I)
|
||
if m2:
|
||
data["igpu_boost_mhz"] = int(m2.group(1))
|
||
|
||
|
||
def _is_captcha_page(page) -> bool:
|
||
"""Detect if the current page is a Google CAPTCHA / unusual-traffic page."""
|
||
try:
|
||
url = page.url
|
||
if "sorry/" in url or "sorry.google.com" in url:
|
||
return True
|
||
content = page.content()
|
||
if re.search(r'unusual.{0,40}traffic|g-recaptcha|recaptcha|captcha', content, re.I):
|
||
return True
|
||
except Exception:
|
||
pass
|
||
return False
|
||
|
||
|
||
def _google_find_intel_url(term: str) -> Optional[str]:
|
||
"""Search Google for the Intel product spec page URL via Playwright.
|
||
|
||
If Google returns a CAPTCHA or 429, the browser window (already non-headless)
|
||
stays open and we wait up to 5 minutes for the user to solve it manually.
|
||
"""
|
||
# Strip parenthetical annotations like "(4 kärnor / 8 trådar)" that some
|
||
# retailers append to CPU names — they break the Google site: search.
|
||
clean_term = re.sub(r'\s*\([^)]*\)', '', term).strip()
|
||
query = url_quote(f"Intel {clean_term} site:intel.com/content/www/us/en/products/sku")
|
||
search_url = GOOGLE_SEARCH.format(query=query)
|
||
log.info(f" → Google: {search_url}")
|
||
|
||
page = None
|
||
html = None
|
||
try:
|
||
ctx = _get_pw_context()
|
||
page = ctx.new_page()
|
||
response = page.goto(search_url, wait_until="networkidle", timeout=30_000)
|
||
status = response.status if response else 200
|
||
|
||
# Google sometimes returns 429 directly; more often it redirects to
|
||
# /sorry/... with a 200 status that contains a CAPTCHA form.
|
||
if status == 429 or _is_captcha_page(page):
|
||
log.warning("=" * 60)
|
||
log.warning(" ⚠ Google CAPTCHA / rate-limit detected!")
|
||
log.warning(" Search term: '%s'", term)
|
||
log.warning(" The browser window is open — please solve the CAPTCHA manually.")
|
||
log.warning(" Waiting up to 5 minutes for you to reach the search results…")
|
||
log.warning("=" * 60)
|
||
try:
|
||
page.wait_for_selector("#search a, #rso a", timeout=300_000)
|
||
log.info(" ✓ CAPTCHA solved — resuming.")
|
||
except Exception:
|
||
log.error(
|
||
" Timed out (5 min) waiting for CAPTCHA solution for '%s'. Giving up.", term
|
||
)
|
||
return None
|
||
|
||
elif status not in (200, 301, 302):
|
||
log.warning(f" Google returned HTTP {status}")
|
||
return None
|
||
|
||
else:
|
||
# Normal page — just wait a moment for JS results to settle
|
||
try:
|
||
page.wait_for_selector("#search a, #rso a", timeout=10_000)
|
||
except Exception:
|
||
pass
|
||
|
||
html = page.content()
|
||
|
||
except Exception as exc:
|
||
log.warning(f" Google search error for '{term}': {exc}")
|
||
return None
|
||
finally:
|
||
if page:
|
||
try:
|
||
page.close()
|
||
except Exception:
|
||
pass
|
||
|
||
return _intel_find_product_url(html)
|
||
|
||
|
||
def lookup_intel(cpu_raw: str) -> dict:
|
||
"""Fetch Intel product specs from intel.com via Google → Playwright."""
|
||
term = _intel_search_term(cpu_raw)
|
||
log.info(f" → Intel lookup via Google: '{term}'")
|
||
|
||
product_url = _google_find_intel_url(term)
|
||
_cpu_sleep()
|
||
|
||
if not product_url:
|
||
log.warning(f" No intel.com product URL found for '{term}'")
|
||
return {"lookup_status": "failed"}
|
||
|
||
if "/products/sku/" in product_url and not product_url.endswith("specifications.html"):
|
||
product_url = re.sub(
|
||
r'/(?:ordering|compatible|downloads|support)(\.html)$', '', product_url
|
||
)
|
||
if not product_url.endswith(".html"):
|
||
product_url += "/specifications.html"
|
||
else:
|
||
product_url = product_url.replace(".html", "/specifications.html")
|
||
|
||
log.info(f" → Intel product: {product_url}")
|
||
html = playwright_get(product_url)
|
||
_cpu_sleep()
|
||
if not html:
|
||
return {"lookup_status": "failed"}
|
||
|
||
result = _intel_parse_specs(html, product_url)
|
||
result["lookup_status"] = (
|
||
"ok" if result.get("total_cores") or result.get("tdp_w") or result.get("full_name")
|
||
else "failed"
|
||
)
|
||
return result
|
||
|
||
|
||
# ── Apple Silicon (static data) ───────────────────────────────────────────────
|
||
|
||
_APPLE_STATIC: dict[str, dict] = {
|
||
"apple-m1": {
|
||
"full_name": "Apple M1", "vendor": "Apple",
|
||
"codename": "Firestorm / Icestorm (M1)", "lithography": "5nm (TSMC N5)",
|
||
"launch_date": "2020-11", "total_cores": 8, "performance_cores": 4,
|
||
"efficient_cores": 4, "total_threads": 8, "base_freq_ghz": 3.2,
|
||
"max_turbo_ghz": 3.2, "l2_cache": "12 MB (P-core), 4 MB (E-core)",
|
||
"l3_cache": "N/A (Unified Memory Architecture)", "tdp_w": 15, "max_tdp_w": 20,
|
||
"pcie_version": "4.0", "pcie_lanes": None, "max_ram_gb": 16,
|
||
"ram_types": "LPDDR4X (Unified Memory)", "ram_speeds_mhz": "4266",
|
||
"ecc_support": 0, "igpu_name": "Apple M1 8-core GPU",
|
||
"igpu_base_mhz": None, "igpu_boost_mhz": None,
|
||
"source_url": "https://www.apple.com/newsroom/2020/11/apple-unleashes-m1/",
|
||
"lookup_status": "ok",
|
||
},
|
||
"apple-m1-pro": {
|
||
"full_name": "Apple M1 Pro", "vendor": "Apple",
|
||
"codename": "Avalanche / Blizzard (M1 Pro)", "lithography": "5nm (TSMC N5P)",
|
||
"launch_date": "2021-10", "total_cores": 10, "performance_cores": 8,
|
||
"efficient_cores": 2, "total_threads": 10, "base_freq_ghz": 3.22,
|
||
"max_turbo_ghz": 3.22, "l2_cache": "28 MB (P-core), 4 MB (E-core)",
|
||
"l3_cache": "N/A (Unified Memory Architecture)", "tdp_w": 30, "max_tdp_w": 60,
|
||
"pcie_version": "4.0", "pcie_lanes": None, "max_ram_gb": 32,
|
||
"ram_types": "LPDDR5 (Unified Memory)", "ram_speeds_mhz": "6400",
|
||
"ecc_support": 0, "igpu_name": "Apple M1 Pro 16-core GPU",
|
||
"igpu_base_mhz": None, "igpu_boost_mhz": None,
|
||
"source_url": "https://www.apple.com/newsroom/2021/10/apple-unveils-m1-pro-and-m1-max-supercharged-for-pros/",
|
||
"lookup_status": "ok",
|
||
},
|
||
"apple-m2": {
|
||
"full_name": "Apple M2", "vendor": "Apple",
|
||
"codename": "Everest / Sawtooth (M2)", "lithography": "5nm 2nd gen (TSMC N5P)",
|
||
"launch_date": "2022-06", "total_cores": 8, "performance_cores": 4,
|
||
"efficient_cores": 4, "total_threads": 8, "base_freq_ghz": 3.49,
|
||
"max_turbo_ghz": 3.49, "l2_cache": "16 MB (P-core), 4 MB (E-core)",
|
||
"l3_cache": "N/A (Unified Memory Architecture)", "tdp_w": 15, "max_tdp_w": 20,
|
||
"pcie_version": "4.0", "pcie_lanes": None, "max_ram_gb": 24,
|
||
"ram_types": "LPDDR5 (Unified Memory)", "ram_speeds_mhz": "6400",
|
||
"ecc_support": 0, "igpu_name": "Apple M2 10-core GPU",
|
||
"igpu_base_mhz": None, "igpu_boost_mhz": None,
|
||
"source_url": "https://www.apple.com/newsroom/2022/06/apple-unveils-m2-with-breakthrough-performance-and-capabilities/",
|
||
"lookup_status": "ok",
|
||
},
|
||
}
|
||
|
||
|
||
def lookup_apple(cpu_key: str) -> dict:
|
||
result = _APPLE_STATIC.get(cpu_key)
|
||
if result:
|
||
log.info(f" → Apple: using static data for '{cpu_key}'")
|
||
return dict(result)
|
||
log.warning(f" No static data for Apple key '{cpu_key}'")
|
||
return {"lookup_status": "failed"}
|
||
|
||
|
||
# ── Entry point ───────────────────────────────────────────────────────────────
|
||
|
||
def run(conn: sqlite3.Connection, reset: bool = False) -> None:
|
||
"""Look up detailed specs for all CPU models found in the laptops table."""
|
||
if reset:
|
||
conn.execute("DROP TABLE IF EXISTS cpu_specs")
|
||
conn.executescript(SCHEMA)
|
||
conn.commit()
|
||
log.info("cpu_specs table reset.")
|
||
|
||
log.info("═══ CPU Lookup – starting ═══")
|
||
|
||
rows = conn.execute(
|
||
"SELECT DISTINCT cpu_full FROM laptops "
|
||
"WHERE cpu_full IS NOT NULL AND cpu_full != '' "
|
||
"ORDER BY cpu_full"
|
||
).fetchall()
|
||
|
||
if not rows:
|
||
log.warning("No cpu_full values found in laptops table. "
|
||
"Has the main scraper finished Phase 2 yet?")
|
||
return
|
||
|
||
key_to_raws: dict[str, list[str]] = {}
|
||
for row in rows:
|
||
raw = row["cpu_full"].strip()
|
||
key = normalize_cpu_key(raw)
|
||
key_to_raws.setdefault(key, []).append(raw)
|
||
|
||
log.info(
|
||
f"Found {len(rows)} distinct cpu_full strings "
|
||
f"→ {len(key_to_raws)} unique CPU keys after normalisation"
|
||
)
|
||
|
||
for key, raws in sorted(key_to_raws.items()):
|
||
if len(raws) > 1:
|
||
log.info(f" Merged {len(raws)} strings → '{key}': {raws}")
|
||
|
||
done_keys = {
|
||
r["cpu_key"]
|
||
for r in conn.execute(
|
||
"SELECT cpu_key FROM cpu_specs WHERE lookup_status = 'ok'"
|
||
).fetchall()
|
||
}
|
||
todo = {k: v for k, v in key_to_raws.items() if k not in done_keys}
|
||
log.info(f"{len(done_keys)} already done, {len(todo)} to look up")
|
||
|
||
if not todo:
|
||
log.info("Nothing to do – all CPUs already in cpu_specs.")
|
||
close_playwright()
|
||
return
|
||
|
||
total = len(todo)
|
||
done_ok = 0
|
||
done_fail = 0
|
||
|
||
for i, (cpu_key, raws) in enumerate(sorted(todo.items()), 1):
|
||
model_raw = max(raws, key=len)
|
||
vendor = detect_vendor(model_raw)
|
||
|
||
log.info(f"[{i}/{total}] {cpu_key} vendor={vendor}")
|
||
log.info(f" raw: '{model_raw}'")
|
||
|
||
try:
|
||
if vendor == "Apple":
|
||
result = lookup_apple(cpu_key)
|
||
elif vendor == "AMD":
|
||
log.info(f" Skipping AMD CPU (not supported): {cpu_key}")
|
||
result = {"lookup_status": "skipped"}
|
||
else:
|
||
result = lookup_intel(model_raw)
|
||
|
||
result["cpu_key"] = cpu_key
|
||
result["cpu_model_raw"] = model_raw
|
||
result.setdefault("vendor", vendor)
|
||
result["scraped_at"] = datetime.now().isoformat()
|
||
|
||
conn.execute(CPU_SPECS_INSERT, _safe_cpu_row(result))
|
||
conn.commit()
|
||
|
||
status = result.get("lookup_status", "?")
|
||
if status == "ok":
|
||
done_ok += 1
|
||
log.info(f" ✓ {result.get('full_name', cpu_key)}")
|
||
else:
|
||
done_fail += 1
|
||
log.warning(f" ✗ lookup_status={status} for {cpu_key}")
|
||
|
||
except Exception as exc:
|
||
log.error(f" Exception processing {cpu_key}: {exc}", exc_info=True)
|
||
try:
|
||
conn.execute(
|
||
"""INSERT OR REPLACE INTO cpu_specs
|
||
(cpu_key, cpu_model_raw, vendor, scraped_at, lookup_status)
|
||
VALUES (?, ?, ?, ?, 'failed')""",
|
||
(cpu_key, model_raw, vendor, datetime.now().isoformat()),
|
||
)
|
||
conn.commit()
|
||
except Exception:
|
||
pass
|
||
done_fail += 1
|
||
|
||
if vendor != "Apple":
|
||
time.sleep(random.uniform(1.0, 2.0))
|
||
|
||
log.info(
|
||
f"═══ CPU Lookup done. ✓ {done_ok} ok ✗ {done_fail} failed "
|
||
f"(total processed: {total}) ═══"
|
||
)
|
||
|
||
summary = conn.execute(
|
||
"SELECT vendor, lookup_status, COUNT(*) as n "
|
||
"FROM cpu_specs GROUP BY vendor, lookup_status ORDER BY vendor, lookup_status"
|
||
).fetchall()
|
||
log.info("Summary:")
|
||
for row in summary:
|
||
log.info(f" {row['vendor']:8s} {row['lookup_status']:8s} {row['n']} entries")
|
||
|
||
close_playwright()
|