Skip to content

Commit bdd564a

Browse files
committed
fix(ingest): correct Wikipedia CPU parsing of fractional TDP, clock ranges, L2 columns and family rows
- '9.5 W' parsed as 5 W and '3.6 W' as 6 W (regex matched the digits after the dot) - '1.7-2.0 GHz' took 2.0 as the base clock; now base 1.7, boost 2.0 - any 'cache' header mapped to l3_cache_mb, so Atom 'L2 cache' columns became L3 - 'September 2013' collapsed to January 1; month is now kept - family-tier rows ('Ryzen 5', 'Core i7') and citation markers no longer become SKUs - quoted section headings are cleaned and '(14 nm)' goes to process_node
1 parent cb47e02 commit bdd564a

4 files changed

Lines changed: 163 additions & 5 deletions

File tree

‎app/ingest/normalize.py‎

Lines changed: 29 additions & 3 deletions
Original file line numberDiff line numberDiff line change
@@ -8,6 +8,7 @@
88

99
from __future__ import annotations
1010

11+
import math
1112
import re
1213
from datetime import date
1314

@@ -18,7 +19,12 @@
1819
_MEMORY_RE = re.compile(r"(\d+(?:\.\d+)?)\s*(GB|MB)\b", re.IGNORECASE)
1920
_BUS_RE = re.compile(r"(\d{2,4})\s*-?\s*bit\b", re.IGNORECASE)
2021
_PCIE_RE = re.compile(r"PCI[-\s]?[Ee]?\s*(?:Gen\s*)?(\d(?:\.\d)?)", re.IGNORECASE)
21-
_TDP_RE = re.compile(r"(\d{1,4})(?:\s*/\s*\d{1,4})?\s*W\b", re.IGNORECASE)
22+
_TDP_RE = re.compile(
23+
r"(?<![\d.])(\d{1,4}(?:\.\d+)?)(?:\s*/\s*\d{1,4}(?:\.\d+)?)?\s*W\b", re.IGNORECASE
24+
)
25+
_FREQ_RANGE_RE = re.compile(
26+
r"(\d+(?:\.\d+)?)\s*(?:–|—|-|to)\s*(\d+(?:\.\d+)?)\s*(GHz|MHz)\b", re.IGNORECASE
27+
)
2228
_RAM_RE = re.compile(r"(\d{1,3}(?:\.\d+)?)\s*(GB|MB)\b", re.IGNORECASE)
2329
_BATTERY_RE = re.compile(r"(\d{3,5})\s*m\s*A\s*h\b", re.IGNORECASE)
2430
_WEIGHT_RE = re.compile(r"(\d{1,3}(?:\.\d+)?)\s*g\b")
@@ -41,6 +47,11 @@
4147
re.IGNORECASE,
4248
)
4349
_QUARTER_RE = re.compile(r"\bQ([1-4])\s*'?(\d{2}|\d{4})\b", re.IGNORECASE)
50+
_MONTH_YEAR_RE = re.compile(
51+
r"\b(January|February|March|April|May|June|July|"
52+
r"August|September|October|November|December)\s+(\d{4})\b",
53+
re.IGNORECASE,
54+
)
4455
_YEAR_ONLY_RE = re.compile(r"\b(19\d{2}|20\d{2})\b")
4556

4657
_MONTHS = {
@@ -67,11 +78,24 @@ def parse_frequency_ghz(text: str) -> float | None:
6778

6879

6980
def parse_tdp_w(text: str) -> int | None:
70-
"""``"65 W"`` → ``65``; ``"65/95 W"`` → ``65`` (takes the lower bound)."""
81+
"""``"65 W"`` → ``65``; ``"65/95 W"`` → ``65``; ``"9.5 W"`` → ``10`` (half-up)."""
7182
if not text:
7283
return None
7384
match = _TDP_RE.search(text)
74-
return int(match.group(1)) if match else None
85+
return math.floor(float(match.group(1)) + 0.5) if match else None
86+
87+
88+
def parse_frequency_range_ghz(text: str) -> tuple[float, float] | None:
89+
"""``"1.7–2.0 GHz"`` → ``(1.7, 2.0)`` (base, boost); ``None`` if not a range."""
90+
if not text:
91+
return None
92+
match = _FREQ_RANGE_RE.search(text)
93+
if not match:
94+
return None
95+
low, high = float(match.group(1)), float(match.group(2))
96+
if match.group(3).lower() == "mhz":
97+
low, high = round(low / 1000, 3), round(high / 1000, 3)
98+
return (low, high) if low < high else None
7599

76100

77101
def parse_cache_mb(text: str) -> float | None:
@@ -136,6 +160,8 @@ def parse_date(text: str) -> date | None:
136160
year_raw = match.group(2)
137161
year = 2000 + int(year_raw) if len(year_raw) == 2 else int(year_raw)
138162
return _safe_date(year, (quarter - 1) * 3 + 1, 1)
163+
if (match := _MONTH_YEAR_RE.search(stripped)):
164+
return _safe_date(int(match.group(2)), _MONTHS[match.group(1).lower()], 1)
139165
if (match := _YEAR_ONLY_RE.search(stripped)):
140166
return _safe_date(int(match.group(1)), 1, 1)
141167
return None

‎app/ingest/sources/wikipedia_cpu.py‎

Lines changed: 40 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -11,6 +11,7 @@
1111

1212
from __future__ import annotations
1313

14+
import re
1415
from collections.abc import Iterator
1516
from pathlib import Path
1617

@@ -25,6 +26,7 @@
2526
parse_cores_threads,
2627
parse_date,
2728
parse_frequency_ghz,
29+
parse_frequency_range_ghz,
2830
parse_int,
2931
parse_tdp_w,
3032
)
@@ -43,6 +45,15 @@
4345
("amd", "List_of_AMD_Threadripper_processors", "AMD Threadripper"),
4446
]
4547

48+
# Citation markers left in model cells ("7501 [ 32 ] [ 33 ]").
49+
_FOOTNOTE_RE = re.compile(r"\s*\[\s*[\w\s]{1,6}\]")
50+
# Rows naming only a family tier ("Ryzen 5", "Core i7", "Xeon 6") are group
51+
# headers, not SKUs; they pass the has-a-digit check but must never become records.
52+
_FAMILY_ONLY_RE = re.compile(
53+
r"(?:ryzen|ryzen-pro|core|core-ultra|core-i|xeon|epyc|athlon|pentium|celeron|atom|opteron)"
54+
r"-?\d{1,2}"
55+
)
56+
4657
# Manufacturer keys are stored lowercase; these are their display forms used to
4758
# synthesize ``name`` when the model string omits the brand. Plain ``.upper()``
4859
# mangles "intel" → "INTEL" (an ingest casing artifact); AMD is genuinely
@@ -59,7 +70,9 @@
5970
"threads": ["threads", "thread"],
6071
"base_clock": ["base", "freq", "clock"],
6172
"boost_clock": ["boost", "turbo", "max"],
62-
"l3_cache": ["l3", "cache"],
73+
# Only an explicit L3 / Smart Cache column is L3. A bare "cache" match used
74+
# to route "L2 cache" columns (e.g. every Atom table) into l3_cache_mb.
75+
"l3_cache": ["l3", "smart cache"],
6376
"tdp": ["tdp", "power", "wattage"],
6477
"release_date": ["released", "release", "launched", "launch", "date"],
6578
"socket": ["socket"],
@@ -99,10 +112,12 @@ def _extract(
99112
for table in soup.select("table.wikitable"):
100113
section_label = _nearest_section_label(table) or fallback_arch
101114
for row in parse_table(table, HEADER_RULES):
102-
model = row.cells.get("model", "")
115+
model = _FOOTNOTE_RE.sub("", row.cells.get("model", "")).strip()
103116
slug = slugify(model, manufacturer=manufacturer)
104117
if len(slug) < 4 or not any(ch.isdigit() for ch in slug):
105118
continue
119+
if _FAMILY_ONLY_RE.fullmatch(slug):
120+
continue
106121
architecture = row.cells.get("architecture") or section_label
107122
yield _build_candidate(
108123
manufacturer=manufacturer,
@@ -114,6 +129,22 @@ def _extract(
114129
)
115130

116131

132+
_HEADING_NODE_RE = re.compile(r"\((\d+(?:\.\d+)?)\s*nm\)")
133+
134+
135+
def _clean_architecture(label: str) -> tuple[str, str | None]:
136+
"""``'" Denverton " (14 nm)'`` → ``("Denverton", "14 nm")``.
137+
138+
Section headings on the Atom/Core list pages quote the codename and carry
139+
the node in parentheses; the raw heading text leaked both into
140+
``architecture``.
141+
"""
142+
node_match = _HEADING_NODE_RE.search(label)
143+
node = f"{node_match.group(1)} nm" if node_match else None
144+
name = _HEADING_NODE_RE.sub("", label).replace('"', "").replace("“", "").replace("”", "")
145+
return " ".join(name.split()) or label, node
146+
147+
117148
def _nearest_section_label(table: Tag) -> str | None:
118149
for prev in table.find_all_previous(["h2", "h3", "h4"]):
119150
text = prev.get_text(" ", strip=True)
@@ -139,10 +170,17 @@ def _build_candidate(
139170
release_date = parse_date(row.get("release_date", ""))
140171
base_clock = parse_frequency_ghz(row.get("base_clock", ""))
141172
boost_clock = parse_frequency_ghz(row.get("boost_clock", ""))
173+
# "1.7–2.0 GHz" in a single frequency cell is base–boost; the plain parser
174+
# would read only the number glued to the unit (2.0) as the base clock.
175+
if (clock_range := parse_frequency_range_ghz(row.get("base_clock", ""))) is not None:
176+
base_clock = clock_range[0]
177+
boost_clock = boost_clock or clock_range[1]
142178
l3_cache = parse_cache_mb(row.get("l3_cache", ""))
143179
tdp = parse_tdp_w(row.get("tdp", ""))
144180
socket = row.get("socket") or None
145181
process_node = row.get("process_node") or None
182+
architecture, node_from_heading = _clean_architecture(architecture)
183+
process_node = process_node or node_from_heading
146184

147185
segment = guess_cpu_segment(model)
148186
brand = _BRAND_DISPLAY.get(manufacturer, manufacturer.title())

‎tests/unit/test_ingest_normalize.py‎

Lines changed: 24 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -12,6 +12,7 @@
1212
parse_cores_threads,
1313
parse_date,
1414
parse_frequency_ghz,
15+
parse_frequency_range_ghz,
1516
parse_int,
1617
parse_tdp_w,
1718
)
@@ -38,12 +39,35 @@ def test_parse_frequency_ghz(text: str, expected: float | None) -> None:
3839
("65/95 W", 65),
3940
("125W", 125),
4041
("none", None),
42+
# Regression: "9.5 W" used to parse as 5 (the digits after the dot).
43+
("9.5 W", 10),
44+
("3.6 W", 4),
45+
("2.2/3 W", 2),
4146
],
4247
)
4348
def test_parse_tdp_w(text: str, expected: int | None) -> None:
4449
assert parse_tdp_w(text) == expected
4550

4651

52+
@pytest.mark.parametrize(
53+
"text,expected",
54+
[
55+
("1.7–2.0 GHz", (1.7, 2.0)),
56+
("1.7-2.0 GHz", (1.7, 2.0)),
57+
("1600–2400 MHz", (1.6, 2.4)),
58+
("2.0 GHz", None),
59+
("2.0–1.7 GHz", None),
60+
("", None),
61+
],
62+
)
63+
def test_parse_frequency_range_ghz(text: str, expected: tuple[float, float] | None) -> None:
64+
assert parse_frequency_range_ghz(text) == expected
65+
66+
67+
def test_parse_date_month_year_keeps_the_month() -> None:
68+
assert parse_date("September 2013") == date(2013, 9, 1)
69+
70+
4771
@pytest.mark.parametrize(
4872
"text,expected",
4973
[

‎tests/unit/test_ingest_wikipedia_cpu.py‎

Lines changed: 70 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -88,3 +88,73 @@ def test_filters_non_model_rows_lacking_a_slug() -> None:
8888
# short and gets filtered out.
8989
slugs = {c.slug for c in candidates}
9090
assert all("raptor" not in slug for slug in slugs)
91+
92+
93+
# Atom-style table rendered from {{cpulist}}: an "L2 cache" column, a
94+
# base–turbo range in one frequency cell, a fractional TDP, a family-tier
95+
# group row and a footnoted model name.
96+
_ATOM_HTML = """
97+
<html><body>
98+
<h3>" Avoton " (22 nm)</h3>
99+
<table class="wikitable">
100+
<tr>
101+
<th>Model</th><th>Cores</th><th>Frequency</th><th>L2 cache</th>
102+
<th>TDP</th><th>Released</th><th>Socket</th>
103+
</tr>
104+
<tr>
105+
<td>Atom C2530</td><td>4</td><td>1.7–2.0 GHz</td><td>2 × 1 MB</td>
106+
<td>9 W</td><td>September 2013</td><td>FC-BGA 1283</td>
107+
</tr>
108+
<tr>
109+
<td>Atom C2508</td><td>4</td><td>1.25 GHz</td><td>2 × 1 MB</td>
110+
<td>9.5 W</td><td>March 2014</td><td>FC-BGA 1283</td>
111+
</tr>
112+
<tr>
113+
<td>Atom 3 [ 12 ]</td><td>4</td><td>1.0 GHz</td><td>1 MB</td>
114+
<td>5 W</td><td>2014</td><td>BGA</td>
115+
</tr>
116+
<tr>
117+
<td>Atom C2338 [ 7 ]</td><td>2</td><td>1.7 GHz</td><td>1 MB</td>
118+
<td>7 W</td><td>2014</td><td>BGA</td>
119+
</tr>
120+
</table>
121+
</body></html>
122+
"""
123+
124+
125+
def _atom() -> dict[str, dict[str, object]]:
126+
candidates = WikipediaCpuIngest._extract(
127+
_ATOM_HTML, "intel", "List_of_Intel_Atom_processors", "Intel Atom"
128+
)
129+
return {c.slug: c.record for c in candidates}
130+
131+
132+
def test_l2_cache_column_is_not_written_as_l3() -> None:
133+
assert _atom()["atom-c2530"]["l3_cache_mb"] is None
134+
135+
136+
def test_frequency_range_splits_into_base_and_boost() -> None:
137+
record = _atom()["atom-c2530"]
138+
assert record["base_clock_ghz"] == 1.7
139+
assert record["boost_clock_ghz"] == 2.0
140+
141+
142+
def test_fractional_tdp_rounds_instead_of_dropping_the_integer_part() -> None:
143+
assert _atom()["atom-c2508"]["tdp_w"] == 10
144+
145+
146+
def test_month_year_release_keeps_the_month() -> None:
147+
assert _atom()["atom-c2530"]["release_date"] == "2013-09-01"
148+
149+
150+
def test_family_tier_rows_and_footnotes_are_handled() -> None:
151+
records = _atom()
152+
assert "atom-3" not in records
153+
assert "atom-c2338" in records
154+
assert records["atom-c2338"]["name"] == "Intel Atom C2338"
155+
156+
157+
def test_quoted_heading_is_cleaned_and_node_extracted() -> None:
158+
record = _atom()["atom-c2530"]
159+
assert record["architecture"] == "Avoton"
160+
assert record["process_node"] == "22 nm"

0 commit comments

Comments
 (0)