From e5da387e74f9debc2e173c659743a875df394f0f Mon Sep 17 00:00:00 2001 From: tiXor-code Date: Tue, 25 Aug 2026 21:33:04 +0300 Subject: [PATCH] deficienta: publish episodes, fix the strazi asymmetry, enforce the contract CMTEB publishes three states - Oprire, Deficienta (pressure/temperature below spec) and normal. The headline `days` counts only oprire+ACC, which is correct and deliberate. But deficienta was being flattened to a bare day count, and on streets not even that. Measured on the current published bundle: 15,747 deficienta PT-days in 2025 against a 26,559-day headline (59%), rising to 73% year-to-date in 2026, with ~85% of them falling on days no outage touched. These three changes MUST land together: validate's new key constants require the fields publish now emits, so landing validate alone would FAIL the next nightly on all ~11,700 entity-years and block the release. 1. Publish deficienta episodes (PT only - streets have no episode array). The display record is hoisted above the severity branch and shared, so oprire and deficienta cannot drift apart. Three additive keys: episodes_deficienta / episodes_count_deficienta / est_hours_deficienta. Strictly additive: the deficienta path still `continue`s before pt_days, city_eps, pt_epn and pt_hours, so `days` and everything derived from it are untouched. Proven by test, which asserts est_hours stays 24.0 and episodes_count stays 1 while the deficienta counterparts fill. 2. Emit days_deficienta on the strazi ndjson year object. `defi` was already computed there as the emission gate and thrown away, so street-years shipped deficienta RUNS with no counter to reconcile against - while ARTIFACTS.md:91 claims the union invariant holds "for every PT-year and street-year". Verified against the live bundle: absent on all 6,670 street-years. 3. Enforce the contract in validate.py, which mentioned deficienta exactly once (in a docstring) and asserted nothing: runs_days_union FAIL - the ARTIFACTS.md:90-91 guarantee ndjson_year_keys FAIL - a missing key renders NaN on the site deficienta_reconciliation FAIL - exact biconditionals by construction rankings_row_keys FAIL - exact contract key set per row deficienta_non_vacuous WARN - heuristic; FAIL only if EVERY namespace-year is zero, since a quiet year is plausible upstream and blocking the release on a CMTEB editorial change would be wrong The union check unions day-of-year numbers rather than summing run lengths: classes legitimately overlap, and summing would FAIL on correct data. The ranking key check is a PRE-PASS, before the existing `break`s. Those stop at the first bad row, so a key regression further down would never be seen; it also fixes a latent crash where a malformed row raised KeyError outside any try and killed validate with a traceback instead of a FAIL. Verified against the live published bundle before enforcing: PT 5,054 entity-years and street 6,670 entity-years, ZERO union mismatches on either. The invariant is real, so these checks will not produce false FAILs. Detail strings carry only slugs, years and integers - never cause_raw - so untrusted CMTEB text cannot forge PASS/FAIL lines in the CI log (SEC049). tests/test_publish_shapes.py now imports the key sets from validate instead of redeclaring them, so the CI-visible check and the db-gated check cannot drift. Its fixtures were also non-vacuous-ified: strazi ranking was [] and both ndjson year objects were {"days": 3}, so several of these checks would have passed without asserting anything. 45 passed, 15 skipped. Co-Authored-By: Claude Opus 5 (1M context) --- ARTIFACTS.md | 16 ++- pipeline/publish.py | 44 ++++++-- pipeline/validate.py | 174 ++++++++++++++++++++++++++++++ tests/test_publish_shapes.py | 10 +- tests/test_validate.py | 201 +++++++++++++++++++++++++++++++++-- 5 files changed, 426 insertions(+), 19 deletions(-) diff --git a/ARTIFACTS.md b/ARTIFACTS.md index 9246ced..588ceda 100644 --- a/ARTIFACTS.md +++ b/ARTIFACTS.md @@ -79,7 +79,12 @@ One JSON object per line per PT ever seen (universe + standalone entities): "episodes": [{"start": "2025-04-24T07:00", "end": "2025-10-16T23:00", "ongoing": false, "uncertain": true, "cause_class": "programat", "cause_raw": "Revizie tehnica CTE Progresu", - "remediere_last": "2025-10-15T23:00"}]}}} + "remediere_last": "2025-10-15T23:00"}], + "episodes_count_deficienta": 2, "est_hours_deficienta": 61.5, + "episodes_deficienta": [{"start": "2025-07-19T08:00", "end": "2025-07-20T08:00", + "ongoing": false, "uncertain": false, "cause_class": "unclassified", + "cause_raw": "Lipsa parametri pentru livrare apa calda de consum", + "remediere_last": null}]}}} ``` `runs` = [start_day_of_year (1-based, local), length_days, cause_class], episodes clipped to the year; ongoing = ended_before null; uncertain = @@ -91,6 +96,14 @@ Run `cause_class` takes one of FOUR values: `avarie`, `programat`, of non-deficienta runs equals `days` for every PT-year and street-year). Consumers painting or summing headline days must exclude `deficienta` runs. +Deficienta also carries its own episode array and counters: +`episodes_deficienta`, `episodes_count_deficienta`, `est_hours_deficienta` +(PT only - streets have no episode array at all). These are DISJOINT from the +headline trio: `episodes` / `episodes_count` / `est_hours` never include a +deficienta episode, and the deficienta three never include an oprire one. They +must never be summed into the headline. `days_deficienta > 0` if and only if +`episodes_count_deficienta > 0`. + ### strazi/all.ndjson.gz ```json {"slug": "sos-pantelimon", "name": "Sos Pantelimon", "type": "sos", @@ -100,6 +113,7 @@ Consumers painting or summing headline days must exclude `deficienta` runs. "inferred_pt": null, "inferred_km": null, "addr": {"64": [3, 0.18], "64a": [3, 0.2], "70": [5, 0.42]}, "years": {"2025": {"days": 181, "days_avarie": 121, "days_programat": 74, + "days_deficienta": 28, "runs": [[10, 3, "avarie"], "..."]}}} ``` Street universe = streets CMTEB named in outages UNION all named Bucharest diff --git a/pipeline/publish.py b/pipeline/publish.py index 6b28b09..2288593 100644 --- a/pipeline/publish.py +++ b/pipeline/publish.py @@ -307,6 +307,12 @@ def build(db_path: str, registry_path: str, harta_html: str, pt_eps: dict[tuple, list[dict]] = defaultdict(list) # (pt, y) display episodes pt_epn: Counter = Counter() # (pt, y) oprire eps touching pt_hours: dict[tuple, float] = defaultdict(float) # (pt, y) split est_hours + # Parallel deficienta structures, deliberately separate from every headline + # map above. Nothing here ever feeds pt_days, city_eps, pt_epn or pt_hours, + # so `days` and everything derived from it stay byte-identical to v1. + pt_eps_defi: dict[tuple, list[dict]] = defaultdict(list) # (pt, y) display eps + pt_epn_defi: Counter = Counter() # (pt, y) eps touching + pt_hours_defi: dict[tuple, float] = defaultdict(float) # (pt, y) split est_hours pt_blocks: dict[str, int] = {} sector_votes: dict[str, Counter] = defaultdict(Counter) city_eps: Counter = Counter() # start-year attributed @@ -329,9 +335,23 @@ def build(db_path: str, registry_path: str, harta_html: str, sector_votes[pt][sector] += 1 if blocks: pt_blocks[pt] = max(pt_blocks.get(pt, 0), blocks) + # The display record is a pure function of the episode row, so it is + # hoisted above the severity branch and shared. Building it for + # deficienta rows is the new behaviour; the oprire record is unchanged. + epd = {"start": fmt_minute(first), "end": fmt_minute(last), + "ongoing": ended is None, "uncertain": bool(gap), + "cause_class": cls, "cause_raw": craw, + "remediere_last": _rem_iso(rem)} if sev == "deficienta": for d in days: pt_cls[(pt, d.year, "deficienta")].add(d) + # Same proportional year-split rule as the oprire path below, and + # keyed off the same `ycount`, so days_deficienta > 0 and + # episodes_count_deficienta > 0 hold on identical year sets. + for y, c in ycount.items(): + pt_eps_defi[(pt, y)].append(epd) + pt_epn_defi[(pt, y)] += 1 + pt_hours_defi[(pt, y)] += est_h * (c / len(days)) continue y0 = local(first).year city_eps[y0] += 1 @@ -341,10 +361,6 @@ def build(db_path: str, registry_path: str, harta_html: str, for d in days: pt_days[(pt, d.year)].add(d) pt_cls[(pt, d.year, cls)].add(d) - epd = {"start": fmt_minute(first), "end": fmt_minute(last), - "ongoing": ended is None, "uncertain": bool(gap), - "cause_class": cls, "cause_raw": craw, - "remediere_last": _rem_iso(rem)} for y, c in ycount.items(): pt_eps[(pt, y)].append(epd) pt_epn[(pt, y)] += 1 @@ -718,6 +734,9 @@ def _longest(days_map: dict, ent, y: int) -> int: write_json(out / "client" / "map" / f"pt-{y}.geojson", {"type": "FeatureCollection", "features": feats}) + def _epkey(e): + return (e["start"], e["end"], e["cause_class"], e["cause_raw"] or "") + pt_lines = [] for pt in sorted(pts_all, key=lambda p: pt_slug[p]): years_obj = {} @@ -726,9 +745,8 @@ def _longest(days_map: dict, ent, y: int) -> int: defi = pt_cls.get((pt, y, "deficienta"), set()) if not union and not defi: continue - eps = sorted(pt_eps.get((pt, y), ()), - key=lambda e: (e["start"], e["end"], e["cause_class"], - e["cause_raw"] or "")) + eps = sorted(pt_eps.get((pt, y), ()), key=_epkey) + eps_defi = sorted(pt_eps_defi.get((pt, y), ()), key=_epkey) years_obj[str(y)] = { "days": len(union), "days_avarie": len(pt_cls.get((pt, y, "avarie"), ())), @@ -739,6 +757,12 @@ def _longest(days_map: dict, ent, y: int) -> int: "est_hours": round(pt_hours[(pt, y)], 1), "runs": _runs(pt_cls, pt, y), "episodes": eps, + # Additive deficienta counterparts. Disjoint from the three + # above: episodes_count / est_hours / episodes never include a + # deficienta episode, and these never include an oprire one. + "episodes_count_deficienta": pt_epn_defi[(pt, y)], + "est_hours_deficienta": round(pt_hours_defi[(pt, y)], 1), + "episodes_deficienta": eps_defi, } c = pt_coord.get(pt) pt_lines.append({ @@ -766,6 +790,12 @@ def _longest(days_map: dict, ent, y: int) -> int: "days": len(union), "days_avarie": len(st_cls.get((k, y, "avarie"), ())), "days_programat": len(st_cls.get((k, y, "programat"), ())), + # `defi` was already computed above as the emission gate but was + # never written out, so street-years shipped deficienta RUNS with + # no counter to reconcile them against - and ARTIFACTS.md:91 + # claims the union invariant holds "for every PT-year and + # street-year". Emitting it makes publish match the contract. + "days_deficienta": len(defi), "runs": _runs(st_cls, k, y), } blocks = [] diff --git a/pipeline/validate.py b/pipeline/validate.py index 7a4f411..dfb7104 100644 --- a/pipeline/validate.py +++ b/pipeline/validate.py @@ -33,6 +33,27 @@ ARTIFACT_KEYS_META = ("generated_at", "data_through", "years", "last_complete_year", "partial_years", "universe_size", "coverage", "sources_cutover_utc") +# Exact key sets for published rows. These ARE the contract, so `set(row) != +# want` is the assertion: an unannounced extra key fails as loudly as a missing +# one. tests/test_publish_shapes.py imports these so there is one definition. +ARTIFACT_KEYS_PT_RANK = frozenset({ + "slug", "name", "sector", "days", "days_avarie", "days_programat", + "days_deficienta", "episodes", "longest_days", "est_day_eq", "delta_prev"}) +ARTIFACT_KEYS_ST_RANK = (ARTIFACT_KEYS_PT_RANK - {"sector"}) | {"sectors", "pt_slugs"} +# Year objects use a SUBSET check instead: they are where additive fields land +# most often, and validate should not block a future additive key. +ARTIFACT_KEYS_PT_YEAR = frozenset({ + "days", "days_avarie", "days_programat", "days_deficienta", + "episodes_count", "longest_days", "est_hours", "runs", "episodes", + "episodes_count_deficienta", "est_hours_deficienta", "episodes_deficienta"}) +ARTIFACT_KEYS_ST_YEAR = frozenset({ + "days", "days_avarie", "days_programat", "days_deficienta", "runs"}) + +# The three headline cause classes. `deficienta` is the pseudo-class feeding the +# secondary counter, excluded from `days` by contract (ARTIFACTS.md:88-92). +NON_DEFICIENTA = ("avarie", "programat", "unclassified") +MAX_DOY = 366 + def _alias(db) -> dict[str, str]: try: @@ -210,6 +231,88 @@ def check_sector_consistency(db, web: Path, harta_html: str, return out +def _run_day_set(runs: list, classes: tuple[str, ...]) -> set[int]: + """Day-of-year numbers covered by runs whose cause is in `classes`. + + Runs of different classes legitimately OVERLAP - a day can carry both an + avarie and a programat episode, which is exactly why ARTIFACTS.md says the + per-class counts may sum above `days`. So this unions DOY numbers rather + than summing run lengths; summing would over-count and make the contract + look breached on perfectly good data. + + Callers must shape-check `runs` first (see _year_object_problems). + """ + out: set[int] = set() + for start, length, cls in runs: + if cls in classes: + out.update(range(start, start + length)) + return out + + +def _year_object_problems(ns: str, yo) -> list[str]: + """Contract violations inside one entity-year object. `ns` is "pt" or "st". + + The load-bearing check is ARTIFACTS.md:90-91, stated there as a hard + guarantee and enforced nowhere until now: the union of non-deficienta run + days equals `days`, and deficienta runs account for exactly + `days_deficienta`. + + Returns strings and never raises - one malformed entity must not abort the + whole ndjson scan. Details carry only slugs, years and integers, never + cause_raw, so untrusted CMTEB text cannot forge PASS/FAIL lines in the CI + log (SEC049). + """ + if not isinstance(yo, dict): + return [f"year object is {type(yo).__name__}, not an object"] + probs: list[str] = [] + want = ARTIFACT_KEYS_PT_YEAR if ns == "pt" else ARTIFACT_KEYS_ST_YEAR + missing = sorted(want - set(yo)) + if missing: + probs.append(f"missing keys {missing}") + + runs = yo.get("runs") + if not isinstance(runs, list): + return probs + ["runs missing or not a list"] + for r in runs: + if not (isinstance(r, list) and len(r) == 3 + and isinstance(r[0], int) and isinstance(r[1], int) + and isinstance(r[2], str) + and r[0] >= 1 and r[1] >= 1 and r[0] + r[1] - 1 <= MAX_DOY): + return probs + [f"malformed run {r}"] + + if isinstance(yo.get("days"), int): + head = len(_run_day_set(runs, NON_DEFICIENTA)) + if head != yo["days"]: + probs.append(f"non-deficienta runs cover {head} days != days {yo['days']}") + if isinstance(yo.get("days_deficienta"), int): + defi = len(_run_day_set(runs, ("deficienta",))) + if defi != yo["days_deficienta"]: + probs.append(f"deficienta runs cover {defi} days " + f"!= days_deficienta {yo['days_deficienta']}") + return probs + + +def _deficienta_problems(yo: dict) -> list[str]: + """PT-only reconciliation between deficienta counters and their episodes. + + publish.py appends to episodes_deficienta and increments + episodes_count_deficienta in one loop, and derives the deficienta day set + from the SAME per-episode year Counter, so both relations below are exact + biconditionals by construction. Any drift is a real bug, never data noise. + """ + probs: list[str] = [] + for cnt_key, eps_key in (("episodes_count", "episodes"), + ("episodes_count_deficienta", "episodes_deficienta")): + n, eps = yo.get(cnt_key), yo.get(eps_key) + if isinstance(eps, list) and isinstance(n, int) and n != len(eps): + probs.append(f"{cnt_key} {n} != len({eps_key}) {len(eps)}") + n, dd = yo.get("episodes_count_deficienta"), yo.get("days_deficienta") + if isinstance(n, int) and isinstance(dd, int) and (n > 0) != (dd > 0): + probs.append(f"episodes_count_deficienta {n} inconsistent " + f"with days_deficienta {dd}") + return probs + + def check_artifacts(web: Path, registry_slugs: set[str] | None = None) -> list[tuple]: out = [] @@ -242,6 +345,12 @@ def load(rel: str): streets_with_addr = 0 addr_numbers = 0 addr_bad: list[str] = [] + year_bad: list[str] = [] # runs_days_union + key_bad: list[str] = [] # ndjson_year_keys + recon_bad: list[str] = [] # deficienta_reconciliation + n_year_bad = n_key_bad = n_recon_bad = 0 + defi_by_nsyear: Counter = Counter() # (ns, "YYYY") -> sum(days_deficienta) + nsyears_seen: set[tuple] = set() for ns, rel in (("pt", "pt/all.ndjson.gz"), ("st", "strazi/all.ndjson.gz")): p = web / rel if not p.exists(): @@ -253,6 +362,25 @@ def load(rel: str): for line in f: o = json.loads(line) bag.add(o["slug"]) + for ystr, yo in (o.get("years") or {}).items(): + nsyears_seen.add((ns, ystr)) + for msg in _year_object_problems(ns, yo): + if msg.startswith("missing keys"): + n_key_bad += 1 + if len(key_bad) < 5: + key_bad.append(f"{o['slug']}/{ystr}: {msg}") + else: + n_year_bad += 1 + if len(year_bad) < 5: + year_bad.append(f"{o['slug']}/{ystr}: {msg}") + if ns == "pt" and isinstance(yo, dict): + for msg in _deficienta_problems(yo): + n_recon_bad += 1 + if len(recon_bad) < 5: + recon_bad.append(f"{o['slug']}/{ystr}: {msg}") + if isinstance(yo, dict) and isinstance( + yo.get("days_deficienta"), int): + defi_by_nsyear[(ns, ystr)] += yo["days_deficienta"] if ns == "st" and o.get("blocks"): streets_with_blocks += 1 block_pts.update(b["pt"] for b in o["blocks"]) @@ -288,7 +416,33 @@ def load(rel: str): f"addr pt_index out of range / bad -1: {addr_bad[:5]}" if addr_bad else f"addr maps on {streets_with_addr} streets, {addr_numbers} numbers, all resolvable")) + # --- deficienta contract (ARTIFACTS.md:88-92) -------------------------- + out.append((FAIL if year_bad else PASS, "runs_days_union", + f"{n_year_bad} entity-years breach the runs/days contract: " + + "; ".join(year_bad) if year_bad + else f"non-deficienta runs == days and deficienta runs == " + f"days_deficienta across {len(nsyears_seen)} namespace-years")) + out.append((FAIL if key_bad else PASS, "ndjson_year_keys", + f"{n_key_bad} entity-years: " + "; ".join(key_bad) if key_bad + else "year objects carry the contract key set")) + out.append((FAIL if recon_bad else PASS, "deficienta_reconciliation", + f"{n_recon_bad} entity-years: " + "; ".join(recon_bad) if recon_bad + else "episode counts reconcile with episode arrays and day counts")) + # Heuristic, not an identity: a genuinely quiet year is conceivable upstream, + # and FAILing on it would let a CMTEB editorial change block the release. + # Every namespace-year at zero simultaneously is a pipeline regression, so + # only that case FAILs. + vac = sorted(f"{ns}-{y}" for (ns, y) in nsyears_seen + if defi_by_nsyear[(ns, y)] == 0) + lvl = FAIL if vac and len(vac) == len(nsyears_seen) else WARN if vac else PASS + out.append((lvl, "deficienta_non_vacuous", + f"days_deficienta is 0 across every entity in: {vac[:6]}" if vac + else f"days_deficienta populated in all {len(nsyears_seen)} " + f"namespace-years")) + rank_bad = [] + row_key_bad: list[str] = [] + n_row_key_bad = 0 for y in years: dist = load(f"city/distribution-{y}.json") if dist is not None: @@ -300,6 +454,23 @@ def load(rel: str): rows = load(f"rankings/{kind}-{y}.json") if rows is None: continue + # Shape pre-pass, deliberately BEFORE the value checks. Two + # reasons: the r["days"]/r["slug"] accesses below would raise + # KeyError on a malformed row and kill validate with a traceback + # instead of a FAIL; and the `break`s below stop at the first bad + # row, so a key regression further down would never be seen. + want = ARTIFACT_KEYS_PT_RANK if kind == "pt" else ARTIFACT_KEYS_ST_RANK + file_bad = False + for i, r in enumerate(rows): + if not isinstance(r, dict) or set(r) != want: + file_bad = True + n_row_key_bad += 1 + if len(row_key_bad) < 5: + diff = (sorted(set(r) ^ want) if isinstance(r, dict) + else type(r).__name__) + row_key_bad.append(f"{kind}-{y} row {i}: {diff}") + if file_bad: + continue days_seq = [r["days"] for r in rows] if days_seq != sorted(days_seq, reverse=True): rank_bad.append(f"{kind}-{y} not sorted desc") @@ -320,6 +491,9 @@ def load(rel: str): f["geometry"]["type"] != "Point" or len(f["geometry"]["coordinates"]) != 2 for f in feats): out.append((FAIL, "map_geojson", f"pt-{y}.geojson malformed")) + out.append((FAIL if row_key_bad else PASS, "rankings_row_keys", + f"{n_row_key_bad} rows: " + "; ".join(row_key_bad) if row_key_bad + else f"ranking rows carry the exact contract keys for {len(years)} years")) out.append((FAIL if rank_bad else PASS, "rankings_integrity", "; ".join(rank_bad[:5]) if rank_bad else f"rankings resolvable + sorted for {len(years)} years")) diff --git a/tests/test_publish_shapes.py b/tests/test_publish_shapes.py index bee3a4d..2186d39 100644 --- a/tests/test_publish_shapes.py +++ b/tests/test_publish_shapes.py @@ -11,15 +11,19 @@ import pytest +from pipeline import validate + DB = Path("db/termo.db") HARTA = Path("data/harta.html") pytestmark = pytest.mark.skipif(not DB.exists() or not HARTA.exists(), reason="real db/harta not present") -PT_KEYS = {"slug", "name", "sector", "days", "days_avarie", "days_programat", - "days_deficienta", "episodes", "longest_days", "est_day_eq", "delta_prev"} -ST_KEYS = (PT_KEYS - {"sector"}) | {"sectors", "pt_slugs"} +# Imported rather than redeclared: pipeline/validate.py enforces these same key +# sets on CI (where this file is skipped for lack of a db), so two copies would +# be free to drift apart silently. +PT_KEYS = set(validate.ARTIFACT_KEYS_PT_RANK) +ST_KEYS = set(validate.ARTIFACT_KEYS_ST_RANK) @pytest.fixture(scope="module") diff --git a/tests/test_validate.py b/tests/test_validate.py index c59d453..c910c28 100644 --- a/tests/test_validate.py +++ b/tests/test_validate.py @@ -1,5 +1,6 @@ """Validate checks against synthetic dbs and a minimal artifact tree.""" +import json import sqlite3 from pipeline import validate @@ -70,6 +71,53 @@ def test_unclassified_12_pct_warns_exit_zero(tmp_path): assert validate.exit_code(results) == 0 +# --- fixture year objects ------------------------------------------------- +# Internally consistent by construction: every day count is exactly the union +# of the matching runs, so any override that breaks one relation trips exactly +# the check under test and nothing else. +_EP = {"start": "2025-05-01T10:00", "end": "2025-05-03T10:00", "ongoing": False, + "uncertain": False, "cause_class": "avarie", "cause_raw": "c", + "remediere_last": None} +_EP_DEFI = {"start": "2025-07-19T08:00", "end": "2025-07-20T08:00", + "ongoing": False, "uncertain": False, "cause_class": "unclassified", + "cause_raw": "presiune scazuta", "remediere_last": None} + + +def _pt_year(**over): + """days=3 <- [121,3,avarie] (doy 121-123); days_deficienta=2 <- [200,2,deficienta].""" + y = {"days": 3, "days_avarie": 3, "days_programat": 0, "days_deficienta": 2, + "episodes_count": 1, "longest_days": 3, "est_hours": 48.0, + "runs": [[121, 3, "avarie"], [200, 2, "deficienta"]], + "episodes": [_EP], + "episodes_count_deficienta": 1, "est_hours_deficienta": 24.0, + "episodes_deficienta": [_EP_DEFI]} + y.update(over) + return y + + +def _st_year(**over): + """days=3 <- [121,3,avarie]; days_deficienta=1 <- [300,1,deficienta].""" + y = {"days": 3, "days_avarie": 3, "days_programat": 0, "days_deficienta": 1, + "runs": [[121, 3, "avarie"], [300, 1, "deficienta"]]} + y.update(over) + return y + + +def _pt_record(**over): + r = {"slug": "pt-a", "name": "A", "sector": 1, "lat": 44.4, "lon": 26.1, + "on_map": True, "blocks_estimate": None, "streets": [], "nearest": [], + "years": {"2025": _pt_year()}} + r.update(over) + return r + + +def _st_record(**over): + r = {"slug": "str-b", "name": "Str B", "type": "str", "sectors": [1], + "pts": ["pt-a"], "neighbors": [], "years": {"2025": _st_year()}} + r.update(over) + return r + + def _minimal_web(tmp_path): web = tmp_path / "web" write_json(web / "meta.json", { @@ -91,7 +139,10 @@ def _minimal_web(tmp_path): {"slug": "pt-a", "name": "A", "sector": 1, "days": 3, "days_avarie": 3, "days_programat": 0, "days_deficienta": 0, "episodes": 1, "longest_days": 3, "est_day_eq": 3.0, "delta_prev": None}]) - write_json(web / "rankings" / "strazi-2025.json", []) + write_json(web / "rankings" / "strazi-2025.json", [ + {"slug": "str-b", "name": "Str B", "sectors": [1], "pt_slugs": ["pt-a"], + "days": 3, "days_avarie": 3, "days_programat": 0, "days_deficienta": 1, + "episodes": 1, "longest_days": 3, "est_day_eq": 3.0, "delta_prev": None}]) write_json(web / "rankings" / "sectoare-2025.json", [ {"sector": s, "pts": 1 if s == 1 else 0, "median_days": 0, "mean_days": 0.0, "mean_days_avarie": 0.0, "mean_days_programat": 0.0, @@ -107,13 +158,8 @@ def _minimal_web(tmp_path): {"t": "st", "n": "Str B", "s": "str-b", "sec": 1, "d": 3}]) write_json(web / "og" / "stats.json", {"pt-a": ["pt", "A", 1, 3, 2025], "str-b": ["st", "Str B", 1, 3, 2025]}) - write_ndjson_gz(web / "pt" / "all.ndjson.gz", [ - {"slug": "pt-a", "name": "A", "sector": 1, "lat": 44.4, "lon": 26.1, - "on_map": True, "blocks_estimate": None, "streets": [], "nearest": [], - "years": {"2025": {"days": 3}}}]) - write_ndjson_gz(web / "strazi" / "all.ndjson.gz", [ - {"slug": "str-b", "name": "Str B", "type": "str", "sectors": [1], - "pts": ["pt-a"], "neighbors": [], "years": {"2025": {"days": 3}}}]) + write_ndjson_gz(web / "pt" / "all.ndjson.gz", [_pt_record()]) + write_ndjson_gz(web / "strazi" / "all.ndjson.gz", [_st_record()]) return web @@ -146,3 +192,142 @@ def test_unsorted_rankings_fail(tmp_path): results = validate.check_artifacts(web, {"pt-a", "str-b"}) assert any(r[1] == "rankings_integrity" and r[0] == validate.FAIL for r in results) + + +# --- deficienta contract (ARTIFACTS.md:88-92) ----------------------------- + +def test_run_day_set_unions_overlapping_classes(): + # A day can carry BOTH an avarie and a programat episode - ARTIFACTS.md says + # the per-class day counts may sum above `days`. So the invariant must union + # day-of-year numbers, never sum run lengths, or it fails on correct data. + runs = [[10, 3, "avarie"], [12, 3, "programat"], [50, 1, "deficienta"]] + assert validate._run_day_set(runs, validate.NON_DEFICIENTA) == {10, 11, 12, 13, 14} + assert validate._run_day_set(runs, ("deficienta",)) == {50} + + +def test_runs_days_union_mismatch_fails(tmp_path): + web = _minimal_web(tmp_path) + # runs still cover only 3 days while `days` claims 4 + write_ndjson_gz(web / "pt" / "all.ndjson.gz", + [_pt_record(years={"2025": _pt_year(days=4)})]) + results = validate.check_artifacts(web, {"pt-a", "str-b"}) + assert any(r[1] == "runs_days_union" and r[0] == validate.FAIL for r in results) + assert validate.exit_code(results) == 1 + + +def test_street_runs_days_union_mismatch_fails(tmp_path): + # The strazi half of the ARTIFACTS.md:91 guarantee. Vacuous before this work, + # because street year objects carried no days_deficienta at all. + web = _minimal_web(tmp_path) + write_ndjson_gz(web / "strazi" / "all.ndjson.gz", + [_st_record(years={"2025": _st_year(days_deficienta=7)})]) + results = validate.check_artifacts(web, {"pt-a", "str-b"}) + assert any(r[1] == "runs_days_union" and r[0] == validate.FAIL for r in results) + + +def test_missing_year_key_fails(tmp_path): + web = _minimal_web(tmp_path) + y = _pt_year() + del y["days_deficienta"] + write_ndjson_gz(web / "pt" / "all.ndjson.gz", [_pt_record(years={"2025": y})]) + results = validate.check_artifacts(web, {"pt-a", "str-b"}) + assert any(r[1] == "ndjson_year_keys" and r[0] == validate.FAIL for r in results) + + +def test_deficienta_episode_count_mismatch_fails(tmp_path): + web = _minimal_web(tmp_path) + write_ndjson_gz(web / "pt" / "all.ndjson.gz", + [_pt_record(years={"2025": _pt_year(episodes_count_deficienta=4)})]) + results = validate.check_artifacts(web, {"pt-a", "str-b"}) + assert any(r[1] == "deficienta_reconciliation" and r[0] == validate.FAIL + for r in results) + + +def test_ranking_row_missing_days_deficienta_fails(tmp_path): + web = _minimal_web(tmp_path) + rows = [{"slug": "pt-a", "name": "A", "sector": 1, "days": 3, "days_avarie": 3, + "days_programat": 0, "episodes": 1, "longest_days": 3, + "est_day_eq": 3.0, "delta_prev": None}] # days_deficienta dropped + write_json(web / "rankings" / "pt-2025.json", rows) + results = validate.check_artifacts(web, {"pt-a", "str-b"}) + assert any(r[1] == "rankings_row_keys" and r[0] == validate.FAIL for r in results) + + +def test_ranking_key_check_survives_a_bad_first_row(tmp_path): + # Guards the pre-existing `break` short-circuit: an unresolvable slug in row 0 + # must not hide a key regression in row 1. + web = _minimal_web(tmp_path) + write_json(web / "rankings" / "pt-2025.json", [ + {"slug": "pt-ghost", "name": "G", "sector": 1, "days": 9, "days_avarie": 9, + "days_programat": 0, "days_deficienta": 0, "episodes": 1, + "longest_days": 9, "est_day_eq": 9.0, "delta_prev": None}, + {"slug": "pt-a", "name": "A", "sector": 1, "days": 3, "days_avarie": 3, + "days_programat": 0, "episodes": 1, "longest_days": 3, + "est_day_eq": 3.0, "delta_prev": None}]) # days_deficienta dropped + results = validate.check_artifacts(web, {"pt-a", "str-b"}) + assert any(r[1] == "rankings_row_keys" and r[0] == validate.FAIL for r in results) + + +def test_all_zero_deficienta_warns_but_does_not_block_release(tmp_path): + # A quiet year is plausible upstream; blocking the nightly on it would let an + # editorial change at CMTEB stop publication. WARN, not FAIL. + web = _minimal_web(tmp_path) + write_ndjson_gz(web / "pt" / "all.ndjson.gz", [_pt_record(years={"2025": _pt_year( + days_deficienta=0, episodes_count_deficienta=0, est_hours_deficienta=0.0, + episodes_deficienta=[], runs=[[121, 3, "avarie"]])})]) + results = validate.check_artifacts(web, {"pt-a", "str-b"}) + assert any(r[1] == "deficienta_non_vacuous" and r[0] == validate.WARN + for r in results) + assert validate.exit_code(results) == 0 + + +def test_deficienta_episodes_published_without_touching_headline(tmp_path): + """The additivity contract, proven end-to-end through publish.build(). + + A deficienta episode must contribute to days_deficienta and the new + episode fields, and to NOTHING else: not days, not episodes_count, not + est_hours, not the episodes array. + """ + from pipeline import publish + db = _db(tmp_path) + _ep(db, pt="pt a", sev="oprire", + first="2025-05-01T10:00:00+03:00", last="2025-05-02T10:00:00+03:00") + _ep(db, pt="pt a", sev="deficienta", + first="2025-07-01T10:00:00+03:00", last="2025-07-03T10:00:00+03:00") + # build() derives the year range from snapshot coverage, so the synthetic db + # needs the two bookend snapshots that bracket 2025. + for i, ts in enumerate(("2025-01-01T00:00:00+00:00", "2025-12-31T00:00:00+00:00")): + db.execute("INSERT INTO snapshot (sha, observed_utc, content_hash, " + "n_records, parse_status, changed) VALUES (?,?,?,?,?,?)", + (f"sha{i}", ts, f"h{i}", 1, "ok", 1)) + db.commit() + db.close() + + # Empty static/ skips the OSM street + address joins, which are irrelevant + # here and turn a 0.0s test into a 45s one. + empty_static = tmp_path / "static" + empty_static.mkdir() + out = tmp_path / "web" + publish.build(str(tmp_path / "t.db"), str(tmp_path / "reg.json"), + "data/harta.html", str(empty_static), str(out)) + + import gzip as _gz + with _gz.open(out / "pt" / "all.ndjson.gz", "rt", encoding="utf-8") as f: + rec = next(r for r in map(json.loads, f) if r["slug"] == "pt-pt-a") + y = rec["years"]["2025"] + + # headline half - every one of these must be blind to the deficienta episode + assert y["days"] == 2 + assert y["episodes_count"] == 1 + assert y["est_hours"] == 24.0 # the deficienta episode's hours are NOT here + assert len(y["episodes"]) == 1 + assert y["episodes"][0]["cause_class"] == "avarie" + + # deficienta half + assert y["days_deficienta"] == 3 + assert y["episodes_count_deficienta"] == 1 + assert len(y["episodes_deficienta"]) == 1 + assert y["est_hours_deficienta"] > 0 + + # and the two never mix + assert y["episodes_deficienta"][0] not in y["episodes"]