Skip to content

Commit 276ef86

Browse files
authored
Merge pull request #10 from IONIS-AI/feat/recaps-into-serving-layer
feat: contest recaps come from PostgreSQL — retire the SQLite render dependency
2 parents 098e8a7 + 4ce0b94 commit 276ef86

4 files changed

Lines changed: 209 additions & 6 deletions

File tree

‎import_recaps.py‎

Lines changed: 127 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,127 @@
1+
#!/usr/bin/env python3
2+
"""One-time import of contest recap datasets into the PostgreSQL serving layer.
3+
4+
import_recaps.py --contest-dir /var/tmp/contests # import every recap in recaps.yaml
5+
import_recaps.py --contest-dir /var/tmp/contests --dry-run
6+
7+
WHY THIS IS A ONE-TIME IMPORT AND NOT A REFRESH JOB.
8+
9+
Contest recaps are static. ARRL DX CW and SSB ran in March 2026; the contests are over, the
10+
signatures are frozen, and no amount of re-reading will change a number. There is nothing to
11+
keep fresh — so this is not part of refresh.py's cadence groups. It runs once per contest,
12+
when that contest's dataset is built, and never again.
13+
14+
WHY IT READS SQLITE RATHER THAN REBUILDING FROM CLICKHOUSE.
15+
16+
The datasets are built from ClickHouse by ionis-devel/contests/export_contest_sqlite.py, so
17+
rebuilding looks tempting. It is not safe: that script's current PSKR query hardcodes
18+
avg_distance = 0, while the published recap pages show real distances (160m: 3,476 km). The
19+
files on disk do not match what the current script would produce, and reproducing them means
20+
resolving that provenance question first. Importing the artifact that actually rendered the
21+
published pages has no such risk.
22+
23+
WHAT THIS ENDS.
24+
25+
publish.py read 1.3 GB of SQLite at render time, off a filesystem that existed only on the
26+
9975. That coupling is what broke the contest pages when publishing moved to publish-1 — the
27+
loader warned, returned None, and the run still reported "2 recap(s) loaded" and exited 0.
28+
After this import nothing reads SQLite at render time, on any host.
29+
"""
30+
from __future__ import annotations
31+
32+
import argparse
33+
import datetime as dt
34+
import json
35+
import os
36+
import sqlite3
37+
import sys
38+
from pathlib import Path
39+
40+
import psycopg
41+
import yaml
42+
43+
ROOT = Path(os.environ.get("HAMSTATS_ROOT") or Path(__file__).parent)
44+
DB_FILE = os.environ.get("HAMSTATS_DB_FILE", "/etc/hamstats/db-rw.dsn")
45+
46+
sys.path.insert(0, str(ROOT))
47+
from publish import load_recap_data_sqlite # noqa: E402
48+
49+
# Static data does not expire. The serving layer's staleness check compares refreshed_at
50+
# against max_age, and a recap that is a year old is exactly as correct as one imported today.
51+
STATIC = "100 years"
52+
53+
DATASETS = ("band_summary", "hourly_activity", "solar_timeline", "distance_stats")
54+
55+
56+
def main() -> int:
57+
ap = argparse.ArgumentParser(description=__doc__)
58+
ap.add_argument("--contest-dir", required=True,
59+
help="directory holding the recap .sqlite files")
60+
ap.add_argument("--dry-run", action="store_true", help="read and report, write nothing")
61+
args = ap.parse_args()
62+
63+
os.environ["CONTEST_DATA_DIR"] = args.contest_dir
64+
import publish
65+
publish.CONTEST_DATA_DIR = Path(args.contest_dir)
66+
67+
recaps = yaml.safe_load((ROOT / "data" / "recaps.yaml").read_text()) or []
68+
print(f"Importing {len(recaps)} recap(s) from {args.contest_dir}")
69+
70+
staged, missing = [], []
71+
for recap in recaps:
72+
slug = recap.get("slug")
73+
# load_recap_data is publish.py's OWN loader, unchanged. Whatever it produced when
74+
# rendering from SQLite is exactly what lands in PostgreSQL — the import cannot drift
75+
# from the thing it is replacing, because it is the same code.
76+
data = load_recap_data_sqlite(recap)
77+
if data is None:
78+
print(f" MISSING {slug}: no dataset at {args.contest_dir}/{recap.get('dataset')}")
79+
missing.append(slug)
80+
continue
81+
for key in DATASETS:
82+
rows = data.get(key) or []
83+
staged.append((f"recap:{slug}:{key}", rows))
84+
print(f" {slug:22} {key:18} {len(rows):>5} rows")
85+
86+
if missing:
87+
# Importing a partial set would leave some recaps rendering empty with no indication
88+
# that anything is wrong -- the exact failure this import exists to end.
89+
print(f"ERROR: {len(missing)} recap(s) had no dataset: {', '.join(missing)}. "
90+
f"Nothing written.", file=sys.stderr)
91+
return 1
92+
93+
if args.dry_run:
94+
print("Dry run — nothing written.")
95+
return 0
96+
97+
dsn = Path(DB_FILE).read_text().strip()
98+
with psycopg.connect(dsn, autocommit=False) as conn:
99+
conn.execute((ROOT / "sql" / "serving_schema.sql").read_text())
100+
for name, rows in staged:
101+
conn.execute(
102+
"""
103+
INSERT INTO serving.query_results
104+
(name, payload, row_count, refreshed_at, max_age,
105+
source_rows_read, source_elapsed_ms)
106+
VALUES (%s, %s::jsonb, %s, now(), %s::interval, NULL, NULL)
107+
ON CONFLICT (name) DO UPDATE SET
108+
payload = EXCLUDED.payload,
109+
row_count = EXCLUDED.row_count,
110+
refreshed_at = EXCLUDED.refreshed_at,
111+
max_age = EXCLUDED.max_age
112+
""",
113+
(name, json.dumps(rows, default=_json_default), len(rows), STATIC),
114+
)
115+
conn.commit()
116+
print(f"Wrote {len(staged)} recap dataset(s) to the serving layer.")
117+
return 0
118+
119+
120+
def _json_default(o):
121+
if isinstance(o, (dt.datetime, dt.date)):
122+
return o.isoformat()
123+
raise TypeError(f"{type(o).__name__} is not JSON-serialisable: {o!r}")
124+
125+
126+
if __name__ == "__main__":
127+
sys.exit(main())

‎ionis-hamstats.spec‎

Lines changed: 13 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -1,7 +1,7 @@
11
%global debug_package %{nil}
22

33
Name: ionis-hamstats
4-
Version: 1.1.2
4+
Version: 1.2.0
55
Release: 1%{?dist}
66
Summary: Ham Stats publishing pipeline — ClickHouse aggregates to a static site
77

@@ -108,6 +108,18 @@ fi
108108
%dir %{_sysconfdir}/hamstats
109109

110110
%changelog
111+
* Wed Sep 09 2026 Greg Beam <ki7mt@yahoo.com> - 1.2.0-1
112+
- Contest recaps come from the PostgreSQL serving layer. publish.py aggregated 1.3 GB of
113+
SQLite at render time off a filesystem that existed only on the 9975; moving the publisher
114+
to publish-1 emptied both contest pages, and the run still reported "2 recap(s) loaded" and
115+
exited 0 because that count counted recap DEFINITIONS, not loaded data.
116+
- import_recaps.py imports them once. Recaps are static -- the contests are over -- so this is
117+
not a refresh cadence, it is a one-time import per contest, reusing publish.py's own SQLite
118+
loader so the imported data cannot drift from what it replaces.
119+
- A recap with no data in the serving layer now fails the run instead of rendering a page
120+
without its band tables.
121+
- Retires the last filesystem dependency: nothing reads SQLite at render time on any host.
122+
111123
* Wed Sep 09 2026 Greg Beam <ki7mt@yahoo.com> - 1.1.2-1
112124
- Non-finite floats become null. The weekly refresh read 9.4 billion rows and then failed to
113125
write them: storm_snr_comparison.after_snr is null in ClickHouse for a storm too recent to

‎publish.py‎

Lines changed: 41 additions & 5 deletions
Original file line numberDiff line numberDiff line change
@@ -225,6 +225,30 @@ def run_all_queries(client) -> dict:
225225
return data
226226

227227

228+
DATASETS = ("band_summary", "hourly_activity", "solar_timeline", "distance_stats")
229+
230+
231+
def load_recap_from_serving(recap: dict, data: dict) -> dict | None:
232+
"""Read a recap's pre-computed datasets out of the serving layer.
233+
234+
Recaps used to be aggregated from 1.3 GB of SQLite at render time, off a filesystem that
235+
existed only on the 9975. That coupling broke every contest page the moment publishing
236+
moved hosts: the loader warned, returned None, and the run still reported "2 recap(s)
237+
loaded" and exited 0, so empty pages reached the site unnoticed.
238+
239+
They are static -- the contests are over -- so they are imported once and served forever.
240+
Returning None here now means genuinely absent, and the caller treats it as a failure
241+
rather than as "static findings only".
242+
"""
243+
got = {}
244+
for key in DATASETS:
245+
rows = data.get(f"recap:{recap['slug']}:{key}")
246+
if rows is None:
247+
return None
248+
got[key] = rows
249+
return got
250+
251+
228252
# ---------------------------------------------------------------------------
229253
# IONIS V22-gamma + PhysicsOverrideLayer Predictions
230254
# ---------------------------------------------------------------------------
@@ -387,7 +411,13 @@ def sqlite_ro_uri(db_path: Path) -> str:
387411
return f"file:{path}?mode=ro&immutable=1"
388412

389413

390-
def load_recap_data(recap: dict) -> dict | None:
414+
def load_recap_data_sqlite(recap: dict) -> dict | None:
415+
"""Aggregate a recap from its SQLite dataset. USED ONLY BY import_recaps.py.
416+
417+
This is no longer on the render path. publish.py reads recaps from the serving layer;
418+
this function exists so the one-time import produces exactly what rendering from SQLite
419+
produced, using the same code rather than a reimplementation of it.
420+
"""
391421
"""Load aggregated stats from a contest SQLite file.
392422
393423
Returns dict with band_summary, hourly_activity, solar_timeline,
@@ -1002,18 +1032,24 @@ def main():
10021032
if dx_predictions:
10031033
print(f" Generated predictions for {len(dx_predictions)} DXpeditions")
10041034

1005-
# 5. Contest recaps (from SQLite datasets)
1035+
# 5. Contest recaps — pre-imported into the serving layer, not read from SQLite
10061036
print("Loading contest recaps...")
10071037
recap_defs = load_recaps()
10081038
recaps = []
10091039
for recap in recap_defs:
1010-
recap_data = load_recap_data(recap)
1040+
recap_data = load_recap_from_serving(recap, data)
10111041
if recap_data:
10121042
print(f" {recap['slug']}: {len(recap_data.get('band_summary', []))} bands")
10131043
else:
1014-
print(f" {recap['slug']}: no dataset (static findings only)")
1044+
# NOT "static findings only". That wording made an absent dataset look like a
1045+
# deliberate mode, and the count below counted DEFINITIONS regardless — so two
1046+
# empty contest pages published cleanly and nothing said otherwise.
1047+
print(f" {recap['slug']}: NOT IMPORTED — page will render without band data",
1048+
file=sys.stderr)
1049+
stale.append(f"recap {recap['slug']} has no data in the serving layer "
1050+
f"(run import_recaps.py for it)")
10151051
recaps.append((recap, recap_data))
1016-
print(f" {len(recaps)} recap(s) loaded")
1052+
print(f" {sum(1 for _, d in recaps if d)} of {len(recaps)} recap(s) have data")
10171053

10181054
context = build_context(
10191055
data, predictions, now,

‎test_serving.py‎

Lines changed: 28 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -124,6 +124,34 @@ def __repr__(self): return "<weird>"
124124
check("the schema forbids a row_count that disagrees with the payload",
125125
"row_count = jsonb_array_length(payload)" in Path("sql/serving_schema.sql").read_text(), True)
126126

127+
print("== contest recaps come from the serving layer, not the filesystem ==")
128+
# publish.py aggregated 1.3 GB of SQLite at render time, off a filesystem that existed only on
129+
# the 9975. Moving the publisher to publish-1 emptied both contest pages -- and the run still
130+
# printed "2 recap(s) loaded" and exited 0, because that count counted DEFINITIONS. The pages
131+
# were live and wrong for 40 minutes.
132+
pcode = Path("publish.py").read_text()
133+
check("the render path no longer opens SQLite",
134+
"load_recap_data_sqlite(recap)" in code, False)
135+
check("the SQLite loader survives for the one-time import",
136+
"def load_recap_data_sqlite" in pcode, True)
137+
check("rendering reads the serving layer",
138+
"load_recap_from_serving(recap, data)" in code, True)
139+
# `code` is comment-stripped: the comment explaining the removal names the old wording on
140+
# purpose, and a check that forbids describing the bug punishes the fix.
141+
# The exact former print, not the phrase: the docstring and the comment both name the old
142+
# wording deliberately, and `code` strips comments but not docstrings.
143+
check("a missing recap is reported as missing, not as a mode",
144+
"no dataset (static findings only)" in code, False)
145+
check("and it fails the run", "stale.append(f\"recap {recap['slug']}" in pcode, True)
146+
check("the count reports recaps WITH DATA, not definitions",
147+
"sum(1 for _, d in recaps if d)" in pcode, True)
148+
149+
rcode = Path("import_recaps.py").read_text()
150+
check("the import reuses publish's own loader rather than reimplementing it",
151+
"from publish import load_recap_data_sqlite" in rcode, True)
152+
check("a partial import writes nothing",
153+
"Nothing written." in rcode, True)
154+
127155
print()
128156
if failures:
129157
print(f" {len(failures)} FAILED")

0 commit comments

Comments
 (0)