From e182836180525d07dc00f6f35088404e4c8830b6 Mon Sep 17 00:00:00 2001 From: Matt Spinola Date: Thu, 30 Jul 2026 21:46:29 -0400 Subject: [PATCH] feat(vintage): seed the CFTC release calendar so derived rows get real dates cotdata-schedule sync now also parses the published CFTC release schedule, which lists one year of RELEASE dates month by month and marks holiday-delayed ones with an asterisk ('*Delayed release date due to a federal holiday.'). What this is actually worth, stated precisely: on a normal week 'derived' (report_date + 3, weekend-adjusted) already lands on the right Friday, so most dates do not move. It buys two things. It CORRECTS the six 2026 weeks delayed by a federal holiday, where derived is wrong by three days (New Year, Juneteenth, Independence Day, Veterans Day, Thanksgiving, Christmas). And it upgrades the PROVENANCE of every matched row from derived (a guess) to scheduled (a published fact), which is what lets strict point-in-time evaluation trust the date at all. report_date is derived from each release as the latest Tuesday at least three days earlier. That one rule covers both the normal Friday release and a holiday-delayed Monday one, which reports the previous week's Tuesday, without needing a holiday calendar of our own. Parsed against a trimmed copy of the live page committed as a fixture, so the parser is pinned to real markup rather than an idealised shape. It reads the year from tag-stripped text because it sits inside nested markup, and it RAISES rather than returning an empty frame on a layout change: a silent empty parse looks exactly like 'CFTC published nothing' and would quietly leave every row on derived. Cross-validated: the calendar says report 2026-07-21 released 2026-07-24, which is the same date the published mechanism independently derives from the weekly static's Last-Modified header. Two unrelated sources agree, and a test pins that. Also fixes a pandas FutureWarning in the announcements merge, where concatenating into an empty store is deprecated and shifts dtype inference. Coverage limit: the page only ever shows the current year, and CFTC does not publish past calendars there, so earlier years cannot be seeded this way. Co-Authored-By: Claude Opus 4.8 --- src/cotdata/vintage_cli.py | 4 + src/cotdata/vintage_schedule.py | 104 +++++++++++++++++- .../fixtures/cftc_release_schedule_2026.html | 7 ++ tests/test_vintage_schedule.py | 73 ++++++++++++ 4 files changed, 187 insertions(+), 1 deletion(-) create mode 100644 tests/fixtures/cftc_release_schedule_2026.html diff --git a/src/cotdata/vintage_cli.py b/src/cotdata/vintage_cli.py index c0874da..165f1e8 100644 --- a/src/cotdata/vintage_cli.py +++ b/src/cotdata/vintage_cli.py @@ -182,8 +182,12 @@ def main(argv=None) -> int: # ── schedule ──────────────────────────────────────────────────────────────── def _cmd_sched_sync(args) -> int: from . import vintage_schedule + cal = vintage_schedule.sync_release_schedule() + print(f"schedule sync: {cal['scheduled']} release date(s) from the published " + f"calendar ({cal['holiday_delayed']} holiday-delayed).") res = vintage_schedule.sync() print(f"schedule sync: {res['announcements']} announcement row(s) scraped.") + print("run 'cotdata-schedule backfill' to apply them to stored observations.") return 0 diff --git a/src/cotdata/vintage_schedule.py b/src/cotdata/vintage_schedule.py index 37cc8f3..aa50e98 100644 --- a/src/cotdata/vintage_schedule.py +++ b/src/cotdata/vintage_schedule.py @@ -21,6 +21,7 @@ import csv import datetime as dt +import re from email.utils import parsedate_to_datetime from pathlib import Path from zoneinfo import ZoneInfo @@ -189,6 +190,103 @@ def sync_published() -> dict: return {"published": len(derived)} +# ── `scheduled`: the CFTC published release calendar ──────────────────────── +# The page lists RELEASE dates for one year, month by month, marking holiday-delayed ones +# with an asterisk ("*Delayed release date due to a federal holiday."). Those asterisks are +# the entire value here: on a normal week `derived` (report_date + 3, weekend-adjusted) +# already lands on the right Friday, so seeding the calendar changes few dates. What it +# changes is (a) the handful of holiday weeks where derived is wrong by one to three days, +# and (b) the PROVENANCE of every matched row, from `derived` (a guess) to `scheduled` (a +# published fact) — which is what lets strict point-in-time evaluation trust the date. +_MONTHS = {m: i for i, m in enumerate( + ["January", "February", "March", "April", "May", "June", "July", + "August", "September", "October", "November", "December"], start=1)} + + +def _strip_tags(html: str) -> str: + return re.sub(r"\s+", " ", re.sub(r"<[^>]+>", " ", html)).strip() + + +def report_date_for_release(release: dt.date) -> dt.date: + """The Tuesday whose positions a given release reports. + + COT is Tuesday-dated and published the following Friday. A federal holiday pushes the + RELEASE later but never moves the as-of date, so the report date is the latest Tuesday + at least three days before the release. That rule handles both the normal Friday case + (Tuesday + 3) and a delayed Monday release (the previous week's Tuesday), without + needing a holiday calendar of our own. + """ + d = pd.Timestamp(release).date() - dt.timedelta(days=3) + while d.weekday() != 1: # 1 == Tuesday + d -= dt.timedelta(days=1) + return d + + +def parse_release_schedule(html: str) -> pd.DataFrame: + """Parse the CFTC release-schedule page into schedule rows. + + Tag-tolerant by design: the year lives inside nested markup + (``

2026 Release Schedule

``), so it is read from tag-stripped + text rather than from an assumed tag shape. Raises rather than returning an empty + frame if the year or table cannot be found — a silent empty parse would look exactly + like "CFTC published nothing", and would quietly leave every row on `derived`. + """ + text = _strip_tags(html) + ym = re.search(r"(20\d{2})\s+Release Schedule", text, re.I) + if not ym: + raise ValueError("release schedule: could not find the ' Release Schedule' " + "heading; the page layout has probably changed.") + year = int(ym.group(1)) + + tm = re.search(r"", html, re.S | re.I) + if not tm: + raise ValueError("release schedule: no found on the page.") + cells = re.findall(r"]*>(.*?)", tm.group(0), re.S | re.I) + + rows, month = [], None + for cell in cells: + val = _strip_tags(cell).replace("\xa0", " ").replace(" ", " ").strip() + if not val: + continue + if val in _MONTHS: + month = _MONTHS[val] + continue + m = re.fullmatch(r"(\d{1,2})\s*(\*?)", val) + if not m or month is None: + continue + release = dt.date(year, month, int(m.group(1))) + delayed = bool(m.group(2)) + rows.append({ + "report_date": pd.Timestamp(report_date_for_release(release)), + "release_date": pd.Timestamp(release), + "source": "scheduled", + "note": ("delayed by a federal holiday" if delayed + else "published CFTC release schedule"), + "ingested_at": pd.Timestamp.now("UTC"), + }) + if not rows: + raise ValueError("release schedule: table found but no dates parsed out of it.") + return pd.DataFrame(rows) + + +def sync_release_schedule(*, fetch_html=None) -> dict: + """Fetch and store the published release calendar. Idempotent. + + Covers ONE year: the page only ever shows the current schedule, and CFTC does not + publish past calendars here, so earlier years cannot be recovered this way and stay + on `announced` or `derived`. + """ + fetch_html = fetch_html or _fetch_html + parsed = parse_release_schedule(fetch_html(RELEASE_SCHEDULE_URL)) + existing = read_release_schedule() + merged = (pd.concat([existing, parsed], ignore_index=True) + if not existing.empty else parsed) + merged = merged.drop_duplicates(subset=["report_date", "source"], keep="last") + write_release_schedule(merged) + delayed = int((parsed["note"] == "delayed by a federal holiday").sum()) + return {"scheduled": len(parsed), "holiday_delayed": delayed} + + # ── Backfill ──────────────────────────────────────────────────────────────── _SOURCE_RANK = {"published": 3, "announced": 2, "scheduled": 1} @@ -296,8 +394,12 @@ def sync(*, fetch_html=_fetch_html, now=None) -> dict: html = fetch_html(ANNOUNCEMENTS_URL) rows = _parse_announcements(html, url=ANNOUNCEMENTS_URL, scraped_at=now) if rows: + fresh = pd.DataFrame(rows) existing = read_announcements() - merged = pd.concat([existing, pd.DataFrame(rows)], ignore_index=True) + # Skip the concat when there is nothing to merge into. Concatenating an all-empty + # frame is deprecated in pandas and changes dtype inference, so an empty store + # would otherwise warn now and silently shift column dtypes later. + merged = pd.concat([existing, fresh], ignore_index=True) if not existing.empty else fresh merged = merged.drop_duplicates(subset=["announcement_date", "raw_text"]) write_announcements(merged) return {"announcements": len(rows)} diff --git a/tests/fixtures/cftc_release_schedule_2026.html b/tests/fixtures/cftc_release_schedule_2026.html new file mode 100644 index 0000000..fb36015 --- /dev/null +++ b/tests/fixtures/cftc_release_schedule_2026.html @@ -0,0 +1,7 @@ + +

The following is a tentative schedule of releases through 2026. Federal holidays may delay release by one or two days.

+

2026 Release Schedule

+
MonthDates
January05*09162330
February06132027 
March06132027 
April03101724 
May0108152229
June 051222*26 
July06*10172431
August07142128 
September04111825 
October0209162330
November0616*2030* 
December04111828* 
+

*Delayed release date due to a federal holiday.

diff --git a/tests/test_vintage_schedule.py b/tests/test_vintage_schedule.py index 5fa8377..b74876a 100644 --- a/tests/test_vintage_schedule.py +++ b/tests/test_vintage_schedule.py @@ -2,6 +2,7 @@ Oct–Dec 2025 backlog week resolving to its true announced release date. """ import datetime as dt +from pathlib import Path import pandas as pd import pytest @@ -228,6 +229,78 @@ def http(url, *, etag=None, last_modified=None): assert len(vs.read_release_schedule()) == 1 # no duplicate row +_SCHED_FIXTURE = Path(__file__).parent / "fixtures" / "cftc_release_schedule_2026.html" + + +def test_parses_the_real_cftc_release_schedule(): + """Against a trimmed copy of the live page, so the parser is pinned to real markup.""" + from cotdata.vintage_schedule import parse_release_schedule + df = parse_release_schedule(_SCHED_FIXTURE.read_text()) + + assert len(df) == 52 # one release per week + assert set(df["source"]) == {"scheduled"} + # every report date is a Tuesday; every release is a Friday unless holiday-delayed + assert all(pd.Timestamp(d).weekday() == 1 for d in df["report_date"]) + normal = df[~df["note"].str.contains("holiday")] + assert all(pd.Timestamp(d).weekday() == 4 for d in normal["release_date"]) + # the six 2026 federal-holiday delays, which is the whole point of seeding this + delayed = df[df["note"].str.contains("holiday")] + assert len(delayed) == 6 + assert all(pd.Timestamp(d).weekday() == 0 for d in delayed["release_date"]) # Mondays + + +def test_release_schedule_agrees_with_the_published_timestamp(): + """Independent cross-check: the calendar and the weekly-static Last-Modified must + give the same release date for the same report date.""" + from cotdata.vintage_schedule import ( + _last_modified_to_release_date, + parse_release_schedule, + ) + df = parse_release_schedule(_SCHED_FIXTURE.read_text()) + row = df[df["report_date"] == pd.Timestamp("2026-07-21")].iloc[0] + from_header = _last_modified_to_release_date("Fri, 24 Jul 2026 19:27:59 GMT") + assert pd.Timestamp(row["release_date"]).date() == from_header == dt.date(2026, 7, 24) + + +def test_report_date_for_release_handles_holiday_shift(): + from cotdata.vintage_schedule import report_date_for_release + # normal: Friday release reports the Tuesday three days earlier + assert report_date_for_release(dt.date(2026, 7, 24)) == dt.date(2026, 7, 21) + # delayed: a Monday release still reports the PREVIOUS week's Tuesday, not a new one + assert report_date_for_release(dt.date(2026, 7, 6)) == dt.date(2026, 6, 30) + assert report_date_for_release(dt.date(2026, 6, 22)) == dt.date(2026, 6, 16) + + +def test_parse_raises_rather_than_returning_empty_on_layout_change(): + """A silent empty parse is indistinguishable from 'CFTC published nothing', and would + quietly leave every row on `derived`.""" + from cotdata.vintage_schedule import parse_release_schedule + with pytest.raises(ValueError, match="Release Schedule"): + parse_release_schedule("no heading here") + with pytest.raises(ValueError, match="no "): + parse_release_schedule("

2026 Release Schedule

nothing

") + + +def test_scheduled_upgrades_derived_rows(store_env): + """End to end: a report date with no other evidence goes from `derived` to + `scheduled`, and a holiday week gets its date CORRECTED, not just relabelled.""" + from cotdata import vintage_ingest as vi + from cotdata import vintage_schedule as vs + + # report 2026-06-30: derived would say Friday 2026-07-03; the calendar says Monday + # 2026-07-06, because Independence Day delayed it. + _ingest_one("2026-06-30", observed_at=dt.datetime(2026, 7, 30, tzinfo=dt.timezone.utc)) + assert vs.derive_release_date("2026-06-30") == dt.date(2026, 7, 3) # the wrong guess + + schedule = vs.parse_release_schedule(_SCHED_FIXTURE.read_text()) + counts = vs.backfill(schedule=schedule) + + obs = vi.read_observations() + assert counts["scheduled"] == len(obs) and counts["derived"] == 0 + assert set(obs["release_date_source"]) == {"scheduled"} + assert pd.Timestamp(obs.iloc[0]["release_date"]).date() == dt.date(2026, 7, 6) + + def test_announcement_parse_is_best_effort(): from cotdata.vintage_schedule import _parse_announcements html = "
  • January 5, 2026: revised gold report
"