diff --git a/src/cotdata/vintage_cli.py b/src/cotdata/vintage_cli.py index c0874da..165f1e8 100644 --- a/src/cotdata/vintage_cli.py +++ b/src/cotdata/vintage_cli.py @@ -182,8 +182,12 @@ def main(argv=None) -> int: # ── schedule ──────────────────────────────────────────────────────────────── def _cmd_sched_sync(args) -> int: from . import vintage_schedule + cal = vintage_schedule.sync_release_schedule() + print(f"schedule sync: {cal['scheduled']} release date(s) from the published " + f"calendar ({cal['holiday_delayed']} holiday-delayed).") res = vintage_schedule.sync() print(f"schedule sync: {res['announcements']} announcement row(s) scraped.") + print("run 'cotdata-schedule backfill' to apply them to stored observations.") return 0 diff --git a/src/cotdata/vintage_schedule.py b/src/cotdata/vintage_schedule.py index 37cc8f3..aa50e98 100644 --- a/src/cotdata/vintage_schedule.py +++ b/src/cotdata/vintage_schedule.py @@ -21,6 +21,7 @@ import csv import datetime as dt +import re from email.utils import parsedate_to_datetime from pathlib import Path from zoneinfo import ZoneInfo @@ -189,6 +190,103 @@ def sync_published() -> dict: return {"published": len(derived)} +# ── `scheduled`: the CFTC published release calendar ──────────────────────── +# The page lists RELEASE dates for one year, month by month, marking holiday-delayed ones +# with an asterisk ("*Delayed release date due to a federal holiday."). Those asterisks are +# the entire value here: on a normal week `derived` (report_date + 3, weekend-adjusted) +# already lands on the right Friday, so seeding the calendar changes few dates. What it +# changes is (a) the handful of holiday weeks where derived is wrong by one to three days, +# and (b) the PROVENANCE of every matched row, from `derived` (a guess) to `scheduled` (a +# published fact) — which is what lets strict point-in-time evaluation trust the date. +_MONTHS = {m: i for i, m in enumerate( + ["January", "February", "March", "April", "May", "June", "July", + "August", "September", "October", "November", "December"], start=1)} + + +def _strip_tags(html: str) -> str: + return re.sub(r"\s+", " ", re.sub(r"<[^>]+>", " ", html)).strip() + + +def report_date_for_release(release: dt.date) -> dt.date: + """The Tuesday whose positions a given release reports. + + COT is Tuesday-dated and published the following Friday. A federal holiday pushes the + RELEASE later but never moves the as-of date, so the report date is the latest Tuesday + at least three days before the release. That rule handles both the normal Friday case + (Tuesday + 3) and a delayed Monday release (the previous week's Tuesday), without + needing a holiday calendar of our own. + """ + d = pd.Timestamp(release).date() - dt.timedelta(days=3) + while d.weekday() != 1: # 1 == Tuesday + d -= dt.timedelta(days=1) + return d + + +def parse_release_schedule(html: str) -> pd.DataFrame: + """Parse the CFTC release-schedule page into schedule rows. + + Tag-tolerant by design: the year lives inside nested markup + (``

2026 Release Schedule

``), so it is read from tag-stripped + text rather than from an assumed tag shape. Raises rather than returning an empty + frame if the year or table cannot be found — a silent empty parse would look exactly + like "CFTC published nothing", and would quietly leave every row on `derived`. + """ + text = _strip_tags(html) + ym = re.search(r"(20\d{2})\s+Release Schedule", text, re.I) + if not ym: + raise ValueError("release schedule: could not find the ' Release Schedule' " + "heading; the page layout has probably changed.") + year = int(ym.group(1)) + + tm = re.search(r"", html, re.S | re.I) + if not tm: + raise ValueError("release schedule: no found on the page.") + cells = re.findall(r"]*>(.*?)", tm.group(0), re.S | re.I) + + rows, month = [], None + for cell in cells: + val = _strip_tags(cell).replace("\xa0", " ").replace(" ", " ").strip() + if not val: + continue + if val in _MONTHS: + month = _MONTHS[val] + continue + m = re.fullmatch(r"(\d{1,2})\s*(\*?)", val) + if not m or month is None: + continue + release = dt.date(year, month, int(m.group(1))) + delayed = bool(m.group(2)) + rows.append({ + "report_date": pd.Timestamp(report_date_for_release(release)), + "release_date": pd.Timestamp(release), + "source": "scheduled", + "note": ("delayed by a federal holiday" if delayed + else "published CFTC release schedule"), + "ingested_at": pd.Timestamp.now("UTC"), + }) + if not rows: + raise ValueError("release schedule: table found but no dates parsed out of it.") + return pd.DataFrame(rows) + + +def sync_release_schedule(*, fetch_html=None) -> dict: + """Fetch and store the published release calendar. Idempotent. + + Covers ONE year: the page only ever shows the current schedule, and CFTC does not + publish past calendars here, so earlier years cannot be recovered this way and stay + on `announced` or `derived`. + """ + fetch_html = fetch_html or _fetch_html + parsed = parse_release_schedule(fetch_html(RELEASE_SCHEDULE_URL)) + existing = read_release_schedule() + merged = (pd.concat([existing, parsed], ignore_index=True) + if not existing.empty else parsed) + merged = merged.drop_duplicates(subset=["report_date", "source"], keep="last") + write_release_schedule(merged) + delayed = int((parsed["note"] == "delayed by a federal holiday").sum()) + return {"scheduled": len(parsed), "holiday_delayed": delayed} + + # ── Backfill ──────────────────────────────────────────────────────────────── _SOURCE_RANK = {"published": 3, "announced": 2, "scheduled": 1} @@ -296,8 +394,12 @@ def sync(*, fetch_html=_fetch_html, now=None) -> dict: html = fetch_html(ANNOUNCEMENTS_URL) rows = _parse_announcements(html, url=ANNOUNCEMENTS_URL, scraped_at=now) if rows: + fresh = pd.DataFrame(rows) existing = read_announcements() - merged = pd.concat([existing, pd.DataFrame(rows)], ignore_index=True) + # Skip the concat when there is nothing to merge into. Concatenating an all-empty + # frame is deprecated in pandas and changes dtype inference, so an empty store + # would otherwise warn now and silently shift column dtypes later. + merged = pd.concat([existing, fresh], ignore_index=True) if not existing.empty else fresh merged = merged.drop_duplicates(subset=["announcement_date", "raw_text"]) write_announcements(merged) return {"announcements": len(rows)} diff --git a/tests/fixtures/cftc_release_schedule_2026.html b/tests/fixtures/cftc_release_schedule_2026.html new file mode 100644 index 0000000..fb36015 --- /dev/null +++ b/tests/fixtures/cftc_release_schedule_2026.html @@ -0,0 +1,7 @@ + +

The following is a tentative schedule of releases through 2026. Federal holidays may delay release by one or two days.

+

2026 Release Schedule

+
MonthDates
January05*09162330
February06132027 
March06132027 
April03101724 
May0108152229
June 051222*26 
July06*10172431
August07142128 
September04111825 
October0209162330
November0616*2030* 
December04111828* 
+

*Delayed release date due to a federal holiday.

diff --git a/tests/test_vintage_schedule.py b/tests/test_vintage_schedule.py index 5fa8377..b74876a 100644 --- a/tests/test_vintage_schedule.py +++ b/tests/test_vintage_schedule.py @@ -2,6 +2,7 @@ Oct–Dec 2025 backlog week resolving to its true announced release date. """ import datetime as dt +from pathlib import Path import pandas as pd import pytest @@ -228,6 +229,78 @@ def http(url, *, etag=None, last_modified=None): assert len(vs.read_release_schedule()) == 1 # no duplicate row +_SCHED_FIXTURE = Path(__file__).parent / "fixtures" / "cftc_release_schedule_2026.html" + + +def test_parses_the_real_cftc_release_schedule(): + """Against a trimmed copy of the live page, so the parser is pinned to real markup.""" + from cotdata.vintage_schedule import parse_release_schedule + df = parse_release_schedule(_SCHED_FIXTURE.read_text()) + + assert len(df) == 52 # one release per week + assert set(df["source"]) == {"scheduled"} + # every report date is a Tuesday; every release is a Friday unless holiday-delayed + assert all(pd.Timestamp(d).weekday() == 1 for d in df["report_date"]) + normal = df[~df["note"].str.contains("holiday")] + assert all(pd.Timestamp(d).weekday() == 4 for d in normal["release_date"]) + # the six 2026 federal-holiday delays, which is the whole point of seeding this + delayed = df[df["note"].str.contains("holiday")] + assert len(delayed) == 6 + assert all(pd.Timestamp(d).weekday() == 0 for d in delayed["release_date"]) # Mondays + + +def test_release_schedule_agrees_with_the_published_timestamp(): + """Independent cross-check: the calendar and the weekly-static Last-Modified must + give the same release date for the same report date.""" + from cotdata.vintage_schedule import ( + _last_modified_to_release_date, + parse_release_schedule, + ) + df = parse_release_schedule(_SCHED_FIXTURE.read_text()) + row = df[df["report_date"] == pd.Timestamp("2026-07-21")].iloc[0] + from_header = _last_modified_to_release_date("Fri, 24 Jul 2026 19:27:59 GMT") + assert pd.Timestamp(row["release_date"]).date() == from_header == dt.date(2026, 7, 24) + + +def test_report_date_for_release_handles_holiday_shift(): + from cotdata.vintage_schedule import report_date_for_release + # normal: Friday release reports the Tuesday three days earlier + assert report_date_for_release(dt.date(2026, 7, 24)) == dt.date(2026, 7, 21) + # delayed: a Monday release still reports the PREVIOUS week's Tuesday, not a new one + assert report_date_for_release(dt.date(2026, 7, 6)) == dt.date(2026, 6, 30) + assert report_date_for_release(dt.date(2026, 6, 22)) == dt.date(2026, 6, 16) + + +def test_parse_raises_rather_than_returning_empty_on_layout_change(): + """A silent empty parse is indistinguishable from 'CFTC published nothing', and would + quietly leave every row on `derived`.""" + from cotdata.vintage_schedule import parse_release_schedule + with pytest.raises(ValueError, match="Release Schedule"): + parse_release_schedule("no heading here") + with pytest.raises(ValueError, match="no "): + parse_release_schedule("

2026 Release Schedule

nothing

") + + +def test_scheduled_upgrades_derived_rows(store_env): + """End to end: a report date with no other evidence goes from `derived` to + `scheduled`, and a holiday week gets its date CORRECTED, not just relabelled.""" + from cotdata import vintage_ingest as vi + from cotdata import vintage_schedule as vs + + # report 2026-06-30: derived would say Friday 2026-07-03; the calendar says Monday + # 2026-07-06, because Independence Day delayed it. + _ingest_one("2026-06-30", observed_at=dt.datetime(2026, 7, 30, tzinfo=dt.timezone.utc)) + assert vs.derive_release_date("2026-06-30") == dt.date(2026, 7, 3) # the wrong guess + + schedule = vs.parse_release_schedule(_SCHED_FIXTURE.read_text()) + counts = vs.backfill(schedule=schedule) + + obs = vi.read_observations() + assert counts["scheduled"] == len(obs) and counts["derived"] == 0 + assert set(obs["release_date_source"]) == {"scheduled"} + assert pd.Timestamp(obs.iloc[0]["release_date"]).date() == dt.date(2026, 7, 6) + + def test_announcement_parse_is_best_effort(): from cotdata.vintage_schedule import _parse_announcements html = ""