diff --git a/src/cotdata/vintage_cli.py b/src/cotdata/vintage_cli.py index c0874da..165f1e8 100644 --- a/src/cotdata/vintage_cli.py +++ b/src/cotdata/vintage_cli.py @@ -182,8 +182,12 @@ def main(argv=None) -> int: # ── schedule ──────────────────────────────────────────────────────────────── def _cmd_sched_sync(args) -> int: from . import vintage_schedule + cal = vintage_schedule.sync_release_schedule() + print(f"schedule sync: {cal['scheduled']} release date(s) from the published " + f"calendar ({cal['holiday_delayed']} holiday-delayed).") res = vintage_schedule.sync() print(f"schedule sync: {res['announcements']} announcement row(s) scraped.") + print("run 'cotdata-schedule backfill' to apply them to stored observations.") return 0 diff --git a/src/cotdata/vintage_schedule.py b/src/cotdata/vintage_schedule.py index 37cc8f3..aa50e98 100644 --- a/src/cotdata/vintage_schedule.py +++ b/src/cotdata/vintage_schedule.py @@ -21,6 +21,7 @@ import csv import datetime as dt +import re from email.utils import parsedate_to_datetime from pathlib import Path from zoneinfo import ZoneInfo @@ -189,6 +190,103 @@ def sync_published() -> dict: return {"published": len(derived)} +# ── `scheduled`: the CFTC published release calendar ──────────────────────── +# The page lists RELEASE dates for one year, month by month, marking holiday-delayed ones +# with an asterisk ("*Delayed release date due to a federal holiday."). Those asterisks are +# the entire value here: on a normal week `derived` (report_date + 3, weekend-adjusted) +# already lands on the right Friday, so seeding the calendar changes few dates. What it +# changes is (a) the handful of holiday weeks where derived is wrong by one to three days, +# and (b) the PROVENANCE of every matched row, from `derived` (a guess) to `scheduled` (a +# published fact) — which is what lets strict point-in-time evaluation trust the date. +_MONTHS = {m: i for i, m in enumerate( + ["January", "February", "March", "April", "May", "June", "July", + "August", "September", "October", "November", "December"], start=1)} + + +def _strip_tags(html: str) -> str: + return re.sub(r"\s+", " ", re.sub(r"<[^>]+>", " ", html)).strip() + + +def report_date_for_release(release: dt.date) -> dt.date: + """The Tuesday whose positions a given release reports. + + COT is Tuesday-dated and published the following Friday. A federal holiday pushes the + RELEASE later but never moves the as-of date, so the report date is the latest Tuesday + at least three days before the release. That rule handles both the normal Friday case + (Tuesday + 3) and a delayed Monday release (the previous week's Tuesday), without + needing a holiday calendar of our own. + """ + d = pd.Timestamp(release).date() - dt.timedelta(days=3) + while d.weekday() != 1: # 1 == Tuesday + d -= dt.timedelta(days=1) + return d + + +def parse_release_schedule(html: str) -> pd.DataFrame: + """Parse the CFTC release-schedule page into schedule rows. + + Tag-tolerant by design: the year lives inside nested markup + (``
| Month | Dates | ||||
| January | 05* | 09 | 16 | 23 | 30 |
| February | 06 | 13 | 20 | 27 | |
| March | 06 | 13 | 20 | 27 | |
| April | 03 | 10 | 17 | 24 | |
| May | 01 | 08 | 15 | 22 | 29 |
| June | 05 | 12 | 22* | 26 | |
| July | 06* | 10 | 17 | 24 | 31 |
| August | 07 | 14 | 21 | 28 | |
| September | 04 | 11 | 18 | 25 | |
| October | 02 | 09 | 16 | 23 | 30 |
| November | 06 | 16* | 20 | 30* | |
| December | 04 | 11 | 18 | 28* | |
*Delayed release date due to a federal holiday.
diff --git a/tests/test_vintage_schedule.py b/tests/test_vintage_schedule.py index 5fa8377..b74876a 100644 --- a/tests/test_vintage_schedule.py +++ b/tests/test_vintage_schedule.py @@ -2,6 +2,7 @@ Oct–Dec 2025 backlog week resolving to its true announced release date. """ import datetime as dt +from pathlib import Path import pandas as pd import pytest @@ -228,6 +229,78 @@ def http(url, *, etag=None, last_modified=None): assert len(vs.read_release_schedule()) == 1 # no duplicate row +_SCHED_FIXTURE = Path(__file__).parent / "fixtures" / "cftc_release_schedule_2026.html" + + +def test_parses_the_real_cftc_release_schedule(): + """Against a trimmed copy of the live page, so the parser is pinned to real markup.""" + from cotdata.vintage_schedule import parse_release_schedule + df = parse_release_schedule(_SCHED_FIXTURE.read_text()) + + assert len(df) == 52 # one release per week + assert set(df["source"]) == {"scheduled"} + # every report date is a Tuesday; every release is a Friday unless holiday-delayed + assert all(pd.Timestamp(d).weekday() == 1 for d in df["report_date"]) + normal = df[~df["note"].str.contains("holiday")] + assert all(pd.Timestamp(d).weekday() == 4 for d in normal["release_date"]) + # the six 2026 federal-holiday delays, which is the whole point of seeding this + delayed = df[df["note"].str.contains("holiday")] + assert len(delayed) == 6 + assert all(pd.Timestamp(d).weekday() == 0 for d in delayed["release_date"]) # Mondays + + +def test_release_schedule_agrees_with_the_published_timestamp(): + """Independent cross-check: the calendar and the weekly-static Last-Modified must + give the same release date for the same report date.""" + from cotdata.vintage_schedule import ( + _last_modified_to_release_date, + parse_release_schedule, + ) + df = parse_release_schedule(_SCHED_FIXTURE.read_text()) + row = df[df["report_date"] == pd.Timestamp("2026-07-21")].iloc[0] + from_header = _last_modified_to_release_date("Fri, 24 Jul 2026 19:27:59 GMT") + assert pd.Timestamp(row["release_date"]).date() == from_header == dt.date(2026, 7, 24) + + +def test_report_date_for_release_handles_holiday_shift(): + from cotdata.vintage_schedule import report_date_for_release + # normal: Friday release reports the Tuesday three days earlier + assert report_date_for_release(dt.date(2026, 7, 24)) == dt.date(2026, 7, 21) + # delayed: a Monday release still reports the PREVIOUS week's Tuesday, not a new one + assert report_date_for_release(dt.date(2026, 7, 6)) == dt.date(2026, 6, 30) + assert report_date_for_release(dt.date(2026, 6, 22)) == dt.date(2026, 6, 16) + + +def test_parse_raises_rather_than_returning_empty_on_layout_change(): + """A silent empty parse is indistinguishable from 'CFTC published nothing', and would + quietly leave every row on `derived`.""" + from cotdata.vintage_schedule import parse_release_schedule + with pytest.raises(ValueError, match="Release Schedule"): + parse_release_schedule("no heading here") + with pytest.raises(ValueError, match="no