Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 4 additions & 0 deletions src/cotdata/vintage_cli.py
Original file line number Diff line number Diff line change
Expand Up @@ -182,8 +182,12 @@ def main(argv=None) -> int:
# ── schedule ────────────────────────────────────────────────────────────────
def _cmd_sched_sync(args) -> int:
from . import vintage_schedule
cal = vintage_schedule.sync_release_schedule()
print(f"schedule sync: {cal['scheduled']} release date(s) from the published "
f"calendar ({cal['holiday_delayed']} holiday-delayed).")
res = vintage_schedule.sync()
print(f"schedule sync: {res['announcements']} announcement row(s) scraped.")
print("run 'cotdata-schedule backfill' to apply them to stored observations.")
return 0


Expand Down
104 changes: 103 additions & 1 deletion src/cotdata/vintage_schedule.py
Original file line number Diff line number Diff line change
Expand Up @@ -21,6 +21,7 @@

import csv
import datetime as dt
import re
from email.utils import parsedate_to_datetime
from pathlib import Path
from zoneinfo import ZoneInfo
Expand Down Expand Up @@ -189,6 +190,103 @@ def sync_published() -> dict:
return {"published": len(derived)}


# ── `scheduled`: the CFTC published release calendar ────────────────────────
# The page lists RELEASE dates for one year, month by month, marking holiday-delayed ones
# with an asterisk ("*Delayed release date due to a federal holiday."). Those asterisks are
# the entire value here: on a normal week `derived` (report_date + 3, weekend-adjusted)
# already lands on the right Friday, so seeding the calendar changes few dates. What it
# changes is (a) the handful of holiday weeks where derived is wrong by one to three days,
# and (b) the PROVENANCE of every matched row, from `derived` (a guess) to `scheduled` (a
# published fact) — which is what lets strict point-in-time evaluation trust the date.
_MONTHS = {m: i for i, m in enumerate(
["January", "February", "March", "April", "May", "June", "July",
"August", "September", "October", "November", "December"], start=1)}


def _strip_tags(html: str) -> str:
return re.sub(r"\s+", " ", re.sub(r"<[^>]+>", " ", html)).strip()


def report_date_for_release(release: dt.date) -> dt.date:
"""The Tuesday whose positions a given release reports.

COT is Tuesday-dated and published the following Friday. A federal holiday pushes the
RELEASE later but never moves the as-of date, so the report date is the latest Tuesday
at least three days before the release. That rule handles both the normal Friday case
(Tuesday + 3) and a delayed Monday release (the previous week's Tuesday), without
needing a holiday calendar of our own.
"""
d = pd.Timestamp(release).date() - dt.timedelta(days=3)
while d.weekday() != 1: # 1 == Tuesday
d -= dt.timedelta(days=1)
return d


def parse_release_schedule(html: str) -> pd.DataFrame:
"""Parse the CFTC release-schedule page into schedule rows.

Tag-tolerant by design: the year lives inside nested markup
(``<h3><strong>2026 Release Schedule</strong></h3>``), so it is read from tag-stripped
text rather than from an assumed tag shape. Raises rather than returning an empty
frame if the year or table cannot be found — a silent empty parse would look exactly
like "CFTC published nothing", and would quietly leave every row on `derived`.
"""
text = _strip_tags(html)
ym = re.search(r"(20\d{2})\s+Release Schedule", text, re.I)
if not ym:
raise ValueError("release schedule: could not find the '<YYYY> Release Schedule' "
"heading; the page layout has probably changed.")
year = int(ym.group(1))

tm = re.search(r"<table.*?</table>", html, re.S | re.I)
if not tm:
raise ValueError("release schedule: no <table> found on the page.")
cells = re.findall(r"<t[dh][^>]*>(.*?)</t[dh]>", tm.group(0), re.S | re.I)

rows, month = [], None
for cell in cells:
val = _strip_tags(cell).replace("\xa0", " ").replace("&nbsp;", " ").strip()
if not val:
continue
if val in _MONTHS:
month = _MONTHS[val]
continue
m = re.fullmatch(r"(\d{1,2})\s*(\*?)", val)
if not m or month is None:
continue
release = dt.date(year, month, int(m.group(1)))
delayed = bool(m.group(2))
rows.append({
"report_date": pd.Timestamp(report_date_for_release(release)),
"release_date": pd.Timestamp(release),
"source": "scheduled",
"note": ("delayed by a federal holiday" if delayed
else "published CFTC release schedule"),
"ingested_at": pd.Timestamp.now("UTC"),
})
if not rows:
raise ValueError("release schedule: table found but no dates parsed out of it.")
return pd.DataFrame(rows)


def sync_release_schedule(*, fetch_html=None) -> dict:
"""Fetch and store the published release calendar. Idempotent.

Covers ONE year: the page only ever shows the current schedule, and CFTC does not
publish past calendars here, so earlier years cannot be recovered this way and stay
on `announced` or `derived`.
"""
fetch_html = fetch_html or _fetch_html
parsed = parse_release_schedule(fetch_html(RELEASE_SCHEDULE_URL))
existing = read_release_schedule()
merged = (pd.concat([existing, parsed], ignore_index=True)
if not existing.empty else parsed)
merged = merged.drop_duplicates(subset=["report_date", "source"], keep="last")
write_release_schedule(merged)
delayed = int((parsed["note"] == "delayed by a federal holiday").sum())
return {"scheduled": len(parsed), "holiday_delayed": delayed}


# ── Backfill ────────────────────────────────────────────────────────────────
_SOURCE_RANK = {"published": 3, "announced": 2, "scheduled": 1}

Expand Down Expand Up @@ -296,8 +394,12 @@ def sync(*, fetch_html=_fetch_html, now=None) -> dict:
html = fetch_html(ANNOUNCEMENTS_URL)
rows = _parse_announcements(html, url=ANNOUNCEMENTS_URL, scraped_at=now)
if rows:
fresh = pd.DataFrame(rows)
existing = read_announcements()
merged = pd.concat([existing, pd.DataFrame(rows)], ignore_index=True)
# Skip the concat when there is nothing to merge into. Concatenating an all-empty
# frame is deprecated in pandas and changes dtype inference, so an empty store
# would otherwise warn now and silently shift column dtypes later.
merged = pd.concat([existing, fresh], ignore_index=True) if not existing.empty else fresh
merged = merged.drop_duplicates(subset=["announcement_date", "raw_text"])
write_announcements(merged)
return {"announcements": len(rows)}
Expand Down
7 changes: 7 additions & 0 deletions tests/fixtures/cftc_release_schedule_2026.html
Original file line number Diff line number Diff line change
@@ -0,0 +1,7 @@
<!-- Trimmed from the live CFTC release-schedule page, fetched 2026-07-30.
Navigation and boilerplate removed; the intro, year heading, table and
footnote are verbatim, because those are what the parser depends on. -->
<p>The following is a tentative schedule of releases through 2026. Federal holidays may delay release by one or two days.</p>
<h3><strong>2026 Release Schedule</strong></h3>
<table class="table" style="border-collapse:collapse;width:308pt;" border="0" cellpadding="0" cellspacing="0" width="410"><tbody><tr style="height:14.5pt;" height="19"><td class="xl71" style="border-top-style:none;height:14.5pt;" height="19">Month</td><td class="xl71" style="border-left-style:none;border-top-style:none;" colspan="5">Dates</td></tr><tr style="height:14.5pt;" height="19"><td class="xl68" style="border-top-style:none;height:14.5pt;" height="19">January</td><td class="xl69" style="border-left-style:none;border-top-style:none;">05*</td><td class="xl69" style="border-left-style:none;border-top-style:none;">09</td><td class="xl69" style="border-left-style:none;border-top-style:none;">16</td><td class="xl69" style="border-left-style:none;border-top-style:none;">23</td><td class="xl69" style="border-left-style:none;border-top-style:none;">30</td></tr><tr style="height:14.5pt;" height="19"><td class="xl68" style="border-top-style:none;height:14.5pt;" height="19">February</td><td class="xl69" style="border-left-style:none;border-top-style:none;">06</td><td class="xl69" style="border-left-style:none;border-top-style:none;">13</td><td class="xl69" style="border-left-style:none;border-top-style:none;">20</td><td class="xl69" style="border-left-style:none;border-top-style:none;">27</td><td class="xl69" style="border-left-style:none;border-top-style:none;">&nbsp;</td></tr><tr style="height:14.5pt;" height="19"><td class="xl68" style="border-top-style:none;height:14.5pt;" height="19">March</td><td class="xl69" style="border-left-style:none;border-top-style:none;">06</td><td class="xl69" style="border-left-style:none;border-top-style:none;">13</td><td class="xl69" style="border-left-style:none;border-top-style:none;">20</td><td class="xl69" style="border-left-style:none;border-top-style:none;">27</td><td class="xl69" style="border-left-style:none;border-top-style:none;">&nbsp;</td></tr><tr style="height:14.5pt;" height="19"><td class="xl68" style="border-top-style:none;height:14.5pt;" height="19">April</td><td class="xl69" style="border-left-style:none;border-top-style:none;">03</td><td class="xl69" style="border-left-style:none;border-top-style:none;">10</td><td class="xl69" style="border-left-style:none;border-top-style:none;">17</td><td class="xl69" style="border-left-style:none;border-top-style:none;">24</td><td class="xl69" style="border-left-style:none;border-top-style:none;">&nbsp;</td></tr><tr style="height:14.5pt;" height="19"><td class="xl68" style="border-top-style:none;height:14.5pt;" height="19">May</td><td class="xl69" style="border-left-style:none;border-top-style:none;">01</td><td class="xl69" style="border-left-style:none;border-top-style:none;">08</td><td class="xl69" style="border-left-style:none;border-top-style:none;">15</td><td class="xl69" style="border-left-style:none;border-top-style:none;">22</td><td class="xl69" style="border-left-style:none;border-top-style:none;">29</td></tr><tr style="height:14.5pt;" height="19"><td class="xl68" style="border-top-style:none;height:14.5pt;" height="19">June<span>&nbsp;</span></td><td class="xl69" style="border-left-style:none;border-top-style:none;">05</td><td class="xl69" style="border-left-style:none;border-top-style:none;">12</td><td class="xl69" style="border-left-style:none;border-top-style:none;">22*</td><td class="xl69" style="border-left-style:none;border-top-style:none;">26</td><td class="xl69" style="border-left-style:none;border-top-style:none;">&nbsp;</td></tr><tr style="height:14.5pt;" height="19"><td class="xl68" style="border-top-style:none;height:14.5pt;" height="19">July</td><td class="xl69" style="border-left-style:none;border-top-style:none;">06*</td><td class="xl69" style="border-left-style:none;border-top-style:none;">10</td><td class="xl69" style="border-left-style:none;border-top-style:none;">17</td><td class="xl69" style="border-left-style:none;border-top-style:none;">24</td><td class="xl69" style="border-left-style:none;border-top-style:none;">31</td></tr><tr style="height:14.5pt;" height="19"><td class="xl68" style="border-top-style:none;height:14.5pt;" height="19">August</td><td class="xl69" style="border-left-style:none;border-top-style:none;">07</td><td class="xl69" style="border-left-style:none;border-top-style:none;">14</td><td class="xl69" style="border-left-style:none;border-top-style:none;">21</td><td class="xl69" style="border-left-style:none;border-top-style:none;">28</td><td class="xl69" style="border-left-style:none;border-top-style:none;">&nbsp;</td></tr><tr style="height:14.5pt;" height="19"><td class="xl68" style="border-top-style:none;height:14.5pt;" height="19">September</td><td class="xl69" style="border-left-style:none;border-top-style:none;">04</td><td class="xl69" style="border-left-style:none;border-top-style:none;">11</td><td class="xl69" style="border-left-style:none;border-top-style:none;">18</td><td class="xl69" style="border-left-style:none;border-top-style:none;">25</td><td class="xl69" style="border-left-style:none;border-top-style:none;">&nbsp;</td></tr><tr style="height:14.5pt;" height="19"><td class="xl68" style="border-top-style:none;height:14.5pt;" height="19">October</td><td class="xl69" style="border-left-style:none;border-top-style:none;">02</td><td class="xl69" style="border-left-style:none;border-top-style:none;">09</td><td class="xl69" style="border-left-style:none;border-top-style:none;">16</td><td class="xl69" style="border-left-style:none;border-top-style:none;">23</td><td class="xl69" style="border-left-style:none;border-top-style:none;">30</td></tr><tr style="height:14.5pt;" height="19"><td class="xl68" style="border-top-style:none;height:14.5pt;" height="19">November</td><td class="xl69" style="border-left-style:none;border-top-style:none;">06</td><td class="xl69" style="border-left-style:none;border-top-style:none;">16*</td><td class="xl69" style="border-left-style:none;border-top-style:none;">20</td><td class="xl69" style="border-left-style:none;border-top-style:none;">30*</td><td class="xl69" style="border-left-style:none;border-top-style:none;">&nbsp;</td></tr><tr style="height:14.5pt;" height="19"><td class="xl68" style="border-top-style:none;height:14.5pt;" height="19">December</td><td class="xl69" style="border-left-style:none;border-top-style:none;">04</td><td class="xl69" style="border-left-style:none;border-top-style:none;">11</td><td class="xl69" style="border-left-style:none;border-top-style:none;">18</td><td class="xl69" style="border-left-style:none;border-top-style:none;">28*</td><td class="xl69" style="border-left-style:none;border-top-style:none;">&nbsp;</td></tr></tbody></table>
<p>*Delayed release date due to a federal holiday.</p>
73 changes: 73 additions & 0 deletions tests/test_vintage_schedule.py
Original file line number Diff line number Diff line change
Expand Up @@ -2,6 +2,7 @@
Oct–Dec 2025 backlog week resolving to its true announced release date.
"""
import datetime as dt
from pathlib import Path

import pandas as pd
import pytest
Expand Down Expand Up @@ -228,6 +229,78 @@ def http(url, *, etag=None, last_modified=None):
assert len(vs.read_release_schedule()) == 1 # no duplicate row


_SCHED_FIXTURE = Path(__file__).parent / "fixtures" / "cftc_release_schedule_2026.html"


def test_parses_the_real_cftc_release_schedule():
"""Against a trimmed copy of the live page, so the parser is pinned to real markup."""
from cotdata.vintage_schedule import parse_release_schedule
df = parse_release_schedule(_SCHED_FIXTURE.read_text())

assert len(df) == 52 # one release per week
assert set(df["source"]) == {"scheduled"}
# every report date is a Tuesday; every release is a Friday unless holiday-delayed
assert all(pd.Timestamp(d).weekday() == 1 for d in df["report_date"])
normal = df[~df["note"].str.contains("holiday")]
assert all(pd.Timestamp(d).weekday() == 4 for d in normal["release_date"])
# the six 2026 federal-holiday delays, which is the whole point of seeding this
delayed = df[df["note"].str.contains("holiday")]
assert len(delayed) == 6
assert all(pd.Timestamp(d).weekday() == 0 for d in delayed["release_date"]) # Mondays


def test_release_schedule_agrees_with_the_published_timestamp():
"""Independent cross-check: the calendar and the weekly-static Last-Modified must
give the same release date for the same report date."""
from cotdata.vintage_schedule import (
_last_modified_to_release_date,
parse_release_schedule,
)
df = parse_release_schedule(_SCHED_FIXTURE.read_text())
row = df[df["report_date"] == pd.Timestamp("2026-07-21")].iloc[0]
from_header = _last_modified_to_release_date("Fri, 24 Jul 2026 19:27:59 GMT")
assert pd.Timestamp(row["release_date"]).date() == from_header == dt.date(2026, 7, 24)


def test_report_date_for_release_handles_holiday_shift():
from cotdata.vintage_schedule import report_date_for_release
# normal: Friday release reports the Tuesday three days earlier
assert report_date_for_release(dt.date(2026, 7, 24)) == dt.date(2026, 7, 21)
# delayed: a Monday release still reports the PREVIOUS week's Tuesday, not a new one
assert report_date_for_release(dt.date(2026, 7, 6)) == dt.date(2026, 6, 30)
assert report_date_for_release(dt.date(2026, 6, 22)) == dt.date(2026, 6, 16)


def test_parse_raises_rather_than_returning_empty_on_layout_change():
"""A silent empty parse is indistinguishable from 'CFTC published nothing', and would
quietly leave every row on `derived`."""
from cotdata.vintage_schedule import parse_release_schedule
with pytest.raises(ValueError, match="Release Schedule"):
parse_release_schedule("<html><body>no heading here</body></html>")
with pytest.raises(ValueError, match="no <table>"):
parse_release_schedule("<h3>2026 Release Schedule</h3><p>nothing</p>")


def test_scheduled_upgrades_derived_rows(store_env):
"""End to end: a report date with no other evidence goes from `derived` to
`scheduled`, and a holiday week gets its date CORRECTED, not just relabelled."""
from cotdata import vintage_ingest as vi
from cotdata import vintage_schedule as vs

# report 2026-06-30: derived would say Friday 2026-07-03; the calendar says Monday
# 2026-07-06, because Independence Day delayed it.
_ingest_one("2026-06-30", observed_at=dt.datetime(2026, 7, 30, tzinfo=dt.timezone.utc))
assert vs.derive_release_date("2026-06-30") == dt.date(2026, 7, 3) # the wrong guess

schedule = vs.parse_release_schedule(_SCHED_FIXTURE.read_text())
counts = vs.backfill(schedule=schedule)

obs = vi.read_observations()
assert counts["scheduled"] == len(obs) and counts["derived"] == 0
assert set(obs["release_date_source"]) == {"scheduled"}
assert pd.Timestamp(obs.iloc[0]["release_date"]).date() == dt.date(2026, 7, 6)


def test_announcement_parse_is_best_effort():
from cotdata.vintage_schedule import _parse_announcements
html = "<ul><li>January 5, 2026: revised gold report</li><li></li></ul>"
Expand Down
Loading