From 264eca2902b55fb3e92135bb252545160088b7fd Mon Sep 17 00:00:00 2001 From: Matt Spinola Date: Thu, 30 Jul 2026 13:50:57 -0400 Subject: [PATCH 01/16] feat(vintage): immutable raw-snapshot capture for COT vintage tracking Step 1 of the vintage/revision-tracking subsystem (handoff v0.2). Capture is the only time-critical piece: CFTC serves current-state files only and git-history recovery found nothing, so the vintage series can only accumulate forward and an uncaptured weekly release is permanently lost. Ingest/diff/PIT are deferred and will run retroactively over the retained raw bytes. - cotdata-vintage fetch: conditional GET (If-None-Match/If-Modified-Since), sha256, immutable retention under vintage/raw/{source_kind}/{year}/, provenance in a self-owned vintage/manifest.json. Byte-identical regenerations and 304s are recorded but deduped (a changed download is not itself a revision). - Provenance lives in its own manifest file, not the cot-half manifest, so store.reconcile_manifest() cannot ghost-prune snapshot ids as missing parquet. - current/ output guarded byte-identical (tests/test_current_baseline.py) with a golden generated from the clean tree before the subsystem lands. - Design: docs/design/cot_vintage.md (incl. the Last-Modified negative result). Persistence/scope decision: cotdata docs/adr stub -> crucible-stack ADR-0008. Co-Authored-By: Claude Opus 4.8 --- .gitignore | 3 + CHANGELOG.md | 9 + ...-0001-cot-vintage-provenance-in-parquet.md | 13 + docs/design/cot_vintage.md | 129 +++++++++ pyproject.toml | 1 + src/cotdata/vintage.py | 274 ++++++++++++++++++ src/cotdata/vintage_cli.py | 49 ++++ tests/_gen_golden.py | 27 ++ .../fixtures/golden/cot_legacy_golden.parquet | Bin 0 -> 11439 bytes tests/test_current_baseline.py | 59 ++++ tests/test_vintage_capture.py | 135 +++++++++ 11 files changed, 699 insertions(+) create mode 100644 docs/adr/ADR-0001-cot-vintage-provenance-in-parquet.md create mode 100644 docs/design/cot_vintage.md create mode 100644 src/cotdata/vintage.py create mode 100644 src/cotdata/vintage_cli.py create mode 100644 tests/_gen_golden.py create mode 100644 tests/fixtures/golden/cot_legacy_golden.parquet create mode 100644 tests/test_current_baseline.py create mode 100644 tests/test_vintage_capture.py diff --git a/.gitignore b/.gitignore index 4429fd0..ccc6329 100644 --- a/.gitignore +++ b/.gitignore @@ -9,6 +9,9 @@ dist/ # never commit the data store itself /store/ *.parquet +# ...except the tiny committed golden fixture that guards current/ byte-identity +# (tests/test_current_baseline.py). Regenerate with tests/_gen_golden.py. +!tests/fixtures/golden/*.parquet # local Task Scheduler / cron wrapper scripts (machine-specific paths, and on # Linux a Databento API key). Copy templates from docs/examples/ into here. diff --git a/CHANGELOG.md b/CHANGELOG.md index 969cc32..d94ebb6 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -7,6 +7,15 @@ to [Semantic Versioning](https://semver.org/spec/v2.0.0.html). ## [Unreleased] ### Added +- **COT vintage capture** (`cotdata-vintage fetch`) — an immutable, hashed landing + zone for as-published CFTC files under `$COTDATA_STORE/vintage/raw/`, with + provenance (etag, last-modified, sha256, size, retrieved-at) recorded in a + self-owned `vintage/manifest.json`. Purely additive: the current-state store is + byte-identical (guarded by `tests/test_current_baseline.py`). This is step 1 of + the vintage/revision-tracking subsystem — capture must start now because an + uncaptured weekly release is irrecoverable; ingest/diff/PIT land next and run + retroactively over retained raw bytes. Decision recorded in crucible-stack + ADR-0008; design in `docs/design/cot_vintage.md`. - **`propadj` price adjustment** — a proportional (ratio) back-adjusted view derived on read from the stored `unadj` + `backadj` series via `get_prices(symbol, adjustment="propadj")`. It preserves daily percentage diff --git a/docs/adr/ADR-0001-cot-vintage-provenance-in-parquet.md b/docs/adr/ADR-0001-cot-vintage-provenance-in-parquet.md new file mode 100644 index 0000000..8301b4e --- /dev/null +++ b/docs/adr/ADR-0001-cot-vintage-provenance-in-parquet.md @@ -0,0 +1,13 @@ +# ADR stub: COT vintage provenance + +> **This decision is recorded as crucible-stack ADR-0008**, beside ADR-0007, because +> its scope half — *is vintage provenance inside narrowed cotdata's boundary?* — is an +> interpretation of ADR-0007 and must be discoverable from ADR-0007's own directory. +> This file is a local pointer so the cotdata worktree still surfaces the decision. + +**Decision, in one line:** vintage provenance IS in scope for the narrowed `cotdata` +(it is CFTC-positioning provenance), and it persists in the existing Parquet + manifest +contract — no database. + +- Full ADR: `crucible-stack/docs/adr/ADR-0008-cot-vintage-provenance-in-parquet.md` +- Design/working notes: [../design/cot_vintage.md](../design/cot_vintage.md) diff --git a/docs/design/cot_vintage.md b/docs/design/cot_vintage.md new file mode 100644 index 0000000..0c83f6f --- /dev/null +++ b/docs/design/cot_vintage.md @@ -0,0 +1,129 @@ +# COT Vintage Store & Revision Tracking + +Session working notes for the `crowdmon-futures` step-1 build (handoff v0.2). +Authored under `docs/` per workspace governance. This is the durable record of the +discovery phase, the `Last-Modified` spike, and the design as adapted to this repo. + +## 1. Why + +CFTC serves current-state COT files only: there is no as-published endpoint and no +official vintage archive. Trader **reclassification** moves positions between +categories after the fact, silently rewriting the historical baseline that every +downstream rolling z-score / percentile is computed against (precedent: the July +2008 restatement reached back to July 2007). Vintage data can only be accumulated +going forward, so capture must start now — every uncaptured week is permanently +lost. + +## 2. Discovery findings (§2) + +The repo is **Parquet-per-symbol + JSON manifest, no database** — a deliberate +producer/consumer contract, and ADR-0007 is actively narrowing `cotdata` to CFTC +positioning only. The vintage layer must sit *alongside* that contract, additively, +leaving current-state output byte-identical. + +| # | Finding | +|---|---| +| Storage | One parquet per symbol under `$COTDATA_STORE/{cot_legacy,cot_disagg,cot_tff}/`, atomic temp+`os.replace` ([store.py](../../src/cotdata/store.py)). No DB. | +| Raw retention | Year zips are cached at `$COTDATA_STORE/_cache/cot_legacy/` but **overwritten in place** on regeneration (conditional on `Last-Modified`) — a destructive cache, not an immutable landing zone. | +| Schema | Parse output is a `Report_Date`-indexed (tz-naive `DatetimeIndex`) table: OI, category long/short, trader counts ([cftc.py:27](../../src/cotdata/providers/cftc.py#L27)). **No `release_date` anywhere.** | +| Sources | Legacy (`dea_fut_xls_.zip`), Disaggregated (`fut_disagg_txt_.zip`, from 2006), TFF (`fut_fin_txt_.zip`) — all **futures-only, all annual zips**. No combined, no supplemental, no Socrata. | +| Date handling | `Report_Date` stored **as reported**, `format='mixed'`, never rounded to Tuesday ([cftc.py:98](../../src/cotdata/providers/cftc.py#L98)) — already §6-compliant. Release date not captured. | +| Idempotency | Re-fetch: HEAD/`Last-Modified` skips download if cache fresh; parse always rebuilds the full per-code table and **overwrites** the parquet. Idempotent in outcome, but **destroys prior state** — no diff, no history. This is the gap the vintage layer closes. | +| Manifest shape | Per-half JSON (`manifests/cot.json`, `manifests/prices.json`) + legacy aggregate fallback. Each domain is `{name: {last_date, n_rows, source, updated_at}}`, `schema_version` at top. `half_for()` refuses undeclared domains ([store.py:123](../../src/cotdata/store.py#L123)). | +| Git archaeology | **Zero data files ever committed** (checked parquet/csv/zip/txt/db/duckdb/json across all history). No historical vintages recoverable by any means. `vintage recover --from-git` would be a no-op — not built. | + +## 3. NEGATIVE RESULT — `Last-Modified` cannot backfill historical release dates + +**Tested 2026-07-30.** Recorded explicitly as a null so it is not re-investigated: +an afternoon spent rediscovering this is the failure this section prevents. The +short version — *annual zips regenerate weekly, so their `Last-Modified` carries no +per-week information, and past weekly statics are not archived.* + +Probed CFTC headers directly (network to cftc.gov works from this host, contrary to +the sandbox caveat noted in CLAUDE.md): + +| File | `Last-Modified` (UTC) | ET | Verdict | +|---|---|---|---| +| `dea_fut_xls_2025.zip` (annual) | Fri 24 Jul 2026 19:27:59 | 15:27 Fri | **Regenerated weekly.** A 2025 file stamped 2026-07-24 proves the annual zip's timestamp tracks the latest regeneration, not any week's publication. Useless as a per-week release date. | +| `deafut.txt` (weekly static) | Fri 24 Jul 2026 19:27:44 | 15:27 Fri | ~3 min before the nominal Fri 15:30 ET Legacy release. **Reflects true publication time — current week only.** | + +**Conclusions** + +1. **The taxonomy does not collapse.** Historical release dates cannot be backfilled + from `Last-Modified`: annual zips are regenerated weekly (worthless per-week), and + past weekly statics are overwritten, not archived. `scheduled` (published release + schedule) + `announced` (Special Announcements) remain the only historical + sources — the §4.6 fallback chain stands as written. +2. **Bonus for forward capture:** the weekly-static `Last-Modified` *is* a real + publication timestamp. Captured at fetch time it gives a cleaner `observed` + release date than "first `observed_at` at poll interval." Refines §4.6; does not + change the plan's shape. +3. Answers the §11 open question negatively: weekly statics are a single + overwritten file, **not a frozen historical archive** — no further historical + vintage recovery is possible through them. + +## 4. Design as adapted to this repo + +Layout under `$COTDATA_STORE/vintage/` (a new subtree, disjoint from `current/` which +is the untouched existing `cot_legacy/` etc.): + +``` +vintage/raw/{source_kind}/{year}/{retrieved_at}_{sha8}.{ext} immutable +vintage/observations/report_year=YYYY/*.parquet change-only rows +vintage/revisions/detected_year=YYYY/*.parquet append-only, field-level +vintage/release_schedule.parquet +vintage/announcements.parquet +``` + +Manifest: a new `vintage` block on the **cot-half** file (`manifests/cot.json`), so it +stays on the CFTC producer's side of the ADR-0007 seam. `store._DOMAIN_HALF` gains +`vintage -> cot`. + +### §4.6 amendment (confirmed, not contingent) +The spike upgrades the `observed` mechanism: capturing the weekly-static +`Last-Modified` **at fetch time** yields a *true publication timestamp*, not merely a +value accurate to the polling interval. So `observed` is now the confirmed primary +path for weeks captured going forward, and §4.6's precedence chain (`observed` > +`announced` > `scheduled` > `derived`) otherwise stands exactly as written — the +fallback path is the confirmed path, not a contingency. + +### Deviations from v0.2 (and why) +- **pandas/pyarrow, not polars.** polars is not a dependency; pandas + `pyarrow>=10` + are. Implementing change-only-insert / PIT in pandas avoids adding a datastore-like + dep, consistent with the no-DB decision. +- **Fixtures are synthetic**, matching the repo's existing test idiom (`tests/test_cot.py` + builds DataFrames directly), rather than trimmed real `.xls` — deterministic, offline, + and avoids an `xlrd` fixture dependency. Capture is tested with small byte payloads. +- **ADR home is crucible-stack** (beside ADR-0007), numbered ADR-0008. cotdata keeps a + one-line stub pointing to it; the full ADR is moved in the crucible-stack worktree. +- **ADR is cotdata-local** (`docs/adr/ADR-0001`), cross-referencing crucible-stack + ADR-0007, rather than editing the sibling repo from this worktree. +- **`vintage recover --from-git` not built** — git archaeology found nothing to + recover. +- **`current/` unchanged**: the vintage layer reads the same parsed frames the + existing providers already build; it does not alter their write path. + +### Canonical schema +Natural key `(report_date, market_code, report_type, combined, category)`. `combined` +is `False` for every row this session (only futures-only is fetched) but is kept in +the key so a future combined series never collides. Controlled vocab per report type +is derived from the provider column constants (Legacy: `commercial/noncommercial/ +nonreportable`; Disagg: `producer_merchant/swap/managed_money/other_reportable`; TFF: +`dealer/asset_manager/leveraged/other_reportable`). + +## 5. Scope this session (§10) + +In: raw snapshot capture; change-only observation writes; field-level revisions; +release schedule + announcements ingest + `release_date`/`release_date_source` +backfill; §8 tests; `current/` byte-identical. Deferred: `vintage stats`, +category-migration detection, tombstone *logic* (column present), weekly-static +fetching beyond the spike. + +## Bottom line + +Discovery and the spike are done. The genuine null on historical recovery holds: no +past vintages exist to recover, from git or from CFTC, so the vintage series starts +from this session's first capture. The one materially useful thing against *existing* +history is the release-date backfill (schedule + announcements), which the spike +confirmed cannot be shortcut via `Last-Modified` and must come from the two published +sources — most valuably for the Oct–Dec 2025 backlog weeks. diff --git a/pyproject.toml b/pyproject.toml index 2eb9fa2..3a2e240 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -45,6 +45,7 @@ dev = ["pytest>=7", "ruff==0.15.22"] cotdata-update = "cotdata.update:main" cotdata-cot = "cotdata.update:main_cot" cotdata-prices = "cotdata.update:main_prices" +cotdata-vintage = "cotdata.vintage_cli:main" [tool.setuptools.packages.find] where = ["src"] diff --git a/src/cotdata/vintage.py b/src/cotdata/vintage.py new file mode 100644 index 0000000..f759146 --- /dev/null +++ b/src/cotdata/vintage.py @@ -0,0 +1,274 @@ +"""COT vintage capture + provenance — step 1 of the crowdmon-futures build (handoff v0.2). + +This module is purely ADDITIVE alongside the current-state store: it never touches the +existing ``cot_legacy/`` etc. write path, so existing consumers see byte-identical output. + +Layout (all under ``$COTDATA_STORE/vintage/``): + + raw/{source_kind}/{year}/{retrieved_at}_{sha8}.{ext} immutable, never rewritten + observations/report_year=YYYY/*.parquet change-only bitemporal rows + revisions/detected_year=YYYY/*.parquet append-only, field-level + release_schedule.parquet + announcements.parquet + manifest.json vintage provenance index + +Provenance lives in its OWN ``vintage/manifest.json`` rather than a block in the cot-half +manifest, because ``store.reconcile_manifest`` ghost-prunes any manifest entry without a +matching ``{name}.parquet`` — raw snapshot ids are not parquet files and would be wiped. +A self-owned file also matches the repo's existing "one writer per manifest file" split. + +See docs/design/cot_vintage.md and crucible-stack ADR-0008. + +COMMIT 1 (this file, first pass): raw snapshot CAPTURE only — fetch, hash, retain, +record. The immutable landing zone is the only time-critical piece: an uncaptured +weekly release is irrecoverable, whereas ingest/diff can run retroactively over +retained raw bytes at any later point. +""" +from __future__ import annotations + +import datetime as dt +import hashlib +import json +import os +import time +from dataclasses import dataclass +from pathlib import Path + +from . import config + +_UA = "cotdata-vintage/0.1 (COT research; contact matt.spinola@gmail.com)" +_RATE_LIMIT_S = 1.0 # polite spacing between real network requests +SCHEMA_VERSION = 1 + + +# ── Paths ─────────────────────────────────────────────────────────────────── +def vintage_root() -> Path: + return config.store_root() / "vintage" + + +def raw_dir(source_kind: str, year: int | str) -> Path: + return vintage_root() / "raw" / source_kind / str(year) + + +def manifest_path() -> Path: + return vintage_root() / "manifest.json" + + +# ── Sources ───────────────────────────────────────────────────────────────── +@dataclass(frozen=True) +class Source: + """One fetchable CFTC file. ``report_year`` is None for the current-week static.""" + report_type: str # legacy | disaggregated | tff + source_kind: str # annual_zip | weekly_static + ext: str # zip | txt + url: str + report_year: int | None = None + + +# Annual-zip URL builders. The Legacy naming convention changed in 2004; disagg/TFF +# history starts in 2006. Duplicated from providers/ deliberately — capture stays +# decoupled from the parse path (non-goal: do not refactor existing fetch logic). +def _legacy_zip_url(year: int) -> str: + if year < 2004: + return f"https://www.cftc.gov/files/dea/history/deafut_xls_{year}.zip" + return f"https://www.cftc.gov/files/dea/history/dea_fut_xls_{year}.zip" + + +def annual_sources(year: int) -> list[Source]: + out = [Source("legacy", "annual_zip", "zip", _legacy_zip_url(year), year)] + if year >= 2006: + out.append(Source("disaggregated", "annual_zip", "zip", + f"https://www.cftc.gov/files/dea/history/fut_disagg_txt_{year}.zip", year)) + out.append(Source("tff", "annual_zip", "zip", + f"https://www.cftc.gov/files/dea/history/fut_fin_txt_{year}.zip", year)) + return out + + +# The current-week static. Its HTTP Last-Modified is a TRUE publication timestamp +# (spike 2026-07-30, docs/design/cot_vintage.md §3), so capturing it upgrades the +# `observed` release-date mechanism from poll-accurate to publication-accurate. +WEEKLY_STATIC = Source("legacy", "weekly_static", "txt", + "https://www.cftc.gov/dea/newcot/deafut.txt", None) + + +# ── HTTP ──────────────────────────────────────────────────────────────────── +@dataclass +class HttpResult: + status: int + content: bytes | None + etag: str | None = None + last_modified: str | None = None + + +def _http_get(url: str, *, etag: str | None, last_modified: str | None) -> HttpResult: + """Conditional GET with If-None-Match / If-Modified-Since. Real network path. + + Injected as ``http_get`` in tests so the capture logic runs fully offline. + """ + import requests # local import: keeps the module importable without network deps + + headers = {"User-Agent": _UA} + if etag: + headers["If-None-Match"] = etag + if last_modified: + headers["If-Modified-Since"] = last_modified + r = requests.get(url, headers=headers, timeout=180) + if r.status_code == 304: + return HttpResult(304, None) + r.raise_for_status() + return HttpResult(r.status_code, r.content, + etag=r.headers.get("ETag"), + last_modified=r.headers.get("Last-Modified")) + + +# ── Manifest (self-owned, append-only snapshot index) ─────────────────────── +def _read_manifest() -> dict: + p = manifest_path() + if not p.exists(): + return {"schema_version": SCHEMA_VERSION, "snapshots": []} + try: + m = json.loads(p.read_text()) + except json.JSONDecodeError: + return {"schema_version": SCHEMA_VERSION, "snapshots": []} + m.setdefault("snapshots", []) + m.setdefault("schema_version", SCHEMA_VERSION) + return m + + +def _write_manifest(m: dict) -> None: + p = manifest_path() + p.parent.mkdir(parents=True, exist_ok=True) + tmp = p.with_suffix(".json.tmp") + tmp.write_text(json.dumps(m, indent=2, sort_keys=True)) + os.replace(tmp, p) + + +def _latest_for_url(snapshots: list[dict], url: str) -> dict | None: + prev = [s for s in snapshots if s.get("source_url") == url] + return prev[-1] if prev else None + + +def read_snapshots() -> list[dict]: + """All recorded snapshot provenance rows, oldest first.""" + return _read_manifest()["snapshots"] + + +# ── Capture ───────────────────────────────────────────────────────────────── +def _utcnow() -> dt.datetime: + return dt.datetime.now(dt.timezone.utc).replace(microsecond=0) + + +def _iso(ts: dt.datetime) -> str: + return ts.isoformat().replace("+00:00", "Z") + + +def capture_source(source: Source, *, snapshots: list[dict], http_get, now: dt.datetime) -> dict: + """Fetch one source, retain its bytes if new, and return a provenance record. + + Always returns a record (a fetch is always recorded, §4.1), including on 304. + A new raw file is written ONLY when the content sha differs from the most recent + snapshot for the same URL — a byte-identical regeneration is deduped (§3.4: a + changed sha does not imply changed data, and an unchanged sha means nothing new to + retain). Raw bytes, once written, are never rewritten. + """ + prev = _latest_for_url(snapshots, source.url) + res = http_get(source.url, + etag=(prev or {}).get("http_etag"), + last_modified=(prev or {}).get("http_last_modified")) + retrieved_at = _iso(now) + base = { + "source_url": source.url, + "source_kind": source.source_kind, + "report_type": source.report_type, + "report_year": source.report_year, + "retrieved_at": retrieved_at, + "parse_status": "pending", + "parse_error": None, + } + + if res.status == 304 or res.content is None: + # Server says unchanged: record the check, retain nothing new, reuse prior file. + return { + **base, + "snapshot_id": f"{retrieved_at}_304", + "http_status": 304, + "http_etag": (prev or {}).get("http_etag"), + "http_last_modified": (prev or {}).get("http_last_modified"), + "content_sha256": (prev or {}).get("content_sha256"), + "byte_size": None, + "local_path": (prev or {}).get("local_path"), + "note": "304 not-modified", + } + + sha = hashlib.sha256(res.content).hexdigest() + if prev and prev.get("content_sha256") == sha: + # Byte-identical to what we already retained (zips regenerate): dedupe, no rewrite. + return { + **base, + "snapshot_id": f"{retrieved_at}_{sha[:8]}", + "http_status": res.status, + "http_etag": res.etag, + "http_last_modified": res.last_modified, + "content_sha256": sha, + "byte_size": len(res.content), + "local_path": prev.get("local_path"), + "note": "unchanged bytes (deduped)", + } + + # New bytes: retain immutably. + compact = retrieved_at.replace("-", "").replace(":", "") + fname = f"{compact}_{sha[:8]}.{source.ext}" + dest = raw_dir(source.source_kind, source.report_year or "current") / fname + dest.parent.mkdir(parents=True, exist_ok=True) + dest.write_bytes(res.content) + rel = str(dest.relative_to(config.store_root())) + return { + **base, + "snapshot_id": f"{retrieved_at}_{sha[:8]}", + "http_status": res.status, + "http_etag": res.etag, + "http_last_modified": res.last_modified, + "content_sha256": sha, + "byte_size": len(res.content), + "local_path": rel, + "note": None, + } + + +def fetch(year: int | None = None, *, all_years: bool = False, + include_weekly: bool = True, sources: list[Source] | None = None, + http_get=_http_get, rate_limit_s: float = _RATE_LIMIT_S, now_fn=_utcnow) -> dict: + """Capture the release-critical CFTC files into the immutable landing zone. + + Default: the current year's three annual zips (which carry the newest release) plus + the Legacy weekly static (for its true-publication Last-Modified). ``all_years`` + walks 1986→current for a cold-start backfill of raw bytes. An explicit ``sources`` + list overrides the year/weekly derivation (targeted capture; used in tests). + + Returns ``{"records": [...], "new_files": n, "checks": n}``. + """ + if sources is None: + this_year = now_fn().year + years = range(1986, this_year + 1) if all_years else [year or this_year] + sources = [] + for y in years: + sources.extend(annual_sources(y)) + if include_weekly: + sources.append(WEEKLY_STATIC) + + m = _read_manifest() + snapshots = m["snapshots"] + new_files = 0 + records = [] + for i, src in enumerate(sources): + rec = capture_source(src, snapshots=snapshots, http_get=http_get, now=now_fn()) + snapshots.append(rec) # visible to the next source's _latest_for_url + records.append(rec) + if rec.get("note") is None and rec.get("byte_size") is not None: + new_files += 1 + if rate_limit_s and i < len(sources) - 1: + time.sleep(rate_limit_s) + + m["schema_version"] = SCHEMA_VERSION + _write_manifest(m) + return {"records": records, "new_files": new_files, "checks": len(records)} diff --git a/src/cotdata/vintage_cli.py b/src/cotdata/vintage_cli.py new file mode 100644 index 0000000..a547a15 --- /dev/null +++ b/src/cotdata/vintage_cli.py @@ -0,0 +1,49 @@ +"""`cotdata-vintage` — capture and query the COT vintage store. + +Subcommands (mirrors handoff §7, adapted to the repo's hyphenated entry-point style): + + cotdata-vintage fetch [--year YYYY | --all] [--no-weekly] + +COMMIT 1 ships ``fetch`` (raw snapshot capture) only. ``ingest``, ``diff``, ``asof`` +and the ``cotdata-schedule`` commands land with the substantive subsystem in commit 2. +""" +import argparse + +from . import config + + +def _cmd_fetch(args) -> int: + from . import vintage + res = vintage.fetch(year=args.year, all_years=args.all, + include_weekly=not args.no_weekly) + print(f"vintage fetch: {res['checks']} source(s) checked, " + f"{res['new_files']} new raw file(s) retained.") + for rec in res["records"]: + tag = rec.get("note") or "NEW" + print(f" {rec['report_type']:<13} {rec['source_kind']:<13} {tag}") + return 0 + + +def main(argv=None) -> int: + p = argparse.ArgumentParser( + prog="cotdata-vintage", + description="Capture and query the COT vintage (as-published) store.") + sub = p.add_subparsers(dest="cmd", required=True) + + f = sub.add_parser("fetch", help="Capture raw CFTC files into the immutable landing zone.") + f.add_argument("--year", type=int, default=None, + help="Report year to capture (default: current year).") + f.add_argument("--all", action="store_true", + help="Capture every year 1986→present (cold-start raw backfill).") + f.add_argument("--no-weekly", action="store_true", + help="Skip the current-week static (whose Last-Modified is the true " + "publication timestamp).") + f.set_defaults(func=_cmd_fetch) + + args = p.parse_args(argv) + config.store_root() # fail fast if COTDATA_STORE unset + return args.func(args) + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tests/_gen_golden.py b/tests/_gen_golden.py new file mode 100644 index 0000000..8968232 --- /dev/null +++ b/tests/_gen_golden.py @@ -0,0 +1,27 @@ +"""Regenerate the current/ golden baseline fixture. Run on a CLEAN tree (before the +vintage subsystem alters anything) so the guard in test_current_baseline.py is meaningful: + + PYTHONPATH=src python tests/_gen_golden.py +""" +import tempfile +from pathlib import Path + +from test_current_baseline import golden_frame # type: ignore + + +def main() -> None: + import os + + with tempfile.TemporaryDirectory() as tmp: + os.environ["COTDATA_STORE"] = tmp + from cotdata import store + store.write_cot_legacy("GOLD_088691", golden_frame(), source="cftc") + src = Path(tmp) / "cot_legacy" / "GOLD_088691.parquet" + dest = Path(__file__).parent / "fixtures" / "golden" / "cot_legacy_golden.parquet" + dest.parent.mkdir(parents=True, exist_ok=True) + dest.write_bytes(src.read_bytes()) + print(f"wrote {dest} ({dest.stat().st_size} bytes)") + + +if __name__ == "__main__": + main() diff --git a/tests/fixtures/golden/cot_legacy_golden.parquet b/tests/fixtures/golden/cot_legacy_golden.parquet new file mode 100644 index 0000000000000000000000000000000000000000..398aeac06e9c38556b8ab0fc12d0b07b7ffd5886 GIT binary patch literal 11439 zcmd5?J!~V#6<+CNnZBHJKR(jSfgOks?J37cN|+i;EN~T)J?P_hxscT@II| z1L+dtj>~!P&3xZ`GxO%nLc2mCGL3lZmZ8SQB@$*_~Lb zSYC_E;o-CuvZNYYd)upwdQ>@14E*JH{MzM{mU>K!+iFo@thoy^;jC# zW7+zcdsC!Nq$|q+;Xr!t)hh()-c934BUu{OW7#4SEuDw->?;G|_dQv$@? zdu}{wBufMHWZIH>GVRZ$^N_fafxwR4kSq=Bv1}2Emd-Nu1J=K^;ouuL`&x(y<;N0>xyJ)SdV3kNVIew z(gPFWp(~Q5VLg^DBGJ-$NDNc@=Wis)q|e{}^0m?`L(h|?VLg^DBGFPB$^XNR2*a)p z>+NP&@xH-$uiaf{{aIyj*wm|v)~Lo$>TN~aZ&q_kw>iAaWbS~!hLxBoidC_v=>w&% zSIs_RuhF!uR$#1~JN>3sO=)^_&>ZSjK~>?50Qu`;uiLHe_J$q3)6<64Oi$ZC*Rm7Z z$L-$0*7;|-o;G3O?EiDxV9F4MW?y%ufJ00eObl_oI8YkR!LVBF*|>QvDIw^j<#XsJ zhGQ+I#ogK0KdGMR?gz3Mq?n>NtIDvN%~oTvY8gIvZ)ZNfb>}`=r@j^c9sdv4gUlW8 zHGhOzz2ToR+umj0itqWd@3ZBg8uVODwcy9KepWdr}><=}U#08W<~2u@2E2A{EWZSWtx68x(N z;OW`};c4l@@U!xQ4gZ^~!B4KD`&30j+^3`qf%NyXabHu%r41^@jMociZv&obv_ z3YmIJx-k5#oMyxS^-qI;c@^;K;sx<(>B9Jv8PkUO_4VNIt^+1r#vmpwT^MtEKDHr$ zb~E^|8-PrgHHb`07e;o;_BQArZUz5*3!v#b2cc={!qBc~i4FX#+rekI0h}&<5S*4S z4DNnX+0g&s3x4Ybbh;cubXvMNy6ZV;L;wEu;GbRxbh<=BbXqzS-S;u;TM2r6ANe5z zv#Wh5o0K8@$Jd?&I$EQ7QmyyYQCAxV-d=k@;7Ts=!Tq&>VOb^Jnxw;%fChCJbk#2Po_|b)})`3fMjgG&Qj8X#3c4 zub6nS6?ppeoQab{z_l#IVI9n7n@$k!)ef37%Hcj0s|i*P;|ru^UC_+yARVA72A2TuxOcjfpv*3d ziX+_4xv>I5xH#-CfsD0jtb*G;G}A0Ek&wgj=>f7BIZQ8*l6zS~o$NDBCn!Og(bPLo zzrZNYTEx;k+I)XB`~dZnCmL2EimnejwUJJ0p(k+i2kL{SqW1=*c2PUEfif6l>Czu7 zgF){Y$2o#GGVn$TrQyg!?jaYzekL)SaszZ5fu{_6O(+z0N^cvlz}WBauH)xa*bxLF z5C8G=D?!*4qWeNZ5cUOepU9$7hyECa@iTqyIBm4yn~XyADM2_EpdUPO5Rnr>2W>#u z1N-P(sjfPzB&AI`oI9#*OL|Sqc6K_^T1ri;^&~%TB=~q)E~s)iaVm46oXi#Ca!9R| zq{EK<_;^3nHT$W2zt%0N&7^diI_T|}Hin&AH=$P^ryiw}kx(spJWPq{u2PbQl>|R3 zmqIGq)ih~X6JZ=J+J^q^dWh$ma+K&&nlw^MxuZ&Quagml4%q7TZMD_dR*%WJV*0eP zoicPKi7S^1?M5<={J}42R1WipN-3ftPD+*b3LS1UrGbySfU# zBd1nDZXYPg#HbRMT1E6p&JxT@S5<4tLhH;=F`X;R=|M#Qnc&K2u|454BKnLNd*Hv& z$#a#xe29}Q_ z4)W917R=4bumx-1xn^J=ltQh&B!8;Ng+9a`_RGl;%>6l!Zlzx-ogB(x+dPGNy(aPb z+D5M5NUBFQE!Qe0`B5drM`6zp{Asav@T2!dzozBGIU`n!t(Pf0_QLI>x@L>xg8Zp| z)Alyxor{fs%a35a|6k058y703Q|4mfzAtJ%@C?~5{OCP_e(ZInW5`+m z^L^pQWzqSV<_FdHMa+j<7{mFf)*-)0Nq%-Nb+}Rk&*U;ydW1laS_xIqu zI)oe!`P10XllgGUxjVPdi^!o8oR?5bjiH8tb2p^Rp;c-QdlC!!1bzw!~v5w9a zB|ctA@KCF(756#_Yi5a4upY}Ot%}`g|Sjf zhhS`SA82k6`9Zxa0Ukfbx)187A=r$eW{!{=U5iHG_qeuEC*wD|u$RJ66LsJYA*RN; z$D7BQ1EGVpm+ig~#$2J4_z@Z7AUn#6o2My_!&^=gzpT9vKXXlZES% zAYwZU4qgyom;%`0dN_g#%Es~uJ5j|LnA7*iD-!5ZV%d6*m!)hO@E@#_n5=cxJ7( z8P5z-x5YDwtm800bLuVdcXR#7?thds+m`>6>#K%HwtVWS$Lw{CuzTu^N9Mq6yTzji zBa?WTXQd+^S%{NH?t5UBhdn^{m$|;y=fwj_$EVx{CijFO#_`n)nj7mODwu669!S!@ z-(+mhxhn`K6MXlrw#)p*acjh<-(L|A5)-79eIa^``N8~<^&7MJM-#@Pot!_CALhOb zK>SP*T;IxLb^yF2WK97QBys+1lNIu06Hm==@wx9za5NYSAn=#)?z;@T)>j4}j+**| jdcXgGyii*+Uzn{KuhQTzhv3h@7XQa6L5A6dKUMw@@qoZL literal 0 HcmV?d00001 diff --git a/tests/test_current_baseline.py b/tests/test_current_baseline.py new file mode 100644 index 0000000..26a1da2 --- /dev/null +++ b/tests/test_current_baseline.py @@ -0,0 +1,59 @@ +"""Guard: the current-state COT write path must stay byte-identical as the vintage +layer is added (acceptance §7). The golden parquet was generated from the tree BEFORE +the vintage subsystem landed; if this fails, the additive change stopped being additive. + +The golden frame is deterministic and defined here; regenerate the fixture with: + + PYTHONPATH=src python tests/_gen_golden.py +""" +import hashlib +from pathlib import Path + +import pandas as pd +import pytest + +GOLDEN = Path(__file__).parent / "fixtures" / "golden" / "cot_legacy_golden.parquet" + + +def golden_frame() -> pd.DataFrame: + """A fixed, realistic Legacy-COT slice (two weeks, one market).""" + idx = pd.to_datetime(["2026-07-14", "2026-07-21"]) + idx.name = "Report_Date_as_MM_DD_YYYY" + return pd.DataFrame( + { + "Market_and_Exchange_Names": ["GOLD - COMMODITY EXCHANGE INC."] * 2, + "CFTC_Contract_Market_Code": ["088691", "088691"], + "Open_Interest_All": [500000, 505000], + "Comm_Positions_Long_All": [200000, 201000], + "Comm_Positions_Short_All": [250000, 251000], + "NonComm_Positions_Long_All": [150000, 151000], + "NonComm_Positions_Short_All": [90000, 91000], + "NonRept_Positions_Long_All": [40000, 41000], + "NonRept_Positions_Short_All": [30000, 31000], + "Traders_Tot_All": [280, 281], + "Traders_Comm_Long_All": [50, 51], + "Traders_Comm_Short_All": [55, 56], + "Traders_NonComm_Long_All": [60, 61], + "Traders_NonComm_Short_All": [45, 46], + }, + index=idx, + ) + + +@pytest.fixture() +def store_env(tmp_path, monkeypatch): + monkeypatch.setenv("COTDATA_STORE", str(tmp_path)) + return tmp_path + + +def test_current_cot_legacy_output_byte_identical(store_env): + from cotdata import store + + store.write_cot_legacy("GOLD_088691", golden_frame(), source="cftc") + produced = (store_env / "cot_legacy" / "GOLD_088691.parquet").read_bytes() + + assert GOLDEN.exists(), "golden baseline missing — run tests/_gen_golden.py on a clean tree" + expected = GOLDEN.read_bytes() + assert hashlib.sha256(produced).hexdigest() == hashlib.sha256(expected).hexdigest(), ( + "current/ cot_legacy output changed — the vintage layer is no longer additive" + ) diff --git a/tests/test_vintage_capture.py b/tests/test_vintage_capture.py new file mode 100644 index 0000000..2873eb1 --- /dev/null +++ b/tests/test_vintage_capture.py @@ -0,0 +1,135 @@ +"""Vintage raw-snapshot capture (commit 1): immutable landing zone + provenance. + +Fully offline — the HTTP layer is injected. Covers acceptance §9.1 (every fetch +recorded, raw bytes retained) and the dedupe rules of §3.4 / §4.1. +""" +import datetime as dt + +import pytest + + +@pytest.fixture() +def store_env(tmp_path, monkeypatch): + monkeypatch.setenv("COTDATA_STORE", str(tmp_path)) + return tmp_path + + +class _FakeHttp: + """Deterministic HTTP: maps url -> queue of (status, content, etag, last_modified).""" + + def __init__(self, responses): + self._responses = {u: list(v) for u, v in responses.items()} + self.calls = [] + + def __call__(self, url, *, etag=None, last_modified=None): + from cotdata.vintage import HttpResult + self.calls.append((url, etag, last_modified)) + status, content, et, lm = self._responses[url].pop(0) + if status == 304: + return HttpResult(304, None) + return HttpResult(status, content, etag=et, last_modified=lm) + + +def _clock(start="2026-07-30T16:00:00+00:00"): + t = [dt.datetime.fromisoformat(start)] + + def now(): + t[0] += dt.timedelta(seconds=1) + return t[0] + return now + + +def _only(url, *tuples): + return {url: list(tuples)} + + +LEGACY_2026 = "https://www.cftc.gov/files/dea/history/dea_fut_xls_2026.zip" + + +def _legacy_only(): + """A single-source capture list, so tests exercise one URL deterministically.""" + from cotdata.vintage import Source + return [Source("legacy", "annual_zip", "zip", LEGACY_2026, 2026)] + + +def _fetch(vintage, http, now): + return vintage.fetch(sources=_legacy_only(), http_get=http, rate_limit_s=0, now_fn=now) + + +def test_fetch_retains_raw_bytes_and_records_provenance(store_env): + from cotdata import vintage + http = _FakeHttp(_only(LEGACY_2026, (200, b"ZIPBYTES-week1", '"etag1"', "Fri, 24 Jul 2026 19:27:59 GMT"))) + + res = _fetch(vintage, http, _clock()) + + assert res["new_files"] == 1 and res["checks"] == 1 + rec = res["records"][0] + # raw bytes retained on disk, immutably, under the year partition + raw = store_env / rec["local_path"] + assert raw.exists() and raw.read_bytes() == b"ZIPBYTES-week1" + assert raw.parent == store_env / "vintage" / "raw" / "annual_zip" / "2026" + # provenance recorded: sha, size, etag, last-modified, status + import hashlib + assert rec["content_sha256"] == hashlib.sha256(b"ZIPBYTES-week1").hexdigest() + assert rec["byte_size"] == len(b"ZIPBYTES-week1") + assert rec["http_etag"] == '"etag1"' + assert rec["http_last_modified"] == "Fri, 24 Jul 2026 19:27:59 GMT" + assert rec["parse_status"] == "pending" + # and persisted in the vintage manifest + assert len(vintage.read_snapshots()) == 1 + + +def test_304_records_check_without_new_file(store_env): + from cotdata import vintage + http = _FakeHttp(_only( + LEGACY_2026, + (200, b"ZIPBYTES-week1", '"etag1"', "Fri, 24 Jul 2026 19:27:59 GMT"), + (304, None, None, None), + )) + now = _clock() + _fetch(vintage, http, now) + res2 = _fetch(vintage, http, now) + + assert res2["new_files"] == 0 + rec = res2["records"][0] + assert rec["http_status"] == 304 and rec["note"] == "304 not-modified" + # conditional GET actually sent the prior validators + assert http.calls[-1] == (LEGACY_2026, '"etag1"', "Fri, 24 Jul 2026 19:27:59 GMT") + # exactly one raw file on disk; the check is recorded but retains nothing new + raws = list((store_env / "vintage" / "raw").rglob("*.zip")) + assert len(raws) == 1 + assert len(vintage.read_snapshots()) == 2 + + +def test_byte_identical_regeneration_is_deduped(store_env): + """A regenerated zip with identical bytes (200, no 304) must not write a second + raw file — a changed download is not itself a revision (§3.4).""" + from cotdata import vintage + http = _FakeHttp(_only( + LEGACY_2026, + (200, b"SAME", '"e1"', "lm1"), + (200, b"SAME", '"e2"', "lm2"), # server didn't honour conditional GET, but bytes match + )) + now = _clock() + _fetch(vintage, http, now) + res2 = _fetch(vintage, http, now) + + assert res2["new_files"] == 0 + assert res2["records"][0]["note"] == "unchanged bytes (deduped)" + assert len(list((store_env / "vintage" / "raw").rglob("*.zip"))) == 1 + + +def test_changed_bytes_writes_second_immutable_snapshot(store_env): + from cotdata import vintage + http = _FakeHttp(_only( + LEGACY_2026, + (200, b"week1", '"e1"', "lm1"), + (200, b"week2-revised", '"e2"', "lm2"), + )) + now = _clock() + _fetch(vintage, http, now) + _fetch(vintage, http, now) + + raws = sorted((store_env / "vintage" / "raw").rglob("*.zip")) + assert len(raws) == 2 # both vintages retained; neither overwritten + assert {p.read_bytes() for p in raws} == {b"week1", b"week2-revised"} From 0ba2a3cb00464582554c74027789d4ec1c84d917 Mon Sep 17 00:00:00 2001 From: Matt Spinola Date: Thu, 30 Jul 2026 13:59:25 -0400 Subject: [PATCH 02/16] feat(vintage): change-only ingest, field-level revisions, PIT asof, release-date backfill MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Commit 2 of the vintage subsystem — the substantive layer, built on top of the raw snapshots captured in the previous commit. Runs retroactively over retained bytes, so nothing here was time-critical. - ingest_canonical: change-only bitemporal writes keyed on (report_date, market_code, report_type, combined, category). A row is written only when its value-hash (row_sha256 over value fields, excluding provenance and release_date) differs from the latest for its key — re-ingesting identical data, including a byte-changed regeneration, is a no-op. - field-level revisions with age_days (revision depth); asof(t) = greatest observed_at <= t per key (no valid_to column). - release-date resolution with provenance (observed > announced > scheduled > derived); backfill flags the Oct-Dec 2025 backlog weeks as `announced`, not silently `derived`. observed is honoured only for live-captured rows. - best-effort Special Announcements scrape (raw text always retained). - CLI: cotdata-vintage ingest|diff|asof, cotdata-schedule sync|backfill. - report_date stored as-reported (Monday holiday weeks not normalised to Tuesday); timestamps tz-naive UTC to match the repo's convention. Tests cover handoff §8: idempotent ingest, byte-change/data-same, single-field revision, PIT query, revision depth, holiday week, backlog-week announced resolution, release-date precedence, and validation-failure-before-write. Co-Authored-By: Claude Opus 4.8 --- CHANGELOG.md | 9 + pyproject.toml | 1 + src/cotdata/vintage.py | 14 ++ src/cotdata/vintage_cli.py | 145 +++++++++++++-- src/cotdata/vintage_ingest.py | 312 ++++++++++++++++++++++++++++++++ src/cotdata/vintage_schedule.py | 210 +++++++++++++++++++++ tests/test_vintage_ingest.py | 132 ++++++++++++++ tests/test_vintage_schedule.py | 88 +++++++++ 8 files changed, 896 insertions(+), 15 deletions(-) create mode 100644 src/cotdata/vintage_ingest.py create mode 100644 src/cotdata/vintage_schedule.py create mode 100644 tests/test_vintage_ingest.py create mode 100644 tests/test_vintage_schedule.py diff --git a/CHANGELOG.md b/CHANGELOG.md index d94ebb6..1f3d67c 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -16,6 +16,15 @@ to [Semantic Versioning](https://semver.org/spec/v2.0.0.html). uncaptured weekly release is irrecoverable; ingest/diff/PIT land next and run retroactively over retained raw bytes. Decision recorded in crucible-stack ADR-0008; design in `docs/design/cot_vintage.md`. +- **COT vintage ingest + revision tracking** (`cotdata-vintage ingest|diff|asof`, + `cotdata-schedule sync|backfill`) — parses retained raw snapshots into a change-only + bitemporal `observations/` table (a row is written only when its value hash differs + from the latest for its natural key, so storage grows with revisions not with time), + emits field-level `revisions/` with `age_days` revision depth, and answers + point-in-time `asof(t)` reads (greatest `observed_at <= t` per key). Release dates are + resolved with explicit provenance (`observed > announced > scheduled > derived`), + including a backfill that flags the Oct–Dec 2025 appropriations-lapse backlog weeks as + `announced` rather than silently `derived`. All pandas/pyarrow, no database. - **`propadj` price adjustment** — a proportional (ratio) back-adjusted view derived on read from the stored `unadj` + `backadj` series via `get_prices(symbol, adjustment="propadj")`. It preserves daily percentage diff --git a/pyproject.toml b/pyproject.toml index 3a2e240..6b4ad41 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -46,6 +46,7 @@ cotdata-update = "cotdata.update:main" cotdata-cot = "cotdata.update:main_cot" cotdata-prices = "cotdata.update:main_prices" cotdata-vintage = "cotdata.vintage_cli:main" +cotdata-schedule = "cotdata.vintage_cli:main_schedule" [tool.setuptools.packages.find] where = ["src"] diff --git a/src/cotdata/vintage.py b/src/cotdata/vintage.py index f759146..c32b645 100644 --- a/src/cotdata/vintage.py +++ b/src/cotdata/vintage.py @@ -153,6 +153,20 @@ def read_snapshots() -> list[dict]: return _read_manifest()["snapshots"] +def update_snapshot(snapshot_id: str, **fields) -> bool: + """Patch fields (e.g. ``parse_status``, ``parse_error``) on a recorded snapshot. + Returns True if a matching snapshot was found and written.""" + m = _read_manifest() + hit = False + for rec in m["snapshots"]: + if rec.get("snapshot_id") == snapshot_id: + rec.update(fields) + hit = True + if hit: + _write_manifest(m) + return hit + + # ── Capture ───────────────────────────────────────────────────────────────── def _utcnow() -> dt.datetime: return dt.datetime.now(dt.timezone.utc).replace(microsecond=0) diff --git a/src/cotdata/vintage_cli.py b/src/cotdata/vintage_cli.py index a547a15..4cc3d3c 100644 --- a/src/cotdata/vintage_cli.py +++ b/src/cotdata/vintage_cli.py @@ -1,17 +1,20 @@ -"""`cotdata-vintage` — capture and query the COT vintage store. +"""`cotdata-vintage` / `cotdata-schedule` — capture and query the COT vintage store. -Subcommands (mirrors handoff §7, adapted to the repo's hyphenated entry-point style): +Subcommands (handoff §7, adapted to the repo's hyphenated entry-point style): - cotdata-vintage fetch [--year YYYY | --all] [--no-weekly] - -COMMIT 1 ships ``fetch`` (raw snapshot capture) only. ``ingest``, ``diff``, ``asof`` -and the ``cotdata-schedule`` commands land with the substantive subsystem in commit 2. + cotdata-vintage fetch [--year YYYY | --all] [--no-weekly] + cotdata-vintage ingest [--snapshot ID | --pending] + cotdata-vintage diff [--since DATE] [--market CODE] [--report-type T] + cotdata-vintage asof --as-of TIMESTAMP --report-date DATE [--market CODE] + cotdata-schedule sync + cotdata-schedule backfill """ import argparse from . import config +# ── vintage ───────────────────────────────────────────────────────────────── def _cmd_fetch(args) -> int: from . import vintage res = vintage.fetch(year=args.year, all_years=args.all, @@ -19,8 +22,73 @@ def _cmd_fetch(args) -> int: print(f"vintage fetch: {res['checks']} source(s) checked, " f"{res['new_files']} new raw file(s) retained.") for rec in res["records"]: - tag = rec.get("note") or "NEW" - print(f" {rec['report_type']:<13} {rec['source_kind']:<13} {tag}") + print(f" {rec['report_type']:<13} {rec['source_kind']:<13} {rec.get('note') or 'NEW'}") + return 0 + + +def _cmd_ingest(args) -> int: + from pathlib import Path + + from . import vintage, vintage_ingest + from .providers import cftc + + snaps = vintage.read_snapshots() + if args.snapshot: + snaps = [s for s in snaps if s.get("snapshot_id") == args.snapshot] + elif args.pending: + snaps = [s for s in snaps if s.get("parse_status") == "pending"] + # Only the Legacy annual zip is wired end-to-end this pass; disagg/TFF canonicalizers + # are a follow-on. Skip (don't fail) snapshots we can't yet parse. + total_obs = total_rev = 0 + for s in snaps: + if s.get("report_type") != "legacy" or s.get("source_kind") != "annual_zip": + continue + path = config.store_root() / s["local_path"] + try: + wide = cftc._parse_zip(Path(path)) + wide = wide.set_index(cftc.REPORT_DATE) + canonical = vintage_ingest.canonicalize_legacy(wide) + res = vintage_ingest.ingest_canonical(canonical, snapshot_id=s["snapshot_id"]) + vintage.update_snapshot(s["snapshot_id"], parse_status="ok", parse_error=None) + total_obs += res["observations"] + total_rev += res["revisions"] + print(f" {s['snapshot_id']}: +{res['observations']} obs, +{res['revisions']} rev") + except Exception as e: # noqa: BLE001 — record the failure, don't abort the batch + vintage.update_snapshot(s["snapshot_id"], parse_status="failed", parse_error=str(e)) + print(f" {s['snapshot_id']}: FAILED — {e}") + print(f"vintage ingest: {total_obs} new observation(s), {total_rev} revision(s).") + return 0 + + +def _cmd_diff(args) -> int: + from . import vintage_ingest + rev = vintage_ingest.read_revisions() + if rev.empty: + print("vintage diff: no revisions recorded yet.") + return 0 + if args.since: + import pandas as pd + rev = rev[rev["detected_at"] >= pd.Timestamp(args.since)] + if args.market: + rev = rev[rev["market_code"] == args.market] + if args.report_type: + rev = rev[rev["report_type"] == args.report_type] + print(f"vintage diff: {len(rev)} revision row(s).") + cols = ["report_date", "market_code", "category", "field", "old_value", + "new_value", "age_days"] + with __import__("pandas").option_context("display.max_rows", 50): + print(rev[cols].to_string(index=False)) + return 0 + + +def _cmd_asof(args) -> int: + from . import vintage_ingest + df = vintage_ingest.asof(args.as_of, report_date=args.report_date, market_code=args.market) + print(f"vintage asof {args.as_of}: {len(df)} row(s) known at that time.") + if not df.empty: + cols = ["report_date", "market_code", "category", "long_contracts", + "short_contracts", "open_interest", "observed_at", "snapshot_id"] + print(df[cols].to_string(index=False)) return 0 @@ -31,19 +99,66 @@ def main(argv=None) -> int: sub = p.add_subparsers(dest="cmd", required=True) f = sub.add_parser("fetch", help="Capture raw CFTC files into the immutable landing zone.") - f.add_argument("--year", type=int, default=None, - help="Report year to capture (default: current year).") - f.add_argument("--all", action="store_true", - help="Capture every year 1986→present (cold-start raw backfill).") - f.add_argument("--no-weekly", action="store_true", - help="Skip the current-week static (whose Last-Modified is the true " - "publication timestamp).") + f.add_argument("--year", type=int, default=None, help="Report year (default: current).") + f.add_argument("--all", action="store_true", help="Every year 1986→present (cold-start).") + f.add_argument("--no-weekly", action="store_true", help="Skip the current-week static.") f.set_defaults(func=_cmd_fetch) + ing = sub.add_parser("ingest", help="Parse retained raw files into change-only observations.") + g = ing.add_mutually_exclusive_group() + g.add_argument("--snapshot", default=None, help="Ingest one snapshot id.") + g.add_argument("--pending", action="store_true", help="Ingest all parse_status=pending.") + ing.set_defaults(func=_cmd_ingest) + + d = sub.add_parser("diff", help="Show recorded field-level revisions.") + d.add_argument("--since", default=None, help="Only revisions detected on/after DATE.") + d.add_argument("--market", default=None, help="Filter by CFTC market code.") + d.add_argument("--report-type", default=None, dest="report_type") + d.set_defaults(func=_cmd_diff) + + a = sub.add_parser("asof", help="Reconstruct the dataset as known at a past timestamp.") + a.add_argument("--as-of", required=True, dest="as_of", help="Point-in-time timestamp.") + a.add_argument("--report-date", default=None, dest="report_date") + a.add_argument("--market", default=None) + a.set_defaults(func=_cmd_asof) + args = p.parse_args(argv) config.store_root() # fail fast if COTDATA_STORE unset return args.func(args) +# ── schedule ──────────────────────────────────────────────────────────────── +def _cmd_sched_sync(args) -> int: + from . import vintage_schedule + res = vintage_schedule.sync() + print(f"schedule sync: {res['announcements']} announcement row(s) scraped.") + return 0 + + +def _cmd_sched_backfill(args) -> int: + from . import vintage_schedule + counts = vintage_schedule.backfill() + total = sum(counts.values()) + print(f"schedule backfill: resolved release_date for {total} observation row(s):") + for src in vintage_schedule.PRECEDENCE: + if counts.get(src): + print(f" {src:<10} {counts[src]}") + return 0 + + +def main_schedule(argv=None) -> int: + p = argparse.ArgumentParser( + prog="cotdata-schedule", + description="Sync the CFTC release schedule / announcements and backfill release_date.") + sub = p.add_subparsers(dest="cmd", required=True) + s = sub.add_parser("sync", help="Scrape Special Announcements into the store.") + s.set_defaults(func=_cmd_sched_sync) + b = sub.add_parser("backfill", help="Resolve release_date/source across all observations.") + b.set_defaults(func=_cmd_sched_backfill) + args = p.parse_args(argv) + config.store_root() + return args.func(args) + + if __name__ == "__main__": raise SystemExit(main()) diff --git a/src/cotdata/vintage_ingest.py b/src/cotdata/vintage_ingest.py new file mode 100644 index 0000000..13b23b8 --- /dev/null +++ b/src/cotdata/vintage_ingest.py @@ -0,0 +1,312 @@ +"""COT vintage ingest — change-only bitemporal observations + field-level revisions. + +Commit 2 of the vintage subsystem. Sits on top of the raw snapshots captured by +``vintage.py``: parse → canonical long schema → validate → change-only write → emit +revisions. pandas/pyarrow throughout (no database — crucible-stack ADR-0008). + +Canonical natural key: ``(report_date, market_code, report_type, combined, category)``. +A point-in-time read is "greatest ``observed_at`` <= t per natural key" — no valid_to +column. Change-only writes mean storage grows with revisions, not with time. + +See docs/design/cot_vintage.md. +""" +from __future__ import annotations + +import datetime as dt +import hashlib +from pathlib import Path + +import pandas as pd + +from . import vintage + +# ── Schema ────────────────────────────────────────────────────────────────── +NATURAL_KEY = ["report_date", "market_code", "report_type", "combined", "category"] + +# Fields that constitute the *value* of an observation. row_sha256 is computed over +# these only: not over provenance (observed_at/snapshot_id), not over release_date +# (resolved separately, so a release-date backfill must NOT read as a data revision), +# not over market_name (descriptive). +VALUE_FIELDS = ["long_contracts", "short_contracts", "spread_contracts", + "trader_count_long", "trader_count_short", "open_interest", + "cr4_net_long", "cr4_net_short", "cr8_net_long", "cr8_net_short"] + +PROVENANCE_FIELDS = ["release_date", "release_date_source", "market_name", + "observed_at", "snapshot_id", "row_sha256", "is_tombstone"] + +ALL_COLUMNS = NATURAL_KEY + ["market_name"] + VALUE_FIELDS + [ + "release_date", "release_date_source", "observed_at", "snapshot_id", + "row_sha256", "is_tombstone"] + +# Controlled vocabulary per report type (validation §5). "nonreportable" is common +# to all three; the rest come from each report's reporting categories. +CATEGORIES = { + "legacy": {"commercial", "noncommercial", "nonreportable"}, + "disaggregated": {"producer_merchant", "swap", "managed_money", + "other_reportable", "nonreportable"}, + "tff": {"dealer", "asset_manager", "leveraged", "other_reportable", "nonreportable"}, +} + + +# ── Paths ─────────────────────────────────────────────────────────────────── +def _obs_dir() -> Path: + return vintage.vintage_root() / "observations" + + +def _rev_dir() -> Path: + return vintage.vintage_root() / "revisions" + + +def _obs_path(report_year: int) -> Path: + return _obs_dir() / f"report_year={report_year}" / "observations.parquet" + + +def _rev_path(detected_year: int) -> Path: + return _rev_dir() / f"detected_year={detected_year}" / "revisions.parquet" + + +# ── Hashing ───────────────────────────────────────────────────────────────── +def _naive_utc(ts) -> pd.Timestamp: + """Normalise a timestamp to tz-naive UTC — the repo stores tz-naive datetimes + throughout (Report_Date, price index), so vintage timestamps match that convention + and never trip tz-aware/tz-naive arithmetic.""" + t = pd.Timestamp(ts) + if t.tzinfo is not None: + t = t.tz_convert("UTC").tz_localize(None) + return t + + +def _norm(v) -> str: + if v is None or (isinstance(v, float) and pd.isna(v)) or v is pd.NA: + return "" + if isinstance(v, float) and float(v).is_integer(): + return str(int(v)) + return str(v) + + +def row_sha256(row: dict) -> str: + payload = "|".join(f"{f}={_norm(row.get(f))}" for f in VALUE_FIELDS) + return hashlib.sha256(payload.encode()).hexdigest() + + +# ── Canonicalisation from the existing wide parsers ───────────────────────── +def canonicalize_legacy(wide: pd.DataFrame, *, combined: bool = False) -> pd.DataFrame: + """Melt a Legacy wide frame (the shape ``providers/cftc.py`` emits, Report_Date + index) into canonical long rows: three category rows per (report_date, market).""" + rows = [] + for report_date, r in wide.iterrows(): + market_code = str(r["CFTC_Contract_Market_Code"]) + market_name = r.get("Market_and_Exchange_Names") + oi = r.get("Open_Interest_All") + cats = { + "commercial": ("Comm_Positions_Long_All", "Comm_Positions_Short_All", + "Traders_Comm_Long_All", "Traders_Comm_Short_All"), + "noncommercial": ("NonComm_Positions_Long_All", "NonComm_Positions_Short_All", + "Traders_NonComm_Long_All", "Traders_NonComm_Short_All"), + "nonreportable": ("NonRept_Positions_Long_All", "NonRept_Positions_Short_All", + None, None), + } + for category, (lc, sc, tl, ts) in cats.items(): + rows.append({ + "report_date": pd.Timestamp(report_date).normalize(), + "market_code": market_code, + "market_name": market_name, + "report_type": "legacy", + "combined": combined, + "category": category, + "long_contracts": r.get(lc), + "short_contracts": r.get(sc), + "spread_contracts": pd.NA, + "trader_count_long": r.get(tl) if tl else pd.NA, + "trader_count_short": r.get(ts) if ts else pd.NA, + "open_interest": oi, + "cr4_net_long": pd.NA, "cr4_net_short": pd.NA, + "cr8_net_long": pd.NA, "cr8_net_short": pd.NA, + }) + return pd.DataFrame(rows) + + +# ── Validation (§5) ───────────────────────────────────────────────────────── +class ValidationError(ValueError): + pass + + +def validate(canonical: pd.DataFrame) -> list[str]: + """Fail loudly (raise) on structural problems; return a list of soft warnings. + + Raises rather than partially ingesting on: missing natural-key columns, null + natural-key values, or categories outside the controlled vocabulary. Warns (does + not fail) on ``long+short+spread > open_interest`` — definitional edge cases exist. + """ + missing = [c for c in NATURAL_KEY if c not in canonical.columns] + if missing: + raise ValidationError(f"missing natural-key columns: {missing}") + if canonical.empty: + raise ValidationError("empty canonical frame — refusing to ingest nothing") + for c in NATURAL_KEY: + if canonical[c].isna().any(): + raise ValidationError(f"null values in natural-key column {c!r}") + for rt, grp in canonical.groupby("report_type"): + vocab = CATEGORIES.get(rt) + if vocab is None: + raise ValidationError(f"unknown report_type {rt!r}") + bad = set(grp["category"]) - vocab + if bad: + raise ValidationError(f"categories {bad} outside vocabulary for {rt!r}") + + warnings = [] + for _, r in canonical.iterrows(): + oi = r.get("open_interest") + parts = [r.get("long_contracts"), r.get("short_contracts"), r.get("spread_contracts")] + s = sum(p for p in parts if p is not None and not pd.isna(p)) + if oi is not None and not pd.isna(oi) and s > oi: + warnings.append(f"{r['report_date'].date()} {r['market_code']} " + f"{r['category']}: long+short+spread ({s}) > OI ({oi})") + return warnings + + +# ── Observation store I/O ─────────────────────────────────────────────────── +def read_observations(report_years=None) -> pd.DataFrame: + d = _obs_dir() + if not d.exists(): + return pd.DataFrame(columns=ALL_COLUMNS) + parts = sorted(d.glob("report_year=*/observations.parquet")) + if report_years is not None: + want = {int(y) for y in report_years} + parts = [p for p in parts if int(p.parent.name.split("=")[1]) in want] + if not parts: + return pd.DataFrame(columns=ALL_COLUMNS) + return pd.concat([pd.read_parquet(p) for p in parts], ignore_index=True) + + +def _latest_by_key(obs: pd.DataFrame) -> pd.DataFrame: + """Most recent (max observed_at) observation per natural key.""" + if obs.empty: + return obs + idx = obs.groupby(NATURAL_KEY, dropna=False)["observed_at"].idxmax() + return obs.loc[idx] + + +def _append_parquet(path: Path, new_rows: pd.DataFrame) -> None: + path.parent.mkdir(parents=True, exist_ok=True) + if path.exists(): + existing = pd.read_parquet(path) + combined = pd.concat([existing, new_rows], ignore_index=True) + else: + combined = new_rows + tmp = path.with_suffix(".parquet.tmp") + combined.to_parquet(tmp) + tmp.replace(path) + + +# ── Ingest (change-only) ──────────────────────────────────────────────────── +def ingest_canonical(canonical: pd.DataFrame, *, snapshot_id: str, + observed_at: dt.datetime | None = None, + validate_input: bool = True) -> dict: + """Change-only write of a canonical frame + field-level revision emission. + + Idempotent: re-ingesting an identical frame writes zero observations and zero + revisions, because every row's ``row_sha256`` matches the latest stored value for + its natural key. Returns ``{"observations": n, "revisions": n, "warnings": [...]}``. + """ + warnings = validate(canonical) if validate_input else [] + if observed_at is None: + observed_at = dt.datetime.now(dt.timezone.utc).replace(microsecond=0) + observed_ts = _naive_utc(observed_at) + + prior = _latest_by_key(read_observations()) + prior_by_key = {} + for _, r in prior.iterrows(): + prior_by_key[tuple(r[k] for k in NATURAL_KEY)] = r + + new_obs: dict[int, list] = {} + revisions: list[dict] = [] + detected_at = observed_ts + + for _, row in canonical.iterrows(): + rec = row.to_dict() + rec["row_sha256"] = row_sha256(rec) + key = tuple(rec[k] for k in NATURAL_KEY) + prev = prior_by_key.get(key) + if prev is not None and prev["row_sha256"] == rec["row_sha256"]: + continue # unchanged value → change-only skip + + rec["observed_at"] = observed_ts + rec["snapshot_id"] = snapshot_id + rec.setdefault("release_date", pd.NaT) + rec.setdefault("release_date_source", "unknown") + rec.setdefault("is_tombstone", False) + ry = pd.Timestamp(rec["report_date"]).year + new_obs.setdefault(ry, []).append(rec) + + if prev is not None: + report_date = pd.Timestamp(rec["report_date"]) + age_days = int((detected_at.normalize() - report_date.normalize()).days) + for f in VALUE_FIELDS: + old, new = prev.get(f), rec.get(f) + if _norm(old) == _norm(new): + continue + delta = pct = None + try: + if not pd.isna(old) and not pd.isna(new): + delta = float(new) - float(old) + pct = (delta / float(old)) if float(old) != 0 else None + except (TypeError, ValueError): + pass + revisions.append({ + **{k: rec[k] for k in NATURAL_KEY}, + "field": f, "old_value": _norm(old), "new_value": _norm(new), + "delta": delta, "pct_delta": pct, + "old_snapshot_id": prev.get("snapshot_id"), + "new_snapshot_id": snapshot_id, + "detected_at": detected_at, "age_days": age_days, + "revision_id": hashlib.sha256( + f"{key}|{f}|{prev.get('snapshot_id')}|{snapshot_id}".encode() + ).hexdigest()[:16], + }) + + n_obs = 0 + for ry, recs in new_obs.items(): + frame = pd.DataFrame(recs).reindex(columns=ALL_COLUMNS) + _append_parquet(_obs_path(ry), frame) + n_obs += len(recs) + + if revisions: + by_year: dict[int, list] = {} + for rv in revisions: + by_year.setdefault(pd.Timestamp(rv["detected_at"]).year, []).append(rv) + for dy, recs in by_year.items(): + _append_parquet(_rev_path(dy), pd.DataFrame(recs)) + + return {"observations": n_obs, "revisions": len(revisions), "warnings": warnings} + + +# ── Point-in-time read (§4 / acceptance §4) ───────────────────────────────── +def asof(t, *, report_date=None, market_code=None, report_type=None) -> pd.DataFrame: + """The dataset as known at timestamp ``t``: for each natural key, the row with the + greatest ``observed_at`` <= t. Rows first observed after ``t`` are absent.""" + obs = read_observations() + if obs.empty: + return obs + t = _naive_utc(t) + obs = obs[obs["observed_at"] <= t] + if report_date is not None: + obs = obs[obs["report_date"] == pd.Timestamp(report_date).normalize()] + if market_code is not None: + obs = obs[obs["market_code"] == str(market_code)] + if report_type is not None: + obs = obs[obs["report_type"] == report_type] + return _latest_by_key(obs).reset_index(drop=True) + + +def read_revisions(detected_years=None) -> pd.DataFrame: + d = _rev_dir() + if not d.exists(): + return pd.DataFrame() + parts = sorted(d.glob("detected_year=*/revisions.parquet")) + if detected_years is not None: + want = {int(y) for y in detected_years} + parts = [p for p in parts if int(p.parent.name.split("=")[1]) in want] + if not parts: + return pd.DataFrame() + return pd.concat([pd.read_parquet(p) for p in parts], ignore_index=True) diff --git a/src/cotdata/vintage_schedule.py b/src/cotdata/vintage_schedule.py new file mode 100644 index 0000000..80c9530 --- /dev/null +++ b/src/cotdata/vintage_schedule.py @@ -0,0 +1,210 @@ +"""Release-schedule + Special-Announcements ingestion, and the release-date backfill. + +The annual CFTC zips carry only ``report_date``. A release date that embeds no lookahead +must be *resolved* and its provenance recorded (§4.6), in precedence order: + + observed > announced > scheduled > derived > unknown + +- ``observed`` is stamped at CAPTURE time from the weekly-static Last-Modified (a true + publication timestamp — spike 2026-07-30). Not manufactured here. +- ``announced`` comes from the Special Announcements page (holiday shifts, and the + Oct–Dec 2025 appropriations-lapse backlog where release trails report_date by weeks). +- ``scheduled`` comes from the CFTC published release schedule (normal weeks). +- ``derived`` is ``report_date + 3d`` (weekend-adjusted) — the fallback that fails on + exactly the weeks that matter, which is why the source flag exists: downstream code + must be able to exclude ``derived`` rows from strict PIT evaluation. + +Scraping is best-effort (store raw text always). The resolution logic is a pure function +so it is fully testable offline. See docs/design/cot_vintage.md. +""" +from __future__ import annotations + +import datetime as dt +from pathlib import Path + +import pandas as pd + +from . import vintage +from . import vintage_ingest as vi + +PRECEDENCE = ("observed", "announced", "scheduled", "derived", "unknown") + + +# ── Paths ─────────────────────────────────────────────────────────────────── +def release_schedule_path() -> Path: + return vintage.vintage_root() / "release_schedule.parquet" + + +def announcements_path() -> Path: + return vintage.vintage_root() / "announcements.parquet" + + +# ── Derivation + resolution (pure) ────────────────────────────────────────── +def derive_release_date(report_date) -> dt.date: + """Fallback: report_date + 3 days, pushed off a weekend. Matches the normal + Tuesday→Friday COT cadence. Deliberately holiday-naive — that is what ``announced`` + and ``scheduled`` are for.""" + d = (pd.Timestamp(report_date).normalize() + pd.Timedelta(days=3)).date() + while d.weekday() >= 5: # Sat/Sun → next Monday + d += dt.timedelta(days=1) + return d + + +def resolve_release_date(report_date, *, observed=None, announced=None, + scheduled=None) -> tuple[dt.date | None, str]: + """Return ``(release_date, source)`` by the §4.6 precedence. ``observed`` / + ``announced`` / ``scheduled`` are optional resolved dates for this report_date.""" + for value, source in ((observed, "observed"), (announced, "announced"), + (scheduled, "scheduled")): + if value is not None and not pd.isna(value): + return pd.Timestamp(value).date(), source + return derive_release_date(report_date), "derived" + + +# ── Store I/O ─────────────────────────────────────────────────────────────── +def _write(path: Path, df: pd.DataFrame) -> None: + path.parent.mkdir(parents=True, exist_ok=True) + tmp = path.with_suffix(".parquet.tmp") + df.to_parquet(tmp) + tmp.replace(path) + + +def read_release_schedule() -> pd.DataFrame: + p = release_schedule_path() + return pd.read_parquet(p) if p.exists() else pd.DataFrame( + columns=["report_date", "release_date", "source", "note", "ingested_at"]) + + +def read_announcements() -> pd.DataFrame: + p = announcements_path() + return pd.read_parquet(p) if p.exists() else pd.DataFrame( + columns=["announcement_date", "raw_text", "affected_report_types", + "affected_markets", "affected_date_from", "affected_date_to", + "url", "scraped_at"]) + + +def write_release_schedule(df: pd.DataFrame) -> None: + _write(release_schedule_path(), df) + + +def write_announcements(df: pd.DataFrame) -> None: + _write(announcements_path(), df) + + +# ── Backfill ──────────────────────────────────────────────────────────────── +def _schedule_map(schedule: pd.DataFrame) -> dict: + """report_date -> (release_date, source) from the schedule table. `announced` rows + win over `scheduled` rows for the same report_date.""" + out: dict = {} + for _, r in schedule.iterrows(): + rd = pd.Timestamp(r["report_date"]).normalize() + src = r.get("source", "scheduled") + prev = out.get(rd) + if prev is None or (prev[1] == "scheduled" and src == "announced"): + out[rd] = (pd.Timestamp(r["release_date"]).date(), src) + return out + + +def backfill(*, schedule: pd.DataFrame | None = None, + observed_window_days: int = 4) -> dict: + """Resolve ``release_date`` / ``release_date_source`` across all stored observations + and write them back in place. Returns coverage counts keyed by source. + + ``observed`` is honoured only when the recorded ``observed_at`` is within + ``observed_window_days`` of ``report_date`` — i.e. the row was captured live, not + bulk-ingested long afterwards (a bulk ingest's observed_at is not a release date). + """ + if schedule is None: + schedule = read_release_schedule() + smap = _schedule_map(schedule) + + obs_dir = vi._obs_dir() + counts = dict.fromkeys(PRECEDENCE, 0) + if not obs_dir.exists(): + return counts + + for part in sorted(obs_dir.glob("report_year=*/observations.parquet")): + df = pd.read_parquet(part) + if df.empty: + continue + rel_dates, rel_srcs = [], [] + for _, r in df.iterrows(): + rd = pd.Timestamp(r["report_date"]).normalize() + observed = None + oa = r.get("observed_at") + if oa is not None and not pd.isna(oa): + oa = vi._naive_utc(oa) + if abs((oa.normalize() - rd).days) <= observed_window_days: + observed = oa + sched = smap.get(rd) + announced = sched[0] if sched and sched[1] == "announced" else None + scheduled = sched[0] if sched and sched[1] == "scheduled" else None + rdate, src = resolve_release_date( + rd, observed=observed, announced=announced, scheduled=scheduled) + rel_dates.append(pd.Timestamp(rdate) if rdate is not None else pd.NaT) + rel_srcs.append(src) + counts[src] += 1 + df["release_date"] = rel_dates + df["release_date_source"] = rel_srcs + tmp = part.with_suffix(".parquet.tmp") + df.to_parquet(tmp) + tmp.replace(part) + return counts + + +# ── Best-effort scrape (network; injectable for offline use) ──────────────── +RELEASE_SCHEDULE_URL = ("https://www.cftc.gov/MarketReports/CommitmentsofTraders/" + "ReleaseSchedule/index.htm") +ANNOUNCEMENTS_URL = ("https://www.cftc.gov/MarketReports/CommitmentsofTraders/" + "HistoricalSpecialAnnouncements/index.htm") + + +def _fetch_html(url: str) -> str: + import requests + r = requests.get(url, headers={"User-Agent": vintage._UA}, timeout=120) + r.raise_for_status() + return r.text + + +def sync(*, fetch_html=_fetch_html, now=None) -> dict: + """Scrape the announcements page into ``announcements.parquet`` (raw text always + retained). Structured extraction is best-effort; attribution only needs the text + and date. Returns ``{"announcements": n}``. The release schedule is seeded by the + caller (small, and its HTML layout changes) rather than scraped heuristically here. + """ + now = now or dt.datetime.now(dt.timezone.utc).replace(microsecond=0) + html = fetch_html(ANNOUNCEMENTS_URL) + rows = _parse_announcements(html, url=ANNOUNCEMENTS_URL, scraped_at=now) + if rows: + existing = read_announcements() + merged = pd.concat([existing, pd.DataFrame(rows)], ignore_index=True) + merged = merged.drop_duplicates(subset=["announcement_date", "raw_text"]) + write_announcements(merged) + return {"announcements": len(rows)} + + +def _parse_announcements(html: str, *, url: str, scraped_at) -> list[dict]: + """Best-effort: pull list items / paragraphs mentioning a date. Never raises on + unrecognised markup — a layout change must degrade to fewer rows, not a crash.""" + import re + rows = [] + for m in re.finditer(r"]*>(.*?)", html, flags=re.S | re.I): + text = re.sub(r"<[^>]+>", " ", m.group(1)) + text = re.sub(r"\s+", " ", text).strip() + if not text: + continue + date = None + dm = re.search(r"([A-Z][a-z]+ \d{1,2}, \d{4})", text) + if dm: + try: + date = pd.Timestamp(dm.group(1)).date() + except (ValueError, TypeError): + date = None + rows.append({ + "announcement_date": pd.Timestamp(date) if date else pd.NaT, + "raw_text": text, "affected_report_types": None, + "affected_markets": None, "affected_date_from": pd.NaT, + "affected_date_to": pd.NaT, "url": url, + "scraped_at": pd.Timestamp(scraped_at), + }) + return rows diff --git a/tests/test_vintage_ingest.py b/tests/test_vintage_ingest.py new file mode 100644 index 0000000..4a9baf5 --- /dev/null +++ b/tests/test_vintage_ingest.py @@ -0,0 +1,132 @@ +"""Vintage ingest (commit 2): change-only observations, field-level revisions, PIT asof. + +Offline — canonical frames are built directly, matching the repo's synthetic-fixture +idiom. Covers handoff §8: idempotent ingest, byte-change/data-same, single-field +revision, PIT query, revision depth, holiday week, and validation failure. +""" +import datetime as dt + +import pandas as pd +import pytest + + +@pytest.fixture() +def store_env(tmp_path, monkeypatch): + monkeypatch.setenv("COTDATA_STORE", str(tmp_path)) + return tmp_path + + +def _wide(report_date, *, code="088691", comm_long=200000, comm_short=250000, + oi=500000): + """One-market, one-week Legacy wide frame (the shape providers/cftc.py emits).""" + idx = pd.to_datetime([report_date]) + idx.name = "Report_Date_as_MM_DD_YYYY" + return pd.DataFrame({ + "Market_and_Exchange_Names": ["GOLD"], + "CFTC_Contract_Market_Code": [code], + "Open_Interest_All": [oi], + "Comm_Positions_Long_All": [comm_long], + "Comm_Positions_Short_All": [comm_short], + "NonComm_Positions_Long_All": [150000], + "NonComm_Positions_Short_All": [90000], + "NonRept_Positions_Long_All": [40000], + "NonRept_Positions_Short_All": [30000], + "Traders_Tot_All": [280], + "Traders_Comm_Long_All": [50], "Traders_Comm_Short_All": [55], + "Traders_NonComm_Long_All": [60], "Traders_NonComm_Short_All": [45], + }, index=idx) + + +def _canon(report_date, **kw): + from cotdata.vintage_ingest import canonicalize_legacy + return canonicalize_legacy(_wide(report_date, **kw)) + + +def test_idempotent_ingest(store_env): + from cotdata import vintage_ingest as vi + c = _canon("2026-07-21") + r1 = vi.ingest_canonical(c, snapshot_id="s1") + assert r1["observations"] == 3 and r1["revisions"] == 0 # 3 categories, first sighting + r2 = vi.ingest_canonical(c, snapshot_id="s1") + assert r2["observations"] == 0 and r2["revisions"] == 0 # nothing changed + + +def test_byte_change_data_same_is_noop(store_env): + """A regenerated archive (new snapshot id) with identical values must produce no + new observations and no revisions.""" + from cotdata import vintage_ingest as vi + c = _canon("2026-07-21") + vi.ingest_canonical(c, snapshot_id="s1") + r = vi.ingest_canonical(c, snapshot_id="s2-regenerated") + assert r["observations"] == 0 and r["revisions"] == 0 + + +def test_single_field_revision(store_env): + from cotdata import vintage_ingest as vi + vi.ingest_canonical(_canon("2026-07-21", comm_short=250000), snapshot_id="s1") + r = vi.ingest_canonical(_canon("2026-07-21", comm_short=251000), snapshot_id="s2") + # only the commercial category row changed → 1 new obs, 1 revision + assert r["observations"] == 1 and r["revisions"] == 1 + rev = vi.read_revisions() + assert len(rev) == 1 + row = rev.iloc[0] + assert row["field"] == "short_contracts" + assert row["old_value"] == "250000" and row["new_value"] == "251000" + assert row["category"] == "commercial" + assert row["delta"] == 1000.0 + + +def test_pit_query_returns_pre_revision_value(store_env): + from cotdata import vintage_ingest as vi + t1 = dt.datetime(2026, 7, 24, 16, 0, tzinfo=dt.timezone.utc) + t2 = dt.datetime(2026, 7, 31, 16, 0, tzinfo=dt.timezone.utc) + vi.ingest_canonical(_canon("2026-07-21", comm_short=250000), snapshot_id="s1", observed_at=t1) + vi.ingest_canonical(_canon("2026-07-21", comm_short=251000), snapshot_id="s2", observed_at=t2) + + between = dt.datetime(2026, 7, 28, tzinfo=dt.timezone.utc) + df = vi.asof(between, report_date="2026-07-21", market_code="088691") + comm = df[df["category"] == "commercial"].iloc[0] + assert comm["short_contracts"] == 250000 # pre-revision value + + after = dt.datetime(2026, 8, 1, tzinfo=dt.timezone.utc) + df2 = vi.asof(after, report_date="2026-07-21", market_code="088691") + assert df2[df2["category"] == "commercial"].iloc[0]["short_contracts"] == 251000 + + +def test_revision_depth_age_days(store_env): + """A restatement of a year-old report gets the correct revision depth.""" + from cotdata import vintage_ingest as vi + t1 = dt.datetime(2025, 7, 18, tzinfo=dt.timezone.utc) + t2 = dt.datetime(2026, 7, 30, tzinfo=dt.timezone.utc) + vi.ingest_canonical(_canon("2025-07-15", comm_short=250000), snapshot_id="s1", observed_at=t1) + vi.ingest_canonical(_canon("2025-07-15", comm_short=260000), snapshot_id="s2", observed_at=t2) + rev = vi.read_revisions() + expected = (pd.Timestamp("2026-07-30") - pd.Timestamp("2025-07-15")).days + assert int(rev.iloc[0]["age_days"]) == expected == 380 + + +def test_holiday_week_not_normalised_to_tuesday(store_env): + """A Monday as-of date is stored as Monday, not rounded to Tuesday.""" + from cotdata import vintage_ingest as vi + vi.ingest_canonical(_canon("2025-12-29"), snapshot_id="s1") # 2025-12-29 is a Monday + obs = vi.read_observations() + stored = pd.Timestamp(obs.iloc[0]["report_date"]) + assert stored.weekday() == 0 and stored.date() == dt.date(2025, 12, 29) + + +def test_validation_failure_raises_before_writing(store_env): + from cotdata import vintage_ingest as vi + bad = _canon("2026-07-21") + bad.loc[0, "category"] = "bogus" # outside the legacy controlled vocabulary + with pytest.raises(vi.ValidationError): + vi.ingest_canonical(bad, snapshot_id="s1") + assert vi.read_observations().empty # nothing partially written + + +def test_oi_over_sum_warns_not_raises(store_env): + from cotdata import vintage_ingest as vi + # commercial long+short (200000+250000) already < OI; force a breach on OI instead + c = _canon("2026-07-21", comm_long=400000, comm_short=400000, oi=100000) + res = vi.ingest_canonical(c, snapshot_id="s1") + assert res["warnings"] # soft warning emitted + assert res["observations"] == 3 # still ingested diff --git a/tests/test_vintage_schedule.py b/tests/test_vintage_schedule.py new file mode 100644 index 0000000..5598280 --- /dev/null +++ b/tests/test_vintage_schedule.py @@ -0,0 +1,88 @@ +"""Release-date resolution + backfill (§4.6, §8): precedence, derivation, and the +Oct–Dec 2025 backlog week resolving to its true announced release date. +""" +import datetime as dt + +import pandas as pd +import pytest + + +@pytest.fixture() +def store_env(tmp_path, monkeypatch): + monkeypatch.setenv("COTDATA_STORE", str(tmp_path)) + return tmp_path + + +def test_release_date_precedence(): + from cotdata.vintage_schedule import resolve_release_date + rd = "2026-07-21" + obs, ann, sched = "2026-07-24", "2026-07-25", "2026-07-26" + assert resolve_release_date(rd, observed=obs, announced=ann, scheduled=sched) \ + == (dt.date(2026, 7, 24), "observed") + assert resolve_release_date(rd, announced=ann, scheduled=sched) \ + == (dt.date(2026, 7, 25), "announced") + assert resolve_release_date(rd, scheduled=sched) == (dt.date(2026, 7, 26), "scheduled") + # nothing resolved → derived (report_date + 3d, weekend-adjusted) + val, src = resolve_release_date(rd) + assert src == "derived" and val == dt.date(2026, 7, 24) + + +def test_derive_pushes_off_weekend(): + from cotdata.vintage_schedule import derive_release_date + # 2026-07-16 (Thu) + 3 = Sun 07-19 → pushed to Mon 07-20 + assert derive_release_date("2026-07-16") == dt.date(2026, 7, 20) + + +def _ingest_one(report_date, observed_at): + from cotdata import vintage_ingest as vi + idx = pd.to_datetime([report_date]) + idx.name = "Report_Date_as_MM_DD_YYYY" + wide = pd.DataFrame({ + "Market_and_Exchange_Names": ["GOLD"], "CFTC_Contract_Market_Code": ["088691"], + "Open_Interest_All": [500000], + "Comm_Positions_Long_All": [200000], "Comm_Positions_Short_All": [250000], + "NonComm_Positions_Long_All": [150000], "NonComm_Positions_Short_All": [90000], + "NonRept_Positions_Long_All": [40000], "NonRept_Positions_Short_All": [30000], + "Traders_Comm_Long_All": [50], "Traders_Comm_Short_All": [55], + "Traders_NonComm_Long_All": [60], "Traders_NonComm_Short_All": [45], + }, index=idx) + vi.ingest_canonical(vi.canonicalize_legacy(wide), snapshot_id="s1", observed_at=observed_at) + + +def test_backlog_week_resolves_to_announced(store_env): + """A report_date in the Oct–Nov 2025 appropriations-lapse backlog resolves to its + true (late) announced release date, flagged `announced` — not `derived`.""" + from cotdata import vintage_schedule as vs + # bulk-ingested long after the fact, so observed_at is NOT a release date + _ingest_one("2025-10-07", observed_at=dt.datetime(2026, 7, 30, tzinfo=dt.timezone.utc)) + + schedule = pd.DataFrame([{ + "report_date": pd.Timestamp("2025-10-07"), + "release_date": pd.Timestamp("2025-11-14"), # republished weeks late + "source": "announced", "note": "appropriations-lapse backlog", + "ingested_at": pd.Timestamp.now("UTC"), + }]) + counts = vs.backfill(schedule=schedule) + assert counts["announced"] == 3 and counts["derived"] == 0 # 3 category rows + + from cotdata import vintage_ingest as vi + row = vi.read_observations().iloc[0] + assert pd.Timestamp(row["release_date"]).date() == dt.date(2025, 11, 14) + assert row["release_date_source"] == "announced" + + +def test_backfill_derives_when_no_schedule(store_env): + from cotdata import vintage_schedule as vs + _ingest_one("2025-10-07", observed_at=dt.datetime(2026, 7, 30, tzinfo=dt.timezone.utc)) + counts = vs.backfill(schedule=pd.DataFrame( + columns=["report_date", "release_date", "source", "note", "ingested_at"])) + assert counts["derived"] == 3 and counts["announced"] == 0 + + +def test_announcement_parse_is_best_effort(): + from cotdata.vintage_schedule import _parse_announcements + html = "
  • January 5, 2026: revised gold report
" + rows = _parse_announcements(html, url="http://x", scraped_at=pd.Timestamp.now("UTC")) + assert len(rows) == 1 # empty
  • skipped + assert rows[0]["raw_text"].startswith("January 5, 2026") + assert pd.Timestamp(rows[0]["announcement_date"]).date() == dt.date(2026, 1, 5) From 5b7dbde127ae01662969162b20eab19a3271d7c3 Mon Sep 17 00:00:00 2001 From: Matt Spinola Date: Thu, 30 Jul 2026 14:14:59 -0400 Subject: [PATCH 03/16] fix(vintage): deterministic asof tie-break; backfill no-downgrade test Review follow-ups. - asof / change-detection: _latest_by_key ordered by (observed_at, snapshot_id) with a stable sort, take last per key. Previously idxmax() returned the FIRST occurrence of the max observed_at, which is file/append-order dependent and undefined when two snapshots share a timestamp. Capture snapshot_ids lead with a compact retrieved_at, so the lexicographic tie-break also tracks retrieval time. - test: two snapshots sharing observed_at resolve to the greater snapshot_id regardless of append order (append order set opposite to the winner). - test: schedule backfill is idempotent and does not downgrade an `observed` release date to `scheduled` on re-run (precedence enforced on every write, observed first). Co-Authored-By: Claude Opus 4.8 --- src/cotdata/vintage_ingest.py | 11 +++++++++-- tests/test_vintage_ingest.py | 20 ++++++++++++++++++++ tests/test_vintage_schedule.py | 29 +++++++++++++++++++++++++++++ 3 files changed, 58 insertions(+), 2 deletions(-) diff --git a/src/cotdata/vintage_ingest.py b/src/cotdata/vintage_ingest.py index 13b23b8..43d945e 100644 --- a/src/cotdata/vintage_ingest.py +++ b/src/cotdata/vintage_ingest.py @@ -180,10 +180,17 @@ def read_observations(report_years=None) -> pd.DataFrame: def _latest_by_key(obs: pd.DataFrame) -> pd.DataFrame: - """Most recent (max observed_at) observation per natural key.""" + """Most recent observation per natural key, with a DETERMINISTIC tie-break. + + Ordering is (observed_at, snapshot_id) ascending → take the last per key. When two + snapshots share an ``observed_at`` (same-second ingests), the lexicographically + greater ``snapshot_id`` wins rather than whichever row happened to land first in + file/append order. Capture snapshot_ids lead with a compact retrieved_at timestamp, + so that ordering also tracks retrieval time. A stable sort keeps it reproducible.""" if obs.empty: return obs - idx = obs.groupby(NATURAL_KEY, dropna=False)["observed_at"].idxmax() + ordered = obs.sort_values(["observed_at", "snapshot_id"], kind="mergesort") + idx = ordered.groupby(NATURAL_KEY, dropna=False, sort=False).tail(1).index return obs.loc[idx] diff --git a/tests/test_vintage_ingest.py b/tests/test_vintage_ingest.py index 4a9baf5..c47f981 100644 --- a/tests/test_vintage_ingest.py +++ b/tests/test_vintage_ingest.py @@ -123,6 +123,26 @@ def test_validation_failure_raises_before_writing(store_env): assert vi.read_observations().empty # nothing partially written +def test_asof_tiebreak_is_deterministic(store_env): + """Two snapshots sharing an observed_at must resolve deterministically: the + lexicographically greater snapshot_id wins, NOT file/append order. Append order is + set opposite to the winner so a naive first-occurrence pick would return 'a-early'.""" + from cotdata import vintage_ingest as vi + t = dt.datetime(2026, 7, 24, 16, 0, tzinfo=dt.timezone.utc) + # appended first (lower index): 'a-early' = 111111 + vi.ingest_canonical(_canon("2026-07-21", comm_short=111111), snapshot_id="a-early", observed_at=t) + # appended second (higher index): 'z-late' = 222222, same observed_at → a genuine tie + vi.ingest_canonical(_canon("2026-07-21", comm_short=222222), snapshot_id="z-late", observed_at=t) + + comm = vi.asof(t, report_date="2026-07-21", market_code="088691") + comm = comm[comm["category"] == "commercial"].iloc[0] + # naive first-occurrence would pick a-early (111111); deterministic picks z-late + assert comm["snapshot_id"] == "z-late" and comm["short_contracts"] == 222222 + # stable across repeated calls + again = vi.asof(t, report_date="2026-07-21", market_code="088691") + assert again[again["category"] == "commercial"].iloc[0]["snapshot_id"] == "z-late" + + def test_oi_over_sum_warns_not_raises(store_env): from cotdata import vintage_ingest as vi # commercial long+short (200000+250000) already < OI; force a breach on OI instead diff --git a/tests/test_vintage_schedule.py b/tests/test_vintage_schedule.py index 5598280..fc5a922 100644 --- a/tests/test_vintage_schedule.py +++ b/tests/test_vintage_schedule.py @@ -79,6 +79,35 @@ def test_backfill_derives_when_no_schedule(store_env): assert counts["derived"] == 3 and counts["announced"] == 0 +def test_backfill_is_idempotent_and_does_not_downgrade_observed(store_env): + """Re-running backfill must not downgrade an `observed` release date to `scheduled`. + Precedence is enforced on every write (full re-resolve), so a live-captured row whose + observed_at is near its report_date stays `observed` even when a schedule entry exists.""" + from cotdata import vintage_ingest as vi + from cotdata import vintage_schedule as vs + # captured live: observed_at 3 days after report_date (within the 4-day window) + _ingest_one("2026-07-21", observed_at=dt.datetime(2026, 7, 24, 16, tzinfo=dt.timezone.utc)) + schedule = pd.DataFrame([{ + "report_date": pd.Timestamp("2026-07-21"), + "release_date": pd.Timestamp("2026-07-25"), # a competing `scheduled` date + "source": "scheduled", "note": "", "ingested_at": pd.Timestamp.now("UTC"), + }]) + + c1 = vs.backfill(schedule=schedule) + obs1 = vi.read_observations() + src1 = set(obs1["release_date_source"]) + rel1 = pd.Timestamp(obs1.iloc[0]["release_date"]).date() + + c2 = vs.backfill(schedule=schedule) # re-run + obs2 = vi.read_observations() + + assert c1 == c2 # idempotent counts + assert src1 == {"observed"} # observed wins over scheduled + assert rel1 == dt.date(2026, 7, 24) # = observed_at, not the scheduled date + assert set(obs2["release_date_source"]) == {"observed"} # not downgraded on re-run + assert list(obs1["release_date"]) == list(obs2["release_date"]) + + def test_announcement_parse_is_best_effort(): from cotdata.vintage_schedule import _parse_announcements html = "
    • January 5, 2026: revised gold report
    " From 87e7d299476d313297cae0d0f7d65bee45ef3028 Mon Sep 17 00:00:00 2001 From: Matt Spinola Date: Thu, 30 Jul 2026 17:00:04 -0400 Subject: [PATCH 04/16] =?UTF-8?q?fix(vintage):=20review=20fixes=20?= =?UTF-8?q?=E2=80=94=20index=20hazard,=20hash=20portability,=20observed-da?= =?UTF-8?q?te=20semantics?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Second review pass. Three real defects, one invariant, one lock. 1. _latest_by_key returned obs.loc[idx], a duplicate-label hazard: a frame built by a naive concat has repeated index labels and .loc returns EVERY match, silently yielding multiple rows per natural key. Verified: 6 rows where 3 were correct. Now returns the grouped rows directly. 2. _norm no longer depends on numpy/pandas scalar formatting: values unwrap via .item() to Python natives and floats format explicitly (".10g", int-valued as int). The hash is a permanent artifact; a numpy repr change would otherwise appear as a mass revision event across every market at once, worst in the cr4/cr8 ratio fields. 3. backfill resolved `observed` from each ROW's observed_at. A row is one vintage, so a row revised months later produced a release date that late. Now uses the FIRST sighting per report_date. Verified: offsets were [3, 103] days for one report. The window check is also directional (0 <= delta <= window) rather than abs(): observed_at cannot precede report_date, so a negative offset is clock skew or a tz bug and must not be absorbed as a valid release date. Added: ConsistencyError raised when row_sha256 says "changed" but no VALUE_FIELDS differ — the hash and field-diff use different comparison paths, and disagreement would write an observation with no revision detail. Turns future dtype drift into a failure. Added: advisory _WriteLock over the vintage subtree. Writes are read-concat-rewrite, so two concurrent ingests would last-writer-wins; the producer is single-writer by design and this makes a second writer fail loudly instead. Added: `published` precedence slot above `observed` for the weekly-static Last-Modified (a true publication timestamp vs a polling-interval approximation). Not yet populated — mapping a weekly static to its report_date needs that file parsed, deferred by §10 — but the slot exists so the wiring lands without a taxonomy migration. Also documents that market_name is excluded from the hash by design, and therefore is never updated once a natural key is written. Co-Authored-By: Claude Opus 4.8 --- src/cotdata/vintage_ingest.py | 123 +++++++++++++++++++++++++++----- src/cotdata/vintage_schedule.py | 34 ++++++--- tests/test_vintage_ingest.py | 50 +++++++++++++ tests/test_vintage_schedule.py | 49 +++++++++++++ 4 files changed, 232 insertions(+), 24 deletions(-) diff --git a/src/cotdata/vintage_ingest.py b/src/cotdata/vintage_ingest.py index 43d945e..0038598 100644 --- a/src/cotdata/vintage_ingest.py +++ b/src/cotdata/vintage_ingest.py @@ -14,6 +14,7 @@ import datetime as dt import hashlib +import os from pathlib import Path import pandas as pd @@ -31,6 +32,10 @@ "trader_count_long", "trader_count_short", "open_interest", "cr4_net_long", "cr4_net_short", "cr8_net_long", "cr8_net_short"] +# market_name is descriptive, not a value: it is deliberately OUT of the hash, so a CFTC +# market rename never registers as a revision. Consequence to know before debugging it: +# once a natural key is written, its stored market_name is never updated by a later +# ingest (that row is a change-only skip). The market_code is the identity. PROVENANCE_FIELDS = ["release_date", "release_date_source", "market_name", "observed_at", "snapshot_id", "row_sha256", "is_tombstone"] @@ -77,10 +82,33 @@ def _naive_utc(ts) -> pd.Timestamp: def _norm(v) -> str: - if v is None or (isinstance(v, float) and pd.isna(v)) or v is pd.NA: + """Canonical string form of one value, for hashing and field comparison. + + Deliberately independent of any third-party scalar formatting: numpy/pandas scalars + are unwrapped to Python natives via ``.item()`` and formatted explicitly. The hash is + a PERMANENT artifact — if it moved with a numpy/pandas formatting change (numpy 2.0 + already changed scalar repr once) every stored row would appear revised at once, a + silent mass revision event across every market. Non-integer floats only occur in the + cr4/cr8 ratios, which is exactly where that would bite. + """ + if v is None: return "" - if isinstance(v, float) and float(v).is_integer(): - return str(int(v)) + try: + if pd.isna(v): + return "" + except (TypeError, ValueError): + pass # non-scalar or unsupported type → fall through to explicit formatting + if hasattr(v, "item"): + try: + v = v.item() # np.int64/np.float64/np.bool_ → int/float/bool + except (AttributeError, ValueError): + pass + if isinstance(v, bool): # must precede int: bool is a subclass of int + return "1" if v else "0" + if isinstance(v, float): + return str(int(v)) if v.is_integer() else format(v, ".10g") + if isinstance(v, int): + return str(v) return str(v) @@ -131,6 +159,18 @@ class ValidationError(ValueError): pass +class ConsistencyError(RuntimeError): + """The change-hash and the field-level diff disagreed. + + These are two different comparison paths: the hash compares a STORED row_sha256 + against a freshly computed one, while the field loop compares parquet-read values + against fresh ones. If they ever disagree we would write an observation row with no + corresponding revision rows — a revision that silently lost its detail. Raising turns + any future dtype/round-trip drift into a loud failure instead of a data-quality + mystery discovered months later. + """ + + def validate(canonical: pd.DataFrame) -> list[str]: """Fail loudly (raise) on structural problems; return a list of soft warnings. @@ -190,8 +230,51 @@ def _latest_by_key(obs: pd.DataFrame) -> pd.DataFrame: if obs.empty: return obs ordered = obs.sort_values(["observed_at", "snapshot_id"], kind="mergesort") - idx = ordered.groupby(NATURAL_KEY, dropna=False, sort=False).tail(1).index - return obs.loc[idx] + # Return the grouped rows DIRECTLY. Round-tripping through obs.loc[idx] would be a + # duplicate-label hazard: any caller that built `obs` without ignore_index=True has + # repeated index labels, and .loc on those returns EVERY matching row — silently + # yielding several rows per natural key, which reads as data corruption rather than + # an indexing bug. + return ordered.groupby(NATURAL_KEY, dropna=False, sort=False).tail(1) + + +class _WriteLock: + """Advisory single-writer lock over the vintage subtree. + + The observation/revision writes are read-concat-rewrite: atomic per file, but two + concurrent ingest processes would resolve last-writer-wins and silently drop one + side's rows. The COT producer is single-writer by design (the same reason the store + manifests are split per half), so this lock exists to convert that silent data-loss + mode into a loud error, not to support concurrency. + """ + + def __init__(self, root: Path): + self.path = root / ".ingest.lock" + + def __enter__(self): + self.path.parent.mkdir(parents=True, exist_ok=True) + try: + fd = os.open(str(self.path), os.O_CREAT | os.O_EXCL | os.O_WRONLY) + except FileExistsError: + holder = "" + try: + holder = f" (held by pid {self.path.read_text().strip()})" + except OSError: + pass + raise RuntimeError( + f"another cotdata vintage ingest is writing this store{holder}. " + f"The vintage writers are single-writer by design. If no such process is " + f"running, remove the stale lock: {self.path}") from None + with os.fdopen(fd, "w") as fh: + fh.write(str(os.getpid())) + return self + + def __exit__(self, *exc): + try: + self.path.unlink() + except OSError: + pass + return False def _append_parquet(path: Path, new_rows: pd.DataFrame) -> None: @@ -249,6 +332,7 @@ def ingest_canonical(canonical: pd.DataFrame, *, snapshot_id: str, if prev is not None: report_date = pd.Timestamp(rec["report_date"]) age_days = int((detected_at.normalize() - report_date.normalize()).days) + n_before = len(revisions) for f in VALUE_FIELDS: old, new = prev.get(f), rec.get(f) if _norm(old) == _norm(new): @@ -271,19 +355,26 @@ def ingest_canonical(canonical: pd.DataFrame, *, snapshot_id: str, f"{key}|{f}|{prev.get('snapshot_id')}|{snapshot_id}".encode() ).hexdigest()[:16], }) + if len(revisions) == n_before: + raise ConsistencyError( + f"row_sha256 changed for {key} but no VALUE_FIELDS differ " + f"(stored {prev['row_sha256'][:12]} vs computed {rec['row_sha256'][:12]}). " + f"The hash and the field-diff comparison paths disagree — refusing to " + f"write an observation with no revision detail.") n_obs = 0 - for ry, recs in new_obs.items(): - frame = pd.DataFrame(recs).reindex(columns=ALL_COLUMNS) - _append_parquet(_obs_path(ry), frame) - n_obs += len(recs) - - if revisions: - by_year: dict[int, list] = {} - for rv in revisions: - by_year.setdefault(pd.Timestamp(rv["detected_at"]).year, []).append(rv) - for dy, recs in by_year.items(): - _append_parquet(_rev_path(dy), pd.DataFrame(recs)) + with _WriteLock(vintage.vintage_root()): + for ry, recs in new_obs.items(): + frame = pd.DataFrame(recs).reindex(columns=ALL_COLUMNS) + _append_parquet(_obs_path(ry), frame) + n_obs += len(recs) + + if revisions: + by_year: dict[int, list] = {} + for rv in revisions: + by_year.setdefault(pd.Timestamp(rv["detected_at"]).year, []).append(rv) + for dy, recs in by_year.items(): + _append_parquet(_rev_path(dy), pd.DataFrame(recs)) return {"observations": n_obs, "revisions": len(revisions), "warnings": warnings} diff --git a/src/cotdata/vintage_schedule.py b/src/cotdata/vintage_schedule.py index 80c9530..86c19ce 100644 --- a/src/cotdata/vintage_schedule.py +++ b/src/cotdata/vintage_schedule.py @@ -27,7 +27,14 @@ from . import vintage from . import vintage_ingest as vi -PRECEDENCE = ("observed", "announced", "scheduled", "derived", "unknown") +# `published` outranks `observed`: it is the weekly-static HTTP Last-Modified, a TRUE +# publication timestamp (spike 2026-07-30), whereas `observed` is only the first time WE +# saw the report — accurate to the polling interval. They must not share a bucket, so it +# stays possible to tell later which weeks carry a real publication time. +# NOTE: nothing populates `published` yet — mapping a weekly static to its report_date +# requires parsing that file, which handoff §10 defers. The slot and precedence are here +# so the wiring lands without a taxonomy migration. +PRECEDENCE = ("published", "observed", "announced", "scheduled", "derived", "unknown") # ── Paths ─────────────────────────────────────────────────────────────────── @@ -50,12 +57,12 @@ def derive_release_date(report_date) -> dt.date: return d -def resolve_release_date(report_date, *, observed=None, announced=None, +def resolve_release_date(report_date, *, published=None, observed=None, announced=None, scheduled=None) -> tuple[dt.date | None, str]: - """Return ``(release_date, source)`` by the §4.6 precedence. ``observed`` / - ``announced`` / ``scheduled`` are optional resolved dates for this report_date.""" - for value, source in ((observed, "observed"), (announced, "announced"), - (scheduled, "scheduled")): + """Return ``(release_date, source)`` by the §4.6 precedence. Each argument is an + optional resolved date for this report_date; the first present wins.""" + for value, source in ((published, "published"), (observed, "observed"), + (announced, "announced"), (scheduled, "scheduled")): if value is not None and not pd.isna(value): return pd.Timestamp(value).date(), source return derive_release_date(report_date), "derived" @@ -127,14 +134,25 @@ def backfill(*, schedule: pd.DataFrame | None = None, df = pd.read_parquet(part) if df.empty: continue + # FIRST sighting per report_date, not the row's own observed_at. A row is one + # VINTAGE of a natural key: a revised row's observed_at can be months after + # publication, so using it per-row would make every `observed` release date + # systematically late by however long that row went unrevised. + norm_rd = df["report_date"].map(lambda x: pd.Timestamp(x).normalize()) + first_seen = df.assign(_rd=norm_rd).groupby("_rd")["observed_at"].min() + rel_dates, rel_srcs = [], [] for _, r in df.iterrows(): rd = pd.Timestamp(r["report_date"]).normalize() observed = None - oa = r.get("observed_at") + oa = first_seen.get(rd) if oa is not None and not pd.isna(oa): oa = vi._naive_utc(oa) - if abs((oa.normalize() - rd).days) <= observed_window_days: + # Directional, NOT abs(): a report cannot be observed BEFORE its own + # as-of date, so a negative offset means clock skew or a timezone bug. + # abs() would silently absorb that as a valid release date. + delta = (oa.normalize() - rd).days + if 0 <= delta <= observed_window_days: observed = oa sched = smap.get(rd) announced = sched[0] if sched and sched[1] == "announced" else None diff --git a/tests/test_vintage_ingest.py b/tests/test_vintage_ingest.py index c47f981..dfb86ac 100644 --- a/tests/test_vintage_ingest.py +++ b/tests/test_vintage_ingest.py @@ -143,6 +143,56 @@ def test_asof_tiebreak_is_deterministic(store_env): assert again[again["category"] == "commercial"].iloc[0]["snapshot_id"] == "z-late" +def test_norm_is_independent_of_numpy_scalar_formatting(): + """The hash must not depend on numpy/pandas str() behaviour: np scalars unwrap to + Python natives, and int-valued floats unify with ints.""" + np = pytest.importorskip("numpy") + from cotdata.vintage_ingest import _norm, row_sha256 + assert _norm(np.int64(250000)) == _norm(250000) == _norm(np.float64(250000.0)) == "250000" + assert _norm(np.bool_(True)) == _norm(True) == "1" + assert _norm(None) == _norm(pd.NA) == _norm(float("nan")) == "" + # a non-integer ratio formats explicitly, not via numpy's repr + assert _norm(np.float64(0.4375)) == _norm(0.4375) == "0.4375" + # whole-row: numpy-typed and python-typed rows hash identically + a = {"long_contracts": np.int64(10), "cr4_net_long": np.float64(0.25)} + b = {"long_contracts": 10, "cr4_net_long": 0.25} + assert row_sha256(a) == row_sha256(b) + + +def test_asof_returns_one_row_per_key_with_duplicate_index(store_env): + """_latest_by_key must not round-trip through .loc on a duplicated index — that + returns every matching label and yields several rows per natural key.""" + from cotdata import vintage_ingest as vi + t = dt.datetime(2026, 7, 24, 16, tzinfo=dt.timezone.utc) + vi.ingest_canonical(_canon("2026-07-21"), snapshot_id="s1", observed_at=t) + obs = vi.read_observations() + dup = pd.concat([obs, obs]) # duplicated index labels, as a naive concat would give + latest = vi._latest_by_key(dup) + assert len(latest) == 3 # 3 categories, not 6 + assert latest.groupby(vi.NATURAL_KEY, dropna=False).size().max() == 1 + + +def test_hash_change_without_field_diff_raises(store_env, monkeypatch): + """The consistency invariant: if the stored hash says 'changed' but no VALUE_FIELDS + differ, refuse to write an observation with no revision detail.""" + from cotdata import vintage_ingest as vi + vi.ingest_canonical(_canon("2026-07-21"), snapshot_id="s1") + # corrupt the comparison: make every freshly computed hash differ, while the actual + # field values stay identical — exactly the drift the invariant guards against. + monkeypatch.setattr(vi, "row_sha256", lambda row: "deadbeef" * 8) + with pytest.raises(vi.ConsistencyError, match="no VALUE_FIELDS differ"): + vi.ingest_canonical(_canon("2026-07-21"), snapshot_id="s2") + + +def test_concurrent_ingest_lock_fails_loudly(store_env): + """A second writer must error, not silently last-writer-wins over the first.""" + from cotdata import vintage + from cotdata import vintage_ingest as vi + with vi._WriteLock(vintage.vintage_root()): + with pytest.raises(RuntimeError, match="single-writer"): + vi.ingest_canonical(_canon("2026-07-21"), snapshot_id="s1") + + def test_oi_over_sum_warns_not_raises(store_env): from cotdata import vintage_ingest as vi # commercial long+short (200000+250000) already < OI; force a breach on OI instead diff --git a/tests/test_vintage_schedule.py b/tests/test_vintage_schedule.py index fc5a922..0997d59 100644 --- a/tests/test_vintage_schedule.py +++ b/tests/test_vintage_schedule.py @@ -108,6 +108,55 @@ def test_backfill_is_idempotent_and_does_not_downgrade_observed(store_env): assert list(obs1["release_date"]) == list(obs2["release_date"]) +def test_observed_uses_first_sighting_not_the_revised_row(store_env): + """A later revision must not drag the `observed` release date forward. The release + date is the FIRST time the report was seen, not when a given vintage was written.""" + from cotdata import vintage_ingest as vi + from cotdata import vintage_schedule as vs + # first seen 3 days after report_date (a live capture)… + _ingest_one("2026-07-21", observed_at=dt.datetime(2026, 7, 24, 16, tzinfo=dt.timezone.utc)) + # …then revised months later + idx = pd.to_datetime(["2026-07-21"]) + idx.name = "Report_Date_as_MM_DD_YYYY" + revised = pd.DataFrame({ + "Market_and_Exchange_Names": ["GOLD"], "CFTC_Contract_Market_Code": ["088691"], + "Open_Interest_All": [500000], + "Comm_Positions_Long_All": [200000], "Comm_Positions_Short_All": [999999], + "NonComm_Positions_Long_All": [150000], "NonComm_Positions_Short_All": [90000], + "NonRept_Positions_Long_All": [40000], "NonRept_Positions_Short_All": [30000], + "Traders_Comm_Long_All": [50], "Traders_Comm_Short_All": [55], + "Traders_NonComm_Long_All": [60], "Traders_NonComm_Short_All": [45], + }, index=idx) + vi.ingest_canonical(vi.canonicalize_legacy(revised), snapshot_id="s2", + observed_at=dt.datetime(2026, 11, 1, tzinfo=dt.timezone.utc)) + + counts = vs.backfill(schedule=pd.DataFrame( + columns=["report_date", "release_date", "source", "note", "ingested_at"])) + obs = vi.read_observations() + # every row (including the revised vintage) carries the FIRST-sighting release date + assert set(obs["release_date_source"]) == {"observed"} + assert {pd.Timestamp(d).date() for d in obs["release_date"]} == {dt.date(2026, 7, 24)} + assert counts["observed"] == len(obs) and counts["derived"] == 0 + + +def test_observed_window_rejects_negative_offset(store_env): + """observed_at before report_date is impossible (clock skew / tz bug) and must NOT + be absorbed as a valid `observed` release date.""" + from cotdata import vintage_schedule as vs + # "observed" two days BEFORE the report's as-of date + _ingest_one("2026-07-21", observed_at=dt.datetime(2026, 7, 19, tzinfo=dt.timezone.utc)) + counts = vs.backfill(schedule=pd.DataFrame( + columns=["report_date", "release_date", "source", "note", "ingested_at"])) + assert counts["observed"] == 0 and counts["derived"] == 3 # fell through to derived + + +def test_published_outranks_observed(): + from cotdata.vintage_schedule import resolve_release_date + val, src = resolve_release_date("2026-07-21", published="2026-07-24", + observed="2026-07-25", scheduled="2026-07-26") + assert src == "published" and val == dt.date(2026, 7, 24) + + def test_announcement_parse_is_best_effort(): from cotdata.vintage_schedule import _parse_announcements html = "
    • January 5, 2026: revised gold report
    " From 8dfd243d36d1851372eb050a72b29f6c120acdde Mon Sep 17 00:00:00 2001 From: Matt Spinola Date: Thu, 30 Jul 2026 17:14:28 -0400 Subject: [PATCH 05/16] =?UTF-8?q?fix(vintage):=20capture=20hardening=20?= =?UTF-8?q?=E2=80=94=20corrupt=20manifest,=20atomic=20raw=20write,=20id=20?= =?UTF-8?q?collision,=20per-source=20guard?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Third review pass on the capture path. 1. A corrupt manifest no longer degrades to an empty one. _read_manifest quarantined the damaged file and returned {} — and the next write then overwrote the record with that empty structure. The manifest is the ONLY map from snapshot_id to url/sha/ retrieval time/parse status, so losing it leaves a directory of opaque blobs while the bytes survive. Now moves it aside to manifest.json.corrupt. and raises CorruptManifestError, per the fail-loudly rule. 2. Raw bytes are written .part + os.replace. A crash during a plain write_bytes left a TRUNCATED file already carrying the full sha in its name, so any sha-keyed recovery would adopt it as valid. 3. snapshot_id carries a url discriminator. _utcnow() truncates to whole seconds, so several sources returning 304 in one second shared "{retrieved_at}_304" — and update_snapshot patches EVERY record with a matching id. Rate limiting hid it in production; rate_limit_s=0 in tests walks into it. 4. fetch guards per source. One 404 propagated out and killed an entire --all run (120+ requests) with nothing recorded. Failures are now recorded as a record with a note and the run continues, matching the ingest path. Also: the manifest is written after EVERY source rather than once at the end, shrinking the crash window from a whole run to one source (small file, atomic replace, no measurable cost) — this removes the need for a reconcile companion. _latest_for_url takes an explicit max over retrieved_at instead of the last list element. The weekly static partitions by CAPTURE year rather than "current/". A minimum-byte floor per source_kind refuses truncated/empty 200 bodies (content is None does not catch b""). The User-Agent moves to COTDATA_USER_AGENT with a repo-URL default, out of source. Documented two measured findings: annual-zip sha churn (closed years are content-frozen and dedupe to nothing even though Last-Modified is re-touched weekly; only the current year genuinely churns, ~20MB/week) and the deliberate futures-only scope that leaves `combined` constant-False. Co-Authored-By: Claude Opus 4.8 --- docs/design/cot_vintage.md | 42 +++++++++ src/cotdata/vintage.py | 148 ++++++++++++++++++++++++++++---- src/cotdata/vintage_schedule.py | 2 +- tests/test_vintage_capture.py | 135 ++++++++++++++++++++++++++--- 4 files changed, 300 insertions(+), 27 deletions(-) diff --git a/docs/design/cot_vintage.md b/docs/design/cot_vintage.md index 0c83f6f..fcbabd9 100644 --- a/docs/design/cot_vintage.md +++ b/docs/design/cot_vintage.md @@ -62,6 +62,48 @@ the sandbox caveat noted in CLAUDE.md): overwritten file, **not a frozen historical archive** — no further historical vintage recovery is possible through them. +## 3b. Measured: sha churn on annual zips (2026-07-30) + +Open question was whether zip regeneration rewrites embedded entry timestamps, which +would make every annual file a new sha every week (byte-dedupe never fires, a fresh +multi-MB copy retained weekly). **Measured, and it does not.** + +| File | HTTP `Last-Modified` | inner entry mtime | Reading | +|---|---|---|---| +| `dea_fut_xls_2015.zip` | — | 2015-12-31 | frozen at year end | +| `dea_fut_xls_2024.zip` | — | 2025-01-03 | frozen just after year end | +| `dea_fut_xls_2025.zip` | **2026-07-24** (this week) | **2026-01-02** | `Last-Modified` churns weekly; CONTENT frozen since January | +| `dea_fut_xls_2026.zip` | 2026-07-24 | 2026-07-23 | current year: genuinely regenerated, real new data | + +Two consequences: + +1. **Closed years dedupe to nothing.** Their `Last-Modified` is re-touched weekly (so a + conditional GET may still transfer), but the bytes are identical, so `content_sha256` + matches and the dedupe branch retains no second copy. `--all` on a schedule is + therefore cheap in storage, contrary to the worry — it costs bandwidth, not disk. + This is precisely the case §3.4's "changed download is not a revision" exists for. +2. **The current year does churn weekly**, correctly: a new report is appended, so the + sha genuinely changes and a new ~5-8 MB snapshot is retained. At the default + (3 annual + 1 weekly static) that is roughly **20 MB/week, ~1 GB/year** — small in + absolute terms but large enough that a retention policy should be a deliberate + decision, not a discovery. Options: keep all (simplest, ~1 GB/yr), or keep the raw + file only when the parsed values differ from the previous vintage. + +## 3c. Decision: futures-only, `combined` is constant-False for now + +`annual_sources` fetches `dea_fut_xls` / `fut_disagg_txt` / `fut_fin_txt` — all +**futures-only**. The futures-and-options-combined files are NOT fetched, so the +`combined` dimension of the natural key is `False` for every row today. + +This is a **deliberate carry-forward, not an omission**: the existing producer has only +ever fetched futures-only (discovery §2.3), and this task's stated non-goal is to avoid +changing existing fetch scope. `combined` stays in the natural key so the two series can +never silently merge into one time series (§6) if/when combined files are added — adding +them later is a fetch-list change, with no schema migration. + +Consequence to be explicit about: **half the reportable universe is absent** until that +happens. Anything needing futures-and-options-combined positioning is blocked on this. + ## 4. Design as adapted to this repo Layout under `$COTDATA_STORE/vintage/` (a new subtree, disjoint from `current/` which diff --git a/src/cotdata/vintage.py b/src/cotdata/vintage.py index c32b645..77032fe 100644 --- a/src/cotdata/vintage.py +++ b/src/cotdata/vintage.py @@ -36,10 +36,23 @@ from . import config -_UA = "cotdata-vintage/0.1 (COT research; contact matt.spinola@gmail.com)" +# Descriptive UA, overridable per deployment via COTDATA_USER_AGENT. Deliberately not a +# personal address in committed source: appropriate to send to CFTC, not to publish. +_DEFAULT_UA = "cotdata-vintage/0.1 (+https://github.com/mspinola/cotdata)" _RATE_LIMIT_S = 1.0 # polite spacing between real network requests SCHEMA_VERSION = 1 +# Byte-level floor per source kind: the analogue of §5's row-count band. A truncated or +# empty 200 body would otherwise be retained as a legitimate snapshot (``content is None`` +# does not catch ``b""``). Deliberately conservative — the smallest real annual zip +# (1986) is far above this, so the floor only ever catches a broken response. +_MIN_BYTES = {"annual_zip": 1024, "weekly_static": 1024} +_MIN_BYTES_DEFAULT = 512 + + +def user_agent() -> str: + return os.environ.get("COTDATA_USER_AGENT", "").strip() or _DEFAULT_UA + # ── Paths ─────────────────────────────────────────────────────────────────── def vintage_root() -> Path: @@ -47,6 +60,9 @@ def vintage_root() -> Path: def raw_dir(source_kind: str, year: int | str) -> Path: + """Partition for retained raw bytes. For sources with no report year (the weekly + static) the caller passes the CAPTURE year — 'current/' would be actively wrong the + moment it stopped being current.""" return vintage_root() / "raw" / source_kind / str(year) @@ -107,7 +123,7 @@ def _http_get(url: str, *, etag: str | None, last_modified: str | None) -> HttpR """ import requests # local import: keeps the module importable without network deps - headers = {"User-Agent": _UA} + headers = {"User-Agent": user_agent()} if etag: headers["If-None-Match"] = etag if last_modified: @@ -122,14 +138,39 @@ def _http_get(url: str, *, etag: str | None, last_modified: str | None) -> HttpR # ── Manifest (self-owned, append-only snapshot index) ─────────────────────── +class CorruptManifestError(RuntimeError): + """The vintage manifest exists but could not be parsed. + + Never recovered from by returning an empty manifest: the next write would then + overwrite the damaged file with that empty structure, and the manifest is the ONLY + mapping from snapshot_id to url / sha / retrieval time / parse status. Losing it + leaves a directory of opaquely-named blobs while the raw bytes themselves survive — + the worst outcome available in this module, and reachable from one interrupted write. + """ + + def _read_manifest() -> dict: p = manifest_path() if not p.exists(): return {"schema_version": SCHEMA_VERSION, "snapshots": []} try: m = json.loads(p.read_text()) - except json.JSONDecodeError: - return {"schema_version": SCHEMA_VERSION, "snapshots": []} + except (json.JSONDecodeError, UnicodeDecodeError) as e: + stamp = _utcnow().strftime("%Y%m%dT%H%M%SZ") + quarantine = p.with_suffix(f".json.corrupt.{stamp}") + try: + os.replace(p, quarantine) + except OSError: + quarantine = None + raise CorruptManifestError( + f"{p} is unreadable ({e}). " + + (f"Moved aside to {quarantine}. " if quarantine else "Could not move it aside. ") + + "Raw bytes under vintage/raw/ are intact and self-describing (each filename " + "carries its sha256 prefix); rebuild the index from them rather than " + "letting an empty manifest overwrite the record." + ) from e + if not isinstance(m, dict): + raise CorruptManifestError(f"{p} does not contain a JSON object.") m.setdefault("snapshots", []) m.setdefault("schema_version", SCHEMA_VERSION) return m @@ -144,8 +185,28 @@ def _write_manifest(m: dict) -> None: def _latest_for_url(snapshots: list[dict], url: str) -> dict | None: - prev = [s for s in snapshots if s.get("source_url") == url] - return prev[-1] if prev else None + """Most recent snapshot for a URL, by an EXPLICIT max over retrieved_at rather than + 'last element'. The list is append-only and chronological today, but an implicit + ordering assumption is exactly what breaks when a merge or dedupe pass is added + later. Ties fall back to list position, which preserves today's behaviour.""" + prev = [(s.get("retrieved_at") or "", i, s) + for i, s in enumerate(snapshots) if s.get("source_url") == url] + if not prev: + return None + return max(prev, key=lambda t: (t[0], t[1]))[2] + + +def _snapshot_id(retrieved_at: str, url: str, tag: str) -> str: + """Unique per (retrieval second, url, content-state). + + The URL discriminator is load-bearing: _utcnow() truncates to whole seconds, so + several sources returning 304 within one second would otherwise share the id + ``{retrieved_at}_304`` — and update_snapshot patches EVERY record matching an id, so + one later parse-status write would hit all of them. Rate limiting hides this in + production; a test with rate_limit_s=0 walks straight into it. + """ + url8 = hashlib.sha256(url.encode()).hexdigest()[:8] + return f"{retrieved_at}_{tag}_{url8}" def read_snapshots() -> list[dict]: @@ -204,7 +265,7 @@ def capture_source(source: Source, *, snapshots: list[dict], http_get, now: dt.d # Server says unchanged: record the check, retain nothing new, reuse prior file. return { **base, - "snapshot_id": f"{retrieved_at}_304", + "snapshot_id": _snapshot_id(retrieved_at, source.url, "304"), "http_status": 304, "http_etag": (prev or {}).get("http_etag"), "http_last_modified": (prev or {}).get("http_last_modified"), @@ -214,12 +275,19 @@ def capture_source(source: Source, *, snapshots: list[dict], http_get, now: dt.d "note": "304 not-modified", } + floor = _MIN_BYTES.get(source.source_kind, _MIN_BYTES_DEFAULT) + if len(res.content) < floor: + raise ValueError( + f"{source.url} returned {len(res.content)} bytes, below the {floor}-byte floor " + f"for {source.source_kind}. Refusing to retain a truncated/empty response as a " + f"legitimate snapshot.") + sha = hashlib.sha256(res.content).hexdigest() if prev and prev.get("content_sha256") == sha: # Byte-identical to what we already retained (zips regenerate): dedupe, no rewrite. return { **base, - "snapshot_id": f"{retrieved_at}_{sha[:8]}", + "snapshot_id": _snapshot_id(retrieved_at, source.url, sha[:8]), "http_status": res.status, "http_etag": res.etag, "http_last_modified": res.last_modified, @@ -229,16 +297,26 @@ def capture_source(source: Source, *, snapshots: list[dict], http_get, now: dt.d "note": "unchanged bytes (deduped)", } - # New bytes: retain immutably. + # New bytes: retain immutably, via a .part file + atomic replace. A crash during a + # plain write_bytes would leave a TRUNCATED file already carrying the full sha in its + # name — the filename asserts an integrity claim the contents don't satisfy, and any + # sha-keyed recovery pass would then adopt it as valid. compact = retrieved_at.replace("-", "").replace(":", "") fname = f"{compact}_{sha[:8]}.{source.ext}" - dest = raw_dir(source.source_kind, source.report_year or "current") / fname + year = source.report_year if source.report_year is not None else now.year + dest = raw_dir(source.source_kind, year) / fname dest.parent.mkdir(parents=True, exist_ok=True) - dest.write_bytes(res.content) + part = dest.with_suffix(dest.suffix + ".part") + try: + part.write_bytes(res.content) + os.replace(part, dest) + finally: + if part.exists(): + part.unlink() rel = str(dest.relative_to(config.store_root())) return { **base, - "snapshot_id": f"{retrieved_at}_{sha[:8]}", + "snapshot_id": _snapshot_id(retrieved_at, source.url, sha[:8]), "http_status": res.status, "http_etag": res.etag, "http_last_modified": res.last_modified, @@ -271,18 +349,56 @@ def fetch(year: int | None = None, *, all_years: bool = False, sources.append(WEEKLY_STATIC) m = _read_manifest() + m["schema_version"] = SCHEMA_VERSION snapshots = m["snapshots"] new_files = 0 + failed = 0 records = [] for i, src in enumerate(sources): - rec = capture_source(src, snapshots=snapshots, http_get=http_get, now=now_fn()) + now = now_fn() + try: + rec = capture_source(src, snapshots=snapshots, http_get=http_get, now=now) + except Exception as e: # noqa: BLE001 + # One bad source must not kill the run. A cold-start --all is 120+ requests + # and dies at the first naming variant or missing disagg year otherwise. The + # failure is RECORDED (so it is visible and retryable), then the run goes on — + # matching what the ingest path already does. + rec = _failure_record(src, now, e) + failed += 1 + print(f" {src.report_type}/{src.source_kind} {src.report_year or ''}: " + f"fetch failed — {e}") snapshots.append(rec) # visible to the next source's _latest_for_url records.append(rec) if rec.get("note") is None and rec.get("byte_size") is not None: new_files += 1 + # Write after EVERY source, not once at the end: the manifest is small and its + # replace is atomic, so this costs nothing measurable and shrinks the crash window + # from a whole run to a single source. Combined with the atomic raw write, an + # interrupted run leaves a consistent store rather than unrecorded blobs. + _write_manifest(m) if rate_limit_s and i < len(sources) - 1: time.sleep(rate_limit_s) - m["schema_version"] = SCHEMA_VERSION - _write_manifest(m) - return {"records": records, "new_files": new_files, "checks": len(records)} + return {"records": records, "new_files": new_files, "checks": len(records), + "failed": failed} + + +def _failure_record(source: Source, now: dt.datetime, error: Exception) -> dict: + retrieved_at = _iso(now) + return { + "source_url": source.url, + "source_kind": source.source_kind, + "report_type": source.report_type, + "report_year": source.report_year, + "retrieved_at": retrieved_at, + "snapshot_id": _snapshot_id(retrieved_at, source.url, "failed"), + "http_status": None, + "http_etag": None, + "http_last_modified": None, + "content_sha256": None, + "byte_size": None, + "local_path": None, + "parse_status": "pending", + "parse_error": None, + "note": f"fetch failed: {error}", + } diff --git a/src/cotdata/vintage_schedule.py b/src/cotdata/vintage_schedule.py index 86c19ce..a944d49 100644 --- a/src/cotdata/vintage_schedule.py +++ b/src/cotdata/vintage_schedule.py @@ -179,7 +179,7 @@ def backfill(*, schedule: pd.DataFrame | None = None, def _fetch_html(url: str) -> str: import requests - r = requests.get(url, headers={"User-Agent": vintage._UA}, timeout=120) + r = requests.get(url, headers={"User-Agent": vintage.user_agent()}, timeout=120) r.raise_for_status() return r.text diff --git a/tests/test_vintage_capture.py b/tests/test_vintage_capture.py index 2873eb1..3738e26 100644 --- a/tests/test_vintage_capture.py +++ b/tests/test_vintage_capture.py @@ -39,6 +39,12 @@ def now(): return now +def _body(marker: bytes) -> bytes: + """A payload above the minimum-size floor. Real annual zips are megabytes; the floor + exists to refuse truncated/empty responses, so fixtures must look plausibly sized.""" + return marker + b"\x00" * 2048 + + def _only(url, *tuples): return {url: list(tuples)} @@ -58,7 +64,7 @@ def _fetch(vintage, http, now): def test_fetch_retains_raw_bytes_and_records_provenance(store_env): from cotdata import vintage - http = _FakeHttp(_only(LEGACY_2026, (200, b"ZIPBYTES-week1", '"etag1"', "Fri, 24 Jul 2026 19:27:59 GMT"))) + http = _FakeHttp(_only(LEGACY_2026, (200, _body(b"ZIPBYTES-week1"), '"etag1"', "Fri, 24 Jul 2026 19:27:59 GMT"))) res = _fetch(vintage, http, _clock()) @@ -66,12 +72,12 @@ def test_fetch_retains_raw_bytes_and_records_provenance(store_env): rec = res["records"][0] # raw bytes retained on disk, immutably, under the year partition raw = store_env / rec["local_path"] - assert raw.exists() and raw.read_bytes() == b"ZIPBYTES-week1" + assert raw.exists() and raw.read_bytes() == _body(b"ZIPBYTES-week1") assert raw.parent == store_env / "vintage" / "raw" / "annual_zip" / "2026" # provenance recorded: sha, size, etag, last-modified, status import hashlib - assert rec["content_sha256"] == hashlib.sha256(b"ZIPBYTES-week1").hexdigest() - assert rec["byte_size"] == len(b"ZIPBYTES-week1") + assert rec["content_sha256"] == hashlib.sha256(_body(b"ZIPBYTES-week1")).hexdigest() + assert rec["byte_size"] == len(_body(b"ZIPBYTES-week1")) assert rec["http_etag"] == '"etag1"' assert rec["http_last_modified"] == "Fri, 24 Jul 2026 19:27:59 GMT" assert rec["parse_status"] == "pending" @@ -83,7 +89,7 @@ def test_304_records_check_without_new_file(store_env): from cotdata import vintage http = _FakeHttp(_only( LEGACY_2026, - (200, b"ZIPBYTES-week1", '"etag1"', "Fri, 24 Jul 2026 19:27:59 GMT"), + (200, _body(b"ZIPBYTES-week1"), '"etag1"', "Fri, 24 Jul 2026 19:27:59 GMT"), (304, None, None, None), )) now = _clock() @@ -107,8 +113,8 @@ def test_byte_identical_regeneration_is_deduped(store_env): from cotdata import vintage http = _FakeHttp(_only( LEGACY_2026, - (200, b"SAME", '"e1"', "lm1"), - (200, b"SAME", '"e2"', "lm2"), # server didn't honour conditional GET, but bytes match + (200, _body(b"SAME"), '"e1"', "lm1"), + (200, _body(b"SAME"), '"e2"', "lm2"), # server didn't honour conditional GET, but bytes match )) now = _clock() _fetch(vintage, http, now) @@ -119,12 +125,121 @@ def test_byte_identical_regeneration_is_deduped(store_env): assert len(list((store_env / "vintage" / "raw").rglob("*.zip"))) == 1 +def test_corrupt_manifest_raises_and_quarantines_rather_than_overwriting(store_env): + """A corrupt manifest must never be silently replaced by an empty one: it is the only + map from snapshot_id to url/sha/parse-status, and the raw bytes outlive it.""" + from cotdata import vintage + http = _FakeHttp(_only(LEGACY_2026, (200, _body(b"ZIPBYTES-week1") * 100, '"e1"', "lm1"))) + _fetch(vintage, http, _clock()) + mpath = vintage.manifest_path() + mpath.write_text("{ this is not json") + + with pytest.raises(vintage.CorruptManifestError, match="unreadable"): + vintage.read_snapshots() + # original moved aside, not destroyed; no empty manifest left in its place + quarantined = list(mpath.parent.glob("manifest.json.corrupt.*")) + assert len(quarantined) == 1 + assert quarantined[0].read_text() == "{ this is not json" + assert not mpath.exists() + # and the retained raw bytes are untouched + assert len(list((store_env / "vintage" / "raw").rglob("*.zip"))) == 1 + + +def test_raw_write_is_atomic_no_part_files_left(store_env): + from cotdata import vintage + http = _FakeHttp(_only(LEGACY_2026, (200, b"Z" * 5000, '"e1"', "lm1"))) + _fetch(vintage, http, _clock()) + assert not list((store_env / "vintage" / "raw").rglob("*.part")) + raw = list((store_env / "vintage" / "raw").rglob("*.zip"))[0] + assert raw.read_bytes() == b"Z" * 5000 + + +def test_304_snapshot_ids_are_unique_within_one_second(store_env): + """Several sources returning 304 in the same second must not share a snapshot_id — + update_snapshot patches every record with a matching id.""" + from cotdata import vintage + u1 = "https://www.cftc.gov/files/dea/history/dea_fut_xls_2026.zip" + u2 = "https://www.cftc.gov/files/dea/history/fut_disagg_txt_2026.zip" + srcs = [vintage.Source("legacy", "annual_zip", "zip", u1, 2026), + vintage.Source("disaggregated", "annual_zip", "zip", u2, 2026)] + http = _FakeHttp({u1: [(304, None, None, None)], u2: [(304, None, None, None)]}) + # a frozen clock: every source is stamped the same whole second + res = vintage.fetch(sources=srcs, http_get=http, rate_limit_s=0, + now_fn=lambda: dt.datetime(2026, 7, 30, 16, 0, tzinfo=dt.timezone.utc)) + ids = [r["snapshot_id"] for r in res["records"]] + assert len(set(ids)) == 2, f"snapshot_id collision: {ids}" + + +def test_one_failing_source_does_not_kill_the_run(store_env): + """A 404 mid-run must be recorded and skipped, not abort the remaining sources.""" + from cotdata import vintage + u1 = "https://www.cftc.gov/files/dea/history/dea_fut_xls_1986.zip" + u2 = "https://www.cftc.gov/files/dea/history/dea_fut_xls_2026.zip" + + class _Http(_FakeHttp): + def __call__(self, url, *, etag=None, last_modified=None): + if url == u1: + raise RuntimeError("404 Not Found") + return super().__call__(url, etag=etag, last_modified=last_modified) + + srcs = [vintage.Source("legacy", "annual_zip", "zip", u1, 1986), + vintage.Source("legacy", "annual_zip", "zip", u2, 2026)] + http = _Http({u2: [(200, b"GOOD" * 500, '"e"', "lm")]}) + res = vintage.fetch(sources=srcs, http_get=http, rate_limit_s=0, now_fn=_clock()) + + assert res["failed"] == 1 and res["new_files"] == 1 and res["checks"] == 2 + notes = [r.get("note") or "" for r in res["records"]] + assert any("fetch failed" in n for n in notes) + # the failure is persisted, so it is visible and retryable + assert any("fetch failed" in (s.get("note") or "") for s in vintage.read_snapshots()) + + +def test_truncated_response_is_refused(store_env): + """An implausibly small 200 body is not retained as a legitimate snapshot.""" + from cotdata import vintage + http = _FakeHttp(_only(LEGACY_2026, (200, b"", '"e"', "lm"))) + res = vintage.fetch(sources=_legacy_only(), http_get=http, rate_limit_s=0, now_fn=_clock()) + assert res["failed"] == 1 and res["new_files"] == 0 + assert not list((store_env / "vintage" / "raw").rglob("*.zip")) + + +def test_manifest_persisted_after_each_source(store_env): + """The manifest is written per source, so an interrupted run keeps what it captured.""" + from cotdata import vintage + u1, u2 = LEGACY_2026, "https://www.cftc.gov/files/dea/history/fut_disagg_txt_2026.zip" + seen = [] + + def http(url, *, etag=None, last_modified=None): + from cotdata.vintage import HttpResult + if url == u2: # simulate a crash after the first source was captured + assert len(vintage.read_snapshots()) == 1, "first source not yet persisted!" + seen.append(url) + raise KeyboardInterrupt("simulated crash") + return HttpResult(200, b"A" * 5000, etag='"e"', last_modified="lm") + + srcs = [vintage.Source("legacy", "annual_zip", "zip", u1, 2026), + vintage.Source("disaggregated", "annual_zip", "zip", u2, 2026)] + with pytest.raises(KeyboardInterrupt): + vintage.fetch(sources=srcs, http_get=http, rate_limit_s=0, now_fn=_clock()) + assert seen == [u2] + assert len(vintage.read_snapshots()) == 1 # survived the crash + + +def test_weekly_static_partitions_by_capture_year(store_env): + from cotdata import vintage + http = _FakeHttp({vintage.WEEKLY_STATIC.url: [(200, b"T" * 5000, '"e"', "lm")]}) + vintage.fetch(sources=[vintage.WEEKLY_STATIC], http_get=http, rate_limit_s=0, + now_fn=_clock("2026-07-30T16:00:00+00:00")) + raws = list((store_env / "vintage" / "raw" / "weekly_static").rglob("*.txt")) + assert len(raws) == 1 and raws[0].parent.name == "2026" # not "current" + + def test_changed_bytes_writes_second_immutable_snapshot(store_env): from cotdata import vintage http = _FakeHttp(_only( LEGACY_2026, - (200, b"week1", '"e1"', "lm1"), - (200, b"week2-revised", '"e2"', "lm2"), + (200, _body(b"week1"), '"e1"', "lm1"), + (200, _body(b"week2-revised"), '"e2"', "lm2"), )) now = _clock() _fetch(vintage, http, now) @@ -132,4 +247,4 @@ def test_changed_bytes_writes_second_immutable_snapshot(store_env): raws = sorted((store_env / "vintage" / "raw").rglob("*.zip")) assert len(raws) == 2 # both vintages retained; neither overwritten - assert {p.read_bytes() for p in raws} == {b"week1", b"week2-revised"} + assert {p.read_bytes() for p in raws} == {_body(b"week1"), _body(b"week2-revised")} From be17291a29c0c982e9fd10f3e197b5e0c313fb21 Mon Sep 17 00:00:00 2001 From: Matt Spinola Date: Thu, 30 Jul 2026 17:33:10 -0400 Subject: [PATCH 06/16] docs(vintage): correct the churn finding, record retention + --all tripwire, live smoke MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Corrects a wrong claim in the previous commit's doc and records the operational decisions the measurement settles. - CORRECTION: "closed years are re-touched weekly" was wrong. A header sweep across 2020-2026 shows only the current year and the immediately-prior year move (2025 and 2026 share an identical Last-Modified — a rolling two-year regeneration window); 2022-2024 got one bulk re-touch in Jan 2026, and 2020/2021 are static. - cftc.gov serves NO ETag on any of these files, so If-None-Match can never fire. If-Modified-Since DOES return 304 (verified against a static year and the current year), so the conditional GET is not decorative: on an --all sweep only ~2 of ~40 annual files transfer. Noted in _http_get so nobody assumes If-None-Match works. - Retention decision: keep everything, no pruning. ~1GB/yr is immaterial against irreplaceability, and those weekly copies ARE the vintage series. - --all is a restatement TRIPWIRE, not a backfill: a content_sha256 change on a frozen closed year is the 2008-style retroactive-restatement signature, and this is the only automated detector for it. Monthly/quarterly, not weekly. - Caveat recorded: inner entry mtime is evidence, not proof (a regeneration preserving source mtimes looks identical). content_sha256 across two weeks is definitive, and capture now does that automatically. - Pre-merge live smoke: default path run against cftc.gov into a throwaway store. All four URLs resolved, UA accepted, four 200s retained with provenance, legacy sha matched an independent manual download; a second run returned 304 on all four and retained nothing. Co-Authored-By: Claude Opus 4.8 --- docs/design/cot_vintage.md | 71 +++++++++++++++++++++++++++++++------- src/cotdata/vintage.py | 5 +++ 2 files changed, 63 insertions(+), 13 deletions(-) diff --git a/docs/design/cot_vintage.md b/docs/design/cot_vintage.md index fcbabd9..3f95dbd 100644 --- a/docs/design/cot_vintage.md +++ b/docs/design/cot_vintage.md @@ -75,19 +75,64 @@ multi-MB copy retained weekly). **Measured, and it does not.** | `dea_fut_xls_2025.zip` | **2026-07-24** (this week) | **2026-01-02** | `Last-Modified` churns weekly; CONTENT frozen since January | | `dea_fut_xls_2026.zip` | 2026-07-24 | 2026-07-23 | current year: genuinely regenerated, real new data | -Two consequences: - -1. **Closed years dedupe to nothing.** Their `Last-Modified` is re-touched weekly (so a - conditional GET may still transfer), but the bytes are identical, so `content_sha256` - matches and the dedupe branch retains no second copy. `--all` on a schedule is - therefore cheap in storage, contrary to the worry — it costs bandwidth, not disk. - This is precisely the case §3.4's "changed download is not a revision" exists for. -2. **The current year does churn weekly**, correctly: a new report is appended, so the - sha genuinely changes and a new ~5-8 MB snapshot is retained. At the default - (3 annual + 1 weekly static) that is roughly **20 MB/week, ~1 GB/year** — small in - absolute terms but large enough that a retention policy should be a deliberate - decision, not a discovery. Options: keep all (simplest, ~1 GB/yr), or keep the raw - file only when the parsed values differ from the previous vintage. +**Caveat on the evidence.** Inner entry mtime is strong evidence, not proof: a +regeneration that preserved source mtimes would look identical. The definitive test is +`content_sha256` compared across two weeks, which the capture system now performs +automatically. Week one settles it. + +### Which files are actually re-touched (measured 2026-07-30) + +An earlier draft of this section claimed "closed years are re-touched weekly." **That was +wrong.** The header sweep shows only the current year and the immediately-prior year move: + +| Year | `Last-Modified` | +|---|---| +| 2020 | 2021-10-29 | +| 2021 | 2022-01-14 | +| 2022 / 2023 / 2024 | 2026-01-15 (one bulk re-touch, not weekly) | +| **2025** | **2026-07-24 19:27:59** | +| **2026** | **2026-07-24 19:27:59** | + +2025 and 2026 share an identical timestamp: the weekly job regenerates a rolling +current-plus-prior-year window. Everything older is static. + +### Conditional GET: works, and `If-None-Match` is dead weight + +- **No `ETag` on any CFTC file** (annual zips or weekly static). `If-None-Match` can + therefore never fire. It is harmless to keep sending (future-proofing if CFTC adds + one), but `If-Modified-Since` is the only live mechanism today. +- **`If-Modified-Since` genuinely returns 304**, verified directly against both a static + year (2015) and the current year. Confirmed end-to-end in the pipeline: a second + `cotdata-vintage fetch` returned 304 on all four default sources, retained nothing new, + and recorded four check-only snapshot records. + +So the conditional GET is NOT decorative: on an `--all` run only the current and prior +year actually transfer, and ~38 of ~40 annual files 304. Sha-dedupe is the backstop for +the two that do transfer (2025 re-downloads weekly, ~8 MB, and dedupes away). + +### Decisions this settles + +1. **Retention: keep everything, no pruning.** ~1 GB/year of current-year churn is + immaterial relative to irreplaceability — those weekly copies *are* the vintage + series, so pruning them destroys the artifact being built. Recorded deliberately + rather than left to be discovered. +2. **Run `--all` monthly or quarterly, not weekly, and not for backfill.** If closed-year + files are genuinely frozen, a `content_sha256` change on a closed year is *precisely* + the 2008-style retroactive-restatement signature. The point of `--all` is not to + collect bytes but to detect a change that should be impossible — the only automated + detector for the failure mode this whole subsystem exists to guard against. Weekly is + pointless (nothing older moves); monthly/quarterly is cheap because almost everything + 304s. + +### Real-network smoke (2026-07-30, pre-merge) + +Default path (`cotdata-vintage fetch` = current year × 3 reports + weekly static) run +against live cftc.gov into a throwaway store: all four URLs resolved, the UA was +accepted, four 200s retained (legacy 4,943,331 B; disagg 1,399,661 B; TFF 403,332 B; +weekly static 417,064 B), provenance recorded with `parse_status=pending`. The legacy sha +matched an independent manual download. A second run returned 304 on all four and +retained nothing. URL builders and headers are therefore verified live, not just in +fixtures. ## 3c. Decision: futures-only, `combined` is constant-False for now diff --git a/src/cotdata/vintage.py b/src/cotdata/vintage.py index 77032fe..b0f312e 100644 --- a/src/cotdata/vintage.py +++ b/src/cotdata/vintage.py @@ -120,6 +120,11 @@ def _http_get(url: str, *, etag: str | None, last_modified: str | None) -> HttpR """Conditional GET with If-None-Match / If-Modified-Since. Real network path. Injected as ``http_get`` in tests so the capture logic runs fully offline. + + MEASURED 2026-07-30: cftc.gov serves NO ETag on any of these files, so If-None-Match + can never fire — it is sent only in case that changes. If-Modified-Since does work + (verified 304 against both a static year and the current year), and is what makes an + --all sweep cheap: only the current and immediately-prior year actually transfer. """ import requests # local import: keeps the module importable without network deps From ab34f2d732bd8540052eeee1cafc83a0695defb9 Mon Sep 17 00:00:00 2001 From: Matt Spinola Date: Thu, 30 Jul 2026 17:45:08 -0400 Subject: [PATCH 07/16] feat(vintage): closed-year restatement tripwire + COTDATA_VINTAGE_ROOT escape hatch MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two things, one of which blocks scheduling capture on a replica. DEPLOYMENT HAZARD found while siting the scheduled job: the Mac replica is fed by robocopy /MIR, which deletes destination-only files and excludes only _cache/_raw/citpy. A vintage/ tree written on the Mac would therefore be DELETED by the next producer sync — the exact trap sync-store.cmd already documents for citpy, except vintage data is irreplaceable (CFTC serves current state only, so a deleted vintage never comes back). - COTDATA_VINTAGE_ROOT relocates the whole tree outside the mirrored store, so capture can run on a replica safely. Preferred placement remains the producer, where the tree syncs outward normally and matches the ADR-0007 cot-half seam. Adding `vintage` to /XD is explicitly NOT recommended: it holds only while every future invocation remembers the flag. - Restatement tripwire: a CLOSED year whose content sha changes is flagged restatement_suspect with a loud message. Closed years are frozen (measured), so this is the 2008-style retroactive-restatement signature, and it falls out of the ordinary dedupe path with no extra machinery. 2025 sits in the rolling regeneration window but is byte-frozen, so it should emit exactly one "unchanged bytes (deduped)" record per week indefinitely; the week it does not is the alert. January year-end finalization is called out in the message as the benign case. Co-Authored-By: Claude Opus 4.8 --- docs/design/cot_vintage.md | 30 ++++++++++++++++++++++++ src/cotdata/vintage.py | 39 ++++++++++++++++++++++++++++++- tests/test_vintage_capture.py | 43 +++++++++++++++++++++++++++++++++++ 3 files changed, 111 insertions(+), 1 deletion(-) diff --git a/docs/design/cot_vintage.md b/docs/design/cot_vintage.md index 3f95dbd..0f2b91b 100644 --- a/docs/design/cot_vintage.md +++ b/docs/design/cot_vintage.md @@ -124,6 +124,36 @@ the two that do transfer (2025 re-downloads weekly, ~8 MB, and dedupes away). pointless (nothing older moves); monthly/quarterly is cheap because almost everything 304s. +## 3d. DEPLOYMENT HAZARD: `robocopy /MIR` deletes a replica-local vintage tree + +**Read before scheduling capture anywhere.** + +This deployment syncs the store from one Windows producer to two read-only replicas +(Mac over SMB, Linux VPS over rsync). The Mac push is `robocopy /MIR`, which deletes +anything present at the destination and absent at the source, excluding only +`/XD _cache _raw citpy` and `/XF manifest.json`. **`vintage` is not excluded.** + +So a `vintage/` tree written on the Mac replica is **destroyed by the next producer +sync**. This is not hypothetical: `sync-store.cmd` already documents exactly this +outcome for `citpy` ("not written by any producer, so /MIR removes it and no producer +run brings it back"). The difference is that vintage data is **irreplaceable** — CFTC +serves current state only, so a deleted vintage cannot be re-fetched, ever. + +Two safe placements: + +1. **Run capture on the producer (preferred).** Vintage capture *is* a producer action + (it fetches from CFTC), so it belongs on the same machine as the COT half, and the + tree then syncs outward to replicas like any other store content. This also matches + the ADR-0007 seam: the CFTC producer owns the `cot` half. +2. **Run on a replica with `COTDATA_VINTAGE_ROOT` pointing outside the mirrored store.** + The override exists for this. The tree then survives `/MIR`, but does not propagate to + other machines — acceptable for a single-box capture, but it makes that box the sole + custodian of irreplaceable data, so it needs its own backup. + +Adding `vintage` to the `/XD` list is a third option and is NOT recommended: it protects +the tree only as long as every future sync invocation remembers the flag, and one +forgotten flag is unrecoverable. + ### Real-network smoke (2026-07-30, pre-merge) Default path (`cotdata-vintage fetch` = current year × 3 reports + weekly static) run diff --git a/src/cotdata/vintage.py b/src/cotdata/vintage.py index b0f312e..f6a4f4f 100644 --- a/src/cotdata/vintage.py +++ b/src/cotdata/vintage.py @@ -56,6 +56,22 @@ def user_agent() -> str: # ── Paths ─────────────────────────────────────────────────────────────────── def vintage_root() -> Path: + """Root of the vintage subtree. + + Defaults to ``$COTDATA_STORE/vintage``, but ``COTDATA_VINTAGE_ROOT`` overrides it, and + on a REPLICA that override is mandatory rather than cosmetic. + + The deployment syncs the store with ``robocopy /MIR``, which deletes anything at the + destination that is absent at the source, excluding only ``_cache _raw citpy``. A + vintage tree written on a replica is therefore destroyed by the next producer sync — + the same trap docs/SYNCING.md already documents for ``citpy``, except that vintage data + is IRREPLACEABLE by construction (CFTC serves current state only). Either run capture + on the producer, where the tree syncs outward normally, or point this override at a + path outside the mirrored store. + """ + override = os.environ.get("COTDATA_VINTAGE_ROOT", "").strip() + if override: + return Path(override) return config.store_root() / "vintage" @@ -318,10 +334,31 @@ def capture_source(source: Source, *, snapshots: list[dict], http_get, now: dt.d finally: if part.exists(): part.unlink() - rel = str(dest.relative_to(config.store_root())) + try: + rel = str(dest.relative_to(config.store_root())) + except ValueError: + rel = str(dest) # COTDATA_VINTAGE_ROOT points outside the store + + # ── Restatement tripwire ──────────────────────────────────────────────── + # A CLOSED year's content is frozen (measured: 2020-2024 static, and 2025 re-touched + # weekly by the rolling regeneration window but byte-identical since January). So a + # closed year whose sha CHANGES is the 2008-style retroactive-restatement signature — + # the failure mode this whole subsystem exists to detect — and it costs no extra + # machinery: it falls out of the ordinary dedupe path. 2025 in particular should + # produce exactly one "unchanged bytes (deduped)" record per week, indefinitely; the + # week it does not is the alert. + restatement_suspect = False + if prev is not None and source.report_year is not None and source.report_year < now.year: + restatement_suspect = True + print(f" *** RESTATEMENT SUSPECT: {source.report_type} {source.report_year} " + f"content changed (sha {(prev.get('content_sha256') or '')[:8]} -> {sha[:8]}). " + f"A closed year should be frozen. In January this may be ordinary year-end " + f"finalization; otherwise treat as a retroactive restatement and diff it.") + return { **base, "snapshot_id": _snapshot_id(retrieved_at, source.url, sha[:8]), + "restatement_suspect": restatement_suspect, "http_status": res.status, "http_etag": res.etag, "http_last_modified": res.last_modified, diff --git a/tests/test_vintage_capture.py b/tests/test_vintage_capture.py index 3738e26..a862996 100644 --- a/tests/test_vintage_capture.py +++ b/tests/test_vintage_capture.py @@ -234,6 +234,49 @@ def test_weekly_static_partitions_by_capture_year(store_env): assert len(raws) == 1 and raws[0].parent.name == "2026" # not "current" +def test_closed_year_sha_change_is_flagged_as_restatement_suspect(store_env): + """A closed year is frozen, so a content change is the retroactive-restatement + signature. It must be flagged, not silently retained as an ordinary new vintage.""" + from cotdata import vintage + url = "https://www.cftc.gov/files/dea/history/dea_fut_xls_2025.zip" + srcs = [vintage.Source("legacy", "annual_zip", "zip", url, 2025)] # closed year + http = _FakeHttp({url: [(200, _body(b"jan-final"), '"e1"', "lm1"), + (200, _body(b"RESTATED"), '"e2"', "lm2")]}) + now = _clock() # clock is in 2026, so 2025 is closed + r1 = vintage.fetch(sources=srcs, http_get=http, rate_limit_s=0, now_fn=now) + r2 = vintage.fetch(sources=srcs, http_get=http, rate_limit_s=0, now_fn=now) + + assert r1["records"][0]["restatement_suspect"] is False # first sighting, no prior + assert r2["records"][0]["restatement_suspect"] is True # frozen year changed + + +def test_current_year_change_is_not_a_restatement_suspect(store_env): + """The current year legitimately gains a report every week.""" + from cotdata import vintage + http = _FakeHttp(_only(LEGACY_2026, + (200, _body(b"week1"), '"e1"', "lm1"), + (200, _body(b"week2"), '"e2"', "lm2"))) + now = _clock() # 2026, and the source's report_year is 2026 + _fetch(vintage, http, now) + res2 = _fetch(vintage, http, now) + assert res2["records"][0]["restatement_suspect"] is False + + +def test_vintage_root_override_keeps_tree_outside_a_mirrored_store(store_env, tmp_path, monkeypatch): + """COTDATA_VINTAGE_ROOT must relocate the whole tree — the escape hatch for a replica + whose store is mirrored with robocopy /MIR (which would delete a store-local tree).""" + from cotdata import vintage + outside = tmp_path / "vintage_elsewhere" + monkeypatch.setenv("COTDATA_VINTAGE_ROOT", str(outside)) + http = _FakeHttp(_only(LEGACY_2026, (200, _body(b"week1"), '"e1"', "lm1"))) + res = _fetch(vintage, http, _clock()) + + assert res["new_files"] == 1 + assert (outside / "manifest.json").exists() + assert list(outside.rglob("*.zip")) + assert not (store_env / "vintage").exists() # nothing written into the synced store + + def test_changed_bytes_writes_second_immutable_snapshot(store_env): from cotdata import vintage http = _FakeHttp(_only( From eff935a5ca5003af6bdda232dbc283d762c02cf0 Mon Sep 17 00:00:00 2001 From: Matt Spinola Date: Thu, 30 Jul 2026 17:58:38 -0400 Subject: [PATCH 08/16] feat(vintage): producer-side capture + per-replica sync policy Sites the scheduled capture on the Windows producer and makes it propagate correctly. - run-vintage.cmd: fetch + ingest --pending, to chain after run-cot.cmd. Documents why it runs on the producer (capture is a producer action, and a replica-written tree is deleted by /MIR) and why the schedule is DAILY not weekly (almost everything 304s, so it is nearly free, and it catches holiday shifts and backlog publications while tightening observed_at from a 7-day to a 1-day bound). - RENAME vintage/manifest.json -> vintage/snapshots.json. Both sync scripts exclude "manifest.json" UNANCHORED (robocopy /XF and rsync --exclude both match by name at any depth), so the provenance index would have been silently stripped in transit and each replica would have received raw archives with no index. The rename means existing deployed sync scripts carry it correctly with no edit. - sync-store.cmd (Mac): carries vintage in full, deliberately. The bytes are irreplaceable and the Mac is the natural second copy (~1GB/yr). Adds *.tmp/*.part to /XF so a sync mid-capture cannot land a partial. - push-to-server.cmd (VPS): excludes vintage/ entirely. cot-analyzer reads prices and COT only, so the dash would carry ~1GB/yr of archives it never opens. Notes that excluding just vintage/raw/ is the right change if the dash ever consumes revisions. - SYNCING.md: new section on where vintage may be written, the per-replica table, and the naming gotcha, next to the existing citpy precedent. Co-Authored-By: Claude Opus 4.8 --- docs/SYNCING.md | 32 +++++++++++++++ docs/examples/windows/push-to-server.cmd | 13 ++++-- docs/examples/windows/run-vintage.cmd | 51 ++++++++++++++++++++++++ docs/examples/windows/sync-store.cmd | 10 ++++- src/cotdata/vintage.py | 15 +++++-- tests/test_vintage_capture.py | 4 +- 6 files changed, 116 insertions(+), 9 deletions(-) create mode 100644 docs/examples/windows/run-vintage.cmd diff --git a/docs/SYNCING.md b/docs/SYNCING.md index f3dbe26..917da22 100644 --- a/docs/SYNCING.md +++ b/docs/SYNCING.md @@ -158,6 +158,38 @@ arrival. The per-half files under `manifests/` are disjoint and merge correctly. Run `cotdata-update --migrate-manifests` once per store, then delete `manifest.json`. +### `vintage/` is irreplaceable, so where it is WRITTEN matters + +The vintage tree (`vintage/raw/`, `observations/`, `revisions/`, `snapshots.json`) records +CFTC data *as published*. CFTC serves current state only and there is no vintage archive, +so a deleted vintage snapshot can never be re-fetched. Treat it as write-once. + +**Capture on the producer, never on a replica.** Vintage capture fetches from CFTC, so it +is a producer action and belongs beside the COT half (`run-vintage.cmd`, chained after +`run-cot.cmd`, daily). Written on the producer it propagates outward like any other store +content. Written on a **replica** it is destroyed by the next `/MIR` or `--delete` pass, +for the same reason `citpy` is (below): the source has no such directory, so the mirror +removes it. `citpy` is regenerable; vintage data is not. + +If a replica genuinely must capture, point `COTDATA_VINTAGE_ROOT` at a path **outside** +the mirrored store. That box then holds the only copy, so give it its own backup. + +Per replica in this deployment: + +| Target | Carries `vintage/`? | Why | +|---|---|---| +| Mac (research) | **Yes, in full** | Natural second copy of irreplaceable bytes, ~1 GB/yr, and research may query revisions | +| Linux dash VPS | **No** | cot-analyzer reads prices and COT only; it would carry ~1 GB/yr of archives it never opens | + +**Naming gotcha, already handled:** the vintage provenance index is `snapshots.json`, not +`manifest.json`. Both sync scripts exclude `manifest.json` *unanchored* (robocopy `/XF` +and rsync `--exclude` both match by name at any depth), so a `vintage/manifest.json` would +have been stripped in transit and the replica would receive raw archives with no index. +Do not rename it back. + +Exclude `*.part` alongside `*.tmp`: raw downloads land via a `.part` file plus an atomic +replace, and a sync running mid-capture must not carry the partial. + ## Check before you mirror `--delete` and `/MIR` are silent when they destroy something. Run the preflight first: diff --git a/docs/examples/windows/push-to-server.cmd b/docs/examples/windows/push-to-server.cmd index b65c975..8e035b0 100644 --- a/docs/examples/windows/push-to-server.cmd +++ b/docs/examples/windows/push-to-server.cmd @@ -55,11 +55,18 @@ REM rides under _raw and so is excluded, per ADR-0006. REM citpy consumer-owned on the server; excluding it from --delete is REM what stops the mirror from wiping it. REM manifest.json legacy aggregate, resolved last-writer-wins across halves. -REM *.tmp a producer's partial-write temp (atomic write via os.replace); -REM never propagate a half-written file. +REM *.tmp, *.part a producer's partial-write temps (atomic write via os.replace, +REM and in-flight raw downloads); never propagate a half-written file. +REM vintage/ NOT pushed here, unlike the Mac sync. cot-analyzer reads prices +REM and COT, never the vintage tree, so the dash would carry roughly +REM 1 GB/year of raw CFTC archives it never opens. The Mac keeps the +REM second copy instead. Drop this exclusion if something on the dash +REM ever consumes revisions -- and if so consider excluding only +REM "vintage/raw/" so the small derived tables still ride along. "%RSYNC%" -az --delete ^ --exclude "_cache/" --exclude "_raw/" --exclude "citpy/" ^ - --exclude "manifest.json" --exclude "*.tmp" --exclude "manifests/" ^ + --exclude "manifest.json" --exclude "*.tmp" --exclude "*.part" ^ + --exclude "manifests/" --exclude "vintage/" ^ -e "%SSH%" "%SRC%/" "%DEST%/" if %ERRORLEVEL% NEQ 0 ( echo push FAILED, rsync code %ERRORLEVEL% & exit /b %ERRORLEVEL% ) diff --git a/docs/examples/windows/run-vintage.cmd b/docs/examples/windows/run-vintage.cmd new file mode 100644 index 0000000..7de44c3 --- /dev/null +++ b/docs/examples/windows/run-vintage.cmd @@ -0,0 +1,51 @@ +@echo off +REM cotdata VINTAGE capture wrapper for Windows Task Scheduler. +REM Copy this file into your scheduler folder and overwrite the two markers below. +REM Do NOT put angle brackets in a .cmd file: cmd reads them as redirection and +REM the file fails with "The syntax of the command is incorrect" even on comment +REM lines, which is why these are plain-text markers you replace. +REM REPLACE_WITH_STORE_PATH = your data store e.g. C:\Users\you\cotdata_store +REM REPLACE_WITH_VENV_PATH = your cotdata venv e.g. C:\Users\you\code\cotdata\.venv +REM +REM WHY THIS RUNS ON THE PRODUCER, not on a replica: vintage capture fetches from +REM CFTC, so it is a producer action and belongs beside the COT half. It also has to +REM be here for safety -- the replicas are fed by robocopy /MIR, which DELETES +REM destination-only files, so a vintage tree written on the Mac would be wiped by the +REM next sync. Written here, it propagates outward like any other store content. +REM (A replica that must capture anyway needs COTDATA_VINTAGE_ROOT pointing outside +REM the mirrored store. See docs/SYNCING.md.) +REM +REM SCHEDULE: DAILY, not weekly. Almost everything returns 304, so a daily run costs +REM close to nothing, and it buys three things a weekly run does not: +REM - holiday-shifted releases are caught with no schedule logic +REM - backlog catch-up publications are caught +REM - observed_at tightens from a 7-day bound to a 1-day bound, which directly +REM improves release-date quality (observed is the top of the precedence order) +REM ~17:00 ET puts capture within about ninety minutes of the 15:30 ET publication. +REM +REM ORDERING: run this AFTER run-cot.cmd, then chain the sync scripts after it, so a +REM single sync carries both the current-state update and the new vintage snapshot. +setlocal +set COTDATA_STORE=REPLACE_WITH_STORE_PATH +REM Optional: identify yourself to CFTC. Defaults to the repo URL if unset. +REM set COTDATA_USER_AGENT=cotdata-vintage/0.1 (+contact you@example.com) + +REM Default path = current year's three annual reports + the Legacy weekly static. +REM The weekly static is fetched for its HTTP Last-Modified, which is a true +REM publication timestamp rather than a polling-interval approximation. +"REPLACE_WITH_VENV_PATH\Scripts\cotdata-vintage.exe" fetch +if %ERRORLEVEL% NEQ 0 ( + echo vintage fetch FAILED with code %ERRORLEVEL% + exit /b %ERRORLEVEL% +) + +REM Parse whatever was just retained into change-only observations + revisions. +REM Safe to re-run: re-ingesting identical bytes writes zero rows. +"REPLACE_WITH_VENV_PATH\Scripts\cotdata-vintage.exe" ingest --pending +if %ERRORLEVEL% NEQ 0 ( + echo vintage ingest FAILED with code %ERRORLEVEL% + exit /b %ERRORLEVEL% +) + +echo vintage ok +exit /b 0 diff --git a/docs/examples/windows/sync-store.cmd b/docs/examples/windows/sync-store.cmd index 53da710..97cea07 100644 --- a/docs/examples/windows/sync-store.cmd +++ b/docs/examples/windows/sync-store.cmd @@ -21,9 +21,17 @@ REM and no producer run brings it back. Kept as a backstop: such REM files belong outside the store. See docs/SYNCING.md. REM /XF excludes the legacy aggregate manifest: nothing writes it, and it is the one REM file a sync would resolve last-writer-wins across two halves. +REM vintage/ IS carried here, deliberately and in full (including vintage/raw). Those +REM bytes are irreplaceable -- CFTC serves current state only, so a lost vintage cannot +REM be re-fetched -- and the Mac is the natural second copy. It costs roughly 1 GB/year. +REM Note the provenance index is vintage\snapshots.json, NOT manifest.json: /XF below +REM matches by NAME AT ANY DEPTH, so had it been called manifest.json this sync would +REM have silently delivered raw bytes with no index. +REM /XF also drops partial-write temps (*.tmp from parquet/JSON writes, *.part from +REM in-flight raw downloads) so a sync mid-capture never lands a truncated file. robocopy "REPLACE_WITH_STORE_PATH" "REPLACE_WITH_DEST_PATH" /MIR /R:2 /W:5 /NFL /NDL /NP ^ /XD _cache _raw citpy ^ - /XF manifest.json + /XF manifest.json *.tmp *.part REM robocopy uses exit codes 0-7 for SUCCESS (1 = files copied, 2 = extras present, REM 3 = both, and so on) and 8+ for failure. Task Scheduler treats any non-zero as a diff --git a/src/cotdata/vintage.py b/src/cotdata/vintage.py index f6a4f4f..425d38f 100644 --- a/src/cotdata/vintage.py +++ b/src/cotdata/vintage.py @@ -10,9 +10,9 @@ revisions/detected_year=YYYY/*.parquet append-only, field-level release_schedule.parquet announcements.parquet - manifest.json vintage provenance index + snapshots.json vintage provenance index -Provenance lives in its OWN ``vintage/manifest.json`` rather than a block in the cot-half +Provenance lives in its OWN ``vintage/snapshots.json`` rather than a block in the cot-half manifest, because ``store.reconcile_manifest`` ghost-prunes any manifest entry without a matching ``{name}.parquet`` — raw snapshot ids are not parquet files and would be wiped. A self-owned file also matches the repo's existing "one writer per manifest file" split. @@ -83,7 +83,16 @@ def raw_dir(source_kind: str, year: int | str) -> Path: def manifest_path() -> Path: - return vintage_root() / "manifest.json" + """The snapshot provenance index. + + Deliberately NOT named manifest.json. Both deployed sync scripts exclude that name + UNANCHORED — robocopy ``/XF manifest.json`` and rsync ``--exclude "manifest.json"`` + match at any depth — so a vintage/manifest.json would be silently stripped in transit, + delivering raw bytes to a replica with no index: the "directory of opaque blobs" + failure this module works hardest to prevent. The distinct name means existing sync + scripts need no edit to carry it correctly. + """ + return vintage_root() / "snapshots.json" # ── Sources ───────────────────────────────────────────────────────────────── diff --git a/tests/test_vintage_capture.py b/tests/test_vintage_capture.py index a862996..0f9d309 100644 --- a/tests/test_vintage_capture.py +++ b/tests/test_vintage_capture.py @@ -137,7 +137,7 @@ def test_corrupt_manifest_raises_and_quarantines_rather_than_overwriting(store_e with pytest.raises(vintage.CorruptManifestError, match="unreadable"): vintage.read_snapshots() # original moved aside, not destroyed; no empty manifest left in its place - quarantined = list(mpath.parent.glob("manifest.json.corrupt.*")) + quarantined = list(mpath.parent.glob("snapshots.json.corrupt.*")) assert len(quarantined) == 1 assert quarantined[0].read_text() == "{ this is not json" assert not mpath.exists() @@ -272,7 +272,7 @@ def test_vintage_root_override_keeps_tree_outside_a_mirrored_store(store_env, tm res = _fetch(vintage, http, _clock()) assert res["new_files"] == 1 - assert (outside / "manifest.json").exists() + assert (outside / "snapshots.json").exists() assert list(outside.rglob("*.zip")) assert not (store_env / "vintage").exists() # nothing written into the synced store From 71a9e3a8ded8eb2af2cd2870e49083d9edc73a9b Mon Sep 17 00:00:00 2001 From: Matt Spinola Date: Thu, 30 Jul 2026 19:05:59 -0400 Subject: [PATCH 09/16] docs(readme): document the vintage subsystem MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The README was untouched by the previous commits while the branch added two CLI entry points, two env vars, a new store subtree and a new scheduled task — all categories the README already documents. - New "COT vintage tracking (as-published history)" section under Concepts & design: why revisions matter, the full CLI, how change-only storage / PIT reads / release-date provenance work, the producer-vs-replica placement rule, and why --all is a restatement tripwire rather than a backfill. - Store layout gains vintage/, flagged opt-in and additive. - Producer command list gains cotdata-vintage fetch; Contents gains the section link. - Sync section notes the irreplaceability rule and the snapshots.json naming, since the usual manifest.json exclusion matches at any depth. - WINDOWS_SCHEDULING.md: run-vintage.cmd wrapper, an optional fourth schtasks entry at 17:00, and the daily-not-weekly rationale. Corrected "Create three tasks", which my addition had made wrong. Co-Authored-By: Claude Opus 4.8 --- README.md | 69 +++++++++++++++++++++++++++++++++++++- docs/WINDOWS_SCHEDULING.md | 20 ++++++++++- 2 files changed, 87 insertions(+), 2 deletions(-) diff --git a/README.md b/README.md index cc8d532..0e6a7ef 100644 --- a/README.md +++ b/README.md @@ -28,7 +28,7 @@ cotdata separates *fetching* data (a "producer" that talks to vendors) from *usi ## Contents -- [Quickstart](#quickstart) · [How it works](#how-it-works) · [Reading data](#reading-data-consumer) · [Producing data](#producing-data-producer) · [Windows setup](docs/WINDOWS_SETUP.md) · [Scheduling on Windows](docs/WINDOWS_SCHEDULING.md) · [Scheduling on Linux](docs/LINUX_SCHEDULING.md) · [Syncing the store](docs/SYNCING.md) · [Operations](#operations) · [Concepts & design](#concepts--design) · [Reference: schemas](#reference-data-schemas) · [Reference: COT formats](#reference-cot-formats-explained) · [Diagnostics](#diagnostics) · [Development](#development) · [Contributing](#contributing) · [License](#license) +- [Quickstart](#quickstart) · [How it works](#how-it-works) · [Reading data](#reading-data-consumer) · [Producing data](#producing-data-producer) · [Windows setup](docs/WINDOWS_SETUP.md) · [Scheduling on Windows](docs/WINDOWS_SCHEDULING.md) · [Scheduling on Linux](docs/LINUX_SCHEDULING.md) · [Syncing the store](docs/SYNCING.md) · [Operations](#operations) · [Concepts & design](#concepts--design) · [COT vintage tracking](#cot-vintage-tracking-as-published-history) · [Reference: schemas](#reference-data-schemas) · [Reference: COT formats](#reference-cot-formats-explained) · [Diagnostics](#diagnostics) · [Development](#development) · [Contributing](#contributing) · [License](#license) ## Quickstart @@ -85,6 +85,9 @@ The store layout: - `metadata/contract_specs.parquet` — Norgate contract specifications (tick size, point value, margin). - `manifest.json` — per-table `last_date`, `n_rows`, `source`, `updated_at`, `schema_version`. - `status.json` — machine-readable new-data signal for downstream tools (see [Operations](#operations)). +- `vintage/` — optional as-published (vintage) capture: retained raw CFTC downloads plus + change-only observations and field-level revisions. Purely additive; the tables above are + unchanged whether or not it is enabled. See [COT vintage tracking](#cot-vintage-tracking-as-published-history). ## Reading data (consumer) @@ -128,6 +131,7 @@ COTDATA_STORE=/store cotdata-update --cot-legacy # CFTC Legacy ( COTDATA_STORE=/store cotdata-update --cot-disagg # CFTC Disaggregated (any OS) COTDATA_STORE=/store cotdata-update --cot-tff # CFTC Traders in Financial Futures (any OS) COTDATA_STORE=/store cotdata-update --cot-all # all three CFTC COT reports +COTDATA_STORE=/store cotdata-vintage fetch # optional: capture as-published COT (any OS) ``` `--prices` with no `--symbols` updates every symbol in the registry; add `--symbols` to scope it. Each run prints a per-symbol line with the date advance (e.g. `ES: … [2026-07-13 -> 2026-07-14]`) and a summary footer (OK/failed counts, rows written, elapsed, newest date). A run **exits non-zero** if a fetch hard-fails (Norgate/CFTC unreachable), so a scheduler can retry — see [Scheduling on Windows](#scheduling-on-windows-task-scheduler). @@ -226,6 +230,12 @@ Anything a consumer put in the store by hand is a **correctness** issue rather t saving. No producer creates it, so a mirroring sync deletes it. Exclude it, but the real fix is to keep it out of the store: the store belongs to its producer. +The same rule bites hardest on `vintage/` if you enable it, because that data cannot be +re-fetched: capture it on the **producer** so it syncs outward, or keep it outside the +mirrored store with `COTDATA_VINTAGE_ROOT`. Its provenance index is deliberately named +`snapshots.json`, since the usual `manifest.json` exclusion matches by name at any depth +and would otherwise strip it in transit, delivering raw archives with no index. + Consumer cloud sync (Dropbox, Google Drive) is a poor fit here: conflict copies land inside the store, and on-demand placeholder files break `read_parquet` on the machine doing research. @@ -311,6 +321,63 @@ The supported futures contracts are defined in a YAML registry, so adding a mark The store uses **atomic writes** (write-temp-then-rename). Consumers can safely query via `get_prices` / `get_cot` even while `cotdata-update` is actively downloading and writing. +### COT vintage tracking (as-published history) + +CFTC revises COT data after publication — most consequentially through **trader +reclassification**, which moves positions between categories retroactively. Because +downstream signals are rolling z-scores and percentiles against years of history, a +restatement silently rewrites the baseline every historical reading was computed against. +There is precedent: in July 2008 the Commission revised reports back to July 3, 2007. + +**CFTC serves current state only.** There is no vintage archive and no as-published +endpoint, so vintage data can only be accumulated going forward — every uncaptured week is +a permanent blind spot in the part of the series most likely to have been revised. + +This is **opt-in and purely additive**: if you never run it, the store behaves exactly as +before. Enabling it adds a `vintage/` subtree. + +```bash +cotdata-vintage fetch # capture current year + weekly static (daily) +cotdata-vintage fetch --all # every year 1986-present (see below) +cotdata-vintage ingest --pending # parse retained raw -> observations + revisions +cotdata-vintage diff --since 2026-01-01 # field-level revisions, with revision depth +cotdata-vintage asof --as-of 2026-07-24T18:00:00 --report-date 2026-07-21 +cotdata-schedule sync # CFTC Special Announcements +cotdata-schedule backfill # resolve release_date + its provenance +``` + +How it works: + +- **Immutable landing zone.** Every fetch is recorded (including 304s) and raw bytes are + retained permanently under `vintage/raw/`, written atomically and never rewritten. A + byte-identical regeneration is deduped — a changed download is not itself a revision. +- **Change-only observations.** A row is written only when its value hash differs from the + latest for its natural key `(report_date, market_code, report_type, combined, category)`, + so storage grows with actual revisions rather than with time. +- **Field-level revisions** carry `age_days` (revision depth): whether revisions stay in + recent weeks or reach back into the calibration window determines how much the rest of + a system has to care. +- **Point-in-time reads.** `asof(t)` returns each key's latest value observed at or before + `t`, reconstructing what was actually knowable then. +- **Release dates with provenance.** `report_date` is stored exactly as reported (never + normalized to Tuesday), and `release_date` is resolved through + `published > observed > announced > scheduled > derived`, with the source recorded — a + release date without provenance is worse than none, since indexing on `report_date` + embeds a lookahead (three days normally, weeks during a backlog). + +**Run capture on the producer, not a replica**, and schedule it **daily**: nearly every +request returns 304, so a daily run is close to free while catching holiday-shifted and +backlog releases with no schedule logic. `--all` is a **restatement tripwire** rather than +a backfill — closed years are byte-frozen, so a checksum change on one is the retroactive- +restatement signature; monthly or quarterly is the right cadence, and it is cheap because +almost everything 304s. Full design notes, including the measured CFTC caching behaviour, +are in [docs/design/cot_vintage.md](docs/design/cot_vintage.md). + +> **Replica warning.** The vintage tree must not be written on a machine whose store is +> mirrored (`robocopy /MIR`, `rsync --delete`) from a producer: the mirror deletes +> destination-only files and the data is irreplaceable. Capture on the producer, or set +> `COTDATA_VINTAGE_ROOT` to a path outside the mirrored store. See [docs/SYNCING.md](docs/SYNCING.md). + ## Local development ```bash diff --git a/docs/WINDOWS_SCHEDULING.md b/docs/WINDOWS_SCHEDULING.md index 4bcd1be..9484f3e 100644 --- a/docs/WINDOWS_SCHEDULING.md +++ b/docs/WINDOWS_SCHEDULING.md @@ -35,9 +35,22 @@ set COTDATA_STORE=REPLACE_WITH_STORE_PATH Using the full venv `\Scripts\cotdata-prices.exe` / `\Scripts\cotdata-cot.exe` path (rather than relying on the command being on `PATH`) matters here: Task Scheduler runs with a different, often bare, environment than your interactive shell, so a bare command name that resolves fine in Command Prompt can fail to resolve under the scheduler. +`run-vintage.cmd` — **optional**, the as-published (vintage) capture. Copy +[`docs/examples/windows/run-vintage.cmd`](examples/windows/run-vintage.cmd); it runs +`cotdata-vintage fetch` then `ingest --pending`, both exit-code guarded. Two things to know +before enabling it: + +- **It belongs on the producer.** Capture fetches from CFTC, so it is a producer action — + and a vintage tree written on a mirrored replica is deleted by the next sync, which is + unrecoverable because CFTC serves current state only. See [SYNCING.md](SYNCING.md). +- **Schedule it daily, not weekly.** Nearly every request returns 304, so a daily run costs + almost nothing while catching holiday-shifted and backlog releases with no schedule logic. + + + ## Creating the tasks -Create three tasks — times are the **machine's local** time; convert from ET if it isn't on Eastern: +Create three tasks (plus an optional fourth if you enable vintage capture) — times are the **machine's local** time; convert from ET if it isn't on Eastern: ```bat :: 1) Prices — fire at the Continuous Futures Final (~8:55pm ET); --require-final + restart @@ -46,6 +59,11 @@ schtasks /Create /TN "cotdata prices" /TR "\run-prices.cmd" /SC DAILY /ST 2 :: 2) COT — daily morning catch-up for holiday-delayed releases and as a safety net schtasks /Create /TN "cotdata COT (catch-up)" /TR "\run-cot.cmd" /SC DAILY /ST 08:10 + +:: 3) Vintage (OPTIONAL) — as-published capture, ~90 min after the 15:30 ET release. +:: Daily is deliberate: almost every request 304s, so it is nearly free, and it +:: tightens the observed release date from a 7-day bound to a 1-day one. +schtasks /Create /TN "cotdata vintage" /TR "\run-vintage.cmd" /SC DAILY /ST 17:00 ``` > **Substitute `` before running these** — with the real folder holding your `.cmd` files, e.g. `C:\Users\you\code\cotdata\scheduler`. `schtasks` takes the quoted `/TR` value as a literal string and **does not check the file exists**, so a leftover `"\run-cot.cmd"` is accepted without error and creates a task that fails only when it fires. Verify each task points somewhere real: From d4ee8ff14870cd46aa52e5e8aefbafd710e0063b Mon Sep 17 00:00:00 2001 From: Matt Spinola Date: Thu, 30 Jul 2026 19:38:27 -0400 Subject: [PATCH 10/16] feat(vintage): wire `published` release dates from the weekly static MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit §10 deferred this on the assumption it needed the weekly static parsed. Measuring the file showed otherwise: it is a headerless positional CSV covering exactly ONE report date (365 rows, 129 columns, a single distinct value in field 2), so mapping a snapshot to its report date reads one field. The publication timestamp was already being captured into snapshots.json. The whole thing was ~30 lines, not a canonicaliser, so it ships. - published_from_snapshots(): retained weekly-static file -> report_date (field 2), its HTTP Last-Modified -> release_date converted to ET (CFTC publishes on ET, so the conversion decides the date for any post-midnight-UTC timestamp). - backfill folds published rows in automatically on the production path, so plain `cotdata-schedule backfill` picks up true publication times with no extra step; tests still pass an explicit schedule to isolate precedence. - _schedule_map now ranks published > announced > scheduled rather than special-casing announced-beats-scheduled. - New `cotdata-schedule published`, and run-vintage.cmd gains published + backfill after ingest (both idempotent). Verified live: report date 2026-07-21 resolves to a 2026-07-24 ET publication date, source=published, beating the poll-derived observed bound. Forward-only by nature: the weekly static holds one week and is overwritten, so this covers weeks from first capture onward. Full canonicalisation of the weekly static INTO observations is genuinely larger and remains out of scope. Co-Authored-By: Claude Opus 4.8 --- README.md | 6 +- docs/design/cot_vintage.md | 22 ++++- docs/examples/windows/run-vintage.cmd | 16 ++++ src/cotdata/vintage_cli.py | 11 +++ src/cotdata/vintage_schedule.py | 120 ++++++++++++++++++++++++-- tests/test_vintage_schedule.py | 71 +++++++++++++++ 6 files changed, 233 insertions(+), 13 deletions(-) diff --git a/README.md b/README.md index 0e6a7ef..e95e7a4 100644 --- a/README.md +++ b/README.md @@ -343,6 +343,7 @@ cotdata-vintage ingest --pending # parse retained raw -> observations cotdata-vintage diff --since 2026-01-01 # field-level revisions, with revision depth cotdata-vintage asof --as-of 2026-07-24T18:00:00 --report-date 2026-07-21 cotdata-schedule sync # CFTC Special Announcements +cotdata-schedule published # true publication dates from retained weekly statics cotdata-schedule backfill # resolve release_date + its provenance ``` @@ -363,7 +364,10 @@ How it works: normalized to Tuesday), and `release_date` is resolved through `published > observed > announced > scheduled > derived`, with the source recorded — a release date without provenance is worse than none, since indexing on `report_date` - embeds a lookahead (three days normally, weeks during a backlog). + embeds a lookahead (three days normally, weeks during a backlog). `published` is the + weekly static's HTTP `Last-Modified`, a true publication timestamp; it is forward-only + (that file holds one week and is overwritten), so weeks predating capture fall back + down the chain. **Run capture on the producer, not a replica**, and schedule it **daily**: nearly every request returns 304, so a daily run is close to free while catching holiday-shifted and diff --git a/docs/design/cot_vintage.md b/docs/design/cot_vintage.md index 0f2b91b..af5d81c 100644 --- a/docs/design/cot_vintage.md +++ b/docs/design/cot_vintage.md @@ -232,9 +232,25 @@ nonreportable`; Disagg: `producer_merchant/swap/managed_money/other_reportable`; In: raw snapshot capture; change-only observation writes; field-level revisions; release schedule + announcements ingest + `release_date`/`release_date_source` -backfill; §8 tests; `current/` byte-identical. Deferred: `vintage stats`, -category-migration detection, tombstone *logic* (column present), weekly-static -fetching beyond the spike. +backfill; §8 tests; `current/` byte-identical. + +**`published` shipped too, contrary to the original deferral.** §10 deferred +weekly-static work on the assumption it required parsing that file. Measurement showed +otherwise: the weekly static is a headerless positional CSV covering exactly ONE report +date (365 rows, 129 columns, a single distinct value in field 2), so mapping a snapshot +to its report date reads one field, and its publication timestamp was already being +captured into `snapshots.json`. Deriving `published` cost ~30 lines rather than a +canonicaliser, so it landed. Verified live: report date 2026-07-21 resolves to a +2026-07-24 ET publication date. (Full canonicalisation of the weekly static *into +observations* remains genuinely larger and is still out of scope.) + +`published` is forward-only by nature — the weekly static holds one week and is +overwritten — so it covers weeks from the first capture onward; historical weeks stay on +`announced` / `scheduled` / `derived`. + +Still deferred: `vintage stats` (needs a quarter of data to measure), +category-migration detection (needs revisions to exist), tombstone *logic* (column +present, needs a real disappearing key to design against). ## Bottom line diff --git a/docs/examples/windows/run-vintage.cmd b/docs/examples/windows/run-vintage.cmd index 7de44c3..3572006 100644 --- a/docs/examples/windows/run-vintage.cmd +++ b/docs/examples/windows/run-vintage.cmd @@ -47,5 +47,21 @@ if %ERRORLEVEL% NEQ 0 ( exit /b %ERRORLEVEL% ) +REM Resolve release dates. `published` reads the true publication timestamp out of the +REM weekly static just captured (its HTTP Last-Modified), which beats a poll-derived +REM `observed` bound; backfill then applies the precedence across all observations. +REM Both are idempotent, so re-running is a cheap no-op. +"REPLACE_WITH_VENV_PATH\Scripts\cotdata-schedule.exe" published +if %ERRORLEVEL% NEQ 0 ( + echo schedule published FAILED with code %ERRORLEVEL% + exit /b %ERRORLEVEL% +) + +"REPLACE_WITH_VENV_PATH\Scripts\cotdata-schedule.exe" backfill +if %ERRORLEVEL% NEQ 0 ( + echo schedule backfill FAILED with code %ERRORLEVEL% + exit /b %ERRORLEVEL% +) + echo vintage ok exit /b 0 diff --git a/src/cotdata/vintage_cli.py b/src/cotdata/vintage_cli.py index 4cc3d3c..92037f9 100644 --- a/src/cotdata/vintage_cli.py +++ b/src/cotdata/vintage_cli.py @@ -135,6 +135,14 @@ def _cmd_sched_sync(args) -> int: return 0 +def _cmd_sched_published(args) -> int: + from . import vintage_schedule + res = vintage_schedule.sync_published() + print(f"schedule published: {res['published']} week(s) resolved from retained " + f"weekly-static Last-Modified headers (true publication timestamps).") + return 0 + + def _cmd_sched_backfill(args) -> int: from . import vintage_schedule counts = vintage_schedule.backfill() @@ -153,6 +161,9 @@ def main_schedule(argv=None) -> int: sub = p.add_subparsers(dest="cmd", required=True) s = sub.add_parser("sync", help="Scrape Special Announcements into the store.") s.set_defaults(func=_cmd_sched_sync) + pub = sub.add_parser("published", + help="Derive true publication dates from retained weekly statics.") + pub.set_defaults(func=_cmd_sched_published) b = sub.add_parser("backfill", help="Resolve release_date/source across all observations.") b.set_defaults(func=_cmd_sched_backfill) args = p.parse_args(argv) diff --git a/src/cotdata/vintage_schedule.py b/src/cotdata/vintage_schedule.py index a944d49..37cc8f3 100644 --- a/src/cotdata/vintage_schedule.py +++ b/src/cotdata/vintage_schedule.py @@ -19,21 +19,22 @@ """ from __future__ import annotations +import csv import datetime as dt +from email.utils import parsedate_to_datetime from pathlib import Path +from zoneinfo import ZoneInfo import pandas as pd -from . import vintage +from . import config, vintage from . import vintage_ingest as vi # `published` outranks `observed`: it is the weekly-static HTTP Last-Modified, a TRUE # publication timestamp (spike 2026-07-30), whereas `observed` is only the first time WE # saw the report — accurate to the polling interval. They must not share a bucket, so it # stays possible to tell later which weeks carry a real publication time. -# NOTE: nothing populates `published` yet — mapping a weekly static to its report_date -# requires parsing that file, which handoff §10 defers. The slot and precedence are here -# so the wiring lands without a taxonomy migration. +# Populated by published_from_snapshots() below, folded into backfill automatically. PRECEDENCE = ("published", "observed", "announced", "scheduled", "derived", "unknown") @@ -98,16 +99,109 @@ def write_announcements(df: pd.DataFrame) -> None: _write(announcements_path(), df) +# ── `published`: true publication time from the weekly static ─────────────── +# The weekly static is a headerless positional CSV covering exactly ONE report date +# (measured 2026-07-30: 365 rows, 129 columns, a single distinct value in field 2), so +# mapping a retained snapshot to its report_date reads one field rather than parsing the +# file. Its HTTP Last-Modified is a true publication timestamp, which is what makes this +# strictly better than `observed` (accurate only to the polling interval). +_WEEKLY_STATIC_DATE_FIELD = 2 +PUBLISH_TZ = "America/New_York" # CFTC publishes on ET; convert before taking a date + + +def report_date_of_weekly_static(path) -> dt.date | None: + """The single report_date a retained weekly-static file covers, or None.""" + try: + with open(path, newline="") as fh: + for row in csv.reader(fh): + if len(row) > _WEEKLY_STATIC_DATE_FIELD: + try: + return pd.Timestamp(row[_WEEKLY_STATIC_DATE_FIELD]).date() + except (ValueError, TypeError): + return None + except OSError: + return None + return None + + +def _last_modified_to_release_date(lm: str) -> dt.date | None: + """RFC 2822 Last-Modified -> publication DATE in ET.""" + try: + ts = parsedate_to_datetime(lm) + except (TypeError, ValueError): + return None + if ts is None: + return None + if ts.tzinfo is None: + ts = ts.replace(tzinfo=dt.timezone.utc) + return ts.astimezone(ZoneInfo(PUBLISH_TZ)).date() + + +def published_from_snapshots(snapshots=None, *, store_root=None) -> pd.DataFrame: + """Derive `published` release-schedule rows from retained weekly-static snapshots. + + Forward-only by nature: the weekly static holds one week and is overwritten, so this + covers weeks captured from the first run onward and can never reach back. Historical + weeks stay on announced/scheduled/derived. + """ + from . import vintage + if snapshots is None: + snapshots = vintage.read_snapshots() + root = Path(store_root) if store_root else config.store_root() + + rows, seen = [], set() + for s in snapshots: + if s.get("source_kind") != "weekly_static": + continue + lp, lm = s.get("local_path"), s.get("http_last_modified") + if not lp or not lm: + continue + p = Path(lp) + if not p.is_absolute(): + p = root / lp + rd = report_date_of_weekly_static(p) + rel = _last_modified_to_release_date(lm) + if rd is None or rel is None or rd in seen: + continue + seen.add(rd) + rows.append({ + "report_date": pd.Timestamp(rd), + "release_date": pd.Timestamp(rel), + "source": "published", + "note": f"weekly-static Last-Modified ({lm})", + "ingested_at": pd.Timestamp(s.get("retrieved_at") or pd.Timestamp.now("UTC")), + }) + if not rows: + return pd.DataFrame(columns=["report_date", "release_date", "source", "note", + "ingested_at"]) + return pd.DataFrame(rows) + + +def sync_published() -> dict: + """Merge derived `published` rows into release_schedule.parquet. Idempotent.""" + derived = published_from_snapshots() + if derived.empty: + return {"published": 0} + existing = read_release_schedule() + merged = pd.concat([existing, derived], ignore_index=True) if not existing.empty else derived + merged = merged.drop_duplicates(subset=["report_date", "source"], keep="last") + write_release_schedule(merged) + return {"published": len(derived)} + + # ── Backfill ──────────────────────────────────────────────────────────────── +_SOURCE_RANK = {"published": 3, "announced": 2, "scheduled": 1} + + def _schedule_map(schedule: pd.DataFrame) -> dict: - """report_date -> (release_date, source) from the schedule table. `announced` rows - win over `scheduled` rows for the same report_date.""" + """report_date -> (release_date, source) from the schedule table, keeping the + highest-precedence source per date (published > announced > scheduled).""" out: dict = {} for _, r in schedule.iterrows(): rd = pd.Timestamp(r["report_date"]).normalize() src = r.get("source", "scheduled") prev = out.get(rd) - if prev is None or (prev[1] == "scheduled" and src == "announced"): + if prev is None or _SOURCE_RANK.get(src, 0) > _SOURCE_RANK.get(prev[1], 0): out[rd] = (pd.Timestamp(r["release_date"]).date(), src) return out @@ -122,7 +216,13 @@ def backfill(*, schedule: pd.DataFrame | None = None, bulk-ingested long afterwards (a bulk ingest's observed_at is not a release date). """ if schedule is None: - schedule = read_release_schedule() + # Production path: fold in `published` rows derived from retained weekly statics, + # so a plain `cotdata-schedule backfill` picks up true publication timestamps + # without a separate step. Tests pass an explicit schedule to isolate precedence. + stored = read_release_schedule() + derived = published_from_snapshots() + parts = [d for d in (stored, derived) if not d.empty] + schedule = pd.concat(parts, ignore_index=True) if parts else stored smap = _schedule_map(schedule) obs_dir = vi._obs_dir() @@ -155,10 +255,12 @@ def backfill(*, schedule: pd.DataFrame | None = None, if 0 <= delta <= observed_window_days: observed = oa sched = smap.get(rd) + published = sched[0] if sched and sched[1] == "published" else None announced = sched[0] if sched and sched[1] == "announced" else None scheduled = sched[0] if sched and sched[1] == "scheduled" else None rdate, src = resolve_release_date( - rd, observed=observed, announced=announced, scheduled=scheduled) + rd, published=published, observed=observed, + announced=announced, scheduled=scheduled) rel_dates.append(pd.Timestamp(rdate) if rdate is not None else pd.NaT) rel_srcs.append(src) counts[src] += 1 diff --git a/tests/test_vintage_schedule.py b/tests/test_vintage_schedule.py index 0997d59..5fa8377 100644 --- a/tests/test_vintage_schedule.py +++ b/tests/test_vintage_schedule.py @@ -157,6 +157,77 @@ def test_published_outranks_observed(): assert src == "published" and val == dt.date(2026, 7, 24) +def _weekly_static_bytes(report_date="2026-07-21"): + """A headerless positional CSV shaped like deafut.txt: field[2] is the ISO report + date, identical on every row (measured: one distinct date per file).""" + rows = [f'"WHEAT-SRW - CHICAGO BOARD OF TRADE",260721,{report_date},001602,CBT ,00,001 ,455433', + f'"GOLD - COMMODITY EXCHANGE INC.",260721,{report_date},088691,CMX ,00,001 ,500000'] + return ("\n".join(rows) + "\n").encode() + + +def test_report_date_read_from_one_field(store_env, tmp_path): + from cotdata.vintage_schedule import report_date_of_weekly_static + p = tmp_path / "wk.txt" + p.write_bytes(_weekly_static_bytes("2026-07-21")) + assert report_date_of_weekly_static(p) == dt.date(2026, 7, 21) + + +def test_last_modified_converts_to_et_publication_date(): + from cotdata.vintage_schedule import _last_modified_to_release_date + # 19:27:59 UTC on a Friday = 15:27 ET the SAME day (the ~15:30 ET release) + assert _last_modified_to_release_date("Fri, 24 Jul 2026 19:27:59 GMT") == dt.date(2026, 7, 24) + # a UTC timestamp after midnight belongs to the PREVIOUS ET day + assert _last_modified_to_release_date("Sat, 25 Jul 2026 02:00:00 GMT") == dt.date(2026, 7, 24) + assert _last_modified_to_release_date("not a date") is None + + +def test_published_beats_observed_end_to_end(store_env): + """A retained weekly static gives a true publication date that outranks the + poll-derived `observed` bound.""" + from cotdata import vintage + from cotdata import vintage_ingest as vi + from cotdata import vintage_schedule as vs + + # capture a weekly static whose Last-Modified is the real publication moment + lm = "Fri, 24 Jul 2026 19:27:59 GMT" + + def http(url, *, etag=None, last_modified=None): + from cotdata.vintage import HttpResult + return HttpResult(200, _weekly_static_bytes("2026-07-21") + b"\x00" * 2048, + etag='"e"', last_modified=lm) + + vintage.fetch(sources=[vintage.WEEKLY_STATIC], http_get=http, rate_limit_s=0, + now_fn=lambda: dt.datetime(2026, 7, 27, 21, tzinfo=dt.timezone.utc)) + # and an observation captured 3 days after the report date (would resolve `observed`) + _ingest_one("2026-07-21", observed_at=dt.datetime(2026, 7, 24, 21, tzinfo=dt.timezone.utc)) + + derived = vs.published_from_snapshots() + assert len(derived) == 1 + assert pd.Timestamp(derived.iloc[0]["report_date"]).date() == dt.date(2026, 7, 21) + assert pd.Timestamp(derived.iloc[0]["release_date"]).date() == dt.date(2026, 7, 24) + + counts = vs.backfill() # production path folds published in automatically + obs = vi.read_observations() + assert set(obs["release_date_source"]) == {"published"} + assert counts["published"] == len(obs) and counts["observed"] == 0 + + +def test_sync_published_is_idempotent(store_env): + from cotdata import vintage + from cotdata import vintage_schedule as vs + + def http(url, *, etag=None, last_modified=None): + from cotdata.vintage import HttpResult + return HttpResult(200, _weekly_static_bytes() + b"\x00" * 2048, etag='"e"', + last_modified="Fri, 24 Jul 2026 19:27:59 GMT") + + vintage.fetch(sources=[vintage.WEEKLY_STATIC], http_get=http, rate_limit_s=0, + now_fn=lambda: dt.datetime(2026, 7, 27, 21, tzinfo=dt.timezone.utc)) + assert vs.sync_published() == {"published": 1} + vs.sync_published() + assert len(vs.read_release_schedule()) == 1 # no duplicate row + + def test_announcement_parse_is_best_effort(): from cotdata.vintage_schedule import _parse_announcements html = "
    • January 5, 2026: revised gold report
    " From fe7f64726ea014f47e3ec895c9e393781f4fad90 Mon Sep 17 00:00:00 2001 From: Matt Spinola Date: Thu, 30 Jul 2026 19:50:23 -0400 Subject: [PATCH 11/16] docs: add crowdmon-futures module design and the COT vintage handoff spec MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Committed verbatim as authored, before any amendment, so the specs-as-written are a distinct point in history. The v0.2 handoff is the document the vintage subsystem was built against; the module design is its parent (the handoff implements §5.3). Both predate implementation and are amended in the following commit with what the build actually established. Co-Authored-By: Claude Opus 4.8 --- docs/design/cot_vintage_store_handoff.md | 293 +++++++++++++++ docs/design/crowdmon_futures_cot_module.md | 402 +++++++++++++++++++++ 2 files changed, 695 insertions(+) create mode 100644 docs/design/cot_vintage_store_handoff.md create mode 100644 docs/design/crowdmon_futures_cot_module.md diff --git a/docs/design/cot_vintage_store_handoff.md b/docs/design/cot_vintage_store_handoff.md new file mode 100644 index 0000000..c0ca326 --- /dev/null +++ b/docs/design/cot_vintage_store_handoff.md @@ -0,0 +1,293 @@ +# Handoff: COT Vintage Store & Revision Tracking + +**Target:** Claude Code session working in `trading_workspace/cotdata` +**Status:** step 1 of the `crowdmon-futures` build (see `crowdmon_futures_cot_module.md` §5.3) +**Version:** v0.2 — amended after repo review +**Estimated scope:** ~400–500 LOC including tests + +--- + +## Changelog v0.1 → v0.2 + +Amended in response to the implementing agent's review. Three substantive changes: + +1. **No database.** v0.1 specified DuckDB/SQLite. That was convenience, not requirement, and it cut against the repo's deliberate Parquet + JSON manifest contract. Persistence is now Parquet throughout (§3). +2. **`release_date` is no longer assumed available.** It now carries an explicit provenance flag, with a resolution order and a cheap spike that may collapse the whole taxonomy (§4.6). +3. **Scope repriced** after git-history recovery returned no historical vintages. Analytics deferred; release-date backfill promoted (§10). + +--- + +## 0. Read this first + +You have the repo; the author of this spec does not. Everything below describes intended behaviour, not existing structure. **Do not assume the described modules exist or that the described layout matches what's there.** Adapt the design to what's actually in the repo rather than restructuring the repo to match this document. + +**Non-goals.** Do not refactor existing fetch/parse logic beyond what's needed to add snapshotting. Do not change existing public APIs. Do not add analytics, metrics, or plotting. This task adds a provenance layer underneath what already works. + +**Compatibility requirement.** The existing current-state Parquet output must remain byte-identical. The vintage layer is purely additive alongside it. Existing consumers see no change. + +--- + +## 1. Problem and motivation + +CFTC revises published COT data after the fact. Four mechanisms: + +1. Contract-level error corrections (narrow — one market, one date) +2. Late or amended filings from reporting firms (usually recent-dated) +3. Contract universe additions (composition, not values) +4. **Trader reclassification** — the one that matters + +Reclassification moves positions *between categories*. Since Managed Money and Leveraged Funds are the sole input to everything downstream, and since all downstream outputs are rolling z-scores and percentiles against three years of history, a retroactive restatement silently rewrites the baseline against which every historical reading was computed. + +Precedent for large retroactive restatements: in July 2008 the Commission revised reports for affected markets going back to July 3, 2007 — over a year of history rewritten in one event. + +CFTC serves current-state files only. No official vintage archive, no as-published endpoint. **Vintage data can only be accumulated going forward,** which is why this is step 1. Git-history recovery has been attempted and returned nothing, so the vintage series starts from the first capture this session produces. + +CFTC does publish a revision log, ingested here as a first-class source: +`https://www.cftc.gov/MarketReports/CommitmentsofTraders/HistoricalSpecialAnnouncements/index.htm` + +--- + +## 2. Discovery — resolved and outstanding + +**Resolved:** git-history archaeology found no recoverable historical vintages. Vintage accumulation is forward-only from this session. + +**Still to report before writing code:** + +1. **Existing storage.** What is persisted today, and are raw downloads retained or discarded after parse? +2. **Existing schema.** Column names, dtypes, index conventions of the parse output. This is the basis for §3.2. +3. **Which sources are fetched.** Report types (legacy / disaggregated / TFF / supplemental), futures-only vs combined, annual zips vs weekly statics vs the Socrata public reporting environment. +4. **Date handling.** Does the code assume the as-of date is always a Tuesday? (It isn't — §6.) Is any release date stored today? +5. **Idempotency.** What happens on re-fetch of an already-ingested week — overwrite, skip, or duplicate? +6. **Manifest shape.** Current `manifest.json` structure, so the vintage block extends it rather than colliding with it. + +--- + +## 3. Persistence — Parquet, no database + +Nothing in the bitemporal design requires SQL. The data is small enough that Parquet is adequate indefinitely: change-only writes over roughly 50k rows per year means the entire vintage history stays in single-digit megabytes. DuckDB can query these files directly when ad-hoc SQL is wanted, without adopting a datastore as the storage format. + +**Record this as an ADR**, since it sits directly on ADR-0007's boundary. The decision being made: *vintage provenance is in scope for the narrowed `cotdata`, because it is CFTC-positioning provenance, and it persists within the existing Parquet + manifest contract rather than introducing a datastore.* + +### 3.1 Layout + +``` +raw/{source_kind}/{year}/{retrieved_at}_{sha8}.{ext} immutable; never rewritten +observations/report_year=YYYY/*.parquet change-only rows +revisions/detected_year=YYYY/*.parquet append-only, field-level +release_schedule.parquet +announcements.parquet +manifest.json gains a `vintage` block +current/ UNCHANGED — existing consumer contract +``` + +Raw bytes are retained permanently; the files are small and reprocessing history without re-hitting CFTC is worth the disk. + +### 3.2 `observations/` — bitemporal facts + +Natural key: `(report_date, market_code, report_type, combined, category)` + +| Column | Type | Notes | +|---|---|---| +| `report_date` | date | as-of date **as reported** — do not normalise to Tuesday | +| `release_date` | date | nullable; see §4.6 | +| `release_date_source` | text | `observed` \| `announced` \| `scheduled` \| `derived` \| `unknown` | +| `market_code` | text | CFTC contract market code | +| `market_name` | text | | +| `report_type` | text | `legacy` \| `disaggregated` \| `tff` \| `supplemental` | +| `combined` | bool | futures-only vs futures-and-options | +| `category` | text | controlled vocabulary per report type | +| `long_contracts` | int | | +| `short_contracts` | int | | +| `spread_contracts` | int | nullable | +| `trader_count_long` | int | nullable (suppressed) | +| `trader_count_short` | int | nullable | +| `open_interest` | int | market total | +| `cr4_net_long` … `cr8_net_short` | float | concentration ratios | +| `observed_at` | timestamp | when we saw this value (UTC) | +| `snapshot_id` | text | provenance → raw file | +| `row_sha256` | text | hash of value tuple only, excluding provenance columns | +| `is_tombstone` | bool | column present this session; handling deferred (§10) | + +**Change-only writes.** On ingest, compute `row_sha256` and compare against the most recent row for that natural key. Write only if it differs. Storage grows with actual revisions, not with time. + +**Point-in-time read.** For each natural key, the row with greatest `observed_at <= t`. No `valid_to` column. In polars this is a filter plus group-by-max — milliseconds at this size. + +### 3.3 `revisions/` — derived, field-level, append-only + +| Column | Type | +|---|---| +| `revision_id` | text (uuid or content hash) | +| natural key columns | as above | +| `field` | text | +| `old_value` | text | +| `new_value` | text | +| `delta` | float (numeric fields only) | +| `pct_delta` | float | +| `old_snapshot_id` / `new_snapshot_id` | text | +| `detected_at` | timestamp | +| `age_days` | int — `detected_at − report_date`, i.e. **revision depth** | + +`age_days` is the field that determines how much the rest of the system needs to care. Confined to recent weeks means PIT discipline matters only for the current observation; reaching into the calibration window means every historical percentile is live. + +### 3.4 `raw/` provenance — recorded in `manifest.json` + +Per snapshot: `snapshot_id`, `source_url`, `source_kind` (`annual_zip` | `weekly_static` | `socrata`), `report_year`, `retrieved_at`, `http_etag`, `http_last_modified`, `content_sha256`, `byte_size`, `local_path`, `parse_status`, `parse_error`. + +**A changed `content_sha256` does not imply changed data** — zips get regenerated. Byte-level change triggers parse-and-diff; it is not itself a revision. + +### 3.5 `release_schedule.parquet` + +| Column | Type | Notes | +|---|---|---| +| `report_date` | date | | +| `release_date` | date | | +| `source` | text | `announced` \| `scheduled` | +| `note` | text | e.g. holiday shift, backlog catch-up | +| `ingested_at` | timestamp | | + +Seeded from the CFTC [release schedule](https://www.cftc.gov/MarketReports/CommitmentsofTraders/ReleaseSchedule/index.htm) and the Special Announcements page. + +### 3.6 `announcements.parquet` + +`announcement_date`, `raw_text`, `affected_report_types` (nullable), `affected_markets` (nullable), `affected_date_from` / `affected_date_to` (nullable), `url`, `scraped_at`. + +Text parsing is best-effort — always store `raw_text`, treat structured extraction as convenience. Purpose is attribution: matching observed diffs to announced causes, and flagging historical windows as known-restated even where pre-revision values are unrecoverable. + +--- + +## 4. Behaviour + +### 4.1 Fetch +Conditional GET with `If-None-Match` / `If-Modified-Since` where supported. Always record a snapshot entry, including on 304 (record the check, skip the download). Rate-limit; set a descriptive User-Agent. + +### 4.2 Ingest +Parse → canonical schema → validate (§5) → change-only write → emit revisions. Must be idempotent: **re-ingesting an identical file produces zero new observation rows and zero revision rows.** Primary correctness test. + +### 4.3 Diff +Computed on parsed values, never on file bytes. Field-level: one changed `short_contracts` produces one revision row, not a whole-row replacement. + +### 4.4 Disappearing rows +A natural key present in an earlier snapshot and absent from a later one covering the same period is not an unchanged value. Column is present this session; write handling deferred (§10). Do not silently carry values forward in the meantime — leave the gap visible. + +### 4.5 Category migration +Deferred to a later session (§10), but note the intent: when multiple categories for the same `(report_date, market_code, report_type, combined)` change in one ingest, a conserved total with shifted composition is the reclassification signature. + +### 4.6 Release-date resolution + +The annual zips carry only `report_date`. Do not manufacture a release date — resolve it and record how, in this precedence order: + +| `release_date_source` | Mechanism | Coverage | +|---|---|---| +| `observed` | First `observed_at` for that report_date | Forward-only, accurate to polling interval | +| `announced` | Special Announcements — holiday shifts, backlog catch-up schedules | Irregular weeks, historical | +| `scheduled` | CFTC published release schedule | Normal weeks, historical | +| `derived` | `report_date + 3d`, holiday-adjusted | Fallback | +| `unknown` | — | Nothing resolved | + +`derived` fails on exactly the weeks that matter, so downstream code must be able to exclude those rows from strict PIT evaluation. That is the entire reason the flag exists — a release date without provenance is worse than none. + +**Spike first, before building any of the above.** Fetch a weekly static file for a known past week and check whether the HTTP `Last-Modified` header reflects true publication time. If CFTC serves accurate modification times, true release dates can be backfilled across all history in one pass and most of this taxonomy collapses to a single mechanism. Ten minutes to test. Do not build weekly-static fetching for any other purpose this session. + +--- + +## 5. Validation on every ingest + +Fail loudly; never silently drop rows. + +- Schema conformance and dtype check +- `long + short + spread <= open_interest` per category (warn, don't fail — definitional edge cases exist) +- Category values within the controlled vocabulary for that `report_type` +- Non-null natural key columns +- Row count within a sane band of the previous snapshot for the same source +- Null-rate per column within a sane band +- **`release_date − report_date` within the normal 3-day gap; flag and log if not** (catches holiday shifts and backlog weeks rather than letting them pass silently) + +--- + +## 6. Known gotchas + +- **The as-of date is not always Tuesday.** Federal holidays shift the release by a day, and at least one December report covered the prior Monday's open interest instead of Tuesday's. Store the reported date as-is; any code deriving report date by rounding to Tuesday will silently misalign those weeks. +- **The Oct–Dec 2025 backlog.** COT processing and publication were interrupted from 1 October to 12 November 2025 by the lapse in federal appropriations; CFTC then republished in chronological order, clearing the backlog by 29 December 2025. For those report dates, `release_date` trails `report_date` by weeks — in some cases over a month. This is the single largest PIT hole in the existing history, and backfilling true release dates for these weeks is the highest-value historical fix available. The 2023 ION cyber incident produced a smaller version of the same thing. +- **Index on release date, not report date,** for anything used in historical evaluation. Report date embeds a lookahead — normally three days, but weeks during backlog periods. +- **Futures-only and futures-and-options-combined are different series.** Never mix within one time series; keep `combined` in the natural key. +- **File regeneration ≠ data revision.** See §3.4. +- **Classifications are not auditable.** CEA Section 8 confidentiality means CFTC does not publish how individual traders are classified; reclassification can only be inferred from aggregate footprints. +- **Program under review.** CFTC opened a request for public comment on the COT Reports program in May 2026 covering potential procedural modifications. Keep ingestion loosely coupled from schema so a format change stays contained. + +--- + +## 7. CLI + +``` +cotdata vintage fetch [--year YYYY | --all] [--source annual|socrata] +cotdata vintage ingest [--snapshot ID | --pending] +cotdata vintage diff [--since DATE] [--market CODE] [--report-type T] +cotdata vintage asof --as-of TIMESTAMP --report-date DATE [--market CODE] +cotdata schedule sync # release schedule + announcements +cotdata schedule backfill # resolve release_date across existing history +``` + +`vintage stats` is deferred (§10). + +--- + +## 8. Tests (required) + +| Test | Assertion | +|---|---| +| Idempotent ingest | Same file twice → 0 new observations, 0 revisions | +| Byte-change, data-same | Regenerated archive, identical values → 0 revisions | +| Single-field revision | Fixture with one changed field → exactly 1 revision row, correct old/new | +| PIT query | `asof(t)` before a revision returns the pre-revision value | +| Revision depth | `age_days` correct across a synthetic year-old restatement | +| Holiday week | Monday as-of fixture parses without Tuesday normalisation | +| Backlog week | Report date in the Oct–Nov 2025 window resolves to its true announced release date, flagged `announced` | +| Release-date precedence | `observed` beats `announced` beats `scheduled` beats `derived` | +| Validation failure | Malformed fixture raises rather than partially ingesting | +| Contract preservation | `current/` output byte-identical to pre-change baseline | + +Build fixtures from real CFTC files trimmed to a handful of markets, plus hand-edited copies for the revision cases. Commit fixtures so tests run offline. + +--- + +## 9. Acceptance criteria + +1. Every fetch recorded in the manifest with raw bytes retained. +2. Re-ingesting unchanged data is a no-op at the observation level. +3. Any value change produces a field-level revision row with correct provenance on both sides. +4. `vintage asof` reconstructs the dataset as known at an arbitrary past timestamp. +5. Release schedule and announcements ingested; `release_date` + `release_date_source` backfilled across existing history, including the Oct–Dec 2025 backlog weeks. +6. No database introduced; all persistence in Parquet + manifest. +7. `current/` output byte-identical; existing repo tests still pass. +8. All §8 tests pass offline against committed fixtures. +9. ADR recorded for the scope and persistence decision (§3). + +--- + +## 10. Scope for this session + +Git recovery returned nothing, so the diff machinery has no input until the next release. Analytics are therefore deferred and the release-date backfill — the only piece with immediate value against existing history — is promoted. + +**In scope:** +1. Raw snapshot capture: hashed, immutable, manifest-recorded. Must start now; every uncaptured week is permanently lost. +2. Change-only observation writes with `observed_at` / `snapshot_id` provenance. +3. Field-level revisions Parquet. +4. Release schedule + announcements ingestion; `release_date` / `release_date_source` backfill. +5. Tests per §8. +6. `current/` byte-identical. + +**Deferred:** +- `vintage stats` and revision analytics — nothing to measure yet; build once a quarter of data exists. +- Category-migration detection — requires revisions to exist first. +- Tombstone handling — add the column, defer the logic. +- Weekly-static fetching beyond the `Last-Modified` spike (§4.6). + +--- + +## 11. Report back + +- Outstanding discovery findings (§2), especially existing schema and manifest shape +- **Result of the `Last-Modified` spike** — this determines whether the release-date taxonomy collapses to one mechanism +- Any deviation from this spec and why +- Date coverage achieved by the release-date backfill, broken down by `release_date_source` +- Whether weekly static reports appear frozen at publication or restated alongside annual files — open empirical question; the answer determines whether any further historical vintage recovery is possible diff --git a/docs/design/crowdmon_futures_cot_module.md b/docs/design/crowdmon_futures_cot_module.md new file mode 100644 index 0000000..9484155 --- /dev/null +++ b/docs/design/crowdmon_futures_cot_module.md @@ -0,0 +1,402 @@ +# crowdmon-futures — COT Positioning & Systematic Flow Model + +**System description v0.1** · companion spec to `crowdmon_system_description.md` · Python + +Written as a peer document rather than an appendix section, because on current priority this is the primary build and the equity monitor is the follow-on. The two share a store, an aggregation layer, a report layer, and most of the exit-capacity engine. + +--- + +## 1. Purpose and rationale + +Measure crowding and forced-exit risk in futures markets, and model the **systematic flow response function**: given a price or volatility move, how many contracts must trend-following and vol-targeting capital mechanically transact, and what does that cost in impact terms. + +### Why futures first + +COT resolves the four structural weaknesses of the 13F approach: + +| Weakness in equities | Status in futures | +|---|---| +| Long-only — the short book is invisible | Long, short and spreading all reported explicitly | +| 45-day lag, quarterly | 3-day lag, weekly | +| Holdings estimated against an estimated float | Zero-sum closed system; open interest known exactly | +| No trader counts, no concentration data | Trader counts per category plus published CR4/CR8 | + +Additionally, futures ADV is exact and public, which removes the softest input in the equity exit-capacity calculation, and systematic capital in futures is *replicable* — trend-following can be modelled to a decent fit, so the reflexivity model is buildable rather than hypothetical. + +### Goals + +| # | Goal | +|---|------| +| G1 | Ingest COT into a point-in-time, vintage-aware store, reusing existing `cotdata` infrastructure | +| G2 | Normalise positioning into risk units comparable across markets and across time | +| G3 | Separate positioning *extremity* from positioning *concentration* and holder *fragility* | +| G4 | Measure cross-market positioning alignment — the macro/CTA book as a single object | +| G5 | Compute exit capacity with futures-specific constraints (roll congestion, limit moves, margin) | +| G6 | Produce a flow map: price level → forced systematic flow → days of ADV → bps of impact | + +### Non-goals + +- Not a trend-following strategy. The CTA replication model exists to estimate *other people's* positions, and must not be repurposed as a signal without separate validation. +- No intraday data, no order-book modelling. +- No options positioning in v0.1 (futures-and-options-combined COT is ingested, but not delta-decomposed). +- No cash/OTC/swap market reconstruction. + +--- + +## 2. Architecture + +```mermaid +flowchart TB + subgraph EXIST["Existing — cotdata"] + X1["Weekly COT fetch
    + parse"] + end + + subgraph L1["Layer 1 — Ingestion"] + A1["Adapter shim
    cotdata to canonical schema"] + A2["Daily OI + volume
    CME / ICE"] + A3["Futures OHLCV
    per-contract"] + A4["Contract specs
    + roll calendar"] + A5["Exchange margin
    SPAN changes"] + A6["CTA indices
    SG Trend, BTOP50"] + end + + subgraph L2["Layer 2 — Normalisation"] + B1["Contract master"] + B2["Continuous series
    ratio-adjusted"] + B3["Notional conversion
    contracts to USD"] + B4["PIT vintaging
    release date, revisions"] + B5["Seasonal adjustment
    ags / commercials"] + end + + subgraph L3["Layer 3 — Engines"] + C1["Positioning engine
    extremity, concentration, fragility"] + C2["Flow decomposition
    new longs vs short covering"] + C3["Cross-market engine
    PCA, trend alignment"] + C4["Exit-capacity engine
    DTL, impact, roll, limits"] + C5["CTA response function
    replication + trigger solver"] + end + + subgraph L4["Layer 4 — Output"] + D1["Flow map
    price level to forced flow"] + D2["Composite scores
    rolling percentiles"] + D3["Store + report"] + end + + X1 --> A1 + A1 --> B4 + A2 --> B1 + A3 --> B1 + A4 --> B1 + A5 --> C4 + A6 --> C5 + B1 --> B2 --> B3 --> B4 --> B5 + B5 --> C1 --> C2 --> C3 + B5 --> C4 + C3 --> C5 + C4 --> C5 + C5 --> D1 + C1 --> D2 + C3 --> D2 + C4 --> D2 + D1 --> D3 + D2 --> D3 +``` + +**Shared with the equity monitor:** store, aggregation/z-scoring, report layer, and the impact-cost core of the exit-capacity engine (§5 of the equity doc). **Distinct:** ingestion, contract master, positioning engine, CTA response function. + +--- + +## 3. Data sources + +| Source | Content | Cadence | Lag | Notes | +|---|---|---|---|---| +| CFTC COT — Legacy | Commercial / non-commercial / non-reportable | Weekly | 3d | Longest history; use for pre-2006 backfill | +| CFTC COT — Disaggregated | Producer-Merchant, Swap Dealer, Managed Money, Other Reportable | Weekly | 3d | Physical commodities; 2006+ | +| CFTC COT — TFF | Dealer, Asset Manager, Leveraged Funds, Other Reportable | Weekly | 3d | Financial futures (rates, FX, equity index) | +| CFTC concentration ratios | CR4 / CR8 gross and net, largest traders | Weekly | 3d | Published within the same files | +| Bank Participation Report | Bank positioning by market | Monthly | — | Useful cross-check in FX and rates | +| CIT Supplemental | Index trader positions, select ags | Weekly | 3d | Separates index flow from spec flow | +| Exchange daily OI + volume | Per-contract OI, volume | Daily | 1d | **Required** for nowcasting between COT releases | +| Futures OHLCV | Per-contract prices | Daily | 1d | Per-contract, not pre-stitched | +| Contract specifications | Multiplier, tick, currency, limits | Static | — | Manually maintained | +| Exchange margin | Initial/maintenance, SPAN changes | Event | — | Margin hikes are a mechanical deleveraging trigger | +| SG Trend / BTOP50 | CTA index returns | Daily/Monthly | Varies | Calibration target for §7 | + +**Release mechanics:** COT publishes Friday 15:30 ET reflecting the prior **Tuesday** close. Both dates must be stored. Holiday weeks shift the schedule. Futures-only and futures-and-options-combined are separate series and must not be mixed within a time series. + +--- + +## 4. Integration with existing `cotdata` infrastructure + +Existing local infrastructure at `.../trading_workspace/cotdata` handles weekly COT consumption. The design assumption is **wrap, don't replace** — an adapter shim normalises whatever that module emits into a canonical schema, and everything downstream depends only on the schema. + +### Canonical positioning schema + +One row per market × report-type × category × vintage: + +``` +report_date date Tuesday as-of date +release_date date Friday publication date -- index on this +vintage int 0 = original, 1..n = revisions +market_code str CFTC contract market code +exchange str +report_type enum legacy | disaggregated | tff +combined bool futures-only vs futures+options +category str e.g. managed_money, leveraged_funds +long_contracts int +short_contracts int +spread_contracts int null where not applicable +trader_count_long int null where suppressed +trader_count_short int +open_interest int total OI for the market/report +cr4_net_long float concentration ratios +cr4_net_short float +cr8_net_long float +cr8_net_short float +``` + +### Adapter contract + +```python +class CotSource(Protocol): + def available_releases(self) -> list[date]: ... + def load(self, release_date: date) -> pd.DataFrame: + """Return rows conforming to the canonical schema above.""" +``` + +Two implementations: `LocalCotData` wrapping the existing module, and `CftcApiCotData` as a fallback and for backfill. Validation on every load — schema conformance, `long + short + spread <= OI` sanity, non-null keys, category vocabulary check. + +*Open items to resolve at implementation:* the existing module's output schema and index conventions, whether history is already backfilled and from what date, and whether revisions are currently retained or overwritten. Revision retention is the one that matters most — see §5.3. + +--- + +## 5. Normalisation + +### 5.1 Contract master and continuous series + +Per market: multiplier, currency, tick size, daily price limit rules, roll calendar, first notice date. Continuous series built ratio-adjusted (not difference-adjusted) so returns are correct, with the unadjusted per-contract series retained separately for notional and margin calculations. + +### 5.2 Positioning in comparable units + +The normalisation ladder, in increasing order of usefulness: + +1. **Net contracts** — raw. Not comparable across time or markets. Do not report. +2. **Net / open interest** — positioning as share of the market. Handles secular OI growth. +3. **Net notional USD** = `net_contracts × multiplier × price`. Comparable across markets. +4. **Vol-scaled notional** = `net_notional × σ_daily`. Positioning expressed in risk units. This is the version that corresponds to what actually forces deleveraging, and should be the default for every cross-market comparison. + +The widely used **COT Index** — stochastic rescaling of net position to 0–100 over a three-year window — is retained for continuity but flagged in output as lookback-sensitive and non-stationary. Rolling z-score of vol-scaled notional over a 3-year window is the primary measure. + +### 5.3 Point-in-time discipline + +Index on **release date**, never as-of date. Using the Tuesday date embeds a three-day lookahead — small, but it is exactly the window in which the largest moves happen, so it flatters every historical result in precisely the wrong way. + +CFTC revises. Store vintages and expose an as-of query so backtests see only what was visible at the time. If the existing `cotdata` module overwrites on re-fetch, this is the first thing to change. + +### 5.4 Seasonal adjustment + +Commercial and producer-merchant positioning in agricultural markets is strongly seasonal — hedging follows the crop calendar, not sentiment. Raw z-scores on those categories are dominated by seasonality and will produce spurious extremes every year at the same time. Apply a seasonal decomposition (or compare year-over-year within week-of-year) before z-scoring commercial categories in ags. Managed Money is less affected but not immune. + +--- + +## 6. Positioning engine + +### 6.1 Extremity + +Rolling z-score and percentile of vol-scaled net notional, per market per category, 3-year window, winsorised. Reported as percentile against own history. + +### 6.2 Concentration and breadth + +The metric set that COT gives away free and that has no cheap equity equivalent: + +- **CR4 / CR8** net long and short — published directly +- **Trader counts** per category and side +- **Average position per trader** = net position / trader count +- **Breadth–depth quadrant**, the key derived view: + +| | Trader count rising | Trader count flat/falling | +|---|---|---| +| **Avg position rising** | Crowd broadening *and* levering. Most dangerous. | Narrow and deep. Existing holders levering into it. Violent unwind. | +| **Avg position flat/falling** | Crowd broadening, individually smaller. Wide and shallow. Grinds. | Position being distributed. Crowding easing. | + +Same net position can sit in any quadrant. The quadrant, not the net, predicts the character of the unwind. + +### 6.3 Holder fragility + +The conceptually important adjustment, and the one most COT analysis omits. Futures are zero-sum — every long is matched by a short, so "everyone is long" is impossible and net imbalance alone says little. What matters is **how much of the open interest sits with holders who face forced-exit functions**. + +Assign a constraint weight per category: + +| Category | Weight | Rationale | +|---|---|---| +| Managed Money / Leveraged Funds | 1.0 | Vol targets, margin, drawdown limits, monthly redemptions | +| Other Reportables | 0.5 | Mixed | +| Dealer / Intermediary | 0.4 | Hedged, but balance-sheet constrained | +| Asset Manager / Institutional | 0.3 | Unlevered, longer horizon | +| Producer / Merchant / Processor | 0.1 | Hedging physical; can stand for delivery | +| Non-reportable | 0.6 | Retail; small but least resilient per unit | + +`fragility_weighted_oi = Σ (category_position × weight) / open_interest` + +Weights are configured, documented as judgement, and subjected to sensitivity analysis rather than presented as estimates. + +### 6.4 Flow decomposition + +Weekly change in net position decomposes into four states, which have different implications: + +| ΔLong | ΔShort | State | Implication | +|---|---|---|---| +| + | ~0 | New longs | Fresh conviction buying. Sustainable while flows continue. | +| ~0 | − | Short covering | Rally with a **finite fuel supply**. Ends when the shorts are gone. | +| ~0 | + | New shorts | Fresh bearish conviction. | +| − | ~0 | Long liquidation | Position exit, not fresh selling. | + +A rally driven by short covering and a rally driven by new longs look identical on a chart and are entirely different setups. This decomposition is one line of code and is among the highest-value outputs in the system. + +--- + +## 7. Cross-market engine + +Five categories cannot yield a manager-to-manager overlap matrix. The substitute is stronger in one respect: you can observe whether the same category is positioned consistently *across correlated markets*, which is the thing crowding actually consists of. + +- **Panel construction.** Matrix of z-scored Managed Money / Leveraged Funds positioning: markets × weeks. +- **Macro-book PCA.** PCA on positioning *changes*. PC1 approximates the aggregate systematic book; its variance share is the futures absorption ratio. Loading rotation indicates the book being redefined. +- **Trend alignment score.** Correlate the cross-market positioning vector against a canonical time-series momentum vector (blended 20/60/250-day TSMOM per market). High alignment means the trend book is fully expressed — little dry powder, maximum vulnerability to reversal. This is the futures analogue of the equity unwind mirror, and it is cleaner because trend-following is genuinely replicable. +- **Correlation clustering.** Cluster markets by return correlation rather than by sector label. "Long energy" and "short JPY" can be the same macro trade in a given regime; sector taxonomy hides that, empirical clustering does not. + +--- + +## 8. Exit-capacity engine (futures) + +The core impact model is shared with the equity spec (§5.2 there — square-root law, `impact ≈ Y·σ·√(Q/ADV)`). Futures-specific additions: + +- **Days-to-liquidate** with exact ADV: `net_position / (participation × ADV_contracts)`. No float estimation required. +- **OI / volume ratio** — how many days of turnover the open interest represents. A structural liquidity descriptor. +- **Roll congestion.** Calendar spread volatility and bid-ask behaviour during roll windows, plus OI migration rate front→next. A crowded position that must roll pays a measurable, predictable tax; congestion in the spread is an early liquidity tell that the outright market does not show. +- **Limit moves.** Ags and some energy contracts have daily price limits. When limit-up or limit-down, available liquidity is *zero* — a hard constraint with no equity equivalent. Model as an absolute cap on daily exit volume, and flag markets currently near limit distance. +- **Margin sensitivity.** Exchange margin increases force deleveraging directly and are typically announced during vol spikes, i.e. exactly when exit capacity is already impaired. Track margin-to-notional and its change; a rising ratio into a crowded position is a scheduled unwind. +- **Stress-conditioned ADV** and the **volume-spike trap** mitigation carry over unchanged from the equity spec §5.4, and matter just as much here. + +--- + +## 9. CTA response function + +The centrepiece, and the reason futures is worth building first. + +### 9.1 Replication model + +Estimate aggregate systematic positioning per market as: + +``` +raw_signal_i = blend of TSMOM over {20, 60, 250}d, squashed (tanh or rank) +vol_target_i = target_vol / σ_i -- position ∝ 1/σ +portfolio_scale = f(correlation matrix, portfolio vol target) +position_i = raw_signal_i × vol_target_i × portfolio_scale × AUM_estimate +``` + +The volatility-targeting term is the important one. Position size scales inversely with realised volatility, which means **a volatility spike forces selling regardless of price direction**. Most of the reflexivity in modern futures markets runs through this channel rather than through signal flips. + +### 9.2 Calibration + +Fit against two targets: + +1. **SG Trend / BTOP50 returns** — regress modelled portfolio returns on index returns; target R² in the 0.6–0.8 range, which is what published replication work achieves. +2. **COT Managed Money positions** — the model should reproduce the observed weekly positioning panel. Fit quality per market tells you where the category is dominated by trend followers and where it is contaminated by discretionary macro. + +Cross-validate the two. Where they disagree, COT is the ground truth for positioning and the index is the ground truth for aggregate risk appetite. + +### 9.3 Trigger solver + +Invert the model: for each market, solve for the price at which the blended signal changes sign, and for the volatility level at which vol targeting forces a given percentage reduction. Output: + +``` +market: GC (Gold) +current MM net: +XXX,XXX contracts (94th pctile, 3y) +fragility-weighted OI: 0.61 +20d signal flips at: $X,XXX (−4.2% from spot) +60d signal flips at: $X,XXX (−9.8% from spot) +est. systematic supply on 60d flip: N contracts = M days ADV at 20% participation +est. impact: P bps +vol-shock sensitivity: +5 vol pts forces −Q% position, independent of price +``` + +That block is the deliverable. It combines positioning extremity, holder fragility, a specific trigger level, and a liquidity-denominated cost estimate — which is the full synthesis the equity monitor can only approximate. + +### 9.4 Standing caution + +The replication model must not become a trading signal by drift. It is calibrated to reproduce *consensus* positioning, so trading it directly means deliberately joining the crowded trade the system exists to warn about. If a directional strategy is ever derived from it, that requires separate out-of-sample validation and its own document. + +--- + +## 10. Validation + +Replay against episodes where positioning is known to have driven the move, and require the composite to elevate **before** the drawdown rather than coincidentally with it: + +| Episode | Test | +|---|---| +| Feb 2018 volatility spike | Vol-target channel: forced selling on σ, not price | +| Mar 2020 | Margin-driven deleveraging, limit moves, liquidity collapse | +| Aug 2024 yen carry unwind | Cross-market alignment; FX positioning extremity | +| 2021 ags / lumber | Limit-move constraint; seasonal adjustment correctness | +| Silver 2021 / gold 2025 | Retail and non-reportable participation, concentration ratios | + +Plus mechanical tests: release-date indexing (no lookahead), vintage replay reproducing historical values exactly, and seasonal adjustment removing the annual cycle in ag commercial z-scores. + +--- + +## 11. What this system does not measure + +1. **Entities.** Five categories, not managers. No overlap matrix, no manager-level concentration. +2. **Category heterogeneity.** Managed Money blends CTAs, discretionary macro, and risk parity. Leveraged Funds in TFF includes relative-value books whose "net" is meaningless in isolation. +3. **Cash, OTC and swap exposure.** Partially visible through Swap Dealer and Dealer/Intermediary categories, but not decomposable. Commercial hedgers are offsetting physical positions you cannot see, which is why treating commercials as a sentiment signal is unsound. +4. **Options.** Combined reports include options on a futures-equivalent basis with no delta decomposition. Dealer gamma is not modelled. +5. **Intra-week dynamics.** Tuesday snapshot with Friday release. A fast unwind lasts days, so COT confirms after the fact — hence the daily OI nowcast, which is a partial fix and not a complete one. +6. **Non-US venues.** Positioning in LME, SGX, and Asian exchanges is not covered by CFTC reporting. +7. **Direction.** Positioning extremes persist for quarters. Every output is a statement about tail shape and forced-flow risk, not about next week's return. + +--- + +## 12. Stack and layout + +Python 3.11+ · polars or pandas · numpy · statsmodels · scikit-learn · scipy (trigger solver) · requests · pyarrow · duckdb · matplotlib · pydantic · pytest + +``` +crowdmon_futures/ + config/ markets.yaml, contract specs, fragility weights, params + ingest/ + cot_adapter.py wraps existing cotdata; canonical schema + cftc_api.py fallback + backfill + exchange_oi.py daily OI/volume nowcast feed + prices.py per-contract OHLCV + margin.py SPAN margin changes + normalize/ + contract_master.py + continuous.py ratio-adjusted stitching + notional.py + vintage.py PIT store, revision handling + seasonal.py + engines/ + positioning.py extremity, concentration, fragility, flow decomposition + crossmarket.py panel, PCA, trend alignment, clustering + liquidity.py DTL, roll congestion, limit moves, impact + cta.py replication, calibration, trigger solver + aggregate/ + report/ flow map, quadrant plots, panel heatmaps + store/ shared duckdb with equity monitor + tests/ + cli.py +``` + +--- + +## 13. Build order + +1. **Adapter + canonical schema + vintage store.** Wrap `cotdata`, backfill history, enforce release-date indexing. Nothing downstream is trustworthy until this is right. +2. **Contract master, notional and vol-scaled normalisation.** Everything else consumes these units. +3. **Positioning engine** — extremity, concentration, breadth–depth quadrant, flow decomposition. First genuinely useful output; earns its keep before any modelling. +4. **Exit-capacity engine** — DTL, impact, limit-move and roll constraints. +5. **Cross-market engine** — panel, PCA, trend alignment. +6. **CTA response function and trigger solver.** Highest value, highest complexity; depends on all of the above. +7. Validation replay, report layer, seasonal adjustment for ags. + +Steps 1–3 constitute a working monitor on their own. Steps 4–6 are what make it a forced-flow model rather than a positioning dashboard. From ba1b99976ddd3e298ecc610e702d87de88968847 Mon Sep 17 00:00:00 2001 From: Matt Spinola Date: Thu, 30 Jul 2026 19:53:24 -0400 Subject: [PATCH 12/16] docs: amend the crowdmon/vintage specs with what the build established MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Moves findings out of commit messages and chat and into the docs, so the docs are the source of truth. The specs above the new sections are left unedited: they are the documents as written before implementation, and the amendments record where they were wrong rather than quietly correcting them. handoff §12 "Outcome" (following the ADR-0006 house pattern): - The Last-Modified spike as an explicit NEGATIVE result: historical release dates cannot be recovered from headers, so §4.6's fallback chain is the confirmed path. - Header sweep 2020-2026 and the ROLLING TWO-YEAR WINDOW: 2025 and 2026 share an identical Last-Modified, so CFTC regenerates current + prior year and everything older is static. Corrects an earlier claim that all closed years were re-touched weekly, and explains why weekly downloads have always grabbed both years. Also: no ETag is served on any file, so If-None-Match can never fire, while If-Modified-Since does 304. - Decisions: retention (keep everything, no pruning), --all as a restatement TRIPWIRE rather than a backfill, daily producer-side capture, and the futures-only carry-forward that leaves `combined` constant-False with half the reportable universe absent. - Deviations table (polars->pandas, manifest.json->snapshots.json, synthetic fixtures, no --from-git, and `published` shipping despite §10 deferring it). - Answers to §11, including that backfill coverage is legitimately empty because no production vintages exist yet. crowdmon_futures_cot_module.md: resolves the §4 open items (schema, backfill span, revisions were overwritten, no release_date, futures-only) and notes that `vintage: int` is not how it was built — the implementation is bitemporal. §5.3 gains the two findings that change what it can assume: release date is resolved with provenance rather than read, and `derived` fails on exactly the weeks that matter; and vintage history is forward-only, so PIT protection cannot be backfilled. Links to cot_vintage.md and ADR-0008 dangle on main until PRs #78 and #13 merge; called out inline rather than left to surprise a reader. Co-Authored-By: Claude Opus 4.8 --- docs/design/cot_vintage_store_handoff.md | 111 +++++++++++++++++++++ docs/design/crowdmon_futures_cot_module.md | 40 ++++++++ 2 files changed, 151 insertions(+) diff --git a/docs/design/cot_vintage_store_handoff.md b/docs/design/cot_vintage_store_handoff.md index c0ca326..394e5ae 100644 --- a/docs/design/cot_vintage_store_handoff.md +++ b/docs/design/cot_vintage_store_handoff.md @@ -291,3 +291,114 @@ Git recovery returned nothing, so the diff machinery has no input until the next - Any deviation from this spec and why - Date coverage achieved by the release-date backfill, broken down by `release_date_source` - Whether weekly static reports appear frozen at publication or restated alongside annual files — open empirical question; the answer determines whether any further historical vintage recovery is possible + +--- + +## 12. Outcome (2026-07-30) + +Everything above is the spec **as written before implementation** and is left unedited. +This section records what the build established, including where the spec was wrong. +Implementation lives on branch `claude/cot-revision-snapshots-9b196f` +([cotdata PR #78](https://github.com/mspinola/cotdata/pull/78)), unmerged at time of +writing. Design detail: [cot_vintage.md](cot_vintage.md). Decision: crucible-stack +ADR-0008 ([crucible-stack PR #13](https://github.com/mspinola/crucible-stack/pull/13)). + +> **Both of those targets arrive with their PRs.** `cot_vintage.md` lands with cotdata +> #78 and `ADR-0008` with crucible-stack #13, so on `main` those two links dangle until +> the PRs merge. Everything stated in this section is independent of that and stands on +> its own. + +### 12.1 The `Last-Modified` spike: a negative result + +§4.6 hoped the spike would collapse the release-date taxonomy to one mechanism. **It did +not.** Historical release dates cannot be recovered from HTTP headers: + +- Annual zips are regenerated, so their `Last-Modified` carries no per-week information. +- Past weekly statics are not archived — one file, overwritten each week. + +`announced` and `scheduled` therefore remain the only sources for historical weeks, and +§4.6's fallback chain stands as written rather than as a contingency. Recorded explicitly +as a null so it is not re-investigated. + +The spike did produce one gain: the weekly static's `Last-Modified` **is** a true +publication timestamp for the current week, which is strictly better than "first +`observed_at` at polling interval." + +### 12.2 Header sweep (measured 2026-07-30) + +| Year | `Last-Modified` | +|---|---| +| 2020 | 2021-10-29 | +| 2021 | 2022-01-14 | +| 2022 / 2023 / 2024 | 2026-01-15 (one bulk re-touch, not weekly) | +| **2025** | **2026-07-24 19:27:59** | +| **2026** | **2026-07-24 19:27:59** | + +**The rolling two-year window.** 2025 and 2026 share an identical timestamp: CFTC's weekly +job regenerates the current year *and the immediately-prior year*. Everything older is +static. An earlier draft claimed all closed years were re-touched weekly; that was wrong. + +This explains a long-standing observation that weekly downloads grab both the current and +prior year: the pipeline conditions on `Last-Modified`, CFTC re-touches it, and identical +bytes are re-fetched. No revision is involved. + +**Conditional GET.** cftc.gov serves **no `ETag` on any file**, so `If-None-Match` can +never fire. `If-Modified-Since` does work (verified 304 against both a static year and the +current year), so on an `--all` sweep only ~2 of ~40 annual files actually transfer. + +**Content churn.** Inner zip entry mtimes show closed years are byte-frozen (2025's content +has not changed since January despite weekly header re-touching); only the current year +genuinely churns. Caveat: inner mtime is strong evidence, not proof — a regeneration +preserving source mtimes would look identical. `content_sha256` across two weeks is +definitive, and capture now does that automatically. + +### 12.3 Decisions taken + +- **Retention: keep everything, no pruning.** ~1 GB/year of current-year churn is + immaterial against irreplaceability, and those weekly copies *are* the vintage series — + pruning them destroys the artifact being built. +- **`--all` is a restatement tripwire, not a backfill.** Since closed years are frozen, a + `content_sha256` change on one is precisely the 2008-style retroactive-restatement + signature, and it is the only automated detector for the failure mode this subsystem + exists to guard against. Monthly or quarterly; weekly is pointless because nothing older + moves. Implemented as a `restatement_suspect` flag on any closed-year content change. +- **Capture runs on the producer, daily.** Daily because nearly everything 304s, so it is + nearly free while catching holiday shifts and backlog publications with no schedule + logic, and it tightens `observed_at` from a seven-day to a one-day bound. On the + producer because the replicas are mirrored (`robocopy /MIR`, `rsync --delete`) and a + vintage tree written on a replica is **deleted** by the next sync — irrecoverable, since + CFTC serves current state only. `COTDATA_VINTAGE_ROOT` relocates the tree for a replica + that must capture anyway. +- **Futures-only carry-forward.** Only `dea_fut_xls` / `fut_disagg_txt` / `fut_fin_txt` are + fetched, all futures-only, so `combined` is constant-`False` in the natural key today. + This is a deliberate carry-forward of existing producer scope under the §0 non-goal, not + an oversight. `combined` stays in the key so the two series can never silently merge + (§6); adding the combined files later is a fetch-list change with no schema migration. + **Consequence: half the reportable universe is absent** until that happens. + +### 12.4 Deviations from this spec + +| Spec said | Built | Why | +|---|---|---| +| polars (§3.2) | pandas + pyarrow | polars is not a dependency; pandas/pyarrow already are, so the no-database decision is not undone by adding a near-equivalent | +| `manifest.json` (§3.1) | `snapshots.json` | Both deployed sync scripts exclude `manifest.json` **unanchored** (robocopy `/XF`, rsync `--exclude` match at any depth), so it would have been stripped in transit, delivering raw archives to a replica with no index | +| fixtures from trimmed real CFTC files (§8) | synthetic frames | Matches the repo's existing test idiom, runs offline, and avoids an `xlrd` fixture dependency | +| `vintage recover --from-git` (§7) | not built | Git archaeology found zero committed data files; there is nothing to recover | +| weekly-static work deferred (§10) | **`published` shipped** | The deferral assumed the file needed parsing. It is a headerless positional CSV covering exactly ONE report date (365 rows, 129 columns, one distinct value in field 2), so the mapping reads one field. ~30 lines, so it landed. Verified live: report date 2026-07-21 → 2026-07-24 ET publication | + +### 12.5 Answers to §11 + +- **Backfill coverage by `release_date_source`: none yet.** No production vintages exist — + capture is forward-only and git recovery returned nothing, so there is nothing to report + coverage over. The machinery is tested (backlog week → `announced`; otherwise + `derived`), but real numbers require captures to accumulate. +- **Are weekly statics frozen at publication or restated?** Neither, and the question + dissolves: the weekly static is a **single file, overwritten** each week, not a per-week + archive. So it offers no route to further historical vintage recovery. + +### 12.6 Still deferred + +`vintage stats` (needs roughly a quarter of data to measure), category-migration detection +(needs revisions to exist), tombstone *logic* (column present; needs a real disappearing +key to design against), and full canonicalisation of the weekly static **into +observations** (129 positional columns, genuinely larger than the `published` mapping). diff --git a/docs/design/crowdmon_futures_cot_module.md b/docs/design/crowdmon_futures_cot_module.md index 9484155..c4d1af7 100644 --- a/docs/design/crowdmon_futures_cot_module.md +++ b/docs/design/crowdmon_futures_cot_module.md @@ -168,6 +168,28 @@ Two implementations: `LocalCotData` wrapping the existing module, and `CftcApiCo *Open items to resolve at implementation:* the existing module's output schema and index conventions, whether history is already backfilled and from what date, and whether revisions are currently retained or overwritten. Revision retention is the one that matters most — see §5.3. +> **Resolved 2026-07-30** by the vintage build (branch `claude/cot-revision-snapshots-9b196f`, +> [PR #78](https://github.com/mspinola/cotdata/pull/78); spec + outcome in +> [cot_vintage_store_handoff.md](cot_vintage_store_handoff.md) §12, design in +> [cot_vintage.md](cot_vintage.md)): +> +> - **Output schema.** Wide, one parquet per symbol per report type, `Report_Date`-indexed +> (tz-naive `DatetimeIndex`): OI, per-category long/short, trader counts. `Report_Date` is +> stored exactly as reported and **never** normalised to Tuesday, so §5.3's holiday-week +> hazard is already avoided by the existing producer. +> - **Backfill.** Full history per report: Legacy from 1986, Disaggregated and TFF from 2006. +> - **Revisions were overwritten**, as §5.3 feared: every run rebuilt the whole per-code table +> and replaced the parquet, so no prior state survived. That is what the vintage layer fixes. +> - **`release_date` did not exist anywhere**, and could not simply be read — see §5.3. +> - **Only futures-only is fetched**, so `combined` is constant-`False` today: the column is +> present and correct but not yet discriminating, and half the reportable universe is absent +> until the combined files are added. +> - **`vintage: int` is not how it was built.** The implementation is bitemporal +> (`observed_at` plus change-only rows), so a point-in-time read is "greatest +> `observed_at <= t` per natural key". An integer vintage ordinal can be derived from that +> if this adapter wants one, but it is not stored — storage grows with revisions rather +> than with time, which an ordinal-per-week scheme would not achieve. + --- ## 5. Normalisation @@ -193,6 +215,24 @@ Index on **release date**, never as-of date. Using the Tuesday date embeds a thr CFTC revises. Store vintages and expose an as-of query so backtests see only what was visible at the time. If the existing `cotdata` module overwrites on re-fetch, this is the first thing to change. +> **Built 2026-07-30.** It did overwrite; `cotdata-vintage` now captures as-published +> snapshots, records field-level revisions with `age_days` depth, and answers +> `asof(t)`. Two findings change what this section can assume: +> +> - **Release date is resolved, not read.** The annual zips carry only `report_date`, and +> HTTP headers cannot recover historical publication times (see +> [handoff §12.1](cot_vintage_store_handoff.md) — a measured negative result). So +> `release_date` carries a provenance flag, +> `published > observed > announced > scheduled > derived`. **`derived` fails on exactly +> the weeks that matter** (holiday shifts, the Oct–Dec 2025 backlog), so anything doing +> strict point-in-time evaluation must be able to exclude `derived` rows rather than +> trusting the date. A release date without provenance is worse than none. +> - **Vintages accumulate forward only.** Git archaeology recovered nothing and CFTC serves +> current state only, so the vintage series begins at first capture. Backtests over +> history predating that see *current-state* data with no as-of protection — the PIT +> discipline this section asks for is available going forward, not retroactively. That is +> a permanent property, not a gap to be filled later. + ### 5.4 Seasonal adjustment Commercial and producer-merchant positioning in agricultural markets is strongly seasonal — hedging follows the crop calendar, not sentiment. Raw z-scores on those categories are dominated by seasonality and will produce spurious extremes every year at the same time. Apply a seasonal decomposition (or compare year-over-year within week-of-year) before z-scoring commercial categories in ags. Managed Money is less affected but not immune. From 5ffb6fc712951e48cae6e8eb1832a2b077e360c2 Mon Sep 17 00:00:00 2001 From: Matt Spinola Date: Thu, 30 Jul 2026 19:56:45 -0400 Subject: [PATCH 13/16] docs: correct the link note now that the docs ship inside PR #78 The warning was written while these docs sat on main and cot_vintage.md did not. They now travel in the same PR, so that link never dangles. Only the cross-repo ADR-0008 reference still depends on a separate merge (crucible-stack #13). Co-Authored-By: Claude Opus 4.8 --- docs/design/cot_vintage_store_handoff.md | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/docs/design/cot_vintage_store_handoff.md b/docs/design/cot_vintage_store_handoff.md index 394e5ae..fdcc19f 100644 --- a/docs/design/cot_vintage_store_handoff.md +++ b/docs/design/cot_vintage_store_handoff.md @@ -303,10 +303,10 @@ Implementation lives on branch `claude/cot-revision-snapshots-9b196f` writing. Design detail: [cot_vintage.md](cot_vintage.md). Decision: crucible-stack ADR-0008 ([crucible-stack PR #13](https://github.com/mspinola/crucible-stack/pull/13)). -> **Both of those targets arrive with their PRs.** `cot_vintage.md` lands with cotdata -> #78 and `ADR-0008` with crucible-stack #13, so on `main` those two links dangle until -> the PRs merge. Everything stated in this section is independent of that and stands on -> its own. +> `cot_vintage.md` ships in this same PR, so that link resolves here and on `main` after +> merge. The cross-repo `ADR-0008` reference is the exception: it lands separately with +> crucible-stack #13, so it resolves only once that merges. Everything stated in this +> section is independent of both and stands on its own. ### 12.1 The `Last-Modified` spike: a negative result From ae45560f0b7efe0044697e34e3ab58aef540640c Mon Sep 17 00:00:00 2001 From: Matt Spinola Date: Thu, 30 Jul 2026 20:06:50 -0400 Subject: [PATCH 14/16] fix(test): compare current/ baseline by CONTENT, not parquet bytes The byte-hash guard passed locally and failed on all five CI Python versions, with a DIFFERENT hash on each (3.10 ebc56905, 3.13 0478a1ca, local 53e939b8). Parquet encoding is a property of the pandas/pyarrow build -- writer metadata, compression, dictionary encoding -- not of this repo's code, so the assertion was about the library rather than about anything the change could keep true. It could never have gone green in CI. What consumers actually depend on is the data read back, so that is what is compared now, against the same pre-vintage golden file. dtypes are compared loosely because CI spans Python 3.10-3.14 and therefore pandas 2 and 3, which disagree about the string dtype; values, columns, order and index are still compared exactly. Byte-stability is kept as a separate test where it IS a real property: writing identical data twice in one environment must produce identical bytes, which is what an atomic write could plausibly break. Co-Authored-By: Claude Opus 4.8 --- tests/test_current_baseline.py | 51 +++++++++++++++++++++++++++------- 1 file changed, 41 insertions(+), 10 deletions(-) diff --git a/tests/test_current_baseline.py b/tests/test_current_baseline.py index 26a1da2..9479175 100644 --- a/tests/test_current_baseline.py +++ b/tests/test_current_baseline.py @@ -1,12 +1,24 @@ -"""Guard: the current-state COT write path must stay byte-identical as the vintage -layer is added (acceptance §7). The golden parquet was generated from the tree BEFORE -the vintage subsystem landed; if this fails, the additive change stopped being additive. +"""Guard: the current-state COT write path must be unchanged by the vintage layer +(acceptance §7). The golden parquet was generated from the tree BEFORE the vintage +subsystem landed, so it pins what a consumer read back then; if this fails, the additive +change stopped being additive. + +**This compares CONTENT, not bytes, and that distinction is the whole lesson here.** The +first version of this guard hashed the parquet bytes. It passed locally and failed on all +five CI Python versions with a *different* hash on each, because parquet encoding is a +property of the pandas/pyarrow build (writer metadata, compression, dictionary encoding), +not of this repo's code. A byte hash therefore asserts something the code does not +control and cannot keep true. What consumers actually depend on is the data that comes +back out, which is what is checked below. + +dtypes are compared loosely for the same reason: CI spans Python 3.10-3.14 and so spans +pandas 2 and 3, which disagree about the string dtype. Values, columns, order and index +are all compared exactly. The golden frame is deterministic and defined here; regenerate the fixture with: PYTHONPATH=src python tests/_gen_golden.py """ -import hashlib from pathlib import Path import pandas as pd @@ -46,14 +58,33 @@ def store_env(tmp_path, monkeypatch): return tmp_path -def test_current_cot_legacy_output_byte_identical(store_env): +def test_current_cot_legacy_output_matches_pre_vintage_baseline(store_env): + """What a consumer reads back must equal the pre-vintage baseline, exactly.""" from cotdata import store + assert GOLDEN.exists(), "golden baseline missing — run tests/_gen_golden.py on a clean tree" store.write_cot_legacy("GOLD_088691", golden_frame(), source="cftc") - produced = (store_env / "cot_legacy" / "GOLD_088691.parquet").read_bytes() - assert GOLDEN.exists(), "golden baseline missing — run tests/_gen_golden.py on a clean tree" - expected = GOLDEN.read_bytes() - assert hashlib.sha256(produced).hexdigest() == hashlib.sha256(expected).hexdigest(), ( - "current/ cot_legacy output changed — the vintage layer is no longer additive" + produced = store.read_cot_legacy("GOLD_088691") + expected = pd.read_parquet(GOLDEN) + + pd.testing.assert_frame_equal( + produced, expected, + check_dtype=False, # CI spans pandas 2 and 3, which disagree on the string dtype + check_freq=False, + obj="current/ cot_legacy output changed — the vintage layer is no longer additive", ) + + +def test_current_write_path_is_deterministic_within_an_environment(store_env): + """Byte-stability is still a real property WITHIN one environment, and it is the thing + an atomic write must not break. Checked here rather than against a committed fixture, + since across environments the bytes legitimately differ.""" + from cotdata import store + + store.write_cot_legacy("GOLD_088691", golden_frame(), source="cftc") + first = (store_env / "cot_legacy" / "GOLD_088691.parquet").read_bytes() + store.write_cot_legacy("GOLD_088691", golden_frame(), source="cftc") + second = (store_env / "cot_legacy" / "GOLD_088691.parquet").read_bytes() + + assert first == second, "rewriting identical data produced different bytes" From e2521574038aabfa070194269064bf83a3e2bc74 Mon Sep 17 00:00:00 2001 From: Matt Spinola Date: Thu, 30 Jul 2026 20:10:02 -0400 Subject: [PATCH 15/16] =?UTF-8?q?fix(test):=20relax=20index=20dtype=20too?= =?UTF-8?q?=20=E2=80=94=20pandas=203=20writes=20us,=20pandas=202=20reads?= =?UTF-8?q?=20ns?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Second half of the same lesson. check_dtype=False does not cover the INDEX, which assert_frame_equal checks separately, so 3.10 still failed on datetime64[ns] vs datetime64[us]: pandas 3 wrote the golden at microsecond resolution and pandas 2 reads it back at nanosecond. Again a property of the library, not of this code. Index and column CONTENT are still asserted exactly, immediately after, so relaxing the dtype check does not quietly widen what this guard accepts. Co-Authored-By: Claude Opus 4.8 --- tests/test_current_baseline.py | 8 +++++++- 1 file changed, 7 insertions(+), 1 deletion(-) diff --git a/tests/test_current_baseline.py b/tests/test_current_baseline.py index 9479175..cc0b7c4 100644 --- a/tests/test_current_baseline.py +++ b/tests/test_current_baseline.py @@ -70,10 +70,16 @@ def test_current_cot_legacy_output_matches_pre_vintage_baseline(store_env): pd.testing.assert_frame_equal( produced, expected, - check_dtype=False, # CI spans pandas 2 and 3, which disagree on the string dtype + check_dtype=False, # CI spans pandas 2 and 3, which disagree on the string dtype + check_index_type=False, # ...and on datetime resolution: pandas 3 writes us, 2 reads ns. + # check_dtype does NOT cover the index, which needs this. check_freq=False, obj="current/ cot_legacy output changed — the vintage layer is no longer additive", ) + # Values, columns, order and index CONTENT are still compared exactly above; only the + # storage dtypes are relaxed, and only because they are the library's choice not ours. + assert list(produced.columns) == list(expected.columns) + assert [str(d.date()) for d in produced.index] == [str(d.date()) for d in expected.index] def test_current_write_path_is_deterministic_within_an_environment(store_env): From e93c69bde5a1ed2b5de180a578da4bd1f418a87c Mon Sep 17 00:00:00 2001 From: Matt Spinola Date: Thu, 30 Jul 2026 20:13:16 -0400 Subject: [PATCH 16/16] feat(vintage): surface revisions instead of only recording them A scheduled run's stdout goes nowhere, so a retroactive restatement -- the exact event this subsystem exists to detect -- would have been recorded and then silently swallowed, with exit 0 and no way to know without running 'diff' by hand. - 'ingest' now prints the changed fields (with age_days, flagging any reaching more than 30 days back into the calibration window) and exits NON-ZERO when it records a revision or sees a closed-year restatement suspect. The message says plainly that this is a notification and the data is committed, so it is not mistaken for a broken run. - The suspect scan is scoped to the snapshots THIS run processed. A store-wide scan would re-fire on every run forever once one restatement was seen, and an alert that never clears is one that gets switched off. Test pins that a prior run's suspect stays quiet. - run-vintage.cmd tees every step to a log, treats ingest's non-zero as 'revised' rather than 'failed' so later steps still run, and re-raises at the end so Task Scheduler shows the task as attention-needed. Co-Authored-By: Claude Opus 4.8 --- docs/examples/windows/run-vintage.cmd | 29 +++++++++----- src/cotdata/vintage_cli.py | 41 +++++++++++++++++++ tests/test_vintage_ingest.py | 57 +++++++++++++++++++++++++++ 3 files changed, 118 insertions(+), 9 deletions(-) diff --git a/docs/examples/windows/run-vintage.cmd b/docs/examples/windows/run-vintage.cmd index 3572006..7ea047d 100644 --- a/docs/examples/windows/run-vintage.cmd +++ b/docs/examples/windows/run-vintage.cmd @@ -25,15 +25,21 @@ REM ~17:00 ET puts capture within about ninety minutes of the 15:30 ET publicati REM REM ORDERING: run this AFTER run-cot.cmd, then chain the sync scripts after it, so a REM single sync carries both the current-state update and the new vintage snapshot. +REM NOTIFICATION: Task Scheduler discards stdout, so everything below is ALSO appended to +REM a log file, and `ingest` exits NON-ZERO whenever it records a revision or spots a +REM closed-year restatement. That is a notification, not a failure -- the data is already +REM committed. Point Task Scheduler's "send mail"/alert at this task, or just read the log. +REM Without it a retroactive restatement would be detected and silently swallowed. setlocal set COTDATA_STORE=REPLACE_WITH_STORE_PATH +set VINTAGE_LOG=REPLACE_WITH_STORE_PATH\vintage\run.log REM Optional: identify yourself to CFTC. Defaults to the repo URL if unset. REM set COTDATA_USER_AGENT=cotdata-vintage/0.1 (+contact you@example.com) REM Default path = current year's three annual reports + the Legacy weekly static. REM The weekly static is fetched for its HTTP Last-Modified, which is a true REM publication timestamp rather than a polling-interval approximation. -"REPLACE_WITH_VENV_PATH\Scripts\cotdata-vintage.exe" fetch +"REPLACE_WITH_VENV_PATH\Scripts\cotdata-vintage.exe" fetch >> "%VINTAGE_LOG%" 2>&1 if %ERRORLEVEL% NEQ 0 ( echo vintage fetch FAILED with code %ERRORLEVEL% exit /b %ERRORLEVEL% @@ -41,27 +47,32 @@ if %ERRORLEVEL% NEQ 0 ( REM Parse whatever was just retained into change-only observations + revisions. REM Safe to re-run: re-ingesting identical bytes writes zero rows. -"REPLACE_WITH_VENV_PATH\Scripts\cotdata-vintage.exe" ingest --pending -if %ERRORLEVEL% NEQ 0 ( - echo vintage ingest FAILED with code %ERRORLEVEL% - exit /b %ERRORLEVEL% -) +"REPLACE_WITH_VENV_PATH\Scripts\cotdata-vintage.exe" ingest --pending >> "%VINTAGE_LOG%" 2>&1 +REM NON-ZERO HERE MEANS "revisions were recorded", not "the run broke". The data is +REM already written. Remember it, keep going, and re-raise at the end so the scheduler +REM shows the task as attention-needed. +set VINTAGE_REVISED=%ERRORLEVEL% REM Resolve release dates. `published` reads the true publication timestamp out of the REM weekly static just captured (its HTTP Last-Modified), which beats a poll-derived REM `observed` bound; backfill then applies the precedence across all observations. REM Both are idempotent, so re-running is a cheap no-op. -"REPLACE_WITH_VENV_PATH\Scripts\cotdata-schedule.exe" published +"REPLACE_WITH_VENV_PATH\Scripts\cotdata-schedule.exe" published >> "%VINTAGE_LOG%" 2>&1 if %ERRORLEVEL% NEQ 0 ( echo schedule published FAILED with code %ERRORLEVEL% exit /b %ERRORLEVEL% ) -"REPLACE_WITH_VENV_PATH\Scripts\cotdata-schedule.exe" backfill +"REPLACE_WITH_VENV_PATH\Scripts\cotdata-schedule.exe" backfill >> "%VINTAGE_LOG%" 2>&1 if %ERRORLEVEL% NEQ 0 ( echo schedule backfill FAILED with code %ERRORLEVEL% exit /b %ERRORLEVEL% ) -echo vintage ok +if NOT "%VINTAGE_REVISED%"=="0" ( + echo vintage ok, but REVISIONS WERE RECORDED -- see "%VINTAGE_LOG%" and run: cotdata-vintage diff + type "%VINTAGE_LOG%" + exit /b %VINTAGE_REVISED% +) +echo vintage ok, no revisions exit /b 0 diff --git a/src/cotdata/vintage_cli.py b/src/cotdata/vintage_cli.py index 92037f9..4ee7263 100644 --- a/src/cotdata/vintage_cli.py +++ b/src/cotdata/vintage_cli.py @@ -57,9 +57,50 @@ def _cmd_ingest(args) -> int: vintage.update_snapshot(s["snapshot_id"], parse_status="failed", parse_error=str(e)) print(f" {s['snapshot_id']}: FAILED — {e}") print(f"vintage ingest: {total_obs} new observation(s), {total_rev} revision(s).") + + # Surface revisions rather than only recording them. A scheduled run's stdout goes + # nowhere, so a silent exit-0 after detecting a retroactive restatement would defeat + # the point of the subsystem. Anything noteworthy exits non-zero, which Task Scheduler + # and cron both report as a failed run, and prints the detail for the log. + # Scoped to the snapshots THIS run processed, not to every snapshot ever recorded. + # A store-wide scan would keep firing on every subsequent run once a single + # restatement had been seen, and an alert that never clears is one that gets ignored. + suspects = [s for s in snaps if s.get("restatement_suspect")] + if total_rev or suspects: + _report_revisions(total_rev, suspects) + raise SystemExit( + f"cotdata-vintage: {total_rev} revision(s)" + + (f", {len(suspects)} closed-year restatement suspect(s)" if suspects else "") + + " — review with 'cotdata-vintage diff'. Non-zero so a scheduler surfaces this; " + "the data IS committed, this is a notification, not a failure.") return 0 +def _report_revisions(total_rev: int, suspects: list) -> None: + from . import vintage_ingest + if suspects: + print("\n*** CLOSED-YEAR RESTATEMENT SUSPECT ***") + for s in suspects[-5:]: + print(f" {s.get('report_type')} {s.get('report_year')} " + f"changed content at {s.get('retrieved_at')}") + print(" A closed year should be frozen. This is the retroactive-restatement") + print(" signature the vintage store exists to detect.") + if not total_rev: + return + rev = vintage_ingest.read_revisions() + if rev.empty: + return + recent = rev.sort_values("detected_at").tail(20) + print(f"\n{len(rev)} revision row(s) recorded; most recent:") + cols = ["report_date", "market_code", "category", "field", "old_value", + "new_value", "age_days"] + print(recent[[c for c in cols if c in recent.columns]].to_string(index=False)) + deep = recent[recent["age_days"] > 30] if "age_days" in recent.columns else None + if deep is not None and not deep.empty: + print(f"\n{len(deep)} of these reach back more than 30 days — revisions inside the") + print("calibration window rewrite the baseline every historical reading used.") + + def _cmd_diff(args) -> int: from . import vintage_ingest rev = vintage_ingest.read_revisions() diff --git a/tests/test_vintage_ingest.py b/tests/test_vintage_ingest.py index dfb86ac..6c27fb6 100644 --- a/tests/test_vintage_ingest.py +++ b/tests/test_vintage_ingest.py @@ -193,6 +193,63 @@ def test_concurrent_ingest_lock_fails_loudly(store_env): vi.ingest_canonical(_canon("2026-07-21"), snapshot_id="s1") +def test_cli_ingest_exits_nonzero_when_revisions_are_recorded(store_env, capsys): + """A scheduled run's stdout goes nowhere, so a silent exit-0 after detecting a + restatement would defeat the subsystem. Revisions must surface as a non-zero exit.""" + from cotdata import vintage, vintage_cli + from cotdata import vintage_ingest as vi + + # two vintages of the same report date, second one revised + vi.ingest_canonical(_canon("2026-07-21", comm_short=250000), snapshot_id="s1", + observed_at=dt.datetime(2026, 7, 24, tzinfo=dt.timezone.utc)) + vi.ingest_canonical(_canon("2026-07-21", comm_short=251000), snapshot_id="s2", + observed_at=dt.datetime(2026, 7, 31, tzinfo=dt.timezone.utc)) + assert not vi.read_revisions().empty + + # drive the real CLI path: a snapshot whose raw file re-parses to the revised values + raw = store_env / "vintage" / "raw" / "annual_zip" / "2026" + raw.mkdir(parents=True, exist_ok=True) + (raw / "f.zip").write_bytes(b"x") + vintage._write_manifest({"schema_version": 1, "snapshots": [{ + "snapshot_id": "s3", "report_type": "legacy", "source_kind": "annual_zip", + "local_path": "vintage/raw/annual_zip/2026/f.zip", "parse_status": "pending", + "restatement_suspect": True, "report_year": 2025, + "retrieved_at": "2026-07-31T21:00:00Z", + }]}) + + with pytest.raises(SystemExit) as exc: + vintage_cli.main(["ingest", "--pending"]) + msg = str(exc.value) + assert "restatement suspect" in msg and "cotdata-vintage diff" in msg + assert "notification, not a failure" in msg # the data IS committed + assert "RESTATEMENT SUSPECT" in capsys.readouterr().out + + +def test_restatement_alert_does_not_fire_forever(store_env): + """A suspect recorded in an EARLIER run must not keep failing every later run — + an alert that never clears is one that gets switched off.""" + from cotdata import vintage, vintage_cli + vintage._write_manifest({"schema_version": 1, "snapshots": [{ + "snapshot_id": "old", "report_type": "legacy", "source_kind": "annual_zip", + "local_path": "vintage/raw/annual_zip/2025/old.zip", "parse_status": "ok", + "restatement_suspect": True, "report_year": 2025, + "retrieved_at": "2026-01-01T00:00:00Z", + }]}) + # --pending selects nothing (that snapshot is already parse_status=ok), so a later + # run is quiet even though the store still remembers the suspect. + assert vintage_cli.main(["ingest", "--pending"]) == 0 + + +def test_restatement_suspect_is_reported_prominently(store_env, capsys): + from cotdata import vintage_cli + suspects = [{"report_type": "legacy", "report_year": 2025, + "retrieved_at": "2026-07-31T21:00:00Z"}] + vintage_cli._report_revisions(0, suspects) + out = capsys.readouterr().out + assert "CLOSED-YEAR RESTATEMENT SUSPECT" in out + assert "legacy 2025" in out + + def test_oi_over_sum_warns_not_raises(store_env): from cotdata import vintage_ingest as vi # commercial long+short (200000+250000) already < OI; force a breach on OI instead