From 4240eccea04821999c0b5a50613644788e6d375f Mon Sep 17 00:00:00 2001 From: eggmasonvalue Date: Mon, 31 Aug 2026 23:50:01 +0530 Subject: [PATCH] refactor: sharpen remaining research skills --- .github/workflows/insider-scan.yml | 21 +- .markdownlint-cli2.jsonc | 2 + skills/README.md | 58 +- skills/market-scout/README.md | 44 +- skills/market-scout/SKILL.md | 104 +- skills/market-scout/scripts/_common.py | 47 +- .../market-scout/scripts/fetch_market_data.py | 264 ++-- .../market-scout/scripts/fetch_transcripts.py | 63 +- skills/pitch-like-lou/README.md | 80 +- skills/pitch-like-lou/SKILL.md | 306 ++--- skills/signal-sweep/README.md | 85 +- skills/signal-sweep/SKILL.md | 142 ++- .../docs/conferences-autoresearch.md | 263 ---- .../docs/flip_buy_difficulty_analysis.md | 228 ---- .../signal-sweep/references/guide_screens.md | 158 +-- skills/signal-sweep/screens.json | 20 +- skills/signal-sweep/scripts/_common.py | 36 +- .../signal-sweep/scripts/scan_conferences.py | 443 +++---- skills/signal-sweep/scripts/scan_insiders.py | 1105 ++++++++--------- skills/signal-sweep/scripts/scan_market.py | 314 +++-- skills/signal-sweep/scripts/search_themes.py | 301 +++-- 21 files changed, 1692 insertions(+), 2392 deletions(-) delete mode 100644 skills/signal-sweep/docs/conferences-autoresearch.md delete mode 100644 skills/signal-sweep/docs/flip_buy_difficulty_analysis.md diff --git a/.github/workflows/insider-scan.yml b/.github/workflows/insider-scan.yml index e417900..b89c176 100644 --- a/.github/workflows/insider-scan.yml +++ b/.github/workflows/insider-scan.yml @@ -2,8 +2,7 @@ name: Daily Insider Scan on: schedule: - # 7:00 AM ET on weekdays (11:00 UTC Nov-Mar, 12:00 UTC Mar-Nov) - # Using 12:00 UTC to cover EDT; during EST this runs at 7 AM still fine + # 12:00 UTC on weekdays (7 AM EST / 8 AM EDT). - cron: "0 12 * * 1-5" workflow_dispatch: inputs: @@ -27,8 +26,14 @@ jobs: with: python-version: "3.12" + - name: Install uv + uses: astral-sh/setup-uv@v5 + with: + enable-cache: true + - name: Install dependencies - run: pip install -r skills/signal-sweep/requirements.txt + working-directory: skills/signal-sweep + run: uv sync --locked - name: Run insider scan env: @@ -40,16 +45,10 @@ jobs: LOOKBACK="${{ github.event.inputs.lookback || '5' }}" ZSCORE="${{ github.event.inputs.zscore || '1.5' }}" - WEBHOOK_ARG="" - if [ -n "$DISCORD_WEBHOOK_URL" ]; then - WEBHOOK_ARG="--webhook $DISCORD_WEBHOOK_URL" - fi - - python scripts/scan_insiders.py \ + uv run python scripts/scan_insiders.py \ --date "$DATE" \ --lookback "$LOOKBACK" \ - --zscore "$ZSCORE" \ - $WEBHOOK_ARG + --zscore "$ZSCORE" - name: Upload scan results if: always() diff --git a/.markdownlint-cli2.jsonc b/.markdownlint-cli2.jsonc index 9e32d27..13d7e2a 100644 --- a/.markdownlint-cli2.jsonc +++ b/.markdownlint-cli2.jsonc @@ -11,7 +11,9 @@ }, "ignores": [ ".venv", + "**/.venv/**", "node_modules", + "**/node_modules/**", ".git", "**/sec-cache/**", "**/signal-sweep-cache/**", diff --git a/skills/README.md b/skills/README.md index 12396a3..cc6a67d 100644 --- a/skills/README.md +++ b/skills/README.md @@ -54,62 +54,8 @@ the skill that owns them. Load only the guide or script required for the current - [`bottom-up-analyst`](bottom-up-analyst/) contains valuation arithmetic and memo frameworks. - [`pitch-like-lou`](pitch-like-lou/) contains the pitch-writing workflow and reference corpus. -Each skill's README documents its own dependencies and usage. Runtime caches are generated -next to the relevant workspace and are git-ignored. - -## Current snapshot - -The following commands measure the entry-point surface and the explicitly referenced skill -resources from the repository root: - -```bash -cloc --by-file --include-lang=Markdown \ - skills/bottom-up-analyst/SKILL.md \ - skills/pitch-like-lou/SKILL.md \ - skills/sec-edgar-skill/SKILL.md \ - skills/signal-sweep/SKILL.md \ - skills/market-scout/SKILL.md -``` - -```bash -cloc \ - skills/bottom-up-analyst/SKILL.md \ - skills/bottom-up-analyst/references/memo_template.md \ - skills/bottom-up-analyst/references/guide_normalization.md \ - skills/bottom-up-analyst/references/guide_competitive.md \ - skills/bottom-up-analyst/references/guide_valuation.md \ - skills/bottom-up-analyst/references/guide_ownership_signals.md \ - skills/bottom-up-analyst/references/archetypes/*.md \ - skills/bottom-up-analyst/scripts/dcf.py \ - skills/bottom-up-analyst/scripts/epv.py \ - skills/market-scout/SKILL.md \ - skills/market-scout/scripts/fetch_market_data.py \ - skills/market-scout/scripts/fetch_transcripts.py \ - skills/pitch-like-lou/SKILL.md \ - skills/pitch-like-lou/references/corpus/*.md \ - skills/sec-edgar-skill/SKILL.md \ - skills/sec-edgar-skill/references/guide_core.md \ - skills/sec-edgar-skill/references/guide_filings.md \ - skills/sec-edgar-skill/references/guide_financials.md \ - skills/sec-edgar-skill/references/guide_ownership.md \ - skills/sec-edgar-skill/references/guide_proxy.md \ - skills/sec-edgar-skill/references/guide_holdings.md \ - skills/sec-edgar-skill/scripts/orient.py \ - skills/sec-edgar-skill/scripts/fetch_filing.py \ - skills/sec-edgar-skill/scripts/fetch_filings.py \ - skills/sec-edgar-skill/scripts/parse_financials.py \ - skills/sec-edgar-skill/scripts/list_headings.py \ - skills/sec-edgar-skill/scripts/fetch_insider_trades.py \ - skills/sec-edgar-skill/scripts/fetch_13f_holders.py \ - skills/sec-edgar-skill/scripts/test_setup.py \ - skills/signal-sweep/SKILL.md \ - skills/signal-sweep/screens.json \ - skills/signal-sweep/references/guide_screens.md \ - skills/signal-sweep/scripts/scan_insiders.py \ - skills/signal-sweep/scripts/scan_market.py \ - skills/signal-sweep/scripts/search_themes.py \ - skills/signal-sweep/scripts/scan_conferences.py -``` +Each skill's README documents its dependencies and usage. Invoke installed artifact-producing +scripts from the research workspace so their git-ignored runtime caches stay beside the work. ## Research scope diff --git a/skills/market-scout/README.md b/skills/market-scout/README.md index 011a365..2093f24 100644 --- a/skills/market-scout/README.md +++ b/skills/market-scout/README.md @@ -1,41 +1,41 @@ # Market Scout -Pull public market data — price, market cap, trailing returns, peers, and sector screens — for -US-listed stocks via [`yfinance`](https://github.com/ranaroussi/yfinance). An unopinionated data -layer: it surfaces facts and rankings; it decides nothing. +Public market context and earnings-call transcript retrieval for US-listed stocks via +[`yfinance`](https://github.com/ranaroussi/yfinance) and Yahoo Finance. The skill returns source +material; it does not decide whether a security is attractive. Part of the [SecStack skills](../README.md) collection. -## Installation (do this first) +## Setup -Install this skill's dependencies and the browser runtime before using the scripts: +The SecStack bootstrap installs this skill's dependencies into the profile environment. For +standalone use: ```bash -uv sync -# The SecStack bootstrap installs agent-browser into the isolated Pi profile. -# Run this once from an active SecStack profile: -agent-browser install +uv sync --project "" ``` -- `yfinance`/`pandas` power market/peer data. -- `agent-browser` is required for earnings-call transcript pages (JS-rendered Yahoo/Quartr). -- In the packaged SecStack profile, its Pi-managed binary is already on PATH. +Activate the resulting environment or prefix script commands with +`uv run --project ""`. -## Use +Transcript retrieval also uses the Pi-managed `agent-browser` binary. Install its browser runtime +once from an active SecStack profile: ```bash -python scripts/fetch_market_data.py --ticker AAPL --peers +agent-browser install ``` -## Earnings call transcripts (Yahoo + Quartr) +## Examples + +Keep the research workspace as the current directory and invoke the installed scripts by their +resolved paths so generated transcript cache stays with the research: ```bash -python scripts/fetch_transcripts.py --ticker AAPL --list -python scripts/fetch_transcripts.py --ticker AAPL --latest 1 +python "/scripts/fetch_market_data.py" --ticker AAPL --peers +python "/scripts/fetch_transcripts.py" --ticker AAPL --list +python "/scripts/fetch_transcripts.py" --ticker AAPL --latest 1 ``` -yfinance offers far more than the script wraps and is self-documenting (`dir()`, `help()`, -`t.info.keys()`) — see [SKILL.md](SKILL.md) for how to discover and construct what you need. - -Market data is best-effort and occasionally stale or missing for thinly-covered names; confirm -anything load-bearing against a primary source. +See [SKILL.md](SKILL.md) for routing, output contracts, and runtime discovery beyond the bundled +report fields. Yahoo data is best-effort and can be stale or incomplete for thinly covered names; +verify load-bearing facts against an issuer or regulatory source. diff --git a/skills/market-scout/SKILL.md b/skills/market-scout/SKILL.md index 98ad7df..cc552a6 100644 --- a/skills/market-scout/SKILL.md +++ b/skills/market-scout/SKILL.md @@ -1,81 +1,85 @@ --- name: market-scout description: >- - Pull public market data for US-listed stocks via Yahoo Finance: price, market cap, shares, - 52-week range, trailing returns, sector/industry peer tables and pre-ranked screens, and - earnings call transcripts. Use this whenever a task needs a quick market snapshot or quote - for a ticker, trailing performance, a company's peer set, a sector/theme shortlist, or - earnings call transcripts — e.g. "what's the price/return on X", "who are X's peers", - "best-performing names in this industry", "get me the latest earnings call", or turning a - theme into a concrete list of tickers. This is an unopinionated data layer; it does not - decide what is cheap, good, or worth buying. + Retrieve current public market data and earnings-call transcripts for US-listed stocks via + Yahoo Finance: quotes, market capitalization, shares, price history and trailing returns, + industry context, peer tables, and transcript text. Use when a task needs a current market + snapshot, performance calculation, Yahoo peer or industry data, a market-data field, or an + earnings-call transcript. Use primary filings for facts that Yahoo does not author or when a + load-bearing figure needs regulatory verification. --- # Market Scout -Pull **public market data**, **sector/peer screening**, and **earnings call transcripts** -for US-listed stocks. Unopinionated: it surfaces prices, returns, peers, rankings, and -management commentary; it does not decide what is cheap or worth buying — leave that to -whatever framework is driving. +Retrieve market context and transcript text without turning the result into an investment +conclusion. -## Setup (install first) +## Runtime and paths -Install this skill's dependencies from this directory before running any script: +Resolve bundled paths relative to this `SKILL.md`. Invoke scripts by absolute path while keeping +the shell working directory at the research workspace. This puts the default +`./transcript-cache` beside the work rather than inside the installed skill; alternatively pass +`--cache-dir`. -```bash -uv sync -# In the packaged SecStack profile, agent-browser is installed by the bootstrap. -agent-browser install -``` +The packaged SecStack profile already exposes this skill's Python dependencies. For standalone +use, run `uv sync --project ""`, then either activate that environment or prefix the +examples below with `uv run --project ""`. `fetch_transcripts.py` also requires the +`agent-browser` binary and its one-time browser installation (`agent-browser install`). No API key +or SEC identity is required. -- `agent-browser` is required for earnings-call transcripts (`fetch_transcripts.py`). -- No API key or identity is needed — Yahoo Finance is public. +## Choose a route -## Market snapshot and peers +### Snapshot, returns, industry, and peers -`fetch_market_data.py` prints a compact Markdown summary to stdout — price, market cap, -shares outstanding, 52-week range, trailing returns, and (optionally) industry overview and -peer tables. Output is live and never cached. `--help` is the authoritative flag reference: +`fetch_market_data.py` prints a live Markdown report to stdout. It does not cache time-sensitive +market data. ```bash -python scripts/fetch_market_data.py --ticker AAPL --industry --peers +python "/scripts/fetch_market_data.py" --ticker AAPL +python "/scripts/fetch_market_data.py" --ticker AAPL --industry --peers ``` -## Earnings call transcripts +The default report includes common trailing windows available inside the requested history period. +Use `--period` to change how much history is fetched. For a different interval or field, use the +runtime-discovery route below rather than treating the bundled report as Yahoo's full schema. + +### Earnings-call transcripts -`fetch_transcripts.py` scrapes Yahoo Finance's Quartr-powered transcript pages via -`agent-browser` (Yahoo requires JS rendering). It lists available transcripts or downloads -them as LLM-friendly Markdown to `transcript-cache//transcripts/`. Files are -named `Q3-FY2026.md` etc., cached and reused across runs. `--help` for all flags: +List the periods Yahoo currently exposes, then request the exact period or latest count needed: ```bash -python scripts/fetch_transcripts.py --ticker AAPL --list # list available -python scripts/fetch_transcripts.py --ticker AAPL --latest 1 # most recent -python scripts/fetch_transcripts.py --ticker AAPL --year 2025 # full fiscal year -python scripts/fetch_transcripts.py --ticker AAPL --quarter Q3 --year 2025 +python "/scripts/fetch_transcripts.py" --ticker AAPL --list +python "/scripts/fetch_transcripts.py" --ticker AAPL --latest 1 +python "/scripts/fetch_transcripts.py" --ticker AAPL --year 2025 +python "/scripts/fetch_transcripts.py" --ticker AAPL --quarter Q3 --year 2025 ``` -Each cached file has a summary, `## Prepared Remarks` with `### Speaker — Title` headings, -and a `## Q&A` section — greppable by speaker name, "guidance", "margin", or any keyword. +Downloads are written to `//transcripts/Q3-FY2026.md` and reused on later runs. +Artifact-producing mode emits one absolute path per completed transcript to stdout and diagnostics +to stderr. A partial or total retrieval failure exits nonzero rather than masquerading as an empty +period. -## Beyond the bundled scripts — yfinance is self-documenting +Each file preserves the Yahoo source URL and, when present, separates prepared remarks from Q&A. +Transcript text and speaker attribution are third-party data; verify a consequential quote against +the issuer's own transcript, webcast, or filing when available. -The scripts above wrap the common cases. yfinance exposes far more (financials, holders, -options, earnings dates, calendar, sector/industry screens, …). When a task needs something -the scripts don't cover, **discover at runtime** rather than guessing field names: +## Runtime discovery beyond the wrappers + +The bundled scripts cover frequent jobs, not the limits of `yfinance`. Inspect the installed API +instead of guessing field names or assuming a fixed metric template: ```python import yfinance as yf -t = yf.Ticker("AAPL") - -print([a for a in dir(t) if not a.startswith("_")]) # all attributes/methods -list(t.info.keys()) # every field in the snapshot +ticker = yf.Ticker("AAPL") +print([name for name in dir(ticker) if not name.startswith("_")]) +print(sorted((ticker.info or {}).keys())) +help(ticker.history) -# Sector/industry screening (theme -> shortlist): -ind = yf.Industry(t.info["industryKey"]) -print([a for a in dir(ind) if not a.startswith("_")]) # top_companies, overview, ... +industry = yf.Industry((ticker.info or {})["industryKey"]) +print([name for name in dir(industry) if not name.startswith("_")]) ``` -When a field or method isn't what you expected, `dir()` / `help()` / `.info.keys()` recover -the answer inline — prefer that over guessing. +Use this route for calendars, options, holders, financial tables, custom return windows, or other +Yahoo fields. Report the field name, period, units, and retrieval date. Treat missing or stale data +as missing; do not silently substitute a different field or period. diff --git a/skills/market-scout/scripts/_common.py b/skills/market-scout/scripts/_common.py index 4c5992f..8d8548b 100644 --- a/skills/market-scout/scripts/_common.py +++ b/skills/market-scout/scripts/_common.py @@ -1,17 +1,30 @@ -"""Minimal shared runtime setup for market-scout scripts.""" - -import sys - -# yfinance/pandas can emit non-ASCII text (e.g. company names like "Société"); -# force UTF-8 so a Windows cp1252 console doesn't raise UnicodeEncodeError. -if sys.platform.startswith("win"): - for _stream in (sys.stdout, sys.stderr): - try: - _stream.reconfigure(encoding="utf-8") - except Exception: - pass - - -def log(msg: str) -> None: - """Progress/diagnostics -> stderr (keeps stdout clean for the result).""" - print(msg, file=sys.stderr, flush=True) +"""Shared runtime and output helpers for market-scout scripts.""" + +import os +import sys +from pathlib import Path + +# Provider data can contain non-ASCII company and speaker names. +if sys.platform.startswith("win"): + for _stream in (sys.stdout, sys.stderr): + try: + _stream.reconfigure(encoding="utf-8") + except Exception: + pass + +try: + import truststore + + truststore.inject_into_ssl() +except Exception: + pass + + +def log(msg: str) -> None: + """Write human-readable progress to stderr.""" + print(msg, file=sys.stderr, flush=True) + + +def emit(path: str | os.PathLike) -> None: + """Write one absolute artifact path to stdout.""" + print(str(Path(path).resolve()), flush=True) diff --git a/skills/market-scout/scripts/fetch_market_data.py b/skills/market-scout/scripts/fetch_market_data.py index 57df703..d34f8ca 100644 --- a/skills/market-scout/scripts/fetch_market_data.py +++ b/skills/market-scout/scripts/fetch_market_data.py @@ -1,109 +1,155 @@ -"""Fetch market data via yfinance: price snapshot, trailing returns, peers. - -The core tool of the market-scout skill (Yahoo Finance). Output is a compact -Markdown summary printed to stdout — market data is small and time-sensitive, so -it is never cached to disk. Two modes compose: a per-name snapshot, and (with -``--industry`` / ``--peers``) a sector view useful as a discovery top-of-funnel. -Run ``--help`` for all flags. -""" - -import argparse -import sys - -import _common as c - -_RETURN_WINDOWS = [("1M", 30), ("3M", 91), ("6M", 182), ("1Y", 365), ("3Y", 1095), ("5Y", 1825)] - - -def _num(v, money=False, pct=False): - if v is None: - return "n/a" - try: - if pct: - return f"{v:+.1f}%" - if money: - return f"{v:,.0f}" - return f"{v:,.2f}" - except Exception: - return str(v) - - -def main(): - p = argparse.ArgumentParser(description=__doc__.splitlines()[0]) - p.add_argument("--ticker", required=True, help="Stock ticker (Yahoo symbol).") - p.add_argument("--period", default="5y", help="History window for returns (default: 5y).") - p.add_argument("--industry", action="store_true", help="Also print the industry overview.") - p.add_argument("--peers", action="store_true", help="Also list industry peers by weight.") - args = p.parse_args() - - try: - import pandas as pd - import yfinance as yf - except Exception as exc: - c.log(f"ERROR: yfinance/pandas not installed: {exc}") - sys.exit(1) - - ticker = yf.Ticker(args.ticker) - try: - info = ticker.info or {} - except Exception as exc: - c.log(f"WARNING: could not fetch .info: {exc}") - info = {} - - print(f"# Market data: {args.ticker.upper()}\n") - print("## Snapshot") - print(f"- Price: {_num(info.get('currentPrice'))}") - print(f"- Market cap: {_num(info.get('marketCap'), money=True)}") - print(f"- Shares outstanding: {_num(info.get('sharesOutstanding'), money=True)}") - print( - f"- 52-week high / low: {_num(info.get('fiftyTwoWeekHigh'))} / " - f"{_num(info.get('fiftyTwoWeekLow'))}" - ) - if info.get("sector") or info.get("industry"): - print( - f"- Sector / industry: {info.get('sector') or 'n/a'} / {info.get('industry') or 'n/a'}" - ) - - try: - close = ticker.history(period=args.period)["Close"].dropna() - if len(close) > 1: - last, last_date = close.iloc[-1], close.index[-1] - print("\n## Trailing returns") - for label, days in _RETURN_WINDOWS: - prior = close[close.index <= last_date - pd.Timedelta(days=days)] - if len(prior): - print(f"- {label}: {_num((last / prior.iloc[-1] - 1) * 100, pct=True)}") - except Exception as exc: - c.log(f"WARNING: returns unavailable: {exc}") - - if args.industry or args.peers: - industry_key = info.get("industryKey") - if not industry_key: - c.log("WARNING: no industryKey in .info; cannot fetch industry/peers.") - else: - try: - industry = yf.Industry(industry_key) - if args.industry: - overview = industry.overview or {} - print(f"\n## Industry: {getattr(industry, 'name', industry_key)}") - if overview.get("market_cap"): - print(f"- Total market cap: {_num(overview['market_cap'], money=True)}") - if overview.get("companies_count"): - print(f"- Companies: {overview['companies_count']}") - if overview.get("description"): - print(f"- {overview['description']}") - if args.peers: - top = industry.top_companies - if top is not None and len(top): - print("\n## Top peers by market weight") - head = top.head(15) - try: - print(head.to_markdown()) - except Exception: - print(head.to_string()) - except Exception as exc: - c.log(f"WARNING: industry/peers unavailable: {exc}") - - -if __name__ == "__main__": - main() +"""Fetch a Yahoo Finance snapshot, trailing returns, industry, and peers. + +The report is printed to stdout because market data is compact and time-sensitive. +Run ``--help`` for the flag reference. +""" + +from __future__ import annotations + +import argparse +import sys + +import _common as c + +_RETURN_WINDOWS = [("1M", 30), ("3M", 91), ("6M", 182), ("1Y", 365), ("3Y", 1095), ("5Y", 1825)] + + +def _num(value, *, money: bool = False, pct: bool = False) -> str: + if value is None: + return "n/a" + try: + if pct: + return f"{value:+.1f}%" + if money: + return f"{value:,.0f}" + return f"{value:,.2f}" + except (TypeError, ValueError): + return str(value) + + +def _first(mapping: dict, *keys: str): + for key in keys: + value = mapping.get(key) + if value is not None: + return value + return None + + +def main() -> None: + """Run the market-data CLI.""" + parser = argparse.ArgumentParser(description=__doc__.splitlines()[0]) + parser.add_argument("--ticker", required=True, help="Stock ticker (Yahoo symbol).") + parser.add_argument("--period", default="5y", help="History period for returns (default: 5y).") + parser.add_argument("--industry", action="store_true", help="Also print the industry overview.") + parser.add_argument( + "--peers", action="store_true", help="Also print Yahoo's industry peer table." + ) + args = parser.parse_args() + + try: + import pandas as pd + import yfinance as yf + except Exception as exc: + c.log(f"ERROR: yfinance/pandas not installed: {exc}") + sys.exit(1) + + symbol = args.ticker.upper() + ticker = yf.Ticker(symbol) + + try: + info = ticker.info or {} + except Exception as exc: + c.log(f"WARNING: Yahoo snapshot unavailable: {exc}") + info = {} + + close = None + history_error = None + try: + history = ticker.history(period=args.period) + if history is not None and "Close" in history: + close = history["Close"].dropna() + except Exception as exc: + history_error = exc + + snapshot_fields = ( + "currentPrice", + "regularMarketPrice", + "marketCap", + "sharesOutstanding", + "fiftyTwoWeekHigh", + "fiftyTwoWeekLow", + ) + has_snapshot = any(info.get(field) is not None for field in snapshot_fields) + if not has_snapshot and (close is None or close.empty): + detail = f": {history_error}" if history_error else "" + c.log(f"ERROR: Yahoo returned no usable snapshot or price history for {symbol}{detail}") + sys.exit(1) + if history_error: + c.log(f"WARNING: price history unavailable: {history_error}") + + price = _first(info, "currentPrice", "regularMarketPrice") + if price is None and close is not None and not close.empty: + price = close.iloc[-1] + currency = _first(info, "currency", "financialCurrency") + + print(f"# Market data: {symbol}\n") + print("## Snapshot") + price_label = _num(price) + if currency: + price_label += f" {currency}" + print(f"- Price: {price_label}") + print(f"- Market cap: {_num(info.get('marketCap'), money=True)}") + print(f"- Shares outstanding: {_num(info.get('sharesOutstanding'), money=True)}") + print( + f"- 52-week high / low: {_num(info.get('fiftyTwoWeekHigh'))} / " + f"{_num(info.get('fiftyTwoWeekLow'))}" + ) + if info.get("sector") or info.get("industry"): + print( + f"- Sector / industry: {info.get('sector') or 'n/a'} / {info.get('industry') or 'n/a'}" + ) + + if close is not None and len(close) > 1: + last = close.iloc[-1] + last_date = close.index[-1] + rendered = [] + for label, days in _RETURN_WINDOWS: + prior = close[close.index <= last_date - pd.Timedelta(days=days)] + if len(prior): + rendered.append(f"- {label}: {_num((last / prior.iloc[-1] - 1) * 100, pct=True)}") + if rendered: + print(f"\n## Trailing returns (through {last_date.date()})") + print("\n".join(rendered)) + + if args.industry or args.peers: + industry_key = info.get("industryKey") + if not industry_key: + c.log("WARNING: Yahoo returned no industryKey; industry and peers are unavailable.") + return + try: + industry = yf.Industry(industry_key) + if args.industry: + overview = industry.overview or {} + print(f"\n## Industry: {getattr(industry, 'name', industry_key)}") + if overview.get("market_cap") is not None: + print(f"- Total market cap: {_num(overview['market_cap'], money=True)}") + if overview.get("companies_count") is not None: + print(f"- Companies: {overview['companies_count']}") + if overview.get("description"): + print(f"- {overview['description']}") + if args.peers: + top = industry.top_companies + if top is not None and len(top): + print("\n## Yahoo industry peers") + try: + print(top.to_markdown()) + except Exception: + print(top.to_string()) + else: + c.log("WARNING: Yahoo returned no industry peers.") + except Exception as exc: + c.log(f"WARNING: industry/peers unavailable: {exc}") + + +if __name__ == "__main__": + main() diff --git a/skills/market-scout/scripts/fetch_transcripts.py b/skills/market-scout/scripts/fetch_transcripts.py index 0e1b116..73dea09 100644 --- a/skills/market-scout/scripts/fetch_transcripts.py +++ b/skills/market-scout/scripts/fetch_transcripts.py @@ -130,21 +130,44 @@ def _run_browser_script(script: str, timeout: int = 45) -> str: # --------------------------------------------------------------------------- -def _fetch_listing(ticker: str) -> list[dict]: - """Fetch the list of available transcripts for a ticker.""" +def _fetch_listing(ticker: str) -> list[dict] | None: + """Fetch available transcripts, or return None when retrieval fails.""" url = f"{_BASE}/quote/{ticker}/earnings-calls/" if not _open_url(url, timeout=45): - return [] + return None time.sleep(3) - script = 'JSON.stringify(Array.from(document.querySelectorAll(\'a[href*="earnings_call"]\')).map(a=>({url:a.getAttribute("href"),title:(a.textContent||"").trim()||a.getAttribute("aria-label")||a.getAttribute("href")})).filter(x=>x.url).filter((x,i,arr)=>arr.findIndex(y=>y.url===x.url)===i))' + script = r""" +JSON.stringify((() => { + const links = Array.from(document.querySelectorAll('a[href*="earnings_call"]')) + .map(a => ({ + url: a.getAttribute("href"), + title: (a.textContent || "").trim() || a.getAttribute("aria-label") || a.getAttribute("href") + })) + .filter(x => x.url) + .filter((x, i, all) => all.findIndex(y => y.url === x.url) === i); + const text = ((document.querySelector("main") || document.body).innerText || "").slice(0, 5000); + return { links, text }; +})()) +""" raw = _run_browser_script(script) if not raw: - return [] + return None try: - return json.loads(raw) + payload = json.loads(raw) + if not isinstance(payload, dict) or not isinstance(payload.get("links"), list): + return None + if payload["links"]: + return payload["links"] + text = str(payload.get("text", "")) + if re.search( + r"no (?:earnings call )?transcripts|transcripts? (?:are )?not available", text, re.I + ): + return [] + c.log("ERROR: Yahoo loaded without a transcript list or an explicit no-data message.") + return None except json.JSONDecodeError: - c.log("WARNING: could not parse listing response.") - return [] + c.log("ERROR: could not parse Yahoo's transcript listing response.") + return None # --------------------------------------------------------------------------- @@ -297,11 +320,19 @@ def main(): ) args = p.parse_args() + if args.latest < 0: + p.error("--latest must be zero or greater.") + if args.quarter and not re.fullmatch(r"Q[1-4]", args.quarter.upper()): + p.error("--quarter must be Q1, Q2, Q3, or Q4.") + ticker = args.ticker.upper() # Step 1: Fetch listing c.log(f"Fetching transcript listing for {ticker}...") transcripts = _fetch_listing(ticker) + if transcripts is None: + c.log(f"ERROR: transcript listing retrieval failed for {ticker}.") + sys.exit(1) if not transcripts: c.log(f"No earnings call transcripts found for {ticker}.") sys.exit(0) @@ -333,6 +364,8 @@ def main(): root = _cache_root(args.cache_dir) out_dir = _transcript_dir(root, ticker) c.log(f"Saving to: {out_dir}") + completed = 0 + failed = 0 for i, t in enumerate(transcripts): fname = _filename_from_title(t["title"]) @@ -340,13 +373,15 @@ def main(): if out_path.exists(): c.log(f" [{i + 1}/{len(transcripts)}] Already cached: {fname}") - print(str(out_path.resolve())) + c.emit(out_path) + completed += 1 continue c.log(f" [{i + 1}/{len(transcripts)}] Downloading: {t['title']}") data = _fetch_transcript(t["url"]) if not data or not data.get("blocks"): - c.log(" WARNING: no transcript content found, skipping.") + c.log(" ERROR: no transcript content found, skipping.") + failed += 1 continue md = _render_markdown(ticker, t["url"], data) @@ -354,12 +389,16 @@ def main(): Path(out_path).write_text(md, encoding="utf-8") block_count = len(data.get("blocks", [])) c.log(f" Saved: {fname} ({block_count} speaker turns)") - print(str(out_path.resolve())) + c.emit(out_path) + completed += 1 if i < len(transcripts) - 1: time.sleep(_DELAY) - c.log("Done.") + if failed: + c.log(f"ERROR: {failed} transcript(s) failed; {completed} completed or cached.") + sys.exit(1) + c.log(f"Done: {completed} transcript(s) completed or cached.") if __name__ == "__main__": diff --git a/skills/pitch-like-lou/README.md b/skills/pitch-like-lou/README.md index a1d872f..b498080 100644 --- a/skills/pitch-like-lou/README.md +++ b/skills/pitch-like-lou/README.md @@ -1,43 +1,37 @@ -# Pitch Like Norbert Lou - -An [agent skill](SKILL.md) that teaches an LLM to **investigate and write an investment -pitch the way Norbert Lou** (username `charlie479` on Value Investors Club) did — the analyst -whose NVR, Quilmes, and Winmill write-ups Joel Greenblatt handed out to his students as -exemplars of high-conviction value investing. - -The skill is deliberately scoped to two reproducible things: a **disposition** (where value -tends to hide, what to read, what to normalize) and a **voice** (how Lou structures and argues -a pitch, and the temperament that makes it credible). It does **not** invent a thesis — the raw -material comes from SEC-filing and market-data tools (it composes naturally with the -[`sec-edgar-skill`](../sec-edgar-skill/) and [`market-scout`](../market-scout/) data skills, and -with [`bottom-up-analyst`](../bottom-up-analyst/) as the framework), and the actual analytical -insight comes from reasoning over real filings. See [SKILL.md](SKILL.md) for the full design, or -the [stack overview](../README.md) for how the skills fit together. - -## Layout - -- `SKILL.md` — the skill itself (the only file an agent needs to load). -- `references/corpus/` — the seven primary-source pitches, used for voice calibration. They ship - with the skill; an agent greps them for a specific rhetorical move rather than loading them whole. - -## A note on the corpus - -`references/corpus/` contains Markdown extractions of seven Value Investors Club write-ups (and -their public discussion threads) authored by `charlie479`: - -| Pitch | Shape | -|---|---| -| NVR, Sportsman's Guide | quality compounder | -| Winmill | cigar-butt asset play | -| Quilmes, MCI, NII Holdings, Telemig | special situation / structural arbitrage | - -These are **legacy ideas (2001–2009)** that VIC itself makes publicly available after a 45-day -delay, and which have circulated freely for years (they are widely reproduced verbatim — e.g., -across investing newsletters and Substacks — and were distributed in classrooms by Joel -Greenblatt). They are archived here **solely** as a small, fixed reference set for an educational -tool, with **no commercial use** intended. - -Copyright in the underlying write-ups remains with their respective authors and Value Investors -Club. This repository is a non-commercial, educational project and is not affiliated with or -endorsed by Value Investors Club. If you are a rights holder and would prefer a pitch not be -included, please open an issue or contact the maintainer and it will be removed promptly. +# Pitch Like Norbert Lou + +An [agent skill](SKILL.md) for rendering an already-researched investment thesis as a concise, +numbers-first pitch inspired by Norbert Lou's Value Investors Club writing. It focuses on the +transferable craft: document-level specificity, reproducible arithmetic, a fair statement of the +objection, and candor about weak evidence. + +The skill is a presentation layer. It does not source a company, form a thesis, or confer conviction +on incomplete work. Within [SecStack](../README.md), [`bottom-up-analyst`](../bottom-up-analyst/) +owns the research and valuation workflow; [`sec-edgar-skill`](../sec-edgar-skill/) and +[`market-scout`](../market-scout/) retrieve source material. + +## Layout + +- `SKILL.md` — the rendering workflow, voice guidance, and evidence pass. +- `references/corpus/` — seven primary-source pitches and public discussion threads for targeted + style calibration. The skill tells the agent when and how narrowly to consult them. + +## Corpus + +The corpus contains Markdown extractions of Value Investors Club write-ups authored by +`charlie479`: + +| Pitch | Broad situation | +|---|---| +| NVR, Sportsman's Guide | Operating business / compounder | +| Winmill | Asset discount | +| Quilmes, MCI, NII Holdings, Telemig | Special situation / capital structure | + +They are legacy ideas from 2001–2009, not current research or templates whose facts and metrics +should be copied into a new pitch. They are included as a small educational reference set; an +agent should retrieve only the passage needed to calibrate a specific writing move. + +Copyright in the underlying write-ups remains with their respective authors and Value Investors +Club. This repository is a non-commercial educational project and is not affiliated with or +endorsed by Value Investors Club. A rights holder may open an issue or contact the maintainer to +request removal. diff --git a/skills/pitch-like-lou/SKILL.md b/skills/pitch-like-lou/SKILL.md index 17b4ec1..d08c2d8 100644 --- a/skills/pitch-like-lou/SKILL.md +++ b/skills/pitch-like-lou/SKILL.md @@ -1,203 +1,103 @@ ---- -name: pitch-like-lou -description: >- - Write or rewrite a stock pitch the way Norbert Lou did in his legendary Value Investors Club - write-ups — numbers-first, high-conviction value investing. Use this skill whenever the user - wants to make the bull or bear case for a stock, draft an investment thesis or write-up, - argue a long or short in a value-investing voice, or whenever Norbert Lou or Value Investors - Club (VIC) is mentioned. Also use it when asked to "pitch me on X", "write up this stock", - "convince me this is a buy/sell", "make the case for X", or render a finished thesis as a - short, compelling pitch. Pairs with SEC-filing tools for the raw numbers. ---- - -# Pitch Like Norbert Lou - -This skill teaches the *craft* behind Norbert Lou's pitches: where he dug, what he refused to -ignore, and how he wrote the verdict so plainly that a skeptic kept reading. It is a lens and a -voice — not an idea generator. - -Be honest with yourself about what makes those pitches great. Read enough of them and the same -truth keeps surfacing: the edge is **situation-specific forensics on primary documents** (the -put/call formula buried in Schedule 1.04 of a Quinsa exhibit; a $750M pink-sheet preferred that -sits *senior* to $30B of WorldCom bonds; a subscription liability that makes GAAP understate real -earnings). No checklist produces those. What a skill *can* carry is the disposition that points -you at them and the voice that renders them. That is this skill's job. The insight itself comes -from you, reasoning over the actual filings. - -## What this is — and is not - -- **It is:** a *disposition* (where value tends to hide, what to read, what to normalize) welded - to a *voice* (how Lou structures and argues a pitch, and the temperament that makes it credible). -- **It is not:** a screener, an "analysis engine," or a promise that a stock is good. It will not - hand you a thesis, and it must never **manufacture conviction you have not earned by doing the work.** -- **The seam:** the raw material — financials, footnotes, proxies, ownership, deal documents — comes - from your SEC-filing tools (the `sec-edgar-skill` skill's `fetch_filing` / `parse_financials`), - and price / peer data from the `market-scout` skill's `fetch_market_data`. This skill tells you - *which* of those to pull and *why*; it does not re-document them. Keep the tools unopinionated; - keep the opinion here. - -## The one inviolable rule: never let the voice outrun the evidence - -Lou's prose is persuasive *because* it is backed by forensic work and radical honesty about the -weak spots. Borrow his cadence and confidence onto a thesis you have not investigated and you have -built a machine for sounding right while being wrong — the exact failure this skill exists to avoid. - -So the temperament below is not decoration. It is the safety mechanism. If you have not done the -digging, **say so and dial the conviction down.** A Lou pitch with hedges is still a Lou pitch; a -Lou pitch with borrowed certainty is propaganda. - -A cautionary tale from the corpus itself: in the MCI QUIPS pitch — a brilliant structural-seniority -thesis — Lou relayed *third-hand* that the books were clean with no material intercompany payables. -A forensic accountant later surfaced ~$24 billion of intercompany claims that nearly sank the trade -on a substantive-consolidation fight; the position was down ~68% at the trough before it resolved. -Even his best work had a blind spot at exactly the point where he leaned on something he had not -verified himself. The lesson is not "be timid" — it's *know which of your claims you actually -checked, mark the ones you didn't, and follow through honestly when a thesis turns against you.* - -## The loop - -1. **Classify the situation.** Almost every Lou idea is one of three shapes: - *quality compounder*, *cigar-butt asset play*, or *special-situation / structural arbitrage*. - The shape decides where you dig and which parts of the pitch carry the weight. -2. **Dig where that shape pays** — using your SEC tools. See **Where to dig**. -3. **Try to kill it.** Lou posts maybe one idea a year; the discipline is mostly *saying no*. Run - the disqualifiers before you fall in love. Most candidates die here, and that is the system working. -4. **Render** it in the structure and voice below — but only at the conviction your digging supports. - -## The voice — the part that is actually him - -Each move below earns its place; the parenthetical is *why* it works. Short quotes are illustrative. - -- **Lead with the punchline, usually a number.** No "I am pleased to present." Open on the - disconnect. (*"Winmill & Company has net cash of $3.91. The stock trades for $1.70."* The reader - is hooked before any throat-clearing could lose them.) -- **Pre-empt the biggest objection in the second sentence, then invert it.** (*"Yes, pink sheet - equities in bankrupt companies are worthless 99.9% of the time. There are special features, - however…"* — disarming the obvious dismissal buys you the rest of the page.) -- **Plain, declarative, numbered. Define your terms inline.** Spell out *FCF = operating cash flow - minus capex*; *net cash = cash and securities minus ALL liabilities*. (Precision is itself an - argument; it signals you did the work and pre-empts a fight over definitions.) -- **Earn conviction through honesty, not adjectives.** Separate *what you know* from *what you - believe*. Concede the weak points out loud. When governance is ugly — option grants, insider - pay — do not wave it away: **estimate the cost of the "value-destructive" actions, subtract it - from your value, and demand a margin of safety that survives anyway.** (*"His mansion was paid - for with wheelbarrows of liberally-issued stock options"* — admitting the flaw is what makes the - bullishness believable.) -- **Signal credibility by being unimpressed with yourself.** Post rarely; admit when your own price - target was laughably low; never hype. (*"I included a target of $28.10… so it shows how much I - know"* — self-deprecation builds trust faster than a swagger ever could.) -- **Answer a risk with judo, not denial.** Acknowledge the *real* version of the objection, then - show it applies equally to something the skeptic already loves. (On emerging-market currency fear: - isn't it curious nobody asks about the same dollar-liability/foreign-revenue mismatch at Coke?) -- **Use one homely analogy to collapse a hard idea.** A deep discount to liquid assets behind a - controlling family becomes *a money-market fund holding $1.50 in cash, locked up for five years — - would you buy it for 40 cents?* Spectrum licenses become *cable-franchise or broadcast rights.* - (One concrete image does the work of a page of exposition.) -- **Intrinsic value only; ignore the chart.** A price decline is not a warning, it is a cheaper - entry. Never reason from the tape. -- **Quiet wit, used once.** (*"It's this fancy new thing called the internet."*) Sparingly — it - flatters the reader and mocks the market's blind spot without breaking the analytical tone. - -## The structure — a shape that serves the argument, not a template to fill - -Use the skeleton, but let the *situation* decide where the weight goes. A cigar-butt leans on -questions 1 and 4; a compounder on 2 and 3; an arbitrage on 1 and 4 with the structural fact as -the spine. Weight each section to the situation. - -### 1. Stat header - -A compact, text-aligned block of the core metrics. Keep it; it orients the reader instantly. - -```text -Price: [px] EPS: [cur / fwd] -Shares Out (M): [sh] P/E: [x] -Market Cap ($M): [mc] P/FCF: [x] -Net Debt ($M): [nd] EBIT: [ebit] -TEV ($M): [tev] TEV/EBIT: [x] -``` - -*Calc discipline:* Net Debt = total debt − cash & marketable securities. TEV = market cap + net -debt + the cost of option dilution (treasury-stock method — bake the dilution in, don't footnote it). -FCF = operating cash flow − capex, and **name any adjustment you make** rather than burying it. - -### 2. The hook - -One paragraph. State the core disconnect in the first breath. - -- *Compounder:* lead with the return-and-multiple gap — *"…an unleveraged return on equity of over - 35% and trades at 4.85x free cash flow."* -- *Asset play / arb:* lead with the value-to-price gap and the reason it's safe — *"…net cash of - $3.91 … trades for $1.70 … If the stock trades to net cash, the total return will be 130%."* - -### 3. The body — the four questions a Lou pitch actually answers - -Write these as argument, not headings to check off: - -1. **Why is it this cheap — and why is that fear wrong?** Name the market's reason (cyclical, - post-bankruptcy neglect, EM currency, "value trap") and dismantle it. For asset plays, prove - there is no catch — the usual culprits are funded debt or an operating burn depleting the cash, - so show that neither applies. -2. **What is the engine, or the hard fact?** Compounders: the moat — local scale, lowest-cost - production, "share of mind" brand, switching costs — explained mechanically, not asserted. - Special situations: the structural fact — the seniority waterfall, the put/call formula, the - minority-protection statute — explained so a non-specialist sees why it nearly *has* to resolve. -3. **What do the numbers say, once they're honest?** ROIC, capex intensity, working-capital - behavior — after you un-distort the accounting. Show the capital-allocation track record as a - **year-by-year table** (the falling share count, the deleveraging), because the trend persuades - where a single number does not. - - ```text - 12/31/95: 15.21M shares 12/31/98: 10.39M - 12/31/96: 13.57M 12/31/99: 9.17M - 12/31/97: 11.09M 04/18/01: 8.14M - ``` - -4. **Who controls the outcome, and are incentives aligned?** Controlling family, founder's age and - estate, the put/call or takeover formula, the option pool. Price the bad parts in as a cost of - ownership; map why the decision-maker is likely to do the value-realizing thing. - -### 4. The catalyst - -A short numbered list of specific, near-dated events that close the gap — a buyback authorization, -an exchange uplisting, a founder's retirement, a put/call exercise date, price hikes hitting next -quarter's margins. Vague "re-rating" is not a catalyst; a date or a mechanism is. - -## Where to dig — route each shape to your SEC tools - -This is the disposition, kept thin on purpose. It tells you *what to pull and what to normalize*; -your filing tools do the pulling. (Tool names below refer to the `sec-edgar-skill` skill; adapt to -whatever hands you have.) - -| Situation | Where the value hides | Read (pull these) | Normalize / adjust | Disqualifiers (kill it if…) | -|---|---|---|---|---| -| **Compounder** | un-distorted ROIC/FCF, a secular cost shift, buyback math | 10-K Item 7 (MD&A) & Item 8; the cash-flow statement; revenue/lease/deferred-revenue footnotes; the proxy (comp) | capex → maintenance level; pull deferred revenue back into earnings; undo treasury-stock-as-asset quirks | growth needs new capital; ROIC is an accounting mirage; buybacks happen only when the stock is dear | -| **Cigar-butt asset play** | price vs. net cash, proof of "no catch," a realization catalyst | balance sheet & liability footnotes; recent 8-Ks (asset sales); proxy (founder age, ownership) | subtract **all** liabilities; back non-recurring items out of operating cash flow; haircut value-destructive options | funded debt, or operating cash burn that eats the pile; no plausible catalyst; the discount is smaller than the governance theft | -| **Special situation / arb** | a structural or legal fact (seniority waterfall, put/call formula, squeeze-out law) | the agreement / indenture / 13D exhibit itself; reorg or plan docs; 20-F & 6-K for foreign issuers; the foreign regulator's filings | compute the formula *yourself*; rebuild the post-event capital structure; reconcile foreign-GAAP to a comparable basis | the legal fact doesn't actually bind; minorities have no protection; the timeline is open-ended with no forcing event | - -For the mechanics of pulling any of these (item codes, sections, exhibits, XBRL financials), defer -to your filing tools — don't reinvent them here. The 8-K's fired items tell you *what happened*; -the 13D's `item4_purpose_of_transaction` tells you *why*; the proxy and ownership forms map the -incentives. Note for foreign private issuers: there is no 10-K/10-Q/DEF 14A — it's 20-F and 6-K, -and the financials are IFRS. - -## Calibrating from the real pitches - -The seven primary sources ship with this skill in `references/corpus/` — each is the full VIC -write-up plus its discussion thread. They are the ground truth for the voice, but they run several -thousand words each, so treat them like a big filing: **don't load them wholesale.** When you need -to nail a specific move — a hook, a rebuttal, a stat header, a buyback table — grep the corpus for -it and read just the matching lines, the same way you'd pull one section out of a 10-K. - -- **Compounder:** `NVR`, `Sportsman's Guide` -- **Cigar-butt asset play:** `Winmill` -- **Special situation / structural arbitrage:** `Quilmes` (put/call + squeeze-out), `MCI` (capital-structure - seniority), `NII Holdings` (post-bankruptcy neglect), `Telemig` (emerging-market minority buyout) - -Each file has the write-up *and* the full discussion thread. The threads are where the temperament -is most visible — watch how he concedes points, separates what he knows from what he suspects, -answers a hostile question without flinching, and (in MCI) handles a thesis going wrong. Calibrate -your *honesty* there, not just your bullishness. - -When in doubt about whether a sentence is "Lou enough," ask the two questions his best writing -always passes: *Is every claim here something I actually verified in a document?* and *Have I been -as honest about what's wrong with this as I am excited about what's right?* If yes to both, ship it. +--- +name: pitch-like-lou +description: >- + Render an already-researched long or short thesis as a concise, numbers-first stock pitch + inspired by Norbert Lou's Value Investors Club writing. Use when the user asks for a + Norbert Lou or VIC-style write-up, or wants a finished value-investing thesis rewritten as + a direct, skeptical pitch. This is a presentation layer, not a substitute for company + research, thesis formation, or valuation work. +--- + +# Pitch Like Norbert Lou + +Turn a finished thesis into a plainspoken, decisive argument without letting the voice outrun the +evidence. The transferable qualities are forensic specificity, numerical clarity, and candor—not +imitation for its own sake. + +## Input gate + +Before drafting, locate the thesis's: + +- core mispricing or structural fact; +- reference price and as-of date; +- supporting evidence and its provenance; +- valuation or payoff math; +- strongest disconfirming evidence and unresolved assumptions; and +- concrete path by which the gap may close, if one exists. + +Use only what the research supports. If a load-bearing element is missing, ask for it or label +the gap; do not fill it with generic value-investing claims. A request to develop or validate +the thesis belongs in the analytical workflow before this rendering pass. + +## Render the argument + +1. **Lead with the disconnect.** Open with the number, contractual term, or operating fact that + makes the situation worth explaining. Give the reader the price/value contrast early. +2. **Name the market's objection fairly.** State the strongest reason the security may deserve + its price. Answer that version rather than a weakened one. +3. **Prove the engine or hard fact.** Explain the mechanism—economics, asset coverage, capital + structure, incentives, or legal terms—using document-level specifics. +4. **Show the arithmetic.** Define adjusted figures, bridge reported numbers to the figures used, + and make valuation or payoff math reproducible. Preserve the thesis's uncertainty rather than + manufacturing a precise target. +5. **Expose the weak points.** Separate verified fact, inference, and open question. Quantify a + governance cost or adverse case when the evidence permits; otherwise say what remains unknown. +6. **End on realization and falsification.** Name genuine catalysts or forcing mechanisms, but do + not invent one. State what evidence would break the thesis. + +Let the situation determine the order and weight. A contractual special situation may revolve +around one exhibit; an asset discount around liabilities and realization; an operating business +around unit economics, reinvestment, and capital allocation. These are examples, not required +buckets or headings. + +## Voice + +- Use short, declarative sentences and concrete nouns. Define nonstandard terms inline. +- Prefer a table, bridge, or formula when it carries the argument more honestly than prose. +- Earn confidence through specificity and concessions, not adjectives or swagger. +- Use first person only when it clarifies a judgment or calculation; never imply that Lou himself + holds the view. +- Pre-empt obvious objections without sounding defensive. Acknowledge what is genuinely bad + before explaining why the price may compensate for it. +- Use analogy or dry wit only when it makes a difficult point clearer. It is optional, not a quota. +- Discuss price history only when it bears on the thesis. A falling price is neither proof of value + nor proof that the thesis is broken. + +## Shape of the finished pitch + +Use the minimum structure the argument needs. A useful default is: + +1. **Title and position** — company/security, long or short, price and date. +2. **Optional orientation block** — only the decision-useful figures for this thesis. Choose the + metrics rather than copying a fixed VIC header; label periods, units, definitions, and adjusted + values. +3. **Hook** — the core disconnect in one compact paragraph. +4. **Case** — evidence and mechanics in the order that best proves the thesis. +5. **Valuation or payoff** — assumptions, arithmetic, downside, and sensitivity. +6. **Risks and open questions** — the strongest contrary evidence, not boilerplate. +7. **Catalysts / realization** — only specific mechanisms supported by the research; omit or state + the absence of a discrete catalyst when compounding or liquidation value is the actual path. + +Headings are navigation, not a form to complete. Collapse or rename them when the argument reads +better another way. + +## Evidence pass + +Before returning the pitch: + +- trace each load-bearing factual claim to the supplied research or a cited source; +- keep filing dates, transaction dates, fiscal periods, and current-price dates distinct; +- recalculate the central valuation or payoff math; +- mark inference and unresolved claims in proportion to their importance; and +- remove claims, metrics, anecdotes, and stylistic flourishes that do not advance the case. + +Dial conviction to the weakest load-bearing evidence. The desired finish is persuasive because it +is auditable, not persuasive despite uncertainty. + +## Corpus calibration + +The seven primary-source VIC write-ups and discussion threads in `references/corpus/` are optional +style references. Do not load them wholesale. When closer calibration is useful, grep for one +specific move—an opening, numerical bridge, objection, risk concession, or discussion reply—and +read only the surrounding passage. Use the corpus to observe technique and temperament, never as +factual support for a current thesis. diff --git a/skills/signal-sweep/README.md b/skills/signal-sweep/README.md index e3b7211..e7c339b 100644 --- a/skills/signal-sweep/README.md +++ b/skills/signal-sweep/README.md @@ -1,68 +1,49 @@ # Signal Sweep -An [agent skill](SKILL.md) that surfaces new investment ideas for a long-only investor. It -scans SEC filings and market data across the $50M–$10B US-listed universe (NYSE, NASDAQ, OTC) -and produces actionable shortlists of tickers with reasons — the top of the funnel that feeds -[`bottom-up-analyst`](../bottom-up-analyst/). +An [agent skill](SKILL.md) that discovers equity-research candidates from SEC filings and Yahoo +Finance market data. It emits shortlists with source links and coverage notes; every match still +requires company-level diligence. -## Where it sits +## Capabilities -```text - signal-sweep (this skill — produces tickers) - │ - ▼ - bottom-up-analyst (deep dive on one ticker) - ├── sec-edgar-skill (SEC filings) - ├── market-scout (price, peers, transcripts) - ▼ - pitch-like-lou (finished pitch) -``` - -The data skills (`sec-edgar-skill`, `market-scout`) research a company you already have in -mind. This skill answers the prior question: *which companies should you look at?* - -## What it scans +| Route | Script | +|---|---| +| Form 4 purchase clusters and event-date move context; Schedule 13D filings | `scan_insiders.py` | +| Configurable Yahoo Finance market screens | `scan_market.py` | +| SEC filing full-text keyword/theme search | `search_themes.py` | +| Investor-event discovery across relevant 8-K disclosures | `scan_conferences.py` | -| Capability | Script | Cadence | -|---|---|---| -| **Insider cluster/rip/dip buys + 13D filings** | `scan_insiders.py` | Daily CI (cron) or on-demand | -| **Market screens** (7 presets, config-driven) | `scan_market.py` | On-demand | -| **Keyword / theme discovery** (EFTS full-text search) | `search_themes.py` | On-demand | -| **Conference discovery** (8-K Item 8.01) | `scan_conferences.py` | On-demand | - -See [SKILL.md](SKILL.md) for invocation details and flags. +Within [SecStack](../README.md), the skill sits upstream of +[`bottom-up-analyst`](../bottom-up-analyst/): it proposes tickers; the analyst investigates them. ## Setup -1. **Install this skill's dependencies** from this directory: - - ```bash - uv sync - ``` +The SecStack bootstrap installs this skill's dependencies into the profile environment. For +standalone use: -2. **Set `EDGAR_IDENTITY`** — required for insider, theme, and conference scans (see - [profile setup](../../README.md#one-time-runtime-setup)). -3. Market screens (`scan_market.py`) use Yahoo Finance only and need no identity. - -## Screen customization +```bash +uv sync --project "" +``` -Screen definitions live in [`screens.json`](screens.json). Edit the JSON to add, remove, or -tweak screens — no Python changes needed. See [`references/guide_screens.md`](references/guide_screens.md) -for the full yfinance field reference. +Activate the resulting environment or prefix script commands with +`uv run --project ""`. -## Daily CI (GitHub Actions) +Set `EDGAR_IDENTITY` for insider, theme, and investor-event scans (see +[profile setup](../../README.md#one-time-runtime-setup)). Market screens use Yahoo Finance only. -The insider scan is designed to run daily on a cron. The workflow at -`.github/workflows/insider-scan.yml` runs at 7 AM ET on weekdays, posts results to -Discord, and uploads the Markdown output as a build artifact (90-day retention). +Keep the research workspace as the current directory and invoke installed scripts by resolved +absolute path. This keeps `signal-sweep-cache/` with the research rather than the installed skill. +See [SKILL.md](SKILL.md) for routes, semantics, and output contracts. -**Required secrets:** +## Screen customization -- `EDGAR_IDENTITY` — your SEC identity (see [profile setup](../../README.md#one-time-runtime-setup)) -- `DISCORD_WEBHOOK_URL` — (optional) Discord webhook for posting alerts +Definitions and universe bounds live in [`screens.json`](screens.json). See +[`references/guide_screens.md`](references/guide_screens.md) when adding or changing a screen; the +guide uses runtime `yfinance` discovery because provider fields can change. -The workflow also supports `workflow_dispatch` for manual runs with custom date, -lookback, and z-score threshold inputs. +## Daily workflow ---- -Part of the [SecStack skills](../README.md) collection. +[`.github/workflows/insider-scan.yml`](../../.github/workflows/insider-scan.yml) runs the insider +scan on weekdays, optionally posts results to Discord, and uploads the Markdown report. It requires +`EDGAR_IDENTITY`; set `DISCORD_WEBHOOK_URL` only when alerts are wanted. The workflow also supports +manual date, lookback, and threshold inputs. diff --git a/skills/signal-sweep/SKILL.md b/skills/signal-sweep/SKILL.md index c818136..1348be5 100644 --- a/skills/signal-sweep/SKILL.md +++ b/skills/signal-sweep/SKILL.md @@ -1,108 +1,106 @@ --- name: signal-sweep description: >- - Surface new investment ideas for a long-only investor by scanning SEC filings and market - data across the $50M–$10B US-listed universe. Use this skill to discover tickers you do not - yet have, via four scans: insider buying (cluster/dip/rip buys from Form 4), market screens - (near-52-week-low, high-short-interest, forgotten, and other presets), keyword/theme - exposure across filing full-text, and conference presenters. Triggers include "find me new - ideas", "show me insider buying", "run the screens", "who's exposed to [theme]", and "who's - presenting this week". This skill *produces* tickers for the rest of the stack to research: - reach for bottom-up-analyst to deep-dive a name you already have, market-scout for a quote, - sec-edgar-skill for a specific filing. + Discover US-listed equity research candidates by scanning Form 4 purchases and Schedule 13D + filings, configurable Yahoo Finance market screens, SEC filing full-text for a keyword or + theme, and 8-K disclosures for investor events. Use when the user wants new tickers or asks + who is buying, what companies match a market condition, which issuers mention a theme, or who + is presenting at investor events. Use company-research skills instead when the ticker is + already known and the task is diligence rather than discovery. --- # Signal Sweep -Top-of-funnel idea surfacing for long-only investors. Scans SEC filings and market data -across a configurable US-listed universe (NYSE, NASDAQ, OTC) and produces shortlists of -tickers with reasons. It sits upstream of the research stack — it *produces* tickers that -`bottom-up-analyst` then deep-dives. The market-cap floor and ceiling are set in -`screens.json` under `universe.market_cap_min` / `universe.market_cap_max` (default -$50M–$10B); all scripts read from that file. +Produce auditable candidate lists for further research. A match is a lead, not an investment +thesis. -## Capabilities +## Runtime and universe -### 1. Insider buying scanner + 13D alerts +Resolve bundled paths relative to this `SKILL.md` and invoke scripts by absolute path while keeping +the shell working directory at the research workspace. Generated reports then land in the +workspace's `./signal-sweep-cache`; pass `--cache-dir` to put them elsewhere. -Scans Form 4 open-market purchases (code `P` only) and detects three signal types: +The packaged SecStack profile already exposes this skill's Python dependencies. For standalone +use, run `uv sync --project ""`, then activate that environment or prefix the commands +below with `uv run --project ""`. SEC-facing routes require `EDGAR_IDENTITY`; market +screens do not: -- **Cluster buys:** 2+ distinct insiders buying the same stock within the lookback - window. Breadth signal — multiple people agree. -- **Dip buys:** an insider buys after an unusually large decline, measured against the - stock's own volatility (trailing 30-day return z-score ≤ -1.5σ). A CEO buying into a - -2σ drawdown on a normally calm stock is high-signal even without a second insider. -- **Rip buys:** an insider buys after an unusually large rally (z-score ≥ +1.5σ). - Insiders usually buy on weakness — buying into strength suggests the move has legs. +```bash +export EDGAR_IDENTITY="Jane Analyst jane@example.com" +``` -The z-score is volatility-adjusted: a 20% drop is routine for a biotech but exceptional -for a utility. The threshold adapts to each stock's personality. +The default universe is Yahoo Finance equities in region `us` between the market-cap bounds in +`screens.json` (currently $50M–$10B). Those bounds are configuration, not a definition of what is +investable, and can be changed for the task. -```bash -# On-demand -python scripts/scan_insiders.py --date yesterday --lookback 5 +## Routes -# Tighter threshold (only flag ≥2σ moves) -python scripts/scan_insiders.py --date yesterday --lookback 5 --zscore 2.0 +### Insider purchases and Schedule 13D filings -# Daily CI with Discord webhook -python scripts/scan_insiders.py --date yesterday --lookback 5 --webhook $DISCORD_WEBHOOK_URL +```bash +python "/scripts/scan_insiders.py" --date yesterday --lookback 5 ``` -Option exercises, tax withholding, awards, gifts, and sales are filtered out. -`--help` for all flags. +The scan counts only Form 4 transaction code `P`. It reports: -### 2. Market screens +- **clusters** — purchases by at least two distinct reporting owners in filings received during + the scan window; and +- **dip/rip context** — a purchase whose trailing 22-trading-day return is unusually negative or + positive relative to that stock's prior rolling returns (default threshold ±1.5 standard + deviations). -Config-driven screens via yfinance. Definitions live in `screens.json` — edit the JSON -to add or tweak screens, no Python changes needed. +Transaction date and filing date remain distinct. The move is measured as of the transaction date, +so a historical scan does not use today's price action. Schedule 13D and 13D/A matches are labeled +as blockholder filings, not presumed activism; inspect the linked filing for purpose and ownership. +Use `--zscore` only when the task calls for a different sensitivity. Prefer the +`DISCORD_WEBHOOK_URL` environment variable for optional alerts. + +### Configurable market screens ```bash -python scripts/scan_market.py --screen near-52wk-low -python scripts/scan_market.py --all -python scripts/scan_market.py --list +python "/scripts/scan_market.py" --list +python "/scripts/scan_market.py" --screen near-52wk-low +python "/scripts/scan_market.py" --all ``` -The 7 presets: `near-52wk-low`, `high-short-interest`, `short-covering`, `insider-heavy`, -`fallen-from-grace`, `low-institutional`, `forgotten`. Each enriches the top results with -P/E, short %, insider %, analyst rating, and sector. `--no-enrich` for -faster runs. See `references/guide_screens.md` for the field reference and how to add -custom screens. - -### 3. Keyword / theme discovery +`screens.json` contains the bundled query definitions. Results are ranked by each screen's declared +sort field and can be enriched with Yahoo snapshot fields. Read `references/guide_screens.md` only +when adding or changing a screen; it shows how to discover fields from the installed `yfinance` +version rather than treating a static list as exhaustive. -Goes from a keyword to a list of exposed companies by searching the full text of SEC -filings via EDGAR's EFTS engine. Finds non-obvious exposures — the REIT that leases to -cannabis growers, the testing lab, the BDC that lends to the sector. +### SEC full-text theme search ```bash -python scripts/search_themes.py --keyword "cannabis" --since 2026-01-01 -python scripts/search_themes.py --keyword "tariff" --since 2025-01-01 --until 2026-06-17 +python "/scripts/search_themes.py" --keyword "cannabis" --since 2026-01-01 +python "/scripts/search_themes.py" --keyword "tariff" --since 2025-01-01 --until 2026-06-17 ``` -Results are deduplicated by company, filtered to the configured universe, and enriched. +This searches EDGAR's full-text index, deduplicates the returned filing matches by issuer, applies +the configured market-cap universe, and links the most recent matching filing. `--limit` caps +filing documents before issuer deduplication; the report discloses when the server had more matches +than were fetched. A keyword match establishes mention, not economic exposure—open the linked filing +and inspect context before carrying a candidate forward. -### 4. Conference discovery - -Finds companies presenting at conferences by scanning 8-K Item 8.01 filings for -conference-related keywords. +### Investor-event discovery ```bash -python scripts/scan_conferences.py --start 2026-06-16 --end 2026-06-20 +python "/scripts/scan_conferences.py" --start 2026-06-16 --end 2026-06-20 ``` -Item 8.01 is a catch-all, so expect some false positives. The interesting follow-ups are -interactive — "which of these also show insider buying?", "any in healthcare?" - -## Output +This searches 8-K full text for third-party conferences, fireside chats, forums, symposia, and +issuer-hosted investor or capital-markets days. It uses both Reg FD and Other Events disclosures; +it is not an Item 8.01-only scan. The report links each source accession and discloses query +truncation or retrieval failures. Classification is heuristic, so verify the event and date in the +filing. -Every script writes a timestamped `.md` to `signal-sweep-cache/` and prints the absolute -path to stdout (same `emit(path)` pattern as `sec-edgar-skill`). Market-cap lookups are -cached to disk with a 24h TTL. +## Output and failure semantics -## Resources +Each successful scan writes a date- or range-keyed Markdown report under +`signal-sweep-cache//` and emits its absolute path to stdout. Re-running the same request +refreshes that report. Shared Yahoo market-cap lookups used by SEC-facing routes have a 24-hour disk +cache. -| File | Purpose | -|------|---------| -| `screens.json` | The 7 preset screen definitions (user-editable) | -| `references/guide_screens.md` | yfinance EquityQuery field reference + custom screen howto | +Reports distinguish a valid empty result from incomplete retrieval. Source-query, index, and total +parser failures exit nonzero. Recoverable omissions—such as an individually unparseable filing or +an unavailable Yahoo market cap—remain explicit coverage notes rather than becoming false no-data +claims. diff --git a/skills/signal-sweep/docs/conferences-autoresearch.md b/skills/signal-sweep/docs/conferences-autoresearch.md deleted file mode 100644 index fcc3a85..0000000 --- a/skills/signal-sweep/docs/conferences-autoresearch.md +++ /dev/null @@ -1,263 +0,0 @@ -# Conference Classifier — Development Notes - -**Last updated:** 2026-06-20 -**Status:** Working classifier in production. Autoresearch loop ready but not yet run (see §5). - -This file is for humans and developer agents doing further work on the conference -detection feature. It is intentionally NOT referenced from SKILL.md or any script -docstring — the agent using the skill in production doesn't need to read this. - ---- - -## 1. What this feature does - -`scripts/scan_conferences.py` scans a date range of SEC 8-K filings to surface -companies that are presenting at investor conferences. Output is a markdown table -of ticker / company / sector / market cap / conference name. - -The intended use is as a top-of-funnel signal — a company presenting at a conference -often means investor access, IR activity, and potential inflection points worth -investigating with the deeper research skills. - ---- - -## 2. Why the original script was broken - -The script that existed before this rewrite had two fundamental bugs: - -**Bug 1 — Wrong item filter.** -It searched only `items="8.01"` (Other Events). Empirical analysis showed ~75% of -real conference attendance filings use **Item 7.01** (Reg FD Disclosure) — companies -furnishing a presentation they shared at a third-party conference are making a Reg FD -disclosure, not an "other event". The old script was structurally blind to the majority -of its target population. - -**Bug 2 — Keyword matching against metadata, not text.** -The old Stage 2 did `_CONF_PATTERN.search(str(r))` where `r` is an EFTS result -object. `str(r)` is the Python string representation of the metadata dict — company -name, accession number, items list — not the filing text. So the keyword filter was -matching on things like `"conference"` appearing in the company name field of an -unrelated filing. It was effectively random. - ---- - -## 3. How the new classifier works - -### Stage 1 — EFTS server-side pre-filter - -Six queries run against EDGAR's full-text search index, merged and deduplicated by -accession number (first-match wins). This is cheap — no filing downloads. - -| Query | Item filter | ~Weekly vol | Rationale | -|---|---|---|---| -| `conference` | none | ~226 | Backbone. Catches ~75% of all conference filings via "presenting at the XYZ Conference" language | -| `"fireside chat"` | none | ~8 | Near-zero noise. Companies say exactly this. | -| `symposium` | none | ~4 | Clean. Medical/scientific conferences. | -| `"forum"` | `7.01` | ~60 | Genuine recall (AGA Financial Forum, Precision Medicine Forum, etc.) that "conference" misses. Item filter cuts boilerplate volume from 408→manageable. | -| `"investor day"` | `8.01` | ~2 | Companies hosting *their own* investor days file under 8.01. Skip Stage 2 (see below). | -| `"capital markets day"` | none | <1 | European-listed US names use this term exclusively. Skip Stage 2. | - -**Queries deliberately excluded:** - -- `summit` — 100% FPs. Summit Therapeutics, Summit Hotel, Summit Midstream flood it at all item levels. Their company name appears in every filing. -- `presentation` — 995/week, far too broad. -- `"analyst day"` — only 2 hits/week in tested period, both keyword-in-exhibit only. No incremental recall over `conference`. -- `"investor day"` without item filter — 2,999 raw hits, mostly unrelated filings where "investor day" appears in exhibits or boilerplate. Item filter `8.01` cuts this to ~14/week of clean own-hosted events. - -### Stage 2 — Client-side text classification - -For each candidate, download the actual filing HTML (`r.get_filing().text()`) and run: - -**2a. Exclusion check** (`_all_occurrences_excluded`): -Find every occurrence of the signal word in the text. If EVERY occurrence sits inside -a known false-positive phrase (±60 char window), reject. Key insight: if a filing says -"conference call" three times AND "presenting at the Goldman Sachs Conference" once, -the single non-excluded occurrence is enough to pass. Only reject if there is zero -non-excluded occurrence. - -Current exclusion list: - -```python -[ - "conference call", - "conference call and webcast", - "exclusive forum", - "forum selection", - "alternative forum", -] -``` - -**2b. Attendance verb check** (`_has_attendance_verb`): -Require at least one regex pattern to match: - -```python -[ - "will present", - "presenting at", - "participate in", - "scheduled to present", - "speak at", - "participation at", - "will attend", - "will be attending", -] -``` - -`"will attend"` was added after a live test showed the AGA Financial Forum filing -("Unitil Corporation will attend the American Gas Association Financial Forum") -failing with only the original 6 patterns. - -**Stage 2 exceptions** (`no_text_check_queries`): -`"investor day"` and `"capital markets day"` skip Stage 2 entirely. Reason: for -these event types, the announcement language is often only in the attached exhibit -(PDF presentation deck), not in the HTML body. The EFTS match + item filter is -reliable enough signal on its own. Attempting Stage 2 would produce false negatives. - -### Ticker extraction - -The EFTS result `company` field already contains the ticker in parentheses: -`"LCI INDUSTRIES (LCII) (CIK 0000763744)"`. Extract with `r'\(([A-Z]{1,5})\)'`, -first match. No API fallback needed. Companies without a parenthesised ticker -(foreign filers, government entities, private companies) are skipped — they won't -be in-universe anyway. - ---- - -## 4. Empirical findings that shaped the design - -All of these were verified by running live EFTS queries and inspecting actual filing -text during a research session in June 2026. - -**Conference seasonality** (relevant for label building and performance expectations): - -- **Jan 6-17**: Highest density. JPM Healthcare Conference alone generates 100+ - filings in one week. -- **May-June**: Highest overall volume. Goldman, JPMorgan TMC, BofA, Wells Fargo, - Needham, Baird all cluster here. ~226 "conference" hits/week. -- **Sep 8-26**: Back-to-school season. Deutsche Bank, Barclays, Jefferies. -- **Late Jul/Aug**: Dead zone. Avoid for labelling. -- **Mid-Oct to mid-Nov**: Pre-earnings blackout. Very few conferences. - -**"forum" noise structure** (informed the exclusion list): -Of 20 sampled `"forum"` + items=7.01 results: - -- 4 TPs (Precision Medicine Forum, AGA Financial Forum, etc.) — 0% overlap with "conference" -- 4 FPs from bylaw boilerplate ("exclusive forum", "forum selection amendment", - "alternative forum") — all filterable with the 3 exclusion phrases added -- 3 FPs from "Investor Forum" (Acadian Asset Management's own investor event, filed - repeatedly) — borderline TP, acceptable to let through -- 9 keyword-in-exhibit only — correctly handled by SKIP_NO_TEXT - -**"investor day" item behaviour**: -`items='7.01'` returns 0 results; `items='8.01'` returns ~14/week. This is because -companies hosting their OWN investor day file it as an "other event" (8.01), whereas -companies presenting at a THIRD-PARTY conference file under Reg FD (7.01). This -distinction is important for the query design. - ---- - -## 5. Autoresearch plan (ready to execute when viable) - -The goal: use an iterative Jules-powered loop to tune the `exclusions` and `patterns` -lists against a precision/recall target, rather than hand-tuning them. - -### What needs to exist in this repo first - -**a) `data/labels.json`** — ground truth dataset. - -Jules built a partial labels dataset on branch -`rewrite-conference-classifier-2032772644618132070` (269 JSONL entries from -Jan 2026). Format: - -```json -{"id": "accession_number", "company": "...", "ticker": "...", "filed": "...", - "items": [...], "matched_query": "...", "text": "...", "label": "CONFERENCE_ATTENDANCE|OTHER", - "confidence": "high|medium|low"} -``` - -**Recommended expansion**: supplement with ~100 filings from Jun 2-6, 2026 -(BofA Global Technology + Jefferies Global Healthcare week). This adds cross-sector -diversity and is representative of production conditions. Target 150-200 total labels, -balanced between CONFERENCE_ATTENDANCE and OTHER. - -**b) `scripts/eval_harness.py`** — scoring script. - -Accepts `--params '{"exclusions": [...], "patterns": [...]}'`, runs the two-stage -classifier against `labels.json`, prints: - -```json -{"metric_value": , "target_met": =0.90 AND recall>=0.70>, - "details": {"precision": ..., "recall": ..., "f1": ..., - "false_positive_ids": [...], "false_negative_ids": [...]}} -``` - -A template is in the `jules-autoresearch` skill at -`references/eval_harness_template.py`. - -### Loop invocation - -```bash -python /scripts/autoresearch.py \ - --source "sources/github/eggmasonvalue/secstack" \ - --eval-script "skills/signal-sweep/scripts/eval_harness.py" \ - --params '{ - "exclusions": ["conference call", "conference call and webcast", - "exclusive forum", "forum selection", "alternative forum"], - "patterns": ["will present", "presenting at", "participate in", - "scheduled to present", "speak at", "participation at", - "will attend", "will be attending"] - }' \ - --target "precision >= 0.90 with recall >= 0.70" \ - --metric-type numeric \ - --target-value 0.90 \ - --parallel 3 \ - --max-iterations 8 \ - --output-dir autoresearch_results/conference_classifier -``` - -**What Jules tunes**: `exclusions` list and `patterns` list only. The EFTS queries -(Stage 1) are fixed — they define the recall ceiling and should not be modified by -the loop. - -**What "parallel 3" means here**: Jules evaluates current params + 2 ablation -variants in a single session (not 3 separate sessions). 1 Jules session per -iteration. - -**Why autoresearch wasn't run yet**: the loop was designed, the infra was built -(`autoresearch.py`, `jules_client.py`), but `labels.json` was incomplete (Jules -built 269 entries but only from Jan 2026, single-sector heavy) and `eval_harness.py` -doesn't exist yet. These are ~1 hour of Jules work to complete before the loop -can run. - -### Expected convergence - -Starting precision is unknown but likely 0.70-0.80 based on manual inspection. -The patterns list is probably under-specified (missing verbs like "invited to present", -"will be hosting", "plan to present"). The exclusion list may need 1-2 more phrases -for "conference call" variants ("quarterly conference call", "earnings conference -call"). The loop should converge in 3-5 iterations. - ---- - -## 6. Known remaining issues - -**Conference name extractor is rough.** `_extract_conference_name` sometimes returns -phrases like "materials to be used during the conference" or truncates at 120 chars. -This is cosmetic — doesn't affect precision/recall — but the output table looks bad. -A tighter regex or extracting from a specific sentence pattern would help. Not worth -fixing until the classifier precision is tuned. - -**`"forum"` query still has ~60% reject rate** at Stage 2. This is expected — the -bylaw boilerplate is getting filtered correctly. But it means 60% of the 408/week -forum candidates are wasted downloads. A potential optimisation: add a quick -pre-screen for forum bylaw language before downloading the full text (check if -the accession's items list includes `5.03` — bylaws amendment — and skip those). - -**EFTS caching** (`edgartools` caches responses in `~/.edgar/_tcache/`). If a query -fails silently on first run (e.g., SSL not yet configured), it caches 0 results. -Subsequent runs with warm cache return 0 even after the SSL issue is resolved. -Workaround: delete the relevant cache file or wait for cache expiry. - -**yfinance 404s** for valid tickers that aren't in Yahoo's database (e.g., YICC, -MODG) are expected and benign — they're caught by `except Exception` in the enrichment -block and the filing still appears in output with missing price/sector data. diff --git a/skills/signal-sweep/docs/flip_buy_difficulty_analysis.md b/skills/signal-sweep/docs/flip_buy_difficulty_analysis.md deleted file mode 100644 index 1abdc56..0000000 --- a/skills/signal-sweep/docs/flip_buy_difficulty_analysis.md +++ /dev/null @@ -1,228 +0,0 @@ -# Technical Analysis: Implementing a Flip-Buy Insider Filter - -> **Not a skill instruction — do not act on this file.** This is a human-facing *design -> proposal* for an unbuilt feature, kept in `docs/` for developers. It is deliberately not -> referenced from `SKILL.md` or any script. An agent running `signal-sweep` must ignore it: -> the flip-buy filter and the draft code below **do not exist** in the skill. Use only the -> capabilities documented in `SKILL.md`. - -This document provides a comprehensive feasibility and difficulty analysis for implementing a **Flip-Buy** filter within the `signal-sweep` workspace. A flip-buy is defined as an open-market purchase (code `P`) by an corporate insider who has a history of recent open-market sales (code `S`). - ---- - -## 1. Investment Thesis of the "Flip-Buy" Signal - -In insider-activity analysis: - -- **Routine Selling:** Insiders sell shares frequently for liquidity, tax management, or portfolio diversification. Sells are often automated under 10b5-1 plans and carry less directional information. -- **Active Buying:** Open-market purchases require active capital deployment and are strong indicators of valuation confidence. -- **The "Flip" Shift:** When an insider who has been consistently selling for months suddenly turns around and buys, it signifies a major sentiment inflection point. It indicates that the stock has reached a price point so low, or the business has reached a turning point so positive, that the insider is willing to break their selling pattern and deploy cash. - ---- - -## 2. Current Architecture vs. Flip-Buy Requirements - -### Current Flow in [scan_insiders.py](../scripts/scan_insiders.py) - -1. Fetches the daily bulk Form 4 index for the lookback window (e.g., last 5 trading days). -2. Filters to the market-cap universe ($50M–$10B) via `_common.in_universe`. -3. Parses open-market purchases (code `P`) from the current filings using `edgartools`. -4. Tags purchases with **Dip** or **Rip** labels using yfinance volatility metrics. -5. Emits the daily summary. - -```mermaid -graph TD - A[SEC Daily Form 4 Index] --> B[Filter by Universe Cap] - B --> C[Extract Code P Purchases] - C --> D[Compute Rip/Dip Volatility-Adjusted Z-Scores] - D --> E[Output Daily Summary & Discord Webhook] -``` - -### Proposed Flip-Buy Flow - -To detect a flip-buy, the system must inspect the *historical context* of the purchasing insider: - -```mermaid -graph TD - A[Daily Scan finds Purchase by Insider I in Ticker T] --> B{Is Insider's Transaction History Cached?} - B -- No --> C[Fetch 1-Year Form 4 filings for Ticker T from SEC] - B -- Yes --> D[Load Cached History & Fetch Only New Filings] - C --> E[Parse all P/S transactions and update cache] - D --> E - E --> F[Analyze Chronological timeline for Insider I] - F --> G{Did Insider have >= K Sells prior to Purchase?} - G -- Yes --> H[Tag as FLIP BUY] - G -- No --> I[Tag as standard Purchase] -``` - ---- - -## 3. Core Implementation Challenges & Difficulty Level - -We rate the overall implementation difficulty as **Moderate**. The mathematical logic is simple, but building a performant, SEC-compliant caching layer is crucial. - -### Challenge A: SEC Rate Limiting & Network Overhead (High Complexity) - -- **Problem:** If the daily scan finds purchases in 30 different tickers, fetching 12 months of Form 4 filings for each ticker at runtime requires making dozens of SEC EDGAR API calls. Under the SEC's fair access policy, requests are limited to **10 requests/sec**, and synchronous retrieval would make the daily scan take several minutes. -- **Solution:** A persistent transaction cache on disk (e.g. `signal-sweep-cache/insiders/{ticker}_txns.json`) is mandatory. For any ticker, the script should load the cache, find the latest cached transaction date, fetch only newer filings, and merge them. -- **Difficulty:** **Medium-High** (requires state management and robust exception handling). - -### Challenge B: Insider Name Normalization (Medium Complexity) - -- **Problem:** Insiders are represented by strings in the filing object (`row.get("Insider")`). Names may vary slightly across filings (e.g. "Smith John", "Smith John A.", "Smith John Jr."). -- **Solution:** Normalization helper to strip punctuation, remove middle initials/suffixes, and standardize formatting. Alternatively, extract the unique reporter CIK (`rptOwnerCik` XML node) from the raw Form 4 XML structure using `edgartools` if exposed. -- **Difficulty:** **Medium**. - -### Challenge C: Defining a "Flip" Algorithmic Rule (Low Complexity) - -- **Problem:** We need a clear mathematical definition of a flip to avoid tagging noise (e.g. an insider who sold $100 of shares for taxes but bought $100,000 of shares is a buy, not a flip). -- **Solution:** Define a parameterized rule: - - **Lookback Period:** 180 to 365 calendar days. - - **Sells Threshold:** Minimum of 2 or 3 distinct open-market sale transactions (code `S`). - - **Consecutive Check:** No intervening purchases (code `P`) between the historical sells and the current purchase. -- **Difficulty:** **Low**. - ---- - -## 4. Implementation Plan & Estimated Sizing - -| Task | Description | Estimated Effort | -| :--- | :--- | :--- | -| **1. Caching Infrastructure** | Create state-management helpers in `_common.py` to save, load, and incrementally update `signal-sweep-cache/insiders/{ticker}_txns.json` using transaction lists. | 3–4 hours | -| **2. Historical Fetcher** | Add a fetcher in `scan_insiders.py` utilizing the cache to retrieve a ticker's 1-year transaction timeline (P and S) and update it incrementally. | 2–3 hours | -| **3. Detection Logic** | Implement name-normalization and the sequential logic check (e.g. checking if the prior transactions for that insider were consecutive sells). | 2 hours | -| **4. Reporting Integration** | Update formatting functions (`_signal_badge`, markdown table generation) and Discord webhook embeds to label and highlight `FLIP BUY` signals. | 2 hours | -| **5. Testing & Validation** | Run backfills against known historic flips (e.g., REFI) to ensure accuracy and measure speed/network usage. | 2 hours | -| **Total Sizing** | **11–13 developer hours (approx. 1.5–2 days of work)** | Moderate | - ---- - -## 5. Draft Implementation Code Structure - -To illustrate the technical execution, here is how the cache and checking logic could be implemented in Python: - -```python -# In scripts/scan_insiders.py - -import json -from pathlib import Path -from datetime import datetime, timedelta - - -def load_ticker_history(ticker: str, cache_dir: Path) -> list[dict]: - """Load cached insider transactions for a ticker.""" - cache_path = cache_dir / "insiders" / f"{ticker}_txns.json" - if cache_path.exists(): - try: - return json.loads(cache_path.read_text(encoding="utf-8")) - except Exception: - return [] - return [] - - -def save_ticker_history(ticker: str, txns: list[dict], cache_dir: Path) -> None: - """Save ticker history to disk.""" - cache_path = cache_dir / "insiders" / f"{ticker}_txns.json" - cache_path.parent.mkdir(parents=True, exist_ok=True) - cache_path.write_text(json.dumps(txns, indent=2), encoding="utf-8") - - -def update_ticker_history(ticker: str, cache_dir: Path) -> list[dict]: - """Incremental fetch of all Form 4 transactions (P and S) over 1 year.""" - from edgar import Company - - txns = load_ticker_history(ticker, cache_dir) - last_date = max([t["date"] for t in txns]) if txns else None - - # Define start date (1 year lookback) - start_dt = datetime.now() - timedelta(days=365) - start_str = start_dt.strftime("%Y-%m-%d") - - # If we have cached transactions, start from the latest cached date to prevent re-fetching - if last_date and last_date > start_str: - fetch_start = last_date - else: - fetch_start = start_str - - date_range = f"{fetch_start}:{datetime.now().strftime('%Y-%m-%d')}" - - company = Company(ticker) - filings = company.get_filings(form="4", date=date_range) - - new_txns = [] - if filings: - for filing in filings: - try: - obj = filing.obj() - df = obj.to_dataframe() - if df is not None and not df.empty: - for _, row in df.iterrows(): - code = row.get("Code", "") - if code in ("P", "S"): - new_txns.append( - { - "date": filing.filing_date, - "insider": row.get("Insider") or obj.insider_name, - "role": row.get("Position") - or getattr(obj, "position", "Unknown"), - "code": code, - "shares": row.get("Shares", 0), - "price": row.get("Price", 0), - "remaining": row.get("Remaining Shares"), - } - ) - except Exception: - continue - - # Merge and deduplicate - seen = set() - merged = [] - for t in txns + new_txns: - # Deduplicate using uniquely identifying fields - key = (t["date"], t["insider"], t["code"], t["shares"], t["price"]) - if key not in seen: - seen.add(key) - merged.append(t) - - merged.sort(key=lambda x: x["date"]) - save_ticker_history(ticker, merged, cache_dir) - return merged - - -def check_flip_buy( - ticker: str, purchase_insider: str, purchase_date: str, txns: list[dict], min_sells: int = 2 -) -> bool: - """Check if the purchase was preceded by a series of sells by this insider.""" - - # Normalize name for comparison - def norm(name): - return "".join(name.upper().split()) - - insider_norm = norm(purchase_insider) - - # Filter and sort prior transactions - prior_txns = [] - for t in txns: - if t["date"] < purchase_date and norm(t["insider"]) == insider_norm: - prior_txns.append(t) - - prior_txns.sort(key=lambda x: x["date"], reverse=True) # newest first - - sells_count = 0 - for t in prior_txns: - if t["code"] == "S": - sells_count += 1 - elif t["code"] == "P": - # An intermediate purchase breaks the "flip" sequence - break - - return sells_count >= min_sells -``` - ---- - -## 6. Recommendations & Trade-offs - -1. **Name Matching Stability:** While string normalization is usually sufficient, we recommend looking into `edgartools`' API to see if the reporting owner's CIK is exposed directly. Using CIKs guarantees zero name-collision bugs. -2. **Programmatic (10b5-1) Sells:** Programmatic sells are scheduled and less discretionary. Consider filtering out sales that are flagged as 10b5-1 (this is represented by a footnote in Form 4s). If the sales were purely 10b5-1, the "flip" is slightly less significant than if they were active, discretionary sells. However, discretionary sells followed by a buy represents a maximum-conviction pivot. -3. **Execution Mode:** To avoid slowing down the default scan, we suggest adding a `--detect-flips` CLI flag to `scan_insiders.py` so that users can opt-in to this intensive analysis step when needed. diff --git a/skills/signal-sweep/references/guide_screens.md b/skills/signal-sweep/references/guide_screens.md index 928d1cf..ffd340d 100644 --- a/skills/signal-sweep/references/guide_screens.md +++ b/skills/signal-sweep/references/guide_screens.md @@ -1,16 +1,33 @@ # Screen Customization Guide -How to add, modify, or remove screens in `screens.json`. No Python changes needed — -the script reads the config and builds `yfinance.EquityQuery` objects dynamically. +Read this guide when adding or changing a query in `screens.json`. Yahoo's screener schema evolves, +so inspect the installed `yfinance` version instead of treating examples here as a complete field +catalog. -## Screen definition format +## Discover the current schema + +```python +import yfinance as yf + +query = yf.EquityQuery("eq", ["region", "us"]) +print(query.valid_fields) # fields grouped by category +print(query.valid_values) # accepted values for equality fields +help(yf.EquityQuery) +help(yf.screen) +``` + +Use the exact field spelling and value units returned at runtime. Before retaining a new screen, +run it with a small `--size`, inspect several raw matches, and confirm that the provider's units and +sort direction mean what the description says. + +## Definition format ```json { "id": "my-screen", "name": "Human-Readable Name", "emoji": "📊", - "description": "Why this screen exists and what it catches", + "description": "The conditions this query applies", "filters": [ {"field": "some.field", "op": "gte", "value": 42} ], @@ -20,69 +37,40 @@ the script reads the config and builds `yfinance.EquityQuery` objects dynamicall } ``` -- **id**: unique slug, used as CLI argument (`--screen my-screen`) -- **name**: display name in output headers -- **emoji**: prefix for the output header -- **description**: shown below the header; explain the thesis behind the screen -- **filters**: list of conditions (all must pass — implicit AND) -- **sort**: which field to sort by and direction -- **size**: max results to return (default 25) -- **enrich**: whether to run the enrichment pass (adds P/E, short %, insider %, etc.) - -## Filter operators - -| op | Meaning | Example | -|----|---------|---------| -| `eq` | equals | `{"field": "sector", "op": "eq", "value": "Technology"}` | -| `gte` | >= | `{"field": "intradaymarketcap", "op": "gte", "value": 200000000}` | -| `lte` | <= | `{"field": "lastclose52weeklow.lasttwelvemonths", "op": "lte", "value": 1.15}` | -| `gt` | > | `{"field": "peratio.lasttwelvemonths", "op": "gt", "value": 0}` | -| `lt` | < | `{"field": "pctheldinst", "op": "lt", "value": 0.30}` | -| `btwn` | between | `{"field": "peratio.lasttwelvemonths", "op": "btwn", "value": [5, 15]}` | - -## Available yfinance EquityQuery fields - -### Equality fields (use with `eq`) - -- `exchange` — NYSE, NMS (NASDAQ), PNK, etc. -- `sector` — Technology, Healthcare, Industrials, etc. -- `industry` — specific industry name -- `region` — "us" (always set by the universe config) -- `peer_group` — peer group identifier - -### Price & return fields - -- `intradaymarketcap` — current market cap in dollars -- `intradayprice` — current price -- `lastclose52weeklow.lasttwelvemonths` — ratio of last close to 52-week low (1.0 = at the low, 1.15 = 15% above) -- `lastclose52weekhigh.lasttwelvemonths` — ratio of last close to 52-week high (1.0 = at the high, 0.50 = 50% below) -- `percentchange` — intraday % change -- `fiftytwowkpercentchange` — 52-week % change (e.g. -10 means down 10%) - -### Trading & ownership fields - -- `pctheldinsider` — insider ownership as decimal (0.15 = 15%) -- `pctheldinst` — institutional ownership as decimal -- `beta` — beta coefficient -- `avgdailyvol3m` — 3-month average daily volume -- `dayvolume` — today's volume - -### Short interest fields - -- `short_percentage_of_float.value` — short % of float (15 = 15%) -- `short_interest_percentage_change.value` — % change in short interest (-20 = SI dropped 20%) -- `days_to_cover_short.value` — days to cover - -### Valuation fields - -- `peratio.lasttwelvemonths` — trailing P/E ratio -- `pegratio_5y` — 5-year PEG ratio -- `lastclosetevtotalrevenue.lasttwelvemonths` — EV/Revenue (trailing) -- `bookvalueshare.lasttwelvemonths` — book value per share +- `id` is the unique CLI slug. +- `name`, `emoji`, and `description` label the report; describe the observable condition rather + than asserting why it occurred. +- `filters` are combined with logical AND. +- `sort` declares the ranking field and direction. +- `size` is the maximum number of screener rows returned before enrichment. +- `enrich` adds Yahoo snapshot fields for each returned ticker. + +Scalar values become `[field, value]` operands. A JSON list becomes `[field, *values]`, which +supports operators such as `btwn` and `is-in` when accepted by the installed API. + +## Fields used by the bundled screens + +These are examples, not the limits of `EquityQuery`: + +| Field | Meaning and units used here | +|---|---| +| `region` | Yahoo region code; the default universe uses `us`. | +| `intradaymarketcap` | Market capitalization in dollars. | +| `lastclose52weeklow.lasttwelvemonths` | Last close divided by the 52-week low (`1.15` = 15% above). | +| `lastclose52weekhigh.lasttwelvemonths` | Last close divided by the 52-week high (`0.50` = 50% below). | +| `fiftytwowkpercentchange` | 52-week percentage change (`-10` = down 10%). | +| `avgdailyvol3m` | Three-month average daily share volume. | +| `pctheldinsider`, `pctheldinst` | Ownership fractions (`0.15` = 15%). | +| `short_percentage_of_float.value` | Short percentage of float (`15` = 15%). | +| `short_interest_percentage_change.value` | Percentage change in short interest (`-20` = down 20%). | + +Do not infer more than the field establishes. For example, low institutional ownership does not by +itself prove that a stock is undiscovered, and a flat annual return does not prove that the market +ignored new information. ## Universe bounds -The `universe` section in `screens.json` is injected into every screen automatically: +The `universe` object is injected into every query: ```json "universe": { @@ -92,37 +80,25 @@ The `universe` section in `screens.json` is injected into every screen automatic } ``` -You don't need to repeat these in individual screen filters. To tighten the floor for -a specific screen (e.g. $200M for "fallen-from-grace"), add an `intradaymarketcap` -filter to that screen — it will override the universe floor since both conditions -must pass. - -## Enrichment columns - -When `"enrich": true`, the script calls `yf.Ticker(symbol).info` for each result and -adds these columns to the output: +An additional screen-level market-cap condition combines with these bounds. Change the universe +object when the whole scan should use different bounds; add a filter when only one screen needs to +be narrower. -- Analyst rating, price targets (mean, median) -- Short % of float -- Insider %, institutional % -- Sector, industry -- P/E (trailing or forward) -- Current price, market cap +## Enrichment -The enrichment pass takes ~2–3 seconds per ticker (two API calls: snapshot + price -history). For 25 results, expect ~1 minute. Use `--no-enrich` for faster runs when -you only need the screener output. +When enabled, enrichment adds best-effort Yahoo snapshot fields such as price, market cap, sector, +industry, P/E, short ownership, insider/institutional ownership, and analyst rating. These fields +are not part of the screen predicate unless the definition explicitly filters on them. Missing +enrichment remains `n/a` and does not invalidate the underlying screener match. -## Example: adding a custom screen - -To add a "cheap on EV/Revenue" screen: +## Example ```json { - "id": "cheap-ev-revenue", - "name": "Low EV/Revenue", - "emoji": "💰", - "description": "Stocks trading below 1x EV/Revenue — potential value if margins expand", + "id": "positive-low-ev-revenue", + "name": "Positive EV/Revenue Below 1x", + "emoji": "📊", + "description": "Positive trailing EV/revenue between 0x and 1x", "filters": [ {"field": "lastclosetevtotalrevenue.lasttwelvemonths", "op": "lte", "value": 1.0}, {"field": "lastclosetevtotalrevenue.lasttwelvemonths", "op": "gt", "value": 0} @@ -133,8 +109,8 @@ To add a "cheap on EV/Revenue" screen: } ``` -Add this object to the `"screens"` array in `screens.json` and run: +Add the object to `screens.json`, then inspect a small run: ```bash -python scripts/scan_market.py --screen cheap-ev-revenue +python "/scripts/scan_market.py" --screen positive-low-ev-revenue --size 5 ``` diff --git a/skills/signal-sweep/screens.json b/skills/signal-sweep/screens.json index a9d01da..b023c50 100644 --- a/skills/signal-sweep/screens.json +++ b/skills/signal-sweep/screens.json @@ -9,7 +9,7 @@ "id": "near-52wk-low", "name": "Near 52-Week Low", "emoji": "📉", - "description": "Stocks within 15% of their 52-week low — price dislocation in a thinly-covered universe", + "description": "Last close no more than 15% above the trailing 52-week low", "filters": [ {"field": "lastclose52weeklow.lasttwelvemonths", "op": "lte", "value": 1.15} ], @@ -21,7 +21,7 @@ "id": "high-short-interest", "name": "High Short Interest", "emoji": "🩳", - "description": "Stocks with >15% short interest and reasonable volume — someone disagrees strongly", + "description": "Short interest above 15% of float and 3-month average daily volume of at least 100,000 shares", "filters": [ {"field": "short_percentage_of_float.value", "op": "gte", "value": 15}, {"field": "avgdailyvol3m", "op": "gte", "value": 100000} @@ -34,7 +34,7 @@ "id": "short-covering", "name": "Shorts Covering (SI Dropping)", "emoji": "📈", - "description": "Stocks where short interest dropped >20% — shorts capitulating, overhang lifting", + "description": "Short interest down more than 20% while at least 5% of float remains short", "filters": [ {"field": "short_interest_percentage_change.value", "op": "lt", "value": -20}, {"field": "short_percentage_of_float.value", "op": "gte", "value": 5} @@ -47,7 +47,7 @@ "id": "insider-heavy", "name": "High Insider Ownership", "emoji": "🏠", - "description": "Stocks with >15% insider ownership — skin in the game, not just options", + "description": "Yahoo-reported insider ownership of at least 15%", "filters": [ {"field": "pctheldinsider", "op": "gte", "value": 0.15} ], @@ -57,9 +57,9 @@ }, { "id": "fallen-from-grace", - "name": "Fallen From Grace", + "name": "40–60% Below 52-Week High", "emoji": "💥", - "description": "Former momentum names 40-60% below their 52-week high ($200M+ market cap)", + "description": "Last close 40% to 60% below the 52-week high, with market cap of at least $200M", "filters": [ {"field": "lastclose52weekhigh.lasttwelvemonths", "op": "gte", "value": 0.40}, {"field": "lastclose52weekhigh.lasttwelvemonths", "op": "lte", "value": 0.60}, @@ -71,9 +71,9 @@ }, { "id": "low-institutional", - "name": "Under-Owned by Institutions", + "name": "Low Institutional Ownership", "emoji": "🔍", - "description": "Stocks with <30% institutional ownership ($200M+) — under-discovered, no index flows", + "description": "Yahoo-reported institutional ownership below 30%, with market cap of at least $200M", "filters": [ {"field": "pctheldinst", "op": "lt", "value": 0.30}, {"field": "intradaymarketcap", "op": "gte", "value": 200000000} @@ -84,9 +84,9 @@ }, { "id": "forgotten", - "name": "Forgotten Stocks", + "name": "Flat 52-Week Return", "emoji": "😴", - "description": "Stocks that went nowhere over 12 months (-10% to +10%) — the market hasn't noticed", + "description": "Trailing 52-week price change between -10% and +10%", "filters": [ {"field": "fiftytwowkpercentchange", "op": "gte", "value": -10}, {"field": "fiftytwowkpercentchange", "op": "lte", "value": 10} diff --git a/skills/signal-sweep/scripts/_common.py b/skills/signal-sweep/scripts/_common.py index b0acd52..65bf6ee 100644 --- a/skills/signal-sweep/scripts/_common.py +++ b/skills/signal-sweep/scripts/_common.py @@ -13,6 +13,7 @@ import argparse import json import os +import re import sys import time from datetime import datetime, timedelta @@ -195,12 +196,12 @@ def load_universe() -> dict: return _universe_cache -def universe_label() -> str: - """Human-readable universe range, e.g. '$50M\u2013$10B'.""" - u = load_universe() - lo = fmt_mcap(u.get("market_cap_min", 50_000_000)) - hi = fmt_mcap(u.get("market_cap_max", 10_000_000_000)) - return f"{lo}\u2013{hi}" +def universe_label(universe: dict | None = None) -> str: + """Return a human-readable market-cap range, such as '$50M to $10B'.""" + bounds = universe or load_universe() + lo = fmt_mcap(bounds.get("market_cap_min", 50_000_000)) + hi = fmt_mcap(bounds.get("market_cap_max", 10_000_000_000)) + return f"{lo}–{hi}" def in_universe(mcap: int | None, floor: int | None = None, ceiling: int | None = None) -> bool: @@ -230,6 +231,29 @@ def fmt_mcap(mcap: int | None) -> str: return f"${mcap:,.0f}" +def extract_ticker(company: str) -> str | None: + """Extract the first Yahoo-style ticker from an EFTS company label.""" + for group in re.findall(r"\(([^()]*)\)", company): + if group.upper().lstrip().startswith("CIK"): + continue + candidate = group.split(",", 1)[0].strip().upper().replace(".", "-") + if re.fullmatch(r"[A-Z][A-Z0-9-]{0,9}", candidate): + return candidate + return None + + +def sec_filing_url(cik: str | int, accession: str) -> str: + """Build the SEC filing-index URL for an accession.""" + bare_cik = str(cik).lstrip("0") + compact = accession.replace("-", "") + return f"https://www.sec.gov/Archives/edgar/data/{bare_cik}/{compact}/{accession}-index.html" + + +def md_cell(value: object) -> str: + """Escape a value for a Markdown table cell.""" + return str(value).replace("|", "\\|").replace("\n", " ").strip() + + def parse_date(s: str) -> str: """Parse 'YYYY-MM-DD', 'today', or 'yesterday' into YYYY-MM-DD string.""" if s.lower() == "today": diff --git a/skills/signal-sweep/scripts/scan_conferences.py b/skills/signal-sweep/scripts/scan_conferences.py index 8533c4b..d202816 100644 --- a/skills/signal-sweep/scripts/scan_conferences.py +++ b/skills/signal-sweep/scripts/scan_conferences.py @@ -1,59 +1,36 @@ -r"""Discover companies presenting at investor conferences via 8-K filings. - -Two-stage pipeline: - Stage 1 — EFTS server-side pre-filter (cheap, no downloads): - Run targeted full-text queries against EDGAR's search index, optionally - filtered by item number. Merge results and deduplicate by accession number. - - Stage 2 — Client-side text classification (download & parse): - For each candidate, download the actual filing HTML and apply: - 2a. Exclusion check — reject if every occurrence of the signal word - is inside a known false-positive phrase. - 2b. Attendance check — accept only if an attendance verb pattern matches. - Some query types (investor day, capital markets day) skip Stage 2 entirely - because the signal is reliable enough from EFTS + item filter alone, and - the keyword often lives in the exhibit rather than the HTML body. - -Usage: - python scripts/scan_conferences.py --start 2026-06-16 --end 2026-06-20 - - # override classifier params at runtime (useful for testing) - python scripts/scan_conferences.py --start 2026-06-16 --end 2026-06-20 \\ - --params '{"exclusions": ["conference call", "conference call and webcast"]}' +"""Discover investor-event announcements through SEC 8-K full-text search. + +Targeted EFTS queries produce candidates. Most candidates are then classified +against filing text; exact investor-day queries can survive unavailable primary +text because the indexed phrase and item filter are the signal. Reports disclose +pagination caps and retrieval failures. """ from __future__ import annotations import argparse -import json import re import sys +from datetime import datetime from pathlib import Path sys.path.insert(0, str(Path(__file__).resolve().parent)) import _common as c -# --------------------------------------------------------------------------- -# Default classifier parameters -# All three lists are tunable — start minimal, add one element at a time. -# --------------------------------------------------------------------------- -DEFAULT_PARAMS: dict = { - # Stage 1: EFTS full-text queries. - # Key = quoted string exactly as passed to edgar.search_filings(query=...) - # Value = item filter string (or None for no filter) +_MAX_RESULTS_PER_QUERY = 300 + +# Specific queries precede the broad ``conference`` query because candidate +# deduplication retains the first matching route. +_CLASSIFIER: dict = { "queries": { - "conference": None, # backbone — ~226/week, needs Stage 2 - '"fireside chat"': None, # ~8/week, near-zero noise - "symposium": None, # ~4/week, clean - '"forum"': "7.01", # ~408/week with 7.01; Stage 2 clears boilerplate - '"investor day"': "8.01", # ~2/week, own-hosted events; skip Stage 2 - '"capital markets day"': None, # <1/week, European names; skip Stage 2 + '"investor day"': "8.01", + '"capital markets day"': None, + '"fireside chat"': None, + "symposium": None, + '"forum"': "7.01", + "conference": None, }, - # Queries where keyword is often in exhibit only or signal is reliable - # enough from EFTS+item filter — skip Stage 2 text classification. - "no_text_check_queries": ['"investor day"', '"capital markets day"'], - # Stage 2a: reject if EVERY occurrence of the signal word sits inside one - # of these phrases (case-insensitive substring match in a ±60-char window). + "trusted_index_queries": ['"investor day"', '"capital markets day"'], "exclusions": [ "conference call", "conference call and webcast", @@ -61,9 +38,7 @@ "forum selection", "alternative forum", ], - # Stage 2b: accept if at least one of these patterns matches (re.IGNORECASE). - # Keep minimal — add only when a real filing fails to match. - "patterns": [ + "attendance_patterns": [ r"will present", r"presenting at", r"participate in", @@ -76,198 +51,172 @@ } -# --------------------------------------------------------------------------- -# Stage 2 helpers -# --------------------------------------------------------------------------- - - def _all_occurrences_excluded(text: str, signal_word: str, exclusions: list[str]) -> bool: - """Return True only if EVERY occurrence of signal_word in text is contained within an exclusion-phrase context (±60 chars). - - Logic: if even one occurrence is NOT in an exclusion context, the filing - may be genuine — don't reject it. - """ - positions = [m.start() for m in re.finditer(re.escape(signal_word), text, re.IGNORECASE)] - + """Return whether every signal occurrence lies near an excluded phrase.""" + positions = [ + match.start() for match in re.finditer(re.escape(signal_word), text, re.IGNORECASE) + ] if not positions: - return False # word not found → can't reject on this basis - - for pos in positions: - window = text[max(0, pos - 60) : pos + 60 + len(signal_word)].lower() - if not any(ex.lower() in window for ex in exclusions): - return False # found at least one occurrence outside exclusions - - return True # every occurrence was inside an exclusion phrase - - -def _has_attendance_verb(text: str, patterns: list[str]) -> bool: - return any(re.search(p, text, re.IGNORECASE) for p in patterns) + return False + for position in positions: + window = text[max(0, position - 60) : position + 60 + len(signal_word)].lower() + if not any(exclusion.lower() in window for exclusion in exclusions): + return False + return True def _classify(text: str, query: str, params: dict) -> str: - """Apply Stage 2 classification. - - Returns one of: "ACCEPT" | "REJECT_EXCLUSION" | "REJECT_NO_PATTERN" | "SKIP_NO_TEXT". - """ - # Some queries trust EFTS + item filter — skip text check entirely. - if query in params["no_text_check_queries"]: + """Classify a downloaded candidate filing.""" + if query in params["trusted_index_queries"]: return "ACCEPT" - # The signal word is the first meaningful word of the query. signal_word = query.strip('"').split()[0] - - # If the signal word isn't in the HTML body at all, the match was in an - # exhibit — we can't classify it, so skip rather than false-accept. if not re.search(re.escape(signal_word), text, re.IGNORECASE): - return "SKIP_NO_TEXT" - - # 2a — exclusion check + return "REJECT_NO_PRIMARY_TEXT_MATCH" if _all_occurrences_excluded(text, signal_word, params["exclusions"]): return "REJECT_EXCLUSION" - - # 2b — attendance verb check - if _has_attendance_verb(text, params["patterns"]): + if any(re.search(pattern, text, re.IGNORECASE) for pattern in params["attendance_patterns"]): return "ACCEPT" - - return "REJECT_NO_PATTERN" - - -# --------------------------------------------------------------------------- -# Stage 1: EFTS candidate retrieval -# --------------------------------------------------------------------------- + return "REJECT_NO_ATTENDANCE_PATTERN" -def _get_candidates(start: str, end: str, params: dict, limit: int = 300) -> dict[str, dict]: - """Run all EFTS queries, merge results, deduplicate by accession number. - - Returns {accession_number: {"result": EFTSResult, "query": str}}. - First-match wins on deduplication (queries are ordered by signal quality). - """ +def _get_candidates( + start: str, + end: str, + params: dict, + max_per_query: int = _MAX_RESULTS_PER_QUERY, +) -> tuple[dict[str, dict], list[dict]]: + """Search, paginate, and deduplicate EFTS candidates by accession.""" import edgar candidates: dict[str, dict] = {} + query_stats = [] for query, item_filter in params["queries"].items(): - c.log(f" EFTS: {query!r}" + (f" items={item_filter!r}" if item_filter else "") + " ...") + label = f"{query!r}" + (f" items={item_filter!r}" if item_filter else "") + c.log(f" EFTS: {label} ...") + stat = { + "query": query, + "item_filter": item_filter, + "total": 0, + "fetched": 0, + "truncated": False, + "error": "", + } try: - results = edgar.search_filings( + search = edgar.search_filings( query=query, forms="8-K", items=item_filter, start_date=start, end_date=end, - limit=limit, + limit=min(max_per_query, 100), ) - if results is None: - continue - n = 0 - for r in results: - acc = getattr(r, "accession_number", "") or "" - if not acc: - continue - if acc not in candidates: - candidates[acc] = {"result": r, "query": query} - n += 1 - c.log(f" → {n} new candidates") + if search is None: + raise RuntimeError("EFTS returned no search object") + stat["total"] = int(getattr(search, "total", 0) or 0) + fetched = len(list(search)) + target = min(stat["total"], max_per_query) + if fetched < target: + search = search.fetch_more(target - fetched) + rows = list(search)[:max_per_query] + stat["fetched"] = len(rows) + stat["truncated"] = stat["fetched"] < stat["total"] except Exception as exc: + stat["error"] = str(exc) c.log(f" ERROR: {exc}") + query_stats.append(stat) + continue - return candidates - - -# --------------------------------------------------------------------------- -# Conference name extractor -# --------------------------------------------------------------------------- + new_count = 0 + for result in rows: + accession = str(getattr(result, "accession_number", "") or "") + if accession and accession not in candidates: + candidates[accession] = {"result": result, "query": query} + new_count += 1 + c.log(f" → {new_count} new candidates; fetched {stat['fetched']} of {stat['total']}") + query_stats.append(stat) + return candidates, query_stats -def _extract_conference_name(text: str) -> str | None: - """Try to pull a conference / event name out of the filing text. - Returns a clean string or None. - """ +def _extract_event_name(text: str) -> str | None: + """Extract a best-effort event name from filing text.""" patterns = [ - r"(?:at|the)\s+([\w\s&\-\']{10,80}?" + r"(?:at|the)\s+([\w\s&\-']{10,80}?" r"(?:Conference|Forum|Symposium|Investor Day|Capital Markets Day|Fireside Chat))", - r"(?:will present at|presenting at|participate in|speak at)\s+(?:the\s+)?([^.]{10,80})", ] - for pat in patterns: - m = re.search(pat, text, re.IGNORECASE) - if m: - name = re.sub(r"[.,;:\s]+$", "", m.group(1)).strip() + for pattern in patterns: + match = re.search(pattern, text, re.IGNORECASE) + if match: + name = re.sub(r"[.,;:\s]+$", "", match.group(1)).strip() if len(name) >= 10: return name[:120] return None -# --------------------------------------------------------------------------- -# Main scan function -# --------------------------------------------------------------------------- - - -def scan_conferences(start: str, end: str, params: dict, mcap_data: dict) -> list[dict]: - """Run the full two-stage pipeline and return a list of conference dicts.""" - c.log(f"Scanning 8-K filings for conferences: {start} to {end}") - - candidates = _get_candidates(start, end, params) - c.log(f" {len(candidates)} unique candidates after EFTS + dedup") +def scan_conferences( + start: str, + end: str, + params: dict, + mcap_data: dict, +) -> tuple[list[dict], dict]: + """Run candidate retrieval, classification, and enrichment.""" + c.log(f"Scanning 8-K filings for investor events: {start} to {end}") + candidates, query_stats = _get_candidates(start, end, params) + if query_stats and all(stat["error"] for stat in query_stats): + raise RuntimeError("all EFTS event queries failed") - if not candidates: - return [] - - conferences = [] + c.log(f" {len(candidates)} unique candidates after EFTS deduplication") stats = { "checked": 0, "no_ticker": 0, - "out_of_universe": 0, - "no_text": 0, + "out_of_universe_or_unresolved": 0, + "text_retrieval_errors": 0, "rejected": 0, "accepted": 0, + "query_stats": query_stats, } + events = [] - for acc, cand in candidates.items(): - r = cand["result"] - query = cand["query"] + for accession, candidate in candidates.items(): + result = candidate["result"] + query = candidate["query"] stats["checked"] += 1 - # ── resolve ticker ────────────────────────────────────────────────── - company_raw = str(getattr(r, "company", "Unknown")) - str(getattr(r, "cik", "")).lstrip("0") - filed = str(getattr(r, "filed", "")) - - # Ticker is in the EFTS company string: "NAME (TICK) (CIK ...)" - m = re.search(r"\(([A-Z]{1,5})\)", company_raw) - if not m: + company_raw = str(getattr(result, "company", "Unknown")) + cik = str(getattr(result, "cik", "")).lstrip("0") + filed = str(getattr(result, "filed", "")) + ticker = c.extract_ticker(company_raw) + if not ticker: stats["no_ticker"] += 1 continue - ticker = m.group(1) - # ── universe filter (market cap $50M-$10B) ────────────────────────── mcap = c.get_market_cap(ticker, mcap_data) if not c.in_universe(mcap): - stats["out_of_universe"] += 1 + stats["out_of_universe_or_unresolved"] += 1 continue - # ── fetch filing text ─────────────────────────────────────────────── + trusted_index_match = query in params["trusted_index_queries"] filing_text = "" try: - filing_text = r.get_filing().text() + filing_text = result.get_filing().text() or "" except Exception as exc: - c.log(f" WARN: could not fetch text for {acc}: {exc}") + stats["text_retrieval_errors"] += 1 + c.log(f" WARNING: could not fetch filing text for {accession}: {exc}") - if not filing_text: - stats["no_text"] += 1 + if trusted_index_match: + verdict = "ACCEPT" + elif not filing_text: continue - - # ── Stage 2 classification ────────────────────────────────────────── - verdict = _classify(filing_text, query, params) + else: + verdict = _classify(filing_text, query, params) if verdict != "ACCEPT": stats["rejected"] += 1 continue - stats["accepted"] += 1 - # ── enrich with yfinance ──────────────────────────────────────────── try: import yfinance as yf @@ -275,110 +224,126 @@ def scan_conferences(start: str, end: str, params: dict, mcap_data: dict) -> lis except Exception: info = {} - conferences.append( + events.append( { "ticker": ticker, "company": info.get("shortName") or info.get("longName") or company_raw, "sector": info.get("sector", "n/a"), "mcap": mcap, - "price": info.get("currentPrice"), + "price": info.get("currentPrice") or info.get("regularMarketPrice"), "filed": filed, - "conference": _extract_conference_name(filing_text) or "(see filing)", - "query": query, # which EFTS query surfaced this + "event": _extract_event_name(filing_text) or "(see filing)", + "matched_query": query.strip('"'), + "accession": accession, + "source_url": c.sec_filing_url(cik, accession), } ) + events.sort(key=lambda event: (event["filed"], event["ticker"]), reverse=True) c.log(f" Done — {stats}") - return conferences - + return events, stats -# --------------------------------------------------------------------------- -# Output -# --------------------------------------------------------------------------- - -def _render_markdown(start: str, end: str, conferences: list[dict]) -> str: +def _render_markdown(start: str, end: str, events: list[dict], stats: dict) -> str: + """Render event candidates with source and coverage notes.""" lines = [ - f"# Conference Discovery: {start} to {end}\n", - f"Found **{len(conferences)}** companies in the {c.universe_label()} universe.\n", + f"# Investor-Event Discovery: {start} to {end}\n", + f"Found **{len(events)}** companies in the {c.universe_label()} universe.\n", + "Matches are heuristic leads. Verify the event name, date, and participation in the linked filing.\n", ] - if not conferences: - lines.append("No conference announcements found.\n") + truncated = [stat for stat in stats["query_stats"] if stat["truncated"]] + query_errors = [stat for stat in stats["query_stats"] if stat["error"]] + metadata_omissions = stats["no_ticker"] or stats["out_of_universe_or_unresolved"] + if truncated or query_errors or stats["text_retrieval_errors"] or metadata_omissions: + lines.append("## Coverage notes\n") + for stat in truncated: + lines.append( + f"- Query `{stat['query']}` fetched {stat['fetched']} of {stat['total']} matches." + ) + for stat in query_errors: + lines.append(f"- Query `{stat['query']}` failed: {stat['error']}") + if stats["text_retrieval_errors"]: + lines.append( + f"- Filing text retrieval failed for {stats['text_retrieval_errors']} candidate(s)." + ) + if stats["no_ticker"]: + lines.append( + f"- EFTS supplied no parseable ticker for {stats['no_ticker']} candidate(s)." + ) + if stats["out_of_universe_or_unresolved"]: + lines.append( + f"- {stats['out_of_universe_or_unresolved']} candidate(s) were outside the " + "market-cap bounds or lacked a Yahoo market cap." + ) + lines.append("") + + if not events: + lines.append( + "No classified investor-event announcements were found in completed coverage.\n" + ) return "\n".join(lines) lines += [ - "| # | Ticker | Company | Sector | Mkt Cap | Price | Filed | Conference |", - "|---|--------|---------|--------|---------|-------|-------|------------|", + "| # | Ticker | Company | Sector | Mkt Cap | Price | Filed | Matched query | Event | Source |", + "|---|---|---|---|---:|---:|---|---|---|---|", ] - for i, conf in enumerate(conferences, 1): - price_s = f"${conf['price']:.2f}" if conf.get("price") else "n/a" + for index, event in enumerate(events, 1): + price = event.get("price") lines.append( - f"| {i} | {conf['ticker']} | {conf['company']} | {conf['sector']} | " - f"{c.fmt_mcap(conf['mcap'])} | {price_s} | {conf['filed']} | " - f"{conf['conference']} |" + f"| {index} | {c.md_cell(event['ticker'])} | {c.md_cell(event['company'])} | " + f"{c.md_cell(event['sector'])} | {c.fmt_mcap(event['mcap'])} | " + f"{f'${price:.2f}' if price is not None else 'n/a'} | {event['filed']} | " + f"{c.md_cell(event['matched_query'])} | {c.md_cell(event['event'])} | " + f"[{event['accession']}]({event['source_url']}) |" ) - - lines.append( - "\n_Sourced from 8-K filings via EFTS full-text search + " - "two-stage client-side classifier._\n" - ) + lines.append("") return "\n".join(lines) -# --------------------------------------------------------------------------- -# CLI -# --------------------------------------------------------------------------- - - def main() -> None: - """Run the CLI to discover companies presenting at investor conferences.""" - p = argparse.ArgumentParser( - description="Conference discovery via 8-K filings (two-stage EFTS classifier)." - ) - p.add_argument("--start", required=True, help="Start date YYYY-MM-DD") - p.add_argument("--end", required=True, help="End date YYYY-MM-DD") - p.add_argument( - "--params", - help="JSON string to override/extend DEFAULT_PARAMS keys. " - 'Example: \'{"exclusions": ["conference call"]}\'', - ) - c.add_identity_arg(p) - c.add_cache_arg(p) - args = p.parse_args() - - c.resolve_identity(args.identity) + """Run the investor-event discovery CLI.""" + parser = argparse.ArgumentParser(description="Investor-event discovery via 8-K filings.") + parser.add_argument("--start", required=True, help="Start date YYYY-MM-DD.") + parser.add_argument("--end", required=True, help="End date YYYY-MM-DD.") + c.add_identity_arg(parser) + c.add_cache_arg(parser) + args = parser.parse_args() - # edgartools needs system certs on corporate networks try: - from edgar import configure_http - - configure_http(use_system_certs=True) - except Exception: - pass - - params = { - k: (v.copy() if isinstance(v, (dict, list)) else v) for k, v in DEFAULT_PARAMS.items() - } - if args.params: - try: - overrides = json.loads(args.params) - params.update(overrides) - except json.JSONDecodeError as exc: - c.log(f"ERROR: invalid --params JSON: {exc}") - sys.exit(1) + start = datetime.strptime(args.start, "%Y-%m-%d").date() + end = datetime.strptime(args.end, "%Y-%m-%d").date() + except ValueError as exc: + parser.error(str(exc)) + if start > end: + parser.error("--start must not be later than --end.") + c.resolve_identity(args.identity) cache = c.cache_root(args.cache_dir) mcap_data = c.load_mcap_cache(cache) - try: - conferences = scan_conferences(args.start, args.end, params, mcap_data) + events, stats = scan_conferences( + start.isoformat(), + end.isoformat(), + _CLASSIFIER, + mcap_data, + ) + except RuntimeError as exc: + c.log(f"ERROR: {exc}") + sys.exit(1) finally: c.save_mcap_cache(cache, mcap_data) - md = _render_markdown(args.start, args.end, conferences) - slug = f"{args.start}_to_{args.end}" - c.write_output(cache, "conferences", slug, md) + report = _render_markdown(start.isoformat(), end.isoformat(), events, stats) + c.write_output(cache, "conferences", f"{start.isoformat()}_to_{end.isoformat()}", report) + + incomplete = ( + any(stat["truncated"] or stat["error"] for stat in stats["query_stats"]) + or stats["text_retrieval_errors"] > 0 + ) + if incomplete: + c.log("ERROR: emitted report has incomplete source coverage.") + sys.exit(1) if __name__ == "__main__": diff --git a/skills/signal-sweep/scripts/scan_insiders.py b/skills/signal-sweep/scripts/scan_insiders.py index 1e4a2da..4958f8f 100644 --- a/skills/signal-sweep/scripts/scan_insiders.py +++ b/skills/signal-sweep/scripts/scan_insiders.py @@ -1,30 +1,15 @@ -"""Scan Form 4 insider purchases for cluster buys, rip/dip buys, and 13D filings. - -Pulls the daily Form 4 bulk index over a rolling lookback window, filters to -the $50M-$10B universe, parses open-market purchases (code P only), and -detects: - - Cluster buys: 2+ distinct insiders buying the same stock within the window - - Rip buys: insider buys after an unusually large rally (z-score based) - - Dip buys: insider buys into an unusually large decline (z-score based) - -Rip/dip detection is volatility-adjusted — a 20% move means nothing for a -biotech that swings 20% monthly, but it's exceptional for a utility. The -trailing 30-day return is measured against the stock's own historical -volatility (annualized stdev of daily returns over the prior year, scaled to -a 30-day window). A z-score beyond the threshold (default ±1.5) flags the -purchase. - -Also pulls SC 13D / 13D/A filings for activist blockholders. - -Usage: - python scripts/scan_insiders.py --date yesterday --lookback 5 - python scripts/scan_insiders.py --date 2026-06-16 --lookback 5 --webhook $DISCORD_WEBHOOK_URL +"""Scan Form 4 purchases and Schedule 13D filings. + +Form 4 transaction code ``P`` rows are preserved individually. Clusters use +distinct reporting-owner identities, and price-move context is measured as of +the transaction date rather than the day the script runs. """ from __future__ import annotations import argparse import math +import os import sys from collections import defaultdict from datetime import datetime, timedelta @@ -33,488 +18,479 @@ sys.path.insert(0, str(Path(__file__).resolve().parent)) import _common as c -# --------------------------------------------------------------------------- -# Helpers -# --------------------------------------------------------------------------- +_TRADING_DAYS_IN_MONTH = 22 -def _trading_dates(end_date: str, lookback: int) -> list[str]: - """Generate dates to scan (calendar days, skipping weekends).""" +def _weekdays(end_date: str, lookback: int) -> list[str]: + """Return the requested number of weekdays ending on or before a date.""" end = datetime.strptime(end_date, "%Y-%m-%d") dates = [] - d = end + current = end while len(dates) < lookback: - if d.weekday() < 5: - dates.append(d.strftime("%Y-%m-%d")) - d -= timedelta(days=1) + if current.weekday() < 5: + dates.append(current.strftime("%Y-%m-%d")) + current -= timedelta(days=1) return list(reversed(dates)) def _post_discord(webhook_url: str, embeds: list[dict]) -> None: - """Post embeds to Discord webhook, batching at 10 per request.""" + """Post Discord embeds in API-sized batches.""" import requests - for i in range(0, len(embeds), 10): - batch = embeds[i : i + 10] - resp = requests.post(webhook_url, json={"embeds": batch}, timeout=30) - if resp.status_code not in (200, 204): - c.log(f"WARNING: Discord webhook returned {resp.status_code}: {resp.text[:200]}") - - -# --------------------------------------------------------------------------- -# Rip / dip detection -# --------------------------------------------------------------------------- - -_TRADING_DAYS_30D = 22 # ~22 trading days in a calendar month - - -def compute_move_zscore(ticker: str) -> dict | None: - """Compute the trailing 30-day return z-score for a stock. + for index in range(0, len(embeds), 10): + try: + response = requests.post( + webhook_url, json={"embeds": embeds[index : index + 10]}, timeout=30 + ) + response.raise_for_status() + except requests.RequestException as exc: + raise RuntimeError(f"Discord webhook failed: {exc}") from exc - Returns {"return_30d": float, "vol_30d": float, "zscore": float} or None - if there isn't enough price history. - - The z-score measures how unusual the recent 30-day move is relative to the - stock's own historical behavior: - - trailing_return = price now / price 22 trading days ago - 1 - - vol_30d = stdev of daily returns over the prior year, scaled to a - 22-trading-day window (multiply by sqrt(22)) - - zscore = trailing_return / vol_30d - """ - import yfinance as yf +def _number(value) -> float | None: + """Return a finite float or None.""" try: - hist = yf.Ticker(ticker).history(period="1y") - if hist is None or len(hist) < 60: - return None - close = hist["Close"].dropna() - if len(close) < 60: - return None - except Exception: + number = float(value) + except (TypeError, ValueError): return None + return number if math.isfinite(number) else None - daily_returns = close.pct_change().dropna() - if len(daily_returns) < 40: - return None - - # Trailing 30-day return - n = min(_TRADING_DAYS_30D, len(close) - 1) - trailing_return = close.iloc[-1] / close.iloc[-1 - n] - 1 - - # Historical daily stdev, scaled to 30-day window - daily_std = daily_returns.std() - if daily_std == 0 or math.isnan(daily_std): - return None - vol_30d = daily_std * math.sqrt(n) - - zscore = trailing_return / vol_30d - return { - "return_30d": trailing_return, - "vol_30d": vol_30d, - "zscore": zscore, - } - - -def tag_rip_dip(purchases_by_ticker: dict[str, list[dict]], zscore_threshold: float = 1.5) -> None: - """Tag each purchase with rip/dip signals based on move z-score. +def _date(value, fallback: str) -> str: + """Return a YYYY-MM-DD date from a dataframe value.""" + try: + import pandas as pd - Mutates purchase dicts in place, adding: - - "return_30d", "vol_30d", "zscore": the raw numbers - - "signal": "rip" | "dip" | None - """ - tickers = list(purchases_by_ticker.keys()) - c.log(f"Computing move z-scores for {len(tickers)} tickers...") + parsed = pd.to_datetime(value, errors="coerce") + if not pd.isna(parsed): + return parsed.date().isoformat() + except Exception: + pass + return fallback + + +def _owner_identity(ownership, insider_name: str) -> tuple[str, str]: + """Return a stable reporting-owner identity and optional CIK.""" + owners = list(getattr(getattr(ownership, "reporting_owners", None), "owners", []) or []) + normalized = "".join(ch for ch in insider_name.upper() if ch.isalnum()) + for owner in owners: + owner_name = str(getattr(owner, "name", "")) + owner_normalized = "".join(ch for ch in owner_name.upper() if ch.isalnum()) + if len(owners) == 1 or owner_normalized == normalized: + cik = str(getattr(owner, "cik", "") or "").lstrip("0") + if cik: + return f"cik:{cik}", cik + return f"name:{normalized}", "" + + +def compute_move_context(ticker: str, as_of_dates: list[str]) -> dict[str, dict | None]: + """Compute event-date 22-trading-day returns and rolling-return z-scores.""" + import pandas as pd + import yfinance as yf - for i, ticker in enumerate(tickers): - if i % 20 == 0 and i > 0: - c.log(f" z-scores: {i}/{len(tickers)}...") + unique_dates = sorted(set(as_of_dates)) + if not unique_dates: + return {} + first = datetime.strptime(unique_dates[0], "%Y-%m-%d") - timedelta(days=450) + last = datetime.strptime(unique_dates[-1], "%Y-%m-%d") + timedelta(days=1) - result = compute_move_zscore(ticker) + try: + history = yf.Ticker(ticker).history( + start=first.strftime("%Y-%m-%d"), + end=last.strftime("%Y-%m-%d"), + auto_adjust=True, + ) + close = history["Close"].dropna().copy() + close.index = pd.DatetimeIndex(close.index).tz_localize(None).normalize() + except Exception: + return {date: None for date in unique_dates} + + contexts: dict[str, dict | None] = {} + for date in unique_dates: + cutoff = pd.Timestamp(date) + series = close[close.index <= cutoff] + if len(series) < 80: + contexts[date] = None + continue - for purchase in purchases_by_ticker[ticker]: - if result is None: - purchase["return_30d"] = None - purchase["vol_30d"] = None - purchase["zscore"] = None - purchase["signal"] = None - continue + rolling_returns = series.pct_change(_TRADING_DAYS_IN_MONTH).dropna() + if len(rolling_returns) < 41: + contexts[date] = None + continue + current_return = float(rolling_returns.iloc[-1]) + baseline = rolling_returns.iloc[:-1].tail(252) + baseline_std = float(baseline.std()) + if not math.isfinite(baseline_std) or baseline_std == 0: + contexts[date] = None + continue + baseline_mean = float(baseline.mean()) + contexts[date] = { + "return_22d": current_return, + "baseline_mean": baseline_mean, + "baseline_std": baseline_std, + "zscore": (current_return - baseline_mean) / baseline_std, + } + return contexts - purchase["return_30d"] = result["return_30d"] - purchase["vol_30d"] = result["vol_30d"] - purchase["zscore"] = result["zscore"] - z = result["zscore"] - if z >= zscore_threshold: +def tag_move_context( + purchases_by_ticker: dict[str, list[dict]], + zscore_threshold: float = 1.5, +) -> None: + """Attach event-date price context and dip/rip labels to purchases.""" + c.log(f"Computing event-date move context for {len(purchases_by_ticker)} tickers...") + for index, (ticker, purchases) in enumerate(purchases_by_ticker.items()): + if index and index % 20 == 0: + c.log(f" Move context: {index}/{len(purchases_by_ticker)}...") + contexts = compute_move_context( + ticker, [purchase["transaction_date"] for purchase in purchases] + ) + for purchase in purchases: + context = contexts.get(purchase["transaction_date"]) + purchase["return_22d"] = context["return_22d"] if context else None + purchase["zscore"] = context["zscore"] if context else None + zscore = purchase["zscore"] + if zscore is not None and zscore >= zscore_threshold: purchase["signal"] = "rip" - elif z <= -zscore_threshold: + elif zscore is not None and zscore <= -zscore_threshold: purchase["signal"] = "dip" else: purchase["signal"] = None - c.log(" z-scores done.") - -# --------------------------------------------------------------------------- -# Form 4 scanning -# --------------------------------------------------------------------------- - - -def scan_form4s(dates: list[str], cache: Path, mcap_data: dict) -> dict[str, list[dict]]: - """Scan Form 4 filings across dates, return all purchases by ticker. - - Returns {ticker: [purchase_dicts]} — downstream code detects clusters - and rip/dip from this raw collection. - """ +def scan_form4s(dates: list[str], mcap_data: dict) -> tuple[dict[str, list[dict]], dict]: + """Collect code-P transactions from Form 4 filings received on given dates.""" import edgar purchases_by_ticker: dict[str, list[dict]] = defaultdict(list) - seen_keys: dict[str, set] = defaultdict(set) + stats = { + "failed_dates": [], + "filings_seen": 0, + "filings_parsed": 0, + "parse_errors": 0, + "unresolved_market_cap": 0, + } - for date_str in dates: - c.log(f"Fetching Form 4 index for {date_str}...") + for requested_date in dates: + c.log(f"Fetching Form 4 index for {requested_date}...") try: - dt = datetime.strptime(date_str, "%Y-%m-%d") + parsed_date = datetime.strptime(requested_date, "%Y-%m-%d") filings = edgar.get_filings( - year=dt.year, - quarter=(dt.month - 1) // 3 + 1, + year=parsed_date.year, + quarter=(parsed_date.month - 1) // 3 + 1, form="4", - filing_date=date_str, + filing_date=requested_date, amendments=False, ) except Exception as exc: - c.log(f" WARNING: could not fetch Form 4 index for {date_str}: {exc}") + c.log(f" ERROR: could not fetch Form 4 index: {exc}") + stats["failed_dates"].append(requested_date) continue - if filings is None: - c.log(f" No Form 4 filings found for {date_str}") continue - try: - df = filings.to_pandas() - except Exception: - c.log(f" WARNING: could not convert filings to DataFrame for {date_str}") - continue - - c.log(f" Found {len(df)} Form 4 filings for {date_str}") - - for idx, filing in enumerate(filings): - if idx >= len(df): - break - + c.log(f" Found {len(filings)} Form 4 filings") + for index, filing in enumerate(filings): + stats["filings_seen"] += 1 try: - obj = filing.obj() + ownership = filing.obj() + transactions = ownership.to_dataframe() + stats["filings_parsed"] += 1 except Exception: + stats["parse_errors"] += 1 + continue + if transactions is None or transactions.empty or "Code" not in transactions: + continue + purchases = transactions[transactions["Code"] == "P"] + if purchases.empty: continue - ticker = None - try: - issuer = obj.issuer - ticker = getattr(issuer, "ticker", None) - if not ticker: - ticker = getattr(issuer, "trading_symbol", None) - except Exception: - pass - + ticker = str(getattr(ownership.issuer, "ticker", "") or "").upper().strip() if not ticker: continue - ticker = ticker.upper().strip() - mcap = c.get_market_cap(ticker, mcap_data) - if not c.in_universe(mcap): - continue - - # Use to_dataframe() to get transactions + remaining shares - try: - txn_df = obj.to_dataframe() - except Exception: + if mcap is None: + stats["unresolved_market_cap"] += 1 continue - - if txn_df is None or len(txn_df) == 0: + if not c.in_universe(mcap): continue - # Filter to open-market purchases (code P) - p_mask = txn_df["Code"] == "P" if "Code" in txn_df.columns else None - if p_mask is None: - continue - purchases_df = txn_df[p_mask] + filing_date = str(getattr(filing, "filing_date", requested_date)) + accession = str( + getattr(filing, "accession_no", "") or getattr(filing, "accession_number", "") or "" + ) + cik = str(getattr(ownership.issuer, "cik", "") or "").lstrip("0") + source_url = c.sec_filing_url(cik, accession) if cik and accession else "" - for _, row in purchases_df.iterrows(): - shares = row.get("Shares", 0) or 0 - price = row.get("Price", 0) or 0 - if shares <= 0: + for _, row in purchases.iterrows(): + shares = _number(row.get("Shares")) + if shares is None or shares <= 0: continue + price = _number(row.get("Price")) + remaining = _number(row.get("Remaining Shares")) + insider = str(row.get("Insider") or getattr(ownership, "insider_name", "Unknown")) + owner_id, owner_cik = _owner_identity(ownership, insider) + role = str(row.get("Position") or getattr(ownership, "position", "Unknown")) + transaction_date = _date(row.get("Date"), filing_date) + purchases_by_ticker[ticker].append( + { + "insider": insider, + "insider_id": owner_id, + "insider_cik": owner_cik, + "role": role, + "shares": shares, + "price": price, + "transaction_date": transaction_date, + "filing_date": filing_date, + "company": str(getattr(ownership.issuer, "name", ticker)), + "mcap": mcap, + "remaining": remaining, + "shares_out": c.get_cached_shares_out(ticker, mcap_data), + "accession": accession, + "source_url": source_url, + } + ) + if index and index % 50 == 0: + c.log(f" Parsed {index}/{len(filings)} filings...") - insider_name = row.get("Insider") or getattr(obj, "insider_name", "Unknown") - position = row.get("Position") or getattr(obj, "position", "Unknown") - try: - company_name = getattr(obj.issuer, "name", ticker) - except Exception: - company_name = ticker - - remaining = row.get("Remaining Shares") - shares_out = c.get_cached_shares_out(ticker, mcap_data) - - key = f"{insider_name}|{date_str}" - if key not in seen_keys[ticker]: - seen_keys[ticker].add(key) - purchases_by_ticker[ticker].append( - { - "insider": insider_name, - "role": position, - "shares": shares, - "price": price, - "date": date_str, - "company": company_name, - "mcap": mcap, - "remaining": remaining, - "shares_out": shares_out, - } - ) - - if idx % 50 == 0 and idx > 0: - c.log(f" Parsed {idx}/{len(df)} filings...") - - c.log(f" Done with {date_str}") - - return dict(purchases_by_ticker) + return dict(purchases_by_ticker), stats def detect_clusters(purchases_by_ticker: dict[str, list[dict]]) -> list[dict]: - """Find cluster buys: 2+ distinct insiders buying the same ticker.""" + """Find tickers with purchases by at least two reporting owners.""" clusters = [] - for ticker, buys in purchases_by_ticker.items(): - unique_insiders = set(b["insider"] for b in buys) - if len(unique_insiders) >= 2: - # Collect signals present on any purchase in the cluster - signals = set() - for b in buys: - if b.get("signal"): - signals.add(b["signal"]) - - clusters.append( - { - "ticker": ticker, - "company": buys[0]["company"], - "mcap": buys[0]["mcap"], - "insiders": buys, - "num_insiders": len(unique_insiders), - "date_range": ( - f"{min(b['date'] for b in buys)} to {max(b['date'] for b in buys)}" - ), - "signals": sorted(signals), - # Use the first purchase's z-score data (same ticker, same values) - "return_30d": buys[0].get("return_30d"), - "zscore": buys[0].get("zscore"), - } - ) - - clusters.sort(key=lambda x: x["num_insiders"], reverse=True) + for ticker, purchases in purchases_by_ticker.items(): + distinct_owners = {purchase["insider_id"] for purchase in purchases} + if len(distinct_owners) < 2: + continue + clusters.append( + { + "ticker": ticker, + "company": purchases[0]["company"], + "mcap": purchases[0]["mcap"], + "insiders": purchases, + "num_insiders": len(distinct_owners), + "date_range": ( + f"{min(p['transaction_date'] for p in purchases)} to " + f"{max(p['transaction_date'] for p in purchases)}" + ), + "signals": sorted({p["signal"] for p in purchases if p.get("signal")}), + } + ) + clusters.sort(key=lambda cluster: cluster["num_insiders"], reverse=True) return clusters def collect_notable_singles( purchases_by_ticker: dict[str, list[dict]], cluster_tickers: set[str] ) -> list[dict]: - """Collect rip/dip-tagged purchases that aren't part of a cluster. - - These are individual insider buys that are notable because of the - volatility-adjusted price context, even without a second insider - confirming. - """ - notable = [] - for ticker, buys in purchases_by_ticker.items(): - if ticker in cluster_tickers: - continue - for b in buys: - if b.get("signal"): - notable.append({**b, "ticker": ticker}) - - # Sort by absolute z-score (most extreme first) - notable.sort(key=lambda x: abs(x.get("zscore") or 0), reverse=True) + """Collect event-context labels outside cluster tickers.""" + notable = [ + {**purchase, "ticker": ticker} + for ticker, purchases in purchases_by_ticker.items() + if ticker not in cluster_tickers + for purchase in purchases + if purchase.get("signal") + ] + notable.sort(key=lambda purchase: abs(purchase.get("zscore") or 0), reverse=True) return notable -# --------------------------------------------------------------------------- -# 13D scanning (unchanged) -# --------------------------------------------------------------------------- - - -def scan_13d(dates: list[str], cache: Path, mcap_data: dict) -> list[dict]: - """Scan SC 13D / 13D/A filings for activist blockholders.""" +def scan_13d(dates: list[str], mcap_data: dict) -> tuple[list[dict], dict]: + """Collect Schedule 13D and 13D/A filings without presuming activism.""" import edgar results = [] - for date_str in dates: - c.log(f"Fetching 13D/13D-A index for {date_str}...") + stats = {"failed_dates": [], "unresolved_ticker": 0, "unresolved_market_cap": 0} + for requested_date in dates: + c.log(f"Fetching Schedule 13D index for {requested_date}...") try: - dt = datetime.strptime(date_str, "%Y-%m-%d") + parsed_date = datetime.strptime(requested_date, "%Y-%m-%d") filings = edgar.get_filings( - year=dt.year, - quarter=(dt.month - 1) // 3 + 1, + year=parsed_date.year, + quarter=(parsed_date.month - 1) // 3 + 1, form=["SC 13D", "SC 13D/A"], - filing_date=date_str, + filing_date=requested_date, ) except Exception as exc: - c.log(f" WARNING: could not fetch 13D index for {date_str}: {exc}") + c.log(f" ERROR: could not fetch Schedule 13D index: {exc}") + stats["failed_dates"].append(requested_date) continue - if filings is None: continue for filing in filings: - company = getattr(filing, "company", "Unknown") - cik = getattr(filing, "cik", "") - + issuer_cik = str(getattr(filing, "cik", "") or "").lstrip("0") ticker = None try: from edgar import Company - co = Company(int(cik)) - tickers = getattr(co, "tickers", []) + tickers = getattr(Company(int(issuer_cik)), "tickers", []) if tickers: - ticker = next(iter(tickers)) + ticker = str(next(iter(tickers))).upper().replace(".", "-") except Exception: pass - if not ticker: + stats["unresolved_ticker"] += 1 continue mcap = c.get_market_cap(ticker, mcap_data) + if mcap is None: + stats["unresolved_market_cap"] += 1 + continue if not c.in_universe(mcap): continue + blockholders = "(see filing)" + try: + schedule = filing.obj() + names = [ + str(person.name) + for person in (getattr(schedule, "reporting_persons", []) or []) + if getattr(person, "name", None) + ] + if names: + blockholders = "; ".join(dict.fromkeys(names)) + except Exception: + pass + + accession = str( + getattr(filing, "accession_no", "") or getattr(filing, "accession_number", "") or "" + ) results.append( { - "ticker": ticker.upper(), - "company": company, + "ticker": ticker, + "company": str(getattr(filing, "company", "Unknown")), "mcap": mcap, - "filer": getattr(filing, "company", "Unknown"), - "date": date_str, - "form": getattr(filing, "form", "SC 13D"), - "accession": ( - getattr(filing, "accession_no", "") - or getattr(filing, "accession_number", "") - ), + "blockholders": blockholders, + "filing_date": str(getattr(filing, "filing_date", requested_date)), + "form": str(getattr(filing, "form", "SC 13D")), + "accession": accession, + "source_url": c.sec_filing_url(issuer_cik, accession), } ) - - return results + return results, stats -# --------------------------------------------------------------------------- -# Markdown output -# --------------------------------------------------------------------------- +def _signal_badge(signal: str | None) -> str: + return " 🚀" if signal == "rip" else " 🔻" if signal == "dip" else "" -def _signal_badge(signal: str | None) -> str: - if signal == "rip": - return " 🚀" - if signal == "dip": - return " 🔻" - return "" +def _zscore(value: float | None) -> str: + return "n/a" if value is None else f"{value:+.1f}σ" -def _zscore_str(z: float | None) -> str: - if z is None: - return "n/a" - return f"{z:+.1f}σ" +def _return(value: float | None) -> str: + return "n/a" if value is None else f"{value * 100:+.1f}%" -def _return_str(r: float | None) -> str: - if r is None: - return "n/a" - return f"{r * 100:+.1f}%" +def _price(value: float | None) -> str: + return "n/a" if value is None else f"${value:.2f}" def _pct_of_holding(shares: float, remaining: float | None) -> str: - """Purchase as % of post-transaction holding.""" - if not remaining or remaining <= 0: - return "n/a" - return f"{shares / remaining * 100:.1f}%" + return "n/a" if not remaining or remaining <= 0 else f"{shares / remaining * 100:.1f}%" def _pct_of_outstanding(shares: float, shares_out: int | None) -> str: - """Purchase as % of total shares outstanding.""" if not shares_out or shares_out <= 0: return "n/a" - pct = shares / shares_out * 100 - if pct < 0.01: - return "<0.01%" - return f"{pct:.2f}%" + percentage = shares / shares_out * 100 + return "<0.01%" if percentage < 0.01 else f"{percentage:.2f}%" def _build_summary( purchases_by_ticker: dict[str, list[dict]], clusters: list[dict], - notable: list[dict], filings_13d: list[dict], mcap_data: dict, - zscore_threshold: float = 1.5, + threshold: float, ) -> list[str]: - """Build a summary header with key stats and sector breakdown.""" - lines = [] - - # Aggregate stats - all_purchases = [b for buys in purchases_by_ticker.values() for b in buys] - total_purchases = len(all_purchases) - unique_tickers = len(purchases_by_ticker) - unique_insiders = len(set(b["insider"] for b in all_purchases)) - total_dollar = sum(b["shares"] * b["price"] for b in all_purchases) - rip_count = sum(1 for b in all_purchases if b.get("signal") == "rip") - dip_count = sum(1 for b in all_purchases if b.get("signal") == "dip") - - lines.append("## Summary\n") - lines.append("| Metric | Value |") - lines.append("|--------|-------|") - lines.append(f"| Purchases (code P, {c.universe_label()}) | {total_purchases} |") - lines.append(f"| Unique tickers | {unique_tickers} |") - lines.append(f"| Unique insiders | {unique_insiders} |") - lines.append(f"| Total dollar volume | ${total_dollar:,.0f} |") - lines.append(f"| Cluster buys | {len(clusters)} |") - lines.append(f"| Dip buys (\u2264 -{zscore_threshold}\u03c3) | {dip_count} |") - lines.append(f"| Rip buys (\u2265 +{zscore_threshold}\u03c3) | {rip_count} |") - lines.append(f"| 13D filings | {len(filings_13d)} |") - lines.append("") - - # Largest purchase - if all_purchases: - largest = max(all_purchases, key=lambda b: b["shares"] * b["price"]) - lval = largest["shares"] * largest["price"] - # Find the ticker for this purchase - lticker = "" - for t, buys in purchases_by_ticker.items(): - if largest in buys: - lticker = t - break + """Build report-level counts and sector totals.""" + purchases = [purchase for rows in purchases_by_ticker.values() for purchase in rows] + priced = [purchase for purchase in purchases if purchase["price"] is not None] + total_dollar = sum(purchase["shares"] * purchase["price"] for purchase in priced) + lines = [ + "## Summary\n", + "| Metric | Value |", + "|---|---:|", + f"| Code-P transaction rows ({c.universe_label()}) | {len(purchases)} |", + f"| Unique tickers | {len(purchases_by_ticker)} |", + f"| Distinct reporting owners | {len({p['insider_id'] for p in purchases})} |", + f"| Priced purchase value | ${total_dollar:,.0f} |", + f"| Cluster tickers | {len(clusters)} |", + f"| Dip rows (≤ -{threshold}σ) | {sum(p.get('signal') == 'dip' for p in purchases)} |", + f"| Rip rows (≥ +{threshold}σ) | {sum(p.get('signal') == 'rip' for p in purchases)} |", + f"| Schedule 13D filings | {len(filings_13d)} |", + "", + ] + + if priced: + largest = max(priced, key=lambda purchase: purchase["shares"] * purchase["price"]) + ticker = next(t for t, rows in purchases_by_ticker.items() if largest in rows) lines.append( - f"**Largest purchase:** {largest['insider']} " - f"({largest['role']}) bought ${lval:,.0f} of " - f"{lticker} ({largest['company']})\n" + f"**Largest priced row:** {c.md_cell(largest['insider'])} bought " + f"${largest['shares'] * largest['price']:,.0f} of {ticker}.\n" ) - # Sector breakdown sector_counts: dict[str, int] = defaultdict(int) sector_dollars: dict[str, float] = defaultdict(float) - for ticker, buys in purchases_by_ticker.items(): + for ticker, rows in purchases_by_ticker.items(): sector = c.get_cached_sector(ticker, mcap_data) - sector_counts[sector] += len(buys) - sector_dollars[sector] += sum(b["shares"] * b["price"] for b in buys) - + sector_counts[sector] += len(rows) + sector_dollars[sector] += sum( + row["shares"] * row["price"] for row in rows if row["price"] is not None + ) if sector_counts: - # Sort by dollar volume descending - sorted_sectors = sorted(sector_dollars.items(), key=lambda x: x[1], reverse=True) - lines.append("### Sector Breakdown\n") - lines.append("| Sector | Purchases | Dollar Volume |") - lines.append("|--------|-----------|---------------|") - for sector, dollars in sorted_sectors: - count = sector_counts[sector] - lines.append(f"| {sector} | {count} | ${dollars:,.0f} |") + lines += [ + "### Sector breakdown\n", + "| Sector | Transaction rows | Priced value |", + "|---|---:|---:|", + ] + for sector, dollars in sorted( + sector_dollars.items(), key=lambda item: item[1], reverse=True + ): + lines.append(f"| {c.md_cell(sector)} | {sector_counts[sector]} | ${dollars:,.0f} |") lines.append("") - return lines +def _coverage_notes(form4_stats: dict, schedule_stats: dict) -> list[str]: + """Render omissions that affect interpretation of an otherwise valid report.""" + notes = [] + if form4_stats["failed_dates"]: + notes.append( + f"Form 4 index retrieval failed for: {', '.join(form4_stats['failed_dates'])}." + ) + if schedule_stats["failed_dates"]: + notes.append( + f"Schedule 13D index retrieval failed for: {', '.join(schedule_stats['failed_dates'])}." + ) + if form4_stats["parse_errors"]: + notes.append( + f"{form4_stats['parse_errors']} of {form4_stats['filings_seen']} Form 4 filings could not be parsed." + ) + if schedule_stats["unresolved_ticker"]: + notes.append( + f"No current ticker resolved for {schedule_stats['unresolved_ticker']} Schedule 13D filing(s); " + "those were omitted." + ) + unresolved = form4_stats["unresolved_market_cap"] + schedule_stats["unresolved_market_cap"] + if unresolved: + notes.append( + f"Yahoo market cap was unavailable for {unresolved} filing ticker lookup(s); those were omitted." + ) + if not notes: + return [] + return ["## Coverage notes\n", *[f"- {note}" for note in notes], ""] + + +def _source(purchase: dict) -> str: + accession = purchase.get("accession") or "filing" + url = purchase.get("source_url") + return f"[{accession}]({url})" if url else accession + + def _build_markdown( purchases_by_ticker: dict[str, list[dict]], clusters: list[dict], @@ -522,320 +498,219 @@ def _build_markdown( filings_13d: list[dict], dates: list[str], mcap_data: dict, - zscore_threshold: float = 1.5, + form4_stats: dict, + schedule_stats: dict, + threshold: float, ) -> str: - """Build Markdown output from scan results.""" - lines = [] + """Build the source-linked scan report.""" date_range = f"{dates[0]} to {dates[-1]}" if len(dates) > 1 else dates[0] - lines.append(f"# Insider Activity Scan ({date_range})\n") - - # Summary header - lines.extend( - _build_summary( - purchases_by_ticker, - clusters, - notable, - filings_13d, - mcap_data, - zscore_threshold=zscore_threshold, - ) - ) + lines = [f"# Insider Filing Scan ({date_range})\n"] + lines.extend(_coverage_notes(form4_stats, schedule_stats)) + lines.extend(_build_summary(purchases_by_ticker, clusters, filings_13d, mcap_data, threshold)) - # --- Cluster buys --- - lines.append(f"## Cluster Buys ({len(clusters)} found)\n") + lines.append(f"## Cluster purchases ({len(clusters)} tickers)\n") + lines.append( + "A cluster has code-P rows from at least two distinct reporting owners in Form 4 filings received during the scan window.\n" + ) if not clusters: - lines.append("No cluster buys detected in this window.\n") - else: - lines.append( - "A cluster buy = 2+ distinct insiders buying the same stock within the window.\n" - ) - for cl in clusters: - badges = "".join(_signal_badge(s) for s in cl.get("signals", [])) - lines.append(f"### {cl['ticker']} — {cl['company']}{badges}") - lines.append(f"- **Market Cap:** {c.fmt_mcap(cl['mcap'])}") - lines.append(f"- **Insiders buying:** {cl['num_insiders']}") - lines.append(f"- **Window:** {cl['date_range']}") - z = cl.get("zscore") - r = cl.get("return_30d") - if z is not None: - lines.append( - f"- **30-day move:** {_return_str(r)} ({_zscore_str(z)} vs own history)" - ) + lines.append("No clusters were detected in completed coverage.\n") + for cluster in clusters: + badges = "".join(_signal_badge(signal) for signal in cluster["signals"]) + lines += [ + f"### {cluster['ticker']} — {c.md_cell(cluster['company'])}{badges}", + f"- **Market cap:** {c.fmt_mcap(cluster['mcap'])}", + f"- **Distinct reporting owners:** {cluster['num_insiders']}", + f"- **Transaction-date range:** {cluster['date_range']}", + "", + "| Insider | Role | Shares | Price | % of Holding | % of O/S | Transaction date | Filed | Source |", + "|---|---|---:|---:|---:|---:|---|---|---|", + ] + for purchase in cluster["insiders"]: lines.append( - f"- **Link:** [Yahoo Finance](https://finance.yahoo.com/quote/{cl['ticker']})" + f"| {c.md_cell(purchase['insider'])} | {c.md_cell(purchase['role'])} | " + f"{purchase['shares']:,.0f} | {_price(purchase['price'])} | " + f"{_pct_of_holding(purchase['shares'], purchase['remaining'])} | " + f"{_pct_of_outstanding(purchase['shares'], purchase['shares_out'])} | " + f"{purchase['transaction_date']} | {purchase['filing_date']} | {_source(purchase)} |" ) - lines.append("") - lines.append("| Insider | Role | Shares | Price | % of Holding | % of O/S | Date |") - lines.append("|---------|------|--------|-------|--------------|----------|------|") - for ins in cl["insiders"]: - lines.append( - f"| {ins['insider']} | {ins['role']} | " - f"{ins['shares']:,.0f} | ${ins['price']:.2f} | " - f"{_pct_of_holding(ins['shares'], ins.get('remaining'))} | " - f"{_pct_of_outstanding(ins['shares'], ins.get('shares_out'))} | " - f"{ins['date']} |" - ) - lines.append("") + lines.append("") - # --- Notable individual purchases (rip/dip) --- if notable: - rips = [n for n in notable if n.get("signal") == "rip"] - dips = [n for n in notable if n.get("signal") == "dip"] - - lines.append(f"## Notable Individual Purchases ({len(rips)} rip, {len(dips)} dip)\n") - lines.append( - "Volatility-adjusted: the 30-day return is measured against " - "the stock's own historical volatility. A z-score beyond " - f"±{zscore_threshold}σ flags the purchase as unusual.\n" - ) - - if dips: - lines.append("### 🔻 Dip Buys (buying into unusual weakness)\n") - lines.append( - "| Ticker | Company | Insider | Role | Shares | Price " - "| % of Holding | 30d Move | Z-Score | Mkt Cap |" - ) - lines.append( - "|--------|---------|---------|------|--------|-------" - "|--------------|---------|---------|---------|" - ) - for n in dips: - lines.append( - f"| {n['ticker']} | {n['company']} | {n['insider']} | " - f"{n['role']} | {n['shares']:,.0f} | ${n['price']:.2f} | " - f"{_pct_of_holding(n['shares'], n.get('remaining'))} | " - f"{_return_str(n.get('return_30d'))} | " - f"{_zscore_str(n.get('zscore'))} | " - f"{c.fmt_mcap(n.get('mcap'))} |" - ) - lines.append("") - - if rips: - lines.append("### 🚀 Rip Buys (buying into unusual strength)\n") - lines.append( - "| Ticker | Company | Insider | Role | Shares | Price " - "| % of Holding | 30d Move | Z-Score | Mkt Cap |" - ) - lines.append( - "|--------|---------|---------|------|--------|-------" - "|--------------|---------|---------|---------|" - ) - for n in rips: + dips = [purchase for purchase in notable if purchase["signal"] == "dip"] + rips = [purchase for purchase in notable if purchase["signal"] == "rip"] + lines += [ + f"## Event-date move context ({len(dips)} dip, {len(rips)} rip)\n", + "The 22-trading-day return is compared with the stock's prior rolling 22-day returns as of the transaction date.\n", + ] + for label, rows in (("Dip", dips), ("Rip", rips)): + if not rows: + continue + lines += [ + f"### {label} rows\n", + "| Ticker | Company | Insider | Shares | Price | 22d Move | Z-score | Transaction date | Filed | Source |", + "|---|---|---|---:|---:|---:|---:|---|---|---|", + ] + for purchase in rows: lines.append( - f"| {n['ticker']} | {n['company']} | {n['insider']} | " - f"{n['role']} | {n['shares']:,.0f} | ${n['price']:.2f} | " - f"{_pct_of_holding(n['shares'], n.get('remaining'))} | " - f"{_return_str(n.get('return_30d'))} | " - f"{_zscore_str(n.get('zscore'))} | " - f"{c.fmt_mcap(n.get('mcap'))} |" + f"| {purchase['ticker']} | {c.md_cell(purchase['company'])} | " + f"{c.md_cell(purchase['insider'])} | {purchase['shares']:,.0f} | " + f"{_price(purchase['price'])} | {_return(purchase['return_22d'])} | " + f"{_zscore(purchase['zscore'])} | {purchase['transaction_date']} | " + f"{purchase['filing_date']} | {_source(purchase)} |" ) lines.append("") - # --- 13D filings --- - lines.append(f"## Activist 13D Filings ({len(filings_13d)} found)\n") + lines.append(f"## Schedule 13D filings ({len(filings_13d)})\n") + lines.append( + "A Schedule 13D is a beneficial-ownership filing under Section 13(d); it does not by itself establish an activist campaign.\n" + ) if not filings_13d: - lines.append( - f"No SC 13D / 13D/A filings in the {c.universe_label()} universe for this window.\n" - ) + lines.append("No Schedule 13D or 13D/A filings were found in completed coverage.\n") else: - lines.append("| Ticker | Company | Market Cap | Filer | Date | Form |") - lines.append("|--------|---------|-----------|-------|------|------|") - for f13 in filings_13d: + lines += [ + "| Ticker | Issuer | Market Cap | Reporting person(s) | Filed | Form | Source |", + "|---|---|---:|---|---|---|---|", + ] + for filing in filings_13d: lines.append( - f"| {f13['ticker']} | {f13['company']} | " - f"{c.fmt_mcap(f13['mcap'])} | {f13['filer']} | " - f"{f13['date']} | {f13['form']} |" + f"| {filing['ticker']} | {c.md_cell(filing['company'])} | " + f"{c.fmt_mcap(filing['mcap'])} | {c.md_cell(filing['blockholders'])} | " + f"{filing['filing_date']} | {filing['form']} | " + f"[{filing['accession']}]({filing['source_url']}) |" ) lines.append("") - return "\n".join(lines) -# --------------------------------------------------------------------------- -# Discord embeds -# --------------------------------------------------------------------------- - - def _build_discord_embeds( clusters: list[dict], notable: list[dict], filings_13d: list[dict] ) -> list[dict]: - """Build Discord embed objects for webhook posting.""" + """Build concise, source-linked Discord alerts.""" embeds = [] - - for cl in clusters: - insider_lines = [] - for ins in cl["insiders"]: - insider_lines.append( - f"• {ins['insider']} ({ins['role']}) — " - f"{ins['shares']:,.0f} shares @ ${ins['price']:.2f}" - ) - signals = cl.get("signals", []) - signal_str = "" - if signals: - tags = [] - if "dip" in signals: - tags.append("DIP BUY") - if "rip" in signals: - tags.append("RIP BUY") - signal_str = " [" + " + ".join(tags) + "]" - - fields = [ - {"name": "Ticker", "value": cl["ticker"], "inline": True}, - {"name": "Market Cap", "value": c.fmt_mcap(cl["mcap"]), "inline": True}, - {"name": "Window", "value": cl["date_range"], "inline": True}, - {"name": "Insiders", "value": str(cl["num_insiders"]), "inline": True}, + for cluster in clusters: + descriptions = [ + f"• {row['insider']} ({row['role']}) — {row['shares']:,.0f} shares @ {_price(row['price'])}" + for row in cluster["insiders"] ] - z = cl.get("zscore") - r = cl.get("return_30d") - if z is not None: - fields.append( - { - "name": "30d Move", - "value": f"{_return_str(r)} ({_zscore_str(z)})", - "inline": True, - } - ) - embeds.append( { - "title": ( - f"\U0001f7e2 Insider Cluster Buy — {cl['ticker']} ({cl['company']}){signal_str}" - ), - "url": f"https://finance.yahoo.com/quote/{cl['ticker']}", + "title": f"Insider purchase cluster — {cluster['ticker']} ({cluster['company']})", + "url": cluster["insiders"][0]["source_url"], "color": 0x2ECC71, - "description": "\n".join(insider_lines), - "fields": fields, + "description": "\n".join(descriptions), + "fields": [ + {"name": "Market Cap", "value": c.fmt_mcap(cluster["mcap"]), "inline": True}, + {"name": "Owners", "value": str(cluster["num_insiders"]), "inline": True}, + {"name": "Transaction dates", "value": cluster["date_range"], "inline": True}, + ], } ) - - # Notable singles — only post the most extreme (top 10) - for n in notable[:10]: - signal = n.get("signal", "") - if signal == "dip": - emoji = "\U0001f4c9" # 📉 - color = 0x3498DB # blue - label = "Dip Buy" - else: - emoji = "\U0001f4c8" # 📈 - color = 0xE74C3C # red - label = "Rip Buy" - + for purchase in notable: + label = "Dip" if purchase["signal"] == "dip" else "Rip" embeds.append( { - "title": (f"{emoji} {label} — {n['ticker']} ({n['company']})"), - "url": f"https://finance.yahoo.com/quote/{n['ticker']}", - "color": color, + "title": f"{label} purchase context — {purchase['ticker']} ({purchase['company']})", + "url": purchase["source_url"], + "color": 0x3498DB if label == "Dip" else 0xE74C3C, "description": ( - f"• {n['insider']} ({n['role']}) — " - f"{n['shares']:,.0f} shares @ ${n['price']:.2f}" + f"{purchase['insider']} ({purchase['role']}) — " + f"{purchase['shares']:,.0f} shares @ {_price(purchase['price'])}" ), "fields": [ - {"name": "30d Move", "value": _return_str(n.get("return_30d")), "inline": True}, - {"name": "Z-Score", "value": _zscore_str(n.get("zscore")), "inline": True}, - {"name": "Market Cap", "value": c.fmt_mcap(n.get("mcap")), "inline": True}, + {"name": "22d move", "value": _return(purchase["return_22d"]), "inline": True}, + {"name": "Z-score", "value": _zscore(purchase["zscore"]), "inline": True}, + { + "name": "Transaction date", + "value": purchase["transaction_date"], + "inline": True, + }, ], } ) - - for f13 in filings_13d: + for filing in filings_13d: embeds.append( { - "title": (f"\U0001f3db\ufe0f Activist 13D — {f13['ticker']} ({f13['company']})"), - "url": f"https://finance.yahoo.com/quote/{f13['ticker']}", + "title": f"Schedule 13D filing — {filing['ticker']} ({filing['company']})", + "url": filing["source_url"], "color": 0xE67E22, "fields": [ - {"name": "Filer", "value": f13["filer"], "inline": True}, - {"name": "Market Cap", "value": c.fmt_mcap(f13["mcap"]), "inline": True}, - {"name": "Date", "value": f13["date"], "inline": True}, + { + "name": "Reporting person(s)", + "value": filing["blockholders"], + "inline": False, + }, + {"name": "Market Cap", "value": c.fmt_mcap(filing["mcap"]), "inline": True}, + {"name": "Filed", "value": filing["filing_date"], "inline": True}, ], } ) - return embeds -# --------------------------------------------------------------------------- -# Main -# --------------------------------------------------------------------------- - - -def main(): - p = argparse.ArgumentParser( - description="Scan Form 4 cluster buys, rip/dip buys, and 13D filings." - ) - p.add_argument( - "--date", - required=True, - help='Filing date to scan (YYYY-MM-DD, "today", or "yesterday").', - ) - p.add_argument( - "--lookback", - type=int, - default=5, - help="Days to look back for cluster detection (default: 5).", +def main() -> None: + """Run the insider-filing scan CLI.""" + parser = argparse.ArgumentParser(description="Scan Form 4 purchases and Schedule 13D filings.") + parser.add_argument( + "--date", required=True, help='End filing date (YYYY-MM-DD, "today", or "yesterday").' ) - p.add_argument( - "--zscore", - type=float, - default=1.5, - help="Z-score threshold for rip/dip tagging (default: 1.5).", + parser.add_argument("--lookback", type=int, default=5, help="Weekdays to scan (default: 5).") + parser.add_argument( + "--zscore", type=float, default=1.5, help="Absolute dip/rip threshold (default: 1.5)." ) - p.add_argument( - "--webhook", - help="Discord webhook URL (optional; if omitted, no Discord post).", - ) - c.add_identity_arg(p) - c.add_cache_arg(p) - args = p.parse_args() + parser.add_argument("--webhook", help="Discord webhook URL (else $DISCORD_WEBHOOK_URL).") + c.add_identity_arg(parser) + c.add_cache_arg(parser) + args = parser.parse_args() + + if args.lookback < 1: + parser.error("--lookback must be at least 1.") + if args.zscore <= 0: + parser.error("--zscore must be greater than zero.") c.resolve_identity(args.identity) cache = c.cache_root(args.cache_dir) - end_date = c.parse_date(args.date) - dates = _trading_dates(end_date, args.lookback) - c.log(f"Scanning {len(dates)} trading days: {dates[0]} to {dates[-1]}") + dates = _weekdays(end_date, args.lookback) + c.log(f"Scanning {len(dates)} weekdays: {dates[0]} to {dates[-1]}") mcap_data = c.load_mcap_cache(cache) - try: - purchases = scan_form4s(dates, cache, mcap_data) - filings_13d = scan_13d(dates, cache, mcap_data) - - # Tag rip/dip on all purchases + purchases, form4_stats = scan_form4s(dates, mcap_data) + filings_13d, schedule_stats = scan_13d(dates, mcap_data) if purchases: - tag_rip_dip(purchases, zscore_threshold=args.zscore) - - # Detect clusters + tag_move_context(purchases, args.zscore) clusters = detect_clusters(purchases) - cluster_tickers = set(cl["ticker"] for cl in clusters) - - # Collect notable singles (rip/dip that aren't in a cluster) - notable = collect_notable_singles(purchases, cluster_tickers) + notable = collect_notable_singles(purchases, {cluster["ticker"] for cluster in clusters}) finally: c.save_mcap_cache(cache, mcap_data) - total_rip = sum(1 for t in purchases.values() for b in t if b.get("signal") == "rip") - total_dip = sum(1 for t in purchases.values() for b in t if b.get("signal") == "dip") - c.log( - f"Found {len(clusters)} cluster buys, " - f"{total_rip} rip buys, {total_dip} dip buys, " - f"{len(filings_13d)} 13D filings" + report = _build_markdown( + purchases, + clusters, + notable, + filings_13d, + dates, + mcap_data, + form4_stats, + schedule_stats, + args.zscore, ) + c.write_output(cache, "insiders", end_date, report) - md = _build_markdown( - purchases, clusters, notable, filings_13d, dates, mcap_data, zscore_threshold=args.zscore - ) - slug = end_date - c.write_output(cache, "insiders", slug, md) - - if args.webhook: + webhook = args.webhook or os.environ.get("DISCORD_WEBHOOK_URL") + if webhook: embeds = _build_discord_embeds(clusters, notable, filings_13d) if embeds: - c.log(f"Posting {len(embeds)} embeds to Discord...") - _post_discord(args.webhook, embeds) - c.log("Discord post complete.") - else: - c.log("No embeds to post to Discord.") + try: + _post_discord(webhook, embeds) + except RuntimeError as exc: + c.log(f"ERROR: {exc}") + sys.exit(1) + + total_parser_failure = form4_stats["filings_seen"] > 0 and form4_stats["filings_parsed"] == 0 + if form4_stats["failed_dates"] or schedule_stats["failed_dates"] or total_parser_failure: + c.log("ERROR: emitted report has incomplete SEC index or parser coverage.") + sys.exit(1) if __name__ == "__main__": diff --git a/skills/signal-sweep/scripts/scan_market.py b/skills/signal-sweep/scripts/scan_market.py index abcb230..6858c51 100644 --- a/skills/signal-sweep/scripts/scan_market.py +++ b/skills/signal-sweep/scripts/scan_market.py @@ -1,13 +1,7 @@ -"""Config-driven market screens via yfinance EquityQuery. +"""Run config-driven Yahoo Finance equity screens. -Reads screen definitions from screens.json, injects universe bounds -($50M-$10B, region=us), runs yf.screen(), and optionally enriches the -top results with Ticker.info data. - -Usage: - python scripts/scan_market.py --screen near-52wk-low - python scripts/scan_market.py --all - python scripts/scan_market.py --list +Definitions and universe bounds come from ``screens.json``. Each successful run +writes a Markdown report and emits its absolute path. """ from __future__ import annotations @@ -22,16 +16,20 @@ import _common as c -def _load_screens(screens_file: Path) -> dict: - """Load screens.json and return the parsed dict.""" - if not screens_file.exists(): - c.log(f"ERROR: screens file not found: {screens_file}") +def _load_screens(path: Path) -> dict: + """Load and parse a screen configuration.""" + if not path.exists(): + c.log(f"ERROR: screens file not found: {path}") + sys.exit(1) + try: + return json.loads(path.read_text(encoding="utf-8")) + except (OSError, json.JSONDecodeError) as exc: + c.log(f"ERROR: invalid screens file {path}: {exc}") sys.exit(1) - return json.loads(screens_file.read_text(encoding="utf-8")) def _build_query(screen: dict, universe: dict): - """Build a yfinance EquityQuery from screen config + universe bounds.""" + """Build a yfinance EquityQuery from one definition and its universe.""" import yfinance as yf conditions = [ @@ -41,222 +39,206 @@ def _build_query(screen: dict, universe: dict): "lte", ["intradaymarketcap", universe.get("market_cap_max", 10_000_000_000)] ), ] - - op_map = {"eq": "eq", "lte": "lte", "gte": "gte", "lt": "lt", "gt": "gt", "btwn": "btwn"} - - for f in screen.get("filters", []): - op = op_map.get(f["op"], f["op"]) - if op == "btwn": - conditions.append(yf.EquityQuery(op, [f["field"], f["value"][0], f["value"][1]])) - else: - conditions.append(yf.EquityQuery(op, [f["field"], f["value"]])) - + for filter_ in screen.get("filters", []): + value = filter_["value"] + operands = ( + [filter_["field"], *value] if isinstance(value, list) else [filter_["field"], value] + ) + conditions.append(yf.EquityQuery(filter_["op"], operands)) return yf.EquityQuery("and", conditions) def _enrich(quotes: list[dict], size: int) -> list[dict]: - """Enrich top results with yf.Ticker().info data.""" + """Add report columns from Yahoo's per-ticker snapshot.""" import yfinance as yf enriched = [] - for i, q in enumerate(quotes[:size]): - symbol = q.get("symbol", "") - c.log(f" Enriching {i + 1}/{min(size, len(quotes))}: {symbol}") + for index, quote in enumerate(quotes[:size]): + symbol = quote.get("symbol", "") + c.log(f" Enriching {index + 1}/{min(size, len(quotes))}: {symbol}") try: info = yf.Ticker(symbol).info or {} except Exception: info = {} - q["analyst_rating"] = info.get("averageAnalystRating", "n/a") - q["target_mean"] = info.get("targetMeanPrice") - q["target_median"] = info.get("targetMedianPrice") - q["short_pct"] = info.get("shortPercentOfFloat") - q["insider_pct"] = info.get("heldPercentInsiders") - q["inst_pct"] = info.get("heldPercentInstitutions") - q["earnings_growth"] = info.get("earningsQuarterlyGrowth") - q["revenue_growth"] = info.get("revenueGrowth") - q["sector"] = info.get("sector", q.get("sector", "")) - q["industry"] = info.get("industry", q.get("industry", "")) - q["pe"] = info.get("trailingPE") or info.get("forwardPE") - q["current_price"] = info.get("currentPrice", q.get("regularMarketPrice")) - q["market_cap"] = info.get("marketCap", q.get("marketCap")) - - # Trailing 2Y/3Y returns for the "forgotten" screen - try: - hist = yf.Ticker(symbol).history(period="3y") - if hist is not None and len(hist) > 1: - close = hist["Close"].dropna() - last = close.iloc[-1] - import pandas as pd - - last_date = close.index[-1] - # 2Y return - prior_2y = close[close.index <= last_date - pd.Timedelta(days=730)] - if len(prior_2y): - q["return_2y"] = (last / prior_2y.iloc[-1] - 1) * 100 - # 3Y return - prior_3y = close[close.index <= last_date - pd.Timedelta(days=1095)] - if len(prior_3y): - q["return_3y"] = (last / prior_3y.iloc[-1] - 1) * 100 - except Exception: - pass - - enriched.append(q) - + quote["analyst_rating"] = info.get("averageAnalystRating", "n/a") + quote["short_pct"] = info.get("shortPercentOfFloat") + quote["insider_pct"] = info.get("heldPercentInsiders") + quote["inst_pct"] = info.get("heldPercentInstitutions") + quote["sector"] = info.get("sector", quote.get("sector", "")) + quote["pe"] = info.get("trailingPE") or info.get("forwardPE") + quote["current_price"] = info.get("currentPrice", quote.get("regularMarketPrice")) + quote["market_cap"] = info.get("marketCap", quote.get("marketCap")) + enriched.append(quote) return enriched -def _fmt_pct(v, mult100=False) -> str: - if v is None: +def _fmt_pct(value, *, decimal: bool = False) -> str: + if value is None: return "n/a" - val = v * 100 if mult100 else v - return f"{val:.1f}%" + return f"{value * 100 if decimal else value:.1f}%" -def _fmt_num(v) -> str: - if v is None: - return "n/a" - return f"{v:.1f}" +def _fmt_num(value) -> str: + return "n/a" if value is None else f"{value:.1f}" -def _render_markdown(screen: dict, quotes: list[dict], enriched: bool) -> str: +def _render_markdown( + screen: dict, + quotes: list[dict], + *, + enriched: bool, + universe: dict, + total: int | str, +) -> str: """Render screen results as a Markdown table.""" - lines = [] emoji = screen.get("emoji", "📊") name = screen.get("name", screen.get("id", "Screen")) - lines.append(f"# {emoji} Screen: {name} ({c.universe_label()})\n") - lines.append(f"_{screen.get('description', '')}_\n") - lines.append(f"**Results:** {len(quotes)}\n") + lines = [f"# {emoji} Screen: {name} ({c.universe_label(universe)})\n"] + if screen.get("description"): + lines.append(f"_{screen['description']}_\n") + lines.append(f"**Results returned:** {len(quotes)} of {total} matching\n") if not quotes: lines.append("No results matched this screen.\n") return "\n".join(lines) if enriched: - lines.append( - "| # | Ticker | Company | Price | Mkt Cap | P/E | Short % | Insider % | Inst % | Analyst | Sector |" - ) - lines.append( - "|---|--------|---------|-------|---------|-----|---------|-----------|--------|---------|--------|" - ) - for i, q in enumerate(quotes, 1): - ticker = q.get("symbol", "?") - company = q.get("shortName") or q.get("longName") or "?" - price = q.get("current_price") or q.get("regularMarketPrice") - price_s = f"${price:.2f}" if price else "n/a" - mcap_s = c.fmt_mcap(q.get("market_cap")) - pe_s = _fmt_num(q.get("pe")) - short_s = _fmt_pct(q.get("short_pct"), mult100=True) - insider_s = _fmt_pct(q.get("insider_pct"), mult100=True) - inst_s = _fmt_pct(q.get("inst_pct"), mult100=True) - analyst = q.get("analyst_rating", "n/a") - sector = q.get("sector", "n/a") + lines += [ + "| # | Ticker | Company | Price | Mkt Cap | P/E | Short % | Insider % | Inst % | Analyst | Sector |", + "|---|---|---|---:|---:|---:|---:|---:|---:|---|---|", + ] + for index, quote in enumerate(quotes, 1): + price = quote.get("current_price") lines.append( - f"| {i} | {ticker} | {company} | {price_s} | {mcap_s} | " - f"{pe_s} | {short_s} | {insider_s} | {inst_s} | {analyst} | {sector} |" + f"| {index} | {c.md_cell(quote.get('symbol', '?'))} | " + f"{c.md_cell(quote.get('shortName') or quote.get('longName') or '?')} | " + f"{f'${price:.2f}' if price is not None else 'n/a'} | " + f"{c.fmt_mcap(quote.get('market_cap'))} | {_fmt_num(quote.get('pe'))} | " + f"{_fmt_pct(quote.get('short_pct'), decimal=True)} | " + f"{_fmt_pct(quote.get('insider_pct'), decimal=True)} | " + f"{_fmt_pct(quote.get('inst_pct'), decimal=True)} | " + f"{c.md_cell(quote.get('analyst_rating', 'n/a'))} | " + f"{c.md_cell(quote.get('sector') or 'n/a')} |" ) else: - lines.append("| # | Ticker | Company | Price | Mkt Cap |") - lines.append("|---|--------|---------|-------|---------|") - for i, q in enumerate(quotes, 1): - ticker = q.get("symbol", "?") - company = q.get("shortName") or q.get("longName") or "?" - price = q.get("regularMarketPrice") - price_s = f"${price:.2f}" if price else "n/a" - mcap_s = c.fmt_mcap(q.get("marketCap")) - lines.append(f"| {i} | {ticker} | {company} | {price_s} | {mcap_s} |") - + lines += [ + "| # | Ticker | Company | Price | Mkt Cap |", + "|---|---|---|---:|---:|", + ] + for index, quote in enumerate(quotes, 1): + price = quote.get("regularMarketPrice") + lines.append( + f"| {index} | {c.md_cell(quote.get('symbol', '?'))} | " + f"{c.md_cell(quote.get('shortName') or quote.get('longName') or '?')} | " + f"{f'${price:.2f}' if price is not None else 'n/a'} | " + f"{c.fmt_mcap(quote.get('marketCap'))} |" + ) lines.append("") return "\n".join(lines) -def run_screen(screen: dict, universe: dict, size: int | None, do_enrich: bool, cache: Path) -> str: - """Run a single screen and return Markdown output.""" +def run_screen( + screen: dict, + universe: dict, + size: int | None, + do_enrich: bool, +) -> tuple[str, bool]: + """Run one screen and return its report plus a success flag.""" import yfinance as yf screen_id = screen.get("id", "unknown") - screen_size = size or screen.get("size", 25) + screen_size = size if size is not None else screen.get("size", 25) c.log(f"Running screen: {screen_id} (size={screen_size})...") - query = _build_query(screen, universe) - sort_field = screen.get("sort", {}).get("field") - sort_asc = screen.get("sort", {}).get("asc", True) - try: + query = _build_query(screen, universe) kwargs = {"query": query, "size": screen_size} - if sort_field: - kwargs["sortField"] = sort_field - kwargs["sortAsc"] = sort_asc + sort = screen.get("sort", {}) + if sort.get("field"): + kwargs["sortField"] = sort["field"] + kwargs["sortAsc"] = sort.get("asc", True) result = yf.screen(**kwargs) except Exception as exc: c.log(f" ERROR: screen failed: {exc}") - return f"# Screen: {screen_id}\n\nError: {exc}\n" + return f"# Screen: {screen_id}\n\nRetrieval failed: {exc}\n", False quotes = result.get("quotes", []) c.log(f" Got {len(quotes)} results (total matching: {result.get('total', '?')})") - should_enrich = do_enrich and screen.get("enrich", True) if should_enrich and quotes: quotes = _enrich(quotes, screen_size) - - return _render_markdown(screen, quotes, enriched=should_enrich) - - -def main(): - p = argparse.ArgumentParser(description="Config-driven market screens.") - p.add_argument("--screen", help="Screen ID from screens.json.") - p.add_argument("--all", action="store_true", help="Run all screens.") - p.add_argument("--list", action="store_true", help="List available screen IDs.") - p.add_argument("--size", type=int, help="Override the screen's default result size.") - p.add_argument("--no-enrich", action="store_true", help="Skip enrichment pass.") - p.add_argument( - "--screens-file", - help="Path to screens.json (default: ./screens.json relative to script).", + return ( + _render_markdown( + screen, + quotes, + enriched=should_enrich, + universe=universe, + total=result.get("total", "unknown"), + ), + True, ) - c.add_cache_arg(p) - args = p.parse_args() - # Resolve screens.json - if args.screens_file: - screens_path = Path(args.screens_file) - else: - screens_path = Path(__file__).resolve().parent.parent / "screens.json" +def main() -> None: + """Run the market-screen CLI.""" + parser = argparse.ArgumentParser(description="Config-driven market screens.") + parser.add_argument("--screen", help="Screen ID from screens.json.") + parser.add_argument("--all", action="store_true", help="Run all screens.") + parser.add_argument("--list", action="store_true", help="List available screen IDs.") + parser.add_argument("--size", type=int, help="Override the screen's result size.") + parser.add_argument("--no-enrich", action="store_true", help="Skip enrichment.") + parser.add_argument("--screens-file", help="Alternate screens.json path.") + c.add_cache_arg(parser) + args = parser.parse_args() + + if args.size is not None and args.size < 1: + parser.error("--size must be at least 1.") + + screens_path = ( + Path(args.screens_file).resolve() + if args.screens_file + else Path(__file__).resolve().parent.parent / "screens.json" + ) config = _load_screens(screens_path) universe = config.get("universe", {}) screens = config.get("screens", []) - cache = c.cache_root(args.cache_dir) if args.list: print("Available screens:\n") - for s in screens: - emoji = s.get("emoji", "📊") - print(f" {emoji} {s['id']:25s} {s.get('name', '')} — {s.get('description', '')}") + for screen in screens: + print( + f" {screen.get('emoji', '📊')} {screen['id']:25s} " + f"{screen.get('name', '')} — {screen.get('description', '')}" + ) return - if not args.screen and not args.all: - p.error("Specify --screen , --all, or --list.") + parser.error("Specify --screen , --all, or --list.") - do_enrich = not args.no_enrich - today = datetime.now().strftime("%Y-%m-%d") + selected = screens if args.all else [next((s for s in screens if s["id"] == args.screen), None)] + if selected == [None]: + available = ", ".join(screen["id"] for screen in screens) + c.log(f"ERROR: unknown screen {args.screen!r}. Available: {available}") + sys.exit(1) - if args.all: - all_md = [] - for s in screens: - md = run_screen(s, universe, args.size, do_enrich, cache) - all_md.append(md) - combined = "\n---\n\n".join(all_md) - slug = f"all-screens_{today}" - c.write_output(cache, "screens", slug, combined) - else: - screen = next((s for s in screens if s["id"] == args.screen), None) - if not screen: - available = ", ".join(s["id"] for s in screens) - c.log(f"ERROR: unknown screen '{args.screen}'. Available: {available}") - sys.exit(1) - md = run_screen(screen, universe, args.size, do_enrich, cache) - slug = f"{args.screen}_{today}" - c.write_output(cache, "screens", slug, md) + reports = [] + failures = 0 + for screen in selected: + report, ok = run_screen(screen, universe, args.size, not args.no_enrich) + reports.append(report) + failures += int(not ok) + + today = datetime.now().strftime("%Y-%m-%d") + slug = f"all-screens_{today}" if args.all else f"{args.screen}_{today}" + c.write_output( + c.cache_root(args.cache_dir), + "screens", + slug, + "\n---\n\n".join(reports), + ) + if failures: + c.log(f"ERROR: {failures} screen(s) failed; emitted report is incomplete.") + sys.exit(1) if __name__ == "__main__": diff --git a/skills/signal-sweep/scripts/search_themes.py b/skills/signal-sweep/scripts/search_themes.py index 5e9e4cc..b30ca5e 100644 --- a/skills/signal-sweep/scripts/search_themes.py +++ b/skills/signal-sweep/scripts/search_themes.py @@ -1,12 +1,7 @@ -"""Keyword / theme discovery via EDGAR EFTS full-text search. +"""Discover issuers through SEC EFTS full-text keyword matches. -Searches the actual text of SEC filings for a keyword or phrase, deduplicates -by company, filters to the $50M-$10B universe, and enriches with market data. -Goes from a keyword to a list of exposed companies — including non-obvious ones. - -Usage: - python scripts/search_themes.py --keyword "cannabis" --since 2026-01-01 - python scripts/search_themes.py --keyword "tariff" --since 2025-01-01 --until 2026-06-17 +The report distinguishes matching filing documents from keyword occurrences and +discloses when ``--limit`` truncates the server result set. """ from __future__ import annotations @@ -21,211 +16,263 @@ import _common as c -def _extract_ticker_from_company(company_str: str) -> str | None: - """Try to extract ticker from EFTS company string like 'SNDL Inc. (SNDL) (CIK ...)'.""" - m = re.search(r"\(([A-Z]{1,5})\)", company_str) - if m: - return m.group(1) - return None - - def _resolve_ticker_for_cik(cik: str) -> str | None: - """Try to resolve a ticker from a CIK via edgartools.""" + """Resolve the first issuer ticker exposed by edgartools.""" try: from edgar import Company - co = Company(int(cik.lstrip("0"))) - tickers = getattr(co, "tickers", []) + company = Company(int(cik.lstrip("0"))) + tickers = getattr(company, "tickers", []) if tickers: - return next(iter(tickers)) + return str(next(iter(tickers))).upper().replace(".", "-") except Exception: pass return None def search_and_filter( - keyword: str, since: str, until: str, limit: int, cache: Path, mcap_data: dict -) -> tuple[list[dict], int]: - """Search EFTS, deduplicate, filter to universe, enrich. - - Returns (enriched_results, total_unique_before_filter). - """ + keyword: str, since: str, until: str, limit: int, mcap_data: dict +) -> tuple[list[dict], dict]: + """Search EFTS, deduplicate by CIK, and apply the configured universe.""" import edgar - c.log(f"Searching EFTS for '{keyword}' ({since} to {until}, limit={limit})...") - - # EFTS search — both start_date AND end_date must be provided together + c.log(f"Searching EFTS for {keyword!r} ({since} to {until}, limit={limit})...") try: - results = edgar.search_filings( - keyword, start_date=since, end_date=until, limit=min(limit, 100) + search = edgar.search_filings( + keyword, + start_date=since, + end_date=until, + limit=min(limit, 100), ) except Exception as exc: - c.log(f"ERROR: EFTS search failed: {exc}") - return [], 0 + raise RuntimeError(f"EFTS search failed: {exc}") from exc - total_server = getattr(results, "total", "?") - c.log(f" Server reports {total_server} total matches") + if search is None: + raise RuntimeError("EFTS returned no search object") - # Fetch more if needed - fetched = len(list(results)) if results else 0 - if fetched < limit and fetched > 0: + server_total = int(getattr(search, "total", 0) or 0) + fetched = len(list(search)) + if fetched < min(limit, server_total): try: - remaining = limit - fetched - results.fetch_more(remaining) - c.log(f" Fetched {remaining} more results") + search = search.fetch_more(min(limit, server_total) - fetched) except Exception as exc: - c.log(f" WARNING: fetch_more failed: {exc}") - - # Deduplicate by CIK -> collect mention counts and most recent filing - companies: dict[ - str, dict - ] = {} # cik -> {company, cik, mentions, forms, latest_date, latest_form} - for r in results: - cik = str(getattr(r, "cik", "")).lstrip("0") + raise RuntimeError(f"EFTS pagination failed after {fetched} results: {exc}") from exc + rows = list(search)[:limit] + fetched = len(rows) + c.log(f" Fetched {fetched} of {server_total} matching filing documents") + + companies: dict[str, dict] = {} + for result in rows: + cik = str(getattr(result, "cik", "")).lstrip("0") if not cik: continue + company_name = str(getattr(result, "company", "Unknown")) + form = str(getattr(result, "form", "")) + filed = str(getattr(result, "filed", "")) + accession = str(getattr(result, "accession_number", "") or "") - company_name = getattr(r, "company", "Unknown") - form = getattr(r, "form", "") - filed = str(getattr(r, "filed", "")) - - if cik not in companies: - companies[cik] = { + entry = companies.setdefault( + cik, + { "cik": cik, "company_raw": company_name, - "mentions": 0, - "forms": set(), + "matching_documents": 0, "latest_date": "", "latest_form": "", - } - - entry = companies[cik] - entry["mentions"] += 1 - entry["forms"].add(form) + "latest_accession": "", + }, + ) + entry["matching_documents"] += 1 if filed > entry["latest_date"]: entry["latest_date"] = filed entry["latest_form"] = form + entry["latest_accession"] = accession total_unique = len(companies) - c.log(f" {total_unique} unique companies found") + c.log(f" {total_unique} unique CIKs in the fetched result set") - # Resolve tickers and filter by market cap enriched = [] - for i, (cik, info) in enumerate(companies.items()): - if i % 20 == 0: - c.log(f" Resolving tickers/market caps: {i}/{total_unique}...") - - # Try to extract ticker from company string first - ticker = _extract_ticker_from_company(info["company_raw"]) - if not ticker: - ticker = _resolve_ticker_for_cik(cik) - + unresolved_tickers = 0 + unresolved_market_caps = 0 + outside_universe = 0 + for index, (cik, info) in enumerate(companies.items()): + if index % 20 == 0: + c.log(f" Resolving tickers/market caps: {index}/{total_unique}...") + + ticker = c.extract_ticker(info["company_raw"]) or _resolve_ticker_for_cik(cik) if not ticker: + unresolved_tickers += 1 continue - ticker = ticker.upper() mcap = c.get_market_cap(ticker, mcap_data) + if mcap is None: + unresolved_market_caps += 1 + continue if not c.in_universe(mcap): + outside_universe += 1 continue - # Enrich with yfinance data import yfinance as yf try: - yf_info = yf.Ticker(ticker).info or {} + yahoo = yf.Ticker(ticker).info or {} except Exception: - yf_info = {} + yahoo = {} + accession = info["latest_accession"] enriched.append( { "ticker": ticker, - "company": yf_info.get("shortName") - or yf_info.get("longName") - or info["company_raw"], - "sector": yf_info.get("sector", "n/a"), - "industry": yf_info.get("industry", "n/a"), + "company": yahoo.get("shortName") or yahoo.get("longName") or info["company_raw"], + "sector": yahoo.get("sector", "n/a"), "mcap": mcap, - "price": yf_info.get("currentPrice"), - "mentions": info["mentions"], + "price": yahoo.get("currentPrice") or yahoo.get("regularMarketPrice"), + "matching_documents": info["matching_documents"], "latest_date": info["latest_date"], "latest_form": info["latest_form"], + "latest_accession": accession, + "source_url": c.sec_filing_url(cik, accession) if accession else "", } ) - # Sort by most recent filing date (recency-first) - enriched.sort(key=lambda x: x["latest_date"], reverse=True) - return enriched, total_unique + enriched.sort(key=lambda item: item["latest_date"], reverse=True) + coverage = { + "server_total": server_total, + "fetched": fetched, + "unique_ciks": total_unique, + "truncated_by_limit": server_total > limit, + "pagination_incomplete": fetched < min(server_total, limit), + "unresolved_tickers": unresolved_tickers, + "unresolved_market_caps": unresolved_market_caps, + "outside_universe": outside_universe, + } + return enriched, coverage def _render_markdown( - keyword: str, since: str, until: str, results: list[dict], total_unique: int + keyword: str, + since: str, + until: str, + results: list[dict], + coverage: dict, ) -> str: - """Render theme search results as Markdown.""" - lines = [] - lines.append(f'# Theme Search: "{keyword}" (since {since})\n') + """Render a source-linked theme-search report.""" + lines = [f'# Theme Search: "{keyword}" ({since} to {until})\n'] lines.append( - f'Found {total_unique} unique companies mentioning "{keyword}" in SEC filings.\n' - f"After universe filter ({c.universe_label()}): **{len(results)} companies**.\n" + f"EFTS returned **{coverage['fetched']} of {coverage['server_total']}** matching filing " + f"documents, representing **{coverage['unique_ciks']} unique CIKs** in the fetched set." ) - - if not results: - lines.append(f"No companies in the {c.universe_label()} universe matched this search.\n") - return "\n".join(lines) - + if coverage["truncated_by_limit"]: + lines.append( + "\n_Coverage is truncated by `--limit`; issuer counts and rankings are not complete._" + ) + if coverage["pagination_incomplete"]: + lines.append( + "\n_EFTS returned fewer documents than requested; source coverage is incomplete._" + ) lines.append( - "| # | Ticker | Company | Sector | Mkt Cap | Price | Mentions | Most Recent Filing |" + f"\nAfter the configured universe filter ({c.universe_label()}): " + f"**{len(results)} companies**.\n" ) + if coverage["unresolved_tickers"] or coverage["unresolved_market_caps"]: + lines.append( + f"_Omitted for unresolved current metadata: {coverage['unresolved_tickers']} CIK(s) " + f"without a ticker and {coverage['unresolved_market_caps']} ticker(s) without a Yahoo " + "market cap._\n" + ) lines.append( - "|---|--------|---------|--------|---------|-------|----------|--------------------|" + "A full-text match shows that the filing contains the term; inspect the linked source to " + "determine context and materiality.\n" ) - for i, r in enumerate(results, 1): - price_s = f"${r['price']:.2f}" if r.get("price") else "n/a" + + if not results: + lines.append("No companies in the configured universe appeared in the fetched matches.\n") + return "\n".join(lines) + + lines += [ + "| # | Ticker | Company | Sector | Mkt Cap | Price | Matching documents | Latest source |", + "|---|---|---|---|---:|---:|---:|---|", + ] + for index, result in enumerate(results, 1): + price = result.get("price") + price_text = f"${price:.2f}" if price is not None else "n/a" + accession = result["latest_accession"] or "source" + source = ( + f"[{result['latest_form']} {result['latest_date']} · {accession}]({result['source_url']})" + if result["source_url"] + else f"{result['latest_form']} {result['latest_date']}" + ) lines.append( - f"| {i} | {r['ticker']} | {r['company']} | {r['sector']} | " - f"{c.fmt_mcap(r['mcap'])} | {price_s} | {r['mentions']} | " - f"{r['latest_form']} {r['latest_date']} |" + f"| {index} | {c.md_cell(result['ticker'])} | {c.md_cell(result['company'])} | " + f"{c.md_cell(result['sector'])} | {c.fmt_mcap(result['mcap'])} | {price_text} | " + f"{result['matching_documents']} | {source} |" ) - lines.append("") return "\n".join(lines) -def main(): - p = argparse.ArgumentParser(description="Keyword/theme discovery via EDGAR EFTS.") - p.add_argument("--keyword", required=True, help="Search term.") - p.add_argument("--since", required=True, help="Start date YYYY-MM-DD.") - p.add_argument("--until", help="End date YYYY-MM-DD (default: today).") - p.add_argument( +def main() -> None: + """Run the theme-search CLI.""" + parser = argparse.ArgumentParser(description="Keyword/theme discovery via SEC EFTS.") + parser.add_argument("--keyword", required=True, help="Search term.") + parser.add_argument("--since", required=True, help="Start date YYYY-MM-DD.") + parser.add_argument("--until", help="End date YYYY-MM-DD (default: today).") + parser.add_argument( "--limit", type=int, default=200, - help="Max EFTS results to fetch before dedup (default: 200).", + help="Maximum matching filing documents before issuer deduplication (default: 200).", ) - c.add_identity_arg(p) - c.add_cache_arg(p) - args = p.parse_args() + c.add_identity_arg(parser) + c.add_cache_arg(parser) + args = parser.parse_args() + + if args.limit < 1: + parser.error("--limit must be at least 1.") + try: + since = datetime.strptime(args.since, "%Y-%m-%d").date() + until = ( + datetime.strptime(args.until, "%Y-%m-%d").date() + if args.until + else datetime.now().date() + ) + except ValueError as exc: + parser.error(str(exc)) + if since > until: + parser.error("--since must not be later than --until.") c.resolve_identity(args.identity) cache = c.cache_root(args.cache_dir) - until = args.until or datetime.now().strftime("%Y-%m-%d") - mcap_data = c.load_mcap_cache(cache) try: - results, total_unique = search_and_filter( - args.keyword, args.since, until, args.limit, cache, mcap_data + results, coverage = search_and_filter( + args.keyword, + since.isoformat(), + until.isoformat(), + args.limit, + mcap_data, ) + except RuntimeError as exc: + c.log(f"ERROR: {exc}") + sys.exit(1) finally: c.save_mcap_cache(cache, mcap_data) c.log(f"Final: {len(results)} companies in universe") - - md = _render_markdown(args.keyword, args.since, until, results, total_unique) - - # Slug: sanitize keyword for filename - slug_kw = re.sub(r"[^a-zA-Z0-9]+", "-", args.keyword).strip("-").lower() - slug = f"{slug_kw}_since-{args.since}" - c.write_output(cache, "themes", slug, md) + markdown = _render_markdown( + args.keyword, + since.isoformat(), + until.isoformat(), + results, + coverage, + ) + slug_keyword = re.sub(r"[^a-zA-Z0-9]+", "-", args.keyword).strip("-").lower() or "search" + slug = f"{slug_keyword}_{since.isoformat()}_to_{until.isoformat()}" + c.write_output(cache, "themes", slug, markdown) + if coverage["pagination_incomplete"]: + c.log("ERROR: emitted report has incomplete EFTS pagination coverage.") + sys.exit(1) if __name__ == "__main__":