From 3b42cae2965ca252fea26ac5907e9cb0f61e6360 Mon Sep 17 00:00:00 2001 From: Krisha Date: Fri, 14 Aug 2026 06:48:03 +0545 Subject: [PATCH] claude-headless: first draft --- .../context/RealRate_logo_horizontal.svg | 85 +++ .../context/RealRate_logo_light.svg | 1 + claude_headless/context/audience.md | 118 ++++ claude_headless/context/brand-core.md | 76 +++ claude_headless/context/brand-voice.md | 76 +++ .../context/company-report-design.md | 94 +++ .../context/competitive-landscape.md | 69 +++ claude_headless/context/design-system.md | 194 ++++++ .../context/industry-report-design.md | 106 ++++ claude_headless/context/infographic-design.md | 99 ++++ claude_headless/context/mindmap-design.md | 95 +++ claude_headless/context/product-offering.md | 64 ++ claude_headless/context/sources.md | 94 +++ .../generate_mindmap.cpython-310.pyc | Bin 0 -> 12355 bytes .../scripts/generate_company_report.py | 554 ++++++++++++++++++ .../scripts/generate_industry_report.py | 548 +++++++++++++++++ .../scripts/generate_infographic.py | 331 +++++++++++ claude_headless/scripts/generate_mindmap.py | 343 +++++++++++ .../scripts/generate_top5_reveal_gif.py | 243 ++++++++ claude_headless/skills/auto-practice-cycle.md | 97 +++ claude_headless/skills/company-report.md | 108 ++++ claude_headless/skills/industry-report.md | 91 +++ claude_headless/skills/mindmap.md | 85 +++ claude_headless/skills/top10-infographic.md | 83 +++ claude_headless/skills/top5-reveal-gif.md | 73 +++ 25 files changed, 3727 insertions(+) create mode 100644 claude_headless/context/RealRate_logo_horizontal.svg create mode 100644 claude_headless/context/RealRate_logo_light.svg create mode 100644 claude_headless/context/audience.md create mode 100644 claude_headless/context/brand-core.md create mode 100644 claude_headless/context/brand-voice.md create mode 100644 claude_headless/context/company-report-design.md create mode 100644 claude_headless/context/competitive-landscape.md create mode 100644 claude_headless/context/design-system.md create mode 100644 claude_headless/context/industry-report-design.md create mode 100644 claude_headless/context/infographic-design.md create mode 100644 claude_headless/context/mindmap-design.md create mode 100644 claude_headless/context/product-offering.md create mode 100644 claude_headless/context/sources.md create mode 100644 claude_headless/scripts/__pycache__/generate_mindmap.cpython-310.pyc create mode 100644 claude_headless/scripts/generate_company_report.py create mode 100644 claude_headless/scripts/generate_industry_report.py create mode 100644 claude_headless/scripts/generate_infographic.py create mode 100644 claude_headless/scripts/generate_mindmap.py create mode 100644 claude_headless/scripts/generate_top5_reveal_gif.py create mode 100644 claude_headless/skills/auto-practice-cycle.md create mode 100644 claude_headless/skills/company-report.md create mode 100644 claude_headless/skills/industry-report.md create mode 100644 claude_headless/skills/mindmap.md create mode 100644 claude_headless/skills/top10-infographic.md create mode 100644 claude_headless/skills/top5-reveal-gif.md diff --git a/claude_headless/context/RealRate_logo_horizontal.svg b/claude_headless/context/RealRate_logo_horizontal.svg new file mode 100644 index 0000000..af6638c --- /dev/null +++ b/claude_headless/context/RealRate_logo_horizontal.svg @@ -0,0 +1,85 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/claude_headless/context/RealRate_logo_light.svg b/claude_headless/context/RealRate_logo_light.svg new file mode 100644 index 0000000..4feaf02 --- /dev/null +++ b/claude_headless/context/RealRate_logo_light.svg @@ -0,0 +1 @@ +RealRate_logo \ No newline at end of file diff --git a/claude_headless/context/audience.md b/claude_headless/context/audience.md new file mode 100644 index 0000000..ac8278d --- /dev/null +++ b/claude_headless/context/audience.md @@ -0,0 +1,118 @@ +# RealRate — Audience & ICP + +> Note: Audience targeting is still being refined. This file holds the current best understanding — treat as a working document, not a finalised strategy. + +--- + +## Audience Overview + +### Primary Audience +CFOs, Heads of Risk / Treasury / Controlling, Board members, IR leaders + +### Secondary Audience +Institutional investors, analysts, M&A / strategy teams, financial journalists + +### Potential Audience *(not yet fully identified)* +- Marketing / PR / HR leaders (seal as brand signal) +- Actuaries and accounting leads +- Asset managers +- Procurement and supply chain directors +- *[To be expanded as PMF becomes clearer]* + +### Funnel Logic +TOFU (rankings / reach) → MOFU (insight / analysis) → BOFU (methodology / trust / conversion) + +--- + +## ICP Breakdown by Product + +> Pain points, value props, and personas below are working hypotheses — being tested via outreach campaigns. + +### ICP 1 — Top-Rated Companies *(Top-Rated Seal)* +**Pain points:** Lack of financial health transparency, lack of stakeholder/investor trust, maintaining brand reputation +**Value proposition:** Full financial evaluation (current + previous years), free PDF, free industry benchmarking +**Primary targets:** CFO, Head/VP of Investor Relations, Head/VP of Finance +**Secondary targets:** Compliance Director, Corporate Strategy/BizDev Executives, CEO (smaller cap) + +### ICP 2 — Non Top-Rated Companies *(Consulting)* +**Pain points:** Lack of financial health transparency, lack of stakeholder/investor trust +**Value proposition:** Full financial evaluation, free PDF, free industry benchmarking +**Primary targets:** CFO, Head/VP of Investor Relations, Head/VP of Finance +**Secondary targets:** *[TBD]* + +### ICP 3 — SME or Private Equity *(Consulting + Seal)* +**Pain points:** Lack of sufficient funding, lack of transparency, lack of knowledge and expertise +**Value proposition:** RealRate Top-Rated Seal, full financial rating report, tailored workshop +**Primary targets:** CEO, CFO, COO, VP +**Secondary targets:** Head of Corporate Strategy, Operations Manager, BizDev Executives + +### ICP 4 — Sales / Data Buyers *(Investor Intelligence)* +**Pain points:** Market volatility, lack of contemporary data, difficulty forecasting, time-consuming manual research +**Value proposition:** Subscription-based access to company database, AI-generated risk assessment & financial analysis reports +**Primary targets:** Marketing Director, Head of Marketing, Head of Sales, Sales Manager +**Secondary targets:** *[TBD]* + +### ICP 5 — Investors *(Investor Intelligence)* +**Pain points:** Lack of transparency, need reliable data to assess true financial health, gauge real risk and long-term stability +**Value proposition:** Database for thousands of company evaluations and benchmarking +**Primary targets:** CFO, Head/VP of Investor Relations, Individual Investors, Venture Capital +**Secondary targets:** *[TBD]* + +### ICP 6 — Supply Chain Management *(Supply Chain / Vendor Vetting)* +**Pain points:** Costly disruptions when suppliers face hidden financial trouble; traditional risk checks miss deep financial health issues +**Value proposition:** AI-powered ratings flag financially weak partners early using only audited data; assess hundreds of partners in minutes; detect risks up to 12 months earlier +**Primary targets:** CSCOs, Procurement Directors, Operations Managers +**Secondary targets:** CFOs, Risk Officers, Consultants + +--- + +## Key Differentiators + +| Differentiator | What it means | +|---|---| +| Full explainability | Every rating shows the causal drivers — not just a score | +| Fully independent | Not issuer-paid. No conflicts of interest. No pressure to issue favorable ratings. | +| Causal AI | Identifies what *drives* financial strength, not just what correlates with it | +| Free public archive | All rankings and data are publicly accessible | +| Top-Rated Seal | A certification product — the only one of its kind in this category | + +--- + +## Target Wedge Segment *(highest-priority entry point)* + +The wedge segment is not an ICP — it's the *entry point* for acquisition. Start here before broadening. + +| Wedge | Why | +|---|---| +| Public companies already performing well | Top-Rated Seal is an easy yes — validates what they already know | +| IR-sensitive firms | Financial transparency is a core business need, not a nice-to-have | +| Companies raising capital | ECR credibility supports investor conversations directly | + +--- + +## Distribution Partnerships *(channel strategy)* + +Build distribution-lite partnerships where partners use RealRate as a tool in their own client work. + +### Financial Advisors & Boutique Investment Firms +They advise clients on capital allocation, risk exposure, and portfolio decisions. +- **Integration model:** Use RealRate reports and ECR insights in client discussions as decision support +- **Value to RealRate:** Credibility via advisor endorsement; reach into their client networks +- **Value to partner:** Differentiated, explainable financial insight they can't produce themselves + +### Strategy & Restructuring Consultancies +They run financial diagnostics, turnaround analysis, and due diligence engagements. +- **Integration model:** RealRate ECR analysis becomes a standard component of financial health assessments +- **Example positioning:** "We include RealRate ECR analysis in every financial health assessment" + +### Investor Networks & Small Funds +- **Integration model:** ECR-based screening for portfolio decisions; RealRate rankings used in investment memos + +--- + +## Priority Industries *(for sales + content targeting)* + +1. **Life Insurance & Financial Services** — Holger's domain, Bayerische testimonial, deepest ECR data (US + Germany) +2. **Technology (Software & Semiconductors)** — 140 companies ranked, Clearwater / Appfolio / Shopify data available +3. **Construction & Real Estate** — high insolvency risk = urgent relevance, proven angle +4. **Consulting & Professional Services** — CFOs and actuaries are core buyers, seal is a strong trust signal diff --git a/claude_headless/context/brand-core.md b/claude_headless/context/brand-core.md new file mode 100644 index 0000000..94e6b6c --- /dev/null +++ b/claude_headless/context/brand-core.md @@ -0,0 +1,76 @@ +# RealRate — Brand Core + +## Company + +**RealRate** is a B2B fintech (Santa Clara, CA + Berlin, Germany) that analyzes financial statements using explainable causal AI to determine how — and *why* — companies perform financially. + +- **Core metric:** Economic Capital Ratio (ECR) = company valuation / total assets. Transparent, comparable, causal. +- **Stage:** Pre-revenue, testing product-market fit. +- **NOT** a credit rating agency or NRSRO. +- **Tagline:** Explainable Financial AI +- **Core promise:** *"We don't just rank companies. We reveal their financial truth."* +- **Positioning:** The transparency-first financial intelligence platform — independent standard for assessing, ranking, and certifying financial health using explainable AI. + +## CEO — Dr. Holger Bartel + +PhD in statistics (Humboldt University + Stockholm School of Economics, 1999). Former Head of Life Insurance Mathematics at Gothaer; Head of ALM/Risk at ERGO Group. Founded Prozentor 1997; founded RealRate 2016, live 2019. Origin story: as an appointed actuary, witnessed a CEO pressure a rating agency for an AAA rating — that moment created RealRate. + +## Links + +| | URL | +|---|---| +| Website | https://realrate.ai | +| Archive *(internal — data verification only, never share publicly)* | https://www.realrate-archive.com | +| LinkedIn Company | https://www.linkedin.com/company/realrate/ | +| LinkedIn Holger | https://www.linkedin.com/in/dr-holger-bartel/ | +| News | https://news.realrate.ai | +| Contact | holger.bartel@realrate.ai | + +--- + +## LinkedIn Strategy + +**Primary channel:** Company page (3–4× per week). Holger's personal page is secondary — only produce when explicitly requested. + +### Posting Cadence +| Day | Content Type | Pillar | +|---|---|---| +| Monday | Data reveal / ranking highlight | P1 | +| Wednesday | Insight / analysis / driver education | P2 or P3 | +| Friday | Risk signal / methodology / trust | P4 or P5 | +| Optional 4th | Industry news reaction | Any | + +### Content Pillars +| Pillar | Goal | Topics | +|---|---|---| +| P1 — Ranking & Recognition (30%) | Reach | Top 10 rankings, new entrants, ECR benchmarks, seal announcements, YoY movements | +| P2 — Insight & Interpretation (25%) | Credibility | Why companies moved, structural changes, chart interpretation, industry comparison | +| P3 — Drivers of Financial Stability (20%) | Education | Capital buffers, debt structure, cash-flow volatility, revenue vs. financial strength | +| P4 — Risk & Warning Signals (15%) | High-stakes relevance | ECR decline patterns, hidden balance-sheet risks, insolvency indicators | +| P5 — Methodology & Trust (10%) | Conversion | ECR methodology, RealRate independence, testimonials | + +### Caption Formula +**Context → Insight → Implication** + +### Caption Keywords (use naturally, no hashtags) +`financial health` · `explainable AI` · `ECR` · `capital stability` · `balance sheet strength` · `financial stability` + +### One-Line Content Filter +*"Would a CFO, risk officer, or institutional investor find this useful, credible, and distinctly RealRate? If not — it doesn't get posted."* + +--- + +## Standing Rules + +- **Always verify ECR data** at realrate-archive.com before finalising any content +- **Never use:** "excited to share," "game-changing," "revolutionary," "best-in-class" without data +- **Never link** to sales pages, pricing, or the archive in public posts — always link to https://realrate.ai/rankings/[industry_slug]/[year] or realrate.ai/methodology +- **Archive is internal only** — data verification, never shared publicly +- **Never tag companies** in captions — tag in first pinned comment only +- **No hashtags** in captions +- **Ranking link in caption** — always end with `Full ranking: https://realrate.ai/rankings/[industry_slug]/[year]` +- **Never use "we"** — always "RealRate" +- **Company page is primary** — only produce Holger personal page content when explicitly requested +- **No emojis on images** — max 2 in captions +- **Tagline on image only** — `"Powered by RealRate: Using Explainable Financial AI"` never in the caption +- **Pinned comments:** Day 0 (ranking carousel) and Day +1 (seal post) only — never on insight, NTR, deep dive, or other posts diff --git a/claude_headless/context/brand-voice.md b/claude_headless/context/brand-voice.md new file mode 100644 index 0000000..7e58310 --- /dev/null +++ b/claude_headless/context/brand-voice.md @@ -0,0 +1,76 @@ +# RealRate — Brand Voice + +## Tone Hierarchy +Analytical & data-first → Authoritative & institutional → Approachable & educational → Bold & challenger brand + +## Content Approach — Storytelling With Data + +Every piece of content follows this principle: **the story opens, the data proves, the visual shows.** + +| Layer | Role | Rule | +|---|---|---| +| **Story** | The hook — creates curiosity, opens a journey or reveals a tension | Always leads. Never starts with a data point alone — starts with what the data means for a company or an industry | +| **Data** | The proof — ECR numbers, driver contributions, rank movements | Always present. Never dropped for the sake of narrative. Specific and verified | +| **Visual** | The evidence — carousel or graphic makes data scannable and shareable | Caption tells the story; visual shows the numbers | + +**The formula:** Story hook → data that proves it → structural explanation → implication for the reader → tagline + +**What storytelling is NOT:** +- It is not fiction or embellishment — every story element must be grounded in verified data +- It is not emotion for its own sake — the narrative serves the insight, not the other way around +- It is not long — a strong story hook is one or two sentences, not a paragraph + +## Brand Voice + +| | RealRate IS | RealRate is NOT | +|---|---|---| +| Tone | Analytical, calm, precise, data-first | Promotional, hype-driven, excited | +| Claims | Specific, data-backed, verifiable | Vague, superlative, unverifiable | +| Stance | Independent observer | Cheerleader, critic, or advocate | +| Language | Executive-level, accessible | Jargon-heavy or oversimplified | + +## Never Use +- "excited to share" +- "game-changing" +- "revolutionary" +- "best-in-class" without data +- Any vague superlative that can't be verified + +## Content Accuracy Rules + +- **Never include any statement, claim, or statistic that has not been verified.** +- For RealRate data (ECR numbers, rankings, company names): verify against https://www.realrate-archive.com/ +- For external claims (industry stats, market data, company facts): verify against trusted sources (company filings, reputable financial media, official reports) before including. +- If a statement cannot be verified, remove it or explicitly flag it for Amneh to confirm before publishing. +- When in doubt, leave it out. +- When referencing big company names, only say what the data confirms — soften unverified comparisons. + +## Verified Big Names by Industry +| Industry | Names verified for use | +|---|---| +| US Software | Microsoft, Salesforce, Oracle (by revenue/market cap) | +| US Life Insurance | Prudential, MetLife, New York Life, Northwestern Mutual (by assets/premium) | +| German Construction | Use ECR data directly — no single dominant name to reference | + +## What Never to Do + +- Never link to sales pages, pricing, or contact forms in posts +- Never post without anchoring to real ECR data, real company names, or real archive findings +- Never use hashtags in captions +- Never use "we" in copy — always "RealRate" +- Never put emojis on images +- Never issue forecasts or speculative claims about companies +- Never promote RealRate directly in the body of any post +- Never tag companies in the caption — always in the first pinned comment +- Never include the tagline in the caption — it belongs on the image only +- Always end the caption with: `Full ranking: https://realrate.ai/rankings/[industry_slug]/[year]` + +## Holger's Personal Voice (secondary — use only when explicitly requested) + +First-person. Precise, calm, slightly contrarian. Never promotional. Academic rigor + practitioner authority. Text-only posts (no images). Max 250 words. No emojis. Short paragraphs. No hashtags ever. + +**Post Framework:** Hook (1 counter-intuitive line) → Context (1–2 lines) → Method (ECR in one sentence) → Finding (named companies, real numbers) → Implication (why it matters to reader) → Close (strong statement OR soft provocation OR focused question) + +**Hook Rule:** Use big, recognisable company names where data supports it. Let the contrast between known names and ECR result do the work. + +**Closing Rule:** A question is one option, not a requirement. Never "What do you think?" or generic engagement bait. Approved example: *"The distinction rarely comes up until a company is already in trouble. By then, it is usually too late to act on it."* diff --git a/claude_headless/context/company-report-design.md b/claude_headless/context/company-report-design.md new file mode 100644 index 0000000..5945cdc --- /dev/null +++ b/claude_headless/context/company-report-design.md @@ -0,0 +1,94 @@ +# Context — Company Report Design & Data Schema + +Backs `skills/company-report.md`. Companion to `industry-report-design.md` — +same verification discipline, same adaptive-section principle, one level +deeper: a single company within an industry instead of the industry as a +whole. + +--- + +## Data source + +Same endpoint as the industry report — one company's entry inside +`company_details` carries everything, plus one extra live SVG fetch for the +causal graph: + +``` +GET https://www.realrate-archive.com/us_//website-ranking.json +GET (a second live fetch, SVG) +``` + +### Per-company fields used (verified 2026-08-14 against `us_air` / Strata Critical Medical Inc) + +| Field | Shape | Notes | +|---|---|---| +| `graph_url` | URL to SVG | RealRate's own causal "Influence Model Explanation" graph — a Graphviz-rendered DAG of balance-sheet variables flowing into ECR. Plain SVG 1.1 (paths, text, polygons) — renders cleanly with `cairosvg`, no exotic features. Render it as-is; don't redraw or reinterpret the diagram. | +| `report_url` | URL to PDF | RealRate's own full company report — link to it in the Takeaway section, don't try to fetch/embed it | +| `report_text` | str | Same field used in the industry report's leader section — quote verbatim | +| `table_records` | `{year: {input_variables: {...}, output_variables: {...}}}` | Per-year balance sheet figures as comma-formatted strings, e.g. `"126,192"`. **No unit is specified anywhere in the payload** — don't assume thousands/millions, see "Unresolved" below | +| `company_id` | str | Key into `ecr_records[year]` for this company's own multi-year ECR history | + +### CONFIRMED DISCREPANCY — two different ECR values for the same company/year +The causal graph SVG has its own `EconomicCapitalRatio` node with a literal +percentage label (e.g. `"54.3%"`) baked into the graph's `` elements. +This **does not match** `company_details.value * 100` for the same +company, same industry, same year (verified: 123.5% vs. 54.3% for the same +company/id/year). Both numbers come directly from RealRate — this isn't a +computation error on this skill's part. + +**Do not silently pick one.** The company report: +1. States the headline ECR from `company_details.value * 100` (same + source used everywhere else in this repo's skills, so it stays + consistent across deliverables) +2. Separately states whatever the causal graph's own node says, labeled as + "per the causal graph" — extracted from the SVG's text nodes (see + `parse_graph_ecr()` in the script), not re-derived +3. Flags the mismatch explicitly in the Data Confidence & Caveats section + — never averaged, never "corrected" to make them agree + +### Unresolved: financial figure units +`table_records` values like `"126,192"` have no unit label anywhere in the +payload (no "$K" / "$M" / currency marker). `industry_box` (industry-level +aggregate) does use suffixed units ("110 B"), but that's a different field +serving a different purpose — don't assume the same scale applies here. +Chart axis is labeled "Reported value (unit not specified in payload)" +rather than guessing. + +--- + +## Causal graph SVG parsing + +The graph's node structure (Graphviz output) splits a label across +multiple `` elements inside one `...NodeName +...` block. To find the model's own stated ECR: + +1. Locate the block whose `` is exactly `EconomicCapitalRatio` (no + spaces — Graphviz strips them from titles even though the visible + label has them) +2. Within that block, the last `<text>...>` matching `-?\d+\.?\d*%` is the + value + +Same technique generalizes to reading any other node's percentage if a +future section needs it (e.g. `EconomicCapitalRatiobeforeLimitedLiability`, +which this graph shows as a distinct intermediate value, one more sign +these are genuinely different quantities, not rounding drift). + +--- + +## Sections are adaptive (same principle as industry-report-design.md) + +| Candidate | Requires | Chart | Text source | +|---|---|---|---| +| Company Overview | always (once a company resolves) | — (stat cards) | `name`, `rank`, `value`, `top_rated`, `trend` | +| Causal ECR Graph | `graph_url` fetches successfully | the real SVG, rasterized | graph's own node values + discrepancy note | +| Strengths & Weaknesses | `report_text` non-empty | — | `report_text`, quoted verbatim | +| Balance Sheet Snapshot | `table_records` has ≥ 1 year | bar chart, latest year's `output_variables` | computed, unit caveat included | +| Multi-Year ECR History | this `company_id` appears in `ecr_records` for ≥ 3 years | line chart | computed | +| Peer Comparison | `company_details` has ≥ 2 companies | bar: this company vs. industry avg vs. #1 | computed | +| Data Confidence & Caveats | always | — | ECR discrepancy, unit caveat, `trend` caveat | +| Takeaway & Full Report | always | — | synthesis + `report_url` + ranking URL | + +## Chart style, page layout, what never appears +Same as `industry-report-design.md` — white/light background only, navy +text (`#003b57`), primary `#00679B` / secondary `#3DBACD`, no hashtags, no +emojis, never "we". diff --git a/claude_headless/context/competitive-landscape.md b/claude_headless/context/competitive-landscape.md new file mode 100644 index 0000000..5e3ef73 --- /dev/null +++ b/claude_headless/context/competitive-landscape.md @@ -0,0 +1,69 @@ +# RealRate — Competitive Landscape + +## Quick Reference + +| | RealRate | RapidRatings | Moody's/S&P | causaLens | +|---|---|---|---|---| +| Causal AI | ✓ | ✗ | ✗ | ✓ | +| Fully explainable | ✓ | ✗ | ✗ | ✓ | +| Independent (not issuer-paid) | ✓ | ✓ | ✗ | N/A | +| Free public archive | ✓ | ✗ | ✗ | ✗ | +| Certification / seal product | ✓ | ✗ (new Badge Program) | ✗ | ✗ | +| Financial ratings focus | ✓ | ✓ | ✓ | ✗ | + +--- + +## RapidRatings *(Primary Direct Competitor)* +**Website:** https://www.rapidratings.com · **Last updated:** April 2026 + +**What they do:** Financial health intelligence platform. FHR® score (0–100) focused on supply chain risk, third-party risk, and credit risk. Claims 500K ratings across 150 countries, 24-month average warning window before bankruptcy. + +**Who they serve:** Enterprise procurement, finance, and risk teams. CPOs, supply chain risk managers, credit teams. Named clients: Verizon, McDonald's, Unilever, Duke Energy, Becton Dickinson, Under Armour, Coinbase — Fortune 500 skew. + +**Products:** FHR® (core score), RiskPulse (real-time monitoring via Creditsafe), ActionPath (workflow tool), Badge Program *(new — direct parallel to RealRate seal)*, Custom Reports, Workshops, Risk Calculator (free lead gen), FHR Exchange. + +**Methodology:** Correlation-based predictive analytics. Not causal. No public driver explanation — black box for rated companies. Data paywalled. + +**Pricing:** Buyer-paid subscription. No public pricing. + +**Brand:** Tagline *"We see what others don't"*. Positions as proactive risk management partner — outcomes-led, not AI/explainability-led. + +**Content angles for RealRate:** +- Explainability vs. black-box scoring (P3/P5) +- Causal drivers vs. correlation (P3/methodology) +- Public archive vs. paywalled data (P1/trust) +- Why knowing your score isn't enough — you need to know *why* (P2/P3) + +--- + +## Moody's / S&P *(Traditional Incumbent)* +**Websites:** https://www.moodys.com · https://www.spglobal.com/ratings · **Last updated:** April 2026 + +**What they do:** NRSROs. Issue credit ratings for bonds, structured products, sovereigns, and large corporates. Embedded in capital markets — ratings are legally required for institutional investment mandates. + +**Who they serve:** Capital markets: institutional investors, investment banks, large corporates with public debt. Not accessible to SMEs or private mid-market. + +**Products:** Credit ratings (letter-grade AAA–D), research & analytics, ESG scores, data feeds/APIs, risk management tools, due diligence platforms. + +**Methodology:** Analyst-driven, qualitative + quantitative hybrid. Not fully algorithmic. Core conflict: companies pay for their own ratings (issuer-paid). This was the direct motivation for founding RealRate — Holger witnessed a CEO pressure a rating agency for an AAA rating. + +**Weaknesses / RealRate opportunities:** +- Issuer-paid = structural conflict of interest +- Analyst-driven — slow, expensive, inaccessible to mid-market +- Failed to flag major collapses (Enron, Lehman, 2008) — credibility damage +- No driver explainability +- No certification product + +**Content angles for RealRate:** +- Independence vs. issuer-paid conflict (P5/trust) +- Why the 2008 crisis still matters for financial ratings (P4/contrarian) +- Holger's origin story — witnessed direct pressure on a rating agency (personal page, P5) + +--- + +## causaLens *(Methodological Competitor — Distant)* +**Website:** https://causalens.com · **Last updated:** April 2026 + +**Current state:** Pivoted significantly from causal AI analytics roots. Now builds Digital Knowledge Workers — multi-agent AI systems for enterprise knowledge work automation. Tagline: *"Human-only companies are obsolete."* No longer a financial ratings or financial intelligence product. No direct competitive overlap with RealRate's core offering. + +**Relevance:** Causal AI methodology is shared — useful as a reference when positioning RealRate's AI approach vs. correlation-based alternatives. Not a market competitor. diff --git a/claude_headless/context/design-system.md b/claude_headless/context/design-system.md new file mode 100644 index 0000000..65bc799 --- /dev/null +++ b/claude_headless/context/design-system.md @@ -0,0 +1,194 @@ +# RealRate — Design System + +Visual specs for all LinkedIn posts. Load this file when building any HTML post. + +--- + +## Canvas +- **Size:** 1200×1200px +- **Margin:** 50px all sides — all content and logo stay within this boundary +- **Font:** Manrope (Google Fonts) — weights 300, 400, 500, 600, 700, 800 +- **Export:** `node export.mjs` — Puppeteer at deviceScaleFactor 2 → PNG + +--- + +## Colors + +### Dark Backgrounds (approved) +| Hex | Name | +|---|---| +| `#003b57` | Deep Navy — primary | +| `#004a6e` | Navy Blue | +| `#005884` | Dark Blue | +| `#3389b1` | Medium Blue | +| `#34a2b3` | Blue Teal | +| `#2c8b9a` | Dark Teal | +| `#0a1628` | Dark Navy (legacy) | +| `#00679B` | Brand Blue (legacy) | + +### Light Backgrounds (approved) +| Hex | Name | Use | +|---|---|---| +| `#f5f5f5` | Light Grey | Data viz, light slides | +| `#e8e8e8` | Off-White | Light variant | +| `#ffffff` | White | Data viz only | + +### Brand Colors +| Role | Hex | +|---|---| +| Primary teal | `#3DBACD` | +| Primary blue | `#00679B` | +| Black | `#000000` | +| Grey | `#AFAFAF` | + +### Semantic / Delta Colors +| Meaning | Hex | +|---|---| +| Strong Positive | `#419453` | +| Light Positive | `#86CC82` | +| Strong Negative | `#C04A3A` | +| Light Negative | `#F08F82` | +| Neutral | `#E8E8E8` | + +**Delta rule:** Any ECR change or pp shift must use semantic color — never neutral/white. Apply via CSS: +```css +.change-negative { color: #C04A3A; } +.change-positive { color: #419453; } +``` + +### Accent Rule +One accent per design. Never mix across background types: +| Background | Accent colors | +|---|---| +| Dark | `#f5f5f5` · `#e8e8e8` | +| Light | `#3DBACD` · `#00679B` | + +--- + +## Logo +- **Dark background** → `RealRate_logo_light.svg` (white), any corner at 50px margin +- **Light background** → `RealRate_logo_horizontal.svg` (colored), any corner at 50px margin +- Choose corner with most available space and least content collision + +--- + +## Typography Scale + +| Element | Size | Weight | Notes | +|---|---|---|---| +| Logo | 44px height | — | | +| Industry tag | 20px | 700 | Uppercase, letter-spaced, white | +| Post label badge | 25px | 700 | `#3DBACD` bg · `#003b57` text · `border-radius: 6px` · `padding: 10px 22px` · `align-self: flex-start` | +| Title | 64–72px | 800 | Uppercase, `letter-spacing: -2px` | +| Subtitle | 30px | 400 | `#ffffff` · `line-height: 1.45` | +| Company rank | 30px | 700 | Uppercase · `#ffffff` | +| Body / Description | 25–30px | 400 | `#ffffff` | +| Stat labels | 35px | 700 | Uppercase · `white-space: nowrap` · `letter-spacing: 0.5px` · `#ffffff` | +| Stat values (ECR) | 50–56px | 800 | `#f5f5f5` or white | +| Stat driver name | 32px | 700 | White | +| Stat driver gain | 28px | 600 | `#e8e8e8` | +| Tagline | 20px | 400 | `rgba(255,255,255,0.24)` · `position: absolute; bottom: 50px; left: 50px` | + +--- + +## Layout Patterns + +| ID | Name | Description | Best for | +|---|---|---|---| +| L1 | Circle Photo — Bottom Left | Dark BG; large circular B&W photo bottom-left; title large right | Cover posts, announcements | +| L2 | Vertical Split | Colored left panel ~55%; B&W photo right ~45% | Thought leadership, insight posts | +| L3 | Geometric Teal | Teal area; dark angular shapes one corner | Brand storytelling | +| L4 | Circle Photo — Corner | Dark BG; circular photo clipped top-right; bold title bottom-left | Ranking posts, insight 1 | +| L5 | Ring Photo | Dark BG; large ring right with photo inside; text left | Deep dives, feature posts | + +### L1 — Circle Photo, Bottom Left +- **Background:** Deep navy (`#003b57`) +- **Photo element:** Large circle anchored bottom-left — partially bleeds off canvas +- **Logo:** White, any corner at 50px margin +- **Title:** Large bold white, centered-right — 50–64px +- **Subtitle / Body:** Stacked right side, white + +### L2 — Vertical Split +- **Left panel (~55%):** Medium blue (`#3389b1`) — all text lives here +- **Right panel (~45%):** Photo, full-height — original color or desaturated depending on design +- **Logo:** White, any corner at 50px margin +- **Title:** Large bold white, left-aligned — 50–64px +- **Subtitle / Body:** Left panel, white, stacked below title + +### L3 — Geometric Teal +- **Background:** Blue Teal (`#34a2b3`) main area +- **Geometric element:** Abstract angular/faceted dark blue shapes filling one corner quadrant (top-left) — mosaic effect against teal +- **Logo:** White, any corner at 50px margin (sits on geometric element area) +- **Title:** Large bold white on teal area — 50–64px +- **Subtitle / Body:** White on teal + +### L4 — Circle Photo, Corner +- **Background:** Deep navy (`#003b57`) +- **Photo element:** Circular masked photo clipped to top-right corner — only partially visible (quarter circle) +- **Logo:** White, any corner at 50px margin +- **Title:** Large bold white, left-aligned bottom area — 50–64px +- **Subtitle / Body:** Below title, white, left-aligned + +### L5 — Ring Photo +- **Background:** Deep navy (`#003b57`) +- **Photo element:** Large ring/donut shape right side — photo fills the ring, dark background through the center cutout; ring takes up ~45% of width +- **Logo:** White, any corner at 50px margin +- **Title:** Large bold white, left side — 50–64px +- **Subtitle / Body:** Left side, stacked below title, white + +### Layout Rules (all patterns) +- Title section: `flex: 1; display: flex; flex-direction: column; justify-content: center` +- Bottom section: `flex-shrink: 0; margin-bottom: 70px` +- Tagline: `position: absolute; bottom: 50px; left: 50px` +- Company logo box: `background: #fff; border-radius: 12px; padding: 12px 22px; display: inline-flex` — logo inside at height 44px + +--- + +## Data Visualization Posts +- **Background:** `#f5f5f5` or `#ffffff` only — never dark +- **Logo:** Colored (`RealRate_logo_horizontal.svg`), top-left +- **Text:** Dark navy `#003b57` +- **Primary series:** `#00679B` bars (US / main dataset) +- **Secondary series:** `#3DBACD` bars (Europe / secondary dataset) +- **Labels:** Right-aligned percentage values, dark navy +- **Section dividers:** Thin horizontal rules between groups where needed +- **Source line:** Bottom-left, small, grey — when citing external data +- **Style:** Horizontal bars, no gridlines, no background decoration + +--- + +## What Never Appears on Images +- Hashtags +- Tagline in captions (image only) +- Emojis +- Red or green — except `.change-negative` / `.change-positive` on delta values +- External company logo URLs — always inline SVG or letter initials + +--- + +## Design References (Competitor-Inspired) + +### Deep Dive Cover Post → RapidRatings stat-led style +- Full-bleed B&W or desaturated industry photo as background +- Large partial arc/ring in `#00679B` — 40–50% of image, partially cropped at edges +- RealRate logo top-left, no wrap +- Lead stat bottom-left in oversized white type (e.g. "78%" or "43 companies") +- 1–2 lines white copy, CTA hook line in `#3DBACD` +- No boxes, no cards, no dividers — just layered type over image + +### Top-Rated Seal Post → Moody's logo pairing style +- Background: flat `#00679B` — no gradients, no decorations +- RealRate logo left, company logo right — equal weight, centered +- No headline, no tagline, no badge — the pairing is the message +- Optional: one small line below in light opacity white, centered ("Top-Rated · US Computers 2026") + +### General Insight Posts → Moody's geometric blocks style +- Background: flat `#00679B` or `#0a1628` +- Decorative squares: solid `#3DBACD` at `opacity: 0.25–0.35` — top-right corner + bottom edge, 2–4 squares, 60–120px +- Content type label in `#3DBACD` uppercase — matches squares +- Headline: white, 60–80px, dominant +- Never use arcs AND geometric squares in the same post — pick one + +### NTR / Report Posts → RapidRatings split layout style +- Vertical two-panel: photo left with blue overlay, dark blue right with white headline +- More editorial/B2B — works well for CFO-targeted content diff --git a/claude_headless/context/industry-report-design.md b/claude_headless/context/industry-report-design.md new file mode 100644 index 0000000..458e070 --- /dev/null +++ b/claude_headless/context/industry-report-design.md @@ -0,0 +1,106 @@ +# Context — Industry Report Design & Data Schema + +Backs `skills/industry-report.md`. Unlike `infographic-design.md` (a trimmed +subset of `design-system.md`), this file documents a **data schema** +discovered by inspecting the live archive response directly — the real +skill's own docs (`skills/industry-ranking-reports.md`) reference fields +this session couldn't confirm exist (`shrinked_graph_json`, +`feature_distribution_plot` as chartable data, etc.) or a shared module +(`rr_shared.py`) that isn't in the repo. Everything below was checked +against a live response before being written down — see "Verified against" +notes. + +--- + +## Data source + +One endpoint carries everything this skill needs — no per-company SVG +scraping required: + +``` +GET https://www.realrate-archive.com/us_<slug>/<balance_sheet_year>/website-ranking.json +``` + +Same walk-back-from-current-year fetch pattern as `top10-infographic` +(see `scripts/generate_infographic.py::fetch_ranking_data`). + +### Fields, and what they actually are (verified 2026-08-14 against `us_air`) + +| Field | Shape | What it is | +|---|---|---| +| `year` | int | **Marketing year** (e.g. `2026`) — display this, not the year in the fetch URL, which is the balance sheet year. Matches `url_segment` (e.g. `"2026-us-air"`). | +| `industry_label` | str | Human-readable industry name (e.g. `"Aviation"`) | +| `num_companies` | int | Total companies tracked in this industry | +| `industry_box` | list of `[label, value]` pairs | Aggregate financials, e.g. `[["Revenues","110 B"],...]` — pre-formatted strings, not raw numbers | +| `rating_box` | list of `[label, value]` pairs | `[["Top rated","2 of 10"],["Best, worst rating","123 %, 24.3 %"]]` | +| `company_details` | list of company dicts | `name`, `rank`, `top_rated`, `trend`, `value` (raw ratio — display as `value * 100`, confirmed bug, see `infographic-design.md` § Resolved), `report_text` (RealRate's own written strength/weakness summary — quote verbatim, don't rewrite), `company_id` | +| `ecr_records` | `{year: {company_id: {rank, ecr}}}` | Multi-year history for every tracked company. `ecr` is already on the 0–100+ percent scale as a string (no ×100 needed here, unlike `company_details.value`) — coverage varies by company and by industry, don't assume every year is populated | +| `scentences` | list of str | RealRate's own generated analyst sentences about rank movement and standout comparisons — use verbatim, don't paraphrase. Can be empty. | + +### Known unverified field +`trend`'s sign convention (positive = improved) has no cross-checkable field +in this payload the way `value` did — carry it as unverified. + +--- + +## Sections are adaptive, not fixed + +Earlier draft of this skill assumed a fixed list of exactly 10 sections, +each hard-bound to a specific field. That breaks the moment an industry's +data doesn't match the shape `us_air` happened to have — e.g. `scentences` +empty, `ecr_records` covering only 2 years, or `report_text` missing for +smaller industries. + +Instead, `generate_industry_report.py` holds a **registry of candidate +sections** (see script's `SECTION_REGISTRY`). Each candidate declares: +- a title and a chart function +- a `requires(data)` check — a short predicate over the fetched JSON +- a `build(data)` function that returns the interpretation text + +The script runs every candidate's `requires()` check against *this run's* +actual data and includes only the ones that pass, in a fixed priority +order. The number of pages in the output PDF is however many candidates +qualified — not a number decided in advance. A data-rich industry might get +9–10 sections; a sparse one might get 5. This is the same principle as the +"Known open question" pattern used throughout this repo: don't assert +something the data doesn't actually support. + +### Current candidate pool (priority order; add more by extending `SECTION_REGISTRY`) + +| Candidate | Requires | Chart | Text source | +|---|---|---|---| +| Industry Overview | `industry_box` non-empty | stat cards (no chart) | `industry_box`, `industry_label`, `num_companies` | +| Top 10 Rankings | `company_details` non-empty | bar vs. average line | computed | +| ECR Distribution | ≥ 3 companies in `company_details` | histogram | computed mean/std/min/max + `rating_box` | +| Top-Rated Snapshot | `rating_box` present | donut | `rating_box` "N of M" | +| Multi-Year ECR Trend | `ecr_records` has ≥ 3 years with ≥ 3 companies each | line, industry avg by year | `ecr_records` | +| Sector Leader Profile | rank-1 company has non-empty `report_text` | single bar, leader vs. avg | leader's `report_text`, quoted | +| Notable Movers | `scentences` non-empty | rank-change bar (from `scentences` companies matched to `company_details`) | `scentences`, quoted verbatim | +| Competitive Compression | ≥ 5 companies in `company_details` | range plot, ranks 6–N | computed spread | +| Data Confidence & Caveats | always | — (confirmed/unverified table) | carried from `infographic-design.md`'s resolved bugs | +| Takeaway & Full Ranking | always | — | synthesis + CTA link | + +--- + +## Chart style (from `context/design-system.md` § Data Visualization Posts) +- Background: white or `#f5f5f5` — **never dark**, unlike the infographic +- Text: dark navy `#003b57` +- Primary series: `#00679B` +- Secondary series: `#3DBACD` +- Delta colors: positive `#419453`, negative `#C04A3A` +- No gridlines, no decorative background — data only + +## Page layout +- **Cover** (unnumbered): logo on light background (`RealRate_logo_horizontal.svg` + per design-system.md's light-background rule — different asset than the + infographic's dark-background logo), industry name, marketing year, + generation date, tagline. +- Each qualifying section is one PDF page: title, one chart (or stat cards + for chart-less sections), one interpretation paragraph. +- All pages assembled into a single PDF via + `matplotlib.backends.backend_pdf.PdfPages` — no docx, no headless browser. + +## What never appears +Same standing rules as everywhere else in this repo: no hashtags, no +emojis, never "we" (always "RealRate"), full ranking URL uses the +**marketing year**: `https://realrate.ai/rankings/us_<slug>/<year>`. diff --git a/claude_headless/context/infographic-design.md b/claude_headless/context/infographic-design.md new file mode 100644 index 0000000..73b09e8 --- /dev/null +++ b/claude_headless/context/infographic-design.md @@ -0,0 +1,99 @@ +# Context — Infographic Design Tokens + +Scoped subset of `context/design-system.md`, containing only what the Top 10 +infographic generator needs. A full skill would load the whole design system; +this practice copy keeps only what one script actually consumes, so the link +between "context fact" and "line of code" stays visible. + +Note this deliverable is a **Pillow-rendered PNG**, not an HTML→Puppeteer +export like the rest of the design system assumes — so it uses its own canvas +size (1200×1500, portrait) rather than the standard 1200×1200 square. + +--- + +## Canvas +- **Size:** 1200×1500px (portrait — taller than the standard square post to + fit a 10-row list plus header) +- **Margin:** 50px all sides + +## Colors (from `context/design-system.md`) +| Role | Hex | RGB | +|---|---|---| +| Deep Navy — background | `#003b57` | `(0, 59, 87)` | +| Primary teal — accent/wordmark | `#3DBACD` | `(61, 186, 205)` | +| White — primary text | `#ffffff` | `(255, 255, 255)` | +| Off-White — secondary text/stats | `#f5f5f5` | `(245, 245, 245)` | +| Light Grey — flat/neutral trend | `#e8e8e8` | `(232, 232, 232)` | +| Strong Positive — trend up | `#419453` | `(65, 148, 83)` | +| Strong Negative — trend down | `#C04A3A` | `(192, 74, 58)` | + +**Rotating accent pool** (blue family only, per the design system's "Dark +Backgrounds" table) — one is picked deterministically per industry slug so the +same industry always gets the same accent: +| Hex | Name | +|---|---| +| `#004a6e` | Navy Blue | +| `#005884` | Dark Blue | +| `#3389b1` | Medium Blue | +| `#34a2b3` | Blue Teal | +| `#2c8b9a` | Dark Teal | + +## Logo (from `context/design-system.md` § Logo) +- Dark background → render `context/RealRate_logo_light.svg` at **44px + height**, top-left, 50px margin — rasterized via `cairosvg`, not + approximated with drawn text. Use the file's own colors as authored (a + blue mark, white lettering, a teal mark) — they are intentional, not a + placeholder to recolor. +- This was originally implemented as hand-drawn "Real"/"Rate" text, which + silently drifted from the actual brand asset — a reminder that "see + design-system.md" in a code comment isn't enough; the code has to actually + load the file the context names. + +## Typography +Real skill files call for Manrope, but that's a *web font* — this script +renders with Pillow directly (no browser), so it falls back to whatever +system font is available (DejaVu Sans / Liberation Sans / Arial). This is a +known gap versus the brand spec, not a deliberate substitution. + +| Element | Size | Weight | +|---|---|---| +| Logo | 44px height | — (rendered from SVG, not a font) | +| "TOP 10" badge | 22px | bold | +| Title | 54px | bold | +| Subtitle | 24px | regular | +| Rank number | 30px | bold | +| Company name | 28px | bold | +| Stat value (ECR) | 30px | bold | +| Tagline | 18px | regular | + +## What never appears on the image +- Hashtags +- Emojis +- Red/green anywhere except the trend arrow (`.change-positive` / `.change-negative` equivalents) + +## Caption rules (applies to the matching .txt file, not the image) +- No hashtags +- Max 2 emoji (this skill uses 0) +- Never "we" — always "RealRate" +- Ends with `Full ranking: https://realrate.ai/rankings/us_<slug>/<year>` + +## Resolved — ECR scale and year (verified against the live archive) +Two assumptions this skill originally carried as unverified turned out to be +wrong when checked against `ecr_records` and `rating_box` in the raw +`website-ranking.json` payload: + +- **ECR scale:** `company_details[i].value` is a raw ratio, not a percent — + display it as `value * 100`. Confirmed by cross-referencing the same + company_id's entry in `ecr_records[year]`, which reports ECR on a 0–100+ + scale directly (e.g. `value: 1.2349...` == `ecr_records: "123"`), and by + `rating_box`'s `"Best, worst rating": "123%, 24.3%"`. +- **Year:** the URL path uses the *balance sheet year* (e.g. `.../2025/...`), + but the payload's own `year` field (and `url_segment`, e.g. + `"2026-us-air"`) is the *marketing year* — that's what belongs in the + title, filename, and `realrate.ai/rankings/...` URL, not the fetch-URL + year. + +`trend`'s sign convention (positive = improved) is still unverified — no +field in the payload cross-checks it the way `ecr_records` did for value. +Standing rule still applies to anything not explicitly confirmed here — see +`CLAUDE.md` → "Always verify ECR data." diff --git a/claude_headless/context/mindmap-design.md b/claude_headless/context/mindmap-design.md new file mode 100644 index 0000000..83f98fa --- /dev/null +++ b/claude_headless/context/mindmap-design.md @@ -0,0 +1,95 @@ +# Context — Mindmap Design & Data Schema + +Backs `skills/mindmap.md`. This is a **deliberate redesign**, not a port, +of the real repo's `skills/mindmap.md` — read "Why redesigned" before +changing anything here. + +--- + +## Why redesigned, not ported + +The real skill hardcodes exactly 8 named companies and pulls executive/ +office photos live from Wikipedia via `wiki_image_url()`, rendered through +Playwright + headless Chromium. Three problems with porting that as-is: + +1. **Not generic** — adding company #9 means hand-editing two `elif` + blocks in a 1873-line file with real financial figures typed in by + hand, which is exactly the "don't hardcode data" mistake this whole + practice project has been correcting. +2. **External, fragile dependency** — Wikipedia's API can rate-limit, an + article can be renamed, an image can be removed; none of that is + under this project's control and none of it is archive data RealRate + can stand behind. +3. **Heavier runtime** — Playwright + a Chromium download is a lot of + weight for a company card, when every other skill in `claude_practice` + already renders adequately with Pillow + cairosvg. + +This version works for **any company already in the live archive**, built +only from fields verified elsewhere in this project (see +`company-report-design.md`), rendered directly with Pillow — same +cross-platform font loader as `generate_infographic.py`. + +--- + +## Branch data sources (all previously verified fields) + +| Branch | Data source | Always present? | +|---|---|---| +| Industry Position | `rank`, `num_companies`/`len(company_details)`, `industry_label` | Yes | +| Financial Health | `value * 100`, industry average | Yes | +| Greatest Strength | `report_text`, regex-extracted | Only if `report_text` present and matches the template | +| Greatest Weakness | `report_text`, regex-extracted | Only if `report_text` present and matches the template | +| Multi-Year Trend | `ecr_records[year][company_id]` across years | Only if ≥ 2 years on record for this company | +| Rating Status | `top_rated` | Yes | + +Branch count is therefore **adaptive, 4–6** — same principle as the report +skills, applied to a single image instead of PDF pages. + +## `report_text` extraction — a confirmed regex bug and its fix + +`report_text` follows one consistent template across every company checked +this session: + +> "...The greatest strength of **{name}** ... is the variable **{X}**, +> increasing the Economic Capital Ratio by **{N}%** points.The greatest +> weakness of **{name}** is the variable **{Y}**, reducing the Economic +> Capital Ratio by **{M}%** points..." + +A first attempt used a bare pattern `variable (.+?), increasing...` / +`variable (.+?), reducing...`. Because `re.search` locks onto the +**first** literal "variable" in the text (the strength sentence) and the +non-greedy `.+?` only stops at the *next* matching suffix — for the +weakness pattern, that meant it swallowed the entire strength sentence in +between, producing a paragraph-long "variable name" that overflowed the +branch circle onto the canvas edge. + +**Fix:** anchor each pattern to its own sentence — `greatest strength of +.+? is the variable (.+?), increasing...` / `greatest weakness of .+? is +the variable (.+?), reducing...` — so each regex only searches within its +own sentence. See `parse_strength_weakness()` in the script. + +**Defense in depth:** `draw_centered_text()` still truncates any line +wider than its circle regardless of source, since a future template change +in the archive's `report_text` wording could reintroduce the same failure +mode silently. The hub company name is exempted from truncation and +shrinks its font instead — a company's own name is the one label that +should never become "Strata Critica…". + +--- + +## Layout +- Canvas 1920×1080, deep navy background (`#003b57`) — same brand dark + background as the Top 10 infographic +- Hub: 380px circle, center, RealRate logo (`RealRate_logo_light.svg`, + same asset/technique as `top10-infographic`) + company name + ECR/rank + badge line +- Branches: evenly spaced around the hub at fixed radius, count = however + many candidates qualified (4–6), each a 300px filled circle in a + blue-family accent color, connected to the hub by a teal line +- No red/green — this is a dark-background asset, so semantic delta + colors don't apply here (see `design-system.md`'s accent rule: dark + backgrounds use light/teal accents only) + +## What never appears +No hashtags, no emojis, never "we", no photos of real people (see "Why +redesigned" above) — RealRate's own logo and real company data only. diff --git a/claude_headless/context/product-offering.md b/claude_headless/context/product-offering.md new file mode 100644 index 0000000..fd0b970 --- /dev/null +++ b/claude_headless/context/product-offering.md @@ -0,0 +1,64 @@ +# RealRate — Product Offering + +## Tagline +**Explainable Financial AI** + +## Core Promise +*"We don't just rank companies. We reveal their financial truth."* + +## Value Proposition +RealRate goes beyond the numbers to explain the *why* behind financial performance — combining causal AI with economic theory to deliver transparent, objective financial evaluations based on the Economic Capital Ratio (ECR). Every rating is explainable, independent, and grounded in audited data. + +--- + +## Primary Products & Services + +> All products are equal priority for sales content. + +| Product | Description | Primary Buyers | +|---|---|---| +| **Top-Rated Seal** | Certification for financially healthy companies to display publicly across print, digital, and presentations | CFOs, marketing/PR, IR leaders | +| **Financial Health Standing Report** | Full evaluation of a company's financial health — current and previous years, ECR breakdown, causal drivers, industry benchmarking | CFOs, finance leads, risk officers | +| **Consulting** | Financial health evaluation + guidance for non-top-rated companies, with actionable improvement path | CFOs, finance leads | +| **Investor Intelligence** | Company database + AI-generated risk and financial analysis reports for investment decisions | Institutional investors, analysts, M&A teams | +| **Supply Chain / Vendor Vetting** | AI-powered financial health screening of suppliers and partners to flag risk early | Procurement, operations, supply chain leaders | +| **Risk Management** | Ongoing financial stability monitoring and risk signal reporting | Risk officers, actuaries, controllers | + +--- + +## Secondary Offering + +**Workshop — Actionable Insights & Plan** +A hands-on session built around a company's financial health evaluation — translating RealRate findings into a concrete action plan. Paired with Consulting or the Financial Health Standing Report. +- *Target:* Same as Consulting — CFOs, finance leads +- *Format:* [TBD — in-person / virtual / duration] +- *Availability:* [TBD] + +--- + +## Pricing Model + +- No public pricing — all engagements are contact-based +- Interested parties contact via email or phone: holger.bartel@realrate.ai / +49 160 957 90 844 +- *[Subscription tiers, per-report pricing, or annual contract structure: TBD]* + +--- + +## Social Proof + +**Testimonial:** Joachim Zech, BY die Bayerische (German insurer) +- Used for: Seal use case content, life insurance industry posts, BOFU conversion posts, outreach DMs to insurers +- Frame as: A practitioner's observation — what changed for them, not just that they were satisfied +- Goal: Build toward a second testimonial from tech or construction to broaden proof points + +--- + +## Email Campaign Target Metrics + +| Metric | Target | +|---|---| +| Bounce Rate | ~5%, max 10% | +| Open Rate | 40–60% | +| Reply Rate | ~6% (including negative) | +| Interested Rate | ~1% | +| Call Booked Rate | 0.66% | diff --git a/claude_headless/context/sources.md b/claude_headless/context/sources.md new file mode 100644 index 0000000..0e331d4 --- /dev/null +++ b/claude_headless/context/sources.md @@ -0,0 +1,94 @@ +# RealRate — Trusted Content Sources + +> Claude should browse these sources when researching content, verifying claims, or finding industry insights. Sources marked ⚠️ may be paywalled — try but results may be limited. + +--- + +## RealRate Data (Always Check First) + +| Source | URL | Use For | +|---|---|---| +| RealRate Archive *(internal)* | https://www.realrate-archive.com | ECR data, rankings, company financials — verify all RealRate claims here first | +| RealRate Website | https://realrate.ai | Product info, public-facing messaging | +| RealRate News | https://news.realrate.ai | Company announcements, publications | + +--- + +## Financial News & Markets + +| Source | URL | Use For | +|---|---|---| +| Reuters | https://www.reuters.com | Breaking financial news, company updates, market data | +| CNBC | https://www.cnbc.com | Market trends, earnings, executive commentary | +| Financial Times ⚠️ | https://www.ft.com | In-depth financial analysis, European markets | +| Wall Street Journal ⚠️ | https://www.wsj.com | US markets, corporate finance, CFO-relevant news | +| Bloomberg ⚠️ | https://www.bloomberg.com | Financial data, market intelligence | +| Barron's ⚠️ | https://www.barrons.com | Investment analysis, financial health angles | + +--- + +## Company Filings & Public Data + +| Source | URL | Use For | +|---|---|---| +| SEC EDGAR | https://www.sec.gov/edgar | US public company filings, 10-K, 10-Q, balance sheets | +| Federal Reserve | https://www.federalreserve.gov | Macroeconomic data, interest rates, financial stability reports | +| US Bureau of Economic Analysis | https://www.bea.gov | GDP, industry output, economic trends | +| World Bank Open Data | https://data.worldbank.org | Global economic indicators | + +--- + +## Fintech & Insurtech + +| Source | URL | Use For | +|---|---|---| +| Fintech Futures | https://www.fintechfutures.com | Fintech industry news, trends | +| Finovate | https://finovate.com | Fintech innovation, product launches | +| InsurTech Insights | https://insurtechinsights.com | Insurance tech trends | +| Insurance Journal | https://www.insurancejournal.com | US insurance industry news | +| Insurance Information Institute | https://www.iii.org | Industry stats, research, life insurance data | +| LIMRA | https://www.limra.com | Life insurance research and data | + +--- + +## AI & Technology + +| Source | URL | Use For | +|---|---|---| +| MIT Technology Review | https://www.technologyreview.com | AI trends, explainable AI, causal AI developments | +| Wired | https://www.wired.com | Tech trends, AI commentary | +| VentureBeat | https://venturebeat.com | AI industry news, enterprise AI | +| Harvard Business Review ⚠️ | https://hbr.org | Business strategy, leadership, CFO-relevant insights | +| McKinsey Insights | https://www.mckinsey.com/insights | Industry reports, financial services research | + +--- + +## Industry-Specific + +| Source | URL | Industry | Use For | +|---|---|---|---| +| Supply Chain Dive | https://www.supplychaindive.com | Supply Chain | Risk, disruption, vendor news | +| Construction Dive | https://www.constructiondive.com | Construction | Industry trends, financial health signals | +| Engineering News-Record | https://www.enr.com | Construction | Market data, company news | +| Fierce Healthcare | https://www.fiercehealthcare.com | Health Services | Industry news | +| Pharma Voice | https://pharmavoice.com | Pharma | Industry trends | +| CIO | https://www.cio.com | Technology | Tech leadership, software trends | + +--- + +## Competitor Monitoring + +| Source | URL | Use For | +|---|---|---| +| RapidRatings | https://www.rapidratings.com | Competitor — monitor messaging, product updates | +| Moody's | https://www.moodys.com | Competitor — monitor positioning, reports | +| S&P Global | https://www.spglobal.com | Competitor — monitor positioning, reports | +| CausalLens | https://www.causalens.com | Competitor — monitor AI methodology messaging | + +--- + +## Notes +- Always verify RealRate-specific data at the archive before using in content +- For paywalled sources, try the URL — some articles are accessible without a subscription +- Add Holger's preferred reading sources here when identified +- Review and update this list quarterly as new relevant sources emerge diff --git a/claude_headless/scripts/__pycache__/generate_mindmap.cpython-310.pyc b/claude_headless/scripts/__pycache__/generate_mindmap.cpython-310.pyc new file mode 100644 index 0000000000000000000000000000000000000000..7b0a904f55c038351674af8fad14fcee588326df GIT binary patch literal 12355 zcmbVSYj7Lab>0^i3lIdK;!Cu=mSmH#L_l(!#BpRflteueiBd?(ieSm`5_dr?31FeS z3sOWEtrM!T^HB3_rV~dgX{Sxyb~=;kOf$`oP9M`wJJa@WXST1IB-1IU{nJU)w8}&M z&Ru{aWhZ@*gR}SEz3+R^Io~;F-R|zBguhQ;`uU|jM<nSxl-c>$h0F)=^PW*8iAi~h z$xNxpd09M_yds`zUKLL*uZd?Y9}`bKuZw3qZ-{3ipAgSvK8dGV>8hskscLt=yPD3Y zWr=C|9u~W<<a_gdOwaeTcz%Ex`9YS*?_tUO5bMehFG?)+hMXU<V@snOV}kA#bliSq zQ{I$WH%qggO)bCAkytP5`-05+*#PJu+XFhphCxTzDCij53p&mo0o}(Q1vS}Yp!?YY z(1Yx8&?ndg=p@U4KFJP&9%fUZS(XDm!j6JI#ooo9X76V2Vee(nu=lZN+55ke_%%8I zC_84G>^a*kY3w+A{<fBXjC}wl)AoL*oRCm@f}KR^0d~qh*sgo9q_Y`z8l{i3Gg0ey z-Lvc*N}pgKv?tK#1X_QHokz(eyRbdVEW3!347-Fe74~5^huo8*_j!!)oP7wrUuIWO zdYCP;tC+_}>?!ngjpc90@>!I<h`z3){5d<dltS$r>?OR(u>yPfwv<1z-Is+Cbhf~X z=<6tB%*NZN?02!!vhrD(kJ(SHOE^zuUi>oU#($}jl)S{PqVE)KbD`l>nECBjzG%+b zR%OoeZS$mCty{G<^MX@jRjY2Ev1>NBe3xgF$vK-E_-2W_RnuE`DitqRjp}8q%$(%t z#N>9}%^WsY%TBRumMzYTF0(z;zJ)O>Yvwb_q8PzJnN?$Ey@6iv)-E=DXT{FBr4r4o zUUq%gi$=QYEIW0ZIhMJC-=&JRw#uEwvY$;}E<2vNVsp=NYv!uUm(jxI=Jf0}Ga6;h zbZVwwwoQu{%jjMV+3+y7Tf<~Zj$L71(yCzQY>j$yP{g7}=o%G8*ON0R=MJ0Psx5<J zecUhj_AP$`HGFQ@7X5NAxoTU>HPlC=ns3*Nws+X9xQlL~!7CZD$2HgVirm6dI?HoQ z=ni{Ux7~WhK5W+9<N|kBJ)4`%uG=++e%7)$9Hcm!F^Ah?7grs>Y+k|vxT}Xv&#KyH zk-MHZRj*h+cA$!FsQJl~RjDjk#bvY3-4(k=%UX<%(=Ir*lDmk>;Fz-YHFJ_S$;Gxh zHLHSkQF~0^!nk6qlEt!JT*efx;=D0Kvu^q2=mb5df)!&onT-w^>ev;}UM*wB8C;qZ z*fumqU5@Q%R-KOdS<~7+p{j#Tt}U9?M#Xog)@+NLI9A-aHFK%q`Ixd#$7Lr=_9|wx zAo|1JS`IVahHu(f8b-IOE;=c@tH^V+N_Y5X!>Tx?HL>|(lj_{56&-AK((=T$hz_%T zmnXKThcg6&xHTN1Rj*g*O873OO!KVRrc<>RL2I6mYm>wwUOIIe8?tD7Snm}NiD#20 z{_1Of85eVV&pW(={XT|m#Nzl`!K*YDkDJHnaur-&sNpb=n=ej98*y!BdhSLtd9mTw z8-CO&mW*_+;T4XFrX3E<9cy3tsK>%F+R5XE?J;YM$t}_-vl%svoud_m3Gq0^t<`q& zG)_H?x0_@X=asl5InQX4lu@~iwp;Hfauwt)b?Z4we~ZXpfn?OJ`*`qFhoIp5l<w{j z6ig$%^+h7{ASn1lN`H{Zvs9qB$Km}{RHLHyz#pN4+oE9WU6lR=k#9r=PgD90BEK6I z9H#WAh<q7@8z4<3n+@YdED(&_9PY@wf?&~#5GUTsVyIdU57lXqSQQcoj|<{qcy7sG z#T5>X`A#Z98{Y6Ce!|4b?PPBWEClv1fQwfk%DiUp>8Qk<Z<mBx$#q$%ohI5-h6M1- z532W^6$KA!cUcTq+IQ&og<aD-GHK7QI#?5H6fsvXOwaGmcY*Q2lU1B3T3*_b(d)L) z-AbrmYClnhD(M%ZX9~UIdZJ&jL6|zZGnXWpWqUr<MYD@G(wq^QP>0Y4?}z%?=m}9@ z^N6d_agw@dr+b*?TLsJwyF-g=_MAe2iTSk)!uSd9F4M-w=dBgg@S5rKPRTaUp%3DJ zyV*bM)~0rsVKLA*RSTz4rhTv~UbAbSrfHr4cX`eJ3(z7(l5jK@ZJHNCsTHvvT4p8@ z#vy*AalL5VP=f@)q1W76L8SD0dko05XpFE+tiG^7V}$+H&KR`7_85@b?J;n{ICh!a zeuLMF#8aI=aG69G^(21YFM$M-FRe&i52SgdH{?xeL!i7(1y7YJn@UR#<PDW+n=*eS zkXejLzo?8#R4Q5?2;_P6v7v2Br=%Ob*OiviQlp*%mHJy&_<I9I0YaFUZcg|yhOu6k zSezLfx*zwYb>O(yrMud?{4qI`n7t>Pp<K8pPlYO^&lZq`)a<`fTdujQq+23+W==E* zC$i@Bise+S1qe$LQr@#>Q#ovAj4*v}_SBX6%X8NX^XIRenGa*oU^Wlq5X2R$XopGc zCHFnjZ(+>w?J7=aehq@QI&;hMLv6`*YN75y1KTX4@e7zk7#DrmU?>_Z)JTNBCGi5@ z-8-8ryHz{4%pI?6<>=x9%-Fe;6|2GQTv0r5P!S%=K|MQl-^;y #3_T0(}_!@(p zMc>{Ch#_n8h@6yDGNpR)Z^+91eG0emlHULcH4?-{!1nFG1XA=9M-99!1u~BX@}gW5 zu>3X-4F{~PE2nS-EoD){u_1k<)Y4c?9FZPqbWnF=tr#6;5Zjbxsd>RymeeIJh%tjD z9GNBWC@o!-#R8pmvD6*qGe=wTAikut?q!wFB&4e_CN%u7t1{9z2mJVk5yUqVxNJ$5 zCLrH30%M~KP+sOQ2gX%N+0`PtJUe@-8Al~=X%+LK|7-YBznbKy*48>@HwRWFznk@n zwe;OpS^q{lP&egatQS{pV0IU0(AE)mkk0VNK^DOeH<B&f@+Emg*_4*lKpv4kr`!b2 z6~~^@LIX0!_MNI7YLp6NzRRpNzJTT#B~-FULfH{@Bl0rJ!{o6FfCowS<D}y}8hBi4 zK2fs$V!1#X8ltE`!Yg~M;*xwhjzp(Pq#<dd&4Xp%uY1qta;vMW+0Lb#YGcqW1|ao) zPtG;><W}tk5BGek-FtS)b8GhiB5Xc2jce%N8=u3aC1^f1Cy?h<2fgQxW{<ojhjHi+ zpfqo*2P0FS77dWF_2<ani#7bx)H7iZEeepRz-%9k=`|-#x(z`8nhP{fs}tZBt21eJ z0z}g_2Q#O@%$GS35TTEIDoCa$jA6FG4?4gRplPTB;#Qmm;^LbPD26aT*Gb@wi#lLg z$l-<u2FCh`>n(;1wg#=4O(sI90_cT_nOj905RsJ5T%$&7nc>{!Jdf365`v?|A%@8! z%m;_g*2Ce_3`2Phl8IMB74x9Zy}CGj>)~_mfBS*HfbKnl4g0{xhTIR9R^-9<v-c;1 zD&>cMsA%Iq(q2jF8kl@c;RG@GCqZ_xH$ao5<3|G0djrJBeP!|}Y%r$YhII|LR9+Jj zoo9Wdz~m8wr&?M779W)Y6|&h3w6<&>^i{}al~~@_0%aXinHhIu#4N8$C7C5y^0q-D z0zJion0JA7ZNylLb>EhvQn0a(KGLs|eD4YLj#P)J=p6-PW%_2hkD2b`JJL?X0>pRU zBh?hGVqSVnnZ@>k9mK}s&dJ4xx7avVv8oHqI=)1WsTV_PKGqST?c2DMi=x`-OI^0t zIO)IUz^uD%Bv`(<p=NUs*O9EgZGBS)lyCNnw}9zU&oC~tyOu%|h~YPYpLY;Mh>&zk z#xY2nkiSeC6)7lOC6FUgl9BmKsEWrGJl;}x2@n2RO3Om?JP;kcjMAI<37K(FY99C* zOWhV24K#S$H0kDe^qU0vC`r1W7N&<O;s)XLFg>PH>aO+zgzHU(X*j>wJ^6sxHGUDz z#Jp#t>HRY0Nf$ntZJ0Pl8tzt|9~UE$%u-%?LTc{+>9g)MXztl@Dpd=jgQUhpvv;S` zCs`rHVMgaKqVq8Bcnr3sAF58x=hvu6h1D2wbzHDeq{{`SwrV)QXf?Fkut;be@Oi@O zz=~)wU&34dGF9n$Xyy(Iy)lqMa5Mdrf+n8r2Y(OZ-;nvQ;x&I2WY-xG*V1Jd%K7J! zXg*DPW=Wyj&lIRt75po759u8SeuY{cZYhDXq%mFat+Ek=0@Yc3Q?B9c9#)$35>({W zZWYPEW>Avxwc{NnNhPaN#+c>Hm<6v8sS<IC)QNB+bc7j$-=g#yktUI!BN7n#1tPBy z`4|WcDq1psjR+Z0Z@U|jaD*{LJ$k($kt7_GC-0{e9lOXkP#DHmtV+Y)wRatfA@+_o zk@jqm<cl)9#gG?+fuS}#=^d9c4v<`^7yWg-6KsSddK2kT_TDxKF^L$2Og}mvIvN7Z zOAxVlv=&ku;%EYCQ*GyTs<EkVsGBsQ_-r$AY=K|*p3FUeTrf>W4r8T?1zRJ2?8TRk z-#BzUl$z>w?-0<00>e-B8Vh0a+?m;nb2BHW=V!tg3?!$H)qD+;cz7H8ixu0d71}yq zxQl4tJ(~HToQ7uMzlCJjwZosZYnfD$M3z`49VBTH$amQX=3&72!2gO!Ln0sz4Z`;Y z3hV&@Dbf;=Ww0)n6x_iWG>gi!0E?aXl!E23{XPcGqeJs(Bwe5vC>^78iR4SXF^t3A zF;`#&@;<?>1DRX%AkzR=OyG2KLD{Ay{88kafg7hS?jaOR6MQWENInVij(?3jke&@E zR%D)cTpCkw96F0N_!h|rMi!GN`k3r^-3lufhySbXgDU{K!|~Ym!$t45b9e`@BzcA4 zuMy5k(N)!2GZ$<V#v~j%CE*i-XA3qxEE=yi0d(Xx;H8{g0IZp^OYp7v<femHs$oF` z`ZmmvZ1Z`{!x7fbWcJYW7zb`9U>o7qBFEYc7UsfQG*4P}$A|d|Jm<LP0{m{**`dti z&F3EO0N%a^EY(FZqyMS-7V+H{={WG&Y$SMN;(GC~qY&VWED#t}i`;G0`Io7p&VQR; zY2*$G<6c3C*f74@P8gj{4UNtgK77N6>lXK54s~|FQ|}J?^->^4ECsP^;(uJ`zel6N z{~}=MF1SR#o+N%MejX5yxabeVC7;Y&y#r^~Wgp^pUET}J$R{BjA0^h%U?~CqY-nWv zpfo8IW>-{~6ef&c0gE=3MWv*W%@lwwMx>T5S|MP8>)-pj#$c29dZ4e%BG(V=X<#|V zFS0?f(jF+`p}X344>I~3hIMq?k7LYXn#pyo6>k}>L@OD@c|-I)62#ePV4&^TU3I%{ zg5MTz_XY_z9wZ-nOS65%kv5f2fZf@?QRx%dn^&b*!P4we?1u50Opp&Nr!{tAwkBrz z-JpvNfO)&vW22Iyv{J2ZA2oy2sPr084%;vGDlN*oca`-7JwbX|<A1^q)Z!v7@xNdP zvF68Hy<(nE1S!mMBIpL440^Gru^`Pd*s&*r9`WpZXlD*#Xa0p9MjunHKG8=u=!4rL z#&Y5`a<DNK^mgR#gENUi;2dqN<37w-Fij$lFZj`%giFD$95!F1TWeMv_}gzBhSaFI zJOiIY+qo?Kz1D+fg3*P)A8c8H-46$U$#mdV25Z~AvbJsaU<h~u;5CLnxei1Dk(T}0 zxR`$$v*dK;Blh^rc1l35mnr)>kh_%lJV-MNx)h>C0jKV;BqI#Wzkq7HQ~_}|VKf2O z#*q-PFhL*$_X20SF1Mi|96lLEXxCZjr8iNPzXSf`zk(NEr4A$HHwI`fLP75|s}ArJ zy*B&Kb^JV+T-dF!)0+AmECDk@pG}R#Ynbk!^MW`q{_7a4;|6sg{vr?OJq=+M*p%6$ zrF-(@&6o)p9l^#hW-TCM5yo)Z?kYO9Ru+Z4YWAIhT!qQ!oAYhCb5B0h>=I4kqDO~2 zs$6ZW0{&r)A9fMGfG5u7Fg@E*sI`K_-f^qnMZ05yh^_7-Ljf!V1Rfa(?d>KYAxAhH z+g?I-{&A4Elu#`?jO<}C+5?aycAi9Gd*}B_{CAMsx{tq5#RNm8K}IV)YH$)mlQw%V zh_J{t3X5Dutcf*9PJa92pKmH~s;Su{r3N8|$4wNOVZwLoa8$rK)=XS>>r>=6V9oAX zD6Ab>(mqfv<Qi>w!kTXCW^oOO1s1lz>%cly{yO@N?$9cwJ_-W&<$@Q<a$qIb@-zJr zc>gBV_)1g*XZ5F)GCFr0roB}MTq2cB51extTa9w7BG*kmGQ5a*4mix=a3g3Hi<}7} zQ&<Ug%Ho-b!eF?25x^Ck_D3{<mucAa?yF4N&N~h=sE$D$X|kp$@_t$Wi4lVpFK3ZT z#s*P7CVnZ!fI}|@d`4EnxcrpduRMV^JVY0#;bL@}a49Xg6}C%>5hHD*wM)prrH2?i z#2(NYPcRiJ4XHGXA*C;>WG|3c779#wvB>?!5BTZ^6cS)+$16L464(*~ff0P4f2UYq zC(KtYTL|&N9wwfpTkvp{k8l?7NF)&3!9KGc?xENuoS#u(w1d(*(LNCl6d}KC^JzG* zcn<DVZs(}n%OOr+`)hSO$Lu9*rI9;@#|w>ltL9Cez(!?#ztqhAA6kh9$9{SP_?l>W zxB$8H+sWO-J^%k{J!daADi(Y)nD5bC<aVYJxW(6U6dUt$^9{rU?J5+5Rmn{Q^i|C5 zB_96YHlYDBdZ@T?fzhUhx_t`(+T(PKLZh>tVT#;jaM_hC$N+e|{}!+L-x0Y@<R6HL z%fGE_!XyG(XhY}nXLud?s5OO*U-8rK8Ab|TYURG6@Nc6=H~>qpK(nCeqX6KD0Vc{e zg~ST=TSXE#oxfeToOlX0M>pMHsB(qDSqycgz&ciIh#pWW@?<Roq-|(^Yy&>5P1M?m zljVUs9h;4;S0zj?tdUele0DMHjzXM;Xbupp!eIR;3>Rv1XHHBbj9?TIr*kP_7YBK& zPLV6b%Hc-jq%1>CI2{$oMI?zjK6g>X{wR!wef$op@b6N+q~kqUGkPhAX*|5C39+XX zDf}BW^2o5&<RLk&49Wc8kSUS`?)(vyjpKI`Kd%G=C9ST@fksB^urx?uQvsk<_DTSr z@j!(Xcoa@{n680A)))3GfgE5hK%uA=#7p8G>@n?*2GAPs^!8r#aI=p*1=!0inugFP zaHAo_So7ht`gfEegHHvpa3Rs@7KKSdQ1HOu(*^zw6l9XnIur<2ins7vgNv5Xm1xfY zgIdIW7k9o;78;JiVl_LIS0CQ0e(I!91b@&63&QOEN3{0RApOD#la}}4{%Jqv_o-I{ zXDx2uu6srpXWIu;$BB~d8_4T)lVsY>mgbXv3p>}3l@woxkslMG3j{*o>C~7qCLx{$ z%2ZWV)Ci6<t(Al&+~p@2d^60zPN$HD_f(|2k?v}jr$u>Grw3+5ub&JM2g6xC*Xr}T zmQr2t#O;-WKBiD^5QfBFob9mG>c{CM{jQDfpnns-50nhxT&X0zT{0N-ZuA5L!C>16 z(U*GB%D^d$vPFb^{JuA2minALC()NwcT{-#X8eBWhF*VwDT4^uU@q^zt|F?pr!@pR z&>CzF2l~<=`sf;wf<Zx%9v10gX;BM?{XOtO_s|{khXS2OV+qvVvk3FR?_1Y*l=q_D zP|)q~q0*b55<SNQs%;30G-@DCJ^Q-f$NJhm5BnqZtrvE8B#3Q{1~7esk)<&@jZSMe zI4ohcBb!*4m<d)dW&-nSRKjf1(YXFzNRT06u#B?dV3du()EGtCSmXh1jbUEn!5EB> zy%=vVa*trNy=}fR*ti%6@BMx35wY$^mrVcBrN=h*2QUC}+WYP(;HcQffdEF;CenhN z%-K+dtclRsAK*GVUxSgo^(CaXhCnU@UXvu=8U#7LwI4M4^kE(;h%CV}u~mopWq8~o z2EtV*uHkX?6glz8OSyHBTF62kZBa;3IBz2l+3!>KTObVz?mQ0n$sd0cWC!5p-@}Wo zNvijK5cseRp-T$xa;RRKJ(D>ofJ?;rLNuy1do?spPS3tDJzu!GojMz8v(qnJ<0S4w zjlKyH>VdC_JiG(tP^mk>wNP3r92;C<!XJybmuIHWhicQQhutUUre{x{Eu6e~{^Hzx z7`G7@s(S$PSI?fiJR?|MNX)QzZmw|t;+czu`4`R<E=^xP8^-EDckRu26W#J35FyzV zCN7>nU5I-4d&(PC3!fqZA_^54#*w6$FTk0#VrRy8AW}HUCH~jcMh2i>un<=JPs-`# z#zFy-E_y=RQJ!}YD^lGNb_JGZ4#HOPQe?@5Iua!L@DxMh#e&7?Tbg>?9`g7ijSxeG ztYU}C0>tS8iBqy%`47b_VX#7kLZL)RIVt)O(h{}bCREde`)%I&mgk?*iFXzx36<Wj zB;^qWs8&uX<Zc1NRZ|eL{rFGIlk$FbpRCLJ{SjqE-ltM3r5=>Wl@yd$8aQ}J>6Q6= zuuVmDxcx`b>1`!OJ4Wi9e5`lm7DO?5StG|cSch^XPT>Y`=M03JBoeDaS&0u2V5=g+ zy#$X6oR)DYKzYd!;yeM7CNCwC2Jl6kxeEe3=BGBgDSQQWm_}SCy-9xC9@Z^{d+$=; zM!(;?F|dhH7GlWo9^6&@!A<yFeDMwf00Azvj)B4-0q3aU@4+})V5|#)!}`e+a~B>Y z5<hIfhsdJ$>rrTJ@GewLTM+D-Z9dhBxIViLtW66J$@UkMZJ*XQjt19d!G~RNUJ67_ zu}q_55I3i`4QLTy#n*~-j}Du5b}?&C0M$(3Gdy$R%Djn>kS5TNR<<j3PWDyv(8<Vs z7KOqlqb5fuGStEn{h07kM<}YPN1u>2_g@6+;SRHfE3xg5Hp$+iZ`y>jA76xNgx|qQ z5!U4IBfeRnF_SO8WKA`XOug^MAx?IUkWQf<K{P%`RZ^XcnFWGs>Q_0w{xO>aC*U8Y zs0}$oh1ca-vynl4bF%Xlpm5YgA3D+(ez5X8AB|8HI+Ga<4SX#l+>Bv*nlCm8cV8l} zBM(y+V+FhwWx}NyCL@O=3ccv_i)UyRB1TI~iQ+x9k;0S}u2KFQl)XaaH;IsViEwou zsZd8qy12^l3Cf4goj+fgoxU(5cm=`=NY#&|DjA0ObVwJY^H3Ax-=K~@Nkj}t<A?aB zDhk4j&s5@sd?2~FkWml8c&!RcZVc1+_Mv48oJtV`Lrw}FPlA8PVXvkX0#p1f>V-xD zo?E9@C}iZ*nWw{ep}^c?p}?P`Js{rZ|4L+lvSO8ArzxeQ$UVMM5P|zBh}5ErBxImL z&Yc&4TBrqKp70aYLl2Q&B7H<eY+x^?_JM@PrHk|DE}wg0CNySe&P-E^kJ4LVvCz(P zlj_CcOc3@DCJ%&aW4hy1?Xnmf_c%JnFH-|LMnF`j=tt+cL4|)w<V_+JONsmvK&EI( zheBq$TZz;d!SqP|kgVI@GU7dq`e64(>Sbzd9IN8<t%{9!9FF0uAows~Bnl>OQw5F# z#Sr3`L@@muiXI!0wfl+&_rm>MKSlWooDb1YRer2>ewy+lt@Dd1Kh!!uUAd>}derL= i`g?}bLjyxS{ayW9S|92jJUrHC7()r;spOs{#`|x-(t(Ho literal 0 HcmV?d00001 diff --git a/claude_headless/scripts/generate_company_report.py b/claude_headless/scripts/generate_company_report.py new file mode 100644 index 0000000..07fa89d --- /dev/null +++ b/claude_headless/scripts/generate_company_report.py @@ -0,0 +1,554 @@ +#!/usr/bin/env python3 +""" +Practice build — RealRate Company Report Generator. + +Same three-stage pipeline and adaptive-section pattern as +generate_industry_report.py, one level deeper: a single company within an +industry. See context/company-report-design.md before changing anything — +in particular the CONFIRMED DISCREPANCY between company_details.value and +the causal graph's own stated ECR. Both are real RealRate outputs; this +script reports both rather than picking one. + +Usage: + python generate_company_report.py <industry_slug> <rank_or_name> [--year YEAR] + +Output: + output/us_<slug>/company-report/<company_slug>_<year>_report.pdf +""" + +import argparse +import datetime +import io +import os +import re +import textwrap +import urllib.request +import json + +import cairosvg +import matplotlib +matplotlib.use("Agg") +import matplotlib.pyplot as plt +from matplotlib.backends.backend_pdf import PdfPages +from PIL import Image + +SKILL_NAME = "company-report" + +NAVY = "#003b57" +PRIMARY = "#00679B" +SECONDARY = "#3DBACD" +POSITIVE = "#419453" +NEGATIVE = "#C04A3A" +BG = "#f5f5f5" + +RR_LOGO_SVG_PATH = os.path.join( + os.path.dirname(os.path.abspath(__file__)), "..", "context", "RealRate_logo_horizontal.svg" +) + +INDUSTRY_SLUGS = { + "air": "Air", "motor": "Motor", "software": "Software", "computers": "Computers", + "finance_services": "Finance Services", "food": "Food", "health_services": "Health Services", + "advertising": "Advertising", "semiconductors": "Semiconductors", "programming": "Programming", + "petrol": "Petrol", "mining": "Mining", "construction": "Construction", + "realestate": "Real Estate", "hotels": "Hotels", "consulting": "Consulting", + "data_processing": "Data Processing", "brokers": "Brokers", "savings": "Savings", + "life": "Life Insurance", "non_life": "Non-Life Insurance", "pharma": "Pharma", + "chemicals": "Chemicals", "state_banks": "State Banks", + "medicinal_products": "Medicinal Products", "recreation": "Recreation", +} + + +# --- Stage 1: FETCH ---------------------------------------------------------- + +def resolve_slug(arg: str) -> str: + if arg in INDUSTRY_SLUGS: + return arg + lowered = arg.lower().replace(" ", "_").replace("-", "_") + if lowered.startswith("us_"): + lowered = lowered[3:] + if lowered in INDUSTRY_SLUGS: + return lowered + for slug, name in INDUSTRY_SLUGS.items(): + if arg.lower() == name.lower(): + return slug + raise SystemExit(f"Unknown industry '{arg}'. Available slugs: {', '.join(sorted(INDUSTRY_SLUGS))}") + + +def fetch_ranking_data(slug: str, year: int | None): + candidates = [year] if year else [datetime.date.today().year - i for i in range(0, 4)] + tried = [] + for y in candidates: + url = f"https://www.realrate-archive.com/us_{slug}/{y}/website-ranking.json" + tried.append(url) + try: + req = urllib.request.Request(url, headers={"User-Agent": "RealRate-Report-Practice/1.0"}) + with urllib.request.urlopen(req, timeout=15) as resp: + if resp.status != 200: + continue + data = json.loads(resp.read().decode("utf-8")) + if data.get("company_details"): + return data + except Exception: + continue + raise RuntimeError(f"Could not fetch ranking data for slug '{slug}'. Tried:\n " + "\n ".join(tried)) + + +def resolve_company(data: dict, arg: str) -> dict: + companies = sorted(data["company_details"], key=lambda c: c["rank"]) + if arg.isdigit(): + rank = int(arg) + for c in companies: + if c["rank"] == rank: + return c + raise SystemExit(f"No company at rank {rank}. This industry has {len(companies)} ranked companies.") + matches = [c for c in companies if arg.lower() in c["name"].lower()] + if not matches: + available = "\n ".join(f"#{c['rank']} {c['name']}" for c in companies) + raise SystemExit(f"No company matching '{arg}'. Available:\n {available}") + return matches[0] + + +def fetch_svg(url: str) -> str | None: + try: + req = urllib.request.Request(url, headers={"User-Agent": "RealRate-Report-Practice/1.0"}) + with urllib.request.urlopen(req, timeout=15) as resp: + if resp.status != 200: + return None + return resp.read().decode("utf-8") + except Exception: + return None + + +# --- Shared helpers ----------------------------------------------------------- + +def ecr_pct(value: float) -> float: + return value * 100 + + +def parse_number(s: str) -> float | None: + """table_records figures are comma-formatted with no unit, e.g. '126,192'.""" + try: + return float(s.replace(",", "").replace("%", "")) + except (ValueError, AttributeError): + return None + + +def clean_archive_text(s: str) -> str: + s = re.sub(r"<br\s*/?>", " ", s, flags=re.IGNORECASE) + s = re.sub(r"<[^>]+>", "", s) + return re.sub(r"\s+", " ", s).strip() + + +def parse_graph_ecr(svg_text: str, node_title: str = "EconomicCapitalRatio") -> float | None: + """Graphviz splits a node's visible label across multiple <text> + elements inside one <g class="node">...<title>NAME... + block (titles have spaces stripped even though labels don't) — find + that block, then the last percentage-shaped inside it. See + context/company-report-design.md § Causal graph SVG parsing.""" + pattern = re.compile( + r"" + re.escape(node_title) + r"(.*?)", re.DOTALL + ) + match = pattern.search(svg_text) + if not match: + return None + texts = re.findall(r">([^<]+)", match.group(1)) + for t in reversed(texts): + t = t.strip() + if re.match(r"^-?\d+\.?\d*%$", t): + return float(t.rstrip("%")) + return None + + +def wrapped_text(ax, text: str, width: int = 86, fontsize: int = 11.5): + ax.axis("off") + wrapped = "\n".join(textwrap.fill(p, width) for p in text.split("\n\n")) + ax.text(0, 1, wrapped, transform=ax.transAxes, va="top", ha="left", + fontsize=fontsize, color=NAVY, linespacing=1.6) + + +def style_axes(ax): + ax.set_facecolor(BG) + for spine in ("top", "right"): + ax.spines[spine].set_visible(False) + ax.tick_params(colors=NAVY, labelsize=9) + ax.xaxis.label.set_color(NAVY) + ax.yaxis.label.set_color(NAVY) + + +def load_svg_array(svg_text_or_path: str, height_px: int = 90, is_url_content=False): + if is_url_content: + png_bytes = cairosvg.svg2png(bytestring=svg_text_or_path.encode("utf-8"), output_height=height_px) + else: + png_bytes = cairosvg.svg2png(url=svg_text_or_path, output_height=height_px) + return Image.open(io.BytesIO(png_bytes)).convert("RGBA") + + +# --- Stage 2: candidate sections ---------------------------------------------- + +def sec_overview_requires(data, company, extra): + return True + + +def sec_overview_chart(ax, data, company, extra): + ax.axis("off") + logo_svg = extra.get("company_logo_svg") + if logo_svg: + try: + logo = load_svg_array(logo_svg, height_px=110, is_url_content=True) + inset = ax.inset_axes([0.35, 0.15, 0.3, 0.75]) + inset.imshow(logo) + inset.axis("off") + except Exception: + pass + + +def sec_overview_text(data, company, extra): + avg = sum(ecr_pct(c["value"]) for c in data["company_details"]) / len(data["company_details"]) + ecr = ecr_pct(company["value"]) + status = "Top-Rated" if company.get("top_rated") else "Not Top-Rated" + trend = company.get("trend") + trend_word = "improved" if (trend and trend > 0) else "declined" if (trend and trend < 0) else "held flat" + return ( + f"{company['name']} ranks #{company['rank']} of {len(data['company_details'])} in " + f"{data.get('industry_label', 'this industry')} for {data.get('year')}, with an ECR " + f"of {ecr:.1f}% against an industry average of {avg:.1f}%. Status: {status}. " + f"Trend this cycle: {trend_word} (sign convention unverified — see Caveats)." + ) + + +def sec_causal_requires(data, company, extra): + return extra.get("causal_svg") is not None + + +def sec_causal_chart(ax, data, company, extra): + ax.axis("off") + try: + img = load_svg_array(extra["causal_svg"], height_px=500, is_url_content=True) + inset = ax.inset_axes([0.5 - 0.35, 0.0, 0.7, 1.0]) + inset.imshow(img) + inset.axis("off") + except Exception: + ax.text(0.5, 0.5, "Graph failed to render", ha="center", transform=ax.transAxes) + + +def sec_causal_text(data, company, extra): + headline = ecr_pct(company["value"]) + graph_ecr = extra.get("graph_ecr") + base = ( + f"RealRate's own causal graph for {company['name']} — each edge is that " + f"variable's percentage-point contribution to Economic Capital Ratio." + ) + if graph_ecr is not None and abs(graph_ecr - headline) > 0.5: + return base + ( + f"\n\nDiscrepancy: the graph's own ECR node reads {graph_ecr:.1f}%, while the " + f"headline figure used elsewhere in this report is {headline:.1f}% " + f"(company_details.value × 100). Both come directly from RealRate for the same " + f"company and year — this report does not attempt to reconcile them. Verify " + f"which applies before using either externally." + ) + return base + + +def sec_strengths_requires(data, company, extra): + return bool(company.get("report_text")) + + +def sec_strengths_chart(ax, data, company, extra): + ax.axis("off") + + +def sec_strengths_text(data, company, extra): + quote = clean_archive_text(company["report_text"]) + if len(quote) > 700: + quote = quote[:700].rsplit(".", 1)[0] + "." + return f'RealRate\'s own analysis of {company["name"]}:\n\n"{quote}"' + + +def sec_balance_sheet_requires(data, company, extra): + return bool(company.get("table_records")) + + +def _latest_year_vars(company): + years = sorted(company["table_records"].keys()) + latest = years[-1] + return latest, company["table_records"][latest].get("output_variables", {}) + + +def sec_balance_sheet_chart(ax, data, company, extra): + style_axes(ax) + _, out_vars = _latest_year_vars(company) + labels, values = [], [] + for label, raw in out_vars.items(): + if "%" in raw: + continue + v = parse_number(raw) + if v is not None: + labels.append(label) + values.append(v) + if not values: + ax.axis("off") + return + ax.barh(labels, values, color=PRIMARY) + ax.set_xlabel("Reported value (unit not specified in payload)") + + +def sec_balance_sheet_text(data, company, extra): + latest, out_vars = _latest_year_vars(company) + return ( + f"Balance sheet output figures for fiscal year {latest}, as reported by RealRate's " + f"archive. No currency unit is specified in the payload (not thousands, millions, " + f"or dollars explicitly) — treat magnitudes as relative to each other within this " + f"chart, not as absolute dollar figures, until confirmed against the source report: " + f"{company.get('report_url', 'not available')}." + ) + + +def sec_ecr_history_requires(data, company, extra): + records = data.get("ecr_records", {}) + cid = company["company_id"] + years_present = [y for y, cos in records.items() if cid in cos] + return len(years_present) >= 3 + + +def _company_ecr_series(data, company): + records = data.get("ecr_records", {}) + cid = company["company_id"] + series = [(int(y), float(records[y][cid]["ecr"])) for y in sorted(records.keys()) if cid in records[y]] + return series + + +def sec_ecr_history_chart(ax, data, company, extra): + style_axes(ax) + series = _company_ecr_series(data, company) + years = [y for y, _ in series] + vals = [v for _, v in series] + ax.plot(years, vals, color=PRIMARY, marker="o", linewidth=2) + ax.set_xticks(years) + ax.set_xlabel("Year") + ax.set_ylabel("ECR %") + + +def sec_ecr_history_text(data, company, extra): + series = _company_ecr_series(data, company) + first_year, first_val = series[0] + last_year, last_val = series[-1] + direction = "risen" if last_val > first_val else "fallen" if last_val < first_val else "held steady" + return ( + f"{company['name']}'s ECR has {direction} from {first_val:.1f}% in {first_year} to " + f"{last_val:.1f}% in {last_year}, across {len(series)} years on record." + ) + + +def sec_peer_requires(data, company, extra): + return len(data["company_details"]) >= 2 + + +def sec_peer_chart(ax, data, company, extra): + style_axes(ax) + companies = sorted(data["company_details"], key=lambda c: c["rank"]) + avg = sum(ecr_pct(c["value"]) for c in companies) / len(companies) + leader = companies[0] + labels = ["Industry Average", company["name"][:20]] + values = [avg, ecr_pct(company["value"])] + colors = [SECONDARY, PRIMARY] + if leader["company_id"] != company["company_id"]: + labels.append(f"#1 {leader['name'][:20]}") + values.append(ecr_pct(leader["value"])) + colors.append("#0a4a6e") + ax.bar(labels, values, color=colors) + ax.set_ylabel("ECR %") + + +def sec_peer_text(data, company, extra): + companies = sorted(data["company_details"], key=lambda c: c["rank"]) + avg = sum(ecr_pct(c["value"]) for c in companies) / len(companies) + gap = ecr_pct(company["value"]) - avg + direction = "above" if gap >= 0 else "below" + return ( + f"{company['name']} sits {abs(gap):.1f} points {direction} the " + f"{len(companies)}-company industry average of {avg:.1f}%, at rank " + f"#{company['rank']}." + ) + + +def sec_caveats_requires(data, company, extra): + return True + + +def sec_caveats_chart(ax, data, company, extra): + ax.axis("off") + rows = [ + ("ECR scale (value × 100)", "Confirmed", POSITIVE), + ("Marketing year vs. balance sheet year", "Confirmed", POSITIVE), + ("trend sign convention", "Unverified", NEGATIVE), + ("Causal graph ECR vs. headline ECR", "Confirmed mismatch", NEGATIVE), + ("Balance sheet figure units", "Unspecified in payload", NEGATIVE), + ] + for i, (label, status, color) in enumerate(rows): + y = 0.82 - i * 0.18 + ax.text(0.02, y, label, fontsize=10.5, color=NAVY, transform=ax.transAxes) + ax.text(0.62, y, status, fontsize=10.5, color=color, weight="bold", transform=ax.transAxes) + + +def sec_caveats_text(data, company, extra): + return ( + "This report states only what has been checked against live archive responses " + "for this company. The causal-graph/headline ECR mismatch and the unlabeled " + "balance-sheet units are not resolved here — they're surfaced so a human can " + "check them, per this project's standing rule to verify data before publishing." + ) + + +def sec_takeaway_requires(data, company, extra): + return True + + +def sec_takeaway_chart(ax, data, company, extra): + ax.axis("off") + + +def sec_takeaway_text(data, company, extra): + industry_id = data.get("industry_id", "") + year = data.get("year") + report_url = company.get("report_url", "not available") + return ( + f"{company['name']} — RealRate rank #{company['rank']} in " + f"{data.get('industry_label', 'this industry')}, {year} marketing cycle.\n\n" + f"RealRate's full company report: {report_url}\n\n" + f"Full industry ranking: https://realrate.ai/rankings/{industry_id}/{year}" + ) + + +SECTION_REGISTRY = [ + {"title": "Company Overview", "requires": sec_overview_requires, + "chart": sec_overview_chart, "text": sec_overview_text}, + {"title": "Causal ECR Graph", "requires": sec_causal_requires, + "chart": sec_causal_chart, "text": sec_causal_text}, + {"title": "Strengths & Weaknesses", "requires": sec_strengths_requires, + "chart": sec_strengths_chart, "text": sec_strengths_text}, + {"title": "Balance Sheet Snapshot", "requires": sec_balance_sheet_requires, + "chart": sec_balance_sheet_chart, "text": sec_balance_sheet_text}, + {"title": "Multi-Year ECR History", "requires": sec_ecr_history_requires, + "chart": sec_ecr_history_chart, "text": sec_ecr_history_text}, + {"title": "Peer Comparison", "requires": sec_peer_requires, + "chart": sec_peer_chart, "text": sec_peer_text}, + {"title": "Data Confidence & Caveats", "requires": sec_caveats_requires, + "chart": sec_caveats_chart, "text": sec_caveats_text}, + {"title": "Takeaway & Full Report", "requires": sec_takeaway_requires, + "chart": sec_takeaway_chart, "text": sec_takeaway_text}, +] + + +# --- Stage 3: RENDER ----------------------------------------------------------- + +def render_cover(pdf: PdfPages, company: dict, industry_display: str, year: int, num_sections: int): + fig = plt.figure(figsize=(8.5, 11)) + fig.patch.set_facecolor("white") + logo = load_svg_array(RR_LOGO_SVG_PATH, height_px=80) + logo_ax = fig.add_axes([0.5 - 0.16, 0.74, 0.32, 0.32 * logo.height / logo.width]) + logo_ax.imshow(logo) + logo_ax.axis("off") + + fig.text(0.5, 0.62, company["name"], ha="center", fontsize=24, weight="bold", color=NAVY, + wrap=True) + fig.text(0.5, 0.56, "COMPANY REPORT", ha="center", fontsize=20, weight="bold", color=SECONDARY) + fig.text(0.5, 0.50, f"US {industry_display.upper()} · {year} Marketing Cycle · " + f"{num_sections} sections", ha="center", fontsize=12, color=NAVY) + fig.text(0.5, 0.08, "Powered by RealRate: Using Explainable Financial AI", + ha="center", fontsize=10, color="#888888") + fig.text(0.5, 0.05, f"Generated {datetime.date.today().isoformat()}", + ha="center", fontsize=9, color="#aaaaaa") + pdf.savefig(fig) + plt.close(fig) + + +def render_section(pdf: PdfPages, index: int, total: int, title: str, chart_fn, text: str, + data: dict, company: dict, extra: dict): + fig = plt.figure(figsize=(8.5, 11)) + fig.patch.set_facecolor("white") + fig.text(0.08, 0.95, f"{index:02d} / {total:02d}", fontsize=10, color=SECONDARY, weight="bold") + fig.text(0.08, 0.91, title, fontsize=20, weight="bold", color=NAVY) + + ax_chart = fig.add_axes([0.24, 0.46, 0.66, 0.40]) + chart_fn(ax_chart, data, company, extra) + + ax_text = fig.add_axes([0.08, 0.06, 0.84, 0.32]) + wrapped_text(ax_text, text) + + pdf.savefig(fig) + plt.close(fig) + + +def generate_for_company(data: dict, company: dict, slug: str, industry_display: str, year: int) -> str: + """Builds one company's PDF and returns its path. Isolated from main() + so `company == "all"` can call this once per company in a loop without + duplicating the render pipeline.""" + extra = {"graph_ecr": None, "causal_svg": None, "company_logo_svg": None} + causal_svg = fetch_svg(company["graph_url"]) if company.get("graph_url") else None + if causal_svg: + extra["causal_svg"] = causal_svg + extra["graph_ecr"] = parse_graph_ecr(causal_svg) + if company.get("logo_url"): + extra["company_logo_svg"] = fetch_svg(company["logo_url"]) + + qualifying = [s for s in SECTION_REGISTRY if s["requires"](data, company, extra)] + skipped = [s["title"] for s in SECTION_REGISTRY if s not in qualifying] + + company_slug = re.sub(r"[^a-z0-9]+", "_", company["name"].lower()).strip("_") + out_dir = os.path.join(os.getcwd(), "output", f"us_{slug}", SKILL_NAME) + os.makedirs(out_dir, exist_ok=True) + pdf_path = os.path.join(out_dir, f"{company_slug}_{year}_report.pdf") + + with PdfPages(pdf_path) as pdf: + render_cover(pdf, company, industry_display, year, len(qualifying)) + for i, section in enumerate(qualifying, start=1): + text = section["text"](data, company, extra) + render_section(pdf, i, len(qualifying), section["title"], section["chart"], text, + data, company, extra) + + print(f"Wrote {pdf_path}") + print(f" Sections included ({len(qualifying)}/{len(SECTION_REGISTRY)}): " + + ", ".join(s["title"] for s in qualifying)) + if skipped: + print(f" Sections skipped (data not available this run): {', '.join(skipped)}") + if extra["graph_ecr"] is not None and abs(extra["graph_ecr"] - ecr_pct(company["value"])) > 0.5: + print(f" NOTE: causal graph ECR ({extra['graph_ecr']:.1f}%) does not match headline " + f"ECR ({ecr_pct(company['value']):.1f}%) — see Causal ECR Graph / Caveats sections.") + return pdf_path + + +def main(): + parser = argparse.ArgumentParser(description="Practice: generate a RealRate company report PDF.") + parser.add_argument("industry", help="Industry slug or name, e.g. 'air' or 'US Air'") + parser.add_argument("company", help="Company rank (e.g. '1'), a name substring (e.g. 'strata'), " + "or 'all' to generate one report per company in the industry") + parser.add_argument("--year", type=int, default=None, help="Override the archive year to fetch") + args = parser.parse_args() + + slug = resolve_slug(args.industry) + industry_display = INDUSTRY_SLUGS[slug] + + data = fetch_ranking_data(slug, args.year) + year = data.get("year") + + if args.company.strip().lower() == "all": + companies = sorted(data["company_details"], key=lambda c: c["rank"]) + print(f"Generating company reports for all {len(companies)} companies in {industry_display}...") + failures = [] + for company in companies: + try: + generate_for_company(data, company, slug, industry_display, year) + except Exception as e: + print(f" FAILED for {company['name']} (rank {company['rank']}): {e}") + failures.append(company["name"]) + print(f"Done: {len(companies) - len(failures)}/{len(companies)} reports generated.") + if failures: + print(f"Failed: {', '.join(failures)}") + return + + company = resolve_company(data, args.company) + generate_for_company(data, company, slug, industry_display, year) + + +if __name__ == "__main__": + main() diff --git a/claude_headless/scripts/generate_industry_report.py b/claude_headless/scripts/generate_industry_report.py new file mode 100644 index 0000000..8a359f5 --- /dev/null +++ b/claude_headless/scripts/generate_industry_report.py @@ -0,0 +1,548 @@ +#!/usr/bin/env python3 +""" +Practice build — RealRate Industry Report Generator. + +Pipeline (same three-stage shape as generate_infographic.py): + 1. FETCH — one live JSON response carries everything this skill needs + 2. ANALYZE — turn raw fields into numbers/text per candidate section + 3. RENDER — one PDF page per section that qualifies (chart + interpretation) + +Sections are ADAPTIVE, not a fixed count — see context/industry-report-design.md +§ "Sections are adaptive, not fixed". Each entry in SECTION_REGISTRY declares +its own `requires(data)` check; only sections whose required data is actually +present for this run are rendered. Don't assert what the data doesn't support. + +Usage: + python generate_industry_report.py [--year YEAR] + +Output: + output/us_/industry-report/__report.pdf +""" + +import argparse +import datetime +import io +import os +import re +import textwrap +import urllib.request +import json + +import cairosvg +import matplotlib +matplotlib.use("Agg") +import matplotlib.pyplot as plt +from matplotlib.backends.backend_pdf import PdfPages +from PIL import Image + +SKILL_NAME = "industry-report" + +# context/industry-report-design.md § Chart style +NAVY = "#003b57" +PRIMARY = "#00679B" +SECONDARY = "#3DBACD" +POSITIVE = "#419453" +NEGATIVE = "#C04A3A" +BG = "#f5f5f5" + +LOGO_SVG_PATH = os.path.join( + os.path.dirname(os.path.abspath(__file__)), "..", "context", "RealRate_logo_horizontal.svg" +) + +INDUSTRY_SLUGS = { + "air": "Air", "motor": "Motor", "software": "Software", "computers": "Computers", + "finance_services": "Finance Services", "food": "Food", "health_services": "Health Services", + "advertising": "Advertising", "semiconductors": "Semiconductors", "programming": "Programming", + "petrol": "Petrol", "mining": "Mining", "construction": "Construction", + "realestate": "Real Estate", "hotels": "Hotels", "consulting": "Consulting", + "data_processing": "Data Processing", "brokers": "Brokers", "savings": "Savings", + "life": "Life Insurance", "non_life": "Non-Life Insurance", "pharma": "Pharma", + "chemicals": "Chemicals", "state_banks": "State Banks", + "medicinal_products": "Medicinal Products", "recreation": "Recreation", +} + + +# --- Stage 1: FETCH ---------------------------------------------------------- + +def resolve_slug(arg: str) -> str: + if arg in INDUSTRY_SLUGS: + return arg + lowered = arg.lower().replace(" ", "_").replace("-", "_") + if lowered.startswith("us_"): + lowered = lowered[3:] + if lowered in INDUSTRY_SLUGS: + return lowered + for slug, name in INDUSTRY_SLUGS.items(): + if arg.lower() == name.lower(): + return slug + raise SystemExit(f"Unknown industry '{arg}'. Available slugs: {', '.join(sorted(INDUSTRY_SLUGS))}") + + +def fetch_ranking_data(slug: str, year: int | None): + """Same walk-back pattern as generate_infographic.py.""" + candidates = [year] if year else [datetime.date.today().year - i for i in range(0, 4)] + tried = [] + for y in candidates: + url = f"https://www.realrate-archive.com/us_{slug}/{y}/website-ranking.json" + tried.append(url) + try: + req = urllib.request.Request(url, headers={"User-Agent": "RealRate-Report-Practice/1.0"}) + with urllib.request.urlopen(req, timeout=15) as resp: + if resp.status != 200: + continue + data = json.loads(resp.read().decode("utf-8")) + if data.get("company_details"): + return data + except Exception: + continue + raise RuntimeError(f"Could not fetch ranking data for slug '{slug}'. Tried:\n " + "\n ".join(tried)) + + +# --- Shared helpers ----------------------------------------------------------- + +def ecr_pct(value: float) -> float: + """company_details[i].value is a raw ratio — confirmed bug fix, see + infographic-design.md § Resolved. ecr_records' own `ecr` field is + already on this scale and needs no conversion.""" + return value * 100 + + +def companies_sorted(data: dict) -> list: + return sorted(data["company_details"], key=lambda c: c["rank"]) + + +def parse_money(s: str) -> float | None: + """'110 B' -> 110.0 (billions). Returns None if unparseable.""" + s = s.strip() + mult = {"T": 1000.0, "B": 1.0, "M": 1 / 1000.0, "K": 1 / 1_000_000.0} + for suffix, factor in mult.items(): + if s.endswith(suffix): + try: + return float(s[:-1].strip().replace(",", "")) * factor + except ValueError: + return None + try: + return float(s.replace(",", "")) + except ValueError: + return None + + +def clean_archive_text(s: str) -> str: + """report_text/scentences come from the archive with raw HTML markup + (e.g. literal '
') — strip tags rather than quoting them verbatim.""" + s = re.sub(r"", " ", s, flags=re.IGNORECASE) + s = re.sub(r"<[^>]+>", "", s) + return re.sub(r"\s+", " ", s).strip() + + +def wrapped_text(ax, text: str, width: int = 86, fontsize: int = 11.5): + ax.axis("off") + # Text is already wrapped by textwrap.fill below — matplotlib's own + # wrap=True would re-wrap on top of that using axes pixel width, which + # conflicts with the character-width wrap and orphans stray words. + wrapped = "\n".join(textwrap.fill(p, width) for p in text.split("\n\n")) + ax.text(0, 1, wrapped, transform=ax.transAxes, va="top", ha="left", + fontsize=fontsize, color=NAVY, linespacing=1.6) + + +def style_axes(ax): + ax.set_facecolor(BG) + for spine in ("top", "right"): + ax.spines[spine].set_visible(False) + ax.tick_params(colors=NAVY, labelsize=9) + ax.xaxis.label.set_color(NAVY) + ax.yaxis.label.set_color(NAVY) + + +# --- Stage 2: candidate sections (requires / chart / text) -------------------- + +def sec_industry_overview_requires(data): + return bool(data.get("industry_box")) + + +def sec_industry_overview_chart(ax, data): + style_axes(ax) + labels, values = [], [] + for label, raw in data["industry_box"]: + v = parse_money(raw) + if v is not None: + labels.append(f"{label}\n({raw})") + values.append(v) + if not values: + ax.axis("off") + return + ax.bar(labels, values, color=PRIMARY) + ax.set_ylabel("$ Billions") + + +def sec_industry_overview_text(data): + label = data.get("industry_label", "This industry") + n = data.get("num_companies", len(data["company_details"])) + fields = ", ".join(f"{k} of {v}" for k, v in data["industry_box"]) + return ( + f"{label} currently tracks {n} companies in RealRate's archive. " + f"Aggregate figures across the tracked companies: {fields}. " + f"These are archive-reported totals, not RealRate's own estimate — " + f"verify against realrate-archive.com before publishing." + ) + + +def sec_top10_requires(data): + return len(data["company_details"]) > 0 + + +def sec_top10_chart(ax, data): + style_axes(ax) + companies = companies_sorted(data)[:10] + names = [c["name"][:16] for c in companies] + values = [ecr_pct(c["value"]) for c in companies] + avg = sum(ecr_pct(c["value"]) for c in companies_sorted(data)) / len(data["company_details"]) + colors = [SECONDARY if c.get("top_rated") else PRIMARY for c in companies] + ax.barh(names[::-1], values[::-1], color=colors[::-1]) + ax.axvline(avg, color=NEGATIVE, linestyle="--", linewidth=1.2, label=f"Average {avg:.1f}%") + ax.set_xlabel("ECR %") + ax.legend(loc="lower right", fontsize=8, frameon=False) + + +def sec_top10_text(data): + companies = companies_sorted(data) + top10 = companies[:10] + leader = top10[0] + avg = sum(ecr_pct(c["value"]) for c in companies) / len(companies) + gap = ecr_pct(leader["value"]) - avg + top_rated = sum(1 for c in top10 if c.get("top_rated")) + return ( + f"{leader['name']} leads with an ECR of {ecr_pct(leader['value']):.1f}%, " + f"{gap:.1f} points above the {len(companies)}-company average of {avg:.1f}%. " + f"{top_rated} of the top {len(top10)} hold Top-Rated status this cycle." + ) + + +def sec_distribution_requires(data): + return len(data["company_details"]) >= 3 + + +def sec_distribution_chart(ax, data): + style_axes(ax) + values = [ecr_pct(c["value"]) for c in data["company_details"]] + ax.hist(values, bins=min(10, len(values)), color=PRIMARY, edgecolor="white") + ax.set_xlabel("ECR %") + ax.set_ylabel("Companies") + + +def sec_distribution_text(data): + values = [ecr_pct(c["value"]) for c in data["company_details"]] + n = len(values) + mean = sum(values) / n + std = (sum((v - mean) ** 2 for v in values) / n) ** 0.5 + rating = dict(data.get("rating_box", [])) + best_worst = rating.get("Best, worst rating", "") + return ( + f"Across {n} tracked companies, ECR ranges from {min(values):.1f}% to " + f"{max(values):.1f}%, averaging {mean:.1f}% with a standard deviation of " + f"{std:.1f} points. RealRate's own summary reports best/worst rating as " + f"{best_worst or 'not available in this response'}." + ) + + +def sec_top_rated_requires(data): + return bool(data.get("rating_box")) + + +def sec_top_rated_chart(ax, data): + rating = dict(data.get("rating_box", [])) + raw = rating.get("Top rated", "") + try: + n, total = [int(x.strip()) for x in raw.split("of")] + except Exception: + ax.axis("off") + return + ax.pie([n, max(total - n, 0)], labels=["Top-Rated", "Not Top-Rated"], + colors=[SECONDARY, "#cfd8dc"], autopct="%1.0f%%", + textprops={"color": NAVY, "fontsize": 10}, + wedgeprops={"width": 0.45}) + + +def sec_top_rated_text(data): + rating = dict(data.get("rating_box", [])) + raw = rating.get("Top rated", "unknown") + names = [c["name"] for c in data["company_details"] if c.get("top_rated")] + named = ", ".join(names) if names else "none listed in this response" + return f"{raw} companies hold Top-Rated status this cycle: {named}." + + +def sec_trend_requires(data): + records = data.get("ecr_records", {}) + qualifying_years = [y for y, cos in records.items() if len(cos) >= 3] + return len(qualifying_years) >= 3 + + +def _trend_series(data): + records = data.get("ecr_records", {}) + series = [] + for year in sorted(records.keys()): + cos = records[year] + if len(cos) < 3: + continue + vals = [float(v["ecr"]) for v in cos.values()] + series.append((int(year), sum(vals) / len(vals))) + return series + + +def sec_trend_chart(ax, data): + style_axes(ax) + series = _trend_series(data) + years = [y for y, _ in series] + avgs = [v for _, v in series] + ax.plot(years, avgs, color=PRIMARY, marker="o", linewidth=2) + ax.xaxis.set_major_locator(matplotlib.ticker.MaxNLocator(integer=True)) + ax.set_xlabel("Year") + ax.set_ylabel("Industry Avg ECR %") + + +def sec_trend_text(data): + series = _trend_series(data) + first_year, first_avg = series[0] + last_year, last_avg = series[-1] + direction = "risen" if last_avg > first_avg else "fallen" if last_avg < first_avg else "held steady" + return ( + f"Industry-average ECR has {direction} from {first_avg:.1f}% in {first_year} " + f"to {last_avg:.1f}% in {last_year}, across {len(series)} years with sufficient " + f"company coverage (3+ reporting companies) to be meaningful." + ) + + +def sec_leader_requires(data): + leader = companies_sorted(data)[0] + return bool(leader.get("report_text")) + + +def sec_leader_chart(ax, data): + style_axes(ax) + companies = companies_sorted(data) + leader = companies[0] + avg = sum(ecr_pct(c["value"]) for c in companies) / len(companies) + ax.bar(["Industry Average", leader["name"][:24]], + [avg, ecr_pct(leader["value"])], color=[SECONDARY, PRIMARY]) + ax.set_ylabel("ECR %") + + +def sec_leader_text(data): + leader = companies_sorted(data)[0] + quote = clean_archive_text(leader["report_text"]) + if len(quote) > 500: + quote = quote[:500].rsplit(".", 1)[0] + "." + return f'RealRate\'s own analysis of {leader["name"]}:\n\n"{quote}"' + + +def sec_movers_requires(data): + return bool(data.get("scentences")) + + +def sec_movers_chart(ax, data): + style_axes(ax) + companies = {c["name"]: c for c in data["company_details"]} + mentioned = [c for name, c in companies.items() + if any(name in s for s in data["scentences"])] + if not mentioned: + ax.axis("off") + return + names = [c["name"][:16] for c in mentioned] + values = [ecr_pct(c["value"]) for c in mentioned] + colors = [POSITIVE if c.get("trend", 0) and c["trend"] > 0 else NEGATIVE for c in mentioned] + ax.barh(names, values, color=colors) + ax.set_xlabel("ECR %") + + +def sec_movers_text(data): + return "RealRate's own notes on notable movement this cycle:\n\n" + "\n\n".join( + f"• {clean_archive_text(s)}" for s in data["scentences"] + ) + + +def sec_compression_requires(data): + return len(data["company_details"]) >= 5 + + +def sec_compression_chart(ax, data): + style_axes(ax) + companies = companies_sorted(data) + tail = companies[5:10] if len(companies) > 5 else companies[-5:] + names = [c["name"][:20] for c in tail] + values = [ecr_pct(c["value"]) for c in tail] + ax.bar(names, values, color=PRIMARY) + ax.set_ylabel("ECR %") + ax.tick_params(axis="x", rotation=30) + + +def sec_compression_text(data): + companies = companies_sorted(data) + tail = companies[5:10] if len(companies) > 5 else companies[-5:] + values = [ecr_pct(c["value"]) for c in tail] + spread = max(values) - min(values) if values else 0 + return ( + f"Ranks {tail[0]['rank']}–{tail[-1]['rank']} span only {spread:.1f} ECR points " + f"({min(values):.1f}% to {max(values):.1f}%) — a tight cluster where small " + f"balance-sheet shifts could reorder several positions next cycle." + ) + + +def sec_caveats_requires(data): + return True + + +def sec_caveats_chart(ax, data): + ax.axis("off") + rows = [ + ("ECR scale (value × 100)", "Confirmed", POSITIVE), + ("Marketing year vs. balance sheet year", "Confirmed", POSITIVE), + ("trend sign convention", "Unverified", NEGATIVE), + ("report_text / scentences wording", "Verbatim from archive", PRIMARY), + ] + for i, (label, status, color) in enumerate(rows): + y = 0.8 - i * 0.22 + ax.text(0.02, y, label, fontsize=11, color=NAVY, transform=ax.transAxes) + ax.text(0.62, y, status, fontsize=11, color=color, weight="bold", transform=ax.transAxes) + + +def sec_caveats_text(data): + return ( + "This report only states what has been checked against the live archive " + "response. Fields marked \"Unverified\" above are used as-is from the API " + "without independent confirmation — treat them as provisional. Per this " + "project's standing rule, verify all figures at realrate-archive.com before " + "using this report externally." + ) + + +def sec_takeaway_requires(data): + return True + + +def sec_takeaway_chart(ax, data): + ax.axis("off") + + +def sec_takeaway_text(data): + label = data.get("industry_label", "this industry") + year = data.get("year") + # `industry_id` is already the full slug (e.g. "us_air") — confirmed + # live. Don't reconstruct it from url_segment, which is + # "-us-" and produces a doubled "us_us-" prefix if + # concatenated with another "us_". + industry_id = data.get("industry_id", "") + n = data.get("num_companies", len(data["company_details"])) + return ( + f"RealRate tracked {n} {label} companies for the {year} marketing cycle, " + f"ranking each by Economic Capital Ratio — a measure of balance-sheet " + f"resilience independent of size or reputation.\n\n" + f"Full ranking: https://realrate.ai/rankings/{industry_id}/{year}" + ) + + +SECTION_REGISTRY = [ + {"title": "Industry Overview", "requires": sec_industry_overview_requires, + "chart": sec_industry_overview_chart, "text": sec_industry_overview_text}, + {"title": "Top 10 Rankings", "requires": sec_top10_requires, + "chart": sec_top10_chart, "text": sec_top10_text}, + {"title": "ECR Distribution", "requires": sec_distribution_requires, + "chart": sec_distribution_chart, "text": sec_distribution_text}, + {"title": "Top-Rated Snapshot", "requires": sec_top_rated_requires, + "chart": sec_top_rated_chart, "text": sec_top_rated_text}, + {"title": "Multi-Year ECR Trend", "requires": sec_trend_requires, + "chart": sec_trend_chart, "text": sec_trend_text}, + {"title": "Sector Leader Profile", "requires": sec_leader_requires, + "chart": sec_leader_chart, "text": sec_leader_text}, + {"title": "Notable Movers", "requires": sec_movers_requires, + "chart": sec_movers_chart, "text": sec_movers_text}, + {"title": "Competitive Compression", "requires": sec_compression_requires, + "chart": sec_compression_chart, "text": sec_compression_text}, + {"title": "Data Confidence & Caveats", "requires": sec_caveats_requires, + "chart": sec_caveats_chart, "text": sec_caveats_text}, + {"title": "Takeaway & Full Ranking", "requires": sec_takeaway_requires, + "chart": sec_takeaway_chart, "text": sec_takeaway_text}, +] + + +# --- Stage 3: RENDER ----------------------------------------------------------- + +def load_logo_array(svg_path: str, height_px: int = 90): + png_bytes = cairosvg.svg2png(url=svg_path, output_height=height_px) + return Image.open(io.BytesIO(png_bytes)).convert("RGBA") + + +def render_cover(pdf: PdfPages, display_name: str, year: int, num_sections: int): + fig = plt.figure(figsize=(8.5, 11)) + fig.patch.set_facecolor("white") + logo = load_logo_array(LOGO_SVG_PATH) + logo_ax = fig.add_axes([0.5 - 0.18, 0.72, 0.36, 0.36 * logo.height / logo.width]) + logo_ax.imshow(logo) + logo_ax.axis("off") + + fig.text(0.5, 0.6, f"US {display_name.upper()}", ha="center", fontsize=30, + weight="bold", color=NAVY) + fig.text(0.5, 0.55, "INDUSTRY REPORT", ha="center", fontsize=22, + weight="bold", color=SECONDARY) + fig.text(0.5, 0.49, f"{year} Marketing Cycle · {num_sections} sections", + ha="center", fontsize=13, color=NAVY) + fig.text(0.5, 0.08, "Powered by RealRate: Using Explainable Financial AI", + ha="center", fontsize=10, color="#888888") + fig.text(0.5, 0.05, f"Generated {datetime.date.today().isoformat()}", + ha="center", fontsize=9, color="#aaaaaa") + pdf.savefig(fig) + plt.close(fig) + + +def render_section(pdf: PdfPages, index: int, total: int, title: str, chart_fn, text: str, data: dict): + fig = plt.figure(figsize=(8.5, 11)) + fig.patch.set_facecolor("white") + fig.text(0.08, 0.95, f"{index:02d} / {total:02d}", fontsize=10, color=SECONDARY, weight="bold") + fig.text(0.08, 0.91, title, fontsize=20, weight="bold", color=NAVY) + + # Left margin is wide enough for horizontal-bar y-tick labels (company + # names) to not get clipped by the page edge — verified against actual + # rendered output, not assumed. + ax_chart = fig.add_axes([0.24, 0.46, 0.66, 0.40]) + chart_fn(ax_chart, data) + + ax_text = fig.add_axes([0.08, 0.06, 0.84, 0.32]) + wrapped_text(ax_text, text) + + pdf.savefig(fig) + plt.close(fig) + + +def main(): + parser = argparse.ArgumentParser(description="Practice: generate a RealRate industry report PDF.") + parser.add_argument("industry", help="Industry slug or name, e.g. 'air' or 'US Air'") + parser.add_argument("--year", type=int, default=None, help="Override the archive year to fetch") + args = parser.parse_args() + + slug = resolve_slug(args.industry) + display_name = INDUSTRY_SLUGS[slug] + + data = fetch_ranking_data(slug, args.year) + year = data.get("year") + + qualifying = [s for s in SECTION_REGISTRY if s["requires"](data)] + skipped = [s["title"] for s in SECTION_REGISTRY if s not in qualifying] + + out_dir = os.path.join(os.getcwd(), "output", f"us_{slug}", SKILL_NAME) + os.makedirs(out_dir, exist_ok=True) + pdf_path = os.path.join(out_dir, f"{slug}_{year}_report.pdf") + + with PdfPages(pdf_path) as pdf: + render_cover(pdf, display_name, year, len(qualifying)) + for i, section in enumerate(qualifying, start=1): + text = section["text"](data) + render_section(pdf, i, len(qualifying), section["title"], section["chart"], text, data) + + print(f"Wrote {pdf_path}") + print(f"Sections included ({len(qualifying)}/{len(SECTION_REGISTRY)}): " + + ", ".join(s["title"] for s in qualifying)) + if skipped: + print(f"Sections skipped (data not available this run): {', '.join(skipped)}") + + +if __name__ == "__main__": + main() diff --git a/claude_headless/scripts/generate_infographic.py b/claude_headless/scripts/generate_infographic.py new file mode 100644 index 0000000..4cfeb18 --- /dev/null +++ b/claude_headless/scripts/generate_infographic.py @@ -0,0 +1,331 @@ +#!/usr/bin/env python3 +""" +Practice build — RealRate Top 10 Ranking Infographic Generator. + +Learning copy of infographics/generate_infographic.py, rebuilt to make the +link between "design context" and "rendering code" explicit. Every constant +below traces back to a line in claude_practice/context/infographic-design.md +— that's the pattern to notice: a script should never invent a color or size, +it should cite where the number came from. + +Pipeline (three stages, same shape as every skill in this repo): + 1. FETCH — pull live data from an external source (no local business logic) + 2. RENDER — turn that data + the design context into a PNG (Pillow, no browser) + 3. CAPTION — turn the same data into the LinkedIn caption text + +Usage: + python generate_infographic.py [--year YEAR] + +Output (written to ./output/, created if missing): + output/_.png + output/linkedin__.txt + +OPEN QUESTION (see context/infographic-design.md → "Known open question"): +this treats the archive's `value` field as already being an ECR percentage, +and `trend` as a signed delta. Unverified — check before publishing. +""" + +import argparse +import datetime +import io +import os +import urllib.request +import json + +import cairosvg +from PIL import Image, ImageDraw, ImageFont + +# context/design-system.md § Logo: "Dark background -> RealRate_logo_light.svg +# (white), any corner at 50px margin". Resolve relative to this file, not cwd, +# so the script works no matter where it's invoked from. +LOGO_SVG_PATH = os.path.join( + os.path.dirname(os.path.abspath(__file__)), "..", "context", "RealRate_logo_light.svg" +) +LOGO_HEIGHT_PX = 44 # context/infographic-design.md § Typography -> Logo + +# --- Canvas — context/infographic-design.md § Canvas ----------------------- +CANVAS_W, CANVAS_H = 1200, 1500 +MARGIN = 50 + +# --- Colors — context/infographic-design.md § Colors ------------------------ +NAVY = (0, 59, 87) # #003b57 — background +TEAL = (61, 186, 205) # #3DBACD — wordmark/accent +WHITE = (255, 255, 255) +OFF_WHITE = (245, 245, 245) # #f5f5f5 — secondary text/stats +LIGHT_GREY = (232, 232, 232) # #e8e8e8 — flat trend +POSITIVE = (65, 148, 83) # #419453 — trend up +NEGATIVE = (192, 74, 58) # #C04A3A — trend down + +# Rotating blue-family accent pool — context/infographic-design.md § rotating accent pool +ACCENT_POOL = [ + (0, 74, 110), # Navy Blue + (0, 88, 132), # Dark Blue + (51, 137, 177), # Medium Blue + (52, 162, 179), # Blue Teal + (44, 139, 154), # Dark Teal +] + +TAGLINE = "Powered by RealRate: Using Explainable Financial AI" + +# Matches this script's skill file name — used to namespace output by skill, +# same convention as skills/auto-post-cycle.md's output/{industry}/{task}/ layout. +SKILL_NAME = "top10-infographic" + +INDUSTRY_SLUGS = { + "air": "Air", "motor": "Motor", "software": "Software", "computers": "Computers", + "finance_services": "Finance Services", "food": "Food", "health_services": "Health Services", + "advertising": "Advertising", "semiconductors": "Semiconductors", "programming": "Programming", + "petrol": "Petrol", "mining": "Mining", "construction": "Construction", + "realestate": "Real Estate", "hotels": "Hotels", "consulting": "Consulting", + "data_processing": "Data Processing", "brokers": "Brokers", "savings": "Savings", + "life": "Life Insurance", "non_life": "Non-Life Insurance", "pharma": "Pharma", + "chemicals": "Chemicals", "state_banks": "State Banks", + "medicinal_products": "Medicinal Products", "recreation": "Recreation", +} + + +# --- Stage 1: FETCH ---------------------------------------------------------- + +def fetch_ranking_data(slug: str, year: int | None): + """Try the given year, then walk backwards a few years, until a valid + response with company_details is found. Returns (data, year_used).""" + candidates = [year] if year else [datetime.date.today().year - i for i in range(0, 4)] + tried = [] + for y in candidates: + url = f"https://www.realrate-archive.com/us_{slug}/{y}/website-ranking.json" + tried.append(url) + try: + req = urllib.request.Request(url, headers={"User-Agent": "RealRate-Infographic-Practice/1.0"}) + with urllib.request.urlopen(req, timeout=15) as resp: + if resp.status != 200: + continue + data = json.loads(resp.read().decode("utf-8")) + if data.get("company_details"): + return data, y + except Exception: + continue + raise RuntimeError( + f"Could not fetch ranking data for slug '{slug}'. Tried:\n " + "\n ".join(tried) + ) + + +def resolve_slug(arg: str) -> str: + if arg in INDUSTRY_SLUGS: + return arg + lowered = arg.lower().replace(" ", "_").replace("-", "_") + if lowered.startswith("us_"): + lowered = lowered[3:] + if lowered in INDUSTRY_SLUGS: + return lowered + for slug, name in INDUSTRY_SLUGS.items(): + if arg.lower() == name.lower(): + return slug + raise SystemExit( + f"Unknown industry '{arg}'. Available slugs: {', '.join(sorted(INDUSTRY_SLUGS))}" + ) + + +# --- Shared formatting helpers ---------------------------------------------- + +def format_ecr(value: float) -> str: + """`value` in company_details is a raw ratio, not a percentage — confirmed + against the archive's own `ecr_records`/`rating_box` fields, which show + the same company's ECR on a 0-100+ scale (e.g. 1.2349... == 123.5%).""" + return f"{value * 100:.1f}%" + + +def trend_arrow(trend) -> tuple[str, tuple]: + if trend is None or trend == 0: + return "–", LIGHT_GREY + if trend > 0: + return "▲", POSITIVE + return "▼", NEGATIVE + + +def initials(name: str) -> str: + parts = [p for p in name.replace(",", " ").split() if p] + letters = [p[0].upper() for p in parts if p[0].isalpha()] + return "".join(letters[:2]) if letters else "?" + + +def industry_accent(slug: str): + """Deterministic by slug, so the same industry always renders with the + same accent color across runs — see design-system.md's accent rule.""" + return ACCENT_POOL[sum(slug.encode()) % len(ACCENT_POOL)] + + +def _load_font(size: int, bold: bool = False): + """Design system specifies Manrope, but that's a web font loaded via + Puppeteer for HTML posts. This script renders directly with Pillow (no + browser), so it falls back to whatever system sans-serif is installed.""" + candidates = [ + "/usr/share/fonts/truetype/dejavu/DejaVuSans-Bold.ttf" if bold else + "/usr/share/fonts/truetype/dejavu/DejaVuSans.ttf", + "/usr/share/fonts/truetype/liberation/LiberationSans-Bold.ttf" if bold else + "/usr/share/fonts/truetype/liberation/LiberationSans-Regular.ttf", + "/System/Library/Fonts/Supplemental/Arial Bold.ttf" if bold else + "/System/Library/Fonts/Supplemental/Arial.ttf", + ] + for path in candidates: + if os.path.exists(path): + return ImageFont.truetype(path, size) + return ImageFont.load_default() + + +# --- Stage 2: RENDER --------------------------------------------------------- + +def load_logo(height_px: int = LOGO_HEIGHT_PX) -> Image.Image: + """Rasterize the real brand SVG at the spec'd height, colors as-authored + in the file (do not recolor — the mark/lettering colors are intentional, + not a guess).""" + png_bytes = cairosvg.svg2png(url=LOGO_SVG_PATH, output_height=height_px) + return Image.open(io.BytesIO(png_bytes)).convert("RGBA") + + +def draw_pattern(draw: ImageDraw.ImageDraw, accent): + """Subtle diagonal-stripe accent pattern confined to the top-right quadrant.""" + step = 46 + overlay_color = accent + (60,) # low-opacity RGBA + for x in range(CANVAS_W - 420, CANVAS_W + 300, step): + draw.line([(x, 0), (x - 420, 420)], fill=overlay_color, width=10) + + +def render_image(slug: str, display_name: str, year: int, top10: list, avg_ecr: float, out_path: str): + base = Image.new("RGB", (CANVAS_W, CANVAS_H), NAVY) + overlay = Image.new("RGBA", (CANVAS_W, CANVAS_H), (0, 0, 0, 0)) + accent = industry_accent(slug) + draw_pattern(ImageDraw.Draw(overlay), accent) + base = Image.alpha_composite(base.convert("RGBA"), overlay).convert("RGB") + draw = ImageDraw.Draw(base) + + f_badge = _load_font(22, bold=True) + f_title = _load_font(54, bold=True) + f_subtitle = _load_font(24, bold=False) + f_rank = _load_font(30, bold=True) + f_name = _load_font(28, bold=True) + f_avatar = _load_font(22, bold=True) + f_stat = _load_font(30, bold=True) + f_tag = _load_font(18, bold=False) + + # Logo (top-left, 50px margin) — the real brand asset, not hand-drawn text + logo = load_logo() + base.paste(logo, (MARGIN, MARGIN), logo) + + # "TOP 10" badge (top-right) + badge_text = "TOP 10" + bw = draw.textlength(badge_text, font=f_badge) + 44 + bh = 46 + bx0, by0 = CANVAS_W - MARGIN - bw, MARGIN + draw.rounded_rectangle([bx0, by0, bx0 + bw, by0 + bh], radius=8, fill=TEAL) + draw.text((bx0 + 22, by0 + 11), badge_text, font=f_badge, fill=NAVY) + + # Title + title_text = f"US {display_name.upper()}" + title_y = MARGIN + 90 + draw.text((MARGIN, title_y), title_text, font=f_title, fill=WHITE) + draw.text((MARGIN, title_y + 66), "TOP 10 RANKING", font=f_title, fill=TEAL) + + # Subtitle + subtitle = f"{year} ECR Ranking · Industry Average {format_ecr(avg_ecr)}" + draw.text((MARGIN, title_y + 150), subtitle, font=f_subtitle, fill=OFF_WHITE) + + # Ranked rows + list_top = title_y + 210 + list_bottom = CANVAS_H - 110 + row_h = (list_bottom - list_top) / 10 + + for i, company in enumerate(top10): + ry = list_top + i * row_h + row_mid = ry + row_h / 2 + + draw.text((MARGIN, row_mid - 18), f"{i + 1:02d}", font=f_rank, fill=OFF_WHITE) + + av_cx, av_r = MARGIN + 90, 26 + draw.ellipse([av_cx - av_r, row_mid - av_r, av_cx + av_r, row_mid + av_r], fill=accent) + init = initials(company["name"]) + iw = draw.textlength(init, font=f_avatar) + draw.text((av_cx - iw / 2, row_mid - 13), init, font=f_avatar, fill=WHITE) + + name_x = av_cx + av_r + 24 + name = company["name"] + max_name_w = 480 + while draw.textlength(name, font=f_name) > max_name_w and len(name) > 3: + name = name[:-2] + "…" + draw.text((name_x, row_mid - 17), name, font=f_name, fill=WHITE) + + if company.get("top_rated"): + draw.text((name_x, row_mid + 12), "★ TOP RATED", font=f_tag, fill=TEAL) + + stat_text = format_ecr(company["value"]) + sw = draw.textlength(stat_text, font=f_stat) + arrow, arrow_color = trend_arrow(company.get("trend")) + stat_x = CANVAS_W - MARGIN - sw + draw.text((stat_x, row_mid - 18), stat_text, font=f_stat, fill=OFF_WHITE) + draw.text((stat_x - 34, row_mid - 15), arrow, font=f_name, fill=arrow_color) + + if i < 9: + draw.line([(MARGIN, ry + row_h), (CANVAS_W - MARGIN, ry + row_h)], fill=(255, 255, 255, 20), width=1) + + draw.text((MARGIN, CANVAS_H - MARGIN - 20), TAGLINE, font=f_tag, fill=WHITE) + + os.makedirs(os.path.dirname(out_path), exist_ok=True) + base.save(out_path, "PNG") + + +# --- Stage 3: CAPTION --------------------------------------------------------- + +def build_caption(slug: str, display_name: str, year: int, top10: list, avg_ecr: float) -> str: + """Caption rules — context/infographic-design.md § Caption rules.""" + leader = top10[0] + gap = (leader["value"] - avg_ecr) * 100 # same raw-ratio-to-percent fix as format_ecr() + top_rated_count = sum(1 for c in top10 if c.get("top_rated")) + + lines = [ + "RealRate's ECR isolates how structurally sound a company's finances are, " + "independent of size or reputation.", + "", + f"In RealRate's {year} Top 10 ranking for {display_name}, {leader['name']} " + f"leads at {format_ecr(leader['value'])} ECR — " + f"{gap:.1f} points above the industry average of {format_ecr(avg_ecr)}.", + "", + f"{top_rated_count} of the top 10 hold Top-Rated status this cycle.", + "", + f"Full ranking: https://realrate.ai/rankings/us_{slug}/{year}", + ] + return "\n".join(lines) + + +def main(): + parser = argparse.ArgumentParser(description="Practice: generate a RealRate Top 10 ranking infographic.") + parser.add_argument("industry", help="Industry slug or name, e.g. 'software' or 'US Software'") + parser.add_argument("--year", type=int, default=None, help="Override the archive year to fetch") + args = parser.parse_args() + + slug = resolve_slug(args.industry) + display_name = INDUSTRY_SLUGS[slug] + + data, fetch_year = fetch_ranking_data(slug, args.year) + # The URL path uses the balance sheet year (`fetch_year`), but the JSON's + # own `year` field is the marketing year (matches `url_segment`, e.g. + # "2026-us-air") — that's what belongs in the title/filename/ranking URL, + # confirmed by comparing both against the live archive. + year = data.get("year", fetch_year) + companies = sorted(data["company_details"], key=lambda c: c["rank"]) + top10 = companies[:10] + avg_ecr = sum(c["value"] for c in companies) / len(companies) + + industry_folder = f"us_{slug}" + out_dir = os.path.join(os.getcwd(), "output", industry_folder, SKILL_NAME) + png_path = os.path.join(out_dir, f"{slug}_{year}.png") + txt_path = os.path.join(out_dir, f"linkedin_{slug}_{year}.txt") + + render_image(slug, display_name, year, top10, avg_ecr, png_path) + with open(txt_path, "w") as f: + f.write(build_caption(slug, display_name, year, top10, avg_ecr)) + + print(f"Wrote {png_path}") + print(f"Wrote {txt_path}") + + +if __name__ == "__main__": + main() diff --git a/claude_headless/scripts/generate_mindmap.py b/claude_headless/scripts/generate_mindmap.py new file mode 100644 index 0000000..8554948 --- /dev/null +++ b/claude_headless/scripts/generate_mindmap.py @@ -0,0 +1,343 @@ +#!/usr/bin/env python3 +""" +Practice build — RealRate Company Mindmap Generator. + +Rebuilt from skills/mindmap.md (real repo), which hardcodes exactly 8 +companies and pulls executive/office photos from Wikipedia via Playwright. +This version works for ANY company in the archive and uses only fields +already verified in this session (ECR, rank, report_text's strength/ +weakness sentences, logo_url) — no scraped photos of real people, no +browser dependency. Rendered directly with Pillow, same cross-platform font +fallback proven in generate_infographic.py (the original's font loader only +checked Windows paths and silently degraded elsewhere). + +Branches are adaptive: a company missing multi-year history just gets one +fewer branch, laid out evenly among however many qualify — same principle +as the report skills' adaptive sections, applied to a single image instead +of PDF pages. + +Usage: + python generate_mindmap.py [--year YEAR] + +Output: + output/us_/mindmap/__mindmap.png +""" + +import argparse +import datetime +import io +import math +import os +import re +import urllib.request +import json + +import cairosvg +from PIL import Image, ImageDraw, ImageFont + +SKILL_NAME = "mindmap" + +CANVAS_W, CANVAS_H = 1920, 1080 +NAVY = (0, 59, 87) +TEAL = (61, 186, 205) +WHITE = (255, 255, 255) +OFF_WHITE = (245, 245, 245) +POSITIVE = (65, 148, 83) +NEGATIVE = (192, 74, 58) + +BRANCH_COLORS = [ + (0, 74, 110), (0, 88, 132), (51, 137, 177), + (52, 162, 179), (44, 139, 154), (61, 186, 205), +] + +RR_LOGO_SVG_PATH = os.path.join( + os.path.dirname(os.path.abspath(__file__)), "..", "context", "RealRate_logo_light.svg" +) + +INDUSTRY_SLUGS = { + "air": "Air", "motor": "Motor", "software": "Software", "computers": "Computers", + "finance_services": "Finance Services", "food": "Food", "health_services": "Health Services", + "advertising": "Advertising", "semiconductors": "Semiconductors", "programming": "Programming", + "petrol": "Petrol", "mining": "Mining", "construction": "Construction", + "realestate": "Real Estate", "hotels": "Hotels", "consulting": "Consulting", + "data_processing": "Data Processing", "brokers": "Brokers", "savings": "Savings", + "life": "Life Insurance", "non_life": "Non-Life Insurance", "pharma": "Pharma", + "chemicals": "Chemicals", "state_banks": "State Banks", + "medicinal_products": "Medicinal Products", "recreation": "Recreation", +} + + +# --- FETCH --------------------------------------------------------------------- + +def resolve_slug(arg: str) -> str: + if arg in INDUSTRY_SLUGS: + return arg + lowered = arg.lower().replace(" ", "_").replace("-", "_") + if lowered.startswith("us_"): + lowered = lowered[3:] + if lowered in INDUSTRY_SLUGS: + return lowered + for slug, name in INDUSTRY_SLUGS.items(): + if arg.lower() == name.lower(): + return slug + raise SystemExit(f"Unknown industry '{arg}'. Available slugs: {', '.join(sorted(INDUSTRY_SLUGS))}") + + +def fetch_ranking_data(slug: str, year: int | None): + candidates = [year] if year else [datetime.date.today().year - i for i in range(0, 4)] + tried = [] + for y in candidates: + url = f"https://www.realrate-archive.com/us_{slug}/{y}/website-ranking.json" + tried.append(url) + try: + req = urllib.request.Request(url, headers={"User-Agent": "RealRate-Report-Practice/1.0"}) + with urllib.request.urlopen(req, timeout=15) as resp: + if resp.status != 200: + continue + data = json.loads(resp.read().decode("utf-8")) + if data.get("company_details"): + return data + except Exception: + continue + raise RuntimeError(f"Could not fetch ranking data for slug '{slug}'. Tried:\n " + "\n ".join(tried)) + + +def resolve_company(data: dict, arg: str) -> dict: + companies = sorted(data["company_details"], key=lambda c: c["rank"]) + if arg.isdigit(): + rank = int(arg) + for c in companies: + if c["rank"] == rank: + return c + raise SystemExit(f"No company at rank {rank}. This industry has {len(companies)} ranked companies.") + matches = [c for c in companies if arg.lower() in c["name"].lower()] + if not matches: + available = "\n ".join(f"#{c['rank']} {c['name']}" for c in companies) + raise SystemExit(f"No company matching '{arg}'. Available:\n {available}") + return matches[0] + + +def fetch_svg(url: str) -> str | None: + try: + req = urllib.request.Request(url, headers={"User-Agent": "RealRate-Report-Practice/1.0"}) + with urllib.request.urlopen(req, timeout=15) as resp: + return resp.read().decode("utf-8") if resp.status == 200 else None + except Exception: + return None + + +# --- ANALYZE --------------------------------------------------------------------- + +def ecr_pct(value: float) -> float: + return value * 100 + + +def clean_archive_text(s: str) -> str: + s = re.sub(r"", " ", s, flags=re.IGNORECASE) + s = re.sub(r"<[^>]+>", "", s) + return re.sub(r"\s+", " ", s).strip() + + +def parse_strength_weakness(report_text: str): + """report_text follows a consistent template across every company + checked this session — see context/mindmap-design.md. Returns + (strength_var, strength_pts, weakness_var, weakness_pts), any of which + may be None if the pattern isn't found (best-effort, not guaranteed). + + Anchored on "greatest strength/weakness of ... is the variable" rather + than a bare "variable (.+?)," — the bare form's non-greedy match still + locks onto the FIRST "variable" in the text (the strength sentence) and + swallows everything up to the second occurrence of the target keyword, + producing a paragraph-long "variable name" for the weakness case.""" + text = clean_archive_text(report_text) + s_match = re.search( + r"greatest strength of .+? is the variable (.+?), increasing the Economic Capital Ratio by (\d+)%", text) + w_match = re.search( + r"greatest weakness of .+? is the variable (.+?), reducing the Economic Capital Ratio by (\d+)%", text) + strength = (s_match.group(1), int(s_match.group(2))) if s_match else (None, None) + weakness = (w_match.group(1), int(w_match.group(2))) if w_match else (None, None) + return strength[0], strength[1], weakness[0], weakness[1] + + +def build_branches(data: dict, company: dict) -> list: + """Each branch is (label, [text lines], color). Adaptive — a branch is + only included if its underlying data is actually present.""" + companies = sorted(data["company_details"], key=lambda c: c["rank"]) + avg = sum(ecr_pct(c["value"]) for c in companies) / len(companies) + ecr = ecr_pct(company["value"]) + branches = [] + + branches.append(( + "Industry Position", + [f"Rank #{company['rank']} of {len(companies)}", data.get("industry_label", "")], + )) + + gap = ecr - avg + branches.append(( + "Financial Health", + [f"ECR {ecr:.1f}%", f"{abs(gap):.1f} pts {'above' if gap >= 0 else 'below'} avg"], + )) + + if company.get("report_text"): + s_var, s_pts, w_var, w_pts = parse_strength_weakness(company["report_text"]) + if s_var: + branches.append(("Greatest Strength", [s_var, f"+{s_pts} pts to ECR"])) + if w_var: + branches.append(("Greatest Weakness", [w_var, f"-{w_pts} pts to ECR"])) + + records = data.get("ecr_records", {}) + cid = company["company_id"] + years_present = sorted(y for y in records if cid in records[y]) + if len(years_present) >= 2: + first, last = years_present[0], years_present[-1] + first_v = float(records[first][cid]["ecr"]) + last_v = float(records[last][cid]["ecr"]) + direction = "up" if last_v > first_v else "down" if last_v < first_v else "flat" + branches.append(("Multi-Year Trend", [f"{first}→{last}: {direction}", f"{first_v:.0f}% → {last_v:.0f}%"])) + + status = "Top-Rated" if company.get("top_rated") else "Not Top-Rated" + branches.append(("Rating Status", [status, f"{company.get('year', data.get('year'))} cycle"])) + + return branches + + +# --- RENDER --------------------------------------------------------------------- + +def _load_font(size: int, bold: bool = False): + """Same cross-platform fallback chain as generate_infographic.py — the + original mindmap script only checked Windows font paths.""" + candidates = [ + "/usr/share/fonts/truetype/dejavu/DejaVuSans-Bold.ttf" if bold else + "/usr/share/fonts/truetype/dejavu/DejaVuSans.ttf", + "/usr/share/fonts/truetype/liberation/LiberationSans-Bold.ttf" if bold else + "/usr/share/fonts/truetype/liberation/LiberationSans-Regular.ttf", + "/System/Library/Fonts/Supplemental/Arial Bold.ttf" if bold else + "/System/Library/Fonts/Supplemental/Arial.ttf", + ] + for path in candidates: + if os.path.exists(path): + return ImageFont.truetype(path, size) + return ImageFont.load_default() + + +def load_logo(svg_path_or_content: str, height_px: int, is_content=False) -> Image.Image: + if is_content: + png_bytes = cairosvg.svg2png(bytestring=svg_path_or_content.encode("utf-8"), output_height=height_px) + else: + png_bytes = cairosvg.svg2png(url=svg_path_or_content, output_height=height_px) + return Image.open(io.BytesIO(png_bytes)).convert("RGBA") + + +def draw_centered_text(draw, cx, cy, lines, font, fill, max_width=340): + """max_width guards against a bad upstream parse (e.g. a regex capture + swallowing a whole sentence) turning one branch into unreadable + canvas-wide text — truncate rather than let it overflow silently.""" + total_h = len(lines) * (font.size + 6) + y = cy - total_h / 2 + for line in lines: + while draw.textlength(line, font=font) > max_width and len(line) > 3: + line = line[:-4] + "…" + w = draw.textlength(line, font=font) + draw.text((cx - w / 2, y), line, font=font, fill=fill) + y += font.size + 6 + + +def render_mindmap(data: dict, company: dict, out_path: str): + branches = build_branches(data, company) + n = len(branches) + + base = Image.new("RGB", (CANVAS_W, CANVAS_H), NAVY) + draw = ImageDraw.Draw(base) + + hub_cx, hub_cy, hub_r = CANVAS_W // 2, CANVAS_H // 2, 190 + branch_r = 420 + branch_radius = 150 + + f_label = _load_font(22, bold=True) + f_line = _load_font(20, bold=False) + f_badge = _load_font(20, bold=True) + + # Shrink the hub name to fit rather than truncate — a company name is + # the one label a reader most needs intact. + f_name = _load_font(34, bold=True) + while draw.textlength(company["name"], font=f_name) > hub_r * 1.7 and f_name.size > 18: + f_name = _load_font(f_name.size - 2, bold=True) + + # Connector lines (drawn first, under everything) + positions = [] + for i in range(n): + angle = -math.pi / 2 + i * (2 * math.pi / n) + bx = hub_cx + branch_r * math.cos(angle) + by = hub_cy + branch_r * math.sin(angle) + positions.append((bx, by)) + draw.line([(hub_cx, hub_cy), (bx, by)], fill=(*TEAL, 255), width=3) + + # Branch circles + labels + for (label, lines), (bx, by), color in zip(branches, positions, BRANCH_COLORS * 2): + draw.ellipse([bx - branch_radius, by - branch_radius, bx + branch_radius, by + branch_radius], + fill=color, outline=WHITE, width=2) + draw_centered_text(draw, bx, by - 20, [label], f_label, TEAL) + draw_centered_text(draw, bx, by + 15, lines, f_line, WHITE) + + # Hub (drawn last, on top of connector lines) + draw.ellipse([hub_cx - hub_r, hub_cy - hub_r, hub_cx + hub_r, hub_cy + hub_r], + fill=NAVY, outline=TEAL, width=4) + + logo = load_logo(RR_LOGO_SVG_PATH, height_px=36) + base.paste(logo, (hub_cx - logo.width // 2, hub_cy - hub_r + 24), logo) + + draw_centered_text(draw, hub_cx, hub_cy - 20, [company["name"]], f_name, WHITE, max_width=hub_r * 1.8) + + ecr = ecr_pct(company["value"]) + badge_text = f"ECR {ecr:.1f}% · Rank #{company['rank']}" + bw = draw.textlength(badge_text, font=f_badge) + draw.text((hub_cx - bw / 2, hub_cy + 40), badge_text, font=f_badge, fill=OFF_WHITE) + + os.makedirs(os.path.dirname(out_path), exist_ok=True) + base.save(out_path, "PNG") + + +def generate_for_company(data: dict, company: dict, slug: str, year: int) -> str: + company_slug = re.sub(r"[^a-z0-9]+", "_", company["name"].lower()).strip("_") + out_dir = os.path.join(os.getcwd(), "output", f"us_{slug}", SKILL_NAME) + out_path = os.path.join(out_dir, f"{company_slug}_{year}_mindmap.png") + + render_mindmap(data, company, out_path) + print(f"Wrote {out_path}") + print(f" Branches included: {len(build_branches(data, company))} (adaptive — depends on available data)") + return out_path + + +def main(): + parser = argparse.ArgumentParser(description="Practice: generate a RealRate company mindmap PNG.") + parser.add_argument("industry", help="Industry slug or name, e.g. 'air' or 'US Air'") + parser.add_argument("company", help="Company rank (e.g. '1'), a name substring, or 'all' to " + "generate one mindmap per company in the industry") + parser.add_argument("--year", type=int, default=None, help="Override the archive year to fetch") + args = parser.parse_args() + + slug = resolve_slug(args.industry) + data = fetch_ranking_data(slug, args.year) + year = data.get("year") + + if args.company.strip().lower() == "all": + companies = sorted(data["company_details"], key=lambda c: c["rank"]) + print(f"Generating mindmaps for all {len(companies)} companies...") + failures = [] + for company in companies: + try: + generate_for_company(data, company, slug, year) + except Exception as e: + print(f" FAILED for {company['name']} (rank {company['rank']}): {e}") + failures.append(company["name"]) + print(f"Done: {len(companies) - len(failures)}/{len(companies)} mindmaps generated.") + if failures: + print(f"Failed: {', '.join(failures)}") + return + + company = resolve_company(data, args.company) + generate_for_company(data, company, slug, year) + + +if __name__ == "__main__": + main() diff --git a/claude_headless/scripts/generate_top5_reveal_gif.py b/claude_headless/scripts/generate_top5_reveal_gif.py new file mode 100644 index 0000000..a374c99 --- /dev/null +++ b/claude_headless/scripts/generate_top5_reveal_gif.py @@ -0,0 +1,243 @@ +#!/usr/bin/env python3 +""" +Practice build — RealRate Top 5 Reveal GIF Generator. + +Ported from "Industry Deep Dive Carousel/generate_cover_gif.py" (real +repo), which is Pillow-only (good — no browser dependency) but its +load_font() only tries Windows font paths (C:\\Windows\\Fonts\\...) and +silently falls back to Pillow's low-quality bitmap default everywhere +else. This version reuses the cross-platform font-fallback chain already +proven in generate_infographic.py. + +Also generalized: the original is a hand-edited template (every constant +marked "# EDIT", meant to be edited in place per industry and left +edited). This version is CLI-driven like every other skill in this +project — no manual editing required to run it for a different industry. + +Animates a rank-5-to-rank-1 reveal (suspense builds toward the leader), +holds on the final frame, then loops. + +Usage: + python generate_top5_reveal_gif.py [--year YEAR] + +Output: + output/us_/top5-reveal-gif/__top5.gif +""" + +import argparse +import datetime +import io +import os +import urllib.request +import json + +import cairosvg +from PIL import Image, ImageDraw, ImageFont + +SKILL_NAME = "top5-reveal-gif" + +CANVAS_W, CANVAS_H = 1080, 1080 +MARGIN = 60 + +NAVY = (0, 59, 87) +TEAL = (61, 186, 205) +WHITE = (255, 255, 255) +OFF_WHITE = (245, 245, 245) +GOLD = (212, 175, 55) +SILVER = (192, 192, 192) +BRONZE = (176, 118, 62) +RANK_COLORS = {1: GOLD, 2: SILVER, 3: BRONZE} +DEFAULT_RANK_COLOR = (0, 88, 132) + +RR_LOGO_SVG_PATH = os.path.join( + os.path.dirname(os.path.abspath(__file__)), "..", "context", "RealRate_logo_light.svg" +) + +INDUSTRY_SLUGS = { + "air": "Air", "motor": "Motor", "software": "Software", "computers": "Computers", + "finance_services": "Finance Services", "food": "Food", "health_services": "Health Services", + "advertising": "Advertising", "semiconductors": "Semiconductors", "programming": "Programming", + "petrol": "Petrol", "mining": "Mining", "construction": "Construction", + "realestate": "Real Estate", "hotels": "Hotels", "consulting": "Consulting", + "data_processing": "Data Processing", "brokers": "Brokers", "savings": "Savings", + "life": "Life Insurance", "non_life": "Non-Life Insurance", "pharma": "Pharma", + "chemicals": "Chemicals", "state_banks": "State Banks", + "medicinal_products": "Medicinal Products", "recreation": "Recreation", +} + + +# --- FETCH --------------------------------------------------------------------- + +def resolve_slug(arg: str) -> str: + if arg in INDUSTRY_SLUGS: + return arg + lowered = arg.lower().replace(" ", "_").replace("-", "_") + if lowered.startswith("us_"): + lowered = lowered[3:] + if lowered in INDUSTRY_SLUGS: + return lowered + for slug, name in INDUSTRY_SLUGS.items(): + if arg.lower() == name.lower(): + return slug + raise SystemExit(f"Unknown industry '{arg}'. Available slugs: {', '.join(sorted(INDUSTRY_SLUGS))}") + + +def fetch_ranking_data(slug: str, year: int | None): + candidates = [year] if year else [datetime.date.today().year - i for i in range(0, 4)] + tried = [] + for y in candidates: + url = f"https://www.realrate-archive.com/us_{slug}/{y}/website-ranking.json" + tried.append(url) + try: + req = urllib.request.Request(url, headers={"User-Agent": "RealRate-Report-Practice/1.0"}) + with urllib.request.urlopen(req, timeout=15) as resp: + if resp.status != 200: + continue + data = json.loads(resp.read().decode("utf-8")) + if data.get("company_details"): + return data + except Exception: + continue + raise RuntimeError(f"Could not fetch ranking data for slug '{slug}'. Tried:\n " + "\n ".join(tried)) + + +def ecr_pct(value: float) -> float: + return value * 100 + + +def initials(name: str) -> str: + parts = [p for p in name.replace(",", " ").split() if p] + letters = [p[0].upper() for p in parts if p[0].isalpha()] + return "".join(letters[:2]) if letters else "?" + + +# --- RENDER --------------------------------------------------------------------- + +def _load_font(size: int, bold: bool = False): + """Cross-platform fallback chain — the original script this was ported + from only checked C:\\Windows\\Fonts\\... paths.""" + candidates = [ + "/usr/share/fonts/truetype/dejavu/DejaVuSans-Bold.ttf" if bold else + "/usr/share/fonts/truetype/dejavu/DejaVuSans.ttf", + "/usr/share/fonts/truetype/liberation/LiberationSans-Bold.ttf" if bold else + "/usr/share/fonts/truetype/liberation/LiberationSans-Regular.ttf", + "/System/Library/Fonts/Supplemental/Arial Bold.ttf" if bold else + "/System/Library/Fonts/Supplemental/Arial.ttf", + ] + for path in candidates: + if os.path.exists(path): + return ImageFont.truetype(path, size) + return ImageFont.load_default() + + +def load_logo(height_px: int = 60) -> Image.Image: + png_bytes = cairosvg.svg2png(url=RR_LOGO_SVG_PATH, output_height=height_px) + return Image.open(io.BytesIO(png_bytes)).convert("RGBA") + + +def build_base(display_name: str, year: int) -> Image.Image: + base = Image.new("RGB", (CANVAS_W, CANVAS_H), NAVY) + draw = ImageDraw.Draw(base) + + logo = load_logo(48) + base.paste(logo, (MARGIN, MARGIN), logo) + + f_title = _load_font(46, bold=True) + f_sub = _load_font(24, bold=False) + draw.text((MARGIN, MARGIN + 80), f"US {display_name.upper()}", font=f_title, fill=WHITE) + draw.text((MARGIN, MARGIN + 136), f"TOP 5 · {year}", font=f_title, fill=TEAL) + draw.text((MARGIN, MARGIN + 196), "Ranked by Economic Capital Ratio", font=f_sub, fill=OFF_WHITE) + return base + + +def draw_row(draw: ImageDraw.ImageDraw, company: dict, row_index: int, top_y: int, row_h: int): + ry = top_y + row_index * row_h + row_mid = ry + row_h / 2 + rank = company["rank"] + color = RANK_COLORS.get(rank, DEFAULT_RANK_COLOR) + + f_rank = _load_font(28, bold=True) + f_name = _load_font(26, bold=True) + f_stat = _load_font(30, bold=True) + f_badge = _load_font(16, bold=True) + + av_cx, av_r = MARGIN + 90, 30 + draw.ellipse([av_cx - av_r, row_mid - av_r, av_cx + av_r, row_mid + av_r], fill=color) + rank_text = str(rank) + rw = draw.textlength(rank_text, font=f_rank) + draw.text((av_cx - rw / 2, row_mid - 17), rank_text, font=f_rank, fill=WHITE if rank > 3 else NAVY) + + name_x = av_cx + av_r + 24 + name = company["name"] + max_name_w = 480 + while draw.textlength(name, font=f_name) > max_name_w and len(name) > 3: + name = name[:-2] + "…" + draw.text((name_x, row_mid - 16), name, font=f_name, fill=WHITE) + + if company.get("top_rated"): + draw.text((name_x, row_mid + 14), "★ TOP RATED", font=f_badge, fill=TEAL) + + stat_text = f"{ecr_pct(company['value']):.1f}%" + sw = draw.textlength(stat_text, font=f_stat) + draw.text((CANVAS_W - MARGIN - sw, row_mid - 18), stat_text, font=f_stat, fill=OFF_WHITE) + + +def build_frames(data: dict, display_name: str, year: int) -> list: + companies = sorted(data["company_details"], key=lambda c: c["rank"])[:5] + base = build_base(display_name, year) + + list_top = MARGIN + 260 + row_h = 130 + reveal_order = list(reversed(companies)) # rank 5 first, rank 1 last + + frames = [] + revealed = [] + for company in reveal_order: + revealed.append(company) + frame = base.copy() + draw = ImageDraw.Draw(frame) + # Draw already-revealed rows in their final rank position, so a + # newly revealed row appears in place rather than the list + # reshuffling frame to frame. + for c in revealed: + row_index = c["rank"] - 1 + draw_row(draw, c, row_index, list_top, row_h) + frames.append(frame) + + return frames + + +def main(): + parser = argparse.ArgumentParser(description="Practice: generate a RealRate Top 5 reveal GIF.") + parser.add_argument("industry", help="Industry slug or name, e.g. 'air' or 'US Air'") + parser.add_argument("--year", type=int, default=None, help="Override the archive year to fetch") + args = parser.parse_args() + + slug = resolve_slug(args.industry) + display_name = INDUSTRY_SLUGS[slug] + + data = fetch_ranking_data(slug, args.year) + year = data.get("year") + + frames = build_frames(data, display_name, year) + + out_dir = os.path.join(os.getcwd(), "output", f"us_{slug}", SKILL_NAME) + os.makedirs(out_dir, exist_ok=True) + gif_path = os.path.join(out_dir, f"{slug}_{year}_top5.gif") + + # Per-frame duration (not duplicate frames) to hold on the final + # state — Pillow's GIF writer collapses pixel-identical consecutive + # frames even with optimize=False, so duplicating frames to create a + # "hold" silently produces a shorter GIF than intended. A duration + # list is the correct way to make one frame linger. + durations = [700] * (len(frames) - 1) + [3000] + frames[0].save( + gif_path, format="GIF", save_all=True, append_images=frames[1:], + duration=durations, loop=0, optimize=False, + ) + print(f"Wrote {gif_path}") + print(f"Frames: {len(frames)} (5 reveal steps, final frame held for {durations[-1]}ms)") + + +if __name__ == "__main__": + main() diff --git a/claude_headless/skills/auto-practice-cycle.md b/claude_headless/skills/auto-practice-cycle.md new file mode 100644 index 0000000..c3dd77e --- /dev/null +++ b/claude_headless/skills/auto-practice-cycle.md @@ -0,0 +1,97 @@ +--- +description: Run the full claude_practice pipeline for one industry in a single pass — top10 infographic, industry report, company report, mindmap, and top5 reveal GIF. Orchestrator, modeled on skills/auto-post-cycle.md (real repo), scoped down to what actually exists and runs in claude_practice. +arguments: industry_slug [company_rank_or_name] [year] +--- + +# Practice Skill — Auto Practice Cycle (Orchestrator) + +Runs every skill built in `claude_practice/` for one industry, in order, +in a single pass. No dashboard upload, no Drive, no `ADMIN_SECRET` — +those are real-repo `auto-post-cycle.md` concerns that don't apply here; +this orchestrator only chains local script runs and reports what was +produced. + +## Arguments + +- `$ARGUMENTS` parses as: `{industry_slug} [company_rank_or_name] [year]` +- `industry_slug` (required) — e.g. `air`, `hotels` (also accepts `us_air` / `us-air`) +- `company_rank_or_name` (optional) — which company(s) to use for the two + per-company skills (`company-report`, `mindmap`). **Defaults to `all`** + — every company in the industry gets its own report and mindmap. Pass a + rank number or name substring instead to scope to one company. +- `year` (optional) — overrides the balance-sheet-year URL fetched; every + script independently resolves the correct marketing year from the + payload regardless + +Both per-company scripts accept `all` natively (they loop internally, one +process — not one invocation per company from this orchestrator), and +skip-and-continue past any single company's failure rather than aborting +the batch. On a large industry this step takes noticeably longer than the +other four combined (one causal-graph SVG fetch per company for +`company-report`) — that's expected, not a hang. + +## Step 0 — Setup (first run only) + +All five scripts share one venv. Create it if it doesn't already exist: + +```bash +python3 -m venv "${CLAUDE_PROJECT_DIR}/claude_practice/.venv" +"${CLAUDE_PROJECT_DIR}/claude_practice/.venv/bin/pip" install --upgrade pip +"${CLAUDE_PROJECT_DIR}/claude_practice/.venv/bin/pip" install Pillow cairosvg matplotlib +``` + +## Step 1 — Run each skill in order + +Run from `claude_practice/` so relative output paths land correctly. If +any single step fails, report the error and continue with the remaining +steps — don't let one skill's failure block the others (same +error-handling principle as `skills/auto-post-cycle.md`). + +```bash +cd "${CLAUDE_PROJECT_DIR}/claude_practice" +VENV="${CLAUDE_PROJECT_DIR}/claude_practice/.venv/bin/python" +COMPANY="" + +"$VENV" scripts/generate_infographic.py +"$VENV" scripts/generate_industry_report.py +"$VENV" scripts/generate_company_report.py "$COMPANY" +"$VENV" scripts/generate_mindmap.py "$COMPANY" +"$VENV" scripts/generate_top5_reveal_gif.py +``` + +Append `--year ` to each command only if `$ARGUMENTS` explicitly +named one. + +## Step 2 — Summary + +All five skills write into the same `output/us_/` folder, namespaced +by skill (`top10-infographic/`, `industry-report/`, `company-report/`, +`mindmap/`, `top5-reveal-gif/`) — same convention established across every +skill in this project. After running, list what actually landed there: + +```bash +find "${CLAUDE_PROJECT_DIR}/claude_practice/output/us_" -type f +``` + +Report: +- Each file produced, grouped by skill +- Section/branch/frame counts each script printed to stdout (industry-report + and company-report report adaptive section counts; mindmap reports branch + count) — these are meaningful, not just log noise; a lower-than-expected + count means that skill's data wasn't fully available for this + industry/company, not that something broke +- Any step that failed, with its error, so it can be re-run individually + once fixed — don't silently omit a failed step from the summary + +## Known caveats to mention in the summary + +Carry forward from the individual skill files — don't re-verify these each +run, just note they apply: +- ECR figures assume `value × 100` and the marketing year from the + payload's `year` field (see `context/infographic-design.md` § Resolved) +- The causal graph's own ECR (in `company-report`) may not match the + headline ECR — see `context/company-report-design.md`. Confirmed on a + full `us_air` `all` run: 5 of 10 companies showed a mismatch, so expect + this on a meaningful fraction of any industry's companies, not just an + isolated case +- `trend`'s sign convention is still unverified everywhere it appears diff --git a/claude_headless/skills/company-report.md b/claude_headless/skills/company-report.md new file mode 100644 index 0000000..9d97bc8 --- /dev/null +++ b/claude_headless/skills/company-report.md @@ -0,0 +1,108 @@ +--- +description: Generate a RealRate company report PDF — an adaptive set of chart+interpretation sections for one company within an industry, including RealRate's own causal ECR graph rendered from its source SVG. Companion to skills/industry-report.md, same adaptive-section architecture one level deeper. +argument-hint: [industry_slug] [rank_or_company_name] [year (optional)] +--- + +# Practice Skill — Company Report + +Generates a multi-page PDF report for a single company: cover page plus +whichever candidate sections have supporting data for this company this +run. Same principle as `skills/industry-report.md` — sections are adaptive, +not a fixed count. + +Design context: `claude_practice/context/company-report-design.md` — read +it before touching anything here. It documents the verified schema, how the +causal graph SVG is parsed, and a **confirmed discrepancy** you need to +know about before extending this skill. + +## The discrepancy this skill surfaces, not hides + +For the same company/year, `company_details.value × 100` and the causal +graph's own `EconomicCapitalRatio` node report *different* ECR values +(verified: 123.5% vs. 54.3% for Strata Critical Medical Inc, `us_air`, +2025). Both come directly from RealRate. This skill states both, labeled by +source, in the Causal ECR Graph and Caveats sections — it does not average +them, does not pick one as "correct," and does not hide the mismatch. +If you extend this skill, preserve that behavior; resolving the +discrepancy is a question for RealRate's data team, not something to guess +at in a chart script. + +## Arguments + +- `$ARGUMENTS` parses as: `{industry_slug} {rank_or_name} [year]` +- `industry_slug` (required) — e.g. `air` (also accepts `us_air` / `us-air`) +- `rank_or_name` (required) — a rank number (`1` = industry leader), a + case-insensitive substring of the company name (e.g. `strata`), or + **`all`** to generate one report per company in the industry (loops + internally — one process, not one invocation per company; a failure on + one company is logged and skipped, not fatal to the rest) +- `year` (optional) — overrides the balance-sheet-year URL fetched; the + report always displays the marketing year from the payload's `year` field + +## Setup (first run only) + +Shares the venv with `top10-infographic` / `industry-report` — same +dependencies: + +```bash +python3 -m venv "${CLAUDE_PROJECT_DIR}/claude_practice/.venv" +"${CLAUDE_PROJECT_DIR}/claude_practice/.venv/bin/pip" install --upgrade pip +"${CLAUDE_PROJECT_DIR}/claude_practice/.venv/bin/pip" install Pillow cairosvg matplotlib +``` + +## Generate + +```bash +cd "${CLAUDE_PROJECT_DIR}/claude_practice" +"${CLAUDE_PROJECT_DIR}/claude_practice/.venv/bin/python" \ + "${CLAUDE_PROJECT_DIR}/claude_practice/scripts/generate_company_report.py" +``` + +Pass `--year ` only if `$ARGUMENTS` explicitly names one. If the +company name doesn't match, the script lists every ranked company in that +industry with its rank — don't guess a company_id, re-run with the printed +name instead. Pass `all` in place of `` to generate a report +for every company in the industry in one run. + +## Sections are adaptive + +Candidates, in render order: Company Overview, Causal ECR Graph, Strengths +& Weaknesses, Balance Sheet Snapshot, Multi-Year ECR History, Peer +Comparison, Data Confidence & Caveats, Takeaway & Full Report. Full +requires/chart/text spec is in `context/company-report-design.md`. A +company missing `report_text`, a fetchable `graph_url`, or 3+ years of +`ecr_records` history gets correspondingly fewer pages — that's correct +behavior, not a bug to patch around. + +Two sections quote RealRate's own text verbatim rather than generating +interpretation: **Strengths & Weaknesses** (`report_text`) and the +discrepancy note in **Causal ECR Graph**. HTML markup (literal `
`) is +stripped before quoting, per the same `clean_archive_text()` helper used in +the industry report. + +## Output contract + +- `claude_practice/output/us_/company-report/__report.pdf` +- With `all`: one such file per company, same folder, all in one run + +Console output on run lists included vs. skipped sections per company, and +prints an explicit `NOTE:` line if the causal-graph/headline ECR mismatch +applies to that company — check that before treating any PDF as complete. +With `all`, a per-company `FAILED` line (with the error) plus a final +`Done: N/M reports generated` count replaces the single-run summary — +check that count, not just that the command exited, since individual +company failures don't stop the batch. + +## Known unresolved data issues (don't silently fix these) + +- Causal graph ECR vs. headline ECR mismatch (see above) +- Balance sheet figures in `table_records` have no stated unit — charted as + relative magnitudes, axis labeled accordingly, never assumed to be + thousands/millions/dollars +- `trend` sign convention still unverified (carried from the infographic + and industry-report skills) + +## After generating + +Report the output PDF path, the section include/skip list, and — if +present — the causal-graph/headline ECR mismatch note from stdout. diff --git a/claude_headless/skills/industry-report.md b/claude_headless/skills/industry-report.md new file mode 100644 index 0000000..520db36 --- /dev/null +++ b/claude_headless/skills/industry-report.md @@ -0,0 +1,91 @@ +--- +description: Generate a RealRate industry report PDF — an adaptive set of chart+interpretation sections covering rankings, distribution, trend, top-rated status, notable movers, and data caveats, built entirely from one live archive JSON response. Practice-built alongside skills/industry-ranking-reports.md, which targets a rr_shared.py module and docx/Chrome pipeline that don't exist in this repo. +argument-hint: [industry_slug] [year (optional)] +--- + +# Practice Skill — Industry Report + +Generates a multi-page PDF industry report: a cover page plus one page per +section that actually has supporting data for this run — see "Sections are +adaptive" below. Every number and quote traces back to one archive response; +nothing is invented. + +Design context: `claude_practice/context/industry-report-design.md` — read +it before adding a new section or changing what a chart shows. It documents +the verified data schema (which fields exist, what they mean, what's still +unverified) and the chart style rules. + +## Arguments + +- `$ARGUMENTS` parses as: `{industry_slug} [year]` +- `industry_slug` (required) — e.g. `air`, `software`, `hotels` (also + accepts `us_air` / `us-air`) +- `year` (optional) — overrides which balance-sheet-year URL to fetch; the + report always *displays* the marketing year from the payload's own `year` + field, not the fetch year (see context file § Fields) + +## Setup (first run only) + +```bash +python3 -m venv "${CLAUDE_PROJECT_DIR}/claude_practice/.venv" +"${CLAUDE_PROJECT_DIR}/claude_practice/.venv/bin/pip" install --upgrade pip +"${CLAUDE_PROJECT_DIR}/claude_practice/.venv/bin/pip" install Pillow cairosvg matplotlib +``` + +Skip if the venv already exists — shared with `top10-infographic`, which +uses the same Pillow/cairosvg but not matplotlib. + +## Generate + +```bash +cd "${CLAUDE_PROJECT_DIR}/claude_practice" +"${CLAUDE_PROJECT_DIR}/claude_practice/.venv/bin/python" \ + "${CLAUDE_PROJECT_DIR}/claude_practice/scripts/generate_industry_report.py" +``` + +Pass `--year ` only if `$ARGUMENTS` explicitly names one. + +## Sections are adaptive, not fixed + +The script holds a `SECTION_REGISTRY` of 10 candidate sections. Each one +declares a `requires(data)` check against *this run's* actual archive +response; only sections that pass are rendered. A data-rich industry may +get all 10 pages, a sparse one fewer — the script prints exactly which +sections were included and which were skipped (and why, implicitly, via +the `requires` check that failed). Never edit a section to force it to +always appear — if the data doesn't support it, skipping is correct. + +Current candidates, in render order: Industry Overview, Top 10 Rankings, +ECR Distribution, Top-Rated Snapshot, Multi-Year ECR Trend, Sector Leader +Profile, Notable Movers, Competitive Compression, Data Confidence & +Caveats, Takeaway & Full Ranking. Full requires/chart/text spec for each is +in `context/industry-report-design.md`. + +Two sections (**Sector Leader Profile**, **Notable Movers**) quote +RealRate's own `report_text` / `scentences` fields verbatim rather than +generating interpretation — real analyst text beats invented prose. HTML +markup in those fields (e.g. literal `
`) is stripped before quoting, +never rendered raw. + +## Output contract + +- `claude_practice/output/us_/industry-report/__report.pdf` + +One file. Console output on run lists which sections were included vs. +skipped — check that before treating the PDF as complete; a section +missing because its data wasn't available is not a bug. + +## Known unverified data + +`trend`'s sign convention (positive = improved) has no field in the +payload that cross-checks it the way `ecr_records` confirmed the ECR +scale — the Data Confidence & Caveats page in every report says so +explicitly. Don't remove that page; it's the report's own disclosure, not +boilerplate. + +## After generating + +Report the output PDF path and the section include/skip list from the +script's stdout. Remind that `report_text`/`scentences` quotes are +RealRate's own generated text, not this skill's — attribute accordingly if +excerpted elsewhere. diff --git a/claude_headless/skills/mindmap.md b/claude_headless/skills/mindmap.md new file mode 100644 index 0000000..1172538 --- /dev/null +++ b/claude_headless/skills/mindmap.md @@ -0,0 +1,85 @@ +--- +description: Generate a RealRate company mindmap PNG — a radial diagram of one company's ECR, rank, strengths/weaknesses, and trend. Redesigned from skills/mindmap.md (real repo), which hardcodes 8 named companies and scrapes Wikipedia photos via Playwright; this version works for any company in the live archive using only verified data fields. +argument-hint: [industry_slug] [rank_or_company_name] [year (optional)] +--- + +# Practice Skill — Company Mindmap + +Generates a single 1920×1080 radial mindmap PNG for one company: a center +hub (name, ECR, rank) with branch circles for whatever data is actually +available — industry position, financial health, strengths/weaknesses +(quoted from RealRate's own `report_text`), multi-year trend, rating +status. + +Design context: `claude_practice/context/mindmap-design.md` — **read the +"Why redesigned" section before assuming this should look like the real +repo's mindmap skill.** It documents a confirmed regex bug this session +found and fixed (a naive strength/weakness extraction pattern locked onto +the wrong sentence and produced a paragraph-long "variable name" that +overflowed the canvas) — the fix and the defense-in-depth truncation are +both load-bearing, not incidental. + +## Arguments + +- `$ARGUMENTS` parses as: `{industry_slug} {rank_or_name} [year]` +- `industry_slug` (required) — e.g. `air` (also accepts `us_air` / `us-air`) +- `rank_or_name` (required) — rank number (`1` = industry leader), a + case-insensitive substring of the company name, or **`all`** to generate + one mindmap per company in the industry (loops internally; one + company's failure is logged and skipped, not fatal to the rest) +- `year` (optional) — overrides the balance-sheet-year URL fetched + +## Setup (first run only) + +Shares the venv with the other report skills: + +```bash +python3 -m venv "${CLAUDE_PROJECT_DIR}/claude_practice/.venv" +"${CLAUDE_PROJECT_DIR}/claude_practice/.venv/bin/pip" install --upgrade pip +"${CLAUDE_PROJECT_DIR}/claude_practice/.venv/bin/pip" install Pillow cairosvg +``` + +No Playwright, no browser install, no `matplotlib` — this skill only needs +what `top10-infographic` needs. + +## Generate + +```bash +cd "${CLAUDE_PROJECT_DIR}/claude_practice" +"${CLAUDE_PROJECT_DIR}/claude_practice/.venv/bin/python" \ + "${CLAUDE_PROJECT_DIR}/claude_practice/scripts/generate_mindmap.py" +``` + +Pass `--year ` only if `$ARGUMENTS` explicitly names one. If the +company name doesn't match, the script lists every ranked company in that +industry — re-run with the printed name, don't guess a company_id. Pass +`all` in place of `` to generate a mindmap for every company +in the industry in one run. + +## Branches are adaptive + +4–6 branches depending on what this company's data actually supports — +see the table in `context/mindmap-design.md`. Console output on run states +the branch count. A company with no multi-year `ecr_records` history or no +parseable `report_text` gets fewer branches, laid out evenly among however +many qualify — that's correct behavior, not a bug to force back to 6. + +## Output contract + +- `claude_practice/output/us_/mindmap/__mindmap.png` +- With `all`: one PNG per company, same folder, plus a final + `Done: N/M mindmaps generated` count on stdout — check that count before + assuming every company got one + +## Known limitation + +`report_text`'s strength/weakness sentence template was verified across +four companies in `us_air` — if a different industry's wording deviates +from that template, `parse_strength_weakness()` returns `None` for the +unmatched part and that branch is silently skipped (adaptive, not a +crash), but it's worth spot-checking a new industry's first run to confirm +the branches you expect actually appeared. + +## After generating + +Report the output PNG path and the branch count from stdout. diff --git a/claude_headless/skills/top10-infographic.md b/claude_headless/skills/top10-infographic.md new file mode 100644 index 0000000..9c820e0 --- /dev/null +++ b/claude_headless/skills/top10-infographic.md @@ -0,0 +1,83 @@ +--- +description: Generate a RealRate "Top 10" ranking LinkedIn infographic for an industry — a 1200x1500 PNG plus a matching caption .txt, pulled live from realrate-archive.com. Practice copy of skills/top10ranking.md. +argument-hint: [industry_slug] [year (optional)] +--- + +# Practice Skill — Top 10 Infographic + +Generates a 1200×1500px portrait LinkedIn infographic plus a matching caption +for a RealRate industry "Top 10" ranking, using live data from +realrate-archive.com. + +This is a **learning copy** — isolated in `claude_practice/`, does not touch +`skills/top10ranking.md` or `infographics/generate_infographic.py`. + +Design context: `claude_practice/context/infographic-design.md` — read it +before changing any color, size, or copy rule below. Never hardcode a value +into the script that isn't traceable to that file. + +**Unverified assumption** (see context file → "Known open question"): the +archive's `value` field is displayed directly as an ECR percentage. Sanity +check the numbers against realrate-archive.com before treating output as +publish-ready — this is a standing rule in the parent project's `CLAUDE.md`, +not optional here either. + +## Arguments + +- `$ARGUMENTS` parses as: `{industry_slug} [year]` +- `industry_slug` (required) — e.g. `air`, `software`, `hotels` +- `year` (optional) — if omitted, the script auto-detects the latest year + with published data, walking backwards from the current calendar year + +## Setup (first run only) + +Create a virtual environment scoped to this practice folder (skip if it +already exists): + +```bash +python3 -m venv "${CLAUDE_PROJECT_DIR}/claude_practice/.venv" +"${CLAUDE_PROJECT_DIR}/claude_practice/.venv/bin/pip" install --upgrade pip +"${CLAUDE_PROJECT_DIR}/claude_practice/.venv/bin/pip" install Pillow cairosvg +``` + +`Pillow` renders the canvas; `cairosvg` rasterizes the real logo SVG named in +`context/design-system.md` so the image uses the actual brand asset instead +of a hand-drawn approximation. The archive fetch uses Python's built-in +`urllib`, no `requests`/`playwright` needed. + +## Generate + +Always run with the practice venv's Python: + +```bash +cd "${CLAUDE_PROJECT_DIR}/claude_practice" +"${CLAUDE_PROJECT_DIR}/claude_practice/.venv/bin/python" \ + "${CLAUDE_PROJECT_DIR}/claude_practice/scripts/generate_infographic.py" +``` + +Pass `--year ` only if `$ARGUMENTS` explicitly names one. + +## Output contract + +Output is namespaced by industry, then by skill — `output/us_/top10-infographic/` +— so multiple skills can later write into the same industry folder without +colliding, matching `skills/auto-post-cycle.md`'s `output/{industry}/{task}/` +convention. Running the script above always produces both files together — +never treat one as done without the other: + +- `claude_practice/output/us_/top10-infographic/_.png` — the infographic +- `claude_practice/output/us_/top10-infographic/linkedin__.txt` — the matching caption + +## Available slugs + +`air` · `motor` · `software` · `computers` · `finance_services` · `food` · +`health_services` · `advertising` · `semiconductors` · `programming` · +`petrol` · `mining` · `construction` · `realestate` · `hotels` · +`consulting` · `data_processing` · `brokers` · `savings` · `life` · +`non_life` · `pharma` · `chemicals` · `state_banks` · +`medicinal_products` · `recreation` + +## After generating + +Report the output file paths, confirm both files were created, and remind +that the ECR numbers are unverified until checked against the archive. diff --git a/claude_headless/skills/top5-reveal-gif.md b/claude_headless/skills/top5-reveal-gif.md new file mode 100644 index 0000000..b198c8d --- /dev/null +++ b/claude_headless/skills/top5-reveal-gif.md @@ -0,0 +1,73 @@ +--- +description: Generate a RealRate Top 5 reveal GIF for an industry — an animated countdown-style reveal from rank 5 up to rank 1. Ported from "Industry Deep Dive Carousel/generate_cover_gif.py" (real repo), fixing its Windows-only font paths and its hand-edit-per-industry template into a CLI-driven, cross-platform skill. +argument-hint: [industry_slug] [year (optional)] +--- + +# Practice Skill — Top 5 Reveal GIF + +Generates a single animated GIF: ranks 5→1 reveal one at a time (suspense +builds toward the industry leader), holds on the fully-revealed frame, +then loops. + +Design source: this is a direct port of the real repo's cover-GIF +generator, not a from-scratch design like `mindmap` — the layout concept +(reveal order, hold-then-loop) is preserved. What changed is fixed, not +reimagined; see "What was fixed" below. + +## Arguments + +- `$ARGUMENTS` parses as: `{industry_slug} [year]` +- `industry_slug` (required) — e.g. `air` (also accepts `us_air` / `us-air`) +- `year` (optional) — overrides the balance-sheet-year URL fetched + +## Setup (first run only) + +Same venv as `top10-infographic` / `mindmap` — Pillow + cairosvg only, no +browser: + +```bash +python3 -m venv "${CLAUDE_PROJECT_DIR}/claude_practice/.venv" +"${CLAUDE_PROJECT_DIR}/claude_practice/.venv/bin/pip" install --upgrade pip +"${CLAUDE_PROJECT_DIR}/claude_practice/.venv/bin/pip" install Pillow cairosvg +``` + +## Generate + +```bash +cd "${CLAUDE_PROJECT_DIR}/claude_practice" +"${CLAUDE_PROJECT_DIR}/claude_practice/.venv/bin/python" \ + "${CLAUDE_PROJECT_DIR}/claude_practice/scripts/generate_top5_reveal_gif.py" +``` + +Pass `--year ` only if `$ARGUMENTS` explicitly names one. + +## What was fixed vs. the real repo's version + +1. **Font loading was Windows-only.** The original's `load_font()` only + tried `C:\Windows\Fonts\...` paths, silently degrading to Pillow's + bitmap default font everywhere else — invisible on Windows, ugly on + Linux/Mac. This version reuses the cross-platform fallback chain + already proven in `generate_infographic.py`. +2. **Hand-edited template → CLI-driven.** The original is meant to be + edited in place (every `# EDIT` constant) per industry and left + edited for the next run. This version takes the industry as an + argument like every other skill here — no manual editing required. +3. **Duplicate-frame "hold" doesn't work in Pillow.** An early version of + this script tried to hold the final frame by appending 4 duplicate + `Image` objects — Pillow's GIF writer collapsed them back down to one + frame even with `optimize=False` (verified: `im.n_frames` came back 5 + instead of the intended 9). Fixed by passing a **per-frame duration + list** instead (`[700, 700, 700, 700, 3000]`) — the correct way to make + one frame linger in a GIF. If you ever need a longer hold elsewhere in + this project, use a duration list, not repeated frames. + +## Output contract + +- `claude_practice/output/us_/top5-reveal-gif/__top5.gif` + +5 frames, 700ms per reveal step, 3000ms hold on the final frame, infinite +loop. + +## After generating + +Report the output GIF path and frame count from stdout.