diff --git a/results/metrics.json b/results/metrics.json index 78d0537..a945da9 100644 --- a/results/metrics.json +++ b/results/metrics.json @@ -1,5 +1,5 @@ { - "_about": "PennyCore \u00e2\u20ac\u201d append-only metrics journal. Each entry is a per-day snapshot; never overwrite past entries. Defined in SKILL Step 4 + the PRIMARY METRICS section. Initialized on Day 7 as a backfill (the journal was missed on Days 5 and 6 when the first measurable code shipped \u00e2\u20ac\u201d those days' numbers are NOT retroactively reconstructed; the day_05 and day_06 entries record only the binary 'feature shipped' fact so future readers know the journal started here). Comparison metrics (quality, cost, quality/$ ratios) start populating Day 12 when Phase 3 benchmarks land.", + "_about": "PennyCore — append-only metrics journal. Each entry is a per-day snapshot; never overwrite past entries. Defined in SKILL Step 4 + the PRIMARY METRICS section. Initialized on Day 7 as a backfill (the journal was missed on Days 5 and 6 when the first measurable code shipped — those days' numbers are NOT retroactively reconstructed; the day_05 and day_06 entries record only the binary 'feature shipped' fact so future readers know the journal started here). Comparison metrics (quality, cost, quality/$ ratios) start populating Day 12 when Phase 3 benchmarks land.", "schema_version": 1, "entries": [ { @@ -57,7 +57,7 @@ "brief_tokens": 1570, "naive_tokens": 1059, "ratio": 1.483, - "interpretation": "at budget=2000 with 50-message history, the recency brief is LARGER than naive concat because it adds channel markers, timestamps, and action segments that naive concat omits. The ratio<1 win the Phase 3 hybrid is supposed to deliver only kicks in when the borrower history exceeds the budget \u00e2\u20ac\u201d the 200-history-pair benchmark on Day 12 quantifies that." + "interpretation": "at budget=2000 with 50-message history, the recency brief is LARGER than naive concat because it adds channel markers, timestamps, and action segments that naive concat omits. The ratio<1 win the Phase 3 hybrid is supposed to deliver only kicks in when the borrower history exceeds the budget — the 200-history-pair benchmark on Day 12 quantifies that." }, "takehome_score": { "context_engine": { @@ -144,7 +144,7 @@ "event_id_published": "evt_963ff24cf2914a90b1318adf", "event_id_received": "evt_963ff24cf2914a90b1318adf", "round_trip_latency_observed_sec": "<1 (sleep 1 between publish and recent-read; not microbenchmarked today)", - "tenant_seeded": "INSERT INTO tenants (id, name, slug) VALUES ('acme', 'Acme Bank', 'acme-bank') \u00e2\u20ac\u201d required by Day-3 schema FK" + "tenant_seeded": "INSERT INTO tenants (id, name, slug) VALUES ('acme', 'Acme Bank', 'acme-bank') — required by Day-3 schema FK" }, "notes": "Day 8 task was orchestrator-side subscriber, but verification revealed the publish side was still hardcoded to InMemoryEventBus. Added context-engine REDIS_URL bus dispatcher (3 new tests) so the two halves actually share a wire. Live docker stack confirms publish-to-Redis -> subscribe-from-Redis -> /events/recent works end-to-end." }, @@ -174,13 +174,13 @@ "status_changed": "notify_loan_officer", "anomaly_detected": "notify_loan_officer", "system_event": "no_op", - "unknown": "notify_loan_officer (conservative default \u00e2\u20ac\u201d escalate to human)" + "unknown": "notify_loan_officer (conservative default — escalate to human)" }, "fallback_signal": "ActionProposal.proposed_by = ProposedBy.FALLBACK on any LLM failure (network, JSON parse, schema violation, unknown action_type)" }, "handler_chain": [ - "make_buffered_handler(_recent_events) \u00e2\u20ac\u201d Day 8", - "make_planning_handler(_recent_proposals) \u00e2\u20ac\u201d Day 9" + "make_buffered_handler(_recent_events) — Day 8", + "make_planning_handler(_recent_proposals) — Day 9" ], "takehome_score": { "context_engine": null, @@ -263,7 +263,7 @@ "crossover_observed": false } }, - "notes": "Day 9 planner is mock-mode-only in the live system (placeholder ANTHROPIC_API_KEY); LLM path covered by fake-client tests. Real-LLM exercise deferred to user-driven key setup or Phase 3 benchmark days. docker-compose up verification re-run after initial commit to close SKILL Step 4 'actually RUN the system' loop \u00e2\u20ac\u201d the bind-mounted orchestrator/ directory picks up planner.py without rebuild." + "notes": "Day 9 planner is mock-mode-only in the live system (placeholder ANTHROPIC_API_KEY); LLM path covered by fake-client tests. Real-LLM exercise deferred to user-driven key setup or Phase 3 benchmark days. docker-compose up verification re-run after initial commit to close SKILL Step 4 'actually RUN the system' loop — the bind-mounted orchestrator/ directory picks up planner.py without rebuild." }, { "day": 10, @@ -309,7 +309,7 @@ "execution", "execution_failed" ], - "idempotency_key": "(tenant_id, event_id) dedup index in pipeline; mirrors planned Postgres UNIQUE constraint per SYSTEM_DESIGN \u00a75.1", + "idempotency_key": "(tenant_id, event_id) dedup index in pipeline; mirrors planned Postgres UNIQUE constraint per SYSTEM_DESIGN §5.1", "failure_isolation": "executor exception -> EXECUTION_FAILED audit + status, no exception leaks to caller; planner LLM failure -> rule-table fallback (Day 9 behavior)" }, "takehome_score": { @@ -405,7 +405,7 @@ "isolation_verified": "acme customer_id (cust_4a3c7166...) disjoint from beta-bank customer_id (cust_92eee230...); action_id sets disjoint per-tenant", "idempotency_verified": "second POST with idempotency_key=live-jane-001 returned created=false deduped=true; orchestrator received 5 events not 6", "approve_path_verified": "POST /approvals/{id}/approve transitioned status to 'executed' and audit_trail captured [proposal, decision, approval, execution]", - "note": "live stack runs MOCK_LLM=true so planner emits rule-table-fallback proposals; the in-process E2E test exercises the same code paths against the same fallback path, so live-vs-test parity is intact. Live default-policy for acme is APPROVAL_REQUIRED (no docker-compose policy seeding shipped this phase \u2014 Phase 4 hardening adds a startup hook to load tenant policies from the policies table)" + "note": "live stack runs MOCK_LLM=true so planner emits rule-table-fallback proposals; the in-process E2E test exercises the same code paths against the same fallback path, so live-vs-test parity is intact. Live default-policy for acme is APPROVAL_REQUIRED (no docker-compose policy seeding shipped this phase — Phase 4 hardening adds a startup hook to load tenant policies from the policies table)" }, "takehome_score": { "context_engine": { @@ -446,9 +446,9 @@ "tests_full_suite_runtime_sec": 1.93, "takehome_context_engine": "5/6 (5/5 non-LLM)", "takehome_orchestrator": "6/6", - "docker_compose_boot": "verified \u2014 all 4 containers healthy", - "e2e_round_trip_in_process": "verified \u2014 3 tests green", - "e2e_round_trip_live_docker": "verified \u2014 Jane scenario end-to-end through Redis + Postgres" + "docker_compose_boot": "verified — all 4 containers healthy", + "e2e_round_trip_in_process": "verified — 3 tests green", + "e2e_round_trip_live_docker": "verified — Jane scenario end-to-end through Redis + Postgres" }, "champion_strategies_locked": { "ingestion": "FastAPI POST /events with X-Idempotency-Key header + (tenant_id, idempotency_key) UNIQUE oracle", @@ -465,7 +465,7 @@ }, "carries_to_phase_3": "Champion = (declarative policy + recency-only retrieval). Phase 3 Days 12-15 build the context-engine benchmark dataset (200 history+query pairs) and implement strategies 1-5 (Naive/Recency/Semantic/Summarized/Hybrid); Days 16-18 the orchestrator policy-engine comparison (Naive-LLM/YAML/Python-rules/LLM-judge). All four orchestrator engines drop into the existing PolicyEngine Protocol without touching the pipeline." }, - "notes": "Phase 2 wraps clean. The Day-11 deliverable is a single in-process E2E test module (3 tests) + a one-line fix to the takehome runner + the canonical scorecard re-run. NO new production code today by design \u2014 the phase wrap day is for locking in numbers, not introducing risk. Full suite 326/326 green (up from 323 Day-10) confirms no regressions. Live docker-compose verification adds a redundant signal that the in-process test isn't lying about what production does." + "notes": "Phase 2 wraps clean. The Day-11 deliverable is a single in-process E2E test module (3 tests) + a one-line fix to the takehome runner + the canonical scorecard re-run. NO new production code today by design — the phase wrap day is for locking in numbers, not introducing risk. Full suite 326/326 green (up from 323 Day-10) confirms no regressions. Live docker-compose verification adds a redundant signal that the in-process test isn't lying about what production does." }, { "day": 11, @@ -484,7 +484,7 @@ "results/phase2_latency_backfill.json" ], "decision_pipeline_latency_ms": { - "what": "DecisionPipeline.handle_proposal() \u2014 synthetic proposal through policy(auto) -> action-store put -> mock executor -> audit writes", + "what": "DecisionPipeline.handle_proposal() — synthetic proposal through policy(auto) -> action-store put -> mock executor -> audit writes", "samples": 500, "warmup_excluded": 50, "policy_path": "auto (longest path: PROPOSAL + DECISION + EXECUTION audit rows)", @@ -496,14 +496,14 @@ "max": 0.6666, "mean": 0.0688, "stdev": 0.0441, - "interpretation": "Sub-millisecond at p99 \u2014 pure-Python dict mutations + in-memory audit appends. The 'max' tail (0.67 ms) is the worst single-sample outlier; p99 is what production SLOs should target." + "interpretation": "Sub-millisecond at p99 — pure-Python dict mutations + in-memory audit appends. The 'max' tail (0.67 ms) is the worst single-sample outlier; p99 is what production SLOs should target." }, "end_to_end_latency_ms": { "what": "POST /events (context-engine) -> InMemoryEventBus -> orchestrator listener -> chain_handlers (buffer + planner + pipeline) -> audit writes. All in-process so the HTTP response time IS the end-to-end latency.", "samples": 500, "warmup_excluded": 50, "transport": "InMemoryEventBus (synchronous subscriber)", - "policy_path": "auto (every event auto-executes \u2014 longest pipeline path)", + "policy_path": "auto (every event auto-executes — longest pipeline path)", "llm_mode": "mock (rule-table fallback proposals)", "min": 1.5314, "p50": 1.7839, @@ -521,7 +521,7 @@ "production_caveat": "in-process bus is a LOWER BOUND on production E2E. Production Redis Pub/Sub adds wire latency + worker scheduling; AND the POST response returns before the orchestrator processes the event (fire-and-forget publish). Real production E2E lands via OTel spans in Phase 6 Day 30." }, "api_docs": { - "what": "docs/API.md \u2014 first comprehensive API reference for both services. Captures every endpoint stable through Day 11 (context-engine: /, /healthz, /readyz, POST /events; orchestrator: /, /healthz, /readyz, /events/recent, /proposals/recent, /approvals, POST /approvals/{id}/approve, POST /approvals/{id}/reject, /actions/{id}) plus the action-type enum, policy-decision enum, audit-kind taxonomy, and the Phase 2 latency budget table.", + "what": "docs/API.md — first comprehensive API reference for both services. Captures every endpoint stable through Day 11 (context-engine: /, /healthz, /readyz, POST /events; orchestrator: /, /healthz, /readyz, /events/recent, /proposals/recent, /approvals, POST /approvals/{id}/approve, POST /approvals/{id}/reject, /actions/{id}) plus the action-type enum, policy-decision enum, audit-kind taxonomy, and the Phase 2 latency budget table.", "rationale": "Listed in the SKILL project folder structure but no specific day in the Day-by-Day table created it. Backfilling now lets the Phase 3 surface (which adds /benchmarks/* and /policy_engines/*) extend a documented baseline rather than build a doc from scratch." }, "tests_passing": 326, @@ -618,7 +618,7 @@ "budget_saturation_pct": 98.6 } }, - "headline": "Recency baseline saturates the 8K budget at 99% on very_long histories (avg 7886 / 8000 tokens). That's the slice where smarter strategies (semantic, summarized, hybrid) need to win on quality/cost \u2014 the room to beat the baseline lives at the long tail." + "headline": "Recency baseline saturates the 8K budget at 99% on very_long histories (avg 7886 / 8000 tokens). That's the slice where smarter strategies (semantic, summarized, hybrid) need to win on quality/cost — the room to beat the baseline lives at the long tail." }, "harness": { "registered_strategies_today": [ @@ -636,7 +636,7 @@ }, "tests_passing": 341, "tests_added_today": 15, - "tests_added_reason": "test_phase3_dataset.py \u2014 manifest invariants, loader contracts, harness boot path, recency-on-5-pairs smoke. Pre-Day-12 baseline was 326 (Day-11 metrics); 326 + 15 = 341.", + "tests_added_reason": "test_phase3_dataset.py — manifest invariants, loader contracts, harness boot path, recency-on-5-pairs smoke. Pre-Day-12 baseline was 326 (Day-11 metrics); 326 + 15 = 341.", "tests_skipped": 10, "tests_skipped_reason": "DATABASE_URL unset for PG-integration tests; gating unchanged from prior days", "notes": "Phase 3 PR opens today (phase/3-comparison-studies). Day 13 wires Strategy 1 (naive dump-everything) + Strategy 2 (recency-only is already wired as the harness baseline; Day 13 promotes it from baseline to first-class strategy in the comparison table). Day 14 adds semantic + summarized; Day 15 adds hybrid + LLM-as-judge quality scoring across all 5. Dataset is reproducible: re-run benchmarks/data/build_dataset.py --seed 42 to regenerate." @@ -658,7 +658,7 @@ "results/EXPERIMENT_LOG.md" ], "modules_modified": [ - "benchmarks/context_engine_bench.py \u2014 registry split out; per-strategy budget override; per-bucket summary" + "benchmarks/context_engine_bench.py — registry split out; per-strategy budget override; per-bucket summary" ], "tests_passing": 356, "tests_added_today": 15, @@ -779,7 +779,7 @@ } }, "results_added": [ - "results/phase3_context_engine_results.json (overwritten \u2014 now contains both strategies' 200 pairs each = 400 result rows)" + "results/phase3_context_engine_results.json (overwritten — now contains both strategies' 200 pairs each = 400 result rows)" ], "notes": "Day 13 punch line: naive vs recency are token-equal on short/medium/long because both fit under 8K; the only place they diverge is the very_long bucket, where recency saturates 98.6% of its budget and drops ~558 tokens per pair on average. That is the headroom Day 14 semantic + summarized strategies will fight to claim. Counterintuitive: the headline cost contrast is 2%, not 6x, because the dataset distribution matches real customer-service traffic (most histories are short)." }, @@ -801,7 +801,7 @@ "results/phase3_day14_analysis.json" ], "modules_modified": [ - "benchmarks/strategies/__init__.py \u2014 register semantic + summarized" + "benchmarks/strategies/__init__.py — register semantic + summarized" ], "tests_passing": 392, "tests_added_today": 36, @@ -809,7 +809,7 @@ "tests_skipped": 10, "tests_skipped_reason": "Postgres integration gated on DATABASE_URL", "comparison_run": { - "dataset": "benchmarks/data/manifest.json \u2014 200 pairs (100 short / 60 medium / 30 long / 10 very_long)", + "dataset": "benchmarks/data/manifest.json — 200 pairs (100 short / 60 medium / 30 long / 10 very_long)", "llm_mode": "mock (deterministic; LLM-as-judge quality scoring lands Day 15)", "strategies": [ "naive_dump", @@ -900,7 +900,7 @@ "note_on_mock_mode_summary": "Summarized average of 211 tokens is inflated by mock-mode summaries (200-char head + suffix). A real LLM (Day 15 re-judge) is expected to produce 600-1000-token summaries, narrowing the ratio vs recency from 7x to ~2-3x. The pattern (compression > saturation) holds either way." }, "results_added": [ - "results/phase3_context_engine_results.json (overwritten \u2014 now 800 rows: 200 pairs x 4 strategies, full brief text preserved for Day-15 re-judge)", + "results/phase3_context_engine_results.json (overwritten — now 800 rows: 200 pairs x 4 strategies, full brief text preserved for Day-15 re-judge)", "results/phase3_day14_analysis.json (per-pair Jaccard + tokens-saved breakdown)" ], "notes": "Day 14 punch line #1: semantic and recency are byte-identical on 190/200 pairs (short/medium/long) because everything fits under 8K - the only place the ranking signal matters is the 10 very_long pairs where the budget saturates. Day 14 punch line #2: summarized averages 211 tokens vs recencys 1421 - 7x compression - but the mock-mode 200-char summary inflates that ratio; the Day-15 re-judge with a real LLM will pin the realistic number. Day 14 punch line #3: semantic costs ~3.4x more retrieval latency than recency (0.8 vs 0.24 ms p50) - the embed+cosine per segment is cheap but adds up. All four strategies stay well under the 10ms target end-to-end retrieval budget." @@ -927,10 +927,10 @@ "results/phase3_day15_analysis.json" ], "modules_modified": [ - "benchmarks/strategies/__init__.py \u2014 register hybrid" + "benchmarks/strategies/__init__.py — register hybrid" ], "comparison_run": { - "dataset": "benchmarks/data/manifest.json \u2014 200 pairs (100 short / 60 medium / 30 long / 10 very_long)", + "dataset": "benchmarks/data/manifest.json — 200 pairs (100 short / 60 medium / 30 long / 10 very_long)", "llm_mode": "mock (retrieval + judge); judge_mode = mock_proxy", "strategies": [ "naive_dump", @@ -1019,7 +1019,7 @@ "judge_proxy_bias_note": "Fact-recall proxy is biased toward strategies that preserve verbatim text. Compression-based strategies (summarized, hybrid cold-tail) score lower under this proxy than they would under a real LLM-as-judge that rewards coherent summarization. Real-LLM re-run (Day 18 / Phase 5) is the canonical Quality column." }, "results_added": [ - "results/phase3_context_engine_results.json (overwritten \u2014 now 1000 rows: 200 pairs x 5 strategies, every row carries quality_score / query_recall / ground_truth_recall / judge_mode / judge_notes)", + "results/phase3_context_engine_results.json (overwritten — now 1000 rows: 200 pairs x 5 strategies, every row carries quality_score / query_recall / ground_truth_recall / judge_mode / judge_notes)", "results/phase3_day15_analysis.json (per-strategy aggregates + hybrid divergence buckets + proxy bias note)" ], "notes": "Day 15 punch line #1: hybrid emits 41% smaller briefs than recency on aggregate (838 vs 1422 tokens) at the same 8K budget, while matching recency's mock-mode quality on 183/200 pairs (91.5%). Day 15 punch line #2: under the mock-mode token-recall proxy, naive/recency/semantic produce identical mean quality (3.035) because their briefs are byte-identical on 95% of pairs - the proxy correctly identifies them as indistinguishable. Day 15 punch line #3: the mock-mode quality proxy is biased toward verbatim-preserving strategies. Real LLM-as-judge re-run (Day 18 with LLM_PROVIDER=anthropic) is the canonical Quality column; today's numbers are the proxy floor that compression strategies are expected to improve upon under a real LLM." @@ -1083,9 +1083,9 @@ "benchmarks/data/orchestrator/tenant_policies.json", "benchmarks/data/orchestrator/manifest.json" ], - "multi_tenant_divergence": "verified \u2014 same action under different tenants yields different decisions (e.g. send_borrower_message=auto under Acme/Globetrek vs approval_required under Jefferson; schedule_call=approval_required under Acme/Globetrek vs reject under Jefferson)" + "multi_tenant_divergence": "verified — same action under different tenants yields different decisions (e.g. send_borrower_message=auto under Acme/Globetrek vs approval_required under Jefferson; schedule_call=approval_required under Acme/Globetrek vs reject under Jefferson)" }, - "notes": "Day-16 ships the orchestrator benchmark dataset only. No engines have been run against it yet \u2014 Day 17 implements + scores the four policy-engine variants (naive LLM / declarative / Python rules / LLM-as-judge) against expected_action_type + expected_decision." + "notes": "Day-16 ships the orchestrator benchmark dataset only. No engines have been run against it yet — Day 17 implements + scores the four policy-engine variants (naive LLM / declarative / Python rules / LLM-as-judge) against expected_action_type + expected_decision." }, { "day": 17, @@ -1168,7 +1168,7 @@ "tests_skipped": 10, "tests_skipped_reason": "DATABASE_URL unset; PG-integration tests gated", "takehome_score": null, - "notes": "Phase-3 wrap. Consolidated Day-15 context-engine and Day-17 orchestrator artifacts into one view at results/phase3_day18_consolidated.json. Named champions: hybrid (context-engine, under mock-proxy caveat \u2014 locked subject to Phase-5/Day-27 real-LLM re-judge) and declarative (orchestrator, outright). 9 strategies x 400 scenarios head-to-head. No new benchmarks or production code today; consolidation only.", + "notes": "Phase-3 wrap. Consolidated Day-15 context-engine and Day-17 orchestrator artifacts into one view at results/phase3_day18_consolidated.json. Named champions: hybrid (context-engine, under mock-proxy caveat — locked subject to Phase-5/Day-27 real-LLM re-judge) and declarative (orchestrator, outright). 9 strategies x 400 scenarios head-to-head. No new benchmarks or production code today; consolidation only.", "phase3_wrap": { "artifacts": [ "results/phase3_day18_consolidated.json", @@ -1183,8 +1183,8 @@ "orchestrator": "declarative" }, "champion_caveats": { - "context_engine": "Locked subject to Phase-5/Day-27 real-LLM re-judge; mock-proxy mean quality 2.910 vs recency 3.035 \u2014 wins on quality-per-1K-tokens (3.47 vs 2.13) and emits 41% fewer brief tokens at parity fact recall on 91.5% of 200 pairs.", - "orchestrator": "None \u2014 wins on every measured axis (correctness 1.000, p50 0.6 us, $0/100 dec, 5/5 audit + maint)." + "context_engine": "Locked subject to Phase-5/Day-27 real-LLM re-judge; mock-proxy mean quality 2.910 vs recency 3.035 — wins on quality-per-1K-tokens (3.47 vs 2.13) and emits 41% fewer brief tokens at parity fact recall on 91.5% of 200 pairs.", + "orchestrator": "None — wins on every measured axis (correctness 1.000, p50 0.6 us, $0/100 dec, 5/5 audit + maint)." }, "strategies_compared_total": 9, "scenarios_total": 400, @@ -1238,7 +1238,7 @@ "tests_skipped": 10, "tests_skipped_reason": "DATABASE_URL unset; PG-integration tests gated", "takehome_score": null, - "notes": "15-test multi-tenant isolation suite + minimal production patches for two real HTTP leak surfaces. GET /actions/{id} and POST /approvals/{id}/{approve,reject} now accept an optional tenant_id query parameter that 404s cross-tenant attempts (same-shape error \u2014 no existence leak per OWASP API1). audit.entries_for_action/event gain optional tenant_id= filter; pipeline.get_action/approve_action/reject_action gain optional expected_tenant_id= kwarg. Back-compat preserved for the takehome adapter (uses no tenant_id; single-tenant per scenario). 533 passed, 0 failed, zero regression on top of Day 19 518-test baseline.", + "notes": "15-test multi-tenant isolation suite + minimal production patches for two real HTTP leak surfaces. GET /actions/{id} and POST /approvals/{id}/{approve,reject} now accept an optional tenant_id query parameter that 404s cross-tenant attempts (same-shape error — no existence leak per OWASP API1). audit.entries_for_action/event gain optional tenant_id= filter; pipeline.get_action/approve_action/reject_action gain optional expected_tenant_id= kwarg. Back-compat preserved for the takehome adapter (uses no tenant_id; single-tenant per scenario). 533 passed, 0 failed, zero regression on top of Day 19 518-test baseline.", "tenant_isolation_hardening": { "test_classes": 6, "tests_total": 15, @@ -1257,7 +1257,7 @@ "orchestrator/api.py": "optional tenant_id query parameter on /actions/{id}, /approvals/{id}/approve, /approvals/{id}/reject" }, "cross_tenant_response_code": 404, - "cross_tenant_response_rationale": "OWASP API1:2023 \u2014 Broken Object Level Authorization \u2014 404 reveals nothing about cross-tenant existence", + "cross_tenant_response_rationale": "OWASP API1:2023 — Broken Object Level Authorization — 404 reveals nothing about cross-tenant existence", "coverage_isolation_modules": { "orchestrator/audit.py": 0.89, "orchestrator/decision_pipeline.py": 0.79, @@ -1281,13 +1281,13 @@ "tests_skipped": 10, "tests_skipped_reason": "DATABASE_URL unset; PG-integration tests gated", "takehome_score": null, - "notes": "10-test integration suite in tests/integration/test_race_conditions.py exercises every concurrent surface on PennyCore's hot paths. Nine of ten races were already safe by construction (Phase-2 RLock + unique-tuple dedup discipline). The tenth \u2014 context_engine/linking.py check-then-create TOCTOU \u2014 was a real bug invisible to vanilla threaded tests because the GIL serializes the tight find-then-create window. Surfaced via _wrap_with_lookup_latency (10ms sleep injected into find_customer_by_identity, simulating Postgres-adapter round-trip): without the production patch, 7 of 8 concurrent callers raise IdentityCollision. Patched resolve_customer to catch IdentityCollision and re-walk the identity priority list, returning the winner's customer as a regular match. Catch-and-retry pattern (Kleppmann DDIA \u00a77.2.3) preserves happy-path performance \u2014 only race losers pay the extra lookup. Stash-and-rerun confirms tests 1-2 fail without the patch, proving the regression guard exercises the right path.", + "notes": "10-test integration suite in tests/integration/test_race_conditions.py exercises every concurrent surface on PennyCore's hot paths. Nine of ten races were already safe by construction (Phase-2 RLock + unique-tuple dedup discipline). The tenth — context_engine/linking.py check-then-create TOCTOU — was a real bug invisible to vanilla threaded tests because the GIL serializes the tight find-then-create window. Surfaced via _wrap_with_lookup_latency (10ms sleep injected into find_customer_by_identity, simulating Postgres-adapter round-trip): without the production patch, 7 of 8 concurrent callers raise IdentityCollision. Patched resolve_customer to catch IdentityCollision and re-walk the identity priority list, returning the winner's customer as a regular match. Catch-and-retry pattern (Kleppmann DDIA §7.2.3) preserves happy-path performance — only race losers pay the extra lookup. Stash-and-rerun confirms tests 1-2 fail without the patch, proving the regression guard exercises the right path.", "race_condition_hardening": { "test_classes": 7, "tests_total": 10, "tests_passed_first_run": 10, "production_patches": { - "context_engine/linking.py": "resolve_customer step 3 catches IdentityCollision from create_customer, re-walks identity priority list, returns winner's customer as regular match (customer_created=False). Pure-additive \u2014 no signature change, no caller change." + "context_engine/linking.py": "resolve_customer step 3 catches IdentityCollision from create_customer, re-walks identity priority list, returns winner's customer as regular match (customer_created=False). Pure-additive — no signature change, no caller change." }, "races_already_safe": [ "InMemoryApprovalQueue.enqueue under N-thread contention (idempotent on action_id)", @@ -1338,8 +1338,8 @@ } }, "back_compat_preserved": true, - "race_resolution_pattern": "catch-and-retry on UNIQUE-violation (Kleppmann DDIA \u00a77.2.3); locks unchanged", - "regression_path_verification": "tests 1-2 verified to exercise the regression path via `git stash` of context_engine/linking.py \u2014 without the patch, tests 1-2 fail with 7-of-8 callers raising IdentityCollision; restoring the patch returns the green. With the patch in place, all 10 tests pass on first run.", + "race_resolution_pattern": "catch-and-retry on UNIQUE-violation (Kleppmann DDIA §7.2.3); locks unchanged", + "regression_path_verification": "tests 1-2 verified to exercise the regression path via `git stash` of context_engine/linking.py — without the patch, tests 1-2 fail with 7-of-8 callers raising IdentityCollision; restoring the patch returns the green. With the patch in place, all 10 tests pass on first run.", "coverage_concurrency_modules_measurement": "pytest --cov, end-of-day, post-patch; reports percent + (statements, missing)" }, "end_of_day_stack_boot": { @@ -1373,7 +1373,7 @@ "tests_skipped": 10, "tests_skipped_reason": "DATABASE_URL unset; PG-integration tests gated", "takehome_score": null, - "notes": "Load test harness (locust file + stdlib twin) ships. SKILL-prescribed workload (100 concurrent customers across 5 tenants, mock LLM, in-memory repo) runs cleanly: 5954 requests over 30.45s, 100% success, p50 313ms, p95 688ms, p99 853ms. Per-tenant variance under 1% (no tenant skew under load \u2014 validates Day 20 isolation work). Scaling sweep at 1/10/25/50/100 users locates the knee between 25 and 50 concurrent users; throughput peaks at 258 rps and DECREASES to 186 rps at 100 users (classic Little's-Law thrashing). Sequential cProfile of the per-request hot path identifies the bottleneck: 4.08 ms/req in starlette.testclient.handle_request (anyio sync\u2192async bridge), with PennyCore handler code (validation \u2192 linker \u2192 repo upsert \u2192 bus publish) sub-millisecond. The bottleneck is the single-process / single-event-loop ceiling \u2014 characteristic of every single-Uvicorn-worker FastAPI app, not specific to PennyCore. Day 29 production dockerization will fix this with multi-worker Uvicorn topology (one worker per 25-50 concurrent customers expected).", + "notes": "Load test harness (locust file + stdlib twin) ships. SKILL-prescribed workload (100 concurrent customers across 5 tenants, mock LLM, in-memory repo) runs cleanly: 5954 requests over 30.45s, 100% success, p50 313ms, p95 688ms, p99 853ms. Per-tenant variance under 1% (no tenant skew under load — validates Day 20 isolation work). Scaling sweep at 1/10/25/50/100 users locates the knee between 25 and 50 concurrent users; throughput peaks at 258 rps and DECREASES to 186 rps at 100 users (classic Little's-Law thrashing). Sequential cProfile of the per-request hot path identifies the bottleneck: 4.08 ms/req in starlette.testclient.handle_request (anyio sync→async bridge), with PennyCore handler code (validation → linker → repo upsert → bus publish) sub-millisecond. The bottleneck is the single-process / single-event-loop ceiling — characteristic of every single-Uvicorn-worker FastAPI app, not specific to PennyCore. Day 29 production dockerization will fix this with multi-worker Uvicorn topology (one worker per 25-50 concurrent customers expected).", "load_test": { "workload": { "transport": "fastapi.testclient (in-process)", @@ -1456,7 +1456,7 @@ "knee_concurrency_users": "25-50", "throughput_ceiling_rps": 258, "bottleneck": { - "location": "starlette.testclient anyio sync\u2192async bridge + Python GIL", + "location": "starlette.testclient anyio sync→async bridge + Python GIL", "evidence": "sequential cProfile: 4.08 ms/req in handle_request; PennyCore handler code below noise floor", "fix_path": "Day 29 production dockerization with uvicorn --workers N (one worker per 25-50 concurrent customers)", "is_pennycore_code": false @@ -1479,7 +1479,7 @@ "locust_install_required": false, "transports_supported": [ "fastapi.testclient (default)", - "httpx \u2192 live Uvicorn (--base-url)" + "httpx → live Uvicorn (--base-url)" ] } }, @@ -1511,7 +1511,7 @@ } }, "torn_down_after_verification": true, - "verification_note": "Day 22 touched only benchmarks/ + tests/ \u2014 no service or schema changes; this boot reproves the Day-21 green stack remains green." + "verification_note": "Day 22 touched only benchmarks/ + tests/ — no service or schema changes; this boot reproves the Day-21 green stack remains green." } }, { @@ -1611,12 +1611,12 @@ "phase": 5, "date": "2026-05-27", "component": "context-engine", - "feature": "two-stage retrieval \u2014 cross-encoder-style re-ranker", + "feature": "two-stage retrieval — cross-encoder-style re-ranker", "feature_shipped": true, "tests_passing": 681, "tests_added_today": 28, "tests_skipped": 10, - "tests_skipped_reason": "Postgres integration tests require DATABASE_URL \u2014 unchanged from Day 23 baseline.", + "tests_skipped_reason": "Postgres integration tests require DATABASE_URL — unchanged from Day 23 baseline.", "coverage_touched_modules_pct": "rerank.py 100% (28 tests cover every branch); semantic.py +1 line covered (timestamp metadata)", "benchmark_run": true, "benchmark_dataset_size": 200, @@ -1644,7 +1644,7 @@ "judge_mode": "mock_proxy", "phase5_pr_opened": true, "phase5_branch": "phase/5-naive-baseline", - "notes": "Re-rank rewrites brief ordering on 145/200 pairs (73%) but mock-judge proxy is recall-based and order-blind \u2014 same quality histograms. Honest negative result; Day 28 real-LLM re-judge may reveal an ordering quality signal the proxy cannot see.", + "notes": "Re-rank rewrites brief ordering on 145/200 pairs (73%) but mock-judge proxy is recall-based and order-blind — same quality histograms. Honest negative result; Day 28 real-LLM re-judge may reveal an ordering quality signal the proxy cannot see.", "system_boot": "pytest sweep green; benchmark harness runs end-to-end on the 200-pair dataset; no docker boot today (no new service or schema change)." }, { @@ -1756,13 +1756,13 @@ "phase": 5, "date": "2026-05-30", "component": "context-engine", - "feature": "frontier baseline comparison part 1 \u2014 naive vs hybrid champion with USD cost + per-bucket frontier", + "feature": "frontier baseline comparison part 1 — naive vs hybrid champion with USD cost + per-bucket frontier", "feature_shipped": true, "tests_passing": 758, "tests_added_today": 8, "tests_skipped": 10, - "tests_skipped_reason": "Postgres integration tests require DATABASE_URL \u2014 unchanged from Day 26 baseline.", - "llm_mode": "mock (Anthropic key 401, Kimi-K2.6 reasoning-content quirk \u2014 bench fell back to mock-proxy judge with projected claude-sonnet-4-6 pricing for cost)", + "tests_skipped_reason": "Postgres integration tests require DATABASE_URL — unchanged from Day 26 baseline.", + "llm_mode": "mock (Anthropic key 401, Kimi-K2.6 reasoning-content quirk — bench fell back to mock-proxy judge with projected claude-sonnet-4-6 pricing for cost)", "modules_added": [ "benchmarks/phase5_naive_vs_champion.py", "benchmarks/phase5_naive_vs_champion_chart.py", @@ -1801,13 +1801,13 @@ } }, "key_findings": [ - "Hybrid is 4x cheaper than naive on the very_long bucket at identical fact-slice quality \u2014 the cost win compounds with history length.", + "Hybrid is 4x cheaper than naive on the very_long bucket at identical fact-slice quality — the cost win compounds with history length.", "Aggregate 24% cheaper number understates the production story; per-bucket framing is the correct one.", "Recency-only looks competitive aggregate (-1.1%) but on long histories it becomes naive-with-extra-steps.", "Cost-model assumptions (pricing row, system-prompt tokens, response tokens) pinned in JSON header for reproducibility." ], "phase5_pr": "https://github.com/Mark007-R/PennyCore/pull/", - "notes": "Closes the Day-18 'Real-LLM re-run scheduled for Day 27' caveat with projected costs + per-bucket breakdown. Real-LLM judge blocked by environment auth \u2014 see Day-27 report \u00a7Failures for full rationale.", + "notes": "Closes the Day-18 'Real-LLM re-run scheduled for Day 27' caveat with projected costs + per-bucket breakdown. Real-LLM judge blocked by environment auth — see Day-27 report §Failures for full rationale.", "system_boot": "pytest 758 passed / 10 skipped (Postgres-gated); bench + chart scripts run end-to-end." }, { @@ -1815,13 +1815,13 @@ "phase": 5, "date": "2026-05-31", "component": "orchestrator (+ Phase 5 wrap)", - "feature": "frontier baseline comparison part 2 \u2014 naive LLM vs declarative champion with prod-payload USD cost + per-tenant failure shape; Phase 5 wrap-up", + "feature": "frontier baseline comparison part 2 — naive LLM vs declarative champion with prod-payload USD cost + per-tenant failure shape; Phase 5 wrap-up", "feature_shipped": true, "tests_passing": 766, "tests_added_today": 8, "tests_skipped": 10, - "tests_skipped_reason": "Postgres integration tests require DATABASE_URL \u2014 unchanged from Day 27 baseline.", - "llm_mode": "mock (forced \u2014 naive engine's mock-mode tenant-agnostic heuristic is the documented baseline; cost projected at claude-sonnet-4-6 prod-payload rates)", + "tests_skipped_reason": "Postgres integration tests require DATABASE_URL — unchanged from Day 27 baseline.", + "llm_mode": "mock (forced — naive engine's mock-mode tenant-agnostic heuristic is the documented baseline; cost projected at claude-sonnet-4-6 prod-payload rates)", "modules_added": [ "benchmarks/phase5_orch_naive_vs_champion.py", "benchmarks/phase5_orch_naive_vs_champion_chart.py", @@ -1893,10 +1893,10 @@ } }, "key_findings": [ - "Declarative is a strict Pareto winner over naive \u2014 higher correctness AND lower latency AND lower cost AND better auditability.", - "Naive's failure mode is tenant-asymmetric in OPPOSITE directions \u2014 over-approves on the strict tenant (security bug), over-rejects on the permissive ones (UX bug). A tenant-agnostic heuristic cannot satisfy tenants with opposite rules by construction.", + "Declarative is a strict Pareto winner over naive — higher correctness AND lower latency AND lower cost AND better auditability.", + "Naive's failure mode is tenant-asymmetric in OPPOSITE directions — over-approves on the strict tenant (security bug), over-rejects on the permissive ones (UX bug). A tenant-agnostic heuristic cannot satisfy tenants with opposite rules by construction.", "Production-payload cost projection widens naive's cost disadvantage 4x vs the mock-payload Day-17 numbers (~$0.14 -> ~$0.57 per 100 decisions).", - "LLM-as-judge with structured output ties declarative on correctness in mock mode but trades cost + rubric for the LLM in the loop \u2014 right LLM design point, wrong policy-engine design point." + "LLM-as-judge with structured output ties declarative on correctness in mock mode but trades cost + rubric for the LLM in the loop — right LLM design point, wrong policy-engine design point." ], "phase5_wrap_up": { "champions_locked": { @@ -1912,8 +1912,90 @@ ] }, "phase5_pr": "merged via squash on Day 28", - "notes": "Closes Phase 5. Champions locked for Phase 6 (Production Polish, Days 29-32). Same auth/Kimi constraints as Day 27 \u2014 real-LLM cost validation deferred but methodology and assumptions pinned in JSON headers for clean re-run.", + "notes": "Closes Phase 5. Champions locked for Phase 6 (Production Polish, Days 29-32). Same auth/Kimi constraints as Day 27 — real-LLM cost validation deferred but methodology and assumptions pinned in JSON headers for clean re-run.", "system_boot": "pytest 766 passed / 10 skipped (Postgres-gated); bench + chart scripts run end-to-end." + }, + { + "day": 29, + "phase": 6, + "date": "2026-06-01", + "component": "deploy", + "feature": "production_docker_plus_fly_deploy_script_plus_info_endpoint", + "feature_shipped": true, + "tests_passing": 796, + "tests_added_today": 30, + "tests_skipped": 10, + "tests_skipped_reason": "DATABASE_URL unset; PG-integration tests gated", + "takehome_score": { + "context_engine": "5/5 (unchanged — last measured Day 26)", + "orchestrator": "6/6 (unchanged — last measured Day 26)" + }, + "notes": "Production polish day, not a comparison day; no latency / cost numbers. Dockerfile.prod (gunicorn + tini + frozen source + OCI labels), docker-compose.prod.yml (no bind-mounts, POSTGRES_PASSWORD:? fail-fast, per-service worker counts), scripts/deploy_fly.sh (six-step dry-run-default, --execute opt-in, --skip-to resumption), /info endpoint surfaces build SHA / date / version on both services. 30 invariant tests guard Dockerfile shape + compose env-template drift + deploy script step contract + .gitignore/.dockerignore for .env.prod." + }, + { + "day": 30, + "phase": 6, + "date": "2026-06-02", + "component": "observability", + "feature": "opentelemetry_spine_instrumentation_plus_otel_compose_overlay", + "feature_shipped": true, + "tests_passing": 813, + "tests_added_today": 17, + "tests_skipped": 10, + "tests_skipped_reason": "DATABASE_URL unset; PG-integration tests gated", + "takehome_score": { + "context_engine": "5/5 (unchanged — last measured Day 26)", + "orchestrator": "6/6 (unchanged — last measured Day 26)" + }, + "notes": "contracts/observability.py exposes trace_span + setup_tracing; three load-bearing spans (pennycore.ingest_event, pennycore.decision.handle_proposal, pennycore.executor.execute) each carry pennycore.tenant_id as first-class attribute. docker-compose.otel.yml overlay (otel-collector-contrib 0.110 + Jaeger all-in-one 1.62) flips spans from no-op to OTLP/HTTP with one env var. /info surfaces tracing_mode. results/samples/day30_phase6_jaeger_trace.png rendered from real captured spans." + }, + { + "day": 31, + "phase": 6, + "date": "2026-06-03", + "component": "ui", + "feature": "streamlit_approver_dashboard_plus_data_adapter", + "feature_shipped": true, + "tests_passing": 834, + "tests_added_today": 21, + "tests_skipped": 10, + "tests_skipped_reason": "DATABASE_URL unset; PG-integration tests gated", + "takehome_score": { + "context_engine": "5/5 (unchanged — last measured Day 26)", + "orchestrator": "6/6 (unchanged — last measured Day 26)" + }, + "notes": "ui/approver_app.py Streamlit dashboard reading from ui/approver_data.py adapter. Three panels: pending-queue with N-of-M quorum progress bars (Day-26 data shape), per-action detail with embedded audit trail, recent audit log. Single ApproverActionError(category=...) surface translates five typed pipeline exceptions. seed_demo_data populates three tenants. 19 adapter tests + 2 Streamlit AppTest smoke. Snapshot at results/samples/day31_phase6_approver_dashboard.png rendered from live adapter." + }, + { + "day": 32, + "phase": 6, + "date": "2026-06-04", + "component": "ui", + "feature": "streamlit_demo_scenario_ui_plus_phase_6_wrap", + "feature_shipped": true, + "phase_wrap": true, + "tests_passing": 845, + "tests_added_today": 11, + "tests_skipped": 10, + "tests_skipped_reason": "DATABASE_URL unset; PG-integration tests gated", + "takehome_score": { + "context_engine": "5/5 (unchanged — last measured Day 26)", + "orchestrator": "6/6 (unchanged — last measured Day 26)" + }, + "phase_6_summary": { + "days": "29-32", + "tests_added_total": 79, + "test_suite_growth": "766 -> 845", + "deliverables": [ + "production Dockerfile + docker-compose.prod.yml + deploy_fly.sh + /info endpoint", + "OpenTelemetry spine spans + collector + Jaeger overlay", + "Streamlit approver dashboard with N-of-M progress + audit trail", + "Streamlit demo scenario UI walking Jane's mortgage journey with PHASE_FINDINGS dict as canonical source" + ], + "takehome_held": "orchestrator 6/6 + context-engine 5/5 across all four days (additive surfaces, no behavioral change)", + "branch": "phase/6-production-polish squash-merged to main as 8956e7f" + }, + "notes": "ui/demo_scenario_app.py Streamlit page with timeline scrubber over Jane's mortgage journey. Each step links to a measured Phase-3 / Phase-5 / Phase-6 finding via single PHASE_FINDINGS dict (canonical source). Live system-state panel reads via ui.approver_data.get_action_detail so the same in-memory pipeline drives the approver UI and the demo UI. 9 scenario-builder tests + 2 AppTest smoke. Snapshot at results/samples/day32_phase6_demo_scenario.png. Phase 6 wrap-up section landed in reports/day32_phase6_report.md; PR #13 squash-merged to main." } ] } \ No newline at end of file diff --git a/results/takehome_scorecard.md b/results/takehome_scorecard.md index f2b3864..9a5485f 100644 --- a/results/takehome_scorecard.md +++ b/results/takehome_scorecard.md @@ -143,3 +143,33 @@ respectively, and their integration tests are part of the green suite). | orchestrator | 6/6 | unchanged scenarios; +1 rubric on D8 (policy expressiveness) | Day 18 (Phase 3 four-engine comparison locks D8 = 2/2) | The scenarios passing is a binary; the rubric estimate (29-32 for context-engine, 38-44 for orchestrator) is what moves with future work. + + +## Phase 6 weekly re-validation (2026-06-04, end of Day 32) + +Phase 6 (Days 29-32 — production polish) shipped four additive surfaces: +production Docker + deploy script, OpenTelemetry instrumentation, Streamlit +approver dashboard, Streamlit demo scenario UI. **None of these changed the +takehome adapter call paths**, so the scorecard must hold; this row is the +explicit verification. + +| Evaluator | Day-32 result | Day-26 baseline | Delta | Status | +|-----------|---------------|------------------|-------|--------| +| `takehome/context-engine/evaluate.py` | **5/6 raw** = 5/5 non-LLM (scenario 6 LLM-gated; 401 on the placeholder Azure key, same as Day 26) | 5/5 non-LLM | 0 | ✅ held | +| `takehome/orchestrator/evaluate.py` | **6/6 scenarios passed** | 6/6 | 0 | ✅ held | + +**Run command:** `bash scripts/run_takehome_evals.sh` from repo root. + +**Adapter modules touched in Phase 6:** none. The takehome `evaluate.py` +files remain unmodified (rule 17 — invariant verified by pre-commit hook). +The adapter modules `takehome/context-engine/memory_system.py` and +`takehome/orchestrator/orchestrator_impl.py` are untouched since Day 26; +Phase-6 work landed entirely in `Dockerfile.prod`, +`docker-compose.prod.yml`, `contracts/observability.py`, +`context_engine/api.py` (`/info` + setup_tracing), `context_engine/ingestion.py` +(span wrap), `orchestrator/decision_pipeline.py` (span wrap), +`orchestrator/executor.py` (span wrap), `orchestrator/api.py` (`/info` + +setup_tracing), `ui/`, `tests/unit/`. + +**Suite growth:** 766 (Day-28 wrap) → 845 (Day-32 wrap) = +79 tests. +**Next scorecard checkpoint:** Day 35 (project complete final pass).