From e2ddd03cefdbfcafcbaa59970db3db6fba5268ab Mon Sep 17 00:00:00 2001 From: jmagar <38927646+jmagar@users.noreply.github.com> Date: Mon, 21 Sep 2026 21:25:15 -0400 Subject: [PATCH 1/8] feat: add session investigation evidence bundle --- CLAUDE.md | 4 +- src/app.rs | 5 + src/app/models/ai_incidents.rs | 1 + src/app/models/ai_sessions.rs | 1 + src/app/models/investigation.rs | 97 + src/app/services.rs | 1 + src/app/services/ai.rs | 2 + src/app/services/ai_correlate_tests.rs | 1 + src/app/services/session_investigation.rs | 657 ++++++ .../services/session_investigation_tests.rs | 130 ++ src/cli/dispatch.rs | 1 + src/cli/dispatch_tests.rs | 2 +- src/db/agent_observatory.rs | 7 +- src/db/agent_observatory_read.rs | 70 +- src/db/agent_observatory_run_commits.rs | 50 +- src/db/models.rs | 2 + src/db/queries.rs | 2027 +---------------- src/db/queries_tests.rs | 1015 +-------- src/mcp/actions.rs | 12 + src/mcp/actions_tests.rs | 4 +- src/mcp/rmcp_server_tests.rs | 1 + src/mcp/tools.rs | 12 +- src/mcp/tools_tests.rs | 31 + 23 files changed, 1107 insertions(+), 3026 deletions(-) create mode 100644 src/app/services/session_investigation.rs create mode 100644 src/app/services/session_investigation_tests.rs diff --git a/CLAUDE.md b/CLAUDE.md index de2ccd42d..8b3544118 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -112,7 +112,7 @@ OTLP HTTP/protobuf `/v1/logs`, `/v1/metrics`, and `/v1/traces` all authenticate ## MCP Tools -One MCP tool: **`cortex`** — dispatches by `action` argument. 60 actions, generated from `ACTION_SPECS` in `src/mcp/actions.rs` (the single authoritative registry — regenerate this table from there). +One MCP tool: **`cortex`** — dispatches by `action` argument. 61 actions, generated from `ACTION_SPECS` in `src/mcp/actions.rs` (the single authoritative registry — regenerate this table from there). Scope taxonomy: every action requires `cortex:read` except the six **admin** actions `artifact_evidence_record`, `ack_error`, `unack_error`, `file_tails`, `notifications_test`, and `llm_invocations`, which require `cortex:admin` (static bearer tokens get read-only unless `CORTEX_STATIC_TOKEN_ADMIN=true`); `help` is info-only (no scope gate). @@ -287,7 +287,7 @@ RUST_LOG=info | `docker-compose.yml` | Production deployment (ports 1514, 3100) | | `docs/SETUP.md` | Setup guide (clone, build, configure, deploy, verify); per-host forwarder configs (rsyslog, UniFi, ATT router, WSL) live in README "Syslog Forwarder Setup" | | `src/db/queries.rs` | All SQL queries and FTS5 search implementation | -| `src/mcp/actions.rs` | `ACTION_SPECS` — authoritative registry of all 60 MCP actions and their scopes | +| `src/mcp/actions.rs` | `ACTION_SPECS` — authoritative registry of all 61 MCP actions and their scopes | | `src/mcp/tools.rs` | Single `cortex` tool with action dispatch | | `config/mcporter.json` | mcporter config (HTTP transport to localhost:3100) | | `config/systemd/` | `cortex-backup.service` / `.timer` — daily WAL-safe backup units | diff --git a/src/app.rs b/src/app.rs index 34139842d..7e082bd6c 100644 --- a/src/app.rs +++ b/src/app.rs @@ -227,6 +227,11 @@ pub use models::{ ServiceJournalEntry, ServiceLogsRequest, ServiceLogsResponse, + SessionExternalReference, + SessionExternalReferenceKind, + SessionInvestigateRequest, + SessionInvestigateResponse, + SessionObservatoryEvidence, SeverityCount, SilentHostsRequest, SilentHostsResponse, diff --git a/src/app/models/ai_incidents.rs b/src/app/models/ai_incidents.rs index cf9fed88a..3d9e1b50b 100644 --- a/src/app/models/ai_incidents.rs +++ b/src/app/models/ai_incidents.rs @@ -419,6 +419,7 @@ pub struct GraphSessionCorrelation { /// discover related hosts; `false` for the time-windowed fallback (session /// not yet projected into the graph). pub used_graph: bool, + pub session_entity_keys: Vec, pub discovered_hosts: Vec, pub discovered_entities: Vec, pub logs: Vec, diff --git a/src/app/models/ai_sessions.rs b/src/app/models/ai_sessions.rs index dcd880f81..97e8be26e 100644 --- a/src/app/models/ai_sessions.rs +++ b/src/app/models/ai_sessions.rs @@ -5,6 +5,7 @@ use super::*; pub struct ListSessionsRequest { pub project: Option, pub tool: Option, + pub session_id: Option, pub host: Option, pub since: Option, pub until: Option, diff --git a/src/app/models/investigation.rs b/src/app/models/investigation.rs index 7eb736ffe..7c421c7e9 100644 --- a/src/app/models/investigation.rs +++ b/src/app/models/investigation.rs @@ -177,6 +177,103 @@ pub struct AskInvestigationResponse { pub logs: Vec, } +#[derive(Debug, Clone, Default, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct SessionInvestigateRequest { + pub session_id: String, + pub tool: Option, + pub project: Option, + pub host: Option, + pub limit: Option, + pub window_minutes: Option, + pub severity_min: Option, +} + +#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq, PartialOrd, Ord)] +#[serde(rename_all = "snake_case")] +pub enum SessionExternalReferenceKind { + GithubPullRequest, + GithubIssue, + GithubReference, + LinearIssue, + Url, + CommitSha, +} + +#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq, PartialOrd, Ord)] +pub struct SessionExternalReference { + pub kind: SessionExternalReferenceKind, + pub value: String, + pub source_position: Option, + pub evidence_kind: String, + pub trust_level: String, + pub verified: bool, +} + +#[derive(Debug, Clone, Serialize, Deserialize)] +pub struct SessionRetentionLineageEntry { + pub log_id: i64, + pub hostname: String, + pub app_name: Option, + pub severity: String, + pub ai_project: Option, + pub ai_tool: Option, + pub ai_session_id: Option, + pub retained_until_epoch: i64, +} + +#[derive(Debug, Clone, Default, Serialize, Deserialize)] +pub struct SessionSourceEvidenceSummary { + pub count: usize, + pub log_ids: Vec, + pub truncated: bool, +} + +#[derive(Debug, Clone, Default, Serialize, Deserialize)] +pub struct SessionObservatoryEvidence { + pub runs: Vec, + pub ambiguous_run: bool, + pub related_runs: Vec, + pub related_runs_truncated: bool, + pub repository: Option, + pub worktree: Option, + pub commits: Vec, + pub events: Vec, + pub spans: Vec, + pub metrics: Vec, + pub events_truncated: bool, + pub spans_truncated: bool, + pub metrics_truncated: bool, +} + +#[derive(Debug, Clone, Serialize, Deserialize)] +pub struct SessionInvestigateResponse { + pub session: AiSessionEntry, + pub transcript: Vec, + pub transcript_has_more: bool, + pub correlation: Option, + pub graph_neighborhood: Option, + pub skill_events: Vec, + pub skill_events_truncated: bool, + pub mcp_events: Vec, + pub mcp_events_truncated: bool, + pub hook_events: Vec, + pub hook_events_truncated: bool, + pub artifact_evidence: Vec, + pub artifact_evidence_truncated: bool, + pub observatory: SessionObservatoryEvidence, + pub incident_context: IncidentContextResponse, + pub notifications: Vec, + pub notifications_truncated: bool, + pub retention_lineage: Vec, + pub retention_lineage_truncated: bool, + pub related_sessions: Vec, + pub external_references: Vec, + pub source_counts: BTreeMap, + pub source_evidence: BTreeMap, + pub partial_reasons: Vec, +} + pub fn app_entity_summary(entity: &GraphEntity) -> AppEntitySummary { AppEntitySummary { id: entity.id, diff --git a/src/app/services.rs b/src/app/services.rs index 777a3fb93..b41422923 100644 --- a/src/app/services.rs +++ b/src/app/services.rs @@ -122,6 +122,7 @@ mod mcp_backfill; mod mcp_events; mod mcp_incidents; mod rag; +mod session_investigation; mod session_pages; mod skill_assessment; mod skill_backfill; diff --git a/src/app/services/ai.rs b/src/app/services/ai.rs index cf5f56631..dde6b5f0b 100644 --- a/src/app/services/ai.rs +++ b/src/app/services/ai.rs @@ -53,6 +53,7 @@ fn build_graph_session_correlation( session_start, session_end, used_graph: inputs.used_graph, + session_entity_keys: inputs.session_entity_keys, discovered_hosts: inputs.discovered_hosts, discovered_entities: inputs.discovered_entities, logs, @@ -77,6 +78,7 @@ impl CortexService { let params = db::ListAiSessionsParams { ai_project: req.project, ai_tool: req.tool, + ai_session_id: req.session_id, host: req.host, since: from, until: to, diff --git a/src/app/services/ai_correlate_tests.rs b/src/app/services/ai_correlate_tests.rs index ceedc63fb..5de3ff03f 100644 --- a/src/app/services/ai_correlate_tests.rs +++ b/src/app/services/ai_correlate_tests.rs @@ -59,6 +59,7 @@ fn row_source_kind_parses_metadata() { fn build_graph_session_correlation_classifies_lanes_and_filters_heartbeats() { let inputs = db::SessionGraphInputs { bounds: Some(("2026-01-01T00:00:00Z".into(), "2026-01-01T00:10:00Z".into())), + session_entity_keys: vec!["cortex:claude:s1".into()], discovered_hosts: vec!["devhost".into()], discovered_entities: vec!["devhost".into(), "cortex".into()], used_graph: true, diff --git a/src/app/services/session_investigation.rs b/src/app/services/session_investigation.rs new file mode 100644 index 000000000..ebc201e5c --- /dev/null +++ b/src/app/services/session_investigation.rs @@ -0,0 +1,657 @@ +use std::time::Instant; + +use super::*; + +impl CortexService { + pub async fn session_investigate( + &self, + req: models::SessionInvestigateRequest, + ) -> ServiceResult> { + let started = Instant::now(); + let budget = InvestigationBudget::default(); + let session_id = req.session_id.trim().to_string(); + if session_id.is_empty() { + return Err(ServiceError::InvalidInput( + "session_id must not be empty".to_string(), + )); + } + let section_limit = req.limit.unwrap_or(100).clamp(1, 200); + + let sessions = self + .list_sessions(ListSessionsRequest { + project: req.project.clone(), + tool: req.tool.clone(), + session_id: Some(session_id.clone()), + host: req.host.clone(), + since: None, + until: None, + limit: Some(20), + }) + .await? + .sessions; + let mut exact = sessions + .into_iter() + .filter(|session| session.session_id == session_id) + .collect::>(); + if exact.is_empty() { + return Err(ServiceError::NotFound(format!( + "AI session not found: {session_id}" + ))); + } + if exact.len() > 1 { + return Err(ServiceError::InvalidInput( + "session_id is ambiguous; provide tool, project, and/or host".to_string(), + )); + } + let session = exact.remove(0); + + let transcript_page = self + .rendered_session_page(models::RenderedSessionPageRequest { + project: session.project.clone(), + tool: session.tool.clone(), + session_id: session.session_id.clone(), + host: session.hostname.clone(), + cursor: None, + limit: Some(section_limit), + }) + .await?; + + let correlated = self + .correlate_ai_logs(AiCorrelateRequest { + project: Some(session.project.clone()), + tool: Some(session.tool.clone()), + session_id: Some(session.session_id.clone()), + host: Some(session.hostname.clone()), + window_minutes: req.window_minutes, + severity_min: req.severity_min.clone(), + limit: Some(10), + events_per_anchor: Some(section_limit.min(100)), + ..Default::default() + }) + .await?; + + let session_entity_keys = correlated + .graph_correlation + .as_ref() + .map(|value| value.session_entity_keys.clone()) + .unwrap_or_default(); + let session_graph_entity_ambiguous = session_entity_keys.len() > 1; + let graph_neighborhood = match session_entity_keys.as_slice() { + [session_key] => { + let around = self + .graph_around(GraphAroundRequest { + mode: Some("around".to_string()), + entity_type: Some("ai_session".to_string()), + key: Some(session_key.clone()), + depth: Some(1), + limit: Some(section_limit.min(100)), + evidence_sample_limit: Some(3), + payload_budget: Some(32_768), + ..Default::default() + }) + .await?; + Some(models::app_graph_from_around_response(&around)) + } + _ => None, + }; + + let incident_context = self + .incident_context(models::IncidentContextRequest { + since: Some(session.first_seen.clone()), + until: Some(session.last_seen.clone()), + host: Some(session.hostname.clone()), + app: None, + query: None, + severity_min: req.severity_min.clone(), + limit: Some(section_limit.min(200)), + }) + .await?; + + let mut notifications = self + .notifications_recent_checked(models::NotificationsRecentRequest { + limit: Some(i64::from(section_limit.saturating_mul(5).min(500))), + rule_id: None, + since: Some(session.first_seen.clone()), + }) + .await? + .into_iter() + .filter(|firing| { + firing.hostname == session.hostname + && timestamp_inclusive_between( + &firing.fired_at, + &session.first_seen, + &session.last_seen, + ) + }) + .collect::>(); + let notifications_truncated = notifications.len() > section_limit as usize; + notifications.truncate(section_limit as usize); + + let retention_project = session.project.clone(); + let retention_tool = session.tool.clone(); + let retention_session_id = session.session_id.clone(); + let retention_host = session.hostname.clone(); + let mut retention_lineage = self + .run_db("session_investigate_retention_lineage", move |pool| { + let conn = pool.get()?; + let mut stmt = conn.prepare( + "SELECT id, hostname, app_name, severity, ai_project, ai_tool, ai_session_id, + deleted_at + 900 + FROM stream_deleted_log_lineage + WHERE ai_project = ?1 + AND lower(ai_tool) = lower(?2) + AND ai_session_id = ?3 + AND hostname = ?4 + ORDER BY id DESC + LIMIT ?5", + )?; + let rows = stmt.query_map( + rusqlite::params![ + retention_project, + retention_tool, + retention_session_id, + retention_host, + i64::from(section_limit) + 1 + ], + |row| { + Ok(models::SessionRetentionLineageEntry { + log_id: row.get(0)?, + hostname: row.get(1)?, + app_name: row.get(2)?, + severity: row.get(3)?, + ai_project: row.get(4)?, + ai_tool: row.get(5)?, + ai_session_id: row.get(6)?, + retained_until_epoch: row.get(7)?, + }) + }, + )?; + Ok(rows.collect::, rusqlite::Error>>()?) + }) + .await?; + let retention_lineage_truncated = retention_lineage.len() > section_limit as usize; + retention_lineage.truncate(section_limit as usize); + + let (skill_events, mcp_events, hook_events) = tokio::try_join!( + self.list_skill_events(models::ListSkillEventsRequest { + tool: Some(session.tool.clone()), + project: Some(session.project.clone()), + session_id: Some(session.session_id.clone()), + hostname: Some(session.hostname.clone()), + limit: Some(section_limit), + ..Default::default() + }), + self.list_mcp_events(models::ListMcpEventsRequest { + tool: Some(session.tool.clone()), + project: Some(session.project.clone()), + session_id: Some(session.session_id.clone()), + hostname: Some(session.hostname.clone()), + limit: Some(section_limit), + ..Default::default() + }), + self.list_hook_events(models::ListHookEventsRequest { + tool: Some(session.tool.clone()), + project: Some(session.project.clone()), + session_id: Some(session.session_id.clone()), + hostname: Some(session.hostname.clone()), + limit: Some(section_limit), + ..Default::default() + }) + )?; + + let (artifact_by_correlation, artifact_by_request) = tokio::try_join!( + self.list_artifact_evidence(models::ListArtifactEvidenceRequest { + correlation_id: Some(session.session_id.clone()), + from: Some(session.first_seen.clone()), + to: Some(session.last_seen.clone()), + limit: Some(section_limit), + ..Default::default() + }), + self.list_artifact_evidence(models::ListArtifactEvidenceRequest { + request_id: Some(session.session_id.clone()), + from: Some(session.first_seen.clone()), + to: Some(session.last_seen.clone()), + limit: Some(section_limit), + ..Default::default() + }) + )?; + let artifact_evidence_truncated = + artifact_by_correlation.truncated || artifact_by_request.truncated; + let mut artifact_evidence_by_id = BTreeMap::new(); + for event in artifact_by_correlation + .events + .into_iter() + .chain(artifact_by_request.events) + { + artifact_evidence_by_id.insert(event.cortex_log_id, event); + } + let artifact_evidence = artifact_evidence_by_id.into_values().collect::>(); + + let observatory_session_id = session.session_id.clone(); + let observatory_tool = session.tool.clone(); + let observatory_host = session.hostname.clone(); + let observatory = self + .run_db("session_investigate_observatory", move |pool| { + let query = db::agent_observatory::AgentRunQuery { + tools: vec![observatory_tool.clone()], + host: Some(observatory_host.clone()), + query: Some(observatory_session_id.clone()), + ..Default::default() + }; + let mut runs = + db::agent_observatory::list_observatory_runs(pool, &query, None, 50, i64::MAX)?; + runs.retain(|run| { + run.native_session_id == observatory_session_id + && run.tool.eq_ignore_ascii_case(&observatory_tool) + && run.hostname == observatory_host + }); + let ambiguous_run = runs.len() > 1; + let Some(run) = runs.first().cloned().filter(|_| !ambiguous_run) else { + return Ok(models::SessionObservatoryEvidence { + runs, + ambiguous_run, + ..Default::default() + }); + }; + let resolved = db::agent_observatory::resolve_observatory_run(pool, &run.run_key)?; + let (run_id, identity) = resolved.ok_or_else(|| { + anyhow::anyhow!( + "Agent Observatory run disappeared during session investigation" + ) + })?; + let worktree = match run.primary_worktree_id { + Some(id) => db::agent_observatory::resolve_observatory_worktree(pool, id)?, + None => None, + }; + let repository = match worktree.as_ref() { + Some(worktree) => db::agent_observatory::resolve_observatory_repository( + pool, + worktree.repository_id, + )?, + None => None, + }; + let commits = + db::agent_observatory::list_agent_run_attributed_commits(pool, run_id)?; + let mut related_runs = match run.primary_worktree_id { + Some(worktree_id) => db::agent_observatory::list_observatory_runs( + pool, + &db::agent_observatory::AgentRunQuery { + worktree_id: Some(worktree_id), + ..Default::default() + }, + None, + section_limit as usize + 1, + i64::MAX, + )?, + None => Vec::new(), + }; + related_runs.retain(|candidate| candidate.id != run_id); + let related_runs_truncated = related_runs.len() > section_limit as usize; + related_runs.truncate(section_limit as usize); + let events = db::agent_observatory::list_observatory_events( + pool, + &run.run_key, + &db::agent_observatory::AgentEventQuery::default(), + None, + section_limit as usize + 1, + true, + i64::MAX, + )?; + let spans = db::agent_observatory::list_observatory_spans( + pool, + run_id, + &identity, + &db::agent_observatory::TelemetryQuery::default(), + None, + section_limit as usize + 1, + i64::MAX, + )?; + let metrics = db::agent_observatory::list_observatory_metrics( + pool, + run_id, + &identity, + &db::agent_observatory::TelemetryQuery::default(), + None, + section_limit as usize + 1, + i64::MAX, + )?; + Ok(models::SessionObservatoryEvidence { + runs, + ambiguous_run, + related_runs, + related_runs_truncated, + repository, + worktree, + commits, + events_truncated: events.len() > section_limit as usize, + spans_truncated: spans.len() > section_limit as usize, + metrics_truncated: metrics.len() > section_limit as usize, + events: events.into_iter().take(section_limit as usize).collect(), + spans: spans.into_iter().take(section_limit as usize).collect(), + metrics: metrics.into_iter().take(section_limit as usize).collect(), + }) + }) + .await?; + + let related_sessions = self + .list_sessions(ListSessionsRequest { + project: Some(session.project.clone()), + tool: None, + session_id: None, + host: None, + since: Some(session.first_seen.clone()), + until: Some(session.last_seen.clone()), + limit: Some(50), + }) + .await? + .sessions + .into_iter() + .filter(|candidate| candidate.session_key != session.session_key) + .take(20) + .collect::>(); + + let external_references = extract_session_external_references(&transcript_page.events); + let source_evidence = summarize_session_source_evidence( + correlated.graph_correlation.as_ref(), + section_limit as usize, + ); + let mut source_counts = source_evidence + .iter() + .map(|(kind, summary)| (kind.clone(), summary.count)) + .collect::>(); + for event in &observatory.events { + *source_counts + .entry(format!("observatory:{}", event.source_kind)) + .or_default() += 1; + } + if !skill_events.events.is_empty() { + source_counts.insert("skill_event".to_string(), skill_events.events.len()); + } + if !mcp_events.events.is_empty() { + source_counts.insert("mcp_event".to_string(), mcp_events.events.len()); + } + if !hook_events.events.is_empty() { + source_counts.insert("hook_event".to_string(), hook_events.events.len()); + } + if !artifact_evidence.is_empty() { + source_counts.insert("artifact_evidence".to_string(), artifact_evidence.len()); + } + if !observatory.spans.is_empty() { + source_counts.insert("otlp_span".to_string(), observatory.spans.len()); + } + if !observatory.metrics.is_empty() { + source_counts.insert("otlp_metric".to_string(), observatory.metrics.len()); + } + if let Some(graph) = graph_neighborhood.as_ref() { + source_counts.insert("graph_entity".to_string(), graph.entities.len()); + source_counts.insert("graph_relationship".to_string(), graph.relationships.len()); + source_counts.insert("graph_evidence".to_string(), graph.evidence.len()); + } + if let Some(correlation) = correlated.graph_correlation.as_ref() + && !correlation.heartbeat_summaries.is_empty() + { + source_counts.insert( + "heartbeat_summary".to_string(), + correlation.heartbeat_summaries.len(), + ); + } + if !incident_context.error_logs.is_empty() { + source_counts.insert( + "incident_error_log".to_string(), + incident_context.error_logs.len(), + ); + } + if !notifications.is_empty() { + source_counts.insert("notification".to_string(), notifications.len()); + } + if !observatory.related_runs.is_empty() { + source_counts.insert( + "same_worktree_run".to_string(), + observatory.related_runs.len(), + ); + } + if !retention_lineage.is_empty() { + source_counts.insert("retention_lineage".to_string(), retention_lineage.len()); + } + + let mut partial_reasons = Vec::new(); + if transcript_page.has_more { + partial_reasons.push("transcript_truncated".to_string()); + } + if correlated + .graph_correlation + .as_ref() + .is_some_and(|value| value.truncated) + { + partial_reasons.push("correlation_truncated".to_string()); + } + if session_graph_entity_ambiguous { + partial_reasons.push("session_graph_entity_ambiguous".to_string()); + } + if skill_events.truncated { + partial_reasons.push("skill_events_truncated".to_string()); + } + if mcp_events.truncated { + partial_reasons.push("mcp_events_truncated".to_string()); + } + if hook_events.truncated { + partial_reasons.push("hook_events_truncated".to_string()); + } + if artifact_evidence_truncated { + partial_reasons.push("artifact_evidence_truncated".to_string()); + } + if observatory.ambiguous_run { + partial_reasons.push("observatory_run_ambiguous".to_string()); + } + if observatory.related_runs_truncated { + partial_reasons.push("observatory_related_runs_truncated".to_string()); + } + if observatory.events_truncated { + partial_reasons.push("observatory_events_truncated".to_string()); + } + if observatory.spans_truncated { + partial_reasons.push("observatory_spans_truncated".to_string()); + } + if observatory.metrics_truncated { + partial_reasons.push("observatory_metrics_truncated".to_string()); + } + if incident_context.error_logs_truncated { + partial_reasons.push("incident_context_errors_truncated".to_string()); + } + if notifications_truncated { + partial_reasons.push("notifications_truncated".to_string()); + } + if retention_lineage_truncated { + partial_reasons.push("retention_lineage_truncated".to_string()); + } + + let metadata = session_investigation_metadata( + &budget, + started, + correlated.graph_correlation.as_ref(), + graph_neighborhood.is_some(), + &partial_reasons, + transcript_page.events.len(), + observatory.events.len(), + ); + Ok(InvestigationEnvelope { + metadata, + result: models::SessionInvestigateResponse { + session, + transcript: transcript_page.events, + transcript_has_more: transcript_page.has_more, + correlation: correlated.graph_correlation, + graph_neighborhood, + skill_events: skill_events.events, + skill_events_truncated: skill_events.truncated, + mcp_events: mcp_events.events, + mcp_events_truncated: mcp_events.truncated, + hook_events: hook_events.events, + hook_events_truncated: hook_events.truncated, + artifact_evidence, + artifact_evidence_truncated, + observatory, + incident_context, + notifications, + notifications_truncated, + retention_lineage, + retention_lineage_truncated, + related_sessions, + external_references, + source_counts, + source_evidence, + partial_reasons, + }, + }) + } +} + +fn session_investigation_metadata( + budget: &InvestigationBudget, + started: Instant, + correlation: Option<&GraphSessionCorrelation>, + has_graph_neighborhood: bool, + partial_reasons: &[String], + transcript_rows: usize, + evidence_rows: usize, +) -> InvestigationMetadata { + let graph_calls = u32::from(correlation.is_some()) + u32::from(has_graph_neighborhood); + let log_rows = correlation.map_or(0, |value| value.logs.len() as u32); + let partial = !partial_reasons.is_empty(); + InvestigationMetadata { + server_version: env!("CARGO_PKG_VERSION").to_string(), + schema_version: INVESTIGATION_UI_VERSION.to_string(), + graph_projection_status: correlation + .map(|value| if value.used_graph { "used" } else { "fallback" }.to_string()), + source_watermark: None, + degraded_reasons: correlation + .filter(|value| !value.used_graph) + .map(|_| vec!["session_graph_entity_unavailable".to_string()]) + .unwrap_or_default(), + truncated: partial, + truncation_reasons: partial_reasons.to_vec(), + partial, + partial_reasons: partial_reasons.to_vec(), + auth_state: "bearer".to_string(), + budget: budget.clone(), + budget_used: InvestigationBudgetUsed { + graph_calls, + log_rows, + evidence_rows: evidence_rows.min(u32::MAX as usize) as u32, + candidate_explanations: 0, + wall_time_ms: started.elapsed().as_millis().min(u32::MAX as u128) as u32, + payload_bytes: 0, + }, + payload_limit_bytes: budget.max_payload_bytes, + version_skew: (transcript_rows > 200).then(|| "transcript_limit_exceeded".to_string()), + } +} + +fn summarize_session_source_evidence( + correlation: Option<&GraphSessionCorrelation>, + requested_limit: usize, +) -> BTreeMap { + let id_limit = requested_limit.clamp(1, 50); + let mut summaries = BTreeMap::::new(); + let Some(correlation) = correlation else { + return summaries; + }; + + for row in &correlation.logs { + let kind = row + .source_kind + .clone() + .unwrap_or_else(|| "unknown".to_string()); + let summary = summaries.entry(kind).or_default(); + summary.count += 1; + if summary.log_ids.len() < id_limit { + summary.log_ids.push(row.entry.id); + } else { + summary.truncated = true; + } + } + + summaries +} + +fn timestamp_inclusive_between(value: &str, start: &str, end: &str) -> bool { + let Ok(value) = chrono::DateTime::parse_from_rfc3339(value) else { + return false; + }; + let Ok(start) = chrono::DateTime::parse_from_rfc3339(start) else { + return false; + }; + let Ok(end) = chrono::DateTime::parse_from_rfc3339(end) else { + return false; + }; + value >= start && value <= end +} + +fn extract_session_external_references( + events: &[models::RenderedSessionEvent], +) -> Vec { + let mut references = std::collections::BTreeSet::new(); + for event in events { + for raw in event.text.split_whitespace() { + let token = raw.trim_matches(|ch: char| { + matches!( + ch, + ',' | '.' | ';' | ':' | ')' | '(' | ']' | '[' | '}' | '{' | '"' | '\'' + ) + }); + if token.is_empty() { + continue; + } + let kind = if token.starts_with("http://") || token.starts_with("https://") { + if token.contains("github.com/") && token.contains("/pull/") { + models::SessionExternalReferenceKind::GithubPullRequest + } else if token.contains("github.com/") && token.contains("/issues/") { + models::SessionExternalReferenceKind::GithubIssue + } else { + models::SessionExternalReferenceKind::Url + } + } else if looks_like_linear_identifier(token) { + models::SessionExternalReferenceKind::LinearIssue + } else if token.strip_prefix('#').is_some_and(|number| { + !number.is_empty() && number.chars().all(|ch| ch.is_ascii_digit()) + }) { + models::SessionExternalReferenceKind::GithubReference + } else if looks_like_commit_sha(token) { + models::SessionExternalReferenceKind::CommitSha + } else { + continue; + }; + references.insert(models::SessionExternalReference { + kind, + value: safe_passive_text(token, 500), + source_position: Some(event.position), + evidence_kind: "transcript_text".to_string(), + trust_level: "claimed".to_string(), + verified: false, + }); + } + } + references.into_iter().collect() +} + +fn looks_like_linear_identifier(token: &str) -> bool { + let Some((prefix, suffix)) = token.split_once('-') else { + return false; + }; + (2..=12).contains(&prefix.len()) + && prefix + .chars() + .all(|ch| ch.is_ascii_uppercase() || ch.is_ascii_digit()) + && !suffix.is_empty() + && suffix.chars().all(|ch| ch.is_ascii_digit()) +} + +fn looks_like_commit_sha(token: &str) -> bool { + (7..=40).contains(&token.len()) + && token.chars().all(|ch| ch.is_ascii_hexdigit()) + && token.chars().any(|ch| ch.is_ascii_alphabetic()) +} + +#[cfg(test)] +#[path = "session_investigation_tests.rs"] +mod tests; diff --git a/src/app/services/session_investigation_tests.rs b/src/app/services/session_investigation_tests.rs new file mode 100644 index 000000000..1f278915a --- /dev/null +++ b/src/app/services/session_investigation_tests.rs @@ -0,0 +1,130 @@ +use super::*; + +fn event(position: i64, text: &str) -> models::RenderedSessionEvent { + models::RenderedSessionEvent { + position, + timestamp: "2026-09-21T00:00:00Z".to_string(), + kind: models::RenderedSessionEventKind::Assistant, + text: text.to_string(), + redacted: false, + parse_warning: None, + } +} + +#[test] +fn external_references_are_classified_and_deduplicated() { + let events = vec![ + event( + 1, + "Worked on CLD-1149 and https://github.com/dinglebear-ai/labby/pull/717 with commit abc1234.", + ), + event( + 2, + "Follow-up #731 https://github.com/dinglebear-ai/labby/issues/723 https://linear.app/lime-technology/issue/CLD-1149/foo", + ), + ]; + + let refs = extract_session_external_references(&events); + + assert!(refs.iter().any(|item| { + item.kind == models::SessionExternalReferenceKind::LinearIssue && item.value == "CLD-1149" + })); + assert!(refs.iter().any(|item| { + item.kind == models::SessionExternalReferenceKind::GithubPullRequest + && item.value.contains("/pull/717") + })); + assert!(refs.iter().any(|item| { + item.kind == models::SessionExternalReferenceKind::GithubIssue + && item.value.contains("/issues/723") + })); + assert!(refs.iter().any(|item| { + item.kind == models::SessionExternalReferenceKind::GithubReference && item.value == "#731" + })); + assert!(refs.iter().any(|item| { + item.kind == models::SessionExternalReferenceKind::CommitSha && item.value == "abc1234" + })); +} + +#[test] +fn linear_identifier_parser_rejects_ordinary_hyphenated_text() { + assert!(looks_like_linear_identifier("U8-1122")); + assert!(looks_like_linear_identifier("CLD-1149")); + assert!(!looks_like_linear_identifier("session-investigate")); + assert!(!looks_like_linear_identifier("abc-123")); + assert!(!looks_like_linear_identifier("CLD-next")); +} + +fn correlated_log(id: i64, source_kind: Option<&str>) -> CorrelatedLogRow { + CorrelatedLogRow { + entry: models::LogEntry { + id, + timestamp: "2026-09-21T00:00:00Z".to_string(), + hostname: "macpoo".to_string(), + facility: None, + severity: "info".to_string(), + app_name: Some("test".to_string()), + process_id: None, + message: format!("log-{id}"), + received_at: "2026-09-21T00:00:00Z".to_string(), + source_ip: "test://source".to_string(), + ai_tool: None, + ai_project: None, + ai_session_id: None, + ai_transcript_path: None, + metadata_json: None, + }, + source_kind: source_kind.map(str::to_string), + discovery: "test".to_string(), + } +} + +#[test] +fn source_evidence_groups_source_kinds_and_bounds_log_ids() { + let correlation = GraphSessionCorrelation { + session_id: "session-1".to_string(), + session_start: "2026-09-21T00:00:00Z".to_string(), + session_end: "2026-09-21T00:10:00Z".to_string(), + used_graph: true, + session_entity_keys: vec!["project:codex:session-1".to_string()], + discovered_hosts: Vec::new(), + discovered_entities: Vec::new(), + logs: vec![ + correlated_log(1, Some("docker-stream")), + correlated_log(2, Some("docker-stream")), + correlated_log(3, Some("agent-command")), + correlated_log(4, None), + ], + agent_command_count: 1, + shell_history_count: 0, + heartbeat_summaries: Vec::new(), + truncated: false, + }; + + let summaries = summarize_session_source_evidence(Some(&correlation), 1); + + assert_eq!(summaries["docker-stream"].count, 2); + assert_eq!(summaries["docker-stream"].log_ids, vec![1]); + assert!(summaries["docker-stream"].truncated); + assert_eq!(summaries["agent-command"].log_ids, vec![3]); + assert_eq!(summaries["unknown"].log_ids, vec![4]); +} + +#[test] +fn timestamp_window_is_inclusive_and_rejects_invalid_values() { + let start = "2026-09-21T00:00:00Z"; + let end = "2026-09-21T00:10:00Z"; + + assert!(timestamp_inclusive_between(start, start, end)); + assert!(timestamp_inclusive_between(end, start, end)); + assert!(timestamp_inclusive_between( + "2026-09-21T00:05:00Z", + start, + end + )); + assert!(!timestamp_inclusive_between( + "2026-09-21T00:11:00Z", + start, + end + )); + assert!(!timestamp_inclusive_between("not-a-time", start, end)); +} diff --git a/src/cli/dispatch.rs b/src/cli/dispatch.rs index 936d589ca..bb7781d8c 100644 --- a/src/cli/dispatch.rs +++ b/src/cli/dispatch.rs @@ -149,6 +149,7 @@ impl SessionsArgs { ListSessionsRequest { project: self.project, tool: self.tool, + session_id: None, host: self.host, since: self.since, until: self.until, diff --git a/src/cli/dispatch_tests.rs b/src/cli/dispatch_tests.rs index 1289362dc..14b99d021 100644 --- a/src/cli/dispatch_tests.rs +++ b/src/cli/dispatch_tests.rs @@ -247,7 +247,7 @@ fn sessions_args_into_request_snapshot() { let req = args.into_request(); assert_eq!( format!("{req:?}"), - "ListSessionsRequest { project: Some(\"/home/me/proj\"), tool: Some(\"claude\"), host: None, since: None, until: None, limit: Some(20) }" + "ListSessionsRequest { project: Some(\"/home/me/proj\"), tool: Some(\"claude\"), session_id: None, host: None, since: None, until: None, limit: Some(20) }" ); } diff --git a/src/db/agent_observatory.rs b/src/db/agent_observatory.rs index e0218b618..d5cb0bb6c 100644 --- a/src/db/agent_observatory.rs +++ b/src/db/agent_observatory.rs @@ -16,8 +16,8 @@ mod run_commits; #[cfg(test)] pub use run_commits::list_agent_run_commits; pub use run_commits::{ - AgentRunCommitUpsert, commit_attribution_evidence, git_commit_by_repository_sha, - upsert_agent_run_commit, + AgentRunAttributedCommit, AgentRunCommitUpsert, commit_attribution_evidence, + git_commit_by_repository_sha, list_agent_run_attributed_commits, upsert_agent_run_commit, }; #[path = "agent_observatory_sources.rs"] mod sources; @@ -66,7 +66,8 @@ pub use read::{ ObservatoryWorktreeRow, RepositoryQuery, RunTelemetryIdentity, TelemetryQuery, list_observatory_events, list_observatory_metrics, list_observatory_repositories, list_observatory_runs, list_observatory_spans, list_observatory_worktrees, - resolve_observatory_run, scoped_evidence_events, + resolve_observatory_repository, resolve_observatory_run, resolve_observatory_worktree, + scoped_evidence_events, }; use crate::db::pool::DbPool; diff --git a/src/db/agent_observatory_read.rs b/src/db/agent_observatory_read.rs index fbb86ad3c..d625ef1a4 100644 --- a/src/db/agent_observatory_read.rs +++ b/src/db/agent_observatory_read.rs @@ -2,7 +2,7 @@ use super::DbPool; use anyhow::{Context, Result}; -use rusqlite::{Row, params_from_iter, types::Value}; +use rusqlite::{OptionalExtension, Row, params_from_iter, types::Value}; #[path = "agent_observatory_read_cursor.rs"] mod cursor; use cursor::{bounded_limit, int_cursor, push_filter, text_cursor}; @@ -73,6 +73,74 @@ pub fn list_observatory_repositories( .context("list observatory repositories") } +pub fn resolve_observatory_repository( + pool: &DbPool, + repository_id: i64, +) -> Result> { + let conn = pool.get()?; + conn.query_row( + "SELECT r.id,r.repository_key,r.hostname,r.primary_path,r.display_name,r.first_seen_at,r.last_seen_at,r.removed_at,(SELECT COUNT(*) FROM repository_worktrees w WHERE w.repository_id=r.id),(SELECT COUNT(*) FROM agent_runs a JOIN agent_run_worktrees rw ON rw.run_id=a.id JOIN repository_worktrees w ON w.id=rw.worktree_id WHERE w.repository_id=r.id AND a.status IN ('starting','active','waiting','idle')) FROM repositories r WHERE r.id=?1", + [repository_id], + |r| { + Ok(ObservatoryRepositoryRow { + id: r.get(0)?, + key: r.get(1)?, + hostname: r.get(2)?, + primary_path: r.get(3)?, + name: r.get(4)?, + first_seen_at: r.get(5)?, + last_seen_at: r.get(6)?, + removed_at: r.get(7)?, + worktree_count: r.get(8)?, + active_run_count: r.get(9)?, + }) + }, + ) + .optional() + .context("resolve observatory repository") +} + +pub fn resolve_observatory_worktree( + pool: &DbPool, + worktree_id: i64, +) -> Result> { + let conn = pool.get()?; + conn.query_row( + "SELECT id,worktree_key,repository_id,hostname,path,branch_ref,branch_name,head_sha,upstream_ref,detached,bare,locked,lock_reason,prunable,prune_reason,dirty,staged_count,unstaged_count,untracked_count,ahead,behind,first_seen_at,last_seen_at,removed_at FROM repository_worktrees WHERE id=?1", + [worktree_id], + |r| { + Ok(ObservatoryWorktreeRow { + id: r.get(0)?, + key: r.get(1)?, + repository_id: r.get(2)?, + hostname: r.get(3)?, + path: r.get(4)?, + branch_ref: r.get(5)?, + branch: r.get(6)?, + head_sha: r.get(7)?, + upstream_ref: r.get(8)?, + detached: r.get(9)?, + bare: r.get(10)?, + locked: r.get(11)?, + lock_reason: r.get(12)?, + prunable: r.get(13)?, + prune_reason: r.get(14)?, + dirty: r.get(15)?, + staged: r.get(16)?, + unstaged: r.get(17)?, + untracked: r.get(18)?, + ahead: r.get(19)?, + behind: r.get(20)?, + first_seen_at: r.get(21)?, + last_seen_at: r.get(22)?, + removed_at: r.get(23)?, + }) + }, + ) + .optional() + .context("resolve observatory worktree") +} + /// Historical and follow-safe evidence projection for one branch/worktree. /// The keyset watermark is the durable `agent_run_events.id`; callers can /// replay a bounded page and then poll with `after_id` without a race window. diff --git a/src/db/agent_observatory_run_commits.rs b/src/db/agent_observatory_run_commits.rs index 1dd2c833e..006830345 100644 --- a/src/db/agent_observatory_run_commits.rs +++ b/src/db/agent_observatory_run_commits.rs @@ -29,6 +29,12 @@ pub struct AgentRunCommitRow { pub metadata_json: String, } +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +pub struct AgentRunAttributedCommit { + pub relation: AgentRunCommitRow, + pub commit: super::GitCommitRow, +} + #[derive(Debug, Clone, PartialEq)] pub struct AgentRunCommitUpsert { pub run_id: i64, @@ -176,7 +182,6 @@ pub fn upsert_agent_run_commit( Ok(relation) } -#[cfg(test)] pub fn list_agent_run_commits(pool: &DbPool, run_id: i64) -> Result> { if run_id <= 0 { bail!("run_id must be positive"); @@ -190,6 +195,49 @@ pub fn list_agent_run_commits(pool: &DbPool, run_id: i64) -> Result>>()?) } +pub fn list_agent_run_attributed_commits( + pool: &DbPool, + run_id: i64, +) -> Result> { + let relations = list_agent_run_commits(pool, run_id)?; + let connection = pool.get().context("acquire database connection")?; + let mut statement = connection.prepare( + "SELECT id, repository_id, sha, parent_shas_json, author_name, author_email_hash, + authored_at, committed_at, subject, changed_files, insertions, deletions, + changed_paths_json, first_observed_at, last_observed_at, reachable, metadata_json + FROM git_commits WHERE id = ?1", + )?; + relations + .into_iter() + .map(|relation| { + let commit = statement + .query_row([relation.commit_id], |row| { + Ok(super::GitCommitRow { + id: row.get(0)?, + repository_id: row.get(1)?, + sha: row.get(2)?, + parent_shas_json: row.get(3)?, + author_name: row.get(4)?, + author_email_hash: row.get(5)?, + authored_at: row.get(6)?, + committed_at: row.get(7)?, + subject: row.get(8)?, + changed_files: row.get(9)?, + insertions: row.get(10)?, + deletions: row.get(11)?, + changed_paths_json: row.get(12)?, + first_observed_at: row.get(13)?, + last_observed_at: row.get(14)?, + reachable: row.get(15)?, + metadata_json: row.get(16)?, + }) + }) + .context("resolve attributed git commit")?; + Ok(AgentRunAttributedCommit { relation, commit }) + }) + .collect() +} + pub fn commit_attribution_evidence( pool: &DbPool, worktree_id: i64, diff --git a/src/db/models.rs b/src/db/models.rs index 4eae843e8..1c632fc5b 100644 --- a/src/db/models.rs +++ b/src/db/models.rs @@ -65,6 +65,7 @@ pub struct DockerCheckpoint { pub struct ListAiSessionsParams { pub ai_project: Option, pub ai_tool: Option, + pub ai_session_id: Option, pub host: Option, pub since: Option, pub until: Option, @@ -280,6 +281,7 @@ pub struct AiRelatedWindow { #[derive(Debug, Clone, Default)] pub struct SessionGraphInputs { pub bounds: Option<(String, String)>, + pub session_entity_keys: Vec, pub discovered_hosts: Vec, pub discovered_entities: Vec, pub used_graph: bool, diff --git a/src/db/queries.rs b/src/db/queries.rs index b40ee6b71..e3fb20bfe 100644 --- a/src/db/queries.rs +++ b/src/db/queries.rs @@ -758,6 +758,11 @@ pub fn list_ai_sessions_live( bindings.push(rusqlite::types::Value::Text(tool.clone())); idx += 1; } + if let Some(session_id) = ¶ms.ai_session_id { + sql.push_str(&format!(" AND ai_session_id = ?{idx}")); + bindings.push(rusqlite::types::Value::Text(session_id.clone())); + idx += 1; + } if let Some(hostname) = ¶ms.host { sql.push_str(&format!(" AND hostname = ?{idx}")); bindings.push(rusqlite::types::Value::Text(hostname.clone())); @@ -854,6 +859,11 @@ fn list_ai_sessions_from_rollup( bindings.push(rusqlite::types::Value::Text(tool.clone())); idx += 1; } + if let Some(session_id) = ¶ms.ai_session_id { + sql.push_str(&format!(" AND ai_session_id = ?{idx}")); + bindings.push(rusqlite::types::Value::Text(session_id.clone())); + idx += 1; + } if let Some(hostname) = ¶ms.host { sql.push_str(&format!(" AND hostname = ?{idx}")); bindings.push(rusqlite::types::Value::Text(hostname.clone())); @@ -2058,6 +2068,7 @@ pub fn correlate_session_graph( Ok(SessionGraphInputs { bounds: Some((start, end)), + session_entity_keys: session_keys, discovered_hosts, discovered_entities, used_graph, @@ -2135,2018 +2146,4 @@ pub fn topic_correlate_inputs( && !covered.contains(entity.canonical_key.as_str()) { entity.resolver_status = ResolverStatus::Degraded; - } - } - instance_keys.extend(linked); - } - instance_keys.sort(); - instance_keys.dedup(); - generic_seeds.sort(); - generic_seeds.dedup(); - - // Graph expansion + host discovery. Service seeds use the bounded - // service-topic walk (proof relationships only); generic seeds keep the - // general walk. - let mut walk_entities = Vec::new(); - let mut service_seeds: Vec = logical_keys.clone(); - service_seeds.extend(instance_keys.iter().cloned()); - service_seeds.sort(); - service_seeds.dedup(); - let mut graph_walk_truncated = false; - if !service_seeds.is_empty() { - let (entities, truncated) = super::graph_resolver_projection::graph_walk_service_topic( - &conn, - &service_seeds, - max_depth, - )?; - walk_entities.extend(entities); - graph_walk_truncated |= truncated; - } - if !generic_seeds.is_empty() { - walk_entities.extend(super::graph::graph_walk_n_hops( - &conn, - &generic_seeds, - max_depth, - )?); - } - let seed_set: std::collections::HashSet<&str> = service_seeds - .iter() - .chain(generic_seeds.iter()) - .map(String::as_str) - .collect(); - let mut expansion: Vec<(String, String)> = Vec::new(); - let mut discovered_hosts: Vec = Vec::new(); - for entity in walk_entities { - match entity.entity_type.as_str() { - super::graph::ENTITY_TYPE_HOST => discovered_hosts.push(entity.canonical_key.clone()), - super::graph::ENTITY_TYPE_CONTAINER => { - if let Some(host) = - super::entity_resolution::container_key_host(&entity.canonical_key) - { - discovered_hosts.push(host.to_string()); - } - } - super::graph::ENTITY_TYPE_SERVICE_INSTANCE => { - if let Some((host, _)) = - super::entity_resolution::split_service_instance_key(&entity.canonical_key) - { - discovered_hosts.push(host.to_string()); - } - } - _ => {} - } - if !seed_set.contains(entity.canonical_key.as_str()) { - expansion.push((entity.entity_type, entity.canonical_key)); - } - } - discovered_hosts.sort(); - discovered_hosts.dedup(); - expansion.sort(); - expansion.dedup(); - drop(conn); - - // Log fan-out: service instances use service-scoped predicates; generic - // seeds use the graph-related fan-out. Never both for the same rows — - // results merge newest-first under the shared limit. - let mut logs: Vec = Vec::new(); - if !instance_keys.is_empty() { - logs.extend( - queries_service_instances::search_logs_for_service_instances( - pool, - &instance_keys, - since, - until, - source_kinds, - limit, - )?, - ); - } - if !generic_seeds.is_empty() { - // SeedHostsOnly: a host reached transitively from an app/container - // seed (e.g. app:plex —emitted_by→ host:nashost) must never drive - // host-wide `l.hostname IN (…)` inclusion labelled `resolved`. Only - // hosts that were exact topic matches themselves fan out. - logs.extend( - search_logs_from_graph_related_entities( - pool, - &generic_seeds, - max_depth, - since, - until, - source_kinds, - limit, - HostFanoutScope::SeedHostsOnly, - )? - .into_iter() - .map(|entry| GraphRelatedLogEntry { - entry, - inclusion_reason: INCLUSION_GRAPH_RELATED.to_string(), - resolver_status: ResolverStatus::Resolved, - fallback_kind: None, - }), - ); - } - - // Explicit degraded host-context fallback: a service topic whose - // instance predicates matched no rows falls back to the instances' host - // context, annotated (`explicit_degraded_host_context`) — never silent. - if logs.is_empty() && !instance_keys.is_empty() && generic_seeds.is_empty() { - let hosts: Vec = instance_keys - .iter() - .filter_map(|key| { - let split = super::entity_resolution::split_service_instance_key(key); - if split.is_none() { - tracing::debug!( - key = %key, - "discarding non-canonical service_instance key in host-context fallback" - ); - } - split.map(|(host, _)| host.to_string()) - }) - .collect(); - if !hosts.is_empty() { - logs.extend( - queries_service_instances::search_logs_by_hostnames( - pool, - &hosts, - since, - until, - source_kinds, - limit, - )? - .into_iter() - .map(|entry| GraphRelatedLogEntry { - entry, - inclusion_reason: INCLUSION_HOST_CONTEXT.to_string(), - resolver_status: ResolverStatus::Degraded, - fallback_kind: Some(FALLBACK_EXPLICIT_DEGRADED_HOST_CONTEXT.to_string()), - }), - ); - } - } - - logs.sort_by(|a, b| { - b.entry - .timestamp - .cmp(&a.entry.timestamp) - .then_with(|| b.entry.id.cmp(&a.entry.id)) - }); - logs.dedup_by_key(|row| row.entry.id); - logs.truncate(limit); - - Ok(TopicGraphInputs { - resolved, - expansion, - discovered_hosts, - logs, - graph_walk_truncated, - }) -} - -fn topic_correlate_ai_project_fallback( - pool: &DbPool, - terms: &[String], - since: Option<&str>, - until: Option<&str>, - source_kinds: Option<&[SourceKind]>, - limit: usize, -) -> Result { - let conn = pool.get()?; - let predicates = terms - .iter() - .map(|_| "(lower(ai_project) = ? OR lower(ai_project) LIKE '%/' || ?)") - .collect::>() - .join(" OR "); - let mut sql = format!( - "SELECT DISTINCT ai_project FROM logs WHERE ai_project IS NOT NULL AND ({predicates})" - ); - let mut bindings = Vec::with_capacity(terms.len() * 2 + 2); - for term in terms { - bindings.push(rusqlite::types::Value::Text(term.clone())); - bindings.push(rusqlite::types::Value::Text(term.clone())); - } - if let Some(value) = since { - sql.push_str(" AND timestamp >= ?"); - bindings.push(rusqlite::types::Value::Text(value.to_string())); - } - if let Some(value) = until { - sql.push_str(" AND timestamp <= ?"); - bindings.push(rusqlite::types::Value::Text(value.to_string())); - } - sql.push_str(" ORDER BY ai_project LIMIT 32"); - - let projects = conn - .prepare(&sql)? - .query_map(rusqlite::params_from_iter(bindings.iter()), |row| { - row.get::<_, String>(0) - })? - .collect::>>()?; - drop(conn); - - let allowed_source_kinds = source_kinds.map(|kinds| { - kinds - .iter() - .map(|kind| kind.as_str()) - .collect::>() - }); - let mut logs = Vec::new(); - let mut discovered_hosts = Vec::new(); - let mut resolved = Vec::new(); - for project in projects { - let key = project - .rsplit('/') - .next() - .unwrap_or(&project) - .to_ascii_lowercase(); - resolved.push(ResolvedTopicEntity { - entity_type: super::graph::ENTITY_TYPE_AI_PROJECT.to_string(), - canonical_key: key, - match_kind: "exact", - resolver_status: ResolverStatus::Degraded, - }); - let rows = search_logs( - pool, - &SearchParams { - ai_project: Some(project), - since: since.map(str::to_string), - until: until.map(str::to_string), - limit: Some(limit.min(1000) as u32), - ..Default::default() - }, - )?; - for entry in rows { - if let Some(allowed) = &allowed_source_kinds { - let source_kind = entry - .metadata_json - .as_deref() - .and_then(|json| serde_json::from_str::(json).ok()) - .and_then(|value| value.get("source_kind")?.as_str().map(str::to_string)); - if !source_kind - .as_deref() - .is_some_and(|kind| allowed.contains(kind)) - { - continue; - } - } - discovered_hosts.push(entry.hostname.clone()); - logs.push(GraphRelatedLogEntry { - entry, - inclusion_reason: "direct_source_identity".to_string(), - resolver_status: ResolverStatus::Degraded, - fallback_kind: Some("direct_source_identity".to_string()), - }); - } - } - resolved.sort_by(|a, b| a.canonical_key.cmp(&b.canonical_key)); - resolved.dedup_by(|a, b| a.canonical_key == b.canonical_key); - discovered_hosts.sort(); - discovered_hosts.dedup(); - logs.sort_by(|a, b| b.entry.timestamp.cmp(&a.entry.timestamp)); - logs.dedup_by_key(|row| row.entry.id); - logs.truncate(limit); - - Ok(TopicGraphInputs { - resolved, - discovered_hosts, - logs, - ..Default::default() - }) -} - -const DEFAULT_AI_ABUSE_TERMS: &[&str] = &[ - "asshole", "bastard", "bitch", "biznitch", "bullshit", "crap", "damn", "dick", "fuck", - "fucked", "fucker", "fucking", "hell", "piss", "shit", "shitty", -]; - -pub fn search_ai_abuse(pool: &DbPool, params: &AiAbuseParams) -> Result { - let limit = params.limit.unwrap_or(20).clamp(1, 100) as usize; - let before = params.before.unwrap_or(2).min(20); - let after = params.after.unwrap_or(2).min(20); - let terms = normalized_abuse_terms(¶ms.terms); - if terms.len() > 16 { - anyhow::bail!("Too many abuse terms ({}); maximum is 16", terms.len()); - } - let conn = pool.get()?; - const CANDIDATE_CAP: usize = 10_000; - - let mut sql = String::from( - "WITH candidates(id) AS MATERIALIZED ( - SELECT l.id - FROM logs_fts - JOIN logs l ON l.id = logs_fts.rowid - WHERE logs_fts MATCH ?1 - AND l.ai_project IS NOT NULL AND l.ai_project != '' - AND l.ai_tool IS NOT NULL AND l.ai_tool != '' - AND l.ai_session_id IS NOT NULL AND l.ai_session_id != ''", - ); - let mut bindings = vec![rusqlite::types::Value::Text(abuse_fts_query(&terms))]; - let mut idx = 2usize; - - if let Some(project) = ¶ms.ai_project { - sql.push_str(&format!(" AND l.ai_project = ?{idx}")); - bindings.push(rusqlite::types::Value::Text(project.clone())); - idx += 1; - } - if let Some(tool) = ¶ms.ai_tool { - sql.push_str(&format!(" AND l.ai_tool = ?{idx}")); - bindings.push(rusqlite::types::Value::Text(tool.clone())); - idx += 1; - } - if let Some(from) = ¶ms.since { - sql.push_str(&format!(" AND l.timestamp >= ?{idx}")); - bindings.push(rusqlite::types::Value::Text(from.clone())); - idx += 1; - } - if let Some(to) = ¶ms.until { - sql.push_str(&format!(" AND l.timestamp <= ?{idx}")); - bindings.push(rusqlite::types::Value::Text(to.clone())); - } - sql.push_str(&format!( - " ORDER BY logs_fts.rowid DESC LIMIT {} - ) - SELECT l.id, l.timestamp, l.hostname, l.facility, l.severity, - l.app_name, l.process_id, l.message, l.received_at, l.source_ip, - l.ai_tool, l.ai_project, l.ai_session_id, l.ai_transcript_path, l.metadata_json - FROM candidates c - JOIN logs l ON l.id = c.id - ORDER BY l.timestamp DESC, l.id DESC", - CANDIDATE_CAP + 1 - )); - - let mut stmt = conn.prepare(&sql)?; - let candidate_rows = stmt - .query_map(rusqlite::params_from_iter(bindings.iter()), map_row)? - .collect::>>()?; - - let candidate_window_truncated = candidate_rows.len() > CANDIDATE_CAP; - let mut matches = Vec::new(); - let mut result_limit_truncated = false; - for entry in candidate_rows.iter().take(CANDIDATE_CAP) { - if let Some(term) = first_abuse_term(&entry.message, &terms) { - if matches.len() == limit { - result_limit_truncated = true; - break; - } - let (before_rows, after_rows) = ai_session_context(&conn, entry, before, after)?; - matches.push(AiAbuseMatch { - term, - entry: entry.clone(), - before: before_rows, - after: after_rows, - }); - } - } - - Ok(AiAbuseResult { - terms, - candidate_rows: candidate_rows.len().min(CANDIDATE_CAP), - candidate_cap: CANDIDATE_CAP, - candidate_window_truncated, - truncated: candidate_window_truncated || result_limit_truncated, - matches, - }) -} - -pub fn list_ai_tools(pool: &DbPool, params: &ListAiToolsParams) -> Result { - let conn = pool.get()?; - const LIMIT: usize = 100; - let mut sql = String::from( - "SELECT ai_tool, - COUNT(*) AS event_count, - COUNT(DISTINCT ai_session_id) AS session_count, - MIN(timestamp) AS first_seen, - MAX(timestamp) AS last_seen - FROM logs - WHERE ai_tool IS NOT NULL - AND ai_tool != ''", - ); - let mut bindings: Vec = vec![]; - let mut idx = 1usize; - - if let Some(project) = ¶ms.ai_project { - sql.push_str(&format!(" AND ai_project = ?{idx}")); - bindings.push(rusqlite::types::Value::Text(project.clone())); - idx += 1; - } - if let Some(from) = ¶ms.since { - sql.push_str(&format!(" AND timestamp >= ?{idx}")); - bindings.push(rusqlite::types::Value::Text(from.clone())); - idx += 1; - } - if let Some(to) = ¶ms.until { - sql.push_str(&format!(" AND timestamp <= ?{idx}")); - bindings.push(rusqlite::types::Value::Text(to.clone())); - } - sql.push_str(&format!( - " GROUP BY ai_tool ORDER BY event_count DESC, ai_tool ASC LIMIT {}", - LIMIT + 1 - )); - - let mut stmt = conn.prepare(&sql)?; - let mut tools = stmt - .query_map(rusqlite::params_from_iter(bindings.iter()), |row| { - Ok(AiToolInventoryEntry { - tool: row.get(0)?, - event_count: row.get(1)?, - session_count: row.get(2)?, - first_seen: row.get(3)?, - last_seen: row.get(4)?, - }) - })? - .collect::>>()?; - let truncated = truncate_to_limit(&mut tools, LIMIT); - Ok(ListAiToolsResult { - total_tools: tools.len(), - truncated, - tools, - }) -} - -pub fn list_ai_projects( - pool: &DbPool, - params: &ListAiProjectsParams, -) -> Result { - let conn = pool.get()?; - const LIMIT: usize = 200; - let mut sql = String::from( - "SELECT ai_project, - GROUP_CONCAT(DISTINCT ai_tool) AS tools, - COUNT(*) AS event_count, - COUNT(DISTINCT ai_session_id) AS session_count, - MIN(timestamp) AS first_seen, - MAX(timestamp) AS last_seen - FROM logs - WHERE ai_project IS NOT NULL - AND ai_project != ''", - ); - let mut bindings: Vec = vec![]; - let mut idx = 1usize; - - if let Some(tool) = ¶ms.ai_tool { - sql.push_str(&format!(" AND ai_tool = ?{idx}")); - bindings.push(rusqlite::types::Value::Text(tool.clone())); - idx += 1; - } - if let Some(from) = ¶ms.since { - sql.push_str(&format!(" AND timestamp >= ?{idx}")); - bindings.push(rusqlite::types::Value::Text(from.clone())); - idx += 1; - } - if let Some(to) = ¶ms.until { - sql.push_str(&format!(" AND timestamp <= ?{idx}")); - bindings.push(rusqlite::types::Value::Text(to.clone())); - } - sql.push_str(&format!( - " GROUP BY ai_project ORDER BY event_count DESC, ai_project ASC LIMIT {}", - LIMIT + 1 - )); - - let mut stmt = conn.prepare(&sql)?; - let mut projects = stmt - .query_map(rusqlite::params_from_iter(bindings.iter()), |row| { - let tools = row - .get::<_, Option>(1)? - .unwrap_or_default() - .split(',') - .filter(|value| !value.is_empty()) - .map(ToString::to_string) - .collect(); - Ok(AiProjectInventoryEntry { - project: row.get(0)?, - tools, - event_count: row.get(2)?, - session_count: row.get(3)?, - first_seen: row.get(4)?, - last_seen: row.get(5)?, - }) - })? - .collect::>>()?; - let truncated = truncate_to_limit(&mut projects, LIMIT); - Ok(ListAiProjectsResult { - total_projects: projects.len(), - truncated, - projects, - }) -} - -fn truncate_to_limit(values: &mut Vec, limit: usize) -> bool { - let truncated = values.len() > limit; - values.truncate(limit); - truncated -} - -pub fn search_ai_incidents(pool: &DbPool, params: &AiIncidentParams) -> Result { - use std::collections::HashMap; - - let limit = params.limit.unwrap_or(20).clamp(1, 100) as usize; - let window_secs = i64::from(params.window_minutes.unwrap_or(10).clamp(1, 120)) * 60; - let terms = normalized_abuse_terms(¶ms.terms); - const CANDIDATE_CAP: usize = 10_000; - - let conn = pool.get()?; - let (sql, bindings) = ai_incident_anchor_sql(params, &terms, CANDIDATE_CAP); - - // Fetch candidate abuse anchor rows (same FTS path as search_ai_abuse, - // no per-hit context needed here). - struct AnchorRow { - id: i64, - timestamp: String, - hostname: String, - tool: String, - project: String, - session_id: String, - message: String, - } - - let mut stmt = conn.prepare(&sql)?; - let candidate_rows: Vec = stmt - .query_map(rusqlite::params_from_iter(bindings.iter()), |row| { - Ok(AnchorRow { - id: row.get(0)?, - timestamp: row.get(1)?, - hostname: row.get(2)?, - tool: row.get(3)?, - project: row.get(4)?, - session_id: row.get(5)?, - message: row.get(6)?, - }) - })? - .collect::>>()?; - - let candidate_window_truncated = candidate_rows.len() > CANDIDATE_CAP; - let raw_candidate_count = candidate_rows.len(); - - // Group by (project, tool, session_id, hostname) + window-minute buckets. - // Key: (project, tool, session_id, hostname, window_bucket) - // window_bucket = unix_secs / window_secs * window_secs (floor to window boundary) - type GroupKey = (String, String, String, String, i64); - let mut groups: HashMap> = HashMap::new(); - - for row in candidate_rows.iter().take(CANDIDATE_CAP) { - // Parse timestamp to unix seconds for bucketing. - let bucket = chrono::DateTime::parse_from_rfc3339(&row.timestamp) - .map(|dt| { - let secs = dt.timestamp(); - (secs / window_secs) * window_secs - }) - .unwrap_or(0); - let key = ( - row.project.clone(), - row.tool.clone(), - row.session_id.clone(), - row.hostname.clone(), - bucket, - ); - groups.entry(key).or_default().push(row); - } - - // Build incidents from groups. - let mut incidents: Vec = groups - .into_iter() - .map( - |((project, tool, session_id, hostname, _bucket), anchors)| { - let abuse_count = anchors.len(); - let first_seen = anchors - .first() - .map(|r| r.timestamp.clone()) - .unwrap_or_default(); - let last_seen = anchors - .last() - .map(|r| r.timestamp.clone()) - .unwrap_or_default(); - - // duration in seconds - let duration_secs = { - let t0 = chrono::DateTime::parse_from_rfc3339(&first_seen) - .map(|dt| dt.timestamp()) - .unwrap_or(0); - let t1 = chrono::DateTime::parse_from_rfc3339(&last_seen) - .map(|dt| dt.timestamp()) - .unwrap_or(0); - (t1 - t0).max(0) - }; - - // Collect unique terms found in this group's messages. - let mut found_terms: Vec = terms - .iter() - .filter(|term| { - anchors.iter().any(|r| { - first_abuse_term(&r.message, std::slice::from_ref(term)).is_some() - }) - }) - .cloned() - .collect(); - found_terms.sort(); - found_terms.dedup(); - - let mut anchor_ids: Vec = anchors.iter().map(|r| r.id).collect(); - anchor_ids.sort(); - - // Score: abuse_count dominates; density and term variety boost. - let density = if duration_secs > 0 { - abuse_count as f64 / (duration_secs as f64 / 60.0) - } else { - abuse_count as f64 - }; - let term_variety = found_terms.len() as f64; - let priority_score = abuse_count as f64 * 10.0 + density * 2.0 + term_variety; - - // Compare the f64 directly: `as u64` truncates and maps NaN - // to 0, which would mislabel a pathological score as "low" - // (full-review QL2). - let priority_label = if priority_score < 15.0 { - "low" - } else if priority_score < 30.0 { - "medium" - } else if priority_score < 50.0 { - "high" - } else { - "critical" - } - .to_string(); - - // Stable incident ID using a deterministic hash of session identity + anchor IDs. - let incident_id = { - use std::collections::hash_map::DefaultHasher; - use std::hash::{Hash, Hasher}; - let mut h = DefaultHasher::new(); - project.hash(&mut h); - tool.hash(&mut h); - session_id.hash(&mut h); - hostname.hash(&mut h); - for id in &anchor_ids { - id.hash(&mut h); - } - format!("inc-{:016x}", h.finish()) - }; - - AbuseIncident { - incident_id, - project, - tool, - session_id, - hostname, - first_seen, - last_seen, - duration_secs, - abuse_count, - terms: found_terms, - anchor_ids, - priority_score, - priority_label, - window_minutes: (window_secs / 60) as u32, - } - }, - ) - .collect(); - - // Sort by priority_score descending, then last_seen descending. - // total_cmp is a total order (NaN sorts deterministically) — the - // partial_cmp/unwrap_or(Equal) idiom can produce a non-total order if a - // NaN ever sneaks into a score (full-review QL3). - incidents.sort_by(|a, b| { - b.priority_score - .total_cmp(&a.priority_score) - .then_with(|| b.last_seen.cmp(&a.last_seen)) - }); - - let total_incidents = incidents.len(); - let truncated = total_incidents > limit || candidate_window_truncated; - incidents.truncate(limit); - - Ok(AiIncidentResult { - incidents, - total_incidents, - candidate_rows: raw_candidate_count.min(CANDIDATE_CAP), - candidate_cap: CANDIDATE_CAP, - candidate_window_truncated, - truncated, - }) -} - -fn ai_incident_anchor_sql( - params: &AiIncidentParams, - terms: &[String], - candidate_cap: usize, -) -> (String, Vec) { - let mut sql = String::from( - "WITH candidates(id) AS MATERIALIZED ( - SELECT l.id - FROM logs_fts - JOIN logs l ON l.id = logs_fts.rowid - WHERE logs_fts MATCH ?1 - AND l.ai_project IS NOT NULL AND l.ai_project != '' - AND l.ai_tool IS NOT NULL AND l.ai_tool != '' - AND l.ai_session_id IS NOT NULL AND l.ai_session_id != ''", - ); - let mut bindings = vec![rusqlite::types::Value::Text(abuse_fts_query(terms))]; - let mut idx = 2usize; - - if let Some(project) = ¶ms.ai_project { - sql.push_str(&format!(" AND l.ai_project = ?{idx}")); - bindings.push(rusqlite::types::Value::Text(project.clone())); - idx += 1; - } - if let Some(tool) = ¶ms.ai_tool { - sql.push_str(&format!(" AND l.ai_tool = ?{idx}")); - bindings.push(rusqlite::types::Value::Text(tool.clone())); - idx += 1; - } - if let Some(from) = ¶ms.since { - sql.push_str(&format!(" AND l.timestamp >= ?{idx}")); - bindings.push(rusqlite::types::Value::Text(from.clone())); - idx += 1; - } - if let Some(to) = ¶ms.until { - sql.push_str(&format!(" AND l.timestamp <= ?{idx}")); - bindings.push(rusqlite::types::Value::Text(to.clone())); - } - let _ = idx; - sql.push_str(&format!( - " ORDER BY logs_fts.rowid ASC LIMIT {} - ) - SELECT l.id, l.timestamp, l.hostname, - l.ai_tool, l.ai_project, l.ai_session_id, l.message - FROM candidates c - JOIN logs l ON l.id = c.id", - candidate_cap + 1 - )); - (sql, bindings) -} - -pub fn investigate_ai_incidents( - pool: &DbPool, - params: &AiInvestigateParams, -) -> Result { - let limit = params.limit.unwrap_or(3).clamp(1, 10) as usize; - let incident_lookup_limit = if params.incident_id.is_some() { - 100 - } else { - limit as u32 - }; - let corr_mins = i64::from(params.correlation_window_minutes.unwrap_or(5).clamp(1, 120)); - - // Reuse incident grouping to find the top incidents. Exact incident - // assessment may target an ID outside the top investigation page, so it - // searches up to the incident-list cap and then builds one evidence bundle. - let incident_result = search_ai_incidents( - pool, - &AiIncidentParams { - ai_project: params.ai_project.clone(), - ai_tool: params.ai_tool.clone(), - since: params.since.clone(), - until: params.until.clone(), - limit: Some(incident_lookup_limit), - window_minutes: params.window_minutes, - terms: params.terms.clone(), - }, - )?; - let total_incidents = incident_result.total_incidents; - let truncated = incident_result.truncated; - let incidents = if let Some(incident_id) = ¶ms.incident_id { - incident_result - .incidents - .into_iter() - .filter(|incident| incident.incident_id == *incident_id) - .collect() - } else { - incident_result.incidents - }; - - let conn = pool.get()?; - let mut evidence = Vec::with_capacity(incidents.len()); - - for incident in incidents { - const TRANSCRIPT_CAP: usize = 20; - const NEARBY_CAP: usize = 50; - - // Fetch anchor log entries. - let anchors = if incident.anchor_ids.is_empty() { - Vec::new() - } else { - let placeholders: Vec = (1..=incident.anchor_ids.len()) - .map(|i| format!("?{i}")) - .collect(); - let sql = format!( - "SELECT id, timestamp, hostname, facility, severity, app_name, - process_id, message, received_at, source_ip, - ai_tool, ai_project, ai_session_id, ai_transcript_path, metadata_json - FROM logs WHERE id IN ({}) ORDER BY timestamp ASC", - placeholders.join(",") - ); - let mut stmt = conn.prepare(&sql)?; - - stmt.query_map( - rusqlite::params_from_iter( - incident - .anchor_ids - .iter() - .map(|id| rusqlite::types::Value::Integer(*id)), - ), - map_row, - )? - .collect::>>()? - }; - - // Transcript context: entries in the same session before first anchor and after last anchor. - let (transcript_before, transcript_before_truncated) = if let Some(first) = anchors.first() - { - let rows = { - let mut stmt = conn.prepare( - "SELECT id, timestamp, hostname, facility, severity, app_name, - process_id, message, received_at, source_ip, - ai_tool, ai_project, ai_session_id, ai_transcript_path, - metadata_json - FROM logs - WHERE ai_session_id = ?1 AND ai_project = ?2 AND ai_tool = ?3 - AND timestamp < ?4 - ORDER BY timestamp DESC - LIMIT 21", - )?; - - stmt.query_map( - rusqlite::params![ - &incident.session_id, - &incident.project, - &incident.tool, - &first.timestamp, - ], - map_row, - )? - .collect::>>()? - }; - let truncated = rows.len() > TRANSCRIPT_CAP; - let mut out = rows; - out.truncate(TRANSCRIPT_CAP); - out.reverse(); // chronological order - (out, truncated) - } else { - (Vec::new(), false) - }; - - let (transcript_after, transcript_after_truncated) = if let Some(last) = anchors.last() { - let rows = { - let mut stmt = conn.prepare( - "SELECT id, timestamp, hostname, facility, severity, app_name, - process_id, message, received_at, source_ip, - ai_tool, ai_project, ai_session_id, ai_transcript_path, metadata_json - FROM logs - WHERE ai_session_id = ?1 AND ai_project = ?2 AND ai_tool = ?3 - AND timestamp > ?4 - ORDER BY timestamp ASC - LIMIT 21", - )?; - - stmt.query_map( - rusqlite::params![ - &incident.session_id, - &incident.project, - &incident.tool, - &last.timestamp, - ], - map_row, - )? - .collect::>>()? - }; - let truncated = rows.len() > TRANSCRIPT_CAP; - let mut out = rows; - out.truncate(TRANSCRIPT_CAP); - (out, truncated) - } else { - (Vec::new(), false) - }; - - // Nearby non-AI logs in the correlation window. - let (nearby_logs, nearby_logs_truncated) = { - // Window: corr_mins before first_seen through corr_mins after last_seen. - let win_from = chrono::DateTime::parse_from_rfc3339(&incident.first_seen) - .map(|dt| { - use chrono::Duration; - (dt.with_timezone(&chrono::Utc) - Duration::minutes(corr_mins)) - .format("%Y-%m-%dT%H:%M:%S%.3fZ") - .to_string() - }) - .unwrap_or_else(|_| incident.first_seen.clone()); - let win_to = chrono::DateTime::parse_from_rfc3339(&incident.last_seen) - .map(|dt| { - use chrono::Duration; - (dt.with_timezone(&chrono::Utc) + Duration::minutes(corr_mins)) - .format("%Y-%m-%dT%H:%M:%S%.3fZ") - .to_string() - }) - .unwrap_or_else(|_| incident.last_seen.clone()); - - let mut stmt = conn.prepare( - "SELECT id, timestamp, hostname, facility, severity, app_name, - process_id, message, received_at, source_ip, - ai_tool, ai_project, ai_session_id, ai_transcript_path, metadata_json - FROM logs - WHERE timestamp >= ?1 AND timestamp <= ?2 - AND (ai_project IS NULL OR ai_project = '') - ORDER BY timestamp ASC - LIMIT 51", - )?; - let rows = stmt - .query_map(rusqlite::params![win_from, win_to], map_row)? - .collect::>>()?; - let truncated = rows.len() > NEARBY_CAP; - let mut out = rows; - out.truncate(NEARBY_CAP); - (out, truncated) - }; - - // Nearby errors: subset of nearby_logs with severity warning+. - let error_sevs = ["emergency", "alert", "critical", "error", "warning"]; - let nearby_errors: Vec = nearby_logs - .iter() - .filter(|e| error_sevs.contains(&e.severity.as_str())) - .cloned() - .collect(); - - evidence.push(IncidentEvidence { - incident, - transcript_before, - transcript_before_truncated, - transcript_after, - transcript_after_truncated, - anchors, - nearby_logs, - nearby_logs_truncated, - nearby_errors, - }); - } - - Ok(AiInvestigateResult { - evidence, - total_incidents, - truncated, - }) -} - -fn normalized_abuse_terms(custom_terms: &[String]) -> Vec { - let source: Vec = if custom_terms.is_empty() { - DEFAULT_AI_ABUSE_TERMS - .iter() - .map(|term| (*term).to_string()) - .collect() - } else { - custom_terms.to_vec() - }; - - let mut terms = source - .into_iter() - .map(|term| term.trim().to_ascii_lowercase()) - .filter(|term| { - !term.is_empty() - && term.len() <= 64 - && term - .chars() - .all(|ch| ch.is_ascii_alphanumeric() || ch == '-' || ch == '_') - }) - .collect::>(); - terms.sort(); - terms.dedup(); - if terms.is_empty() { - DEFAULT_AI_ABUSE_TERMS - .iter() - .map(|term| (*term).to_string()) - .collect() - } else { - terms - } -} - -fn abuse_fts_query(terms: &[String]) -> String { - // FTS5 escapes a literal " inside a phrase by doubling it: "" → match one " - terms - .iter() - .map(|term| format!("\"{}\"", term.replace('"', "\"\""))) - .collect::>() - .join(" OR ") -} - -fn first_abuse_term(message: &str, terms: &[String]) -> Option { - let lower = message.to_ascii_lowercase(); - terms - .iter() - .filter_map(|term| first_term_index(&lower, term).map(|idx| (idx, term))) - .min_by_key(|(idx, _)| *idx) - .map(|(_, term)| term.clone()) -} - -fn first_term_index(message: &str, term: &str) -> Option { - let mut offset = 0usize; - while let Some(relative) = message[offset..].find(term) { - let start = offset + relative; - let end = start + term.len(); - if is_abuse_boundary(message[..start].chars().next_back()) - && is_abuse_boundary(message[end..].chars().next()) - { - return Some(start); - } - offset = end; - } - None -} - -fn is_abuse_boundary(ch: Option) -> bool { - ch.is_none_or(|ch| !ch.is_ascii_alphanumeric() && ch != '_') -} - -fn ai_session_context( - conn: &rusqlite::Connection, - entry: &LogEntry, - before: u32, - after: u32, -) -> Result<(Vec, Vec)> { - let Some(tool) = entry.ai_tool.as_deref() else { - return Ok((Vec::new(), Vec::new())); - }; - let Some(project) = entry.ai_project.as_deref() else { - return Ok((Vec::new(), Vec::new())); - }; - let Some(session_id) = entry.ai_session_id.as_deref() else { - return Ok((Vec::new(), Vec::new())); - }; - - let mut before_stmt = conn.prepare( - "SELECT id, timestamp, hostname, facility, severity, - app_name, process_id, message, received_at, source_ip, - ai_tool, ai_project, ai_session_id, ai_transcript_path, metadata_json - FROM logs - WHERE hostname = ?1 - AND ai_tool = ?2 - AND ai_project = ?3 - AND ai_session_id = ?4 - AND (timestamp < ?5 OR (timestamp = ?5 AND id < ?6)) - ORDER BY timestamp DESC, id DESC - LIMIT ?7", - )?; - let mut before_rows = before_stmt - .query_map( - params![ - &entry.hostname, - tool, - project, - session_id, - &entry.timestamp, - entry.id, - before - ], - map_row, - )? - .collect::>>()?; - before_rows.reverse(); - - let mut after_stmt = conn.prepare( - "SELECT id, timestamp, hostname, facility, severity, - app_name, process_id, message, received_at, source_ip, - ai_tool, ai_project, ai_session_id, ai_transcript_path, metadata_json - FROM logs - WHERE hostname = ?1 - AND ai_tool = ?2 - AND ai_project = ?3 - AND ai_session_id = ?4 - AND (timestamp > ?5 OR (timestamp = ?5 AND id > ?6)) - ORDER BY timestamp ASC, id ASC - LIMIT ?7", - )?; - let after_rows = after_stmt - .query_map( - params![ - &entry.hostname, - tool, - project, - session_id, - &entry.timestamp, - entry.id, - after - ], - map_row, - )? - .collect::>>()?; - - Ok((before_rows, after_rows)) -} - -/// Get database stats -pub fn get_stats(pool: &DbPool, config: &StorageConfig) -> Result { - get_stats_with_options(pool, config, false) -} - -/// `get_stats`, but `include_fts_diagnostics` controls whether the -/// `phantom_fts_rows` field is computed. That value requires -/// `COUNT(*) FROM logs_fts` — an external-content FTS5 index scan that is -/// cheap on small DBs but expensive on very large ones (the index has no -/// O(1) row counter). The default `stats` path passes `false` so the common -/// query stays fast; callers that specifically need the FTS merge-health -/// diagnostic pass `true`. -pub fn get_stats_with_options( - pool: &DbPool, - config: &StorageConfig, - include_fts_diagnostics: bool, -) -> Result { - let metrics = get_storage_metrics(pool, config)?; - let write_blocked = exceeds_trigger(&metrics, config); - let mut conn = pool.get()?; - - // Deferred read transaction ensures the log stats form a consistent snapshot - let tx = conn.transaction_with_behavior(rusqlite::TransactionBehavior::Deferred)?; - // total_logs reads the timeline_hourly rollup (O(#buckets)) plus the live - // delta of rows ingested since the rollup watermark, instead of the O(#rows) - // `COUNT(*) FROM logs` (~7s on multi-million-row DBs). The rollup covers - // `logs.id <= source_max_id`; the delta covers `id > source_max_id`, so the - // sum is exact at the current snapshot for ADDs. It is NOT perfectly exact - // under concurrent retention DELETEs of rows the rollup already counted: the - // retention prune (spawn_retention_task) trims whole stale buckets, leaving - // at most a transient single-boundary-hour overcount — accepted as a - // negligible drift for a stats counter. (bead syslog-mcp-kcvq) - let rollup_max_id: i64 = tx.query_row( - "SELECT source_max_id FROM timeline_hourly_meta WHERE id = 1", - [], - |r| r.get(0), - )?; - let rollup_total: i64 = tx.query_row( - "SELECT COALESCE(SUM(event_count), 0) FROM timeline_hourly", - [], - |r| r.get(0), - )?; - let live_delta: i64 = tx.query_row( - "SELECT COUNT(*) FROM logs WHERE id > ?1", - [rollup_max_id], - |r| r.get(0), - )?; - let total_logs: i64 = rollup_total + live_delta; - let total_hosts: i64 = tx.query_row("SELECT COUNT(*) FROM hosts", [], |r| r.get(0))?; - let phantom_fts_rows = if include_fts_diagnostics { - let fts_rows: i64 = tx - .query_row("SELECT COUNT(*) FROM logs_fts", [], |r| r.get(0)) - .unwrap_or(0); - Some((fts_rows - total_logs).max(0)) - } else { - None - }; - // MIN/MAX return a single nullable row; use get::<_, Option<_>> so NULL becomes - // None while real query errors (e.g. missing table) still propagate via `?`. - // Both use the covering index idx_logs_timestamp (SEARCH, O(log n)). - let oldest: Option = tx.query_row("SELECT MIN(timestamp) FROM logs", [], |r| { - r.get::<_, Option>(0) - })?; - let newest: Option = tx.query_row("SELECT MAX(timestamp) FROM logs", [], |r| { - r.get::<_, Option>(0) - })?; - tx.finish()?; - - Ok(DbStats { - total_logs, - total_hosts, - oldest_log: oldest, - newest_log: newest, - logical_db_size_mb: format!("{:.2}", metrics.logical_db_size_bytes as f64 / 1_048_576.0), - physical_db_size_mb: format!("{:.2}", metrics.physical_db_size_bytes as f64 / 1_048_576.0), - free_disk_mb: metrics - .free_disk_bytes - .map(|bytes| format!("{:.2}", bytes as f64 / 1_048_576.0)), - max_db_size_mb: config.max_db_size_mb, - min_free_disk_mb: config.min_free_disk_mb, - write_blocked, - phantom_fts_rows, - }) -} - -/// Syslog severity level names ordered by numeric value (0=emerg, 7=debug). -/// Used by both the MCP layer (for threshold filtering) and the syslog parser (for decoding). -pub const SEVERITY_LEVELS: &[&str] = &[ - "emerg", "alert", "crit", "err", "warning", "notice", "info", "debug", -]; - -/// Convert a severity name to its numeric syslog level (0=emerg, 7=debug). -/// Accepts the canonical RFC 5424 keywords (case-insensitive) plus common -/// aliases: `error`/`fatal`/`panic` for `err`, `warn` for `warning`, -/// `critical` for `crit`, `emergency` for `emerg`. -/// Returns `None` for unrecognised names. -pub fn severity_to_num(s: &str) -> Option { - let canonical = match s.to_ascii_lowercase().as_str() { - "emergency" => "emerg", - "critical" => "crit", - "error" | "fatal" | "panic" => "err", - "warn" => "warning", - other => { - return SEVERITY_LEVELS - .iter() - .position(|&l| l == other) - .map(|i| i as u8); - } - }; - SEVERITY_LEVELS - .iter() - .position(|&l| l == canonical) - .map(|i| i as u8) -} - -fn append_filters( - sql: &mut String, - bindings: &mut Vec, - idx: &mut usize, - params: &SearchParams, -) { - if let Some(ref h) = params.host { - append_host_selector(sql, bindings, idx, "l.hostname", h); - } - if let Some(ref source_ip) = params.source { - sql.push_str(&format!(" AND l.source_ip = ?{}", *idx)); - bindings.push(rusqlite::types::Value::Text(source_ip.clone())); - *idx += 1; - } - if let Some(ref prefix) = params.source_ip_prefix - && !prefix.is_empty() - { - sql.push_str(&format!(" AND l.source_ip >= ?{}", *idx)); - bindings.push(rusqlite::types::Value::Text(prefix.clone())); - *idx += 1; - if let Some(upper) = prefix_upper_bound(prefix) { - sql.push_str(&format!(" AND l.source_ip < ?{}", *idx)); - bindings.push(rusqlite::types::Value::Text(upper)); - *idx += 1; - } - } - if let Some(ref prefixes) = params.source_ip_prefixes { - let prefixes = prefixes - .iter() - .filter(|prefix| !prefix.is_empty()) - .collect::>(); - if !prefixes.is_empty() { - sql.push_str(" AND ("); - for (position, prefix) in prefixes.iter().enumerate() { - if position > 0 { - sql.push_str(" OR "); - } - sql.push_str(&format!("(l.source_ip >= ?{}", *idx)); - bindings.push(rusqlite::types::Value::Text((*prefix).clone())); - *idx += 1; - if let Some(upper) = prefix_upper_bound(prefix) { - sql.push_str(&format!(" AND l.source_ip < ?{}", *idx)); - bindings.push(rusqlite::types::Value::Text(upper)); - *idx += 1; - } - sql.push(')'); - } - sql.push(')'); - } - } - if let Some(ref s) = params.severity { - sql.push_str(&format!(" AND l.severity = ?{}", *idx)); - bindings.push(rusqlite::types::Value::Text(s.clone())); - *idx += 1; - } - if let Some(ref levels) = params.severity_in - && !levels.is_empty() - { - let placeholders: Vec = levels - .iter() - .enumerate() - .map(|(i, _)| format!("?{}", *idx + i)) - .collect(); - sql.push_str(&format!(" AND l.severity IN ({})", placeholders.join(", "))); - for level in levels { - bindings.push(rusqlite::types::Value::Text(level.clone())); - *idx += 1; - } - } - if let Some(ref a) = params.app { - sql.push_str(&format!(" AND l.app_name = ?{}", *idx)); - bindings.push(rusqlite::types::Value::Text(a.clone())); - *idx += 1; - } - if let Some(ref f) = params.facility { - sql.push_str(&format!(" AND l.facility = ?{}", *idx)); - bindings.push(rusqlite::types::Value::Text(f.clone())); - *idx += 1; - } - if let Some(ref f) = params.exclude_facility { - sql.push_str(&format!( - " AND (l.facility IS NULL OR l.facility != ?{})", - *idx - )); - bindings.push(rusqlite::types::Value::Text(f.clone())); - *idx += 1; - } - if let Some(ref pid) = params.process_id { - sql.push_str(&format!(" AND l.process_id = ?{}", *idx)); - bindings.push(rusqlite::types::Value::Text(pid.clone())); - *idx += 1; - } - if let Some(ref from) = params.since { - sql.push_str(&format!(" AND l.timestamp >= ?{}", *idx)); - bindings.push(rusqlite::types::Value::Text(from.clone())); - *idx += 1; - } - if let Some(ref to) = params.until { - sql.push_str(&format!(" AND l.timestamp <= ?{}", *idx)); - bindings.push(rusqlite::types::Value::Text(to.clone())); - *idx += 1; - } - if let Some(ref from) = params.received_since { - sql.push_str(&format!(" AND l.received_at >= ?{}", *idx)); - bindings.push(rusqlite::types::Value::Text(from.clone())); - *idx += 1; - } - if let Some(ref to) = params.received_until { - sql.push_str(&format!(" AND l.received_at <= ?{}", *idx)); - bindings.push(rusqlite::types::Value::Text(to.clone())); - *idx += 1; - } - if let Some(ref tool) = params.ai_tool { - sql.push_str(&format!(" AND l.ai_tool = ?{}", *idx)); - bindings.push(rusqlite::types::Value::Text(tool.clone())); - *idx += 1; - } - if let Some(ref project) = params.ai_project { - sql.push_str(&format!(" AND l.ai_project = ?{}", *idx)); - bindings.push(rusqlite::types::Value::Text(project.clone())); - *idx += 1; - } - if let Some(ref session_id) = params.ai_session_id { - sql.push_str(&format!(" AND l.ai_session_id = ?{}", *idx)); - bindings.push(rusqlite::types::Value::Text(session_id.clone())); - *idx += 1; - } - if let Some(ref event_action) = params.event_action { - sql.push_str(&format!(" AND l.event_action = ?{}", *idx)); - bindings.push(rusqlite::types::Value::Text(event_action.clone())); - *idx += 1; - } - if params.exclude_ai { - sql.push_str( - " AND (l.ai_project IS NULL OR l.ai_project = '') - AND (l.ai_tool IS NULL OR l.ai_tool = '') - AND (l.ai_session_id IS NULL OR l.ai_session_id = '') - AND (l.ai_transcript_path IS NULL OR l.ai_transcript_path = '') - AND ( - l.app_name IS NULL - OR l.app_name NOT IN ( - 'ai-transcript', - 'claude-transcript', - 'codex-transcript', - 'gemini-transcript' - ) - )", - ); - } -} - -/// Append an indexed hostname selector that accepts the canonical names -/// returned by [`list_hosts`]. Normalization is confined to the tiny `hosts` -/// aggregate table; the outer log lookup still uses exact stored values and -/// can probe the existing hostname indexes. -fn append_host_selector( - sql: &mut String, - bindings: &mut Vec, - idx: &mut usize, - column: &str, - hostname: &str, -) { - let param = format!("?{}", *idx); - let normalized_param = format!("lower(rtrim(trim({param}), '.'))"); - let normalized_host = "lower(rtrim(trim(h.hostname), '.'))"; - let normalized_bare = "lower(rtrim(trim(bare.hostname), '.'))"; - sql.push_str(&format!( - " AND {column} IN ( - SELECT h.hostname - FROM hosts h - WHERE {normalized_host} = {normalized_param} - OR ( - {normalized_host} LIKE {normalized_param} || '.%' - AND EXISTS ( - SELECT 1 - FROM hosts bare - WHERE {normalized_bare} = {normalized_param} - AND instr({normalized_bare}, '.') = 0 - ) - ) - )" - )); - bindings.push(rusqlite::types::Value::Text(hostname.to_string())); - *idx += 1; -} - -fn prefix_upper_bound(prefix: &str) -> Option { - let mut bytes = prefix.as_bytes().to_vec(); - for idx in (0..bytes.len()).rev() { - if bytes[idx] != u8::MAX { - bytes[idx] += 1; - bytes.truncate(idx + 1); - return String::from_utf8(bytes).ok(); - } - } - None -} - -pub(super) fn map_row(row: &rusqlite::Row) -> rusqlite::Result { - map_row_offset(row, 0) -} - -fn map_row_offset(row: &rusqlite::Row, offset: usize) -> rusqlite::Result { - Ok(LogEntry { - id: row.get(offset)?, - timestamp: row.get(offset + 1)?, - hostname: row.get(offset + 2)?, - facility: row.get(offset + 3)?, - severity: row.get(offset + 4)?, - app_name: row.get(offset + 5)?, - process_id: row.get(offset + 6)?, - message: row.get(offset + 7)?, - received_at: row.get(offset + 8)?, - source_ip: row.get(offset + 9)?, - ai_tool: row.get(offset + 10)?, - ai_project: row.get(offset + 11)?, - ai_session_id: row.get(offset + 12)?, - ai_transcript_path: row.get(offset + 13)?, - metadata_json: row.get(offset + 14)?, - }) -} - -/// Map a row that includes the unparsed `raw` syslog frame (column index 8). -pub(super) fn map_row_with_raw( - row: &rusqlite::Row, -) -> rusqlite::Result { - Ok(super::analytics::LogEntryWithRaw { - id: row.get(0)?, - timestamp: row.get(1)?, - hostname: row.get(2)?, - facility: row.get(3)?, - severity: row.get(4)?, - app_name: row.get(5)?, - process_id: row.get(6)?, - message: row.get(7)?, - raw: row.get(8)?, - received_at: row.get(9)?, - source_ip: row.get(10)?, - ai_tool: row.get(11)?, - ai_project: row.get(12)?, - ai_session_id: row.get(13)?, - ai_transcript_path: row.get(14)?, - metadata_json: row.get(15)?, - }) -} - -// --------------------------------------------------------------------------- -// RAG v1: similar_incidents, incident_context -// --------------------------------------------------------------------------- - -use super::models::{ - AppLogCount, CorrelatedSession, IncidentCluster, IncidentContextParams, IncidentContextResult, - SeverityCount, SimilarIncidentsParams, SimilarIncidentsResult, -}; - -/// Return incident clusters from FTS5 hits, grouped by hostname + app_name in -/// non-overlapping windows of `window_minutes` minutes (default 30). -/// -/// Algorithm: -/// 1. FTS5 MATCH over non-AI log rows (optionally filtered by host/app/time). -/// 2. Group hits by (hostname, app_name, floor(unix_epoch / window_secs)). -/// 3. For each cluster: derive severity_peak (min numeric rank = highest sev), -/// collect up to 3 representative message snippets, and look up correlated -/// AI sessions whose transcript timestamps overlap the cluster window. -pub fn similar_incidents_clusters( - pool: &DbPool, - params: &SimilarIncidentsParams, -) -> Result { - validate_fts_query(¶ms.query)?; - - let conn = pool.get()?; - let window_minutes = params.window_minutes.unwrap_or(30).clamp(5, 120); - let limit = params.limit.unwrap_or(10).clamp(1, 50) as usize; - let window_secs = i64::from(window_minutes) * 60; - - // Build the FTS5 + optional filter query. - // Exclude AI transcript rows so clusters contain only system logs. - let mut sql = String::from( - "WITH hits AS MATERIALIZED ( - SELECT l.id, l.timestamp, l.hostname, l.app_name, l.severity, l.message - FROM logs_fts - JOIN logs l ON l.id = logs_fts.rowid - WHERE logs_fts MATCH ?1 - AND (l.ai_project IS NULL OR l.ai_project = '')", - ); - - let mut query_params = SqlParams::new(2); - query_params - .bindings - .push(rusqlite::types::Value::Text(params.query.clone())); - - if let Some(hostname) = ¶ms.host { - let idx = query_params.push_text(hostname.clone()); - sql.push_str(&format!(" AND l.hostname = ?{idx}")); - } - if let Some(app_name) = ¶ms.app { - let idx = query_params.push_text(app_name.clone()); - sql.push_str(&format!(" AND l.app_name = ?{idx}")); - } - if let Some(from) = ¶ms.since { - let idx = query_params.push_text(from.clone()); - sql.push_str(&format!(" AND l.timestamp >= ?{idx}")); - } - if let Some(to) = ¶ms.until { - let idx = query_params.push_text(to.clone()); - sql.push_str(&format!(" AND l.timestamp <= ?{idx}")); - } - // Apply severity_min filter: include only logs at or above the threshold. - if let Some(severity_min) = ¶ms.severity_min { - let threshold = severity_to_num(severity_min).ok_or_else(|| { - anyhow::anyhow!( - "invalid severity_min '{}': must be one of {}", - severity_min, - SEVERITY_LEVELS.join(", ") - ) - })?; - let levels_in: Vec = SEVERITY_LEVELS[..=threshold as usize] - .iter() - .map(|s| s.to_string()) - .collect(); - let placeholders: Vec = levels_in - .iter() - .map(|s| { - let idx = query_params.push_text(s.clone()); - format!("?{idx}") - }) - .collect(); - sql.push_str(&format!(" AND l.severity IN ({})", placeholders.join(", "))); - } - - sql.push_str(&format!( - " ORDER BY l.id DESC LIMIT {SIMILAR_INCIDENT_FTS_CANDIDATE_CAP} - ), - bucketed AS ( - SELECT - hostname, - app_name, - CAST(strftime('%s', timestamp) AS INTEGER) / {window_secs} AS bucket, - MIN(timestamp) AS window_start, - MAX(timestamp) AS window_end, - COUNT(*) AS log_count, - GROUP_CONCAT(severity, ',') AS severities, - GROUP_CONCAT(SUBSTR(message, 1, 256), '|||') AS messages - FROM hits - GROUP BY hostname, app_name, bucket - ) - SELECT hostname, app_name, window_start, window_end, log_count, severities, messages - FROM bucketed - ORDER BY log_count DESC, window_start DESC - LIMIT {}", - limit + 1 - )); - - let mut stmt = conn.prepare(&sql).map_err(|e| { - tracing::error!(error = %e, "similar_incidents_clusters prepare failed"); - anyhow::anyhow!("similar_incidents query failed") - })?; - let rows = stmt - .query_map( - rusqlite::params_from_iter(query_params.bindings.iter()), - |row| { - Ok(( - row.get::<_, String>(0)?, // hostname - row.get::<_, Option>(1)?, // app_name - row.get::<_, String>(2)?, // window_start - row.get::<_, String>(3)?, // window_end - row.get::<_, i64>(4)?, // log_count - row.get::<_, String>(5)?, // severities (comma-joined) - row.get::<_, String>(6)?, // messages (|||joined) - )) - }, - ) - .map_err(|e| { - tracing::error!(error = %e, "similar_incidents_clusters query failed"); - anyhow::anyhow!("similar_incidents query failed") - })?; - - // Collect raw cluster rows first; keep one extra to detect truncation. - struct RawCluster { - hostname: String, - app_name: Option, - window_start: String, - window_end: String, - log_count: i64, - severity_peak: String, - representative_messages: Vec, - } - let mut raw: Vec = Vec::new(); - for row in rows { - let (hostname, app_name, window_start, window_end, log_count, severities, messages) = - row.map_err(|e| { - tracing::error!(error = %e, "similar_incidents_clusters row mapping failed"); - anyhow::anyhow!("similar_incidents row mapping failed") - })?; - - // Find peak severity (lowest numeric value = highest severity). - let severity_peak = severities - .split(',') - .filter_map(|s| severity_to_num(s).map(|n| (n, s.to_string()))) - .min_by_key(|(n, _)| *n) - .map(|(_, s)| s) - .unwrap_or_else(|| "info".to_string()); - - // Collect up to 3 representative messages. - let representative_messages: Vec = messages - .split("|||") - .take(3) - .map(|m| m.to_string()) - .collect(); - - raw.push(RawCluster { - hostname, - app_name, - window_start, - window_end, - log_count, - severity_peak, - representative_messages, - }); - } - - // Detect truncation (we queried limit+1 rows) and trim to the true limit. - let truncated = raw.len() > limit; - raw.truncate(limit); - - // Build one UNION ALL query across all cluster windows so each window gets - // its own per-session match_count. This is O(1) roundtrips while keeping - // counts accurate (no global-span inflation when a session spans clusters). - let per_cluster_sessions = find_correlated_sessions_per_cluster( - &conn, - &raw.iter() - .map(|c| (c.window_start.as_str(), c.window_end.as_str())) - .collect::>(), - )?; - - let clusters: Vec = raw - .into_iter() - .map(|rc| { - let key = (rc.window_start.clone(), rc.window_end.clone()); - let correlated_sessions = per_cluster_sessions.get(&key).cloned().unwrap_or_default(); - IncidentCluster { - hostname: rc.hostname, - app_name: rc.app_name, - window_start: rc.window_start, - window_end: rc.window_end, - log_count: rc.log_count, - severity_peak: rc.severity_peak, - representative_messages: rc.representative_messages, - correlated_sessions, - } - }) - .collect(); - - let total_clusters = clusters.len(); - Ok(SimilarIncidentsResult { - query: params.query.clone(), - total_clusters, - truncated, - clusters, - }) -} - -/// Per-cluster session lookup using a single UNION ALL query so each cluster -/// window gets its own accurate match_count rather than an inflated global count. -/// Returns a map keyed by (window_start, window_end) → top-5 sessions. -fn find_correlated_sessions_per_cluster( - conn: &rusqlite::Connection, - windows: &[(&str, &str)], -) -> Result>> { - use std::collections::HashMap; - - if windows.is_empty() { - return Ok(HashMap::new()); - } - - // Build UNION ALL: one SELECT per cluster window, tagging each row with ws/we. - // Parameters use stride-2 (?{p} = ws, ?{p+1} = we for window i). SQLite ?N - // numbered bindings reuse the same value within one arm's subquery without - // requiring duplicate params in the binding list. - let mut arms: Vec = Vec::with_capacity(windows.len()); - for (i, _) in windows.iter().enumerate() { - let p = 1 + i * 2; - arms.push(format!( - "SELECT ?{p} AS ws, ?{p1} AS we, - l.ai_project, l.ai_tool, l.ai_session_id, - COUNT(*) AS match_count, - (SELECT l2.message FROM logs l2 - WHERE l2.ai_project = l.ai_project - AND l2.ai_tool = l.ai_tool - AND l2.ai_session_id = l.ai_session_id - AND l2.timestamp BETWEEN ?{p} AND ?{p1} - ORDER BY l2.timestamp DESC LIMIT 1) AS best_snippet - FROM logs l - WHERE l.ai_project IS NOT NULL AND l.ai_project != '' - AND l.ai_tool IS NOT NULL AND l.ai_tool != '' - AND l.ai_session_id IS NOT NULL AND l.ai_session_id != '' - AND l.timestamp BETWEEN ?{p} AND ?{p1} - GROUP BY l.ai_project, l.ai_tool, l.ai_session_id", - p = p, - p1 = p + 1, - )); - } - let sql = arms.join("\nUNION ALL\n"); - - let mut stmt = conn - .prepare(&sql) - .map_err(|e| anyhow::anyhow!("find_correlated_sessions_per_cluster prepare: {e}"))?; - - // Two params per window (ws, we); ?N reuse within each arm handles the rest. - let params: Vec<&dyn rusqlite::ToSql> = windows - .iter() - .flat_map(|(ws, we)| { - let v: [&dyn rusqlite::ToSql; 2] = [ws, we]; - v - }) - .collect(); - - let rows = stmt - .query_map(params.as_slice(), |row| { - let ws: String = row.get(0)?; - let we: String = row.get(1)?; - let project: String = row.get(2)?; - let tool: String = row.get(3)?; - let session_id: String = row.get(4)?; - let match_count: i64 = row.get(5)?; - let best_snippet: Option = row.get(6)?; - Ok((ws, we, project, tool, session_id, match_count, best_snippet)) - }) - .map_err(|e| anyhow::anyhow!("find_correlated_sessions_per_cluster query: {e}"))?; - - // Collect all sessions per cluster before sorting — UNION ALL rows arrive - // unordered, so the top-5 cap must come after sorting, not during insertion. - let mut map: HashMap<(String, String), Vec> = HashMap::new(); - for row in rows { - let (ws, we, project, tool, session_id, match_count, best_snippet) = - row.map_err(|e| anyhow::anyhow!("find_correlated_sessions_per_cluster row: {e}"))?; - map.entry((ws, we)).or_default().push(CorrelatedSession { - project, - tool, - session_id, - match_count, - best_snippet, - }); - } - // Sort by match_count descending, then cap at 5 per cluster. - for sessions in map.values_mut() { - sessions.sort_by_key(|b| std::cmp::Reverse(b.match_count)); - sessions.truncate(5); - } - Ok(map) -} - -/// Return aggregate log statistics + error logs + correlated AI sessions for a -/// given time window. -pub fn incident_context_summary( - pool: &DbPool, - params: &IncidentContextParams, -) -> Result { - let conn = pool.get()?; - let limit = params.limit.unwrap_or(50).clamp(1, 200) as usize; - if let Some(query) = params.query.as_deref() { - validate_fts_query(query)?; - } - - // Resolve severity threshold. Default to "warning" (numeric 4). - let severity_threshold = params - .severity_min - .as_deref() - .map(|s| { - severity_to_num(s).ok_or_else(|| { - anyhow::anyhow!( - "invalid severity_min '{}': must be one of emerg, alert, crit, err, warning, notice, info, debug", - s - ) - }) - }) - .transpose()? - .unwrap_or_else(|| severity_to_num("warning").unwrap()); - - // Build reusable aggregate params with host/app/AI-exclusion filters. - // Params: ?1=from, ?2=to, then optional host/app starting at ?3. - // All aggregate queries exclude AI transcript rows (ai_project IS NULL or ''). - let mut agg_params = SqlParams::new(3); - agg_params - .bindings - .push(rusqlite::types::Value::Text(params.since.clone())); - agg_params - .bindings - .push(rusqlite::types::Value::Text(params.until.clone())); - let mut agg_host_clause = String::new(); - let mut agg_app_clause = String::new(); - if let Some(hostname) = ¶ms.host { - let idx = agg_params.push_text(hostname.clone()); - agg_host_clause = format!(" AND hostname = ?{idx}"); - } - if let Some(app_name) = ¶ms.app { - let idx = agg_params.push_text(app_name.clone()); - agg_app_clause = format!(" AND app_name = ?{idx}"); - } - let agg_base_filter = format!( - "WHERE (ai_project IS NULL OR ai_project = '') - AND timestamp BETWEEN ?1 AND ?2{agg_host_clause}{agg_app_clause}" - ); - - // Total log count in window (system logs only, scoped by host/app). - let total_logs: i64 = conn - .query_row( - &format!("SELECT COUNT(*) FROM logs INDEXED BY idx_logs_timestamp {agg_base_filter}"), - rusqlite::params_from_iter(agg_params.bindings.iter()), - |r| r.get(0), - ) - .map_err(|e| anyhow::anyhow!("incident_context total_logs: {e}"))?; - - // Counts by severity (system logs only, scoped by host/app). - let mut by_sev_stmt = conn - .prepare(&format!( - "SELECT severity, COUNT(*) FROM logs INDEXED BY idx_logs_timestamp - {agg_base_filter} - GROUP BY severity - ORDER BY COUNT(*) DESC" - )) - .map_err(|e| anyhow::anyhow!("incident_context by_severity prepare: {e}"))?; - let by_severity: Vec = by_sev_stmt - .query_map( - rusqlite::params_from_iter(agg_params.bindings.iter()), - |row| { - Ok(SeverityCount { - severity: row.get(0)?, - count: row.get(1)?, - }) - }, - ) - .map_err(|e| anyhow::anyhow!("incident_context by_severity query: {e}"))? - .collect::>>()?; - - // Counts by app_name (top 20, system logs only, scoped by host/app). - let mut by_app_stmt = conn - .prepare(&format!( - "SELECT app_name, COUNT(*) FROM logs INDEXED BY idx_logs_timestamp - {agg_base_filter} - GROUP BY app_name - ORDER BY COUNT(*) DESC - LIMIT 20" - )) - .map_err(|e| anyhow::anyhow!("incident_context by_app prepare: {e}"))?; - let by_app: Vec = by_app_stmt - .query_map( - rusqlite::params_from_iter(agg_params.bindings.iter()), - |row| { - Ok(AppLogCount { - app_name: row.get(0)?, - count: row.get(1)?, - }) - }, - ) - .map_err(|e| anyhow::anyhow!("incident_context by_app query: {e}"))? - .collect::>>()?; - - // Error logs: system logs at or above severity threshold in the window. - let error_severities: Vec = SEVERITY_LEVELS[..=severity_threshold as usize] - .iter() - .map(|s| s.to_string()) - .collect(); - - // Build parameterized query for error logs. - // Params: ?1=from, ?2=to, ?3..=?N=severities, then optional host/app. - // SqlParams::new(3) sets next_idx=3 so push_text calls start at ?3, after - // the two manually-pushed bindings for from (?1) and to (?2). - let mut err_params = SqlParams::new(3); - err_params - .bindings - .push(rusqlite::types::Value::Text(params.since.clone())); - err_params - .bindings - .push(rusqlite::types::Value::Text(params.until.clone())); - - let query_idx = params - .query - .as_ref() - .map(|query| err_params.push_text(query.clone())); - - let sev_placeholders: Vec = error_severities - .iter() - .map(|s| { - let idx = err_params.push_text(s.clone()); - format!("?{idx}") - }) - .collect(); - - let mut err_sql = format!("SELECT {FTS_SELECT_COLS} "); - match query_idx { - Some(idx) => err_sql.push_str(&format!( - "FROM logs_fts - JOIN logs l ON l.id = logs_fts.rowid - WHERE logs_fts MATCH ?{idx} - AND l.timestamp BETWEEN ?1 AND ?2" - )), - None => err_sql.push_str( - "FROM logs l INDEXED BY idx_logs_timestamp - WHERE l.timestamp BETWEEN ?1 AND ?2", - ), - } - err_sql.push_str(&format!( - " AND l.severity IN ({}) - AND (l.ai_project IS NULL OR l.ai_project = '')", - sev_placeholders.join(", ") - )); - - if let Some(hostname) = ¶ms.host { - let idx = err_params.push_text(hostname.clone()); - err_sql.push_str(&format!(" AND l.hostname = ?{idx}")); - } - if let Some(app_name) = ¶ms.app { - let idx = err_params.push_text(app_name.clone()); - err_sql.push_str(&format!(" AND l.app_name = ?{idx}")); - } - // Query limit+1 rows so we can detect true truncation. - err_sql.push_str(&format!(" ORDER BY l.timestamp DESC LIMIT {}", limit + 1)); - - let mut err_stmt = conn.prepare(&err_sql).map_err(|e| { - tracing::error!(error = %e, "incident_context error_logs prepare failed"); - anyhow::anyhow!("incident_context error_logs query failed") - })?; - let error_rows = err_stmt - .query_map( - rusqlite::params_from_iter(err_params.bindings.iter()), - map_row, - ) - .map_err(|e| { - tracing::error!(error = %e, "incident_context error_logs query failed"); - anyhow::anyhow!("incident_context error_logs query failed") - })?; - let mut error_logs: Vec = - error_rows.collect::>>()?; - let error_logs_truncated = error_logs.len() > limit; - error_logs.truncate(limit); - - // AI sessions active in the window — query on the already-held conn to - // avoid a second pool.get() call (which deadlocks on single-connection test pools). - let ai_sessions = { - let mut ai_sql = String::from( - "SELECT ai_project, ai_tool, ai_session_id, - MIN(ai_transcript_path) AS ai_transcript_path, - hostname, - MIN(timestamp) AS first_seen, - MAX(timestamp) AS last_seen, - COUNT(*) AS event_count - FROM logs - WHERE ai_project IS NOT NULL AND ai_project != '' - AND ai_tool IS NOT NULL AND ai_tool != '' - AND ai_session_id IS NOT NULL AND ai_session_id != '' - AND timestamp BETWEEN ?1 AND ?2", - ); - let mut ai_bindings: Vec = vec![ - rusqlite::types::Value::Text(params.since.clone()), - rusqlite::types::Value::Text(params.until.clone()), - ]; - if let Some(hostname) = ¶ms.host { - ai_bindings.push(rusqlite::types::Value::Text(hostname.clone())); - ai_sql.push_str(&format!(" AND hostname = ?{}", ai_bindings.len())); - } - ai_sql.push_str( - " GROUP BY ai_project, ai_tool, ai_session_id, hostname - ORDER BY last_seen DESC LIMIT 20", - ); - let mut ai_stmt = conn - .prepare(&ai_sql) - .map_err(|e| anyhow::anyhow!("incident_context ai_sessions prepare: {e}"))?; - let rows = ai_stmt - .query_map(rusqlite::params_from_iter(ai_bindings.iter()), |row| { - Ok(super::models::AiSessionEntry { - ai_project: row.get(0)?, - ai_tool: row.get(1)?, - ai_session_id: row.get(2)?, - ai_transcript_path: row.get(3)?, - hostname: row.get(4)?, - first_seen: row.get(5)?, - last_seen: row.get(6)?, - event_count: row.get(7)?, - title: None, - title_provenance: None, - }) - }) - .map_err(|e| anyhow::anyhow!("incident_context ai_sessions query: {e}"))?; - rows.collect::>>()? - }; - - Ok(IncidentContextResult { - window_from: params.since.clone(), - window_to: params.until.clone(), - total_logs, - by_severity, - by_app, - error_logs, - error_logs_truncated, - ai_sessions, - }) -} - -#[cfg(test)] -#[path = "queries_tests.rs"] -mod tests; - -#[cfg(test)] -#[path = "queries_graph_tests.rs"] -mod graph_tests; + } \ No newline at end of file diff --git a/src/db/queries_tests.rs b/src/db/queries_tests.rs index ba727198d..7ae474979 100644 --- a/src/db/queries_tests.rs +++ b/src/db/queries_tests.rs @@ -1924,6 +1924,7 @@ fn ai_session_queries_respect_filters() { &ListAiSessionsParams { ai_project: Some("/tmp/a".into()), ai_tool: Some("claude".into()), + ai_session_id: Some("s1".into()), host: Some("host-a".into()), since: Some("2026-01-01T00:00:00Z".into()), until: Some("2026-01-01T23:59:59Z".into()), @@ -1934,6 +1935,19 @@ fn ai_session_queries_respect_filters() { assert_eq!(listed.len(), 1); assert_eq!(listed[0].ai_session_id, "s1"); + let by_session = list_ai_sessions( + &pool, + &ListAiSessionsParams { + ai_session_id: Some("s2".into()), + limit: Some(10), + ..Default::default() + }, + ) + .unwrap(); + assert_eq!(by_session.len(), 1); + assert_eq!(by_session[0].ai_session_id, "s2"); + assert_eq!(by_session[0].ai_project, "/tmp/b"); + let searched = search_ai_sessions( &pool, &SearchAiSessionsParams { @@ -2259,6 +2273,7 @@ fn default_session_params() -> ListAiSessionsParams { ListAiSessionsParams { ai_project: None, ai_tool: None, + ai_session_id: None, host: None, since: None, until: None, @@ -2564,1002 +2579,4 @@ fn rollup_empty_when_only_broad_project_rows_present_succeeds() { // The watermark sees these rows (broad predicate) so src_count > 0; the // rollup predicate excludes them all so staging is legitimately empty. - // Pre-fix this raised the R1 error; post-fix it must SUCCEED with 0 rows. - let total = refresh_ai_session_rollup(&pool).unwrap(); - assert_eq!( - total, 0, - "rollup must be legitimately empty (no rollup-eligible rows)" - ); - - // The meta/fingerprint MUST still be stamped on an empty rollup so - // refresh_ai_session_rollup_if_stale can skip subsequent no-op refreshes. - let status = ai_session_rollup_status(&pool).unwrap(); - assert_eq!(status.row_count, 0, "rollup row_count must be 0"); - assert!( - status.refreshed_at.is_some(), - "meta/fingerprint must be stamped even for an empty rollup" - ); -} - -#[test] -fn rollup_read_uses_last_seen_index_no_temp_btree() { - let (pool, _dir) = test_pool(); - seed_ai_sessions(&pool); - refresh_ai_session_rollup(&pool).unwrap(); - // The unbounded rollup read must be served by the last_seen index, NOT a - // temp b-tree sort (the cost that plagued the live aggregation). The query - // mirrors list_ai_sessions_from_rollup. - let plan = query_plan( - &pool, - "SELECT ai_project, ai_tool, ai_session_id, ai_transcript_path, - hostname, first_seen, last_seen, event_count - FROM ai_session_rollup - WHERE 1=1 - ORDER BY last_seen DESC LIMIT 100", - &[], - ); - assert!( - plan.contains("idx_ai_session_rollup_last_seen"), - "rollup read must use the last_seen index; plan was:\n{plan}" - ); - assert!( - !plan.contains("TEMP B-TREE"), - "rollup read must avoid a temp b-tree sort; plan was:\n{plan}" - ); -} - -#[test] -fn rollup_respects_project_and_tool_filters() { - let (pool, _dir) = test_pool(); - seed_ai_sessions(&pool); - refresh_ai_session_rollup(&pool).unwrap(); - - for (project, tool) in [ - (Some("/proj/0".to_string()), None), - (None, Some("codex".to_string())), - (Some("/proj/1".to_string()), Some("claude".to_string())), - ] { - let params = ListAiSessionsParams { - ai_project: project.clone(), - ai_tool: tool.clone(), - ..default_session_params() - }; - let live = list_ai_sessions_live(&pool, ¶ms).unwrap(); - let rolled = list_ai_sessions(&pool, ¶ms).unwrap(); - assert_eq!( - live.iter().map(|s| &s.ai_session_id).collect::>(), - rolled.iter().map(|s| &s.ai_session_id).collect::>(), - "filtered rollup ({project:?},{tool:?}) must match live" - ); - } -} - -#[test] -fn time_windowed_sessions_always_use_live_path() { - let (pool, _dir) = test_pool(); - seed_ai_sessions(&pool); - refresh_ai_session_rollup(&pool).unwrap(); - - // Insert a brand-new event AFTER the rollup was built. A time-windowed - // query must see it live (rollup is stale and must be bypassed). - insert_logs_batch( - &pool, - &[make_ai_entry( - "2026-06-01T12:00:00Z", - "host0", - "codex", - "/proj/0", - "sess-0", - "fresh post-refresh event", - )], - ) - .unwrap(); - - let windowed = ListAiSessionsParams { - since: Some("2026-06-01T00:00:00Z".into()), - until: Some("2026-06-02T00:00:00Z".into()), - ..default_session_params() - }; - let rows = list_ai_sessions(&pool, &windowed).unwrap(); - assert_eq!( - rows.len(), - 1, - "windowed query must see the fresh event live" - ); - assert_eq!(rows[0].last_seen, "2026-06-01T12:00:00Z"); - assert_eq!( - rows[0].event_count, 1, - "windowed count must be live, not rollup" - ); -} - -// --------------------------------------------------------------------------- -// Rollup source-watermark dirty-check (bead cortex-g33v) -// --------------------------------------------------------------------------- - -/// Helper: insert a single non-AI log row (no ai_* fields). Such rows must NOT -/// move the AI watermark, so a refresh that follows them is skipped. -fn insert_plain_log(pool: &DbPool, ts: &str, msg: &str) { - insert_logs_batch(pool, &[make_entry(ts, "host0", "info", msg)]).unwrap(); -} - -#[test] -fn stale_check_first_refresh_runs_then_noop_is_skipped() { - let (pool, _dir) = test_pool(); - seed_ai_sessions(&pool); - - // Never refreshed yet => must refresh. - match refresh_ai_session_rollup_if_stale(&pool).unwrap() { - RollupRefresh::Refreshed { row_count } => assert!(row_count > 0), - RollupRefresh::Skipped => panic!("first refresh must not be skipped"), - } - - // Nothing changed => the expensive re-aggregation must be skipped. - assert_eq!( - refresh_ai_session_rollup_if_stale(&pool).unwrap(), - RollupRefresh::Skipped, - "unchanged source must skip the refresh" - ); -} - -#[test] -fn stale_check_detects_new_ai_row() { - let (pool, _dir) = test_pool(); - seed_ai_sessions(&pool); - refresh_ai_session_rollup_if_stale(&pool).unwrap(); - assert_eq!( - refresh_ai_session_rollup_if_stale(&pool).unwrap(), - RollupRefresh::Skipped - ); - - // A new AI row advances MAX(id) => must refresh. - insert_logs_batch( - &pool, - &[make_ai_entry( - "2026-07-01T00:00:00Z", - "host9", - "codex", - "/proj/new", - "sess-new", - "brand new ai event", - )], - ) - .unwrap(); - assert!( - matches!( - refresh_ai_session_rollup_if_stale(&pool).unwrap(), - RollupRefresh::Refreshed { .. } - ), - "a new AI row must trigger a refresh" - ); - // And the new session is now visible from the rollup path. - let rows = list_ai_sessions(&pool, &default_session_params()).unwrap(); - assert!(rows.iter().any(|s| s.ai_session_id == "sess-new")); -} - -#[test] -fn stale_check_detects_deleted_ai_row() { - let (pool, _dir) = test_pool(); - seed_ai_sessions(&pool); - refresh_ai_session_rollup_if_stale(&pool).unwrap(); - assert_eq!( - refresh_ai_session_rollup_if_stale(&pool).unwrap(), - RollupRefresh::Skipped - ); - - // Deleting an AI row changes COUNT(*) (and likely MAX(id)) => must refresh. - { - let conn = pool.get().unwrap(); - let deleted = conn - .execute( - "DELETE FROM logs WHERE id IN ( - SELECT id FROM logs WHERE ai_session_id IS NOT NULL LIMIT 1 - )", - [], - ) - .unwrap(); - assert_eq!(deleted, 1, "test must delete exactly one AI row"); - } - assert!( - matches!( - refresh_ai_session_rollup_if_stale(&pool).unwrap(), - RollupRefresh::Refreshed { .. } - ), - "a deleted AI row must trigger a refresh" - ); -} - -#[test] -fn stale_check_ignores_non_ai_rows() { - let (pool, _dir) = test_pool(); - seed_ai_sessions(&pool); - refresh_ai_session_rollup_if_stale(&pool).unwrap(); - - // Plain syslog rows (the overwhelming majority of ingest) must NOT force a - // re-aggregation — that is the whole point of an AI-scoped watermark. - insert_plain_log(&pool, "2026-07-01T00:00:00Z", "ordinary syslog line"); - insert_plain_log(&pool, "2026-07-01T00:00:01Z", "another ordinary line"); - assert_eq!( - refresh_ai_session_rollup_if_stale(&pool).unwrap(), - RollupRefresh::Skipped, - "non-AI ingest must not trigger a rollup refresh" - ); -} - -#[test] -fn stale_check_watermark_is_index_only_no_table_scan() { - // The watermark fingerprint must be cheap: served from the partial index - // idx_logs_ai_project_time (WHERE ai_project IS NOT NULL), NOT a full scan - // of `logs`. That index-only cost is the whole reason the dirty-check is - // worth running on every cadence tick. Mirrors ai_rows_watermark's query. - let (pool, _dir) = test_pool(); - seed_ai_sessions(&pool); - let plan = query_plan( - &pool, - "SELECT COUNT(*), COALESCE(MAX(id), 0) FROM logs - WHERE ai_project IS NOT NULL AND ai_project != ''", - &[], - ); - assert!( - plan.contains("idx_logs_ai_project_time"), - "watermark must use the AI partial index; plan was:\n{plan}" - ); - assert!( - !plan.contains("SCAN logs\n") && !plan.ends_with("SCAN logs"), - "watermark must not full-scan logs; plan was:\n{plan}" - ); -} - -#[test] -fn stale_check_skip_keeps_rollup_correct_vs_live() { - // A skipped refresh must leave the rollup serving results identical to a - // fresh live aggregation (i.e. skipping never serves stale data when the - // source genuinely did not change). - let (pool, _dir) = test_pool(); - seed_ai_sessions(&pool); - refresh_ai_session_rollup_if_stale(&pool).unwrap(); - assert_eq!( - refresh_ai_session_rollup_if_stale(&pool).unwrap(), - RollupRefresh::Skipped - ); - - let live = list_ai_sessions_live(&pool, &default_session_params()).unwrap(); - let rolled = list_ai_sessions(&pool, &default_session_params()).unwrap(); - assert_eq!(live.len(), rolled.len()); - for (l, r) in live.iter().zip(rolled.iter()) { - assert_eq!(l.ai_session_id, r.ai_session_id); - assert_eq!(l.last_seen, r.last_seen); - assert_eq!(l.event_count, r.event_count); - } -} - -// --------------------------------------------------------------------------- -// RAG v1 tests -// --------------------------------------------------------------------------- - -fn make_app_entry(ts: &str, host: &str, severity: &str, app: &str, msg: &str) -> LogBatchEntry { - LogBatchEntry { - timestamp: ts.to_string(), - hostname: host.to_string(), - facility: None, - severity: severity.to_string(), - app_name: Some(app.to_string()), - process_id: None, - message: msg.to_string(), - raw: msg.to_string(), - source_ip: "10.0.0.1:514".to_string(), - docker_checkpoint: None, - ai_tool: None, - ai_project: None, - ai_session_id: None, - ai_transcript_path: None, - metadata_json: None, - http_status: None, - auth_outcome: None, - dns_blocked: None, - event_action: None, - parse_error: None, - } -} - -#[test] -fn similar_incidents_clusters_returns_clusters_for_matching_logs() { - let (pool, _dir) = test_pool(); - - let logs = vec![ - make_app_entry( - "2024-01-15T10:00:00Z", - "web-01", - "err", - "nginx", - "upstream connect error timeout", - ), - make_app_entry( - "2024-01-15T10:05:00Z", - "web-01", - "crit", - "nginx", - "upstream connect error connection refused", - ), - ]; - insert_logs_batch(&pool, &logs).unwrap(); - - let params = SimilarIncidentsParams { - query: "upstream".into(), - host: None, - app: None, - severity_min: None, - since: None, - until: None, - window_minutes: Some(30), - limit: Some(10), - }; - let result = similar_incidents_clusters(&pool, ¶ms).unwrap(); - assert!(!result.clusters.is_empty(), "expected at least one cluster"); - let cluster = &result.clusters[0]; - assert_eq!(cluster.hostname, "web-01"); - assert_eq!(cluster.app_name.as_deref(), Some("nginx")); - assert!(cluster.log_count >= 2); - // "crit" is more severe than "err" - assert_eq!(cluster.severity_peak, "crit"); -} - -#[test] -fn similar_incidents_clusters_filters_by_hostname() { - let (pool, _dir) = test_pool(); - - let logs = vec![ - make_app_entry( - "2024-01-15T10:00:00Z", - "web-01", - "err", - "nginx", - "upstream connect error", - ), - make_app_entry( - "2024-01-15T10:01:00Z", - "web-02", - "err", - "nginx", - "upstream connect error", - ), - ]; - insert_logs_batch(&pool, &logs).unwrap(); - - let params = SimilarIncidentsParams { - query: "upstream".into(), - host: Some("web-01".into()), - ..Default::default() - }; - let result = similar_incidents_clusters(&pool, ¶ms).unwrap(); - assert!(result.clusters.iter().all(|c| c.hostname == "web-01")); -} - -#[test] -fn similar_incidents_applies_fts_before_candidate_cap() { - let (pool, _dir) = test_pool(); - insert_logs_batch( - &pool, - &[make_app_entry( - "2024-01-01T00:00:00Z", - "web-01", - "err", - "nginx", - "historicalincidentneedle connection refused", - )], - ) - .unwrap(); - - // Populate exactly the old raw-log candidate cap with newer nonmatches. - // A recency cap applied before FTS excludes the historical match above. - let conn = pool.get().unwrap(); - conn.execute_batch( - "WITH digits(d) AS ( - VALUES (0),(1),(2),(3),(4),(5),(6),(7),(8),(9) - ), - rows(n) AS ( - SELECT a.d + 10*b.d + 100*c.d + 1000*d.d + 10000*e.d - FROM digits a - CROSS JOIN digits b - CROSS JOIN digits c - CROSS JOIN digits d - CROSS JOIN digits e - ) - INSERT INTO logs - (timestamp, hostname, severity, app_name, message, raw, source_ip) - SELECT '2024-01-02T00:00:00Z', 'web-01', 'info', 'nginx', - 'routine health check', 'routine health check', '10.0.0.1:514' - FROM rows;", - ) - .unwrap(); - drop(conn); - - let result = similar_incidents_clusters( - &pool, - &SimilarIncidentsParams { - query: "historicalincidentneedle".into(), - ..Default::default() - }, - ) - .unwrap(); - - assert_eq!(result.clusters.len(), 1); - assert_eq!(result.clusters[0].window_start, "2024-01-01T00:00:00Z"); - assert_eq!(result.clusters[0].log_count, 1); -} - -#[test] -fn incident_context_summary_returns_window_stats() { - let (pool, _dir) = test_pool(); - - let logs = vec![ - make_app_entry( - "2024-02-01T08:00:00Z", - "db-01", - "err", - "postgres", - "FATAL: out of shared memory", - ), - make_app_entry( - "2024-02-01T08:01:00Z", - "db-01", - "info", - "postgres", - "database system is ready", - ), - ]; - insert_logs_batch(&pool, &logs).unwrap(); - - let params = IncidentContextParams { - since: "2024-02-01T07:00:00Z".into(), - until: "2024-02-01T09:00:00Z".into(), - host: None, - app: None, - query: None, - severity_min: Some("err".into()), - limit: Some(10), - }; - let result = incident_context_summary(&pool, ¶ms).unwrap(); - assert_eq!(result.total_logs, 2); - assert!(!result.by_severity.is_empty()); - // Only the "err" row should be in error_logs (not "info") - assert_eq!(result.error_logs.len(), 1); - assert_eq!(result.error_logs[0].message, "FATAL: out of shared memory"); -} - -#[test] -fn incident_context_summary_empty_window_returns_zero() { - let (pool, _dir) = test_pool(); - - let params = IncidentContextParams { - since: "2020-01-01T00:00:00Z".into(), - until: "2020-01-02T00:00:00Z".into(), - ..Default::default() - }; - let result = incident_context_summary(&pool, ¶ms).unwrap(); - assert_eq!(result.total_logs, 0); - assert!(result.error_logs.is_empty()); - assert!(result.ai_sessions.is_empty()); -} - -#[test] -fn incident_context_summary_filters_error_logs_by_fts_query() { - let (pool, _dir) = test_pool(); - insert_logs_batch( - &pool, - &[ - make_app_entry( - "2024-02-01T08:00:00Z", - "db-01", - "err", - "postgres", - "shared memory exhausted", - ), - make_app_entry( - "2024-02-01T08:01:00Z", - "db-01", - "err", - "postgres", - "connection pool exhausted", - ), - ], - ) - .unwrap(); - - let result = incident_context_summary( - &pool, - &IncidentContextParams { - since: "2024-02-01T07:00:00Z".into(), - until: "2024-02-01T09:00:00Z".into(), - query: Some("memory".into()), - ..Default::default() - }, - ) - .unwrap(); - - assert_eq!(result.total_logs, 2, "window aggregates remain unfiltered"); - assert_eq!(result.error_logs.len(), 1); - assert_eq!(result.error_logs[0].message, "shared memory exhausted"); -} - -#[test] -fn incident_context_window_queries_force_timestamp_index() { - let (pool, _dir) = test_pool(); - let plan = query_plan( - &pool, - "SELECT COUNT(*) - FROM logs INDEXED BY idx_logs_timestamp - WHERE (ai_project IS NULL OR ai_project = '') - AND timestamp BETWEEN ?1 AND ?2", - &[ - rusqlite::types::Value::Text("2024-02-01T07:00:00Z".into()), - rusqlite::types::Value::Text("2024-02-01T09:00:00Z".into()), - ], - ); - assert!( - plan.contains("idx_logs_timestamp"), - "incident context window scan should be timestamp-index driven; got:\n{plan}" - ); -} - -// ─────────────────────────────────────────────────────────────────────────── -// Performance benchmark harness (Issue 4 / bead cortex-2vre). -// -// Builds a synthetic on-disk SQLite DB with a realistic row count and times -// `get_stats` and `list_ai_sessions` before/after the optimization work. -// -// IGNORED by default — it builds millions of rows and takes minutes, so it -// must never run in the normal `cargo nextest` suite. Run explicitly: -// -// CORTEX_BENCH_ROWS=10000000 cargo test --lib \ -// db::queries::tests::bench_stats_and_sessions -- --ignored --nocapture -// -// Row count is controlled by CORTEX_BENCH_ROWS (default 5_000_000). -// ─────────────────────────────────────────────────────────────────────────── - -/// Insert `n` synthetic log rows through the live schema (FTS + inventory + -/// counter triggers all fire), in large transactions for throughput. ~20% of -/// rows carry AI session fields spread across many (project, tool, session) -/// groups so the sessions query has realistic cardinality. -fn bench_seed_rows(pool: &DbPool, n: usize) { - use std::time::Instant; - let started = Instant::now(); - const CHUNK: usize = 50_000; - let mut inserted = 0usize; - while inserted < n { - let this = CHUNK.min(n - inserted); - let mut conn = pool.get().unwrap(); - let tx = conn.transaction().unwrap(); - { - let mut stmt = tx - .prepare_cached( - "INSERT INTO logs (timestamp, hostname, facility, severity, app_name, - process_id, message, raw, received_at, source_ip, - ai_tool, ai_project, ai_session_id, ai_transcript_path) - VALUES (?1,?2,?3,?4,?5,?6,?7,?8,?9,?10,?11,?12,?13,?14)", - ) - .unwrap(); - for i in 0..this { - let x = inserted + i; - // Strictly-increasing timestamp per row (base + x seconds, full - // date rollover). Monotonic in `x` => each session's MAX(ts) is - // globally unique (no last_seen ties), so the rollup vs live - // top-N comparison is deterministic at the LIMIT boundary. - let base = chrono::DateTime::parse_from_rfc3339("2026-01-01T00:00:00Z").unwrap(); - let dt = base + chrono::TimeDelta::seconds(x as i64); - let ts = dt.format("%Y-%m-%dT%H:%M:%SZ").to_string(); - let recv = ts.clone(); - let host = format!("host{:02}", x % 25); - let app = format!("app{:02}", x % 60); - let msg = format!("synthetic log line {x} some error retry connection text"); - let is_ai = x.is_multiple_of(5); - let (tool, proj, sess, tpath): ( - Option, - Option, - Option, - Option, - ) = if is_ai { - let proj = format!("/proj/{}", x % 40); - let tool = if x.is_multiple_of(2) { - "codex" - } else { - "claude" - } - .to_string(); - // ~20k distinct sessions => realistic group cardinality. - let sess = format!("sess-{}", x % 20_000); - let tpath = format!("{proj}/{sess}.jsonl"); - (Some(tool), Some(proj), Some(sess), Some(tpath)) - } else { - (None, None, None, None) - }; - stmt.execute(rusqlite::params![ - ts, - host, - Option::::None, - "info", - app, - Option::::None, - msg, - "raw", - recv, - "10.0.0.1:514", - tool, - proj, - sess, - tpath, - ]) - .unwrap(); - } - } - tx.commit().unwrap(); - inserted += this; - } - eprintln!( - "[bench] seeded {inserted} rows in {:.1}s", - started.elapsed().as_secs_f64() - ); -} - -#[derive(Debug)] -struct BenchPercentiles { - p50_ms: f64, - p95_ms: f64, -} - -/// p50/p95 of N timed runs of `f`, in milliseconds. One warm-up run first. -/// Uses nearest-rank p95 so the emitted values remain reproducible from the -/// captured sample count without interpolating measurements that never ran. -fn bench_percentiles_ms(runs: usize, mut f: impl FnMut()) -> BenchPercentiles { - use std::time::Instant; - assert!(runs > 0); - f(); // warm-up - let mut samples: Vec = Vec::with_capacity(runs); - for _ in 0..runs { - let t = Instant::now(); - f(); - samples.push(t.elapsed().as_secs_f64() * 1000.0); - } - samples.sort_by(|a, b| a.partial_cmp(b).unwrap()); - let p50_index = (samples.len() - 1) / 2; - let p95_index = ((samples.len() * 95).div_ceil(100)).saturating_sub(1); - BenchPercentiles { - p50_ms: samples[p50_index], - p95_ms: samples[p95_index], - } -} - -#[test] -fn session_rollup_query_plan_uses_ordering_index_without_temp_sort() { - let dir = tempfile::tempdir().unwrap(); - let pool = init_pool(&test_storage_config(dir.path().join("plan.db"))).unwrap(); - let conn = pool.get().unwrap(); - let plan = conn - .prepare( - "EXPLAIN QUERY PLAN - SELECT ai_project, ai_tool, ai_session_id, hostname, last_seen - FROM ai_session_rollup - WHERE 1=1 - ORDER BY last_seen DESC LIMIT 100", - ) - .unwrap() - .query_map([], |row| row.get::<_, String>(3)) - .unwrap() - .collect::>>() - .unwrap(); - assert!( - plan.iter() - .any(|detail| detail.contains("idx_ai_session_rollup_last_seen")), - "rollup query must use its last-seen index: {plan:?}" - ); - assert!( - plan.iter() - .all(|detail| !detail.contains("USE TEMP B-TREE")), - "rollup query must not perform a temporary sort: {plan:?}" - ); -} - -#[test] -#[ignore = "performance benchmark; builds millions of rows. Run with --ignored."] -fn bench_stats_and_sessions() { - let rows: usize = crate::env::var("CORTEX_BENCH_ROWS") - .ok() - .and_then(|v| v.parse().ok()) - .unwrap_or(5_000_000); - - // Optional persistent DB path so a seeded DB can be reused across runs - // (seeding 10M rows takes ~13 min). When unset, use a throwaway tempdir. - let _guard_dir; - let (pool, cfg) = if let Ok(path) = crate::env::var("CORTEX_BENCH_DB") { - let db_path = std::path::PathBuf::from(&path); - let fresh = !db_path.exists(); - let cfg = test_storage_config(db_path); - let pool = init_pool(&cfg).unwrap(); - if fresh { - bench_seed_rows(&pool, rows); - } else { - eprintln!("[bench] reusing existing DB at {path} (skipping seed)"); - } - (pool, cfg) - } else { - let dir = tempfile::tempdir().unwrap(); - let cfg = test_storage_config(dir.path().join("test.db")); - let pool = init_pool(&cfg).unwrap(); - bench_seed_rows(&pool, rows); - _guard_dir = dir; // keep alive - (pool, cfg) - }; - - let ground_truth: i64 = { - let conn = pool.get().unwrap(); - conn.query_row("SELECT COUNT(*) FROM logs", [], |r| r.get(0)) - .unwrap() - }; - eprintln!("[bench] ground-truth COUNT(*) FROM logs = {ground_truth}"); - - // --- stats (default: FTS diagnostic skipped) --- - let mut last_stats = None; - let stats_latency = bench_percentiles_ms(20, || { - last_stats = Some(get_stats(&pool, &cfg).unwrap()); - }); - let stats = last_stats.unwrap(); - eprintln!( - "[bench] get_stats (default, FTS skipped): p50={:.1} ms p95={:.1} ms (total_logs={})", - stats_latency.p50_ms, stats_latency.p95_ms, stats.total_logs - ); - assert_eq!( - stats.total_logs, ground_truth, - "stats total_logs must equal ground-truth COUNT(*)" - ); - - // --- stats with FTS diagnostic ON (the expensive COUNT(*) FROM logs_fts) --- - let stats_fts_latency = bench_percentiles_ms(20, || { - let _ = get_stats_with_options(&pool, &cfg, true).unwrap(); - }); - eprintln!( - "[bench] get_stats (FTS diagnostic ON): p50={:.1} ms p95={:.1} ms", - stats_fts_latency.p50_ms, stats_fts_latency.p95_ms - ); - - // --- sessions BEFORE: live aggregation (GROUP BY + temp-btree sort) --- - let params = ListAiSessionsParams { - ai_project: None, - ai_tool: None, - host: None, - since: None, - until: None, - limit: Some(100), - }; - let mut live_rows = 0usize; - let sessions_live_latency = bench_percentiles_ms(20, || { - live_rows = list_ai_sessions_live(&pool, ¶ms).unwrap().len(); - }); - eprintln!( - "[bench] BEFORE list_ai_sessions_live(limit=100): p50={:.1} ms p95={:.1} ms ({live_rows} rows)", - sessions_live_latency.p50_ms, sessions_live_latency.p95_ms - ); - - // --- refresh cost (background cadence; not on the request path) --- - let mut rollup_total = 0usize; - let refresh_latency = bench_percentiles_ms(10, || { - rollup_total = refresh_ai_session_rollup(&pool).unwrap(); - }); - eprintln!( - "[bench] refresh_ai_session_rollup: p50={:.1} ms p95={:.1} ms ({rollup_total} session rows total)", - refresh_latency.p50_ms, refresh_latency.p95_ms - ); - - // --- sessions AFTER: indexed read from the rollup materialization --- - let mut rollup_rows = 0usize; - let sessions_rollup_latency = bench_percentiles_ms(20, || { - rollup_rows = list_ai_sessions(&pool, ¶ms).unwrap().len(); - }); - eprintln!( - "[bench] AFTER list_ai_sessions(rollup, limit=100): p50={:.1} ms p95={:.1} ms ({rollup_rows} rows)", - sessions_rollup_latency.p50_ms, sessions_rollup_latency.p95_ms - ); - - // Correctness: the rollup-served top-N must equal the live top-N. Both - // paths order by `last_seen DESC` only, so rows that TIE on last_seen may - // appear in different relative order between the two plans. Compare in a - // tie-order-independent way: (a) the multiset of last_seen ordering keys - // must be identical, and (b) the per-session (last_seen, event_count) facts - // must match for every returned session. - let live = list_ai_sessions_live(&pool, ¶ms).unwrap(); - let rollup = list_ai_sessions(&pool, ¶ms).unwrap(); - assert_eq!(live.len(), rollup.len(), "rollup/live row count mismatch"); - let mut live_keys: Vec<&String> = live.iter().map(|s| &s.last_seen).collect(); - let mut rollup_keys: Vec<&String> = rollup.iter().map(|s| &s.last_seen).collect(); - live_keys.sort(); - rollup_keys.sort(); - assert_eq!( - live_keys, rollup_keys, - "rollup/live last_seen ordering-key multisets differ" - ); - let live_facts: std::collections::HashMap<_, _> = live - .iter() - .map(|s| { - ( - (&s.ai_project, &s.ai_tool, &s.ai_session_id, &s.hostname), - (&s.last_seen, s.event_count), - ) - }) - .collect(); - for r in &rollup { - let key = (&r.ai_project, &r.ai_tool, &r.ai_session_id, &r.hostname); - match live_facts.get(&key) { - Some((last_seen, count)) => { - assert_eq!(*last_seen, &r.last_seen, "last_seen mismatch for {key:?}"); - assert_eq!(*count, r.event_count, "event_count mismatch for {key:?}"); - } - None => panic!("rollup returned session not in live top-N: {key:?}"), - } - } - - let speedup = sessions_live_latency.p50_ms / sessions_rollup_latency.p50_ms.max(0.001); - let minimum_speedup = crate::env::var("CORTEX_BENCH_MIN_SESSION_SPEEDUP") - .ok() - .map(|value| { - value - .parse::() - .expect("CORTEX_BENCH_MIN_SESSION_SPEEDUP must be a number") - }); - let maximum_p95 = crate::env::var("CORTEX_BENCH_MAX_SESSION_P95_MS") - .ok() - .map(|value| { - value - .parse::() - .expect("CORTEX_BENCH_MAX_SESSION_P95_MS must be a number") - }); - let speedup_passed = minimum_speedup.is_none_or(|minimum| speedup >= minimum); - let p95_passed = maximum_p95.is_none_or(|maximum| sessions_rollup_latency.p95_ms <= maximum); - if let Ok(path) = crate::env::var("CORTEX_BENCH_ARTIFACT") { - let artifact = serde_json::json!({ - "rows": rows, - "stats_default": {"p50_ms": stats_latency.p50_ms, "p95_ms": stats_latency.p95_ms}, - "stats_fts": {"p50_ms": stats_fts_latency.p50_ms, "p95_ms": stats_fts_latency.p95_ms}, - "sessions_live": {"p50_ms": sessions_live_latency.p50_ms, "p95_ms": sessions_live_latency.p95_ms}, - "sessions_rollup": {"p50_ms": sessions_rollup_latency.p50_ms, "p95_ms": sessions_rollup_latency.p95_ms}, - "refresh": {"p50_ms": refresh_latency.p50_ms, "p95_ms": refresh_latency.p95_ms}, - "session_speedup": speedup, - "minimum_session_speedup": minimum_speedup, - "maximum_session_p95_ms": maximum_p95, - "outcome": {"speedup_passed": speedup_passed, "p95_passed": p95_passed}, - }); - std::fs::write(path, serde_json::to_vec_pretty(&artifact).unwrap()).unwrap(); - } - if let Some(minimum_speedup) = minimum_speedup { - assert!( - speedup_passed, - "session rollup speedup {speedup:.2}x is below the {minimum_speedup:.2}x configured threshold" - ); - } - assert!( - p95_passed, - "session rollup p95 {:.1}ms exceeds the configured diagnostic threshold", - sessions_rollup_latency.p95_ms - ); - eprintln!( - "[bench] SUMMARY rows={rows} \ - stats_default_p50_ms={:.1} stats_default_p95_ms={:.1} \ - stats_fts_on_p50_ms={:.1} stats_fts_on_p95_ms={:.1} \ - sessions_BEFORE_live_p50_ms={:.1} sessions_BEFORE_live_p95_ms={:.1} \ - sessions_AFTER_rollup_p50_ms={:.1} sessions_AFTER_rollup_p95_ms={:.1} \ - refresh_p50_ms={:.1} refresh_p95_ms={:.1} sessions_p50_speedup={speedup:.1}x", - stats_latency.p50_ms, - stats_latency.p95_ms, - stats_fts_latency.p50_ms, - stats_fts_latency.p95_ms, - sessions_live_latency.p50_ms, - sessions_live_latency.p95_ms, - sessions_rollup_latency.p50_ms, - sessions_rollup_latency.p95_ms, - refresh_latency.p50_ms, - refresh_latency.p95_ms, - ); -} - -/// full-review QM2: the 15-column log projection is written inline at ~14 -/// sites across queries.rs / analytics.rs / ingest.rs, and `map_row` / -/// `map_row_offset` / `map_row_with_raw` read columns BY ORDINAL POSITION — -/// reordering or inserting a column at one site without updating the readers -/// silently mis-maps fields with no compile error. This drift test extracts -/// every projection that ends in `metadata_json` from the source text and -/// asserts it carries the canonical column order. (`map_row_with_raw` selects -/// `..., metadata_json, raw`; the canonical prefix still applies.) -#[test] -fn inline_log_projections_match_map_row_column_order() { - // Two canonical shapes exist: `map_row` (15 cols) and `map_row_with_raw` - // (16 cols, `raw` between `message` and `received_at`). - const CANON: &str = "id timestamp hostname facility severity app_name process_id message \ - received_at source_ip ai_tool ai_project ai_session_id ai_transcript_path metadata_json"; - const CANON_WITH_RAW: &str = "id timestamp hostname facility severity app_name process_id \ - message raw received_at source_ip ai_tool ai_project ai_session_id ai_transcript_path \ - metadata_json"; - let canon_tokens: Vec<&str> = CANON.split_whitespace().collect(); - let canon_raw_tokens: Vec<&str> = CANON_WITH_RAW.split_whitespace().collect(); - - let sources = [ - ("queries.rs", include_str!("queries.rs")), - ("analytics.rs", include_str!("analytics.rs")), - ("ingest.rs", include_str!("ingest.rs")), - ]; - let re = regex::Regex::new( - r"SELECT\s+((?:[a-zA-Z_][a-zA-Z_0-9]*\.)?id[\sa-zA-Z_0-9,.\\]*?metadata_json)", - ) - .unwrap(); - - let mut checked = 0usize; - for (name, src) in sources { - for cap in re.captures_iter(src) { - let projection = &cap[1]; - let tokens: Vec = projection - .split([',', '\\']) - .map(|t| t.trim()) - .filter(|t| !t.is_empty()) - .map(|t| { - // Strip any table alias prefix ("l.id" -> "id"). - t.rsplit('.').next().unwrap_or(t).to_string() - }) - .collect(); - assert!( - tokens == canon_tokens || tokens == canon_raw_tokens, - "{name}: inline log projection diverges from map_row / \ - map_row_with_raw column order — update the projection AND the \ - row readers together:\n{projection}\ngot: {tokens:?}" - ); - checked += 1; - } - } - assert!( - checked >= 10, - "expected to find at least 10 inline projections; the extraction regex \ - may have rotted (found {checked})" - ); -} - -#[test] -fn lint_flags_unquoted_infix_hyphen_term() { - let err = validate_fts_query("smoke-test").unwrap_err().to_string(); - assert!( - err.contains("NOT operator"), - "should explain hyphen trap: {err}" - ); - assert!( - err.contains("--grep") || err.contains("\"smoke-test\""), - "should suggest a fix: {err}" - ); -} - -#[test] -fn lint_accepts_quoted_phrase() { - // Already-quoted hyphenated phrase is valid FTS5 and must pass. - assert!(validate_fts_query("\"smoke-test\"").is_ok()); -} - -#[test] -fn lint_accepts_normal_boolean_query() { - assert!(validate_fts_query("error AND nginx").is_ok()); -} - -#[test] -fn lint_leaves_leading_hyphen_not_term_alone() { - // `-nginx` is an intentional FTS5 NOT, not the hyphenated-word trap. - assert!(validate_fts_query("error -nginx").is_ok()); -} - -#[test] -fn lint_flags_unbalanced_quote() { - let err = validate_fts_query("\"oops").unwrap_err().to_string(); - assert!(err.contains("unbalanced quote"), "{err}"); -} - -#[test] -fn lint_flags_unquoted_hyphen_term_alongside_a_quoted_phrase() { - // The hyphen check is per-term: a quoted phrase elsewhere must not mask an - // unquoted hyphenated term (regression for a query-wide quote gate). - let err = validate_fts_query("\"disk full\" smoke-test") - .unwrap_err() - .to_string(); - assert!(err.contains("NOT operator"), "{err}"); -} + // Pre-fix this raised the R1 error; post-fix it must SUCCEED with 0 rows. \ No newline at end of file diff --git a/src/mcp/actions.rs b/src/mcp/actions.rs index 71a913904..6170f4928 100644 --- a/src/mcp/actions.rs +++ b/src/mcp/actions.rs @@ -72,6 +72,7 @@ pub(super) enum ActionHandler { ListApps, ListSessions, SearchSessions, + SessionInvestigate, EvidenceScope, SearchAbuse, AbuseIncidents, @@ -381,6 +382,17 @@ pub(super) const ACTION_SPECS: &[ActionSpec] = &[ Cheap, SearchSessions ), + action_spec!( + "session_investigate", + Read, + "Build a bounded evidence bundle rooted at one AI session", + Expensive, + SessionInvestigate, + exact: { + allowed: &["session_id", "tool", "project", "host", "limit", "window_minutes", "severity_min"], + required: &["session_id"] + } + ), action_spec!( "evidence_scope", Read, diff --git a/src/mcp/actions_tests.rs b/src/mcp/actions_tests.rs index 2693650f8..cca4f7f8e 100644 --- a/src/mcp/actions_tests.rs +++ b/src/mcp/actions_tests.rs @@ -2,9 +2,9 @@ use super::*; #[test] fn documented_action_count_matches_registry() { - assert_eq!(ACTION_SPECS.len(), 60); + assert_eq!(ACTION_SPECS.len(), 61); let claude = include_str!("../../CLAUDE.md"); - assert!(claude.contains("authoritative registry of all 60 MCP actions")); + assert!(claude.contains("authoritative registry of all 61 MCP actions")); } // PR 4 of GH #94 / GH #105: LLM skill/abuse/hook assessment is CLI-only. See diff --git a/src/mcp/rmcp_server_tests.rs b/src/mcp/rmcp_server_tests.rs index 8354d3283..491d826ea 100644 --- a/src/mcp/rmcp_server_tests.rs +++ b/src/mcp/rmcp_server_tests.rs @@ -159,6 +159,7 @@ fn minimal_args_for_action(action: &str) -> Value { match action { "correlate" => json!({"action": action, "reference_time": "2026-01-01T00:00:00Z"}), "search_sessions" => json!({"action": action, "query": "mounted"}), + "session_investigate" => json!({"action": action, "session_id": "mounted-session"}), "ai_correlate" => json!({"action": action, "project": "/tmp/project"}), "project_context" => json!({"action": action, "project": "/tmp/project"}), "context" => json!({"action": action, "log_id": 1}), diff --git a/src/mcp/tools.rs b/src/mcp/tools.rs index ccc041458..f6ec873bd 100644 --- a/src/mcp/tools.rs +++ b/src/mcp/tools.rs @@ -24,8 +24,9 @@ use crate::app::{ ListAiToolsRequest, ListAppsRequest, ListArtifactEvidenceRequest, ListHookEventsRequest, ListMcpEventsRequest, ListSessionsRequest, ListSkillEventsRequest, ListSourceIpsRequest, LlmInvocationsRequest, NotificationsRecentRequest, PatternsRequest, ProjectContextRequest, - SearchLogsRequest, SearchSessionsRequest, SilentHostsRequest, TailLogsRequest, TimelineRequest, - TopicCorrelateRequest, UnaddressedErrorsRequest, UsageBlocksRequest, + SearchLogsRequest, SearchSessionsRequest, SessionInvestigateRequest, SilentHostsRequest, + TailLogsRequest, TimelineRequest, TopicCorrelateRequest, UnaddressedErrorsRequest, + UsageBlocksRequest, }; use crate::artifact_evidence::ArtifactEvidenceInput; @@ -96,6 +97,7 @@ async fn dispatch_cortex_action( H::ListApps => tool_list_apps(state, args).await, H::ListSessions => tool_list_sessions(state, args).await, H::SearchSessions => tool_search_sessions(state, args).await, + H::SessionInvestigate => tool_session_investigate(state, args).await, H::EvidenceScope => tool_evidence_scope(state, args).await, H::SearchAbuse => tool_search_abuse(state, args).await, H::AbuseIncidents => tool_abuse_incidents(state, args).await, @@ -262,6 +264,12 @@ async fn tool_search_sessions(state: &AppState, args: Value) -> anyhow::Result anyhow::Result { + let req: SessionInvestigateRequest = action_payload(args, "session_investigate")?; + let response = state.service.session_investigate(req).await?; + Ok(serde_json::to_value(response)?) +} + async fn tool_search_abuse(state: &AppState, args: Value) -> anyhow::Result { let req: AbuseSearchRequest = action_payload(args, "abuse")?; let response = state.service.search_abuse(req).await?; diff --git a/src/mcp/tools_tests.rs b/src/mcp/tools_tests.rs index 49e396185..4a5cd2605 100644 --- a/src/mcp/tools_tests.rs +++ b/src/mcp/tools_tests.rs @@ -1375,6 +1375,7 @@ fn sample_args_for_action(action: &str) -> Option { json!({"action": action, "reference_time": "2026-01-01T00:00:00Z"}) } "search_sessions" => json!({"action": action, "query": "schema"}), + "session_investigate" => json!({"action": action, "session_id": "schema-session"}), "evidence_scope" => json!({"action": action, "branch": "codex/schema-test"}), "ai_correlate" => json!({"action": action, "project": "/tmp/project"}), "topic_correlate" => json!({"action": action, "topic": "schema-test"}), @@ -1484,6 +1485,7 @@ fn typed_unknown_field_samples() -> Vec { "apps", "sessions", "search_sessions", + "session_investigate", "abuse", "abuse_incidents", "abuse_investigate", @@ -1606,6 +1608,35 @@ async fn schema_actions_are_dispatchable() { ) .unwrap(); } + db::insert_logs_batch( + &h.pool, + &[db::LogBatchEntry { + timestamp: "2026-01-01T00:00:00Z".to_string(), + hostname: "schema-session-host".to_string(), + facility: Some("agent".to_string()), + severity: "info".to_string(), + app_name: Some("codex".to_string()), + process_id: None, + message: "schema session transcript".to_string(), + raw: "schema session transcript".to_string(), + source_ip: "agent-command://schema-session-host/codex/schema-session".to_string(), + docker_checkpoint: None, + ai_tool: Some("codex".to_string()), + ai_project: Some("/schema/project".to_string()), + ai_session_id: Some("schema-session".to_string()), + ai_transcript_path: Some("/schema/transcript.jsonl".to_string()), + metadata_json: Some( + r#"{"source_kind":"agent-command","agent_command":{"cwd":"/schema/project"}}"# + .to_string(), + ), + http_status: None, + auth_outcome: None, + dns_blocked: None, + event_action: Some("command".to_string()), + parse_error: None, + }], + ) + .unwrap(); { let _guard = db::graph::GRAPH_TEST_LOCK.lock(); db::graph::refresh_graph_projection(&h.pool).unwrap(); From f5ddc3714f334c7d4cb45c7e35ee1e9cb40eba69 Mon Sep 17 00:00:00 2001 From: jmagar <38927646+jmagar@users.noreply.github.com> Date: Mon, 21 Sep 2026 21:33:12 -0400 Subject: [PATCH 2/8] feat: add session actor and run lineage evidence --- src/app/models/investigation.rs | 4 ++ src/app/services/session_investigation.rs | 31 +++++++++ src/db/agent_observatory.rs | 13 ++-- src/db/agent_observatory_read.rs | 80 ++++++++++++++++++----- src/db/agent_observatory_read_models.rs | 20 ++++++ src/db/agent_observatory_read_tests.rs | 52 +++++++++++++++ 6 files changed, 179 insertions(+), 21 deletions(-) diff --git a/src/app/models/investigation.rs b/src/app/models/investigation.rs index 7c421c7e9..7f8fcd8c2 100644 --- a/src/app/models/investigation.rs +++ b/src/app/models/investigation.rs @@ -233,6 +233,10 @@ pub struct SessionSourceEvidenceSummary { pub struct SessionObservatoryEvidence { pub runs: Vec, pub ambiguous_run: bool, + pub parent_run: Option, + pub previous_run: Option, + pub actors: Vec, + pub actors_truncated: bool, pub related_runs: Vec, pub related_runs_truncated: bool, pub repository: Option, diff --git a/src/app/services/session_investigation.rs b/src/app/services/session_investigation.rs index ebc201e5c..9893a72dd 100644 --- a/src/app/services/session_investigation.rs +++ b/src/app/services/session_investigation.rs @@ -259,6 +259,21 @@ impl CortexService { "Agent Observatory run disappeared during session investigation" ) })?; + let parent_run = match run.parent_run_id { + Some(id) => db::agent_observatory::resolve_observatory_run_row(pool, id)?, + None => None, + }; + let previous_run = match run.previous_run_id { + Some(id) => db::agent_observatory::resolve_observatory_run_row(pool, id)?, + None => None, + }; + let mut actors = db::agent_observatory::list_observatory_run_actors( + pool, + run_id, + section_limit as usize + 1, + )?; + let actors_truncated = actors.len() > section_limit as usize; + actors.truncate(section_limit as usize); let worktree = match run.primary_worktree_id { Some(id) => db::agent_observatory::resolve_observatory_worktree(pool, id)?, None => None, @@ -318,6 +333,10 @@ impl CortexService { Ok(models::SessionObservatoryEvidence { runs, ambiguous_run, + parent_run, + previous_run, + actors, + actors_truncated, related_runs, related_runs_truncated, repository, @@ -404,6 +423,15 @@ impl CortexService { if !notifications.is_empty() { source_counts.insert("notification".to_string(), notifications.len()); } + if !observatory.actors.is_empty() { + source_counts.insert("observatory_actor".to_string(), observatory.actors.len()); + } + if observatory.parent_run.is_some() { + source_counts.insert("parent_run".to_string(), 1); + } + if observatory.previous_run.is_some() { + source_counts.insert("previous_run".to_string(), 1); + } if !observatory.related_runs.is_empty() { source_counts.insert( "same_worktree_run".to_string(), @@ -443,6 +471,9 @@ impl CortexService { if observatory.ambiguous_run { partial_reasons.push("observatory_run_ambiguous".to_string()); } + if observatory.actors_truncated { + partial_reasons.push("observatory_actors_truncated".to_string()); + } if observatory.related_runs_truncated { partial_reasons.push("observatory_related_runs_truncated".to_string()); } diff --git a/src/db/agent_observatory.rs b/src/db/agent_observatory.rs index d5cb0bb6c..3cdeb2937 100644 --- a/src/db/agent_observatory.rs +++ b/src/db/agent_observatory.rs @@ -61,12 +61,13 @@ pub use queries::{get_worktree_by_key, reconcile_repository}; #[path = "agent_observatory_read.rs"] mod read; pub use read::{ - AgentEventQuery, AgentRunQuery, EvidenceScopePage, EvidenceScopeQuery, ObservatoryEventRow, - ObservatoryMetricRow, ObservatoryRepositoryRow, ObservatoryRunRow, ObservatorySpanRow, - ObservatoryWorktreeRow, RepositoryQuery, RunTelemetryIdentity, TelemetryQuery, - list_observatory_events, list_observatory_metrics, list_observatory_repositories, - list_observatory_runs, list_observatory_spans, list_observatory_worktrees, - resolve_observatory_repository, resolve_observatory_run, resolve_observatory_worktree, + AgentEventQuery, AgentRunQuery, EvidenceScopePage, EvidenceScopeQuery, ObservatoryActorRow, + ObservatoryEventRow, ObservatoryMetricRow, ObservatoryRepositoryRow, ObservatoryRunRow, + ObservatorySpanRow, ObservatoryWorktreeRow, RepositoryQuery, RunTelemetryIdentity, + TelemetryQuery, list_observatory_events, list_observatory_metrics, + list_observatory_repositories, list_observatory_run_actors, list_observatory_runs, + list_observatory_spans, list_observatory_worktrees, resolve_observatory_repository, + resolve_observatory_run, resolve_observatory_run_row, resolve_observatory_worktree, scoped_evidence_events, }; diff --git a/src/db/agent_observatory_read.rs b/src/db/agent_observatory_read.rs index d625ef1a4..ca8bf6164 100644 --- a/src/db/agent_observatory_read.rs +++ b/src/db/agent_observatory_read.rs @@ -301,7 +301,7 @@ pub fn list_observatory_runs( ) -> Result> { let conn = pool.get()?; let mut values = Vec::new(); - let mut sql="SELECT DISTINCT a.id,a.run_key,a.native_session_id,a.tool,a.provider_tool,a.hostname,a.status,a.status_reason,a.status_observed_at,a.started_at,a.last_activity_at,a.ended_at,a.transcript_path,a.primary_worktree_id,a.primary_branch,a.start_head_sha,a.current_head_sha,a.event_count,a.error_count,a.freshness_json FROM agent_runs a WHERE 1=1".to_string(); + let mut sql="SELECT DISTINCT a.id,a.run_key,a.native_session_id,a.tool,a.provider_tool,a.hostname,a.parent_run_id,a.previous_run_id,a.status,a.status_reason,a.status_observed_at,a.started_at,a.last_activity_at,a.ended_at,a.transcript_path,a.primary_worktree_id,a.primary_branch,a.start_head_sha,a.current_head_sha,a.event_count,a.error_count,a.freshness_json FROM agent_runs a WHERE 1=1".to_string(); push_filter(&mut sql, &mut values, "a.id <= ?", high_water); if let Some(id) = q.worktree_id { push_filter( @@ -372,6 +372,54 @@ pub fn list_observatory_runs( .collect::>() .context("list observatory runs") } +pub fn resolve_observatory_run_row( + pool: &DbPool, + run_id: i64, +) -> Result> { + pool.get()? + .query_row( + "SELECT a.id,a.run_key,a.native_session_id,a.tool,a.provider_tool,a.hostname,a.parent_run_id,a.previous_run_id,a.status,a.status_reason,a.status_observed_at,a.started_at,a.last_activity_at,a.ended_at,a.transcript_path,a.primary_worktree_id,a.primary_branch,a.start_head_sha,a.current_head_sha,a.event_count,a.error_count,a.freshness_json FROM agent_runs a WHERE a.id=?1", + [run_id], + run_row, + ) + .optional() + .context("resolve observatory run row") +} + +pub fn list_observatory_run_actors( + pool: &DbPool, + run_id: i64, + limit: usize, +) -> Result> { + let conn = pool.get()?; + let mut stmt = conn.prepare( + "SELECT id,actor_key,run_id,native_actor_id,actor_type,display_name,started_at,last_activity_at,ended_at,metadata_json + FROM agent_run_actors + WHERE run_id=?1 + ORDER BY COALESCE(last_activity_at,started_at,'') DESC,id DESC + LIMIT ?2", + )?; + stmt.query_map( + rusqlite::params![run_id, (bounded_limit(limit, 200) + 1) as i64], + |r| { + Ok(ObservatoryActorRow { + id: r.get(0)?, + actor_key: r.get(1)?, + run_id: r.get(2)?, + native_actor_id: r.get(3)?, + actor_type: r.get(4)?, + display_name: r.get(5)?, + started_at: r.get(6)?, + last_activity_at: r.get(7)?, + ended_at: r.get(8)?, + metadata_json: r.get(9)?, + }) + }, + )? + .collect::>() + .context("list observatory run actors") +} + fn run_row(r: &Row<'_>) -> rusqlite::Result { Ok(ObservatoryRunRow { id: r.get(0)?, @@ -380,20 +428,22 @@ fn run_row(r: &Row<'_>) -> rusqlite::Result { tool: r.get(3)?, provider_tool: r.get(4)?, hostname: r.get(5)?, - status: r.get(6)?, - status_reason: r.get(7)?, - status_observed_at: r.get(8)?, - started_at: r.get(9)?, - last_activity_at: r.get(10)?, - ended_at: r.get(11)?, - transcript_path: r.get(12)?, - primary_worktree_id: r.get(13)?, - primary_branch: r.get(14)?, - start_head_sha: r.get(15)?, - current_head_sha: r.get(16)?, - event_count: r.get(17)?, - error_count: r.get(18)?, - freshness_json: r.get(19)?, + parent_run_id: r.get(6)?, + previous_run_id: r.get(7)?, + status: r.get(8)?, + status_reason: r.get(9)?, + status_observed_at: r.get(10)?, + started_at: r.get(11)?, + last_activity_at: r.get(12)?, + ended_at: r.get(13)?, + transcript_path: r.get(14)?, + primary_worktree_id: r.get(15)?, + primary_branch: r.get(16)?, + start_head_sha: r.get(17)?, + current_head_sha: r.get(18)?, + event_count: r.get(19)?, + error_count: r.get(20)?, + freshness_json: r.get(21)?, }) } diff --git a/src/db/agent_observatory_read_models.rs b/src/db/agent_observatory_read_models.rs index 4d25b276b..0cb195a59 100644 --- a/src/db/agent_observatory_read_models.rs +++ b/src/db/agent_observatory_read_models.rs @@ -132,6 +132,10 @@ pub struct ObservatoryRunRow { pub tool: String, pub provider_tool: Option, pub hostname: String, + #[serde(serialize_with = "serialize_optional_id")] + pub parent_run_id: Option, + #[serde(serialize_with = "serialize_optional_id")] + pub previous_run_id: Option, pub status: String, pub status_reason: String, pub status_observed_at: String, @@ -148,6 +152,22 @@ pub struct ObservatoryRunRow { pub error_count: i64, pub freshness_json: String, } +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +pub struct ObservatoryActorRow { + #[serde(serialize_with = "serialize_id")] + pub id: i64, + pub actor_key: String, + #[serde(serialize_with = "serialize_id")] + pub run_id: i64, + pub native_actor_id: String, + pub actor_type: Option, + pub display_name: Option, + pub started_at: Option, + pub last_activity_at: Option, + pub ended_at: Option, + pub metadata_json: String, +} + #[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] pub struct ObservatoryEventRow { #[serde(serialize_with = "serialize_id")] diff --git a/src/db/agent_observatory_read_tests.rs b/src/db/agent_observatory_read_tests.rs index 77a2ae69b..213b9c3a1 100644 --- a/src/db/agent_observatory_read_tests.rs +++ b/src/db/agent_observatory_read_tests.rs @@ -311,3 +311,55 @@ fn run_resolution_reports_unknown_runs_as_none() { assert_eq!(identity.provider_tool.as_deref(), Some("openai")); assert_eq!(identity.native_session_id, "session"); } + +#[test] +fn run_reads_expose_lineage_and_bounded_actors() { + let dir = tempfile::tempdir().unwrap(); + let pool = init_pool(&StorageConfig::for_test(dir.path().join("run-lineage.db"))).unwrap(); + let conn = pool.get().unwrap(); + conn.execute("INSERT INTO agent_runs(run_key,native_session_id,tool,hostname,status,status_observed_at,started_at,last_activity_at) VALUES('parent','parent-session','codex','host','completed','2026-08-21T09:00:00Z','2026-08-21T09:00:00Z','2026-08-21T09:30:00Z')", []).unwrap(); + let parent_id = conn.last_insert_rowid(); + conn.execute("INSERT INTO agent_runs(run_key,native_session_id,tool,hostname,status,status_observed_at,started_at,last_activity_at) VALUES('previous','previous-session','codex','host','completed','2026-08-21T09:30:00Z','2026-08-21T09:30:00Z','2026-08-21T09:45:00Z')", []).unwrap(); + let previous_id = conn.last_insert_rowid(); + conn.execute( + "INSERT INTO agent_runs(run_key,native_session_id,tool,hostname,parent_run_id,previous_run_id,status,status_observed_at,started_at,last_activity_at) VALUES('current','session','codex','host',?1,?2,'active','2026-08-21T10:00:00Z','2026-08-21T10:00:00Z','2026-08-21T10:10:00Z')", + rusqlite::params![parent_id, previous_id], + ).unwrap(); + let run_id = conn.last_insert_rowid(); + for (key, native, activity) in [ + ("actor-a", "subagent-a", "2026-08-21T10:05:00Z"), + ("actor-b", "subagent-b", "2026-08-21T10:06:00Z"), + ] { + conn.execute( + "INSERT INTO agent_run_actors(actor_key,run_id,native_actor_id,actor_type,display_name,started_at,last_activity_at,metadata_json) VALUES(?1,?2,?3,'subagent',?3,'2026-08-21T10:00:00Z',?4,'{}')", + rusqlite::params![key, run_id, native, activity], + ).unwrap(); + } + drop(conn); + + let current = resolve_observatory_run_row(&pool, run_id).unwrap().unwrap(); + assert_eq!(current.parent_run_id, Some(parent_id)); + assert_eq!(current.previous_run_id, Some(previous_id)); + assert_eq!( + resolve_observatory_run_row(&pool, parent_id) + .unwrap() + .unwrap() + .run_key, + "parent" + ); + assert_eq!( + resolve_observatory_run_row(&pool, previous_id) + .unwrap() + .unwrap() + .run_key, + "previous" + ); + + let actors = list_observatory_run_actors(&pool, run_id, 1).unwrap(); + assert_eq!( + actors.len(), + 2, + "bounded reads return limit + 1 for truncation detection" + ); + assert_eq!(actors[0].native_actor_id, "subagent-b"); +} From 6951319b1f71e6d550640e7b7dae1704c17363ae Mon Sep 17 00:00:00 2001 From: jmagar <38927646+jmagar@users.noreply.github.com> Date: Wed, 23 Sep 2026 13:16:16 -0400 Subject: [PATCH 3/8] fix: restore truncated session queries --- src/db/queries.rs | 2047 ++++++++++++++++++++++++++++++++++++++++++++- 1 file changed, 2044 insertions(+), 3 deletions(-) diff --git a/src/db/queries.rs b/src/db/queries.rs index e3fb20bfe..02225e409 100644 --- a/src/db/queries.rs +++ b/src/db/queries.rs @@ -1383,6 +1383,19 @@ pub fn search_ai_sessions( const CANDIDATE_CAP: usize = 5_000; +fn ai_session_fts_query(query: &str, tool: Option<&str>) -> String { + let mut scoped = format!("message : ({query})"); + if let Some(tool) = tool.filter(|tool| { + !tool.is_empty() + && tool + .chars() + .all(|ch| ch.is_ascii_alphanumeric() || matches!(ch, '_' | '-' | '.')) + }) { + scoped.push_str(&format!(" AND ai_tool : \"{tool}\"")); + } + scoped +} + fn search_ai_sessions_sql( params: &SearchAiSessionsParams, limit: usize, @@ -1392,7 +1405,21 @@ fn search_ai_sessions_sql( let mut query_params = SqlParams::new(2); query_params .bindings - .push(rusqlite::types::Value::Text(params.query.clone())); + .push(rusqlite::types::Value::Text(ai_session_fts_query( + ¶ms.query, + params.ai_tool.as_deref(), + ))); + let mut fts_rowid_floor = String::new(); + if let Some(since) = ¶ms.since { + let idx = query_params.push_text(since.clone()); + fts_rowid_floor = format!( + " AND ai_logs_fts.rowid >= COALESCE( + (SELECT MIN(id) + FROM logs INDEXED BY idx_logs_ai_timestamp_tool + WHERE ai_tool IS NOT NULL AND timestamp >= ?{idx}), + 9223372036854775807)" + ); + } push_ai_scope_filters( &mut filters, &mut query_params, @@ -1420,7 +1447,7 @@ fn search_ai_sessions_sql( l.message FROM ai_logs_fts JOIN logs l ON l.id = ai_logs_fts.rowid - WHERE ai_logs_fts MATCH ?1{filters} + WHERE ai_logs_fts MATCH ?1{fts_rowid_floor}{filters} ORDER BY ai_logs_fts.rowid DESC LIMIT {} ), @@ -2146,4 +2173,2018 @@ pub fn topic_correlate_inputs( && !covered.contains(entity.canonical_key.as_str()) { entity.resolver_status = ResolverStatus::Degraded; - } \ No newline at end of file + } + } + instance_keys.extend(linked); + } + instance_keys.sort(); + instance_keys.dedup(); + generic_seeds.sort(); + generic_seeds.dedup(); + + // Graph expansion + host discovery. Service seeds use the bounded + // service-topic walk (proof relationships only); generic seeds keep the + // general walk. + let mut walk_entities = Vec::new(); + let mut service_seeds: Vec = logical_keys.clone(); + service_seeds.extend(instance_keys.iter().cloned()); + service_seeds.sort(); + service_seeds.dedup(); + let mut graph_walk_truncated = false; + if !service_seeds.is_empty() { + let (entities, truncated) = super::graph_resolver_projection::graph_walk_service_topic( + &conn, + &service_seeds, + max_depth, + )?; + walk_entities.extend(entities); + graph_walk_truncated |= truncated; + } + if !generic_seeds.is_empty() { + walk_entities.extend(super::graph::graph_walk_n_hops( + &conn, + &generic_seeds, + max_depth, + )?); + } + let seed_set: std::collections::HashSet<&str> = service_seeds + .iter() + .chain(generic_seeds.iter()) + .map(String::as_str) + .collect(); + let mut expansion: Vec<(String, String)> = Vec::new(); + let mut discovered_hosts: Vec = Vec::new(); + for entity in walk_entities { + match entity.entity_type.as_str() { + super::graph::ENTITY_TYPE_HOST => discovered_hosts.push(entity.canonical_key.clone()), + super::graph::ENTITY_TYPE_CONTAINER => { + if let Some(host) = + super::entity_resolution::container_key_host(&entity.canonical_key) + { + discovered_hosts.push(host.to_string()); + } + } + super::graph::ENTITY_TYPE_SERVICE_INSTANCE => { + if let Some((host, _)) = + super::entity_resolution::split_service_instance_key(&entity.canonical_key) + { + discovered_hosts.push(host.to_string()); + } + } + _ => {} + } + if !seed_set.contains(entity.canonical_key.as_str()) { + expansion.push((entity.entity_type, entity.canonical_key)); + } + } + discovered_hosts.sort(); + discovered_hosts.dedup(); + expansion.sort(); + expansion.dedup(); + drop(conn); + + // Log fan-out: service instances use service-scoped predicates; generic + // seeds use the graph-related fan-out. Never both for the same rows — + // results merge newest-first under the shared limit. + let mut logs: Vec = Vec::new(); + if !instance_keys.is_empty() { + logs.extend( + queries_service_instances::search_logs_for_service_instances( + pool, + &instance_keys, + since, + until, + source_kinds, + limit, + )?, + ); + } + if !generic_seeds.is_empty() { + // SeedHostsOnly: a host reached transitively from an app/container + // seed (e.g. app:plex —emitted_by→ host:nashost) must never drive + // host-wide `l.hostname IN (…)` inclusion labelled `resolved`. Only + // hosts that were exact topic matches themselves fan out. + logs.extend( + search_logs_from_graph_related_entities( + pool, + &generic_seeds, + max_depth, + since, + until, + source_kinds, + limit, + HostFanoutScope::SeedHostsOnly, + )? + .into_iter() + .map(|entry| GraphRelatedLogEntry { + entry, + inclusion_reason: INCLUSION_GRAPH_RELATED.to_string(), + resolver_status: ResolverStatus::Resolved, + fallback_kind: None, + }), + ); + } + + // Explicit degraded host-context fallback: a service topic whose + // instance predicates matched no rows falls back to the instances' host + // context, annotated (`explicit_degraded_host_context`) — never silent. + if logs.is_empty() && !instance_keys.is_empty() && generic_seeds.is_empty() { + let hosts: Vec = instance_keys + .iter() + .filter_map(|key| { + let split = super::entity_resolution::split_service_instance_key(key); + if split.is_none() { + tracing::debug!( + key = %key, + "discarding non-canonical service_instance key in host-context fallback" + ); + } + split.map(|(host, _)| host.to_string()) + }) + .collect(); + if !hosts.is_empty() { + logs.extend( + queries_service_instances::search_logs_by_hostnames( + pool, + &hosts, + since, + until, + source_kinds, + limit, + )? + .into_iter() + .map(|entry| GraphRelatedLogEntry { + entry, + inclusion_reason: INCLUSION_HOST_CONTEXT.to_string(), + resolver_status: ResolverStatus::Degraded, + fallback_kind: Some(FALLBACK_EXPLICIT_DEGRADED_HOST_CONTEXT.to_string()), + }), + ); + } + } + + logs.sort_by(|a, b| { + b.entry + .timestamp + .cmp(&a.entry.timestamp) + .then_with(|| b.entry.id.cmp(&a.entry.id)) + }); + logs.dedup_by_key(|row| row.entry.id); + logs.truncate(limit); + + Ok(TopicGraphInputs { + resolved, + expansion, + discovered_hosts, + logs, + graph_walk_truncated, + }) +} + +fn topic_correlate_ai_project_fallback( + pool: &DbPool, + terms: &[String], + since: Option<&str>, + until: Option<&str>, + source_kinds: Option<&[SourceKind]>, + limit: usize, +) -> Result { + let conn = pool.get()?; + let predicates = terms + .iter() + .map(|_| "(lower(ai_project) = ? OR lower(ai_project) LIKE '%/' || ?)") + .collect::>() + .join(" OR "); + let mut sql = format!( + "SELECT DISTINCT ai_project FROM logs WHERE ai_project IS NOT NULL AND ({predicates})" + ); + let mut bindings = Vec::with_capacity(terms.len() * 2 + 2); + for term in terms { + bindings.push(rusqlite::types::Value::Text(term.clone())); + bindings.push(rusqlite::types::Value::Text(term.clone())); + } + if let Some(value) = since { + sql.push_str(" AND timestamp >= ?"); + bindings.push(rusqlite::types::Value::Text(value.to_string())); + } + if let Some(value) = until { + sql.push_str(" AND timestamp <= ?"); + bindings.push(rusqlite::types::Value::Text(value.to_string())); + } + sql.push_str(" ORDER BY ai_project LIMIT 32"); + + let projects = conn + .prepare(&sql)? + .query_map(rusqlite::params_from_iter(bindings.iter()), |row| { + row.get::<_, String>(0) + })? + .collect::>>()?; + drop(conn); + + let allowed_source_kinds = source_kinds.map(|kinds| { + kinds + .iter() + .map(|kind| kind.as_str()) + .collect::>() + }); + let mut logs = Vec::new(); + let mut discovered_hosts = Vec::new(); + let mut resolved = Vec::new(); + for project in projects { + let key = project + .rsplit('/') + .next() + .unwrap_or(&project) + .to_ascii_lowercase(); + resolved.push(ResolvedTopicEntity { + entity_type: super::graph::ENTITY_TYPE_AI_PROJECT.to_string(), + canonical_key: key, + match_kind: "exact", + resolver_status: ResolverStatus::Degraded, + }); + let rows = search_logs( + pool, + &SearchParams { + ai_project: Some(project), + since: since.map(str::to_string), + until: until.map(str::to_string), + limit: Some(limit.min(1000) as u32), + ..Default::default() + }, + )?; + for entry in rows { + if let Some(allowed) = &allowed_source_kinds { + let source_kind = entry + .metadata_json + .as_deref() + .and_then(|json| serde_json::from_str::(json).ok()) + .and_then(|value| value.get("source_kind")?.as_str().map(str::to_string)); + if !source_kind + .as_deref() + .is_some_and(|kind| allowed.contains(kind)) + { + continue; + } + } + discovered_hosts.push(entry.hostname.clone()); + logs.push(GraphRelatedLogEntry { + entry, + inclusion_reason: "direct_source_identity".to_string(), + resolver_status: ResolverStatus::Degraded, + fallback_kind: Some("direct_source_identity".to_string()), + }); + } + } + resolved.sort_by(|a, b| a.canonical_key.cmp(&b.canonical_key)); + resolved.dedup_by(|a, b| a.canonical_key == b.canonical_key); + discovered_hosts.sort(); + discovered_hosts.dedup(); + logs.sort_by(|a, b| b.entry.timestamp.cmp(&a.entry.timestamp)); + logs.dedup_by_key(|row| row.entry.id); + logs.truncate(limit); + + Ok(TopicGraphInputs { + resolved, + discovered_hosts, + logs, + ..Default::default() + }) +} + +const DEFAULT_AI_ABUSE_TERMS: &[&str] = &[ + "asshole", "bastard", "bitch", "biznitch", "bullshit", "crap", "damn", "dick", "fuck", + "fucked", "fucker", "fucking", "hell", "piss", "shit", "shitty", +]; + +pub fn search_ai_abuse(pool: &DbPool, params: &AiAbuseParams) -> Result { + let limit = params.limit.unwrap_or(20).clamp(1, 100) as usize; + let before = params.before.unwrap_or(2).min(20); + let after = params.after.unwrap_or(2).min(20); + let terms = normalized_abuse_terms(¶ms.terms); + if terms.len() > 16 { + anyhow::bail!("Too many abuse terms ({}); maximum is 16", terms.len()); + } + let conn = pool.get()?; + const CANDIDATE_CAP: usize = 10_000; + + let mut sql = String::from( + "WITH candidates(id) AS MATERIALIZED ( + SELECT l.id + FROM logs_fts + JOIN logs l ON l.id = logs_fts.rowid + WHERE logs_fts MATCH ?1 + AND l.ai_project IS NOT NULL AND l.ai_project != '' + AND l.ai_tool IS NOT NULL AND l.ai_tool != '' + AND l.ai_session_id IS NOT NULL AND l.ai_session_id != ''", + ); + let mut bindings = vec![rusqlite::types::Value::Text(abuse_fts_query(&terms))]; + let mut idx = 2usize; + + if let Some(project) = ¶ms.ai_project { + sql.push_str(&format!(" AND l.ai_project = ?{idx}")); + bindings.push(rusqlite::types::Value::Text(project.clone())); + idx += 1; + } + if let Some(tool) = ¶ms.ai_tool { + sql.push_str(&format!(" AND l.ai_tool = ?{idx}")); + bindings.push(rusqlite::types::Value::Text(tool.clone())); + idx += 1; + } + if let Some(from) = ¶ms.since { + sql.push_str(&format!(" AND l.timestamp >= ?{idx}")); + bindings.push(rusqlite::types::Value::Text(from.clone())); + idx += 1; + } + if let Some(to) = ¶ms.until { + sql.push_str(&format!(" AND l.timestamp <= ?{idx}")); + bindings.push(rusqlite::types::Value::Text(to.clone())); + } + sql.push_str(&format!( + " ORDER BY logs_fts.rowid DESC LIMIT {} + ) + SELECT l.id, l.timestamp, l.hostname, l.facility, l.severity, + l.app_name, l.process_id, l.message, l.received_at, l.source_ip, + l.ai_tool, l.ai_project, l.ai_session_id, l.ai_transcript_path, l.metadata_json + FROM candidates c + JOIN logs l ON l.id = c.id + ORDER BY l.timestamp DESC, l.id DESC", + CANDIDATE_CAP + 1 + )); + + let mut stmt = conn.prepare(&sql)?; + let candidate_rows = stmt + .query_map(rusqlite::params_from_iter(bindings.iter()), map_row)? + .collect::>>()?; + + let candidate_window_truncated = candidate_rows.len() > CANDIDATE_CAP; + let mut matches = Vec::new(); + let mut result_limit_truncated = false; + for entry in candidate_rows.iter().take(CANDIDATE_CAP) { + if let Some(term) = first_abuse_term(&entry.message, &terms) { + if matches.len() == limit { + result_limit_truncated = true; + break; + } + let (before_rows, after_rows) = ai_session_context(&conn, entry, before, after)?; + matches.push(AiAbuseMatch { + term, + entry: entry.clone(), + before: before_rows, + after: after_rows, + }); + } + } + + Ok(AiAbuseResult { + terms, + candidate_rows: candidate_rows.len().min(CANDIDATE_CAP), + candidate_cap: CANDIDATE_CAP, + candidate_window_truncated, + truncated: candidate_window_truncated || result_limit_truncated, + matches, + }) +} + +pub fn list_ai_tools(pool: &DbPool, params: &ListAiToolsParams) -> Result { + let conn = pool.get()?; + const LIMIT: usize = 100; + let mut sql = String::from( + "SELECT ai_tool, + COUNT(*) AS event_count, + COUNT(DISTINCT ai_session_id) AS session_count, + MIN(timestamp) AS first_seen, + MAX(timestamp) AS last_seen + FROM logs + WHERE ai_tool IS NOT NULL + AND ai_tool != ''", + ); + let mut bindings: Vec = vec![]; + let mut idx = 1usize; + + if let Some(project) = ¶ms.ai_project { + sql.push_str(&format!(" AND ai_project = ?{idx}")); + bindings.push(rusqlite::types::Value::Text(project.clone())); + idx += 1; + } + if let Some(from) = ¶ms.since { + sql.push_str(&format!(" AND timestamp >= ?{idx}")); + bindings.push(rusqlite::types::Value::Text(from.clone())); + idx += 1; + } + if let Some(to) = ¶ms.until { + sql.push_str(&format!(" AND timestamp <= ?{idx}")); + bindings.push(rusqlite::types::Value::Text(to.clone())); + } + sql.push_str(&format!( + " GROUP BY ai_tool ORDER BY event_count DESC, ai_tool ASC LIMIT {}", + LIMIT + 1 + )); + + let mut stmt = conn.prepare(&sql)?; + let mut tools = stmt + .query_map(rusqlite::params_from_iter(bindings.iter()), |row| { + Ok(AiToolInventoryEntry { + tool: row.get(0)?, + event_count: row.get(1)?, + session_count: row.get(2)?, + first_seen: row.get(3)?, + last_seen: row.get(4)?, + }) + })? + .collect::>>()?; + let truncated = truncate_to_limit(&mut tools, LIMIT); + Ok(ListAiToolsResult { + total_tools: tools.len(), + truncated, + tools, + }) +} + +pub fn list_ai_projects( + pool: &DbPool, + params: &ListAiProjectsParams, +) -> Result { + let conn = pool.get()?; + const LIMIT: usize = 200; + let mut sql = String::from( + "SELECT ai_project, + GROUP_CONCAT(DISTINCT ai_tool) AS tools, + COUNT(*) AS event_count, + COUNT(DISTINCT ai_session_id) AS session_count, + MIN(timestamp) AS first_seen, + MAX(timestamp) AS last_seen + FROM logs + WHERE ai_project IS NOT NULL + AND ai_project != ''", + ); + let mut bindings: Vec = vec![]; + let mut idx = 1usize; + + if let Some(tool) = ¶ms.ai_tool { + sql.push_str(&format!(" AND ai_tool = ?{idx}")); + bindings.push(rusqlite::types::Value::Text(tool.clone())); + idx += 1; + } + if let Some(from) = ¶ms.since { + sql.push_str(&format!(" AND timestamp >= ?{idx}")); + bindings.push(rusqlite::types::Value::Text(from.clone())); + idx += 1; + } + if let Some(to) = ¶ms.until { + sql.push_str(&format!(" AND timestamp <= ?{idx}")); + bindings.push(rusqlite::types::Value::Text(to.clone())); + } + sql.push_str(&format!( + " GROUP BY ai_project ORDER BY event_count DESC, ai_project ASC LIMIT {}", + LIMIT + 1 + )); + + let mut stmt = conn.prepare(&sql)?; + let mut projects = stmt + .query_map(rusqlite::params_from_iter(bindings.iter()), |row| { + let tools = row + .get::<_, Option>(1)? + .unwrap_or_default() + .split(',') + .filter(|value| !value.is_empty()) + .map(ToString::to_string) + .collect(); + Ok(AiProjectInventoryEntry { + project: row.get(0)?, + tools, + event_count: row.get(2)?, + session_count: row.get(3)?, + first_seen: row.get(4)?, + last_seen: row.get(5)?, + }) + })? + .collect::>>()?; + let truncated = truncate_to_limit(&mut projects, LIMIT); + Ok(ListAiProjectsResult { + total_projects: projects.len(), + truncated, + projects, + }) +} + +fn truncate_to_limit(values: &mut Vec, limit: usize) -> bool { + let truncated = values.len() > limit; + values.truncate(limit); + truncated +} + +pub fn search_ai_incidents(pool: &DbPool, params: &AiIncidentParams) -> Result { + use std::collections::HashMap; + + let limit = params.limit.unwrap_or(20).clamp(1, 100) as usize; + let window_secs = i64::from(params.window_minutes.unwrap_or(10).clamp(1, 120)) * 60; + let terms = normalized_abuse_terms(¶ms.terms); + const CANDIDATE_CAP: usize = 10_000; + + let conn = pool.get()?; + let (sql, bindings) = ai_incident_anchor_sql(params, &terms, CANDIDATE_CAP); + + // Fetch candidate abuse anchor rows (same FTS path as search_ai_abuse, + // no per-hit context needed here). + struct AnchorRow { + id: i64, + timestamp: String, + hostname: String, + tool: String, + project: String, + session_id: String, + message: String, + } + + let mut stmt = conn.prepare(&sql)?; + let candidate_rows: Vec = stmt + .query_map(rusqlite::params_from_iter(bindings.iter()), |row| { + Ok(AnchorRow { + id: row.get(0)?, + timestamp: row.get(1)?, + hostname: row.get(2)?, + tool: row.get(3)?, + project: row.get(4)?, + session_id: row.get(5)?, + message: row.get(6)?, + }) + })? + .collect::>>()?; + + let candidate_window_truncated = candidate_rows.len() > CANDIDATE_CAP; + let raw_candidate_count = candidate_rows.len(); + + // Group by (project, tool, session_id, hostname) + window-minute buckets. + // Key: (project, tool, session_id, hostname, window_bucket) + // window_bucket = unix_secs / window_secs * window_secs (floor to window boundary) + type GroupKey = (String, String, String, String, i64); + let mut groups: HashMap> = HashMap::new(); + + for row in candidate_rows.iter().take(CANDIDATE_CAP) { + // Parse timestamp to unix seconds for bucketing. + let bucket = chrono::DateTime::parse_from_rfc3339(&row.timestamp) + .map(|dt| { + let secs = dt.timestamp(); + (secs / window_secs) * window_secs + }) + .unwrap_or(0); + let key = ( + row.project.clone(), + row.tool.clone(), + row.session_id.clone(), + row.hostname.clone(), + bucket, + ); + groups.entry(key).or_default().push(row); + } + + // Build incidents from groups. + let mut incidents: Vec = groups + .into_iter() + .map( + |((project, tool, session_id, hostname, _bucket), anchors)| { + let abuse_count = anchors.len(); + let first_seen = anchors + .first() + .map(|r| r.timestamp.clone()) + .unwrap_or_default(); + let last_seen = anchors + .last() + .map(|r| r.timestamp.clone()) + .unwrap_or_default(); + + // duration in seconds + let duration_secs = { + let t0 = chrono::DateTime::parse_from_rfc3339(&first_seen) + .map(|dt| dt.timestamp()) + .unwrap_or(0); + let t1 = chrono::DateTime::parse_from_rfc3339(&last_seen) + .map(|dt| dt.timestamp()) + .unwrap_or(0); + (t1 - t0).max(0) + }; + + // Collect unique terms found in this group's messages. + let mut found_terms: Vec = terms + .iter() + .filter(|term| { + anchors.iter().any(|r| { + first_abuse_term(&r.message, std::slice::from_ref(term)).is_some() + }) + }) + .cloned() + .collect(); + found_terms.sort(); + found_terms.dedup(); + + let mut anchor_ids: Vec = anchors.iter().map(|r| r.id).collect(); + anchor_ids.sort(); + + // Score: abuse_count dominates; density and term variety boost. + let density = if duration_secs > 0 { + abuse_count as f64 / (duration_secs as f64 / 60.0) + } else { + abuse_count as f64 + }; + let term_variety = found_terms.len() as f64; + let priority_score = abuse_count as f64 * 10.0 + density * 2.0 + term_variety; + + // Compare the f64 directly: `as u64` truncates and maps NaN + // to 0, which would mislabel a pathological score as "low" + // (full-review QL2). + let priority_label = if priority_score < 15.0 { + "low" + } else if priority_score < 30.0 { + "medium" + } else if priority_score < 50.0 { + "high" + } else { + "critical" + } + .to_string(); + + // Stable incident ID using a deterministic hash of session identity + anchor IDs. + let incident_id = { + use std::collections::hash_map::DefaultHasher; + use std::hash::{Hash, Hasher}; + let mut h = DefaultHasher::new(); + project.hash(&mut h); + tool.hash(&mut h); + session_id.hash(&mut h); + hostname.hash(&mut h); + for id in &anchor_ids { + id.hash(&mut h); + } + format!("inc-{:016x}", h.finish()) + }; + + AbuseIncident { + incident_id, + project, + tool, + session_id, + hostname, + first_seen, + last_seen, + duration_secs, + abuse_count, + terms: found_terms, + anchor_ids, + priority_score, + priority_label, + window_minutes: (window_secs / 60) as u32, + } + }, + ) + .collect(); + + // Sort by priority_score descending, then last_seen descending. + // total_cmp is a total order (NaN sorts deterministically) — the + // partial_cmp/unwrap_or(Equal) idiom can produce a non-total order if a + // NaN ever sneaks into a score (full-review QL3). + incidents.sort_by(|a, b| { + b.priority_score + .total_cmp(&a.priority_score) + .then_with(|| b.last_seen.cmp(&a.last_seen)) + }); + + let total_incidents = incidents.len(); + let truncated = total_incidents > limit || candidate_window_truncated; + incidents.truncate(limit); + + Ok(AiIncidentResult { + incidents, + total_incidents, + candidate_rows: raw_candidate_count.min(CANDIDATE_CAP), + candidate_cap: CANDIDATE_CAP, + candidate_window_truncated, + truncated, + }) +} + +fn ai_incident_anchor_sql( + params: &AiIncidentParams, + terms: &[String], + candidate_cap: usize, +) -> (String, Vec) { + let mut sql = String::from( + "WITH candidates(id) AS MATERIALIZED ( + SELECT l.id + FROM logs_fts + JOIN logs l ON l.id = logs_fts.rowid + WHERE logs_fts MATCH ?1 + AND l.ai_project IS NOT NULL AND l.ai_project != '' + AND l.ai_tool IS NOT NULL AND l.ai_tool != '' + AND l.ai_session_id IS NOT NULL AND l.ai_session_id != ''", + ); + let mut bindings = vec![rusqlite::types::Value::Text(abuse_fts_query(terms))]; + let mut idx = 2usize; + + if let Some(project) = ¶ms.ai_project { + sql.push_str(&format!(" AND l.ai_project = ?{idx}")); + bindings.push(rusqlite::types::Value::Text(project.clone())); + idx += 1; + } + if let Some(tool) = ¶ms.ai_tool { + sql.push_str(&format!(" AND l.ai_tool = ?{idx}")); + bindings.push(rusqlite::types::Value::Text(tool.clone())); + idx += 1; + } + if let Some(from) = ¶ms.since { + sql.push_str(&format!(" AND l.timestamp >= ?{idx}")); + bindings.push(rusqlite::types::Value::Text(from.clone())); + idx += 1; + } + if let Some(to) = ¶ms.until { + sql.push_str(&format!(" AND l.timestamp <= ?{idx}")); + bindings.push(rusqlite::types::Value::Text(to.clone())); + } + let _ = idx; + sql.push_str(&format!( + " ORDER BY logs_fts.rowid ASC LIMIT {} + ) + SELECT l.id, l.timestamp, l.hostname, + l.ai_tool, l.ai_project, l.ai_session_id, l.message + FROM candidates c + JOIN logs l ON l.id = c.id", + candidate_cap + 1 + )); + (sql, bindings) +} + +pub fn investigate_ai_incidents( + pool: &DbPool, + params: &AiInvestigateParams, +) -> Result { + let limit = params.limit.unwrap_or(3).clamp(1, 10) as usize; + let incident_lookup_limit = if params.incident_id.is_some() { + 100 + } else { + limit as u32 + }; + let corr_mins = i64::from(params.correlation_window_minutes.unwrap_or(5).clamp(1, 120)); + + // Reuse incident grouping to find the top incidents. Exact incident + // assessment may target an ID outside the top investigation page, so it + // searches up to the incident-list cap and then builds one evidence bundle. + let incident_result = search_ai_incidents( + pool, + &AiIncidentParams { + ai_project: params.ai_project.clone(), + ai_tool: params.ai_tool.clone(), + since: params.since.clone(), + until: params.until.clone(), + limit: Some(incident_lookup_limit), + window_minutes: params.window_minutes, + terms: params.terms.clone(), + }, + )?; + let total_incidents = incident_result.total_incidents; + let truncated = incident_result.truncated; + let incidents = if let Some(incident_id) = ¶ms.incident_id { + incident_result + .incidents + .into_iter() + .filter(|incident| incident.incident_id == *incident_id) + .collect() + } else { + incident_result.incidents + }; + + let conn = pool.get()?; + let mut evidence = Vec::with_capacity(incidents.len()); + + for incident in incidents { + const TRANSCRIPT_CAP: usize = 20; + const NEARBY_CAP: usize = 50; + + // Fetch anchor log entries. + let anchors = if incident.anchor_ids.is_empty() { + Vec::new() + } else { + let placeholders: Vec = (1..=incident.anchor_ids.len()) + .map(|i| format!("?{i}")) + .collect(); + let sql = format!( + "SELECT id, timestamp, hostname, facility, severity, app_name, + process_id, message, received_at, source_ip, + ai_tool, ai_project, ai_session_id, ai_transcript_path, metadata_json + FROM logs WHERE id IN ({}) ORDER BY timestamp ASC", + placeholders.join(",") + ); + let mut stmt = conn.prepare(&sql)?; + + stmt.query_map( + rusqlite::params_from_iter( + incident + .anchor_ids + .iter() + .map(|id| rusqlite::types::Value::Integer(*id)), + ), + map_row, + )? + .collect::>>()? + }; + + // Transcript context: entries in the same session before first anchor and after last anchor. + let (transcript_before, transcript_before_truncated) = if let Some(first) = anchors.first() + { + let rows = { + let mut stmt = conn.prepare( + "SELECT id, timestamp, hostname, facility, severity, app_name, + process_id, message, received_at, source_ip, + ai_tool, ai_project, ai_session_id, ai_transcript_path, + metadata_json + FROM logs + WHERE ai_session_id = ?1 AND ai_project = ?2 AND ai_tool = ?3 + AND timestamp < ?4 + ORDER BY timestamp DESC + LIMIT 21", + )?; + + stmt.query_map( + rusqlite::params![ + &incident.session_id, + &incident.project, + &incident.tool, + &first.timestamp, + ], + map_row, + )? + .collect::>>()? + }; + let truncated = rows.len() > TRANSCRIPT_CAP; + let mut out = rows; + out.truncate(TRANSCRIPT_CAP); + out.reverse(); // chronological order + (out, truncated) + } else { + (Vec::new(), false) + }; + + let (transcript_after, transcript_after_truncated) = if let Some(last) = anchors.last() { + let rows = { + let mut stmt = conn.prepare( + "SELECT id, timestamp, hostname, facility, severity, app_name, + process_id, message, received_at, source_ip, + ai_tool, ai_project, ai_session_id, ai_transcript_path, metadata_json + FROM logs + WHERE ai_session_id = ?1 AND ai_project = ?2 AND ai_tool = ?3 + AND timestamp > ?4 + ORDER BY timestamp ASC + LIMIT 21", + )?; + + stmt.query_map( + rusqlite::params![ + &incident.session_id, + &incident.project, + &incident.tool, + &last.timestamp, + ], + map_row, + )? + .collect::>>()? + }; + let truncated = rows.len() > TRANSCRIPT_CAP; + let mut out = rows; + out.truncate(TRANSCRIPT_CAP); + (out, truncated) + } else { + (Vec::new(), false) + }; + + // Nearby non-AI logs in the correlation window. + let (nearby_logs, nearby_logs_truncated) = { + // Window: corr_mins before first_seen through corr_mins after last_seen. + let win_from = chrono::DateTime::parse_from_rfc3339(&incident.first_seen) + .map(|dt| { + use chrono::Duration; + (dt.with_timezone(&chrono::Utc) - Duration::minutes(corr_mins)) + .format("%Y-%m-%dT%H:%M:%S%.3fZ") + .to_string() + }) + .unwrap_or_else(|_| incident.first_seen.clone()); + let win_to = chrono::DateTime::parse_from_rfc3339(&incident.last_seen) + .map(|dt| { + use chrono::Duration; + (dt.with_timezone(&chrono::Utc) + Duration::minutes(corr_mins)) + .format("%Y-%m-%dT%H:%M:%S%.3fZ") + .to_string() + }) + .unwrap_or_else(|_| incident.last_seen.clone()); + + let mut stmt = conn.prepare( + "SELECT id, timestamp, hostname, facility, severity, app_name, + process_id, message, received_at, source_ip, + ai_tool, ai_project, ai_session_id, ai_transcript_path, metadata_json + FROM logs + WHERE timestamp >= ?1 AND timestamp <= ?2 + AND (ai_project IS NULL OR ai_project = '') + ORDER BY timestamp ASC + LIMIT 51", + )?; + let rows = stmt + .query_map(rusqlite::params![win_from, win_to], map_row)? + .collect::>>()?; + let truncated = rows.len() > NEARBY_CAP; + let mut out = rows; + out.truncate(NEARBY_CAP); + (out, truncated) + }; + + // Nearby errors: subset of nearby_logs with severity warning+. + let error_sevs = ["emergency", "alert", "critical", "error", "warning"]; + let nearby_errors: Vec = nearby_logs + .iter() + .filter(|e| error_sevs.contains(&e.severity.as_str())) + .cloned() + .collect(); + + evidence.push(IncidentEvidence { + incident, + transcript_before, + transcript_before_truncated, + transcript_after, + transcript_after_truncated, + anchors, + nearby_logs, + nearby_logs_truncated, + nearby_errors, + }); + } + + Ok(AiInvestigateResult { + evidence, + total_incidents, + truncated, + }) +} + +fn normalized_abuse_terms(custom_terms: &[String]) -> Vec { + let source: Vec = if custom_terms.is_empty() { + DEFAULT_AI_ABUSE_TERMS + .iter() + .map(|term| (*term).to_string()) + .collect() + } else { + custom_terms.to_vec() + }; + + let mut terms = source + .into_iter() + .map(|term| term.trim().to_ascii_lowercase()) + .filter(|term| { + !term.is_empty() + && term.len() <= 64 + && term + .chars() + .all(|ch| ch.is_ascii_alphanumeric() || ch == '-' || ch == '_') + }) + .collect::>(); + terms.sort(); + terms.dedup(); + if terms.is_empty() { + DEFAULT_AI_ABUSE_TERMS + .iter() + .map(|term| (*term).to_string()) + .collect() + } else { + terms + } +} + +fn abuse_fts_query(terms: &[String]) -> String { + // FTS5 escapes a literal " inside a phrase by doubling it: "" → match one " + terms + .iter() + .map(|term| format!("\"{}\"", term.replace('"', "\"\""))) + .collect::>() + .join(" OR ") +} + +fn first_abuse_term(message: &str, terms: &[String]) -> Option { + let lower = message.to_ascii_lowercase(); + terms + .iter() + .filter_map(|term| first_term_index(&lower, term).map(|idx| (idx, term))) + .min_by_key(|(idx, _)| *idx) + .map(|(_, term)| term.clone()) +} + +fn first_term_index(message: &str, term: &str) -> Option { + let mut offset = 0usize; + while let Some(relative) = message[offset..].find(term) { + let start = offset + relative; + let end = start + term.len(); + if is_abuse_boundary(message[..start].chars().next_back()) + && is_abuse_boundary(message[end..].chars().next()) + { + return Some(start); + } + offset = end; + } + None +} + +fn is_abuse_boundary(ch: Option) -> bool { + ch.is_none_or(|ch| !ch.is_ascii_alphanumeric() && ch != '_') +} + +fn ai_session_context( + conn: &rusqlite::Connection, + entry: &LogEntry, + before: u32, + after: u32, +) -> Result<(Vec, Vec)> { + let Some(tool) = entry.ai_tool.as_deref() else { + return Ok((Vec::new(), Vec::new())); + }; + let Some(project) = entry.ai_project.as_deref() else { + return Ok((Vec::new(), Vec::new())); + }; + let Some(session_id) = entry.ai_session_id.as_deref() else { + return Ok((Vec::new(), Vec::new())); + }; + + let mut before_stmt = conn.prepare( + "SELECT id, timestamp, hostname, facility, severity, + app_name, process_id, message, received_at, source_ip, + ai_tool, ai_project, ai_session_id, ai_transcript_path, metadata_json + FROM logs + WHERE hostname = ?1 + AND ai_tool = ?2 + AND ai_project = ?3 + AND ai_session_id = ?4 + AND (timestamp < ?5 OR (timestamp = ?5 AND id < ?6)) + ORDER BY timestamp DESC, id DESC + LIMIT ?7", + )?; + let mut before_rows = before_stmt + .query_map( + params![ + &entry.hostname, + tool, + project, + session_id, + &entry.timestamp, + entry.id, + before + ], + map_row, + )? + .collect::>>()?; + before_rows.reverse(); + + let mut after_stmt = conn.prepare( + "SELECT id, timestamp, hostname, facility, severity, + app_name, process_id, message, received_at, source_ip, + ai_tool, ai_project, ai_session_id, ai_transcript_path, metadata_json + FROM logs + WHERE hostname = ?1 + AND ai_tool = ?2 + AND ai_project = ?3 + AND ai_session_id = ?4 + AND (timestamp > ?5 OR (timestamp = ?5 AND id > ?6)) + ORDER BY timestamp ASC, id ASC + LIMIT ?7", + )?; + let after_rows = after_stmt + .query_map( + params![ + &entry.hostname, + tool, + project, + session_id, + &entry.timestamp, + entry.id, + after + ], + map_row, + )? + .collect::>>()?; + + Ok((before_rows, after_rows)) +} + +/// Get database stats +pub fn get_stats(pool: &DbPool, config: &StorageConfig) -> Result { + get_stats_with_options(pool, config, false) +} + +/// `get_stats`, but `include_fts_diagnostics` controls whether the +/// `phantom_fts_rows` field is computed. That value requires +/// `COUNT(*) FROM logs_fts` — an external-content FTS5 index scan that is +/// cheap on small DBs but expensive on very large ones (the index has no +/// O(1) row counter). The default `stats` path passes `false` so the common +/// query stays fast; callers that specifically need the FTS merge-health +/// diagnostic pass `true`. +pub fn get_stats_with_options( + pool: &DbPool, + config: &StorageConfig, + include_fts_diagnostics: bool, +) -> Result { + let metrics = get_storage_metrics(pool, config)?; + let write_blocked = exceeds_trigger(&metrics, config); + let mut conn = pool.get()?; + + // Deferred read transaction ensures the log stats form a consistent snapshot + let tx = conn.transaction_with_behavior(rusqlite::TransactionBehavior::Deferred)?; + // total_logs reads the timeline_hourly rollup (O(#buckets)) plus the live + // delta of rows ingested since the rollup watermark, instead of the O(#rows) + // `COUNT(*) FROM logs` (~7s on multi-million-row DBs). The rollup covers + // `logs.id <= source_max_id`; the delta covers `id > source_max_id`, so the + // sum is exact at the current snapshot for ADDs. It is NOT perfectly exact + // under concurrent retention DELETEs of rows the rollup already counted: the + // retention prune (spawn_retention_task) trims whole stale buckets, leaving + // at most a transient single-boundary-hour overcount — accepted as a + // negligible drift for a stats counter. (bead syslog-mcp-kcvq) + let rollup_max_id: i64 = tx.query_row( + "SELECT source_max_id FROM timeline_hourly_meta WHERE id = 1", + [], + |r| r.get(0), + )?; + let rollup_total: i64 = tx.query_row( + "SELECT COALESCE(SUM(event_count), 0) FROM timeline_hourly", + [], + |r| r.get(0), + )?; + let live_delta: i64 = tx.query_row( + "SELECT COUNT(*) FROM logs WHERE id > ?1", + [rollup_max_id], + |r| r.get(0), + )?; + let total_logs: i64 = rollup_total + live_delta; + let total_hosts: i64 = tx.query_row("SELECT COUNT(*) FROM hosts", [], |r| r.get(0))?; + let phantom_fts_rows = if include_fts_diagnostics { + let fts_rows: i64 = tx + .query_row("SELECT COUNT(*) FROM logs_fts", [], |r| r.get(0)) + .unwrap_or(0); + Some((fts_rows - total_logs).max(0)) + } else { + None + }; + // MIN/MAX return a single nullable row; use get::<_, Option<_>> so NULL becomes + // None while real query errors (e.g. missing table) still propagate via `?`. + // Both use the covering index idx_logs_timestamp (SEARCH, O(log n)). + let oldest: Option = tx.query_row("SELECT MIN(timestamp) FROM logs", [], |r| { + r.get::<_, Option>(0) + })?; + let newest: Option = tx.query_row("SELECT MAX(timestamp) FROM logs", [], |r| { + r.get::<_, Option>(0) + })?; + tx.finish()?; + + Ok(DbStats { + total_logs, + total_hosts, + oldest_log: oldest, + newest_log: newest, + logical_db_size_mb: format!("{:.2}", metrics.logical_db_size_bytes as f64 / 1_048_576.0), + physical_db_size_mb: format!("{:.2}", metrics.physical_db_size_bytes as f64 / 1_048_576.0), + free_disk_mb: metrics + .free_disk_bytes + .map(|bytes| format!("{:.2}", bytes as f64 / 1_048_576.0)), + max_db_size_mb: config.max_db_size_mb, + min_free_disk_mb: config.min_free_disk_mb, + write_blocked, + phantom_fts_rows, + }) +} + +/// Syslog severity level names ordered by numeric value (0=emerg, 7=debug). +/// Used by both the MCP layer (for threshold filtering) and the syslog parser (for decoding). +pub const SEVERITY_LEVELS: &[&str] = &[ + "emerg", "alert", "crit", "err", "warning", "notice", "info", "debug", +]; + +/// Convert a severity name to its numeric syslog level (0=emerg, 7=debug). +/// Accepts the canonical RFC 5424 keywords (case-insensitive) plus common +/// aliases: `error`/`fatal`/`panic` for `err`, `warn` for `warning`, +/// `critical` for `crit`, `emergency` for `emerg`. +/// Returns `None` for unrecognised names. +pub fn severity_to_num(s: &str) -> Option { + let canonical = match s.to_ascii_lowercase().as_str() { + "emergency" => "emerg", + "critical" => "crit", + "error" | "fatal" | "panic" => "err", + "warn" => "warning", + other => { + return SEVERITY_LEVELS + .iter() + .position(|&l| l == other) + .map(|i| i as u8); + } + }; + SEVERITY_LEVELS + .iter() + .position(|&l| l == canonical) + .map(|i| i as u8) +} + +fn append_filters( + sql: &mut String, + bindings: &mut Vec, + idx: &mut usize, + params: &SearchParams, +) { + if let Some(ref h) = params.host { + append_host_selector(sql, bindings, idx, "l.hostname", h); + } + if let Some(ref source_ip) = params.source { + sql.push_str(&format!(" AND l.source_ip = ?{}", *idx)); + bindings.push(rusqlite::types::Value::Text(source_ip.clone())); + *idx += 1; + } + if let Some(ref prefix) = params.source_ip_prefix + && !prefix.is_empty() + { + sql.push_str(&format!(" AND l.source_ip >= ?{}", *idx)); + bindings.push(rusqlite::types::Value::Text(prefix.clone())); + *idx += 1; + if let Some(upper) = prefix_upper_bound(prefix) { + sql.push_str(&format!(" AND l.source_ip < ?{}", *idx)); + bindings.push(rusqlite::types::Value::Text(upper)); + *idx += 1; + } + } + if let Some(ref prefixes) = params.source_ip_prefixes { + let prefixes = prefixes + .iter() + .filter(|prefix| !prefix.is_empty()) + .collect::>(); + if !prefixes.is_empty() { + sql.push_str(" AND ("); + for (position, prefix) in prefixes.iter().enumerate() { + if position > 0 { + sql.push_str(" OR "); + } + sql.push_str(&format!("(l.source_ip >= ?{}", *idx)); + bindings.push(rusqlite::types::Value::Text((*prefix).clone())); + *idx += 1; + if let Some(upper) = prefix_upper_bound(prefix) { + sql.push_str(&format!(" AND l.source_ip < ?{}", *idx)); + bindings.push(rusqlite::types::Value::Text(upper)); + *idx += 1; + } + sql.push(')'); + } + sql.push(')'); + } + } + if let Some(ref s) = params.severity { + sql.push_str(&format!(" AND l.severity = ?{}", *idx)); + bindings.push(rusqlite::types::Value::Text(s.clone())); + *idx += 1; + } + if let Some(ref levels) = params.severity_in + && !levels.is_empty() + { + let placeholders: Vec = levels + .iter() + .enumerate() + .map(|(i, _)| format!("?{}", *idx + i)) + .collect(); + sql.push_str(&format!(" AND l.severity IN ({})", placeholders.join(", "))); + for level in levels { + bindings.push(rusqlite::types::Value::Text(level.clone())); + *idx += 1; + } + } + if let Some(ref a) = params.app { + sql.push_str(&format!(" AND l.app_name = ?{}", *idx)); + bindings.push(rusqlite::types::Value::Text(a.clone())); + *idx += 1; + } + if let Some(ref f) = params.facility { + sql.push_str(&format!(" AND l.facility = ?{}", *idx)); + bindings.push(rusqlite::types::Value::Text(f.clone())); + *idx += 1; + } + if let Some(ref f) = params.exclude_facility { + sql.push_str(&format!( + " AND (l.facility IS NULL OR l.facility != ?{})", + *idx + )); + bindings.push(rusqlite::types::Value::Text(f.clone())); + *idx += 1; + } + if let Some(ref pid) = params.process_id { + sql.push_str(&format!(" AND l.process_id = ?{}", *idx)); + bindings.push(rusqlite::types::Value::Text(pid.clone())); + *idx += 1; + } + if let Some(ref from) = params.since { + sql.push_str(&format!(" AND l.timestamp >= ?{}", *idx)); + bindings.push(rusqlite::types::Value::Text(from.clone())); + *idx += 1; + } + if let Some(ref to) = params.until { + sql.push_str(&format!(" AND l.timestamp <= ?{}", *idx)); + bindings.push(rusqlite::types::Value::Text(to.clone())); + *idx += 1; + } + if let Some(ref from) = params.received_since { + sql.push_str(&format!(" AND l.received_at >= ?{}", *idx)); + bindings.push(rusqlite::types::Value::Text(from.clone())); + *idx += 1; + } + if let Some(ref to) = params.received_until { + sql.push_str(&format!(" AND l.received_at <= ?{}", *idx)); + bindings.push(rusqlite::types::Value::Text(to.clone())); + *idx += 1; + } + if let Some(ref tool) = params.ai_tool { + sql.push_str(&format!(" AND l.ai_tool = ?{}", *idx)); + bindings.push(rusqlite::types::Value::Text(tool.clone())); + *idx += 1; + } + if let Some(ref project) = params.ai_project { + sql.push_str(&format!(" AND l.ai_project = ?{}", *idx)); + bindings.push(rusqlite::types::Value::Text(project.clone())); + *idx += 1; + } + if let Some(ref session_id) = params.ai_session_id { + sql.push_str(&format!(" AND l.ai_session_id = ?{}", *idx)); + bindings.push(rusqlite::types::Value::Text(session_id.clone())); + *idx += 1; + } + if let Some(ref event_action) = params.event_action { + sql.push_str(&format!(" AND l.event_action = ?{}", *idx)); + bindings.push(rusqlite::types::Value::Text(event_action.clone())); + *idx += 1; + } + if params.exclude_ai { + sql.push_str( + " AND (l.ai_project IS NULL OR l.ai_project = '') + AND (l.ai_tool IS NULL OR l.ai_tool = '') + AND (l.ai_session_id IS NULL OR l.ai_session_id = '') + AND (l.ai_transcript_path IS NULL OR l.ai_transcript_path = '') + AND ( + l.app_name IS NULL + OR l.app_name NOT IN ( + 'ai-transcript', + 'claude-transcript', + 'codex-transcript', + 'gemini-transcript' + ) + )", + ); + } +} + +/// Append an indexed hostname selector that accepts the canonical names +/// returned by [`list_hosts`]. Normalization is confined to the tiny `hosts` +/// aggregate table; the outer log lookup still uses exact stored values and +/// can probe the existing hostname indexes. +fn append_host_selector( + sql: &mut String, + bindings: &mut Vec, + idx: &mut usize, + column: &str, + hostname: &str, +) { + let param = format!("?{}", *idx); + let normalized_param = format!("lower(rtrim(trim({param}), '.'))"); + let normalized_host = "lower(rtrim(trim(h.hostname), '.'))"; + let normalized_bare = "lower(rtrim(trim(bare.hostname), '.'))"; + sql.push_str(&format!( + " AND {column} IN ( + SELECT h.hostname + FROM hosts h + WHERE {normalized_host} = {normalized_param} + OR ( + {normalized_host} LIKE {normalized_param} || '.%' + AND EXISTS ( + SELECT 1 + FROM hosts bare + WHERE {normalized_bare} = {normalized_param} + AND instr({normalized_bare}, '.') = 0 + ) + ) + )" + )); + bindings.push(rusqlite::types::Value::Text(hostname.to_string())); + *idx += 1; +} + +fn prefix_upper_bound(prefix: &str) -> Option { + let mut bytes = prefix.as_bytes().to_vec(); + for idx in (0..bytes.len()).rev() { + if bytes[idx] != u8::MAX { + bytes[idx] += 1; + bytes.truncate(idx + 1); + return String::from_utf8(bytes).ok(); + } + } + None +} + +pub(super) fn map_row(row: &rusqlite::Row) -> rusqlite::Result { + map_row_offset(row, 0) +} + +fn map_row_offset(row: &rusqlite::Row, offset: usize) -> rusqlite::Result { + Ok(LogEntry { + id: row.get(offset)?, + timestamp: row.get(offset + 1)?, + hostname: row.get(offset + 2)?, + facility: row.get(offset + 3)?, + severity: row.get(offset + 4)?, + app_name: row.get(offset + 5)?, + process_id: row.get(offset + 6)?, + message: row.get(offset + 7)?, + received_at: row.get(offset + 8)?, + source_ip: row.get(offset + 9)?, + ai_tool: row.get(offset + 10)?, + ai_project: row.get(offset + 11)?, + ai_session_id: row.get(offset + 12)?, + ai_transcript_path: row.get(offset + 13)?, + metadata_json: row.get(offset + 14)?, + }) +} + +/// Map a row that includes the unparsed `raw` syslog frame (column index 8). +pub(super) fn map_row_with_raw( + row: &rusqlite::Row, +) -> rusqlite::Result { + Ok(super::analytics::LogEntryWithRaw { + id: row.get(0)?, + timestamp: row.get(1)?, + hostname: row.get(2)?, + facility: row.get(3)?, + severity: row.get(4)?, + app_name: row.get(5)?, + process_id: row.get(6)?, + message: row.get(7)?, + raw: row.get(8)?, + received_at: row.get(9)?, + source_ip: row.get(10)?, + ai_tool: row.get(11)?, + ai_project: row.get(12)?, + ai_session_id: row.get(13)?, + ai_transcript_path: row.get(14)?, + metadata_json: row.get(15)?, + }) +} + +// --------------------------------------------------------------------------- +// RAG v1: similar_incidents, incident_context +// --------------------------------------------------------------------------- + +use super::models::{ + AppLogCount, CorrelatedSession, IncidentCluster, IncidentContextParams, IncidentContextResult, + SeverityCount, SimilarIncidentsParams, SimilarIncidentsResult, +}; + +/// Return incident clusters from FTS5 hits, grouped by hostname + app_name in +/// non-overlapping windows of `window_minutes` minutes (default 30). +/// +/// Algorithm: +/// 1. FTS5 MATCH over non-AI log rows (optionally filtered by host/app/time). +/// 2. Group hits by (hostname, app_name, floor(unix_epoch / window_secs)). +/// 3. For each cluster: derive severity_peak (min numeric rank = highest sev), +/// collect up to 3 representative message snippets, and look up correlated +/// AI sessions whose transcript timestamps overlap the cluster window. +pub fn similar_incidents_clusters( + pool: &DbPool, + params: &SimilarIncidentsParams, +) -> Result { + validate_fts_query(¶ms.query)?; + + let conn = pool.get()?; + let window_minutes = params.window_minutes.unwrap_or(30).clamp(5, 120); + let limit = params.limit.unwrap_or(10).clamp(1, 50) as usize; + let window_secs = i64::from(window_minutes) * 60; + + // Build the FTS5 + optional filter query. + // Exclude AI transcript rows so clusters contain only system logs. + let mut sql = String::from( + "WITH hits AS MATERIALIZED ( + SELECT l.id, l.timestamp, l.hostname, l.app_name, l.severity, l.message + FROM logs_fts + JOIN logs l ON l.id = logs_fts.rowid + WHERE logs_fts MATCH ?1 + AND (l.ai_project IS NULL OR l.ai_project = '')", + ); + + let mut query_params = SqlParams::new(2); + query_params + .bindings + .push(rusqlite::types::Value::Text(params.query.clone())); + + if let Some(hostname) = ¶ms.host { + let idx = query_params.push_text(hostname.clone()); + sql.push_str(&format!(" AND l.hostname = ?{idx}")); + } + if let Some(app_name) = ¶ms.app { + let idx = query_params.push_text(app_name.clone()); + sql.push_str(&format!(" AND l.app_name = ?{idx}")); + } + if let Some(from) = ¶ms.since { + let idx = query_params.push_text(from.clone()); + sql.push_str(&format!(" AND l.timestamp >= ?{idx}")); + } + if let Some(to) = ¶ms.until { + let idx = query_params.push_text(to.clone()); + sql.push_str(&format!(" AND l.timestamp <= ?{idx}")); + } + // Apply severity_min filter: include only logs at or above the threshold. + if let Some(severity_min) = ¶ms.severity_min { + let threshold = severity_to_num(severity_min).ok_or_else(|| { + anyhow::anyhow!( + "invalid severity_min '{}': must be one of {}", + severity_min, + SEVERITY_LEVELS.join(", ") + ) + })?; + let levels_in: Vec = SEVERITY_LEVELS[..=threshold as usize] + .iter() + .map(|s| s.to_string()) + .collect(); + let placeholders: Vec = levels_in + .iter() + .map(|s| { + let idx = query_params.push_text(s.clone()); + format!("?{idx}") + }) + .collect(); + sql.push_str(&format!(" AND l.severity IN ({})", placeholders.join(", "))); + } + + sql.push_str(&format!( + " ORDER BY l.id DESC LIMIT {SIMILAR_INCIDENT_FTS_CANDIDATE_CAP} + ), + bucketed AS ( + SELECT + hostname, + app_name, + CAST(strftime('%s', timestamp) AS INTEGER) / {window_secs} AS bucket, + MIN(timestamp) AS window_start, + MAX(timestamp) AS window_end, + COUNT(*) AS log_count, + GROUP_CONCAT(severity, ',') AS severities, + GROUP_CONCAT(SUBSTR(message, 1, 256), '|||') AS messages + FROM hits + GROUP BY hostname, app_name, bucket + ) + SELECT hostname, app_name, window_start, window_end, log_count, severities, messages + FROM bucketed + ORDER BY log_count DESC, window_start DESC + LIMIT {}", + limit + 1 + )); + + let mut stmt = conn.prepare(&sql).map_err(|e| { + tracing::error!(error = %e, "similar_incidents_clusters prepare failed"); + anyhow::anyhow!("similar_incidents query failed") + })?; + let rows = stmt + .query_map( + rusqlite::params_from_iter(query_params.bindings.iter()), + |row| { + Ok(( + row.get::<_, String>(0)?, // hostname + row.get::<_, Option>(1)?, // app_name + row.get::<_, String>(2)?, // window_start + row.get::<_, String>(3)?, // window_end + row.get::<_, i64>(4)?, // log_count + row.get::<_, String>(5)?, // severities (comma-joined) + row.get::<_, String>(6)?, // messages (|||joined) + )) + }, + ) + .map_err(|e| { + tracing::error!(error = %e, "similar_incidents_clusters query failed"); + anyhow::anyhow!("similar_incidents query failed") + })?; + + // Collect raw cluster rows first; keep one extra to detect truncation. + struct RawCluster { + hostname: String, + app_name: Option, + window_start: String, + window_end: String, + log_count: i64, + severity_peak: String, + representative_messages: Vec, + } + let mut raw: Vec = Vec::new(); + for row in rows { + let (hostname, app_name, window_start, window_end, log_count, severities, messages) = + row.map_err(|e| { + tracing::error!(error = %e, "similar_incidents_clusters row mapping failed"); + anyhow::anyhow!("similar_incidents row mapping failed") + })?; + + // Find peak severity (lowest numeric value = highest severity). + let severity_peak = severities + .split(',') + .filter_map(|s| severity_to_num(s).map(|n| (n, s.to_string()))) + .min_by_key(|(n, _)| *n) + .map(|(_, s)| s) + .unwrap_or_else(|| "info".to_string()); + + // Collect up to 3 representative messages. + let representative_messages: Vec = messages + .split("|||") + .take(3) + .map(|m| m.to_string()) + .collect(); + + raw.push(RawCluster { + hostname, + app_name, + window_start, + window_end, + log_count, + severity_peak, + representative_messages, + }); + } + + // Detect truncation (we queried limit+1 rows) and trim to the true limit. + let truncated = raw.len() > limit; + raw.truncate(limit); + + // Build one UNION ALL query across all cluster windows so each window gets + // its own per-session match_count. This is O(1) roundtrips while keeping + // counts accurate (no global-span inflation when a session spans clusters). + let per_cluster_sessions = find_correlated_sessions_per_cluster( + &conn, + &raw.iter() + .map(|c| (c.window_start.as_str(), c.window_end.as_str())) + .collect::>(), + )?; + + let clusters: Vec = raw + .into_iter() + .map(|rc| { + let key = (rc.window_start.clone(), rc.window_end.clone()); + let correlated_sessions = per_cluster_sessions.get(&key).cloned().unwrap_or_default(); + IncidentCluster { + hostname: rc.hostname, + app_name: rc.app_name, + window_start: rc.window_start, + window_end: rc.window_end, + log_count: rc.log_count, + severity_peak: rc.severity_peak, + representative_messages: rc.representative_messages, + correlated_sessions, + } + }) + .collect(); + + let total_clusters = clusters.len(); + Ok(SimilarIncidentsResult { + query: params.query.clone(), + total_clusters, + truncated, + clusters, + }) +} + +/// Per-cluster session lookup using a single UNION ALL query so each cluster +/// window gets its own accurate match_count rather than an inflated global count. +/// Returns a map keyed by (window_start, window_end) → top-5 sessions. +fn find_correlated_sessions_per_cluster( + conn: &rusqlite::Connection, + windows: &[(&str, &str)], +) -> Result>> { + use std::collections::HashMap; + + if windows.is_empty() { + return Ok(HashMap::new()); + } + + // Build UNION ALL: one SELECT per cluster window, tagging each row with ws/we. + // Parameters use stride-2 (?{p} = ws, ?{p+1} = we for window i). SQLite ?N + // numbered bindings reuse the same value within one arm's subquery without + // requiring duplicate params in the binding list. + let mut arms: Vec = Vec::with_capacity(windows.len()); + for (i, _) in windows.iter().enumerate() { + let p = 1 + i * 2; + arms.push(format!( + "SELECT ?{p} AS ws, ?{p1} AS we, + l.ai_project, l.ai_tool, l.ai_session_id, + COUNT(*) AS match_count, + (SELECT l2.message FROM logs l2 + WHERE l2.ai_project = l.ai_project + AND l2.ai_tool = l.ai_tool + AND l2.ai_session_id = l.ai_session_id + AND l2.timestamp BETWEEN ?{p} AND ?{p1} + ORDER BY l2.timestamp DESC LIMIT 1) AS best_snippet + FROM logs l + WHERE l.ai_project IS NOT NULL AND l.ai_project != '' + AND l.ai_tool IS NOT NULL AND l.ai_tool != '' + AND l.ai_session_id IS NOT NULL AND l.ai_session_id != '' + AND l.timestamp BETWEEN ?{p} AND ?{p1} + GROUP BY l.ai_project, l.ai_tool, l.ai_session_id", + p = p, + p1 = p + 1, + )); + } + let sql = arms.join("\nUNION ALL\n"); + + let mut stmt = conn + .prepare(&sql) + .map_err(|e| anyhow::anyhow!("find_correlated_sessions_per_cluster prepare: {e}"))?; + + // Two params per window (ws, we); ?N reuse within each arm handles the rest. + let params: Vec<&dyn rusqlite::ToSql> = windows + .iter() + .flat_map(|(ws, we)| { + let v: [&dyn rusqlite::ToSql; 2] = [ws, we]; + v + }) + .collect(); + + let rows = stmt + .query_map(params.as_slice(), |row| { + let ws: String = row.get(0)?; + let we: String = row.get(1)?; + let project: String = row.get(2)?; + let tool: String = row.get(3)?; + let session_id: String = row.get(4)?; + let match_count: i64 = row.get(5)?; + let best_snippet: Option = row.get(6)?; + Ok((ws, we, project, tool, session_id, match_count, best_snippet)) + }) + .map_err(|e| anyhow::anyhow!("find_correlated_sessions_per_cluster query: {e}"))?; + + // Collect all sessions per cluster before sorting — UNION ALL rows arrive + // unordered, so the top-5 cap must come after sorting, not during insertion. + let mut map: HashMap<(String, String), Vec> = HashMap::new(); + for row in rows { + let (ws, we, project, tool, session_id, match_count, best_snippet) = + row.map_err(|e| anyhow::anyhow!("find_correlated_sessions_per_cluster row: {e}"))?; + map.entry((ws, we)).or_default().push(CorrelatedSession { + project, + tool, + session_id, + match_count, + best_snippet, + }); + } + // Sort by match_count descending, then cap at 5 per cluster. + for sessions in map.values_mut() { + sessions.sort_by_key(|b| std::cmp::Reverse(b.match_count)); + sessions.truncate(5); + } + Ok(map) +} + +/// Return aggregate log statistics + error logs + correlated AI sessions for a +/// given time window. +pub fn incident_context_summary( + pool: &DbPool, + params: &IncidentContextParams, +) -> Result { + let conn = pool.get()?; + let limit = params.limit.unwrap_or(50).clamp(1, 200) as usize; + if let Some(query) = params.query.as_deref() { + validate_fts_query(query)?; + } + + // Resolve severity threshold. Default to "warning" (numeric 4). + let severity_threshold = params + .severity_min + .as_deref() + .map(|s| { + severity_to_num(s).ok_or_else(|| { + anyhow::anyhow!( + "invalid severity_min '{}': must be one of emerg, alert, crit, err, warning, notice, info, debug", + s + ) + }) + }) + .transpose()? + .unwrap_or_else(|| severity_to_num("warning").unwrap()); + + // Build reusable aggregate params with host/app/AI-exclusion filters. + // Params: ?1=from, ?2=to, then optional host/app starting at ?3. + // All aggregate queries exclude AI transcript rows (ai_project IS NULL or ''). + let mut agg_params = SqlParams::new(3); + agg_params + .bindings + .push(rusqlite::types::Value::Text(params.since.clone())); + agg_params + .bindings + .push(rusqlite::types::Value::Text(params.until.clone())); + let mut agg_host_clause = String::new(); + let mut agg_app_clause = String::new(); + if let Some(hostname) = ¶ms.host { + let idx = agg_params.push_text(hostname.clone()); + agg_host_clause = format!(" AND hostname = ?{idx}"); + } + if let Some(app_name) = ¶ms.app { + let idx = agg_params.push_text(app_name.clone()); + agg_app_clause = format!(" AND app_name = ?{idx}"); + } + let agg_base_filter = format!( + "WHERE (ai_project IS NULL OR ai_project = '') + AND timestamp BETWEEN ?1 AND ?2{agg_host_clause}{agg_app_clause}" + ); + + // Total log count in window (system logs only, scoped by host/app). + let total_logs: i64 = conn + .query_row( + &format!("SELECT COUNT(*) FROM logs INDEXED BY idx_logs_timestamp {agg_base_filter}"), + rusqlite::params_from_iter(agg_params.bindings.iter()), + |r| r.get(0), + ) + .map_err(|e| anyhow::anyhow!("incident_context total_logs: {e}"))?; + + // Counts by severity (system logs only, scoped by host/app). + let mut by_sev_stmt = conn + .prepare(&format!( + "SELECT severity, COUNT(*) FROM logs INDEXED BY idx_logs_timestamp + {agg_base_filter} + GROUP BY severity + ORDER BY COUNT(*) DESC" + )) + .map_err(|e| anyhow::anyhow!("incident_context by_severity prepare: {e}"))?; + let by_severity: Vec = by_sev_stmt + .query_map( + rusqlite::params_from_iter(agg_params.bindings.iter()), + |row| { + Ok(SeverityCount { + severity: row.get(0)?, + count: row.get(1)?, + }) + }, + ) + .map_err(|e| anyhow::anyhow!("incident_context by_severity query: {e}"))? + .collect::>>()?; + + // Counts by app_name (top 20, system logs only, scoped by host/app). + let mut by_app_stmt = conn + .prepare(&format!( + "SELECT app_name, COUNT(*) FROM logs INDEXED BY idx_logs_timestamp + {agg_base_filter} + GROUP BY app_name + ORDER BY COUNT(*) DESC + LIMIT 20" + )) + .map_err(|e| anyhow::anyhow!("incident_context by_app prepare: {e}"))?; + let by_app: Vec = by_app_stmt + .query_map( + rusqlite::params_from_iter(agg_params.bindings.iter()), + |row| { + Ok(AppLogCount { + app_name: row.get(0)?, + count: row.get(1)?, + }) + }, + ) + .map_err(|e| anyhow::anyhow!("incident_context by_app query: {e}"))? + .collect::>>()?; + + // Error logs: system logs at or above severity threshold in the window. + let error_severities: Vec = SEVERITY_LEVELS[..=severity_threshold as usize] + .iter() + .map(|s| s.to_string()) + .collect(); + + // Build parameterized query for error logs. + // Params: ?1=from, ?2=to, ?3..=?N=severities, then optional host/app. + // SqlParams::new(3) sets next_idx=3 so push_text calls start at ?3, after + // the two manually-pushed bindings for from (?1) and to (?2). + let mut err_params = SqlParams::new(3); + err_params + .bindings + .push(rusqlite::types::Value::Text(params.since.clone())); + err_params + .bindings + .push(rusqlite::types::Value::Text(params.until.clone())); + + let query_idx = params + .query + .as_ref() + .map(|query| err_params.push_text(query.clone())); + + let sev_placeholders: Vec = error_severities + .iter() + .map(|s| { + let idx = err_params.push_text(s.clone()); + format!("?{idx}") + }) + .collect(); + + let mut err_sql = format!("SELECT {FTS_SELECT_COLS} "); + match query_idx { + Some(idx) => err_sql.push_str(&format!( + "FROM logs_fts + JOIN logs l ON l.id = logs_fts.rowid + WHERE logs_fts MATCH ?{idx} + AND l.timestamp BETWEEN ?1 AND ?2" + )), + None => err_sql.push_str( + "FROM logs l INDEXED BY idx_logs_timestamp + WHERE l.timestamp BETWEEN ?1 AND ?2", + ), + } + err_sql.push_str(&format!( + " AND l.severity IN ({}) + AND (l.ai_project IS NULL OR l.ai_project = '')", + sev_placeholders.join(", ") + )); + + if let Some(hostname) = ¶ms.host { + let idx = err_params.push_text(hostname.clone()); + err_sql.push_str(&format!(" AND l.hostname = ?{idx}")); + } + if let Some(app_name) = ¶ms.app { + let idx = err_params.push_text(app_name.clone()); + err_sql.push_str(&format!(" AND l.app_name = ?{idx}")); + } + // Query limit+1 rows so we can detect true truncation. + err_sql.push_str(&format!(" ORDER BY l.timestamp DESC LIMIT {}", limit + 1)); + + let mut err_stmt = conn.prepare(&err_sql).map_err(|e| { + tracing::error!(error = %e, "incident_context error_logs prepare failed"); + anyhow::anyhow!("incident_context error_logs query failed") + })?; + let error_rows = err_stmt + .query_map( + rusqlite::params_from_iter(err_params.bindings.iter()), + map_row, + ) + .map_err(|e| { + tracing::error!(error = %e, "incident_context error_logs query failed"); + anyhow::anyhow!("incident_context error_logs query failed") + })?; + let mut error_logs: Vec = + error_rows.collect::>>()?; + let error_logs_truncated = error_logs.len() > limit; + error_logs.truncate(limit); + + // AI sessions active in the window — query on the already-held conn to + // avoid a second pool.get() call (which deadlocks on single-connection test pools). + let ai_sessions = { + let mut ai_sql = String::from( + "SELECT ai_project, ai_tool, ai_session_id, + MIN(ai_transcript_path) AS ai_transcript_path, + hostname, + MIN(timestamp) AS first_seen, + MAX(timestamp) AS last_seen, + COUNT(*) AS event_count + FROM logs + WHERE ai_project IS NOT NULL AND ai_project != '' + AND ai_tool IS NOT NULL AND ai_tool != '' + AND ai_session_id IS NOT NULL AND ai_session_id != '' + AND timestamp BETWEEN ?1 AND ?2", + ); + let mut ai_bindings: Vec = vec![ + rusqlite::types::Value::Text(params.since.clone()), + rusqlite::types::Value::Text(params.until.clone()), + ]; + if let Some(hostname) = ¶ms.host { + ai_bindings.push(rusqlite::types::Value::Text(hostname.clone())); + ai_sql.push_str(&format!(" AND hostname = ?{}", ai_bindings.len())); + } + ai_sql.push_str( + " GROUP BY ai_project, ai_tool, ai_session_id, hostname + ORDER BY last_seen DESC LIMIT 20", + ); + let mut ai_stmt = conn + .prepare(&ai_sql) + .map_err(|e| anyhow::anyhow!("incident_context ai_sessions prepare: {e}"))?; + let rows = ai_stmt + .query_map(rusqlite::params_from_iter(ai_bindings.iter()), |row| { + Ok(super::models::AiSessionEntry { + ai_project: row.get(0)?, + ai_tool: row.get(1)?, + ai_session_id: row.get(2)?, + ai_transcript_path: row.get(3)?, + hostname: row.get(4)?, + first_seen: row.get(5)?, + last_seen: row.get(6)?, + event_count: row.get(7)?, + title: None, + title_provenance: None, + }) + }) + .map_err(|e| anyhow::anyhow!("incident_context ai_sessions query: {e}"))?; + rows.collect::>>()? + }; + + Ok(IncidentContextResult { + window_from: params.since.clone(), + window_to: params.until.clone(), + total_logs, + by_severity, + by_app, + error_logs, + error_logs_truncated, + ai_sessions, + }) +} + +#[cfg(test)] +#[path = "queries_tests.rs"] +mod tests; + +#[cfg(test)] +#[path = "queries_graph_tests.rs"] +mod graph_tests; From 2ac8d71a244804836c53f725c460d6b76424e78e Mon Sep 17 00:00:00 2001 From: jmagar <38927646+jmagar@users.noreply.github.com> Date: Wed, 23 Sep 2026 13:17:36 -0400 Subject: [PATCH 4/8] refactor: split session investigation support --- src/app/services.rs | 1 + src/app/services/session_investigation.rs | 261 +++--------------- .../services/session_investigation_support.rs | 247 +++++++++++++++++ .../services/session_investigation_tests.rs | 2 + 4 files changed, 282 insertions(+), 229 deletions(-) create mode 100644 src/app/services/session_investigation_support.rs diff --git a/src/app/services.rs b/src/app/services.rs index b41422923..522499ab6 100644 --- a/src/app/services.rs +++ b/src/app/services.rs @@ -123,6 +123,7 @@ mod mcp_events; mod mcp_incidents; mod rag; mod session_investigation; +mod session_investigation_support; mod session_pages; mod skill_assessment; mod skill_backfill; diff --git a/src/app/services/session_investigation.rs b/src/app/services/session_investigation.rs index 9893a72dd..fa3d339fc 100644 --- a/src/app/services/session_investigation.rs +++ b/src/app/services/session_investigation.rs @@ -2,6 +2,13 @@ use std::time::Instant; use super::*; +use super::session_investigation_support::{ + build_session_source_counts, extract_session_external_references, + related_sessions_for_investigation, session_investigation_metadata, + summarize_session_source_evidence, timestamp_inclusive_between, +}; + + impl CortexService { pub async fn session_investigate( &self, @@ -352,95 +359,37 @@ impl CortexService { }) .await?; - let related_sessions = self - .list_sessions(ListSessionsRequest { - project: Some(session.project.clone()), - tool: None, - session_id: None, - host: None, - since: Some(session.first_seen.clone()), - until: Some(session.last_seen.clone()), - limit: Some(50), - }) - .await? - .sessions - .into_iter() - .filter(|candidate| candidate.session_key != session.session_key) - .take(20) - .collect::>(); + let related_sessions = related_sessions_for_investigation(self, &session).await?; let external_references = extract_session_external_references(&transcript_page.events); let source_evidence = summarize_session_source_evidence( correlated.graph_correlation.as_ref(), section_limit as usize, ); - let mut source_counts = source_evidence - .iter() - .map(|(kind, summary)| (kind.clone(), summary.count)) - .collect::>(); - for event in &observatory.events { - *source_counts - .entry(format!("observatory:{}", event.source_kind)) - .or_default() += 1; - } - if !skill_events.events.is_empty() { - source_counts.insert("skill_event".to_string(), skill_events.events.len()); - } - if !mcp_events.events.is_empty() { - source_counts.insert("mcp_event".to_string(), mcp_events.events.len()); - } - if !hook_events.events.is_empty() { - source_counts.insert("hook_event".to_string(), hook_events.events.len()); - } - if !artifact_evidence.is_empty() { - source_counts.insert("artifact_evidence".to_string(), artifact_evidence.len()); - } - if !observatory.spans.is_empty() { - source_counts.insert("otlp_span".to_string(), observatory.spans.len()); - } - if !observatory.metrics.is_empty() { - source_counts.insert("otlp_metric".to_string(), observatory.metrics.len()); - } - if let Some(graph) = graph_neighborhood.as_ref() { - source_counts.insert("graph_entity".to_string(), graph.entities.len()); - source_counts.insert("graph_relationship".to_string(), graph.relationships.len()); - source_counts.insert("graph_evidence".to_string(), graph.evidence.len()); - } - if let Some(correlation) = correlated.graph_correlation.as_ref() - && !correlation.heartbeat_summaries.is_empty() - { - source_counts.insert( - "heartbeat_summary".to_string(), - correlation.heartbeat_summaries.len(), - ); - } - if !incident_context.error_logs.is_empty() { - source_counts.insert( - "incident_error_log".to_string(), - incident_context.error_logs.len(), - ); - } - if !notifications.is_empty() { - source_counts.insert("notification".to_string(), notifications.len()); - } - if !observatory.actors.is_empty() { - source_counts.insert("observatory_actor".to_string(), observatory.actors.len()); - } - if observatory.parent_run.is_some() { - source_counts.insert("parent_run".to_string(), 1); - } - if observatory.previous_run.is_some() { - source_counts.insert("previous_run".to_string(), 1); - } - if !observatory.related_runs.is_empty() { - source_counts.insert( - "same_worktree_run".to_string(), - observatory.related_runs.len(), - ); - } - if !retention_lineage.is_empty() { - source_counts.insert("retention_lineage".to_string(), retention_lineage.len()); - } + let graph_counts = graph_neighborhood.as_ref().map(|graph| { + ( + graph.entities.len(), + graph.relationships.len(), + graph.evidence.len(), + ) + }); + let heartbeat_summary_count = correlated + .graph_correlation + .as_ref() + .map_or(0, |value| value.heartbeat_summaries.len()); + let source_counts = build_session_source_counts( + &source_evidence, + &observatory, + skill_events.events.len(), + mcp_events.events.len(), + hook_events.events.len(), + artifact_evidence.len(), + graph_counts, + heartbeat_summary_count, + incident_context.error_logs.len(), + notifications.len(), + retention_lineage.len(), + ); let mut partial_reasons = Vec::new(); if transcript_page.has_more { @@ -537,152 +486,6 @@ impl CortexService { } } -fn session_investigation_metadata( - budget: &InvestigationBudget, - started: Instant, - correlation: Option<&GraphSessionCorrelation>, - has_graph_neighborhood: bool, - partial_reasons: &[String], - transcript_rows: usize, - evidence_rows: usize, -) -> InvestigationMetadata { - let graph_calls = u32::from(correlation.is_some()) + u32::from(has_graph_neighborhood); - let log_rows = correlation.map_or(0, |value| value.logs.len() as u32); - let partial = !partial_reasons.is_empty(); - InvestigationMetadata { - server_version: env!("CARGO_PKG_VERSION").to_string(), - schema_version: INVESTIGATION_UI_VERSION.to_string(), - graph_projection_status: correlation - .map(|value| if value.used_graph { "used" } else { "fallback" }.to_string()), - source_watermark: None, - degraded_reasons: correlation - .filter(|value| !value.used_graph) - .map(|_| vec!["session_graph_entity_unavailable".to_string()]) - .unwrap_or_default(), - truncated: partial, - truncation_reasons: partial_reasons.to_vec(), - partial, - partial_reasons: partial_reasons.to_vec(), - auth_state: "bearer".to_string(), - budget: budget.clone(), - budget_used: InvestigationBudgetUsed { - graph_calls, - log_rows, - evidence_rows: evidence_rows.min(u32::MAX as usize) as u32, - candidate_explanations: 0, - wall_time_ms: started.elapsed().as_millis().min(u32::MAX as u128) as u32, - payload_bytes: 0, - }, - payload_limit_bytes: budget.max_payload_bytes, - version_skew: (transcript_rows > 200).then(|| "transcript_limit_exceeded".to_string()), - } -} - -fn summarize_session_source_evidence( - correlation: Option<&GraphSessionCorrelation>, - requested_limit: usize, -) -> BTreeMap { - let id_limit = requested_limit.clamp(1, 50); - let mut summaries = BTreeMap::::new(); - let Some(correlation) = correlation else { - return summaries; - }; - - for row in &correlation.logs { - let kind = row - .source_kind - .clone() - .unwrap_or_else(|| "unknown".to_string()); - let summary = summaries.entry(kind).or_default(); - summary.count += 1; - if summary.log_ids.len() < id_limit { - summary.log_ids.push(row.entry.id); - } else { - summary.truncated = true; - } - } - - summaries -} - -fn timestamp_inclusive_between(value: &str, start: &str, end: &str) -> bool { - let Ok(value) = chrono::DateTime::parse_from_rfc3339(value) else { - return false; - }; - let Ok(start) = chrono::DateTime::parse_from_rfc3339(start) else { - return false; - }; - let Ok(end) = chrono::DateTime::parse_from_rfc3339(end) else { - return false; - }; - value >= start && value <= end -} - -fn extract_session_external_references( - events: &[models::RenderedSessionEvent], -) -> Vec { - let mut references = std::collections::BTreeSet::new(); - for event in events { - for raw in event.text.split_whitespace() { - let token = raw.trim_matches(|ch: char| { - matches!( - ch, - ',' | '.' | ';' | ':' | ')' | '(' | ']' | '[' | '}' | '{' | '"' | '\'' - ) - }); - if token.is_empty() { - continue; - } - let kind = if token.starts_with("http://") || token.starts_with("https://") { - if token.contains("github.com/") && token.contains("/pull/") { - models::SessionExternalReferenceKind::GithubPullRequest - } else if token.contains("github.com/") && token.contains("/issues/") { - models::SessionExternalReferenceKind::GithubIssue - } else { - models::SessionExternalReferenceKind::Url - } - } else if looks_like_linear_identifier(token) { - models::SessionExternalReferenceKind::LinearIssue - } else if token.strip_prefix('#').is_some_and(|number| { - !number.is_empty() && number.chars().all(|ch| ch.is_ascii_digit()) - }) { - models::SessionExternalReferenceKind::GithubReference - } else if looks_like_commit_sha(token) { - models::SessionExternalReferenceKind::CommitSha - } else { - continue; - }; - references.insert(models::SessionExternalReference { - kind, - value: safe_passive_text(token, 500), - source_position: Some(event.position), - evidence_kind: "transcript_text".to_string(), - trust_level: "claimed".to_string(), - verified: false, - }); - } - } - references.into_iter().collect() -} - -fn looks_like_linear_identifier(token: &str) -> bool { - let Some((prefix, suffix)) = token.split_once('-') else { - return false; - }; - (2..=12).contains(&prefix.len()) - && prefix - .chars() - .all(|ch| ch.is_ascii_uppercase() || ch.is_ascii_digit()) - && !suffix.is_empty() - && suffix.chars().all(|ch| ch.is_ascii_digit()) -} - -fn looks_like_commit_sha(token: &str) -> bool { - (7..=40).contains(&token.len()) - && token.chars().all(|ch| ch.is_ascii_hexdigit()) - && token.chars().any(|ch| ch.is_ascii_alphabetic()) -} - #[cfg(test)] #[path = "session_investigation_tests.rs"] mod tests; diff --git a/src/app/services/session_investigation_support.rs b/src/app/services/session_investigation_support.rs new file mode 100644 index 000000000..2c8001fbc --- /dev/null +++ b/src/app/services/session_investigation_support.rs @@ -0,0 +1,247 @@ +use std::time::Instant; + +use super::*; + +pub(super) async fn related_sessions_for_investigation( + service: &CortexService, + session: &AiSessionEntry, +) -> ServiceResult> { + Ok(service + .list_sessions(ListSessionsRequest { + project: Some(session.project.clone()), + tool: None, + session_id: None, + host: None, + since: Some(session.first_seen.clone()), + until: Some(session.last_seen.clone()), + limit: Some(50), + }) + .await? + .sessions + .into_iter() + .filter(|candidate| candidate.session_key != session.session_key) + .take(20) + .collect()) +} + +#[allow(clippy::too_many_arguments)] +pub(super) fn build_session_source_counts( + source_evidence: &std::collections::BTreeMap, + observatory: &models::SessionObservatoryEvidence, + skill_event_count: usize, + mcp_event_count: usize, + hook_event_count: usize, + artifact_evidence_count: usize, + graph_counts: Option<(usize, usize, usize)>, + heartbeat_summary_count: usize, + incident_error_log_count: usize, + notification_count: usize, + retention_lineage_count: usize, +) -> std::collections::BTreeMap { + let mut source_counts = source_evidence + .iter() + .map(|(kind, summary)| (kind.clone(), summary.count)) + .collect::>(); + for event in &observatory.events { + *source_counts + .entry(format!("observatory:{}", event.source_kind)) + .or_default() += 1; + } + if skill_event_count > 0 { + source_counts.insert("skill_event".to_string(), skill_event_count); + } + if mcp_event_count > 0 { + source_counts.insert("mcp_event".to_string(), mcp_event_count); + } + if hook_event_count > 0 { + source_counts.insert("hook_event".to_string(), hook_event_count); + } + if artifact_evidence_count > 0 { + source_counts.insert("artifact_evidence".to_string(), artifact_evidence_count); + } + if !observatory.spans.is_empty() { + source_counts.insert("otlp_span".to_string(), observatory.spans.len()); + } + if !observatory.metrics.is_empty() { + source_counts.insert("otlp_metric".to_string(), observatory.metrics.len()); + } + if let Some((entities, relationships, evidence)) = graph_counts { + source_counts.insert("graph_entity".to_string(), entities); + source_counts.insert("graph_relationship".to_string(), relationships); + source_counts.insert("graph_evidence".to_string(), evidence); + } + if heartbeat_summary_count > 0 { + source_counts.insert("heartbeat_summary".to_string(), heartbeat_summary_count); + } + if incident_error_log_count > 0 { + source_counts.insert("incident_error_log".to_string(), incident_error_log_count); + } + if notification_count > 0 { + source_counts.insert("notification".to_string(), notification_count); + } + if !observatory.actors.is_empty() { + source_counts.insert("observatory_actor".to_string(), observatory.actors.len()); + } + if observatory.parent_run.is_some() { + source_counts.insert("parent_run".to_string(), 1); + } + if observatory.previous_run.is_some() { + source_counts.insert("previous_run".to_string(), 1); + } + if !observatory.related_runs.is_empty() { + source_counts.insert( + "same_worktree_run".to_string(), + observatory.related_runs.len(), + ); + } + if retention_lineage_count > 0 { + source_counts.insert("retention_lineage".to_string(), retention_lineage_count); + } + source_counts +} + +pub(super) fn session_investigation_metadata( + budget: &InvestigationBudget, + started: Instant, + correlation: Option<&GraphSessionCorrelation>, + has_graph_neighborhood: bool, + partial_reasons: &[String], + transcript_rows: usize, + evidence_rows: usize, +) -> InvestigationMetadata { + let graph_calls = u32::from(correlation.is_some()) + u32::from(has_graph_neighborhood); + let log_rows = correlation.map_or(0, |value| value.logs.len() as u32); + let partial = !partial_reasons.is_empty(); + InvestigationMetadata { + server_version: env!("CARGO_PKG_VERSION").to_string(), + schema_version: INVESTIGATION_UI_VERSION.to_string(), + graph_projection_status: correlation + .map(|value| if value.used_graph { "used" } else { "fallback" }.to_string()), + source_watermark: None, + degraded_reasons: correlation + .filter(|value| !value.used_graph) + .map(|_| vec!["session_graph_entity_unavailable".to_string()]) + .unwrap_or_default(), + truncated: partial, + truncation_reasons: partial_reasons.to_vec(), + partial, + partial_reasons: partial_reasons.to_vec(), + auth_state: "bearer".to_string(), + budget: budget.clone(), + budget_used: InvestigationBudgetUsed { + graph_calls, + log_rows, + evidence_rows: evidence_rows.min(u32::MAX as usize) as u32, + candidate_explanations: 0, + wall_time_ms: started.elapsed().as_millis().min(u32::MAX as u128) as u32, + payload_bytes: 0, + }, + payload_limit_bytes: budget.max_payload_bytes, + version_skew: (transcript_rows > 200).then(|| "transcript_limit_exceeded".to_string()), + } +} + +pub(super) fn summarize_session_source_evidence( + correlation: Option<&GraphSessionCorrelation>, + requested_limit: usize, +) -> BTreeMap { + let id_limit = requested_limit.clamp(1, 50); + let mut summaries = BTreeMap::::new(); + let Some(correlation) = correlation else { + return summaries; + }; + + for row in &correlation.logs { + let kind = row + .source_kind + .clone() + .unwrap_or_else(|| "unknown".to_string()); + let summary = summaries.entry(kind).or_default(); + summary.count += 1; + if summary.log_ids.len() < id_limit { + summary.log_ids.push(row.entry.id); + } else { + summary.truncated = true; + } + } + + summaries +} + +pub(super) fn timestamp_inclusive_between(value: &str, start: &str, end: &str) -> bool { + let Ok(value) = chrono::DateTime::parse_from_rfc3339(value) else { + return false; + }; + let Ok(start) = chrono::DateTime::parse_from_rfc3339(start) else { + return false; + }; + let Ok(end) = chrono::DateTime::parse_from_rfc3339(end) else { + return false; + }; + value >= start && value <= end +} + +pub(super) fn extract_session_external_references( + events: &[models::RenderedSessionEvent], +) -> Vec { + let mut references = std::collections::BTreeSet::new(); + for event in events { + for raw in event.text.split_whitespace() { + let token = raw.trim_matches(|ch: char| { + matches!( + ch, + ',' | '.' | ';' | ':' | ')' | '(' | ']' | '[' | '}' | '{' | '"' | '\'' + ) + }); + if token.is_empty() { + continue; + } + let kind = if token.starts_with("http://") || token.starts_with("https://") { + if token.contains("github.com/") && token.contains("/pull/") { + models::SessionExternalReferenceKind::GithubPullRequest + } else if token.contains("github.com/") && token.contains("/issues/") { + models::SessionExternalReferenceKind::GithubIssue + } else { + models::SessionExternalReferenceKind::Url + } + } else if looks_like_linear_identifier(token) { + models::SessionExternalReferenceKind::LinearIssue + } else if token.strip_prefix('#').is_some_and(|number| { + !number.is_empty() && number.chars().all(|ch| ch.is_ascii_digit()) + }) { + models::SessionExternalReferenceKind::GithubReference + } else if looks_like_commit_sha(token) { + models::SessionExternalReferenceKind::CommitSha + } else { + continue; + }; + references.insert(models::SessionExternalReference { + kind, + value: safe_passive_text(token, 500), + source_position: Some(event.position), + evidence_kind: "transcript_text".to_string(), + trust_level: "claimed".to_string(), + verified: false, + }); + } + } + references.into_iter().collect() +} + +pub(super) fn looks_like_linear_identifier(token: &str) -> bool { + let Some((prefix, suffix)) = token.split_once('-') else { + return false; + }; + (2..=12).contains(&prefix.len()) + && prefix + .chars() + .all(|ch| ch.is_ascii_uppercase() || ch.is_ascii_digit()) + && !suffix.is_empty() + && suffix.chars().all(|ch| ch.is_ascii_digit()) +} + +pub(super) fn looks_like_commit_sha(token: &str) -> bool { + (7..=40).contains(&token.len()) + && token.chars().all(|ch| ch.is_ascii_hexdigit()) + && token.chars().any(|ch| ch.is_ascii_alphabetic()) +} diff --git a/src/app/services/session_investigation_tests.rs b/src/app/services/session_investigation_tests.rs index 1f278915a..d614f57d7 100644 --- a/src/app/services/session_investigation_tests.rs +++ b/src/app/services/session_investigation_tests.rs @@ -1,4 +1,6 @@ use super::*; +use super::super::session_investigation_support::looks_like_linear_identifier; + fn event(position: i64, text: &str) -> models::RenderedSessionEvent { models::RenderedSessionEvent { From 2774adfa3489780a42931ef5499ae2329f239a84 Mon Sep 17 00:00:00 2001 From: jmagar <38927646+jmagar@users.noreply.github.com> Date: Wed, 23 Sep 2026 13:21:06 -0400 Subject: [PATCH 5/8] fix: restore truncated query tests --- src/db/queries_tests.rs | 1084 ++++++++++++++++++++++++++++++++++++++- 1 file changed, 1083 insertions(+), 1 deletion(-) diff --git a/src/db/queries_tests.rs b/src/db/queries_tests.rs index 7ae474979..731927987 100644 --- a/src/db/queries_tests.rs +++ b/src/db/queries_tests.rs @@ -1206,6 +1206,90 @@ fn search_ai_sessions_query_plan_uses_session_host_time_index() { ); } +#[test] +fn search_ai_sessions_pushes_tool_and_time_scope_into_fts_candidates() { + let (pool, _dir) = test_pool(); + insert_logs_batch( + &pool, + &[ + make_ai_entry( + "2026-09-10T12:00:00Z", + "host-a", + "claude", + "/tmp/project", + "old-claude", + "error before the requested window", + ), + make_ai_entry( + "2026-09-12T12:00:00Z", + "host-a", + "claude", + "/tmp/project", + "recent-claude", + "error inside the requested window", + ), + make_ai_entry( + "2026-09-12T12:01:00Z", + "host-a", + "codex", + "/tmp/project", + "recent-codex", + "error from a different provider", + ), + ], + ) + .unwrap(); + + let params = SearchAiSessionsParams { + query: "error".into(), + ai_tool: Some("claude".into()), + since: Some("2026-09-11T00:00:00Z".into()), + limit: Some(5), + ..Default::default() + }; + let result = search_ai_sessions(&pool, ¶ms).unwrap(); + assert_eq!(result.sessions.len(), 1); + assert_eq!(result.sessions[0].ai_session_id, "recent-claude"); + + let (sql, bindings) = search_ai_sessions_sql(¶ms, 5); + assert!( + sql.contains("INDEXED BY idx_logs_ai_timestamp_tool"), + "time-bounded session search must derive a safe FTS rowid floor from the AI-only time index:\n{sql}" + ); + assert!( + sql.contains("ai_logs_fts.rowid >="), + "time-bounded session search must push its safe rowid floor into FTS:\n{sql}" + ); + + let Some(rusqlite::types::Value::Text(fts_query)) = bindings.first() else { + panic!("first search_sessions binding must be the FTS query"); + }; + assert!( + fts_query.contains("message : (error)") && fts_query.contains("ai_tool : \"claude\""), + "tool scope must be pushed into the transcript FTS query itself; got {fts_query:?}" + ); +} + +#[test] +fn search_ai_sessions_does_not_interpolate_unsafe_tool_scope_into_fts() { + let params = SearchAiSessionsParams { + query: "needle".into(), + ai_tool: Some("claude\" OR message:error".into()), + limit: Some(5), + ..Default::default() + }; + + let (sql, bindings) = search_ai_sessions_sql(¶ms, 5); + let Some(rusqlite::types::Value::Text(fts_query)) = bindings.first() else { + panic!("first search_sessions binding must be the FTS query"); + }; + assert_eq!(fts_query, "message : (needle)"); + assert!( + sql.contains("l.ai_tool = ?"), + "unsafe provider values must fall back to the parameterized relational equality filter" + ); +} + #[test] fn search_ai_sessions_finds_rows_inserted_after_rollup_refresh() { let (pool, _dir) = test_pool(); @@ -2579,4 +2663,1002 @@ fn rollup_empty_when_only_broad_project_rows_present_succeeds() { // The watermark sees these rows (broad predicate) so src_count > 0; the // rollup predicate excludes them all so staging is legitimately empty. - // Pre-fix this raised the R1 error; post-fix it must SUCCEED with 0 rows. \ No newline at end of file + // Pre-fix this raised the R1 error; post-fix it must SUCCEED with 0 rows. + let total = refresh_ai_session_rollup(&pool).unwrap(); + assert_eq!( + total, 0, + "rollup must be legitimately empty (no rollup-eligible rows)" + ); + + // The meta/fingerprint MUST still be stamped on an empty rollup so + // refresh_ai_session_rollup_if_stale can skip subsequent no-op refreshes. + let status = ai_session_rollup_status(&pool).unwrap(); + assert_eq!(status.row_count, 0, "rollup row_count must be 0"); + assert!( + status.refreshed_at.is_some(), + "meta/fingerprint must be stamped even for an empty rollup" + ); +} + +#[test] +fn rollup_read_uses_last_seen_index_no_temp_btree() { + let (pool, _dir) = test_pool(); + seed_ai_sessions(&pool); + refresh_ai_session_rollup(&pool).unwrap(); + // The unbounded rollup read must be served by the last_seen index, NOT a + // temp b-tree sort (the cost that plagued the live aggregation). The query + // mirrors list_ai_sessions_from_rollup. + let plan = query_plan( + &pool, + "SELECT ai_project, ai_tool, ai_session_id, ai_transcript_path, + hostname, first_seen, last_seen, event_count + FROM ai_session_rollup + WHERE 1=1 + ORDER BY last_seen DESC LIMIT 100", + &[], + ); + assert!( + plan.contains("idx_ai_session_rollup_last_seen"), + "rollup read must use the last_seen index; plan was:\n{plan}" + ); + assert!( + !plan.contains("TEMP B-TREE"), + "rollup read must avoid a temp b-tree sort; plan was:\n{plan}" + ); +} + +#[test] +fn rollup_respects_project_and_tool_filters() { + let (pool, _dir) = test_pool(); + seed_ai_sessions(&pool); + refresh_ai_session_rollup(&pool).unwrap(); + + for (project, tool) in [ + (Some("/proj/0".to_string()), None), + (None, Some("codex".to_string())), + (Some("/proj/1".to_string()), Some("claude".to_string())), + ] { + let params = ListAiSessionsParams { + ai_project: project.clone(), + ai_tool: tool.clone(), + ..default_session_params() + }; + let live = list_ai_sessions_live(&pool, ¶ms).unwrap(); + let rolled = list_ai_sessions(&pool, ¶ms).unwrap(); + assert_eq!( + live.iter().map(|s| &s.ai_session_id).collect::>(), + rolled.iter().map(|s| &s.ai_session_id).collect::>(), + "filtered rollup ({project:?},{tool:?}) must match live" + ); + } +} + +#[test] +fn time_windowed_sessions_always_use_live_path() { + let (pool, _dir) = test_pool(); + seed_ai_sessions(&pool); + refresh_ai_session_rollup(&pool).unwrap(); + + // Insert a brand-new event AFTER the rollup was built. A time-windowed + // query must see it live (rollup is stale and must be bypassed). + insert_logs_batch( + &pool, + &[make_ai_entry( + "2026-06-01T12:00:00Z", + "host0", + "codex", + "/proj/0", + "sess-0", + "fresh post-refresh event", + )], + ) + .unwrap(); + + let windowed = ListAiSessionsParams { + since: Some("2026-06-01T00:00:00Z".into()), + until: Some("2026-06-02T00:00:00Z".into()), + ..default_session_params() + }; + let rows = list_ai_sessions(&pool, &windowed).unwrap(); + assert_eq!( + rows.len(), + 1, + "windowed query must see the fresh event live" + ); + assert_eq!(rows[0].last_seen, "2026-06-01T12:00:00Z"); + assert_eq!( + rows[0].event_count, 1, + "windowed count must be live, not rollup" + ); +} + +// --------------------------------------------------------------------------- +// Rollup source-watermark dirty-check (bead cortex-g33v) +// --------------------------------------------------------------------------- + +/// Helper: insert a single non-AI log row (no ai_* fields). Such rows must NOT +/// move the AI watermark, so a refresh that follows them is skipped. +fn insert_plain_log(pool: &DbPool, ts: &str, msg: &str) { + insert_logs_batch(pool, &[make_entry(ts, "host0", "info", msg)]).unwrap(); +} + +#[test] +fn stale_check_first_refresh_runs_then_noop_is_skipped() { + let (pool, _dir) = test_pool(); + seed_ai_sessions(&pool); + + // Never refreshed yet => must refresh. + match refresh_ai_session_rollup_if_stale(&pool).unwrap() { + RollupRefresh::Refreshed { row_count } => assert!(row_count > 0), + RollupRefresh::Skipped => panic!("first refresh must not be skipped"), + } + + // Nothing changed => the expensive re-aggregation must be skipped. + assert_eq!( + refresh_ai_session_rollup_if_stale(&pool).unwrap(), + RollupRefresh::Skipped, + "unchanged source must skip the refresh" + ); +} + +#[test] +fn stale_check_detects_new_ai_row() { + let (pool, _dir) = test_pool(); + seed_ai_sessions(&pool); + refresh_ai_session_rollup_if_stale(&pool).unwrap(); + assert_eq!( + refresh_ai_session_rollup_if_stale(&pool).unwrap(), + RollupRefresh::Skipped + ); + + // A new AI row advances MAX(id) => must refresh. + insert_logs_batch( + &pool, + &[make_ai_entry( + "2026-07-01T00:00:00Z", + "host9", + "codex", + "/proj/new", + "sess-new", + "brand new ai event", + )], + ) + .unwrap(); + assert!( + matches!( + refresh_ai_session_rollup_if_stale(&pool).unwrap(), + RollupRefresh::Refreshed { .. } + ), + "a new AI row must trigger a refresh" + ); + // And the new session is now visible from the rollup path. + let rows = list_ai_sessions(&pool, &default_session_params()).unwrap(); + assert!(rows.iter().any(|s| s.ai_session_id == "sess-new")); +} + +#[test] +fn stale_check_detects_deleted_ai_row() { + let (pool, _dir) = test_pool(); + seed_ai_sessions(&pool); + refresh_ai_session_rollup_if_stale(&pool).unwrap(); + assert_eq!( + refresh_ai_session_rollup_if_stale(&pool).unwrap(), + RollupRefresh::Skipped + ); + + // Deleting an AI row changes COUNT(*) (and likely MAX(id)) => must refresh. + { + let conn = pool.get().unwrap(); + let deleted = conn + .execute( + "DELETE FROM logs WHERE id IN ( + SELECT id FROM logs WHERE ai_session_id IS NOT NULL LIMIT 1 + )", + [], + ) + .unwrap(); + assert_eq!(deleted, 1, "test must delete exactly one AI row"); + } + assert!( + matches!( + refresh_ai_session_rollup_if_stale(&pool).unwrap(), + RollupRefresh::Refreshed { .. } + ), + "a deleted AI row must trigger a refresh" + ); +} + +#[test] +fn stale_check_ignores_non_ai_rows() { + let (pool, _dir) = test_pool(); + seed_ai_sessions(&pool); + refresh_ai_session_rollup_if_stale(&pool).unwrap(); + + // Plain syslog rows (the overwhelming majority of ingest) must NOT force a + // re-aggregation — that is the whole point of an AI-scoped watermark. + insert_plain_log(&pool, "2026-07-01T00:00:00Z", "ordinary syslog line"); + insert_plain_log(&pool, "2026-07-01T00:00:01Z", "another ordinary line"); + assert_eq!( + refresh_ai_session_rollup_if_stale(&pool).unwrap(), + RollupRefresh::Skipped, + "non-AI ingest must not trigger a rollup refresh" + ); +} + +#[test] +fn stale_check_watermark_is_index_only_no_table_scan() { + // The watermark fingerprint must be cheap: served from the partial index + // idx_logs_ai_project_time (WHERE ai_project IS NOT NULL), NOT a full scan + // of `logs`. That index-only cost is the whole reason the dirty-check is + // worth running on every cadence tick. Mirrors ai_rows_watermark's query. + let (pool, _dir) = test_pool(); + seed_ai_sessions(&pool); + let plan = query_plan( + &pool, + "SELECT COUNT(*), COALESCE(MAX(id), 0) FROM logs + WHERE ai_project IS NOT NULL AND ai_project != ''", + &[], + ); + assert!( + plan.contains("idx_logs_ai_project_time"), + "watermark must use the AI partial index; plan was:\n{plan}" + ); + assert!( + !plan.contains("SCAN logs\n") && !plan.ends_with("SCAN logs"), + "watermark must not full-scan logs; plan was:\n{plan}" + ); +} + +#[test] +fn stale_check_skip_keeps_rollup_correct_vs_live() { + // A skipped refresh must leave the rollup serving results identical to a + // fresh live aggregation (i.e. skipping never serves stale data when the + // source genuinely did not change). + let (pool, _dir) = test_pool(); + seed_ai_sessions(&pool); + refresh_ai_session_rollup_if_stale(&pool).unwrap(); + assert_eq!( + refresh_ai_session_rollup_if_stale(&pool).unwrap(), + RollupRefresh::Skipped + ); + + let live = list_ai_sessions_live(&pool, &default_session_params()).unwrap(); + let rolled = list_ai_sessions(&pool, &default_session_params()).unwrap(); + assert_eq!(live.len(), rolled.len()); + for (l, r) in live.iter().zip(rolled.iter()) { + assert_eq!(l.ai_session_id, r.ai_session_id); + assert_eq!(l.last_seen, r.last_seen); + assert_eq!(l.event_count, r.event_count); + } +} + +// --------------------------------------------------------------------------- +// RAG v1 tests +// --------------------------------------------------------------------------- + +fn make_app_entry(ts: &str, host: &str, severity: &str, app: &str, msg: &str) -> LogBatchEntry { + LogBatchEntry { + timestamp: ts.to_string(), + hostname: host.to_string(), + facility: None, + severity: severity.to_string(), + app_name: Some(app.to_string()), + process_id: None, + message: msg.to_string(), + raw: msg.to_string(), + source_ip: "10.0.0.1:514".to_string(), + docker_checkpoint: None, + ai_tool: None, + ai_project: None, + ai_session_id: None, + ai_transcript_path: None, + metadata_json: None, + http_status: None, + auth_outcome: None, + dns_blocked: None, + event_action: None, + parse_error: None, + } +} + +#[test] +fn similar_incidents_clusters_returns_clusters_for_matching_logs() { + let (pool, _dir) = test_pool(); + + let logs = vec![ + make_app_entry( + "2024-01-15T10:00:00Z", + "web-01", + "err", + "nginx", + "upstream connect error timeout", + ), + make_app_entry( + "2024-01-15T10:05:00Z", + "web-01", + "crit", + "nginx", + "upstream connect error connection refused", + ), + ]; + insert_logs_batch(&pool, &logs).unwrap(); + + let params = SimilarIncidentsParams { + query: "upstream".into(), + host: None, + app: None, + severity_min: None, + since: None, + until: None, + window_minutes: Some(30), + limit: Some(10), + }; + let result = similar_incidents_clusters(&pool, ¶ms).unwrap(); + assert!(!result.clusters.is_empty(), "expected at least one cluster"); + let cluster = &result.clusters[0]; + assert_eq!(cluster.hostname, "web-01"); + assert_eq!(cluster.app_name.as_deref(), Some("nginx")); + assert!(cluster.log_count >= 2); + // "crit" is more severe than "err" + assert_eq!(cluster.severity_peak, "crit"); +} + +#[test] +fn similar_incidents_clusters_filters_by_hostname() { + let (pool, _dir) = test_pool(); + + let logs = vec![ + make_app_entry( + "2024-01-15T10:00:00Z", + "web-01", + "err", + "nginx", + "upstream connect error", + ), + make_app_entry( + "2024-01-15T10:01:00Z", + "web-02", + "err", + "nginx", + "upstream connect error", + ), + ]; + insert_logs_batch(&pool, &logs).unwrap(); + + let params = SimilarIncidentsParams { + query: "upstream".into(), + host: Some("web-01".into()), + ..Default::default() + }; + let result = similar_incidents_clusters(&pool, ¶ms).unwrap(); + assert!(result.clusters.iter().all(|c| c.hostname == "web-01")); +} + +#[test] +fn similar_incidents_applies_fts_before_candidate_cap() { + let (pool, _dir) = test_pool(); + insert_logs_batch( + &pool, + &[make_app_entry( + "2024-01-01T00:00:00Z", + "web-01", + "err", + "nginx", + "historicalincidentneedle connection refused", + )], + ) + .unwrap(); + + // Populate exactly the old raw-log candidate cap with newer nonmatches. + // A recency cap applied before FTS excludes the historical match above. + let conn = pool.get().unwrap(); + conn.execute_batch( + "WITH digits(d) AS ( + VALUES (0),(1),(2),(3),(4),(5),(6),(7),(8),(9) + ), + rows(n) AS ( + SELECT a.d + 10*b.d + 100*c.d + 1000*d.d + 10000*e.d + FROM digits a + CROSS JOIN digits b + CROSS JOIN digits c + CROSS JOIN digits d + CROSS JOIN digits e + ) + INSERT INTO logs + (timestamp, hostname, severity, app_name, message, raw, source_ip) + SELECT '2024-01-02T00:00:00Z', 'web-01', 'info', 'nginx', + 'routine health check', 'routine health check', '10.0.0.1:514' + FROM rows;", + ) + .unwrap(); + drop(conn); + + let result = similar_incidents_clusters( + &pool, + &SimilarIncidentsParams { + query: "historicalincidentneedle".into(), + ..Default::default() + }, + ) + .unwrap(); + + assert_eq!(result.clusters.len(), 1); + assert_eq!(result.clusters[0].window_start, "2024-01-01T00:00:00Z"); + assert_eq!(result.clusters[0].log_count, 1); +} + +#[test] +fn incident_context_summary_returns_window_stats() { + let (pool, _dir) = test_pool(); + + let logs = vec![ + make_app_entry( + "2024-02-01T08:00:00Z", + "db-01", + "err", + "postgres", + "FATAL: out of shared memory", + ), + make_app_entry( + "2024-02-01T08:01:00Z", + "db-01", + "info", + "postgres", + "database system is ready", + ), + ]; + insert_logs_batch(&pool, &logs).unwrap(); + + let params = IncidentContextParams { + since: "2024-02-01T07:00:00Z".into(), + until: "2024-02-01T09:00:00Z".into(), + host: None, + app: None, + query: None, + severity_min: Some("err".into()), + limit: Some(10), + }; + let result = incident_context_summary(&pool, ¶ms).unwrap(); + assert_eq!(result.total_logs, 2); + assert!(!result.by_severity.is_empty()); + // Only the "err" row should be in error_logs (not "info") + assert_eq!(result.error_logs.len(), 1); + assert_eq!(result.error_logs[0].message, "FATAL: out of shared memory"); +} + +#[test] +fn incident_context_summary_empty_window_returns_zero() { + let (pool, _dir) = test_pool(); + + let params = IncidentContextParams { + since: "2020-01-01T00:00:00Z".into(), + until: "2020-01-02T00:00:00Z".into(), + ..Default::default() + }; + let result = incident_context_summary(&pool, ¶ms).unwrap(); + assert_eq!(result.total_logs, 0); + assert!(result.error_logs.is_empty()); + assert!(result.ai_sessions.is_empty()); +} + +#[test] +fn incident_context_summary_filters_error_logs_by_fts_query() { + let (pool, _dir) = test_pool(); + insert_logs_batch( + &pool, + &[ + make_app_entry( + "2024-02-01T08:00:00Z", + "db-01", + "err", + "postgres", + "shared memory exhausted", + ), + make_app_entry( + "2024-02-01T08:01:00Z", + "db-01", + "err", + "postgres", + "connection pool exhausted", + ), + ], + ) + .unwrap(); + + let result = incident_context_summary( + &pool, + &IncidentContextParams { + since: "2024-02-01T07:00:00Z".into(), + until: "2024-02-01T09:00:00Z".into(), + query: Some("memory".into()), + ..Default::default() + }, + ) + .unwrap(); + + assert_eq!(result.total_logs, 2, "window aggregates remain unfiltered"); + assert_eq!(result.error_logs.len(), 1); + assert_eq!(result.error_logs[0].message, "shared memory exhausted"); +} + +#[test] +fn incident_context_window_queries_force_timestamp_index() { + let (pool, _dir) = test_pool(); + let plan = query_plan( + &pool, + "SELECT COUNT(*) + FROM logs INDEXED BY idx_logs_timestamp + WHERE (ai_project IS NULL OR ai_project = '') + AND timestamp BETWEEN ?1 AND ?2", + &[ + rusqlite::types::Value::Text("2024-02-01T07:00:00Z".into()), + rusqlite::types::Value::Text("2024-02-01T09:00:00Z".into()), + ], + ); + assert!( + plan.contains("idx_logs_timestamp"), + "incident context window scan should be timestamp-index driven; got:\n{plan}" + ); +} + +// ─────────────────────────────────────────────────────────────────────────── +// Performance benchmark harness (Issue 4 / bead cortex-2vre). +// +// Builds a synthetic on-disk SQLite DB with a realistic row count and times +// `get_stats` and `list_ai_sessions` before/after the optimization work. +// +// IGNORED by default — it builds millions of rows and takes minutes, so it +// must never run in the normal `cargo nextest` suite. Run explicitly: +// +// CORTEX_BENCH_ROWS=10000000 cargo test --lib \ +// db::queries::tests::bench_stats_and_sessions -- --ignored --nocapture +// +// Row count is controlled by CORTEX_BENCH_ROWS (default 5_000_000). +// ─────────────────────────────────────────────────────────────────────────── + +/// Insert `n` synthetic log rows through the live schema (FTS + inventory + +/// counter triggers all fire), in large transactions for throughput. ~20% of +/// rows carry AI session fields spread across many (project, tool, session) +/// groups so the sessions query has realistic cardinality. +fn bench_seed_rows(pool: &DbPool, n: usize) { + use std::time::Instant; + let started = Instant::now(); + const CHUNK: usize = 50_000; + let mut inserted = 0usize; + while inserted < n { + let this = CHUNK.min(n - inserted); + let mut conn = pool.get().unwrap(); + let tx = conn.transaction().unwrap(); + { + let mut stmt = tx + .prepare_cached( + "INSERT INTO logs (timestamp, hostname, facility, severity, app_name, + process_id, message, raw, received_at, source_ip, + ai_tool, ai_project, ai_session_id, ai_transcript_path) + VALUES (?1,?2,?3,?4,?5,?6,?7,?8,?9,?10,?11,?12,?13,?14)", + ) + .unwrap(); + for i in 0..this { + let x = inserted + i; + // Strictly-increasing timestamp per row (base + x seconds, full + // date rollover). Monotonic in `x` => each session's MAX(ts) is + // globally unique (no last_seen ties), so the rollup vs live + // top-N comparison is deterministic at the LIMIT boundary. + let base = chrono::DateTime::parse_from_rfc3339("2026-01-01T00:00:00Z").unwrap(); + let dt = base + chrono::TimeDelta::seconds(x as i64); + let ts = dt.format("%Y-%m-%dT%H:%M:%SZ").to_string(); + let recv = ts.clone(); + let host = format!("host{:02}", x % 25); + let app = format!("app{:02}", x % 60); + let msg = format!("synthetic log line {x} some error retry connection text"); + let is_ai = x.is_multiple_of(5); + let (tool, proj, sess, tpath): ( + Option, + Option, + Option, + Option, + ) = if is_ai { + let proj = format!("/proj/{}", x % 40); + let tool = if x.is_multiple_of(2) { + "codex" + } else { + "claude" + } + .to_string(); + // ~20k distinct sessions => realistic group cardinality. + let sess = format!("sess-{}", x % 20_000); + let tpath = format!("{proj}/{sess}.jsonl"); + (Some(tool), Some(proj), Some(sess), Some(tpath)) + } else { + (None, None, None, None) + }; + stmt.execute(rusqlite::params![ + ts, + host, + Option::::None, + "info", + app, + Option::::None, + msg, + "raw", + recv, + "10.0.0.1:514", + tool, + proj, + sess, + tpath, + ]) + .unwrap(); + } + } + tx.commit().unwrap(); + inserted += this; + } + eprintln!( + "[bench] seeded {inserted} rows in {:.1}s", + started.elapsed().as_secs_f64() + ); +} + +#[derive(Debug)] +struct BenchPercentiles { + p50_ms: f64, + p95_ms: f64, +} + +/// p50/p95 of N timed runs of `f`, in milliseconds. One warm-up run first. +/// Uses nearest-rank p95 so the emitted values remain reproducible from the +/// captured sample count without interpolating measurements that never ran. +fn bench_percentiles_ms(runs: usize, mut f: impl FnMut()) -> BenchPercentiles { + use std::time::Instant; + assert!(runs > 0); + f(); // warm-up + let mut samples: Vec = Vec::with_capacity(runs); + for _ in 0..runs { + let t = Instant::now(); + f(); + samples.push(t.elapsed().as_secs_f64() * 1000.0); + } + samples.sort_by(|a, b| a.partial_cmp(b).unwrap()); + let p50_index = (samples.len() - 1) / 2; + let p95_index = ((samples.len() * 95).div_ceil(100)).saturating_sub(1); + BenchPercentiles { + p50_ms: samples[p50_index], + p95_ms: samples[p95_index], + } +} + +#[test] +fn session_rollup_query_plan_uses_ordering_index_without_temp_sort() { + let dir = tempfile::tempdir().unwrap(); + let pool = init_pool(&test_storage_config(dir.path().join("plan.db"))).unwrap(); + let conn = pool.get().unwrap(); + let plan = conn + .prepare( + "EXPLAIN QUERY PLAN + SELECT ai_project, ai_tool, ai_session_id, hostname, last_seen + FROM ai_session_rollup + WHERE 1=1 + ORDER BY last_seen DESC LIMIT 100", + ) + .unwrap() + .query_map([], |row| row.get::<_, String>(3)) + .unwrap() + .collect::>>() + .unwrap(); + assert!( + plan.iter() + .any(|detail| detail.contains("idx_ai_session_rollup_last_seen")), + "rollup query must use its last-seen index: {plan:?}" + ); + assert!( + plan.iter() + .all(|detail| !detail.contains("USE TEMP B-TREE")), + "rollup query must not perform a temporary sort: {plan:?}" + ); +} + +#[test] +#[ignore = "performance benchmark; builds millions of rows. Run with --ignored."] +fn bench_stats_and_sessions() { + let rows: usize = crate::env::var("CORTEX_BENCH_ROWS") + .ok() + .and_then(|v| v.parse().ok()) + .unwrap_or(5_000_000); + + // Optional persistent DB path so a seeded DB can be reused across runs + // (seeding 10M rows takes ~13 min). When unset, use a throwaway tempdir. + let _guard_dir; + let (pool, cfg) = if let Ok(path) = crate::env::var("CORTEX_BENCH_DB") { + let db_path = std::path::PathBuf::from(&path); + let fresh = !db_path.exists(); + let cfg = test_storage_config(db_path); + let pool = init_pool(&cfg).unwrap(); + if fresh { + bench_seed_rows(&pool, rows); + } else { + eprintln!("[bench] reusing existing DB at {path} (skipping seed)"); + } + (pool, cfg) + } else { + let dir = tempfile::tempdir().unwrap(); + let cfg = test_storage_config(dir.path().join("test.db")); + let pool = init_pool(&cfg).unwrap(); + bench_seed_rows(&pool, rows); + _guard_dir = dir; // keep alive + (pool, cfg) + }; + + let ground_truth: i64 = { + let conn = pool.get().unwrap(); + conn.query_row("SELECT COUNT(*) FROM logs", [], |r| r.get(0)) + .unwrap() + }; + eprintln!("[bench] ground-truth COUNT(*) FROM logs = {ground_truth}"); + + // --- stats (default: FTS diagnostic skipped) --- + let mut last_stats = None; + let stats_latency = bench_percentiles_ms(20, || { + last_stats = Some(get_stats(&pool, &cfg).unwrap()); + }); + let stats = last_stats.unwrap(); + eprintln!( + "[bench] get_stats (default, FTS skipped): p50={:.1} ms p95={:.1} ms (total_logs={})", + stats_latency.p50_ms, stats_latency.p95_ms, stats.total_logs + ); + assert_eq!( + stats.total_logs, ground_truth, + "stats total_logs must equal ground-truth COUNT(*)" + ); + + // --- stats with FTS diagnostic ON (the expensive COUNT(*) FROM logs_fts) --- + let stats_fts_latency = bench_percentiles_ms(20, || { + let _ = get_stats_with_options(&pool, &cfg, true).unwrap(); + }); + eprintln!( + "[bench] get_stats (FTS diagnostic ON): p50={:.1} ms p95={:.1} ms", + stats_fts_latency.p50_ms, stats_fts_latency.p95_ms + ); + + // --- sessions BEFORE: live aggregation (GROUP BY + temp-btree sort) --- + let params = ListAiSessionsParams { + ai_project: None, + ai_tool: None, + host: None, + since: None, + until: None, + limit: Some(100), + }; + let mut live_rows = 0usize; + let sessions_live_latency = bench_percentiles_ms(20, || { + live_rows = list_ai_sessions_live(&pool, ¶ms).unwrap().len(); + }); + eprintln!( + "[bench] BEFORE list_ai_sessions_live(limit=100): p50={:.1} ms p95={:.1} ms ({live_rows} rows)", + sessions_live_latency.p50_ms, sessions_live_latency.p95_ms + ); + + // --- refresh cost (background cadence; not on the request path) --- + let mut rollup_total = 0usize; + let refresh_latency = bench_percentiles_ms(10, || { + rollup_total = refresh_ai_session_rollup(&pool).unwrap(); + }); + eprintln!( + "[bench] refresh_ai_session_rollup: p50={:.1} ms p95={:.1} ms ({rollup_total} session rows total)", + refresh_latency.p50_ms, refresh_latency.p95_ms + ); + + // --- sessions AFTER: indexed read from the rollup materialization --- + let mut rollup_rows = 0usize; + let sessions_rollup_latency = bench_percentiles_ms(20, || { + rollup_rows = list_ai_sessions(&pool, ¶ms).unwrap().len(); + }); + eprintln!( + "[bench] AFTER list_ai_sessions(rollup, limit=100): p50={:.1} ms p95={:.1} ms ({rollup_rows} rows)", + sessions_rollup_latency.p50_ms, sessions_rollup_latency.p95_ms + ); + + // Correctness: the rollup-served top-N must equal the live top-N. Both + // paths order by `last_seen DESC` only, so rows that TIE on last_seen may + // appear in different relative order between the two plans. Compare in a + // tie-order-independent way: (a) the multiset of last_seen ordering keys + // must be identical, and (b) the per-session (last_seen, event_count) facts + // must match for every returned session. + let live = list_ai_sessions_live(&pool, ¶ms).unwrap(); + let rollup = list_ai_sessions(&pool, ¶ms).unwrap(); + assert_eq!(live.len(), rollup.len(), "rollup/live row count mismatch"); + let mut live_keys: Vec<&String> = live.iter().map(|s| &s.last_seen).collect(); + let mut rollup_keys: Vec<&String> = rollup.iter().map(|s| &s.last_seen).collect(); + live_keys.sort(); + rollup_keys.sort(); + assert_eq!( + live_keys, rollup_keys, + "rollup/live last_seen ordering-key multisets differ" + ); + let live_facts: std::collections::HashMap<_, _> = live + .iter() + .map(|s| { + ( + (&s.ai_project, &s.ai_tool, &s.ai_session_id, &s.hostname), + (&s.last_seen, s.event_count), + ) + }) + .collect(); + for r in &rollup { + let key = (&r.ai_project, &r.ai_tool, &r.ai_session_id, &r.hostname); + match live_facts.get(&key) { + Some((last_seen, count)) => { + assert_eq!(*last_seen, &r.last_seen, "last_seen mismatch for {key:?}"); + assert_eq!(*count, r.event_count, "event_count mismatch for {key:?}"); + } + None => panic!("rollup returned session not in live top-N: {key:?}"), + } + } + + let speedup = sessions_live_latency.p50_ms / sessions_rollup_latency.p50_ms.max(0.001); + let minimum_speedup = crate::env::var("CORTEX_BENCH_MIN_SESSION_SPEEDUP") + .ok() + .map(|value| { + value + .parse::() + .expect("CORTEX_BENCH_MIN_SESSION_SPEEDUP must be a number") + }); + let maximum_p95 = crate::env::var("CORTEX_BENCH_MAX_SESSION_P95_MS") + .ok() + .map(|value| { + value + .parse::() + .expect("CORTEX_BENCH_MAX_SESSION_P95_MS must be a number") + }); + let speedup_passed = minimum_speedup.is_none_or(|minimum| speedup >= minimum); + let p95_passed = maximum_p95.is_none_or(|maximum| sessions_rollup_latency.p95_ms <= maximum); + if let Ok(path) = crate::env::var("CORTEX_BENCH_ARTIFACT") { + let artifact = serde_json::json!({ + "rows": rows, + "stats_default": {"p50_ms": stats_latency.p50_ms, "p95_ms": stats_latency.p95_ms}, + "stats_fts": {"p50_ms": stats_fts_latency.p50_ms, "p95_ms": stats_fts_latency.p95_ms}, + "sessions_live": {"p50_ms": sessions_live_latency.p50_ms, "p95_ms": sessions_live_latency.p95_ms}, + "sessions_rollup": {"p50_ms": sessions_rollup_latency.p50_ms, "p95_ms": sessions_rollup_latency.p95_ms}, + "refresh": {"p50_ms": refresh_latency.p50_ms, "p95_ms": refresh_latency.p95_ms}, + "session_speedup": speedup, + "minimum_session_speedup": minimum_speedup, + "maximum_session_p95_ms": maximum_p95, + "outcome": {"speedup_passed": speedup_passed, "p95_passed": p95_passed}, + }); + std::fs::write(path, serde_json::to_vec_pretty(&artifact).unwrap()).unwrap(); + } + if let Some(minimum_speedup) = minimum_speedup { + assert!( + speedup_passed, + "session rollup speedup {speedup:.2}x is below the {minimum_speedup:.2}x configured threshold" + ); + } + assert!( + p95_passed, + "session rollup p95 {:.1}ms exceeds the configured diagnostic threshold", + sessions_rollup_latency.p95_ms + ); + eprintln!( + "[bench] SUMMARY rows={rows} \ + stats_default_p50_ms={:.1} stats_default_p95_ms={:.1} \ + stats_fts_on_p50_ms={:.1} stats_fts_on_p95_ms={:.1} \ + sessions_BEFORE_live_p50_ms={:.1} sessions_BEFORE_live_p95_ms={:.1} \ + sessions_AFTER_rollup_p50_ms={:.1} sessions_AFTER_rollup_p95_ms={:.1} \ + refresh_p50_ms={:.1} refresh_p95_ms={:.1} sessions_p50_speedup={speedup:.1}x", + stats_latency.p50_ms, + stats_latency.p95_ms, + stats_fts_latency.p50_ms, + stats_fts_latency.p95_ms, + sessions_live_latency.p50_ms, + sessions_live_latency.p95_ms, + sessions_rollup_latency.p50_ms, + sessions_rollup_latency.p95_ms, + refresh_latency.p50_ms, + refresh_latency.p95_ms, + ); +} + +/// full-review QM2: the 15-column log projection is written inline at ~14 +/// sites across queries.rs / analytics.rs / ingest.rs, and `map_row` / +/// `map_row_offset` / `map_row_with_raw` read columns BY ORDINAL POSITION — +/// reordering or inserting a column at one site without updating the readers +/// silently mis-maps fields with no compile error. This drift test extracts +/// every projection that ends in `metadata_json` from the source text and +/// asserts it carries the canonical column order. (`map_row_with_raw` selects +/// `..., metadata_json, raw`; the canonical prefix still applies.) +#[test] +fn inline_log_projections_match_map_row_column_order() { + // Two canonical shapes exist: `map_row` (15 cols) and `map_row_with_raw` + // (16 cols, `raw` between `message` and `received_at`). + const CANON: &str = "id timestamp hostname facility severity app_name process_id message \ + received_at source_ip ai_tool ai_project ai_session_id ai_transcript_path metadata_json"; + const CANON_WITH_RAW: &str = "id timestamp hostname facility severity app_name process_id \ + message raw received_at source_ip ai_tool ai_project ai_session_id ai_transcript_path \ + metadata_json"; + let canon_tokens: Vec<&str> = CANON.split_whitespace().collect(); + let canon_raw_tokens: Vec<&str> = CANON_WITH_RAW.split_whitespace().collect(); + + let sources = [ + ("queries.rs", include_str!("queries.rs")), + ("analytics.rs", include_str!("analytics.rs")), + ("ingest.rs", include_str!("ingest.rs")), + ]; + let re = regex::Regex::new( + r"SELECT\s+((?:[a-zA-Z_][a-zA-Z_0-9]*\.)?id[\sa-zA-Z_0-9,.\\]*?metadata_json)", + ) + .unwrap(); + + let mut checked = 0usize; + for (name, src) in sources { + for cap in re.captures_iter(src) { + let projection = &cap[1]; + let tokens: Vec = projection + .split([',', '\\']) + .map(|t| t.trim()) + .filter(|t| !t.is_empty()) + .map(|t| { + // Strip any table alias prefix ("l.id" -> "id"). + t.rsplit('.').next().unwrap_or(t).to_string() + }) + .collect(); + assert!( + tokens == canon_tokens || tokens == canon_raw_tokens, + "{name}: inline log projection diverges from map_row / \ + map_row_with_raw column order — update the projection AND the \ + row readers together:\n{projection}\ngot: {tokens:?}" + ); + checked += 1; + } + } + assert!( + checked >= 10, + "expected to find at least 10 inline projections; the extraction regex \ + may have rotted (found {checked})" + ); +} + +#[test] +fn lint_flags_unquoted_infix_hyphen_term() { + let err = validate_fts_query("smoke-test").unwrap_err().to_string(); + assert!( + err.contains("NOT operator"), + "should explain hyphen trap: {err}" + ); + assert!( + err.contains("--grep") || err.contains("\"smoke-test\""), + "should suggest a fix: {err}" + ); +} + +#[test] +fn lint_accepts_quoted_phrase() { + // Already-quoted hyphenated phrase is valid FTS5 and must pass. + assert!(validate_fts_query("\"smoke-test\"").is_ok()); +} + +#[test] +fn lint_accepts_normal_boolean_query() { + assert!(validate_fts_query("error AND nginx").is_ok()); +} + +#[test] +fn lint_leaves_leading_hyphen_not_term_alone() { + // `-nginx` is an intentional FTS5 NOT, not the hyphenated-word trap. + assert!(validate_fts_query("error -nginx").is_ok()); +} + +#[test] +fn lint_flags_unbalanced_quote() { + let err = validate_fts_query("\"oops").unwrap_err().to_string(); + assert!(err.contains("unbalanced quote"), "{err}"); +} + +#[test] +fn lint_flags_unquoted_hyphen_term_alongside_a_quoted_phrase() { + // The hyphen check is per-term: a quoted phrase elsewhere must not mask an + // unquoted hyphenated term (regression for a query-wide quote gate). + let err = validate_fts_query("\"disk full\" smoke-test") + .unwrap_err() + .to_string(); + assert!(err.contains("NOT operator"), "{err}"); +} From 94e2c6970045fa804c4224bc753cf938d7bb02a8 Mon Sep 17 00:00:00 2001 From: jmagar <38927646+jmagar@users.noreply.github.com> Date: Wed, 23 Sep 2026 13:23:53 -0400 Subject: [PATCH 6/8] style: apply rustfmt to investigation modules --- src/app/services/session_investigation.rs | 1 - src/app/services/session_investigation_tests.rs | 3 +-- 2 files changed, 1 insertion(+), 3 deletions(-) diff --git a/src/app/services/session_investigation.rs b/src/app/services/session_investigation.rs index fa3d339fc..9efe1a2ba 100644 --- a/src/app/services/session_investigation.rs +++ b/src/app/services/session_investigation.rs @@ -8,7 +8,6 @@ use super::session_investigation_support::{ summarize_session_source_evidence, timestamp_inclusive_between, }; - impl CortexService { pub async fn session_investigate( &self, diff --git a/src/app/services/session_investigation_tests.rs b/src/app/services/session_investigation_tests.rs index d614f57d7..957292072 100644 --- a/src/app/services/session_investigation_tests.rs +++ b/src/app/services/session_investigation_tests.rs @@ -1,6 +1,5 @@ -use super::*; use super::super::session_investigation_support::looks_like_linear_identifier; - +use super::*; fn event(position: i64, text: &str) -> models::RenderedSessionEvent { models::RenderedSessionEvent { From 0d509e373b605adc27f38d4ee3da906381171e5c Mon Sep 17 00:00:00 2001 From: jmagar <38927646+jmagar@users.noreply.github.com> Date: Wed, 23 Sep 2026 13:25:12 -0400 Subject: [PATCH 7/8] test: include session id in benchmark params --- src/db/queries_tests.rs | 1 + 1 file changed, 1 insertion(+) diff --git a/src/db/queries_tests.rs b/src/db/queries_tests.rs index 731927987..713f12276 100644 --- a/src/db/queries_tests.rs +++ b/src/db/queries_tests.rs @@ -3424,6 +3424,7 @@ fn bench_stats_and_sessions() { let params = ListAiSessionsParams { ai_project: None, ai_tool: None, + ai_session_id: None, host: None, since: None, until: None, From bf7c46057878b6514cca32632c3637034dbf5320 Mon Sep 17 00:00:00 2001 From: Jake Magar Date: Wed, 30 Sep 2026 02:13:43 -0400 Subject: [PATCH 8/8] fix: bound session investigation evidence and resolve review findings --- CLAUDE.md | 1 + README.md | 2 +- docs/INVENTORY.md | 1 + docs/mcp/SCHEMA.md | 4 +- docs/mcp/TESTS.md | 4 +- docs/mcp/TOOLS.md | 16 + plugins/cortex/skills/cortex/SKILL.md | 1 + src/api.rs | 1 + src/app/models/investigation.rs | 24 +- src/app/models/investigation_tests.rs | 25 + src/app/services.rs | 3 + src/app/services/ai.rs | 66 +-- src/app/services/ai_correlate_tests.rs | 1 + src/app/services/session_graph_correlation.rs | 103 ++++ src/app/services/session_investigation.rs | 270 ++++------- .../services/session_investigation_budget.rs | 454 ++++++++++++++++++ .../session_investigation_budget_tests.rs | 145 ++++++ .../session_investigation_sections.rs | 188 ++++++++ .../services/session_investigation_support.rs | 65 +-- .../services/session_investigation_tests.rs | 132 ++++- src/db.rs | 13 +- src/db/agent_observatory_read.rs | 26 +- src/db/agent_observatory_read_models.rs | 2 + src/db/agent_observatory_read_tests.rs | 53 ++ src/db/agent_observatory_run_commits.rs | 73 +-- src/db/models.rs | 1 + src/db/queries.rs | 7 +- src/db/queries_session_graph.rs | 168 +++++++ src/db/queries_session_graph_tests.rs | 218 +++++++++ src/db/queries_tests.rs | 40 ++ src/filetail/supervisor_tests.rs | 9 +- src/mcp/actions.rs | 2 +- src/mcp/tools_tests.rs | 50 ++ src/surfaces.rs | 1 + tests/live/phases/mcp/run.sh | 1 + tests/live/phases/mcp/scenarios.json | 3 +- 36 files changed, 1814 insertions(+), 359 deletions(-) create mode 100644 src/app/models/investigation_tests.rs create mode 100644 src/app/services/session_graph_correlation.rs create mode 100644 src/app/services/session_investigation_budget.rs create mode 100644 src/app/services/session_investigation_budget_tests.rs create mode 100644 src/app/services/session_investigation_sections.rs create mode 100644 src/db/queries_session_graph.rs create mode 100644 src/db/queries_session_graph_tests.rs diff --git a/CLAUDE.md b/CLAUDE.md index 74a68e5e6..efdbfbb7e 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -133,6 +133,7 @@ Scope taxonomy: every action requires `cortex:read` except the six **admin** act | `apps` | Enumerate all known application names | | `sessions` | List AI transcript sessions | | `search_sessions` | Full-text search over AI transcript sessions | +| `session_investigate` | Build a bounded evidence bundle rooted at one AI session | | `evidence_scope` | Historical Agent Observatory evidence for a Git branch or worktree | | `abuse` | Detect resource-abuse patterns in AI sessions | | `abuse_incidents` | List detected abuse incidents | diff --git a/README.md b/README.md index c73f682b5..aac8b8556 100644 --- a/README.md +++ b/README.md @@ -494,7 +494,7 @@ The current scope split is: | Discovery and health | `hosts`, `apps`, `source_ips`, `status`, `stats`, `ingest_rate`, `silent_hosts`, `clock_skew` | | Analytics and correlation | `timeline`, `patterns`, `anomalies`, `compare`, `correlate`, `topic_correlate`, `similar_incidents`, `recurring_error_comparison`, `incident_context` | | Fleet and topology | `map`, `host_state`, `fleet_state`, `correlate_state`, `graph`, `compose_status`, `compose_doctor` | -| AI sessions and scoped evidence | `sessions`, `search_sessions`, `evidence_scope`, `abuse`, `abuse_incidents`, `abuse_investigate`, `ai_correlate`, `usage_blocks`, `project_context`, `list_ai_tools`, `list_ai_projects` | +| AI sessions and scoped evidence | `sessions`, `search_sessions`, `session_investigate`, `evidence_scope`, `abuse`, `abuse_incidents`, `abuse_investigate`, `ai_correlate`, `usage_blocks`, `project_context`, `list_ai_tools`, `list_ai_projects` | | AI operational events | `skill_events`, `skill_incidents`, `skill_investigate`, `mcp_events`, `mcp_incidents`, `mcp_investigate`, `hook_events`, `hook_incidents`, `hook_investigate` | | Errors and administration | `unaddressed_errors`, `ack_error`, `unack_error`, `notifications_recent`, `notifications_test`, `file_tails`, `llm_invocations`, `artifact_evidence_record` | | Reference | `help` | diff --git a/docs/INVENTORY.md b/docs/INVENTORY.md index ee87f44f7..0a029b857 100644 --- a/docs/INVENTORY.md +++ b/docs/INVENTORY.md @@ -86,6 +86,7 @@ that registry by `src/mcp/schemas.rs::tool_definitions()`. | `status` | Lightweight runtime status: DB health, queue/backpressure state, listener/writer counters, OTLP counters | no | | `sessions` | AI transcript sessions grouped by project/tool/session/host | no | | `search_sessions` | Ranked grouped session search | no | +| `session_investigate` | Bounded evidence bundle rooted at one AI session | no | | `evidence_scope` | Historical Agent Observatory evidence for a Git branch or worktree | no | | `abuse` | Abuse-term detector with same-session context | no | | `abuse_incidents` | Groups abuse hits into scored incident candidates | no | diff --git a/docs/mcp/SCHEMA.md b/docs/mcp/SCHEMA.md index 906ad90c0..3f6f54994 100644 --- a/docs/mcp/SCHEMA.md +++ b/docs/mcp/SCHEMA.md @@ -45,6 +45,7 @@ selects one of the actions below. The mechanically generated current count is in | `apps` | `cortex:read` | cheap | Distinct application names with counts | | `sessions` | `cortex:read` | cheap | AI transcript session inventory | | `search_sessions` | `cortex:read` | cheap | FTS5 search over AI transcript sessions | +| `session_investigate` | `cortex:read` | expensive | Bounded evidence bundle rooted at one AI session | | `evidence_scope` | `cortex:read` | moderate | Historical Agent Observatory evidence for a Git branch or worktree | | `abuse` | `cortex:read` | moderate | Abuse-term hits with same-session context | | `abuse_incidents` | `cortex:read` | moderate | Grouped abuse incident candidates | @@ -175,12 +176,13 @@ boundary. | `query` | `search`, `search_sessions`, `correlate`, `similar_incidents` | | `hostname` | `search`, `filter`, `tail`, `correlate`, `host_state`, `ai_correlate`, `apps`, `sessions`, `timeline`, `patterns`, `context`, `similar_incidents`, `incident_context` | | `host_id` | Authoritative heartbeat identity for `host_state` | -| `host` | Optional host_id-or-hostname filter for `correlate_state` | +| `host` | Optional host_id-or-hostname filter for `correlate_state`; exact session identity qualifier for `session_investigate` | | `reference_time` | Required window center for `correlate_state`; for `correlate`, required unless `query` is given (then derived from an AI-session search) | | `source_ip` | `search`, `filter`, `tail`, `correlate`, `ai_correlate` | | `source_kind` | `filter` only; aliases Docker, file-tail, command-history, shell-history, transcript, and AI-tool rows | | `project` | `filter`, `sessions`, `search_sessions`, `abuse`, `ai_correlate`, `usage_blocks`, `project_context`, `list_ai_tools` | | `tool` | `filter`, `sessions`, `search_sessions`, `abuse`, `ai_correlate`, `usage_blocks`, `project_context`, `list_ai_projects` | +| `session_id` | Required for `session_investigate`; optional exact native identity filter for `sessions` | | `branch`, `worktree` | `evidence_scope`; at least one is required by service validation | | `kinds`, `include_payload`, `after_id` | `evidence_scope` filtering and durable pagination | | `session_id` | `filter`, `ai_correlate` | diff --git a/docs/mcp/TESTS.md b/docs/mcp/TESTS.md index f46fa181c..e2ac93282 100644 --- a/docs/mcp/TESTS.md +++ b/docs/mcp/TESTS.md @@ -76,7 +76,7 @@ the repo-local debug binary at `target/debug/cortex`, so repo-local builds do not require an installed shell binary. Action registry covered by live/script references: `search`, `filter`, `tail`, `errors`, -`hosts`, `map`, `host_state`, `fleet_state`, `correlate_state`, `topic_correlate`, `sessions`, `search_sessions`, `evidence_scope`, `abuse`, `abuse_incidents`, `abuse_investigate`, `ai_correlate`, `usage_blocks`, `project_context`, +`hosts`, `map`, `host_state`, `fleet_state`, `correlate_state`, `topic_correlate`, `sessions`, `search_sessions`, `session_investigate`, `evidence_scope`, `abuse`, `abuse_incidents`, `abuse_investigate`, `ai_correlate`, `usage_blocks`, `project_context`, `list_ai_tools`, `list_ai_projects`, `correlate`, `stats`, `status`, `apps`, `source_ips`, `timeline`, `patterns`, `context`, `get`, `ingest_rate`, `silent_hosts`, `clock_skew`, `anomalies`, `compare`, `compose_status`, @@ -194,7 +194,7 @@ curl -s -X POST http://localhost:3100/mcp \ ## Testing checklist - [ ] **All actions return expected shape** -- cortex search, cortex tail, cortex errors, cortex hosts, cortex host_state, cortex sessions, cortex correlate, cortex stats, cortex status, cortex help -- [ ] **AI session analytics and scoped lifecycle evidence return expected shape and seeded rows** -- cortex search_sessions, cortex evidence_scope, cortex abuse, cortex sessions_correlate, cortex usage_blocks, cortex project_context, cortex list_ai_tools, cortex list_ai_projects +- [ ] **AI session analytics and scoped lifecycle evidence return expected shape and seeded rows** -- cortex search_sessions, cortex session_investigate, cortex evidence_scope, cortex abuse, cortex sessions_correlate, cortex usage_blocks, cortex project_context, cortex list_ai_tools, cortex list_ai_projects - [ ] **Auth: valid token** -- 200 with correct Bearer token - [ ] **Auth: invalid token** -- 401 Unauthorized - [ ] **Auth: no token when required** -- 401 Unauthorized diff --git a/docs/mcp/TOOLS.md b/docs/mcp/TOOLS.md index efda909c6..6c3e11030 100644 --- a/docs/mcp/TOOLS.md +++ b/docs/mcp/TOOLS.md @@ -24,6 +24,7 @@ cortex exposes one MCP tool named `cortex`. The required | `correlate_state` | Correlate logs with heartbeat window summaries around a reference time | | `sessions` | AI transcript sessions by project | | `search_sessions` | Ranked grouped session search | +| `session_investigate` | Bounded evidence bundle rooted at one AI session | | `evidence_scope` | Historical Agent Observatory evidence for a Git branch or worktree | | `abuse` | Abuse hits in AI transcripts with same-session context | | `abuse_incidents` | Groups abuse hits into scored incident candidates | @@ -182,6 +183,21 @@ Required arguments: `action = "search_sessions"`, `query` Optional arguments: `project`, `tool`, `from`, `to`, `limit`. +## cortex session_investigate + +Build an evidence bundle for one exact AI session. Required arguments: +`action = "session_investigate"`, `session_id`. Optional arguments: `tool`, +`project`, `host`, `limit` (1–200), and `severity_min`. Provide identity qualifiers +when a native session ID is shared by several transcripts. + +The response contains `metadata` and `result`, including rendered transcript, +scoped graph correlation, Observatory lineage, tools/skills/hooks, artifacts, +notifications, incidents, and retained deletion lineage. Each section reports +truncation; `partial_reasons` records omitted evidence. The complete JSON envelope +is capped at 64 KiB and wall time at two seconds; exhausted wall time returns a +retryable busy error. Transcript references are passive claims with +`verified = false`. + ## cortex evidence_scope Return a bounded historical page of Agent Observatory events associated with an diff --git a/plugins/cortex/skills/cortex/SKILL.md b/plugins/cortex/skills/cortex/SKILL.md index 4af4fab65..3687582fd 100644 --- a/plugins/cortex/skills/cortex/SKILL.md +++ b/plugins/cortex/skills/cortex/SKILL.md @@ -24,6 +24,7 @@ A single MCP tool, `mcp__cortex__cortex`, dispatches on a required `action` argu | `correlate_state` | Correlate logs with heartbeat summaries around a reference time | | `sessions` | AI transcript sessions by project | | `search_sessions` | Ranked grouped session search | +| `session_investigate` | Bounded evidence bundle rooted at one AI session | | `evidence_scope` | Historical Agent Observatory evidence for a Git branch or worktree | | `abuse` | Abuse hits in AI transcripts with same-session context | | `abuse_incidents` | Groups abuse hits into scored incident candidates | diff --git a/src/api.rs b/src/api.rs index 3461bbb21..6277b0af8 100644 --- a/src/api.rs +++ b/src/api.rs @@ -730,6 +730,7 @@ async fn observatory_runs( since: query.since, until: query.until, active_only: query.active_only.unwrap_or(false), + ..Default::default() }, query.cursor, query.limit.unwrap_or(50), diff --git a/src/app/models/investigation.rs b/src/app/models/investigation.rs index 7f8fcd8c2..2e2b4d162 100644 --- a/src/app/models/investigation.rs +++ b/src/app/models/investigation.rs @@ -185,7 +185,6 @@ pub struct SessionInvestigateRequest { pub project: Option, pub host: Option, pub limit: Option, - pub window_minutes: Option, pub severity_min: Option, } @@ -242,6 +241,7 @@ pub struct SessionObservatoryEvidence { pub repository: Option, pub worktree: Option, pub commits: Vec, + pub commits_truncated: bool, pub events: Vec, pub spans: Vec, pub metrics: Vec, @@ -272,7 +272,9 @@ pub struct SessionInvestigateResponse { pub retention_lineage: Vec, pub retention_lineage_truncated: bool, pub related_sessions: Vec, + pub related_sessions_truncated: bool, pub external_references: Vec, + pub external_references_truncated: bool, pub source_counts: BTreeMap, pub source_evidence: BTreeMap, pub partial_reasons: Vec, @@ -389,22 +391,22 @@ pub fn app_graph_from_around_response(around: &GraphAroundResponse) -> AppGraphR } pub fn safe_passive_text(input: &str, max_chars: usize) -> String { - let mut out = input + static BEARER_VALUE: std::sync::LazyLock = std::sync::LazyLock::new(|| { + regex::Regex::new(r"(?i)\bBearer\s+[^\s,;]+").expect("static bearer redaction pattern") + }); + let scrubbed = crate::receiver::enrichment::scrub_ai_message(input, None); + let scrubbed = BEARER_VALUE.replace_all(&scrubbed, "[REDACTED]"); + let mut out = crate::assessment::redact_secrets(&scrubbed) .chars() .filter(|ch| !ch.is_control() || matches!(ch, '\n' | '\t')) .collect::(); - for marker in [ - "sk-proj-", - "Bearer ", - "password=", - "token=", - "CORTEX_API_TOKEN=", - ] { - out = out.replace(marker, "[redacted]"); - } if out.chars().count() > max_chars { out = out.chars().take(max_chars).collect::(); out.push_str("..."); } out } + +#[cfg(test)] +#[path = "investigation_tests.rs"] +mod tests; diff --git a/src/app/models/investigation_tests.rs b/src/app/models/investigation_tests.rs new file mode 100644 index 000000000..8a98fd5b5 --- /dev/null +++ b/src/app/models/investigation_tests.rs @@ -0,0 +1,25 @@ +use super::*; + +#[test] +fn passive_text_redacts_secret_values_before_truncation() { + for text in [ + "password=passive-secret-value", + "Authorization: Bearer passive-secret-value", + "Bearer passive-secret-value", + "CORTEX_API_TOKEN=passive-secret-value", + "token=passive-secret-value", + "sk-proj-passive-secret-value-0123456789abcdefgh", + ] { + let output = safe_passive_text(text, 500); + assert!( + !output.contains("passive-secret-value"), + "secret value survived redaction" + ); + let short = safe_passive_text(text, 16); + assert!( + !short.contains("passive-secret"), + "truncation exposed a partial secret" + ); + } + assert_eq!(safe_passive_text("safe\u{1b} text", 500), "safe text"); +} diff --git a/src/app/services.rs b/src/app/services.rs index 522499ab6..fbd238147 100644 --- a/src/app/services.rs +++ b/src/app/services.rs @@ -122,7 +122,10 @@ mod mcp_backfill; mod mcp_events; mod mcp_incidents; mod rag; +mod session_graph_correlation; mod session_investigation; +mod session_investigation_budget; +mod session_investigation_sections; mod session_investigation_support; mod session_pages; mod skill_assessment; diff --git a/src/app/services/ai.rs b/src/app/services/ai.rs index dde6b5f0b..ba55020ba 100644 --- a/src/app/services/ai.rs +++ b/src/app/services/ai.rs @@ -1,68 +1,6 @@ use super::*; -/// Log fan-out cap for the graph-anchored session lane of `ai_correlate`. -/// Clamped again to `[1, 1000]` inside `db::correlate_session_graph`. -const GRAPH_SESSION_LOG_LIMIT: usize = 500; - -/// Shape the DB-layer `SessionGraphInputs` into the API response, classifying -/// each row into a source lane (`agent_command` / `shell_history` / -/// `graph:host:`) and counting the agent-command and shell-history lanes. -/// Heartbeat summaries are filtered to the discovered hosts. Returns `None` when -/// the session has no rows at all (empty bounds). -fn build_graph_session_correlation( - session_id: String, - inputs: db::SessionGraphInputs, - summaries: Vec, -) -> Option { - let (session_start, session_end) = inputs.bounds?; - - let truncated = inputs.logs.len() >= GRAPH_SESSION_LOG_LIMIT; - let mut agent_command_count = 0usize; - let mut shell_history_count = 0usize; - let logs: Vec = inputs - .logs - .into_iter() - .map(|entry| { - let source_kind = row_source_kind(&entry); - let discovery = if entry.source_ip.starts_with("agent-command://") { - agent_command_count += 1; - "agent_command".to_string() - } else if source_kind.as_deref() == Some("shell-history") { - shell_history_count += 1; - "shell_history".to_string() - } else { - format!("graph:host:{}", entry.hostname) - }; - CorrelatedLogRow { - entry: entry.into(), - source_kind, - discovery, - } - }) - .collect(); - - let discovered: std::collections::HashSet<&str> = - inputs.discovered_hosts.iter().map(String::as_str).collect(); - let heartbeat_summaries: Vec = summaries - .into_iter() - .filter(|s| discovered.contains(s.hostname.as_str())) - .collect(); - - Some(GraphSessionCorrelation { - session_id, - session_start, - session_end, - used_graph: inputs.used_graph, - session_entity_keys: inputs.session_entity_keys, - discovered_hosts: inputs.discovered_hosts, - discovered_entities: inputs.discovered_entities, - logs, - agent_command_count, - shell_history_count, - heartbeat_summaries, - truncated, - }) -} +use super::session_graph_correlation::{GRAPH_SESSION_LOG_LIMIT, build_graph_session_correlation}; impl CortexService { pub async fn list_sessions( @@ -74,7 +12,7 @@ impl CortexService { // The unbounded (no time-window) path reads from the periodically // refreshed rollup; expose its staleness so callers know the `as_of`. // Time-windowed queries run live, so no staleness applies. - let unbounded = from.is_none() && to.is_none(); + let unbounded = from.is_none() && to.is_none() && req.session_id.is_none(); let params = db::ListAiSessionsParams { ai_project: req.project, ai_tool: req.tool, diff --git a/src/app/services/ai_correlate_tests.rs b/src/app/services/ai_correlate_tests.rs index 5de3ff03f..f0b347cff 100644 --- a/src/app/services/ai_correlate_tests.rs +++ b/src/app/services/ai_correlate_tests.rs @@ -63,6 +63,7 @@ fn build_graph_session_correlation_classifies_lanes_and_filters_heartbeats() { discovered_hosts: vec!["devhost".into()], discovered_entities: vec!["devhost".into(), "cortex".into()], used_graph: true, + source_fields_truncated: false, logs: vec![ db_log( "agent-command://devhost/claude/s1", diff --git a/src/app/services/session_graph_correlation.rs b/src/app/services/session_graph_correlation.rs new file mode 100644 index 000000000..3bf810ffd --- /dev/null +++ b/src/app/services/session_graph_correlation.rs @@ -0,0 +1,103 @@ +use super::*; + +/// Log fan-out cap for the graph-anchored session lane of `ai_correlate`. +/// Clamped again to `[1, 1000]` inside `db::correlate_session_graph`. +pub(super) const GRAPH_SESSION_LOG_LIMIT: usize = 500; + +/// Shape the DB-layer `SessionGraphInputs` into the API response, classifying +/// each row into a source lane (`agent_command` / `shell_history` / +/// `graph:host:`) and counting the agent-command and shell-history lanes. +/// Heartbeat summaries are filtered to the discovered hosts. Returns `None` when +/// the session has no rows at all (empty bounds). +pub(super) fn build_graph_session_correlation( + session_id: String, + inputs: db::SessionGraphInputs, + summaries: Vec, +) -> Option { + let (session_start, session_end) = inputs.bounds?; + + let truncated = inputs.source_fields_truncated || inputs.logs.len() >= GRAPH_SESSION_LOG_LIMIT; + let mut agent_command_count = 0usize; + let mut shell_history_count = 0usize; + let logs: Vec = inputs + .logs + .into_iter() + .map(|entry| { + let source_kind = row_source_kind(&entry); + let discovery = if entry.source_ip.starts_with("agent-command://") { + agent_command_count += 1; + "agent_command".to_string() + } else if source_kind.as_deref() == Some("shell-history") { + shell_history_count += 1; + "shell_history".to_string() + } else { + format!("graph:host:{}", entry.hostname) + }; + CorrelatedLogRow { + entry: entry.into(), + source_kind, + discovery, + } + }) + .collect(); + + let discovered: std::collections::HashSet<&str> = + inputs.discovered_hosts.iter().map(String::as_str).collect(); + let heartbeat_summaries: Vec = summaries + .into_iter() + .filter(|s| discovered.contains(s.hostname.as_str())) + .collect(); + + Some(GraphSessionCorrelation { + session_id, + session_start, + session_end, + used_graph: inputs.used_graph, + session_entity_keys: inputs.session_entity_keys, + discovered_hosts: inputs.discovered_hosts, + discovered_entities: inputs.discovered_entities, + logs, + agent_command_count, + shell_history_count, + heartbeat_summaries, + truncated, + }) +} + +impl CortexService { + pub(super) async fn session_graph_correlation( + &self, + session: &AiSessionEntry, + severity_min: Option<&str>, + ) -> ServiceResult> { + let levels = severity_at_or_above(severity_min.unwrap_or("info"))?; + let session_id = session.session_id.clone(); + let sid = session_id.clone(); + let scope = db::SessionGraphScope { + project: session.project.clone(), + tool: session.tool.clone(), + host: session.hostname.clone(), + severity_in: levels, + }; + let (inputs, summaries) = self + .run_db("session_investigation_graph", move |pool| { + let inputs = db::correlate_session_graph_scoped( + pool, + &sid, + &scope, + GRAPH_SESSION_LOG_LIMIT, + )?; + let summaries = match &inputs.bounds { + Some((start, end)) if inputs.used_graph => { + db::heartbeat_window_summaries(pool, start, end, None)? + } + _ => Vec::new(), + }; + Ok((inputs, summaries)) + }) + .await?; + Ok(build_graph_session_correlation( + session_id, inputs, summaries, + )) + } +} diff --git a/src/app/services/session_investigation.rs b/src/app/services/session_investigation.rs index 9efe1a2ba..bc07b274c 100644 --- a/src/app/services/session_investigation.rs +++ b/src/app/services/session_investigation.rs @@ -5,7 +5,7 @@ use super::*; use super::session_investigation_support::{ build_session_source_counts, extract_session_external_references, related_sessions_for_investigation, session_investigation_metadata, - summarize_session_source_evidence, timestamp_inclusive_between, + summarize_session_source_evidence, }; impl CortexService { @@ -13,8 +13,20 @@ impl CortexService { &self, req: models::SessionInvestigateRequest, ) -> ServiceResult> { - let started = Instant::now(); let budget = InvestigationBudget::default(); + super::session_investigation_budget::within_session_budget( + budget.max_wall_time_ms, + self.session_investigate_inner(req), + ) + .await + } + + async fn session_investigate_inner( + &self, + req: models::SessionInvestigateRequest, + ) -> ServiceResult> { + let started = Instant::now(); + let mut budget = InvestigationBudget::default(); let session_id = req.session_id.trim().to_string(); if session_id.is_empty() { return Err(ServiceError::InvalidInput( @@ -22,6 +34,9 @@ impl CortexService { )); } let section_limit = req.limit.unwrap_or(100).clamp(1, 200); + // The composition has multiple independently bounded evidence lanes. + budget.max_log_rows = 1_000 + section_limit; + budget.max_evidence_rows = section_limit * 18 + 512; let sessions = self .list_sessions(ListSessionsRequest { @@ -62,26 +77,15 @@ impl CortexService { }) .await?; - let correlated = self - .correlate_ai_logs(AiCorrelateRequest { - project: Some(session.project.clone()), - tool: Some(session.tool.clone()), - session_id: Some(session.session_id.clone()), - host: Some(session.hostname.clone()), - window_minutes: req.window_minutes, - severity_min: req.severity_min.clone(), - limit: Some(10), - events_per_anchor: Some(section_limit.min(100)), - ..Default::default() - }) + let correlation = self + .session_graph_correlation(&session, req.severity_min.as_deref()) .await?; - - let session_entity_keys = correlated - .graph_correlation + let session_entity_keys = correlation .as_ref() .map(|value| value.session_entity_keys.clone()) .unwrap_or_default(); let session_graph_entity_ambiguous = session_entity_keys.len() > 1; + let mut graph_neighborhood_truncated = false; let graph_neighborhood = match session_entity_keys.as_slice() { [session_key] => { let around = self @@ -96,6 +100,7 @@ impl CortexService { ..Default::default() }) .await?; + graph_neighborhood_truncated = around.metadata.truncated; Some(models::app_graph_from_around_response(&around)) } _ => None, @@ -113,23 +118,12 @@ impl CortexService { }) .await?; - let mut notifications = self - .notifications_recent_checked(models::NotificationsRecentRequest { - limit: Some(i64::from(section_limit.saturating_mul(5).min(500))), - rule_id: None, - since: Some(session.first_seen.clone()), - }) - .await? - .into_iter() - .filter(|firing| { - firing.hostname == session.hostname - && timestamp_inclusive_between( - &firing.fired_at, - &session.first_seen, - &session.last_seen, - ) - }) - .collect::>(); + let mut notifications = super::session_investigation_sections::session_notifications( + self, + &session, + section_limit, + ) + .await?; let notifications_truncated = notifications.len() > section_limit as usize; notifications.truncate(section_limit as usize); @@ -221,150 +215,27 @@ impl CortexService { ..Default::default() }) )?; - let artifact_evidence_truncated = - artifact_by_correlation.truncated || artifact_by_request.truncated; - let mut artifact_evidence_by_id = BTreeMap::new(); - for event in artifact_by_correlation - .events - .into_iter() - .chain(artifact_by_request.events) - { - artifact_evidence_by_id.insert(event.cortex_log_id, event); - } - let artifact_evidence = artifact_evidence_by_id.into_values().collect::>(); + let (artifact_evidence, artifact_evidence_truncated) = + super::session_investigation_sections::merge_artifact_evidence( + artifact_by_correlation, + artifact_by_request, + section_limit as usize, + ); - let observatory_session_id = session.session_id.clone(); - let observatory_tool = session.tool.clone(); - let observatory_host = session.hostname.clone(); - let observatory = self - .run_db("session_investigate_observatory", move |pool| { - let query = db::agent_observatory::AgentRunQuery { - tools: vec![observatory_tool.clone()], - host: Some(observatory_host.clone()), - query: Some(observatory_session_id.clone()), - ..Default::default() - }; - let mut runs = - db::agent_observatory::list_observatory_runs(pool, &query, None, 50, i64::MAX)?; - runs.retain(|run| { - run.native_session_id == observatory_session_id - && run.tool.eq_ignore_ascii_case(&observatory_tool) - && run.hostname == observatory_host - }); - let ambiguous_run = runs.len() > 1; - let Some(run) = runs.first().cloned().filter(|_| !ambiguous_run) else { - return Ok(models::SessionObservatoryEvidence { - runs, - ambiguous_run, - ..Default::default() - }); - }; - let resolved = db::agent_observatory::resolve_observatory_run(pool, &run.run_key)?; - let (run_id, identity) = resolved.ok_or_else(|| { - anyhow::anyhow!( - "Agent Observatory run disappeared during session investigation" - ) - })?; - let parent_run = match run.parent_run_id { - Some(id) => db::agent_observatory::resolve_observatory_run_row(pool, id)?, - None => None, - }; - let previous_run = match run.previous_run_id { - Some(id) => db::agent_observatory::resolve_observatory_run_row(pool, id)?, - None => None, - }; - let mut actors = db::agent_observatory::list_observatory_run_actors( - pool, - run_id, - section_limit as usize + 1, - )?; - let actors_truncated = actors.len() > section_limit as usize; - actors.truncate(section_limit as usize); - let worktree = match run.primary_worktree_id { - Some(id) => db::agent_observatory::resolve_observatory_worktree(pool, id)?, - None => None, - }; - let repository = match worktree.as_ref() { - Some(worktree) => db::agent_observatory::resolve_observatory_repository( - pool, - worktree.repository_id, - )?, - None => None, - }; - let commits = - db::agent_observatory::list_agent_run_attributed_commits(pool, run_id)?; - let mut related_runs = match run.primary_worktree_id { - Some(worktree_id) => db::agent_observatory::list_observatory_runs( - pool, - &db::agent_observatory::AgentRunQuery { - worktree_id: Some(worktree_id), - ..Default::default() - }, - None, - section_limit as usize + 1, - i64::MAX, - )?, - None => Vec::new(), - }; - related_runs.retain(|candidate| candidate.id != run_id); - let related_runs_truncated = related_runs.len() > section_limit as usize; - related_runs.truncate(section_limit as usize); - let events = db::agent_observatory::list_observatory_events( - pool, - &run.run_key, - &db::agent_observatory::AgentEventQuery::default(), - None, - section_limit as usize + 1, - true, - i64::MAX, - )?; - let spans = db::agent_observatory::list_observatory_spans( - pool, - run_id, - &identity, - &db::agent_observatory::TelemetryQuery::default(), - None, - section_limit as usize + 1, - i64::MAX, - )?; - let metrics = db::agent_observatory::list_observatory_metrics( - pool, - run_id, - &identity, - &db::agent_observatory::TelemetryQuery::default(), - None, - section_limit as usize + 1, - i64::MAX, - )?; - Ok(models::SessionObservatoryEvidence { - runs, - ambiguous_run, - parent_run, - previous_run, - actors, - actors_truncated, - related_runs, - related_runs_truncated, - repository, - worktree, - commits, - events_truncated: events.len() > section_limit as usize, - spans_truncated: spans.len() > section_limit as usize, - metrics_truncated: metrics.len() > section_limit as usize, - events: events.into_iter().take(section_limit as usize).collect(), - spans: spans.into_iter().take(section_limit as usize).collect(), - metrics: metrics.into_iter().take(section_limit as usize).collect(), - }) - }) - .await?; + let observatory = super::session_investigation_sections::session_observatory( + self, + &session, + section_limit, + ) + .await?; - let related_sessions = related_sessions_for_investigation(self, &session).await?; + let (related_sessions, related_sessions_truncated) = + related_sessions_for_investigation(self, &session).await?; - let external_references = extract_session_external_references(&transcript_page.events); - let source_evidence = summarize_session_source_evidence( - correlated.graph_correlation.as_ref(), - section_limit as usize, - ); + let (external_references, external_references_truncated) = + extract_session_external_references(&transcript_page.events, section_limit as usize); + let source_evidence = + summarize_session_source_evidence(correlation.as_ref(), section_limit as usize); let graph_counts = graph_neighborhood.as_ref().map(|graph| { ( graph.entities.len(), @@ -372,8 +243,7 @@ impl CortexService { graph.evidence.len(), ) }); - let heartbeat_summary_count = correlated - .graph_correlation + let heartbeat_summary_count = correlation .as_ref() .map_or(0, |value| value.heartbeat_summaries.len()); let source_counts = build_session_source_counts( @@ -391,16 +261,24 @@ impl CortexService { ); let mut partial_reasons = Vec::new(); + if external_references_truncated { + partial_reasons.push("external_references_truncated".to_string()); + } if transcript_page.has_more { partial_reasons.push("transcript_truncated".to_string()); } - if correlated - .graph_correlation - .as_ref() - .is_some_and(|value| value.truncated) - { + if correlation.as_ref().is_some_and(|value| value.truncated) { partial_reasons.push("correlation_truncated".to_string()); } + if graph_neighborhood_truncated { + partial_reasons.push("graph_neighborhood_truncated".to_string()); + } + if related_sessions_truncated { + partial_reasons.push("related_sessions_truncated".to_string()); + } + if observatory.commits_truncated { + partial_reasons.push("observatory_commits_truncated".to_string()); + } if session_graph_entity_ambiguous { partial_reasons.push("session_graph_entity_ambiguous".to_string()); } @@ -447,19 +325,19 @@ impl CortexService { let metadata = session_investigation_metadata( &budget, started, - correlated.graph_correlation.as_ref(), + correlation.as_ref(), graph_neighborhood.is_some(), &partial_reasons, transcript_page.events.len(), observatory.events.len(), ); - Ok(InvestigationEnvelope { + let envelope = InvestigationEnvelope { metadata, result: models::SessionInvestigateResponse { session, transcript: transcript_page.events, transcript_has_more: transcript_page.has_more, - correlation: correlated.graph_correlation, + correlation, graph_neighborhood, skill_events: skill_events.events, skill_events_truncated: skill_events.truncated, @@ -476,12 +354,34 @@ impl CortexService { retention_lineage, retention_lineage_truncated, related_sessions, + related_sessions_truncated, external_references, + external_references_truncated, source_counts, source_evidence, partial_reasons, }, + }; + self.run_db("session_investigate_finalize", move |_| { + let mut envelope = + super::session_investigation_budget::finalize_session_envelope(envelope) + .map_err(anyhow::Error::from)?; + envelope.metadata.budget_used.wall_time_ms = + started.elapsed().as_millis().min(u32::MAX as u128) as u32; + if started.elapsed() + > std::time::Duration::from_millis(u64::from( + envelope.metadata.budget.max_wall_time_ms, + )) + { + return Err(anyhow::Error::from(ServiceError::Busy( + "session_investigation_wall_time_budget_exceeded".into(), + ))); + } + // Updating elapsed usage can change the escaped envelope byte count. + super::session_investigation_budget::finalize_session_envelope(envelope) + .map_err(anyhow::Error::from) }) + .await } } diff --git a/src/app/services/session_investigation_budget.rs b/src/app/services/session_investigation_budget.rs new file mode 100644 index 000000000..b5945af86 --- /dev/null +++ b/src/app/services/session_investigation_budget.rs @@ -0,0 +1,454 @@ +use super::session_investigation_support::{ + build_session_source_counts, summarize_session_source_evidence, +}; +use super::*; + +pub(super) async fn within_session_budget( + wall_time_ms: u32, + work: impl std::future::Future>, +) -> ServiceResult { + let started = std::time::Instant::now(); + let deadline = std::time::Duration::from_millis(u64::from(wall_time_ms)); + let result = tokio::time::timeout(deadline, work).await.map_err(|_| { + ServiceError::Busy("session_investigation_wall_time_budget_exceeded".into()) + })?; + // A future that blocks within a ready poll can overshoot a Tokio timer. + if started.elapsed() > deadline { + return Err(ServiceError::Busy( + "session_investigation_wall_time_budget_exceeded".into(), + )); + } + result +} + +type SessionEnvelope = InvestigationEnvelope; + +/// Bound the complete escaped JSON envelope, including its metadata. Preserve +/// section omission reasons and rebuild counts after removing returned rows. +pub(super) fn finalize_session_envelope( + mut envelope: SessionEnvelope, +) -> ServiceResult { + loop { + refresh_counts(&mut envelope); + // The byte count contributes digits to its own serialized envelope. + // Iterate to a fixed point before comparing with the advertised limit. + loop { + let size = serde_json::to_vec(&envelope) + .map_err(|error| ServiceError::Internal(error.into()))? + .len(); + if envelope.metadata.budget_used.payload_bytes as usize == size { + break; + } + envelope.metadata.budget_used.payload_bytes = u32::try_from(size).unwrap_or(u32::MAX); + } + if envelope.metadata.budget_used.payload_bytes <= envelope.metadata.payload_limit_bytes { + return Ok(envelope); + } + if !prune_largest_section(&mut envelope.result)? { + return Err(ServiceError::Busy( + "session_investigation_identity_exceeds_payload_budget".into(), + )); + } + let reason = "payload_budget_truncated".to_string(); + if !envelope.result.partial_reasons.contains(&reason) { + envelope.result.partial_reasons.push(reason); + } + envelope.metadata.partial = true; + envelope.metadata.truncated = true; + envelope.metadata.partial_reasons = envelope.result.partial_reasons.clone(); + envelope.metadata.truncation_reasons = envelope + .result + .partial_reasons + .iter() + .filter(|reason| reason.contains("truncated")) + .cloned() + .collect(); + } +} + +fn refresh_counts(envelope: &mut SessionEnvelope) { + let result = &mut envelope.result; + let source_id_limit = result + .source_evidence + .values() + .map(|summary| summary.log_ids.len()) + .max() + .unwrap_or(50) + .max(1); + result.source_evidence = + summarize_session_source_evidence(result.correlation.as_ref(), source_id_limit); + result.source_counts = build_session_source_counts( + &result.source_evidence, + &result.observatory, + result.skill_events.len(), + result.mcp_events.len(), + result.hook_events.len(), + result.artifact_evidence.len(), + result.graph_neighborhood.as_ref().map(|graph| { + ( + graph.entities.len(), + graph.relationships.len(), + graph.evidence.len(), + ) + }), + result + .correlation + .as_ref() + .map_or(0, |correlation| correlation.heartbeat_summaries.len()), + result.incident_context.error_logs.len(), + result.notifications.len(), + result.retention_lineage.len(), + ); + result + .source_counts + .insert("transcript".into(), result.transcript.len()); + result + .source_counts + .insert("related_session".into(), result.related_sessions.len()); + result.source_counts.insert( + "external_reference".into(), + result.external_references.len(), + ); + result + .source_counts + .insert("attributed_commit".into(), result.observatory.commits.len()); + result + .source_counts + .insert("observatory_run".into(), result.observatory.runs.len()); + envelope.metadata.budget_used.log_rows = (result.incident_context.error_logs.len() + + result + .correlation + .as_ref() + .map_or(0, |correlation| correlation.logs.len())) + as u32; + let observatory = &result.observatory; + let graph_rows = result.graph_neighborhood.as_ref().map_or(0, |graph| { + graph.entities.len() + graph.relationships.len() + graph.evidence.len() + }); + envelope.metadata.budget_used.evidence_rows = (result.transcript.len() + + result.skill_events.len() + + result.mcp_events.len() + + result.hook_events.len() + + result.artifact_evidence.len() + + result.notifications.len() + + result.retention_lineage.len() + + result.related_sessions.len() + + result.external_references.len() + + result.incident_context.ai_sessions.len() + + observatory.runs.len() + + observatory.actors.len() + + observatory.related_runs.len() + + observatory.commits.len() + + observatory.events.len() + + observatory.spans.len() + + observatory.metrics.len() + + graph_rows + + usize::from(observatory.parent_run.is_some()) + + usize::from(observatory.previous_run.is_some()) + + usize::from(observatory.repository.is_some()) + + usize::from(observatory.worktree.is_some()) + + result + .correlation + .as_ref() + .map_or(0, |correlation| correlation.heartbeat_summaries.len())) + as u32; +} + +fn prune_largest_section(result: &mut models::SessionInvestigateResponse) -> ServiceResult { + let mut largest = (0, 0); + macro_rules! candidate { + ($id:expr, $value:expr, $present:expr) => { + if $present { + let bytes = serde_json::to_vec(&$value) + .map_err(|error| ServiceError::Internal(error.into()))? + .len(); + if bytes > largest.1 { + largest = ($id, bytes); + } + } + }; + } + candidate!(1, result.transcript, !result.transcript.is_empty()); + candidate!(2, result.correlation, result.correlation.is_some()); + candidate!( + 3, + result.graph_neighborhood, + result.graph_neighborhood.is_some() + ); + candidate!(4, result.skill_events, !result.skill_events.is_empty()); + candidate!(5, result.mcp_events, !result.mcp_events.is_empty()); + candidate!(6, result.hook_events, !result.hook_events.is_empty()); + candidate!( + 7, + result.artifact_evidence, + !result.artifact_evidence.is_empty() + ); + candidate!( + 8, + result.observatory.runs, + !result.observatory.runs.is_empty() + ); + candidate!( + 9, + result.observatory.actors, + !result.observatory.actors.is_empty() + ); + candidate!( + 10, + result.observatory.related_runs, + !result.observatory.related_runs.is_empty() + ); + candidate!( + 11, + result.observatory.commits, + !result.observatory.commits.is_empty() + ); + candidate!( + 12, + result.observatory.events, + !result.observatory.events.is_empty() + ); + candidate!( + 13, + result.observatory.spans, + !result.observatory.spans.is_empty() + ); + candidate!( + 14, + result.observatory.metrics, + !result.observatory.metrics.is_empty() + ); + candidate!( + 15, + result.incident_context.error_logs, + !result.incident_context.error_logs.is_empty() + ); + candidate!(16, result.notifications, !result.notifications.is_empty()); + candidate!( + 17, + result.retention_lineage, + !result.retention_lineage.is_empty() + ); + candidate!( + 18, + result.related_sessions, + !result.related_sessions.is_empty() + ); + candidate!( + 19, + result.external_references, + !result.external_references.is_empty() + ); + candidate!( + 20, + result.observatory.parent_run, + result.observatory.parent_run.is_some() + ); + candidate!( + 21, + result.observatory.previous_run, + result.observatory.previous_run.is_some() + ); + candidate!( + 22, + result.observatory.repository, + result.observatory.repository.is_some() + ); + candidate!( + 23, + result.observatory.worktree, + result.observatory.worktree.is_some() + ); + candidate!( + 24, + result.incident_context.ai_sessions, + !result.incident_context.ai_sessions.is_empty() + ); + candidate!( + 25, + result.incident_context.by_app, + !result.incident_context.by_app.is_empty() + ); + candidate!(26, result.session.title, result.session.title.is_some()); + candidate!( + 27, + result.session.transcript_path, + result.session.transcript_path.is_some() + ); + let section = match largest.0 { + 0 => return Ok(false), + 1 => { + result.transcript.truncate(result.transcript.len() / 2); + result.transcript_has_more = true; + "transcript" + } + 2 => { + result.correlation = None; + "correlation" + } + 3 => { + result.graph_neighborhood = None; + "graph_neighborhood" + } + 4 => { + result.skill_events.truncate(result.skill_events.len() / 2); + result.skill_events_truncated = true; + "skill_events" + } + 5 => { + result.mcp_events.truncate(result.mcp_events.len() / 2); + result.mcp_events_truncated = true; + "mcp_events" + } + 6 => { + result.hook_events.truncate(result.hook_events.len() / 2); + result.hook_events_truncated = true; + "hook_events" + } + 7 => { + result + .artifact_evidence + .truncate(result.artifact_evidence.len() / 2); + result.artifact_evidence_truncated = true; + "artifact_evidence" + } + 8 => { + result + .observatory + .runs + .truncate(result.observatory.runs.len() / 2); + "observatory_runs" + } + 9 => { + result + .observatory + .actors + .truncate(result.observatory.actors.len() / 2); + result.observatory.actors_truncated = true; + "observatory_actors" + } + 10 => { + result + .observatory + .related_runs + .truncate(result.observatory.related_runs.len() / 2); + result.observatory.related_runs_truncated = true; + "observatory_related_runs" + } + 11 => { + result + .observatory + .commits + .truncate(result.observatory.commits.len() / 2); + result.observatory.commits_truncated = true; + "observatory_commits" + } + 12 => { + result + .observatory + .events + .truncate(result.observatory.events.len() / 2); + result.observatory.events_truncated = true; + "observatory_events" + } + 13 => { + result + .observatory + .spans + .truncate(result.observatory.spans.len() / 2); + result.observatory.spans_truncated = true; + "observatory_spans" + } + 14 => { + result + .observatory + .metrics + .truncate(result.observatory.metrics.len() / 2); + result.observatory.metrics_truncated = true; + "observatory_metrics" + } + 15 => { + result + .incident_context + .error_logs + .truncate(result.incident_context.error_logs.len() / 2); + result.incident_context.error_logs_truncated = true; + "incident_context_errors" + } + 16 => { + result + .notifications + .truncate(result.notifications.len() / 2); + result.notifications_truncated = true; + "notifications" + } + 17 => { + result + .retention_lineage + .truncate(result.retention_lineage.len() / 2); + result.retention_lineage_truncated = true; + "retention_lineage" + } + 18 => { + result + .related_sessions + .truncate(result.related_sessions.len() / 2); + result.related_sessions_truncated = true; + "related_sessions" + } + 19 => { + result + .external_references + .truncate(result.external_references.len() / 2); + result.external_references_truncated = true; + "external_references" + } + 20 => { + result.observatory.parent_run = None; + "parent_run" + } + 21 => { + result.observatory.previous_run = None; + "previous_run" + } + 22 => { + result.observatory.repository = None; + "repository" + } + 23 => { + result.observatory.worktree = None; + "worktree" + } + 24 => { + result + .incident_context + .ai_sessions + .truncate(result.incident_context.ai_sessions.len() / 2); + "incident_ai_sessions" + } + 25 => { + result + .incident_context + .by_app + .truncate(result.incident_context.by_app.len() / 2); + "incident_apps" + } + 26 => { + result.session.title = None; + "session_title" + } + _ => { + result.session.transcript_path = None; + "transcript_path" + } + }; + let reason = format!("{section}_payload_truncated"); + if !result.partial_reasons.contains(&reason) { + result.partial_reasons.push(reason); + } + Ok(true) +} + +#[cfg(test)] +#[path = "session_investigation_budget_tests.rs"] +mod tests; diff --git a/src/app/services/session_investigation_budget_tests.rs b/src/app/services/session_investigation_budget_tests.rs new file mode 100644 index 000000000..a323b9b1a --- /dev/null +++ b/src/app/services/session_investigation_budget_tests.rs @@ -0,0 +1,145 @@ +use super::*; + +fn minimal_envelope() -> SessionEnvelope { + let session = AiSessionEntry { + session_key: "h:codex:p:s".into(), + project: "p".into(), + tool: "codex".into(), + session_id: "s".into(), + hostname: "h".into(), + transcript_path: None, + first_seen: "2026-09-21T00:00:00Z".into(), + last_seen: "2026-09-21T00:10:00Z".into(), + event_count: 1, + title: None, + title_provenance: None, + }; + SessionEnvelope { + metadata: super::super::session_investigation_support::session_investigation_metadata( + &InvestigationBudget::default(), + std::time::Instant::now(), + None, + false, + &[], + 1, + 0, + ), + result: models::SessionInvestigateResponse { + session, + transcript: vec![], + transcript_has_more: false, + correlation: None, + graph_neighborhood: None, + skill_events: vec![], + skill_events_truncated: false, + mcp_events: vec![], + mcp_events_truncated: false, + hook_events: vec![], + hook_events_truncated: false, + artifact_evidence: vec![], + artifact_evidence_truncated: false, + observatory: Default::default(), + incident_context: IncidentContextResponse { + window_from: "2026-09-21T00:00:00Z".into(), + window_to: "2026-09-21T00:10:00Z".into(), + total_logs: 0, + by_severity: vec![], + by_app: vec![], + error_logs: vec![], + error_logs_truncated: false, + ai_sessions: vec![], + }, + notifications: vec![], + notifications_truncated: false, + retention_lineage: vec![], + retention_lineage_truncated: false, + related_sessions: vec![], + related_sessions_truncated: false, + external_references: vec![], + external_references_truncated: false, + source_counts: Default::default(), + source_evidence: Default::default(), + partial_reasons: vec![], + }, + } +} + +#[test] +fn escaped_payload_budget_includes_metadata_and_refreshes_counts() { + let mut envelope = minimal_envelope(); + envelope + .result + .transcript + .push(models::RenderedSessionEvent { + position: 1, + timestamp: "2026-09-21T00:00:00Z".into(), + kind: models::RenderedSessionEventKind::Assistant, + text: "\"🦀".repeat(40_000), + redacted: false, + parse_warning: None, + }); + envelope.result.source_counts.insert("transcript".into(), 1); + let result = finalize_session_envelope(envelope).unwrap(); + let size = serde_json::to_vec(&result).unwrap().len(); + assert!(size <= result.metadata.payload_limit_bytes as usize); + assert_eq!(size, result.metadata.budget_used.payload_bytes as usize); + assert!(result.metadata.partial && result.metadata.truncated); + assert!(result.result.transcript_has_more); + assert_eq!( + result.result.source_counts["transcript"], + result.result.transcript.len() + ); + assert!( + result + .result + .partial_reasons + .iter() + .any(|reason| reason == "transcript_payload_truncated") + ); +} + +#[test] +fn oversized_required_identity_fails_closed() { + let mut envelope = minimal_envelope(); + envelope.result.session.project = "p".repeat(100_000); + assert!(matches!( + finalize_session_envelope(envelope), + Err(ServiceError::Busy(_)) + )); +} + +#[test] +fn ambiguity_is_partial_without_claiming_truncation() { + let reasons = vec!["observatory_run_ambiguous".to_string()]; + let metadata = super::super::session_investigation_support::session_investigation_metadata( + &InvestigationBudget::default(), + std::time::Instant::now(), + None, + false, + &reasons, + 0, + 0, + ); + assert!(metadata.partial); + assert!(!metadata.truncated); + assert!(metadata.truncation_reasons.is_empty()); + assert_eq!(metadata.auth_state, "unknown"); +} + +#[tokio::test] +async fn elapsed_budget_returns_retryable_busy() { + let result: ServiceResult<()> = within_session_budget(1, std::future::pending()).await; + assert!( + matches!(result, Err(ServiceError::Busy(message)) if message == "session_investigation_wall_time_budget_exceeded") + ); +} + +#[tokio::test] +async fn ready_future_that_overshoots_wall_budget_fails_closed() { + let result = within_session_budget(1, async { + std::thread::sleep(std::time::Duration::from_millis(5)); + Ok(()) + }) + .await; + assert!(matches!(result, Err(ServiceError::Busy(_)))); +} diff --git a/src/app/services/session_investigation_sections.rs b/src/app/services/session_investigation_sections.rs new file mode 100644 index 000000000..d753927cb --- /dev/null +++ b/src/app/services/session_investigation_sections.rs @@ -0,0 +1,188 @@ +use super::*; + +pub(super) async fn session_notifications( + service: &CortexService, + session: &AiSessionEntry, + limit: u32, +) -> ServiceResult> { + let host = session.hostname.clone(); + let start = session.first_seen.clone(); + let end = session.last_seen.clone(); + service + .run_db("session_investigate_notifications", move |pool| { + let conn = pool.get()?; + let mut stmt = conn.prepare( + "SELECT id,outbox_id,rule_id,hostname,fired_at,status_code + FROM notification_firings + WHERE hostname=?1 AND julianday(fired_at)>=julianday(?2) + AND julianday(fired_at)<=julianday(?3) + ORDER BY fired_at DESC,id DESC LIMIT ?4", + )?; + Ok(stmt + .query_map( + rusqlite::params![host, start, end, i64::from(limit) + 1], + |row| { + Ok(db::notifications::FiringRow { + id: row.get(0)?, + outbox_id: row.get(1)?, + rule_id: row.get(2)?, + hostname: row.get(3)?, + fired_at: row.get(4)?, + status_code: row.get(5)?, + }) + }, + )? + .collect::>>()?) + }) + .await +} + +pub(super) fn merge_artifact_evidence( + correlation: models::ListArtifactEvidenceResponse, + request: models::ListArtifactEvidenceResponse, + limit: usize, +) -> (Vec, bool) { + let source_truncated = correlation.truncated || request.truncated; + let mut events = BTreeMap::new(); + for event in correlation.events.into_iter().chain(request.events) { + events.insert(event.cortex_log_id, event); + } + let truncated = source_truncated || events.len() > limit; + (events.into_values().rev().take(limit).collect(), truncated) +} + +pub(super) async fn session_observatory( + service: &CortexService, + session: &AiSessionEntry, + section_limit: u32, +) -> ServiceResult { + let observatory_session_id = session.session_id.clone(); + let observatory_tool = session.tool.clone(); + let observatory_host = session.hostname.clone(); + service + .run_db("session_investigate_observatory", move |pool| { + let query = db::agent_observatory::AgentRunQuery { + tools: vec![observatory_tool.clone()], + host: Some(observatory_host.clone()), + native_session_id: Some(observatory_session_id.clone()), + ..Default::default() + }; + let mut runs = + db::agent_observatory::list_observatory_runs(pool, &query, None, 50, i64::MAX)?; + runs.retain(|run| { + run.native_session_id == observatory_session_id + && run.tool.eq_ignore_ascii_case(&observatory_tool) + && run.hostname == observatory_host + }); + let ambiguous_run = runs.len() > 1; + let Some(run) = runs.first().cloned().filter(|_| !ambiguous_run) else { + return Ok(models::SessionObservatoryEvidence { + runs, + ambiguous_run, + ..Default::default() + }); + }; + let resolved = db::agent_observatory::resolve_observatory_run(pool, &run.run_key)?; + let (run_id, identity) = resolved.ok_or_else(|| { + anyhow::anyhow!("Agent Observatory run disappeared during session investigation") + })?; + let parent_run = match run.parent_run_id { + Some(id) => db::agent_observatory::resolve_observatory_run_row(pool, id)?, + None => None, + }; + let previous_run = match run.previous_run_id { + Some(id) => db::agent_observatory::resolve_observatory_run_row(pool, id)?, + None => None, + }; + let mut actors = db::agent_observatory::list_observatory_run_actors( + pool, + run_id, + section_limit as usize, + )?; + let actors_truncated = actors.len() > section_limit as usize; + actors.truncate(section_limit as usize); + let worktree = match run.primary_worktree_id { + Some(id) => db::agent_observatory::resolve_observatory_worktree(pool, id)?, + None => None, + }; + let repository = match worktree.as_ref() { + Some(worktree) => db::agent_observatory::resolve_observatory_repository( + pool, + worktree.repository_id, + )?, + None => None, + }; + let mut commits = db::agent_observatory::list_agent_run_attributed_commits( + pool, + run_id, + section_limit as usize, + )?; + let commits_truncated = commits.len() > section_limit as usize; + commits.truncate(section_limit as usize); + let mut related_runs = match run.primary_worktree_id { + Some(worktree_id) => db::agent_observatory::list_observatory_runs( + pool, + &db::agent_observatory::AgentRunQuery { + worktree_id: Some(worktree_id), + exclude_run_id: Some(run_id), + ..Default::default() + }, + None, + section_limit as usize, + i64::MAX, + )?, + None => Vec::new(), + }; + related_runs.retain(|candidate| candidate.id != run_id); + let related_runs_truncated = related_runs.len() > section_limit as usize; + related_runs.truncate(section_limit as usize); + let events = db::agent_observatory::list_observatory_events( + pool, + &run.run_key, + &db::agent_observatory::AgentEventQuery::default(), + None, + section_limit as usize, + true, + i64::MAX, + )?; + let spans = db::agent_observatory::list_observatory_spans( + pool, + run_id, + &identity, + &db::agent_observatory::TelemetryQuery::default(), + None, + section_limit as usize, + i64::MAX, + )?; + let metrics = db::agent_observatory::list_observatory_metrics( + pool, + run_id, + &identity, + &db::agent_observatory::TelemetryQuery::default(), + None, + section_limit as usize, + i64::MAX, + )?; + Ok(models::SessionObservatoryEvidence { + runs, + ambiguous_run, + parent_run, + previous_run, + actors, + actors_truncated, + related_runs, + related_runs_truncated, + repository, + worktree, + commits, + commits_truncated, + events_truncated: events.len() > section_limit as usize, + spans_truncated: spans.len() > section_limit as usize, + metrics_truncated: metrics.len() > section_limit as usize, + events: events.into_iter().take(section_limit as usize).collect(), + spans: spans.into_iter().take(section_limit as usize).collect(), + metrics: metrics.into_iter().take(section_limit as usize).collect(), + }) + }) + .await +} diff --git a/src/app/services/session_investigation_support.rs b/src/app/services/session_investigation_support.rs index 2c8001fbc..42ea8970e 100644 --- a/src/app/services/session_investigation_support.rs +++ b/src/app/services/session_investigation_support.rs @@ -5,8 +5,8 @@ use super::*; pub(super) async fn related_sessions_for_investigation( service: &CortexService, session: &AiSessionEntry, -) -> ServiceResult> { - Ok(service +) -> ServiceResult<(Vec, bool)> { + let mut sessions = service .list_sessions(ListSessionsRequest { project: Some(session.project.clone()), tool: None, @@ -14,14 +14,16 @@ pub(super) async fn related_sessions_for_investigation( host: None, since: Some(session.first_seen.clone()), until: Some(session.last_seen.clone()), - limit: Some(50), + limit: Some(22), }) .await? .sessions .into_iter() .filter(|candidate| candidate.session_key != session.session_key) - .take(20) - .collect()) + .collect::>(); + let truncated = sessions.len() > 20; + sessions.truncate(20); + Ok((sessions, truncated)) } #[allow(clippy::too_many_arguments)] @@ -122,11 +124,17 @@ pub(super) fn session_investigation_metadata( .filter(|value| !value.used_graph) .map(|_| vec!["session_graph_entity_unavailable".to_string()]) .unwrap_or_default(), - truncated: partial, - truncation_reasons: partial_reasons.to_vec(), + truncated: partial_reasons + .iter() + .any(|reason| reason.contains("truncated")), + truncation_reasons: partial_reasons + .iter() + .filter(|reason| reason.contains("truncated")) + .cloned() + .collect(), partial, partial_reasons: partial_reasons.to_vec(), - auth_state: "bearer".to_string(), + auth_state: "unknown".to_string(), budget: budget.clone(), budget_used: InvestigationBudgetUsed { graph_calls, @@ -168,23 +176,11 @@ pub(super) fn summarize_session_source_evidence( summaries } -pub(super) fn timestamp_inclusive_between(value: &str, start: &str, end: &str) -> bool { - let Ok(value) = chrono::DateTime::parse_from_rfc3339(value) else { - return false; - }; - let Ok(start) = chrono::DateTime::parse_from_rfc3339(start) else { - return false; - }; - let Ok(end) = chrono::DateTime::parse_from_rfc3339(end) else { - return false; - }; - value >= start && value <= end -} - pub(super) fn extract_session_external_references( events: &[models::RenderedSessionEvent], -) -> Vec { - let mut references = std::collections::BTreeSet::new(); + limit: usize, +) -> (Vec, bool) { + let mut references = BTreeMap::new(); for event in events { for raw in event.text.split_whitespace() { let token = raw.trim_matches(|ch: char| { @@ -215,17 +211,22 @@ pub(super) fn extract_session_external_references( } else { continue; }; - references.insert(models::SessionExternalReference { - kind, - value: safe_passive_text(token, 500), - source_position: Some(event.position), - evidence_kind: "transcript_text".to_string(), - trust_level: "claimed".to_string(), - verified: false, - }); + references + .entry((kind.clone(), safe_passive_text(token, 500))) + .or_insert(models::SessionExternalReference { + kind, + value: safe_passive_text(token, 500), + source_position: Some(event.position), + evidence_kind: "transcript_text".to_string(), + trust_level: "claimed".to_string(), + verified: false, + }); + if references.len() > limit { + return (references.into_values().take(limit).collect(), true); + } } } - references.into_iter().collect() + (references.into_values().collect(), false) } pub(super) fn looks_like_linear_identifier(token: &str) -> bool { diff --git a/src/app/services/session_investigation_tests.rs b/src/app/services/session_investigation_tests.rs index 957292072..8fcdc1c33 100644 --- a/src/app/services/session_investigation_tests.rs +++ b/src/app/services/session_investigation_tests.rs @@ -25,7 +25,7 @@ fn external_references_are_classified_and_deduplicated() { ), ]; - let refs = extract_session_external_references(&events); + let (refs, _) = extract_session_external_references(&events, 100); assert!(refs.iter().any(|item| { item.kind == models::SessionExternalReferenceKind::LinearIssue && item.value == "CLD-1149" @@ -110,22 +110,120 @@ fn source_evidence_groups_source_kinds_and_bounds_log_ids() { assert_eq!(summaries["unknown"].log_ids, vec![4]); } +fn fixture_service() -> (tempfile::TempDir, CortexService) { + let dir = tempfile::tempdir().unwrap(); + let storage = crate::config::StorageConfig::for_test(dir.path().join("investigate.db")); + let pool = std::sync::Arc::new(db::init_pool(&storage).unwrap()); + (dir, CortexService::new(pool, storage)) +} + +fn fixture_session() -> AiSessionEntry { + db::AiSessionEntry { + ai_project: "p".into(), + ai_tool: "codex".into(), + ai_session_id: "s".into(), + ai_transcript_path: None, + hostname: "h".into(), + first_seen: "2026-09-21T00:00:00Z".into(), + last_seen: "2026-09-21T00:10:00Z".into(), + event_count: 1, + title: None, + title_provenance: None, + } + .into() +} + +#[tokio::test] +async fn notification_identity_and_end_window_are_applied_before_limit() { + let (_dir, service) = fixture_service(); + let conn = service.pool.get().unwrap(); + for (host, time) in [ + ("h", "2026-09-21T00:00:00Z"), + ("h", "2026-09-21T00:05:00Z"), + ("h", "2026-09-21T00:10:00Z"), + ("h", "2026-09-21T01:00:00Z"), + ] { + conn.execute("INSERT INTO notification_firings(outbox_id,rule_id,severity,hostname,fired_at) VALUES(1,'r','err',?1,?2)", rusqlite::params![host,time]).unwrap(); + } + for _ in 0..501 { + conn.execute("INSERT INTO notification_firings(outbox_id,rule_id,severity,hostname,fired_at) VALUES(1,'r','err','other','2026-09-21T00:09:00Z')", []).unwrap(); + } + drop(conn); + let rows = super::super::session_investigation_sections::session_notifications( + &service, + &fixture_session(), + 2, + ) + .await + .unwrap(); + assert_eq!( + rows.len(), + 3, + "sentinel detects truncation among matching rows" + ); + assert!(rows.iter().all(|row| row.hostname == "h")); + assert!( + rows.iter() + .all(|row| row.fired_at.as_str() <= "2026-09-21T00:10:00Z") + ); +} + #[test] -fn timestamp_window_is_inclusive_and_rejects_invalid_values() { - let start = "2026-09-21T00:00:00Z"; - let end = "2026-09-21T00:10:00Z"; +fn artifact_union_is_deduplicated_bounded_and_signals_overflow() { + let entry = |id| models::ArtifactEvidenceEntry { + cortex_log_id: id, + event: serde_json::from_value(serde_json::json!({ + "schemaVersion": "dinglebear.cortex-artifact-evidence/v1", "eventId": format!("e-{id}"), + "eventKind": "installed", "sourceSystem": "test", "sourceIssuer": "fixture", + "observedAt": "2026-09-21T00:05:00Z" + })) + .unwrap(), + }; + let (events, truncated) = super::super::session_investigation_sections::merge_artifact_evidence( + models::ListArtifactEvidenceResponse { + events: vec![entry(1), entry(2)], + truncated: false, + }, + models::ListArtifactEvidenceResponse { + events: vec![entry(2), entry(3)], + truncated: false, + }, + 2, + ); + assert_eq!( + events + .iter() + .map(|item| item.cortex_log_id) + .collect::>(), + vec![3, 2] + ); + assert!(truncated); +} - assert!(timestamp_inclusive_between(start, start, end)); - assert!(timestamp_inclusive_between(end, start, end)); - assert!(timestamp_inclusive_between( - "2026-09-21T00:05:00Z", - start, - end - )); - assert!(!timestamp_inclusive_between( - "2026-09-21T00:11:00Z", - start, - end - )); - assert!(!timestamp_inclusive_between("not-a-time", start, end)); +#[test] +fn external_references_deduplicate_across_positions_and_cap_unique_values() { + let events = vec![event(1, "#123 #456"), event(2, "#123 #789")]; + let (refs, truncated) = extract_session_external_references(&events, 2); + assert!(truncated); + assert_eq!(refs.len(), 2); + assert_eq!(refs.iter().filter(|item| item.value == "#123").count(), 1); + assert!( + refs.iter() + .all(|item| !item.verified && item.trust_level == "claimed") + ); +} + +#[tokio::test] +async fn related_sessions_signal_truncation() { + let (_dir, service) = fixture_service(); + let conn = service.pool.get().unwrap(); + for id in 0..23 { + conn.execute("INSERT INTO logs(timestamp,hostname,severity,message,raw,source_ip,ai_tool,ai_project,ai_session_id) VALUES('2026-09-21T00:05:00Z','h','info','message','','fixture','codex','p',?1)", [format!("session-{id}")]).unwrap(); + } + drop(conn); + let (rows, truncated) = related_sessions_for_investigation(&service, &fixture_session()) + .await + .unwrap(); + assert_eq!(rows.len(), 20); + assert!(truncated); } diff --git a/src/db.rs b/src/db.rs index 0a0435a6c..67a2b4de4 100644 --- a/src/db.rs +++ b/src/db.rs @@ -130,12 +130,13 @@ pub use pool::{ }; pub(crate) use pool::{WriteConnBusy, is_pool_acquire_failure, try_write_conn_for}; pub use queries::{ - RollupRefresh, SEVERITY_LEVELS, ai_session_rollup_status, correlate_session_graph, - durable_stream_page, get_error_summary, get_stats, incident_context_summary, - investigate_ai_incidents, list_ai_projects, list_ai_sessions, list_ai_tools, list_hosts, - prune_expired_stream_lineage, prune_timeline_rollup, refresh_ai_session_rollup_if_stale, - refresh_timeline_rollup, rendered_session_page, search_ai_abuse, search_ai_anchors, - search_ai_incidents, search_ai_related_logs, search_ai_sessions, search_logs, severity_to_num, + RollupRefresh, SEVERITY_LEVELS, SessionGraphScope, ai_session_rollup_status, + correlate_session_graph, correlate_session_graph_scoped, durable_stream_page, + get_error_summary, get_stats, incident_context_summary, investigate_ai_incidents, + list_ai_projects, list_ai_sessions, list_ai_tools, list_hosts, prune_expired_stream_lineage, + prune_timeline_rollup, refresh_ai_session_rollup_if_stale, refresh_timeline_rollup, + rendered_session_page, search_ai_abuse, search_ai_anchors, search_ai_incidents, + search_ai_related_logs, search_ai_sessions, search_logs, severity_to_num, similar_incidents_clusters, tail_logs, timeline_rollup_status, topic_correlate_inputs, }; pub(crate) use skill_events::insert_skill_events_in_tx; diff --git a/src/db/agent_observatory_read.rs b/src/db/agent_observatory_read.rs index ca8bf6164..eb320caca 100644 --- a/src/db/agent_observatory_read.rs +++ b/src/db/agent_observatory_read.rs @@ -303,6 +303,17 @@ pub fn list_observatory_runs( let mut values = Vec::new(); let mut sql="SELECT DISTINCT a.id,a.run_key,a.native_session_id,a.tool,a.provider_tool,a.hostname,a.parent_run_id,a.previous_run_id,a.status,a.status_reason,a.status_observed_at,a.started_at,a.last_activity_at,a.ended_at,a.transcript_path,a.primary_worktree_id,a.primary_branch,a.start_head_sha,a.current_head_sha,a.event_count,a.error_count,a.freshness_json FROM agent_runs a WHERE 1=1".to_string(); push_filter(&mut sql, &mut values, "a.id <= ?", high_water); + if let Some(session_id) = &q.native_session_id { + push_filter( + &mut sql, + &mut values, + "a.native_session_id = ?", + session_id.clone(), + ); + } + if let Some(run_id) = q.exclude_run_id { + push_filter(&mut sql, &mut values, "a.id != ?", run_id); + } if let Some(id) = q.worktree_id { push_filter( &mut sql, @@ -333,11 +344,22 @@ pub fn list_observatory_runs( values.extend(q.statuses.iter().cloned().map(Value::from)); } if !q.tools.is_empty() { + let tool_column = if q.native_session_id.is_some() { + "lower(a.tool)" + } else { + "a.tool" + }; sql.push_str(&format!( - " AND a.tool IN ({})", + " AND {tool_column} IN ({})", vec!["?"; q.tools.len()].join(",") )); - values.extend(q.tools.iter().cloned().map(Value::from)); + values.extend(q.tools.iter().map(|tool| { + Value::from(if q.native_session_id.is_some() { + tool.to_ascii_lowercase() + } else { + tool.clone() + }) + })); } if q.active_only { sql.push_str(" AND a.status IN ('starting','active','waiting','idle')"); diff --git a/src/db/agent_observatory_read_models.rs b/src/db/agent_observatory_read_models.rs index 0cb195a59..f04b24686 100644 --- a/src/db/agent_observatory_read_models.rs +++ b/src/db/agent_observatory_read_models.rs @@ -27,6 +27,8 @@ pub struct RepositoryQuery { } #[derive(Debug, Clone, Default, Serialize)] pub struct AgentRunQuery { + pub native_session_id: Option, + pub exclude_run_id: Option, pub repository_id: Option, pub worktree_id: Option, pub branch: Option, diff --git a/src/db/agent_observatory_read_tests.rs b/src/db/agent_observatory_read_tests.rs index 213b9c3a1..53cae81f5 100644 --- a/src/db/agent_observatory_read_tests.rs +++ b/src/db/agent_observatory_read_tests.rs @@ -363,3 +363,56 @@ fn run_reads_expose_lineage_and_bounded_actors() { ); assert_eq!(actors[0].native_actor_id, "subagent-b"); } + +#[test] +fn exact_native_identity_is_filtered_before_run_limit() { + let dir = tempfile::tempdir().unwrap(); + let pool = init_pool(&StorageConfig::for_test(dir.path().join("exact.db"))).unwrap(); + let conn = pool.get().unwrap(); + conn.execute("INSERT INTO agent_runs(run_key,native_session_id,tool,hostname,status,status_observed_at,started_at,last_activity_at) VALUES('exact','s','codex','h','completed','2026-09-21T00:00:00Z','2026-09-21T00:00:00Z','2026-09-21T00:00:00Z')", []).unwrap(); + for id in 0..60 { + conn.execute("INSERT INTO agent_runs(run_key,native_session_id,tool,hostname,status,status_observed_at,started_at,last_activity_at) VALUES(?1,?1,'codex','h','active','2026-09-21T01:00:00Z','2026-09-21T01:00:00Z','2026-09-21T01:00:00Z')", [format!("prefix-s-{id}")]).unwrap(); + } + drop(conn); + let query = AgentRunQuery { + native_session_id: Some("s".into()), + tools: vec!["Codex".into()], + host: Some("h".into()), + ..Default::default() + }; + let rows = list_observatory_runs(&pool, &query, None, 1, i64::MAX).unwrap(); + assert_eq!(rows.len(), 1); + assert_eq!(rows[0].run_key, "exact"); + pool.get().unwrap().execute("INSERT INTO agent_runs(run_key,native_session_id,tool,hostname,status,status_observed_at,started_at,last_activity_at) VALUES('case-variant','s','CODEX','h','active','2026-09-21T01:00:00Z','2026-09-21T01:00:00Z','2026-09-21T01:00:00Z')", []).unwrap(); + assert_eq!( + list_observatory_runs(&pool, &query, None, 1, i64::MAX) + .unwrap() + .len(), + 2, + "sentinel retains exact ambiguity" + ); +} + +#[test] +fn attributed_commit_join_is_bounded_with_sentinel_and_preserves_fields() { + let dir = tempfile::tempdir().unwrap(); + let pool = init_pool(&StorageConfig::for_test(dir.path().join("commits.db"))).unwrap(); + let conn = pool.get().unwrap(); + conn.execute("INSERT INTO repositories(repository_key,hostname,common_git_dir,primary_path,display_name,first_seen_at,last_seen_at) VALUES('repo','h','/git','/p','p','2026-09-21T00:00:00Z','2026-09-21T00:00:00Z')", []).unwrap(); + let repository_id = conn.last_insert_rowid(); + conn.execute("INSERT INTO agent_runs(run_key,native_session_id,tool,hostname,status,status_observed_at,started_at,last_activity_at) VALUES('run','s','codex','h','active','2026-09-21T00:00:00Z','2026-09-21T00:00:00Z','2026-09-21T00:00:00Z')", []).unwrap(); + let run_id = conn.last_insert_rowid(); + for id in 0..3 { + conn.execute("INSERT INTO git_commits(repository_id,sha,subject,first_observed_at,last_observed_at) VALUES(?1,?2,?3,'2026-09-21T00:00:00Z','2026-09-21T00:00:00Z')", rusqlite::params![repository_id,format!("sha-{id}"),format!("subject-{id}")]).unwrap(); + let commit_id = conn.last_insert_rowid(); + conn.execute("INSERT INTO agent_run_commits(relation_key,run_id,commit_id,evidence_kind,evidence_source,trust_level,confidence,first_seen_at,last_seen_at) VALUES(?1,?2,?3,'attributed','test','claimed',0.5,'2026-09-21T00:00:00Z','2026-09-21T00:00:00Z')", rusqlite::params![format!("relation-{id}"),run_id,commit_id]).unwrap(); + } + drop(conn); + let commits = + crate::db::agent_observatory::list_agent_run_attributed_commits(&pool, run_id, 1).unwrap(); + assert_eq!(commits.len(), 2); + assert_eq!(commits[0].commit.sha, "sha-0"); + assert_eq!(commits[0].commit.subject, "subject-0"); + assert_eq!(commits[0].relation.commit_id, commits[0].commit.id); + assert_eq!(commits[0].relation.confidence, 0.5); +} diff --git a/src/db/agent_observatory_run_commits.rs b/src/db/agent_observatory_run_commits.rs index 006830345..3e58f8f86 100644 --- a/src/db/agent_observatory_run_commits.rs +++ b/src/db/agent_observatory_run_commits.rs @@ -182,6 +182,7 @@ pub fn upsert_agent_run_commit( Ok(relation) } +#[cfg(test)] pub fn list_agent_run_commits(pool: &DbPool, run_id: i64) -> Result> { if run_id <= 0 { bail!("run_id must be positive"); @@ -198,44 +199,48 @@ pub fn list_agent_run_commits(pool: &DbPool, run_id: i64) -> Result Result> { - let relations = list_agent_run_commits(pool, run_id)?; + if run_id <= 0 { + bail!("run_id must be positive"); + } let connection = pool.get().context("acquire database connection")?; let mut statement = connection.prepare( - "SELECT id, repository_id, sha, parent_shas_json, author_name, author_email_hash, - authored_at, committed_at, subject, changed_files, insertions, deletions, - changed_paths_json, first_observed_at, last_observed_at, reachable, metadata_json - FROM git_commits WHERE id = ?1", + "SELECT r.id,r.relation_key,r.run_id,r.commit_id,r.worktree_id,r.evidence_kind, + r.evidence_source,r.trust_level,r.confidence,r.first_seen_at,r.last_seen_at,r.metadata_json, + c.id,c.repository_id,c.sha,c.parent_shas_json,c.author_name,c.author_email_hash, + c.authored_at,c.committed_at,c.subject,c.changed_files,c.insertions,c.deletions, + c.changed_paths_json,c.first_observed_at,c.last_observed_at,c.reachable,c.metadata_json + FROM agent_run_commits r JOIN git_commits c ON c.id=r.commit_id + WHERE r.run_id=?1 ORDER BY r.first_seen_at,r.id LIMIT ?2", )?; - relations - .into_iter() - .map(|relation| { - let commit = statement - .query_row([relation.commit_id], |row| { - Ok(super::GitCommitRow { - id: row.get(0)?, - repository_id: row.get(1)?, - sha: row.get(2)?, - parent_shas_json: row.get(3)?, - author_name: row.get(4)?, - author_email_hash: row.get(5)?, - authored_at: row.get(6)?, - committed_at: row.get(7)?, - subject: row.get(8)?, - changed_files: row.get(9)?, - insertions: row.get(10)?, - deletions: row.get(11)?, - changed_paths_json: row.get(12)?, - first_observed_at: row.get(13)?, - last_observed_at: row.get(14)?, - reachable: row.get(15)?, - metadata_json: row.get(16)?, - }) - }) - .context("resolve attributed git commit")?; - Ok(AgentRunAttributedCommit { relation, commit }) - }) - .collect() + statement + .query_map(params![run_id, (limit.clamp(1, 200) + 1) as i64], |item| { + Ok(AgentRunAttributedCommit { + relation: row(item)?, + commit: super::GitCommitRow { + id: item.get(12)?, + repository_id: item.get(13)?, + sha: item.get(14)?, + parent_shas_json: item.get(15)?, + author_name: item.get(16)?, + author_email_hash: item.get(17)?, + authored_at: item.get(18)?, + committed_at: item.get(19)?, + subject: item.get(20)?, + changed_files: item.get(21)?, + insertions: item.get(22)?, + deletions: item.get(23)?, + changed_paths_json: item.get(24)?, + first_observed_at: item.get(25)?, + last_observed_at: item.get(26)?, + reachable: item.get(27)?, + metadata_json: item.get(28)?, + }, + }) + })? + .collect::>() + .context("list attributed git commits") } pub fn commit_attribution_evidence( diff --git a/src/db/models.rs b/src/db/models.rs index 1c632fc5b..f0c73d07e 100644 --- a/src/db/models.rs +++ b/src/db/models.rs @@ -286,6 +286,7 @@ pub struct SessionGraphInputs { pub discovered_entities: Vec, pub used_graph: bool, pub logs: Vec, + pub source_fields_truncated: bool, } /// A graph entity matched while resolving a topic string, with how it matched diff --git a/src/db/queries.rs b/src/db/queries.rs index 02225e409..927cc395d 100644 --- a/src/db/queries.rs +++ b/src/db/queries.rs @@ -37,6 +37,10 @@ use super::models::{ use super::pool::DbPool; use super::queries_service_instances; +#[path = "queries_session_graph.rs"] +mod session_graph; +pub use session_graph::{SessionGraphScope, correlate_session_graph_scoped}; + const SEARCH_FTS_CANDIDATE_CAP: usize = 10_000; const SIMILAR_INCIDENT_FTS_CANDIDATE_CAP: usize = 5_000; /// Cap on the FTS match-set materialization in the fast (index-led) search @@ -510,7 +514,7 @@ pub fn list_ai_sessions( params: &ListAiSessionsParams, ) -> Result> { let time_filtered = params.since.is_some() || params.until.is_some(); - if !time_filtered && ai_session_rollup_is_populated(pool)? { + if !time_filtered && params.ai_session_id.is_none() && ai_session_rollup_is_populated(pool)? { return list_ai_sessions_from_rollup(pool, params); } list_ai_sessions_live(pool, params) @@ -2100,6 +2104,7 @@ pub fn correlate_session_graph( discovered_entities, used_graph, logs, + source_fields_truncated: false, }) } diff --git a/src/db/queries_session_graph.rs b/src/db/queries_session_graph.rs new file mode 100644 index 000000000..7438dc3b9 --- /dev/null +++ b/src/db/queries_session_graph.rs @@ -0,0 +1,168 @@ +use super::*; + +#[derive(Debug, Clone)] +pub struct SessionGraphScope { + pub project: String, + pub tool: String, + pub host: String, + pub severity_in: Vec, +} + +/// Correlate an already resolved session without treating its native ID as a +/// globally unique identity. Graph keys omit the host, so shared keys fall back +/// to the selected session's own evidence instead of attributing other hosts. +pub fn correlate_session_graph_scoped( + pool: &DbPool, + session_id: &str, + scope: &SessionGraphScope, + limit: usize, +) -> Result { + let limit = limit.clamp(1, 1000); + let conn = pool.get()?; + let bounds = conn.query_row( + "SELECT MIN(timestamp), MAX(timestamp) FROM logs + WHERE ai_session_id=?1 AND ai_project=?2 AND lower(ai_tool)=lower(?3) + AND hostname=?4", + params![session_id, scope.project, scope.tool, scope.host], + |row| { + Ok(( + row.get::<_, Option>(0)?, + row.get::<_, Option>(1)?, + )) + }, + )?; + let (Some(start), Some(end)) = bounds else { + return Ok(SessionGraphInputs::default()); + }; + let key = format!( + "{}:{}:{}", + scope.project.trim().to_ascii_lowercase(), + scope.tool.trim().to_ascii_lowercase(), + session_id + ); + let shared_key: bool = conn.query_row( + "SELECT EXISTS(SELECT 1 FROM logs WHERE ai_session_id=?1 + AND lower(trim(ai_project))=lower(trim(?2)) AND lower(trim(ai_tool))=lower(trim(?3)) + AND hostname<>?4)", + params![session_id, scope.project, scope.tool, scope.host], + |row| row.get(0), + )?; + let exists: bool = conn.query_row( + "SELECT EXISTS(SELECT 1 FROM graph_entities WHERE entity_type='ai_session' AND canonical_key=?1)", + [&key], |row| row.get(0), + )?; + let used_graph = exists && !shared_key; + let mut discovered_hosts = Vec::new(); + let mut discovered_entities = Vec::new(); + if used_graph { + for entity in super::super::graph::graph_walk_n_hops(&conn, std::slice::from_ref(&key), 2)? + { + match entity.entity_type.as_str() { + super::super::graph::ENTITY_TYPE_HOST => { + discovered_hosts.push(entity.canonical_key.clone()) + } + super::super::graph::ENTITY_TYPE_CONTAINER => { + if let Some(host) = + super::super::entity_resolution::container_key_host(&entity.canonical_key) + { + discovered_hosts.push(host.to_string()); + } + } + super::super::graph::ENTITY_TYPE_SERVICE_INSTANCE => { + if let Some((host, _)) = + super::super::entity_resolution::split_service_instance_key( + &entity.canonical_key, + ) + { + discovered_hosts.push(host.to_string()); + } + } + _ => {} + } + discovered_entities.push(entity.canonical_key); + } + discovered_hosts.sort(); + discovered_hosts.dedup(); + discovered_entities.sort(); + discovered_entities.dedup(); + } + let mut bindings: Vec = vec![ + session_id.to_string().into(), + scope.project.clone().into(), + scope.tool.clone().into(), + scope.host.clone().into(), + start.clone().into(), + end.clone().into(), + ]; + let own = + "(l.ai_session_id=?1 AND l.ai_project=?2 AND lower(l.ai_tool)=lower(?3) AND l.hostname=?4)"; + let mut selection = own.to_string(); + if used_graph && !discovered_hosts.is_empty() { + let first_host = bindings.len() + 1; + bindings.extend( + discovered_hosts + .iter() + .cloned() + .map(rusqlite::types::Value::Text), + ); + let hosts = (first_host..=bindings.len()) + .map(|index| format!("?{index}")) + .collect::>() + .join(","); + // Host evidence is a temporal correlation, never another AI session's + // transcript presented as part of the selected identity. + selection = format!("({own} OR (l.ai_session_id IS NULL AND l.hostname IN ({hosts})))"); + } + let severity = if scope.severity_in.is_empty() { + String::new() + } else { + let first = bindings.len() + 1; + bindings.extend( + scope + .severity_in + .iter() + .cloned() + .map(rusqlite::types::Value::Text), + ); + let placeholders = (first..=bindings.len()) + .map(|index| format!("?{index}")) + .collect::>() + .join(","); + format!(" AND l.severity IN ({placeholders})") + }; + bindings.push((limit as i64).into()); + // Cap text before materializing rows. JSON metadata is omitted intact + // rather than returning an invalid JSON prefix; propagate the omission. + let columns = FTS_SELECT_COLS + .replace("l.message", "substr(l.message,1,2048)") + .replace( + "l.metadata_json", + "CASE WHEN length(l.metadata_json)>4096 THEN NULL ELSE l.metadata_json END", + ); + let sql = format!( + "SELECT {columns},(length(l.message)>2048 OR COALESCE(length(l.metadata_json)>4096,0)) FROM logs l WHERE {selection} + AND l.timestamp>=?5 AND l.timestamp<=?6{severity} ORDER BY l.timestamp DESC,l.id DESC LIMIT ?{}", + bindings.len() + ); + let rows = conn + .prepare(&sql)? + .query_map(rusqlite::params_from_iter(bindings.iter()), |row| { + Ok((map_row(row)?, row.get::<_, bool>(15)?)) + })? + .collect::>>()?; + let source_fields_truncated = rows.iter().any(|(_, truncated)| *truncated); + let logs = rows.into_iter().map(|(entry, _)| entry).collect(); + Ok(SessionGraphInputs { + bounds: Some((start, end)), + session_entity_keys: if used_graph { vec![key] } else { Vec::new() }, + discovered_hosts, + discovered_entities, + used_graph, + logs, + source_fields_truncated, + }) +} + +#[cfg(test)] +#[path = "queries_session_graph_tests.rs"] +mod tests; diff --git a/src/db/queries_session_graph_tests.rs b/src/db/queries_session_graph_tests.rs new file mode 100644 index 000000000..b57726105 --- /dev/null +++ b/src/db/queries_session_graph_tests.rs @@ -0,0 +1,218 @@ +use super::*; +use crate::db::{LogBatchEntry, init_pool, insert_logs_batch}; + +fn fixture() -> (DbPool, tempfile::TempDir) { + let dir = tempfile::tempdir().unwrap(); + let pool = init_pool(&StorageConfig::for_test(dir.path().join("scope.db"))).unwrap(); + (pool, dir) +} + +fn row(time: &str, host: &str, project: &str, tool: &str, session: &str) -> LogBatchEntry { + LogBatchEntry { + timestamp: time.into(), + hostname: host.into(), + severity: "info".into(), + message: format!("{project}/{tool}/{host}"), + raw: "fixture".into(), + source_ip: "fixture://scope".into(), + ai_project: Some(project.into()), + ai_tool: Some(tool.into()), + ai_session_id: Some(session.into()), + facility: None, + app_name: None, + process_id: None, + docker_checkpoint: None, + ai_transcript_path: None, + metadata_json: None, + http_status: None, + auth_outcome: None, + dns_blocked: None, + event_action: None, + parse_error: None, + } +} + +fn scope() -> SessionGraphScope { + SessionGraphScope { + project: "chosen".into(), + tool: "codex".into(), + host: "host-a".into(), + severity_in: vec!["info".into()], + } +} + +fn graph_entity(conn: &rusqlite::Connection, kind: &str, key: &str) -> i64 { + conn.execute( + "INSERT INTO graph_entities(entity_type,canonical_key,display_label,trust_level) + VALUES(?1,?2,?2,'verified')", + params![kind, key], + ) + .unwrap(); + conn.last_insert_rowid() +} + +#[test] +fn fallback_reused_native_ids_preserve_selected_project_tool_host_and_bounds() { + let (pool, _dir) = fixture(); + insert_logs_batch( + &pool, + &[ + row( + "2026-09-21T08:00:00Z", + "host-a", + "chosen", + "CODEX", + "reused_%", + ), + row( + "2026-09-21T09:00:00Z", + "host-b", + "chosen", + "codex", + "reused_%", + ), + row( + "2026-09-21T10:00:00Z", + "host-a", + "other", + "codex", + "reused_%", + ), + row( + "2026-09-21T11:00:00Z", + "host-a", + "chosen", + "claude", + "reused_%", + ), + ], + ) + .unwrap(); + let result = correlate_session_graph_scoped(&pool, "reused_%", &scope(), 10).unwrap(); + assert!(!result.used_graph); + assert_eq!(result.logs.len(), 1); + assert_eq!(result.logs[0].message, "chosen/CODEX/host-a"); + assert_eq!( + result.bounds, + Some(("2026-09-21T08:00:00Z".into(), "2026-09-21T08:00:00Z".into())) + ); +} + +#[test] +fn graph_uses_exact_seed_and_keeps_host_context_without_other_session_transcripts() { + let (pool, _dir) = fixture(); + let mut context = row( + "2026-09-21T08:01:00Z", + "host-a", + "unused", + "unused", + "unused", + ); + context.ai_session_id = None; + context.ai_project = None; + context.ai_tool = None; + context.message = "host context".into(); + insert_logs_batch( + &pool, + &[ + row( + "2026-09-21T08:00:00Z", + "host-a", + "chosen", + "codex", + "reused_%", + ), + row( + "2026-09-21T08:02:00Z", + "host-a", + "chosen", + "codex", + "reused_%", + ), + row( + "2026-09-21T08:01:00Z", + "host-a", + "other", + "codex", + "reused_%", + ), + context, + ], + ) + .unwrap(); + { + let conn = pool.get().unwrap(); + let selected = graph_entity(&conn, "ai_session", "chosen:codex:reused_%"); + graph_entity(&conn, "ai_session", "other:codex:reused_%"); + let host = graph_entity(&conn, "host", "host-a"); + conn.execute("INSERT INTO graph_relationships(relationship_key,src_entity_id,dst_entity_id, + relationship_type,reason_code,trust_level,confidence,last_seen_at) + VALUES('scope-link',?1,?2,'runs_on','log_app_name','inferred',0.5,'2026-09-21T08:00:00Z')", params![selected,host]).unwrap(); + } + let result = correlate_session_graph_scoped(&pool, "reused_%", &scope(), 10).unwrap(); + assert!(result.used_graph); + assert_eq!(result.session_entity_keys, ["chosen:codex:reused_%"]); + assert_eq!(result.logs.len(), 3); + assert!(result.logs.iter().any(|row| row.message == "host context")); + assert!( + result + .logs + .iter() + .all(|row| row.ai_project.as_deref() != Some("other")) + ); +} + +#[test] +fn graph_key_shared_by_hosts_falls_back_without_claiming_other_host_evidence() { + let (pool, _dir) = fixture(); + insert_logs_batch( + &pool, + &[ + row("2026-09-21T08:00:00Z", "host-a", "chosen", "codex", "same"), + row("2026-09-21T08:01:00Z", "host-b", "chosen", "codex", "same"), + ], + ) + .unwrap(); + graph_entity(&pool.get().unwrap(), "ai_session", "chosen:codex:same"); + let result = correlate_session_graph_scoped(&pool, "same", &scope(), 10).unwrap(); + assert!(!result.used_graph); + assert!(result.session_entity_keys.is_empty()); + assert!(result.discovered_hosts.is_empty()); + assert_eq!(result.logs.len(), 1); + assert_eq!(result.logs[0].hostname, "host-a"); +} + +#[test] +fn severity_filter_precedes_the_correlation_row_limit() { + let (pool, _dir) = fixture(); + let mut error = row("2026-09-21T08:00:00Z", "host-a", "chosen", "codex", "same"); + error.severity = "err".into(); + insert_logs_batch( + &pool, + &[ + error, + row("2026-09-21T08:01:00Z", "host-a", "chosen", "codex", "same"), + row("2026-09-21T08:02:00Z", "host-a", "chosen", "codex", "same"), + ], + ) + .unwrap(); + let mut scope = scope(); + scope.severity_in = vec!["err".into()]; + let result = correlate_session_graph_scoped(&pool, "same", &scope, 1).unwrap(); + assert_eq!(result.logs.len(), 1); + assert_eq!(result.logs[0].severity, "err"); +} + +#[test] +fn oversized_source_fields_are_bounded_with_explicit_partial_evidence() { + let (pool, _dir) = fixture(); + let mut source = row("2026-09-21T08:00:00Z", "host-a", "chosen", "codex", "large"); + source.message = "😀".repeat(10_000); + source.metadata_json = + Some(serde_json::json!({"source_kind":"fixture", "large":"x".repeat(10_000)}).to_string()); + insert_logs_batch(&pool, &[source]).unwrap(); + let result = correlate_session_graph_scoped(&pool, "large", &scope(), 10).unwrap(); + assert!(result.source_fields_truncated); + assert_eq!(result.logs[0].message.chars().count(), 2048); + assert!(result.logs[0].metadata_json.is_none()); +} diff --git a/src/db/queries_tests.rs b/src/db/queries_tests.rs index 713f12276..f3e468533 100644 --- a/src/db/queries_tests.rs +++ b/src/db/queries_tests.rs @@ -3663,3 +3663,43 @@ fn lint_flags_unquoted_hyphen_term_alongside_a_quoted_phrase() { .to_string(); assert!(err.contains("NOT operator"), "{err}"); } + +#[test] +fn exact_session_lookup_finds_new_evidence_after_rollup_refresh() { + let (pool, _dir) = test_pool(); + insert_logs_batch( + &pool, + &[make_ai_entry( + "2026-01-01T00:00:00Z", + "host-a", + "claude", + "/tmp/project", + "old", + "old event", + )], + ) + .unwrap(); + refresh_ai_session_rollup(&pool).unwrap(); + insert_logs_batch( + &pool, + &[make_ai_entry( + "2026-01-01T00:01:00Z", + "host-a", + "claude", + "/tmp/project", + "fresh", + "fresh event", + )], + ) + .unwrap(); + let result = list_ai_sessions( + &pool, + &ListAiSessionsParams { + ai_session_id: Some("fresh".into()), + ..Default::default() + }, + ) + .unwrap(); + assert_eq!(result.len(), 1); + assert_eq!(result[0].ai_session_id, "fresh"); +} diff --git a/src/filetail/supervisor_tests.rs b/src/filetail/supervisor_tests.rs index fa036b39e..c3d7e9785 100644 --- a/src/filetail/supervisor_tests.rs +++ b/src/filetail/supervisor_tests.rs @@ -257,6 +257,7 @@ async fn supervisor_coalesces_checkpoint_writes_for_a_burst() { configured.start_at_end = false; registry.upsert(configured).unwrap(); let baseline_writes = registry.write_count(); + let burst_started = std::time::Instant::now(); supervisor.reconcile().await.unwrap(); let mut writer = tokio::fs::OpenOptions::new() @@ -291,8 +292,14 @@ async fn supervisor_coalesces_checkpoint_writes_for_a_burst() { .await .unwrap(); + // The supervisor may legitimately flush every 250ms while the fixture's + // asynchronous writes and reads are delayed by other tests. Allow those + // elapsed intervals plus initialization/final persistence, while rejecting + // a checkpoint write for each line. + let elapsed_flushes = burst_started.elapsed().as_millis().div_ceil(250) as usize; + let allowed_writes = (elapsed_flushes + 2).min(199); assert!( - registry.write_count() - baseline_writes <= 4, + registry.write_count() - baseline_writes <= allowed_writes, "checkpoint persistence scaled with line count" ); token.cancel(); diff --git a/src/mcp/actions.rs b/src/mcp/actions.rs index 6170f4928..4fc22f403 100644 --- a/src/mcp/actions.rs +++ b/src/mcp/actions.rs @@ -389,7 +389,7 @@ pub(super) const ACTION_SPECS: &[ActionSpec] = &[ Expensive, SessionInvestigate, exact: { - allowed: &["session_id", "tool", "project", "host", "limit", "window_minutes", "severity_min"], + allowed: &["session_id", "tool", "project", "host", "limit", "severity_min"], required: &["session_id"] } ), diff --git a/src/mcp/tools_tests.rs b/src/mcp/tools_tests.rs index 4a5cd2605..8f1cb9dce 100644 --- a/src/mcp/tools_tests.rs +++ b/src/mcp/tools_tests.rs @@ -1984,3 +1984,53 @@ async fn help_action_dispatch_returns_the_tool_reference() { assert!(reference.contains("## cortex search\n")); assert!(reference.contains("## cortex ack_error\n")); } + +#[tokio::test] +async fn session_investigate_rejects_unsupported_window_minutes() { + let harness = TestHarness::new(); + let error = execute_tool( + &harness.state, + "cortex", + json!({ + "action": "session_investigate", "session_id": "s", "window_minutes": 5 + }), + None, + ) + .await + .unwrap_err(); + assert!(error.to_string().contains("unknown field `window_minutes`")); +} + +#[tokio::test] +async fn session_investigate_dispatch_preserves_selected_identity_and_bounded_metadata() { + let harness = TestHarness::new(); + let conn = harness.pool.get().unwrap(); + for host in ["selected", "other"] { + conn.execute("INSERT INTO logs(timestamp,hostname,severity,message,raw,source_ip,ai_tool,ai_project,ai_session_id) VALUES('2026-09-21T00:00:00Z',?1,'info','user message','','fixture','codex','p','shared')", [host]).unwrap(); + } + drop(conn); + let response = execute_tool(&harness.state, "cortex", json!({ + "action": "session_investigate", "session_id": "shared", "host": "selected", "project": "p", "tool": "codex" + }), None).await.unwrap(); + assert_eq!(response["result"]["session"]["hostname"], "selected"); + assert!( + response["result"]["correlation"]["logs"] + .as_array() + .unwrap() + .iter() + .all(|log| log["entry"]["hostname"] == "selected") + ); + assert_eq!(response["metadata"]["auth_state"], "unknown"); + assert!( + response["metadata"]["budget_used"]["payload_bytes"] + .as_u64() + .unwrap() + <= 65_536 + ); + assert_eq!( + response["metadata"]["budget_used"]["payload_bytes"] + .as_u64() + .unwrap() as usize, + serde_json::to_vec(&response).unwrap().len() + ); +} diff --git a/src/surfaces.rs b/src/surfaces.rs index bbba6d7f2..40ab2ec90 100644 --- a/src/surfaces.rs +++ b/src/surfaces.rs @@ -332,6 +332,7 @@ pub const SURFACE_SPECS: &[SurfaceSpec] = &[ mcp!("abuse_incidents", Sessions, Canonical, Read), mcp!("abuse_investigate", Sessions, Canonical, Read), mcp!("ai_correlate", Sessions, Canonical, Read), + mcp!("session_investigate", Sessions, Canonical, Read), mcp!( "topic_correlate", Correlate, diff --git a/tests/live/phases/mcp/run.sh b/tests/live/phases/mcp/run.sh index 7f078ec61..161ceca65 100644 --- a/tests/live/phases/mcp/run.sh +++ b/tests/live/phases/mcp/run.sh @@ -214,6 +214,7 @@ mcp_phase_run() { artifact_evidence_record) jq -e '.result.structuredContent.inserted==true and .result.structuredContent.event.eventId=="mcp-live-event" and .result.structuredContent.event.artifactId=="mcp-live-artifact"' "$output" >/dev/null || result=fail ;; artifact_evidence) jq -e 'any(.result.structuredContent.events[]?;.eventId=="mcp-live-event" and .artifactId=="mcp-live-artifact")' "$output" >/dev/null || result=fail ;; search_sessions) jq -e --arg s "$MCP_LIVE_SESSION" 'any(.result.structuredContent.sessions[]?;.session_id==$s and .event_count>0)' "$output" >/dev/null || result=fail ;; + session_investigate) jq -e --arg s "$MCP_LIVE_SESSION" '.result.structuredContent.result.session.session_id==$s and (.result.structuredContent.result.transcript|length)>0 and .result.structuredContent.metadata.budget_used.payload_bytes<=65536' "$output" >/dev/null || result=fail ;; sessions) jq -e --arg s "$MCP_LIVE_SESSION" 'any(.result.structuredContent.sessions[]?;.session_id==$s and .event_count>0)' "$output" >/dev/null || result=fail ;; correlate) jq -e --arg h "$MCP_LIVE_ERROR_HOST" '.result.structuredContent.total_events>0 and any(.result.structuredContent.hosts[]?;.hostname==$h and (.events|length)>0)' "$output" >/dev/null || result=fail ;; correlate_state) jq -e --arg h "$MCP_LIVE_TOPIC_HOST" 'any(.result.structuredContent.hosts[]?;.hostname==$h and .heartbeat_summary!=null and (.logs|length)>0)' "$output" >/dev/null || result=fail ;; diff --git a/tests/live/phases/mcp/scenarios.json b/tests/live/phases/mcp/scenarios.json index cdd2e6f91..e32dd00a4 100644 --- a/tests/live/phases/mcp/scenarios.json +++ b/tests/live/phases/mcp/scenarios.json @@ -3,6 +3,7 @@ "arguments": { "search": {"query":"cortex"}, "evidence_scope": {"branch":"main"}, + "session_investigate": {"session_id":"mcp-live-session"}, "host_state": {"host":"missing-live-host"}, "correlate": {"query":"cortex","reference_time":"2026-08-27T12:00:00Z"}, "correlate_state": {"reference_time":"2026-08-27T12:00:00Z"}, @@ -43,7 +44,7 @@ "hook_investigate":"evidence","hosts":"hosts","incident_context":"total_logs","ingest_rate":"buckets","list_ai_projects":"projects", "list_ai_tools":"tools","llm_invocations":"$array","map":"schema","mcp_events":"events","mcp_incidents":"incidents","mcp_investigate":"evidence", "notifications_recent":"$array","patterns":"patterns","project_context":"project","search":"logs","search_sessions":"sessions","sessions":"sessions", - "evidence_scope":"items","recurring_error_comparison":"comparisons","silent_hosts":"hosts","similar_incidents":"clusters","skill_events":"events","skill_incidents":"incidents","skill_investigate":"evidence", + "session_investigate":"result","evidence_scope":"items","recurring_error_comparison":"comparisons","silent_hosts":"hosts","similar_incidents":"clusters","skill_events":"events","skill_incidents":"incidents","skill_investigate":"evidence", "source_ips":"source_ips","stats":"total_logs","status":"status","tail":"logs","timeline":"points","topic_correlate":"topic", "unaddressed_errors":"signatures","usage_blocks":"blocks","ack_error":"signature_hash","unack_error":"signature_hash","host_state":"host_id","compose_doctor":"diagnostics","notifications_test":"result","graph":"resolved_entity" },