diff --git a/arcane/home/honeypot-dashboard/backend-service/Dockerfile b/arcane/home/honeypot-dashboard/backend-service/Dockerfile index d47d4e24..3867ed78 100644 --- a/arcane/home/honeypot-dashboard/backend-service/Dockerfile +++ b/arcane/home/honeypot-dashboard/backend-service/Dockerfile @@ -26,6 +26,15 @@ RUN apt-get update \ COPY --from=build /src/target/release/apiary-backend /usr/local/bin/apiary-backend ENV LISTEN_ADDR=0.0.0.0:8081 EXPOSE 8081 +# Liveness, deliberately NOT /readyz (#3317). /livez is the bare question +# "is this process up" and never touches Elasticsearch; /readyz answers 503 +# whenever ES is unreachable or write-blocked, and curling that from a +# container healthcheck would restart a perfectly healthy backend every time +# Elasticsearch had a bad minute -- turning a dependency's outage into a +# crash loop in the thing that was not broken. Anything that wants to know +# whether the backend can actually do its job probes /readyz itself +# (diagnostics, the deploy verifier); see docs/OPERATIONS.md. +# # Port derived from $LISTEN_ADDR at check time, not hardcoded to the 8081 # default: every non-default LISTEN_ADDR consumer of this image (backend- # worker at 127.0.0.1:8099, backend-worker-importer at 127.0.0.1:8098, @@ -36,6 +45,6 @@ EXPOSE 8081 # not by any test -- CI never runs the image under a non-default # LISTEN_ADDR). HEALTHCHECK --interval=15s --timeout=5s --retries=5 --start-period=20s \ - CMD curl -sf "http://127.0.0.1:${LISTEN_ADDR##*:}/healthz" || exit 1 + CMD curl -sf "http://127.0.0.1:${LISTEN_ADDR##*:}/livez" || exit 1 USER nobody CMD ["apiary-backend"] diff --git a/arcane/home/honeypot-dashboard/backend-service/src/agent_intrusion.rs b/arcane/home/honeypot-dashboard/backend-service/src/agent_intrusion.rs index 651c8dc6..09f51cba 100644 --- a/arcane/home/honeypot-dashboard/backend-service/src/agent_intrusion.rs +++ b/arcane/home/honeypot-dashboard/backend-service/src/agent_intrusion.rs @@ -296,7 +296,7 @@ async fn run_cycle(state: &AppState, fetch_window: Duration, max_events_per_sour // rules::evaluate_event's per-event decode/hash rule chain, run once per // fetched event — up to 20,000 under the default fetch caps) starves // every other task sharing this process's tokio runtime, including - // main.rs's /healthz handler: confirmed live during #1628's preflight — + // main.rs's probe handlers: confirmed live during #1628's preflight — // agent-intrusion run alone reliably made /healthz stop responding // entirely (a direct curl timed out at 45s with zero response) within // ~90s of every cycle start. The container's own cpu.max cgroup quota diff --git a/arcane/home/honeypot-dashboard/backend-service/src/es.rs b/arcane/home/honeypot-dashboard/backend-service/src/es.rs index 8442e651..12885927 100644 --- a/arcane/home/honeypot-dashboard/backend-service/src/es.rs +++ b/arcane/home/honeypot-dashboard/backend-service/src/es.rs @@ -7,6 +7,7 @@ use elasticsearch::{ cluster::ClusterHealthParts, http::transport::{SingleNodeConnectionPool, TransportBuilder}, + indices::IndicesGetSettingsParts, params::{OpType, Refresh}, BulkOperation, BulkParts, Elasticsearch, SearchParts, }; @@ -52,6 +53,63 @@ pub const EVENT_INDICES: &[&str] = &[ "traefik-v1-*", ]; +/// Every index family THIS TIER WRITES TO, as the index expressions +/// `/readyz` asks Elasticsearch about (`Es::write_blocked`, #3317). +/// +/// The complement of `EVENT_INDICES` above, not a copy of it: a read-only +/// index somebody else owns going write-blocked is that tier's outage, but +/// a write-blocked index in this list means the dashboard can no longer +/// record what it observes — which is the difference between "the process +/// is up" (what /livez says) and "the process can do its job" (what +/// /readyz has to answer). +/// +/// `dashboard-*` is deliberately one wildcard rather than the twenty-odd +/// concrete `dashboard-*-v1` names behind it: every serving-tier index +/// uses that prefix, and a pattern cannot fall out of date the way a +/// hand-copied list of them can. The unprefixed families are named +/// individually because nothing about their names is predictable — each is +/// the sole write target of one bundled worker loop. +/// +/// The list drifts in one direction harmlessly and in the other loudly. An +/// index missing from it goes unchecked, so readiness stays green on a +/// block nobody is watching; an entry for an index this tier only *reads* +/// reports not-ready during another component's outage, which trains +/// operators to ignore the endpoint. So when a new write target is added, +/// add it here, and when in doubt leave it out rather than widening the +/// claim. +pub const WRITE_TARGET_FAMILIES: &[&str] = &[ + // Serving tier: config, users, credentials, IP blocks, report + // definitions/generated, workbench runs+recipes, payload inventory and + // bytes, canary tokens, problem reports, webhook delivery, ML acks. + "dashboard-*", + // Bundled worker loops. + "attackers-v1", // attacker_identity + "attacker-clusters-v1", // correlator + "campaigns-v1", // correlator + "cred-reuse-v1", // correlations + "flow-links-v1", // correlations + "ioc-verdicts-v1", // attacker_identity + "overview-rollup-v1", // rollups + "geo-rollup-v1", // rollups + "attack-rollup-v1", // rollups + "gpu-job-queue", // gpu_queue, vault_rag + "agent-intrusion-campaigns", // agent_intrusion + // es_importer's mirrors of the external analysis workers. Its own + // sources table keys these off env vars, so a deployment that never + // set one simply has no such index and never gets asked about it. + "yara-analysis-v1", + "ghidra-analysis-v1", + "ghidra-report-artifacts-v1", + "sandbox-analysis-v1", + "sandbox-export-artifacts-v1", + "revdeck-analysis-v1", + "cape-analysis-v1", + "cowrie-ttylog-v1", + "mailoney-mail-v1", + "github-analysis-v1", + "reporter-metrics-v1", +]; + /// The "is this a login/auth attempt" query fragment (#1611 workstream C), /// shared by every login-counting aggregation (aggregates.rs's sources /// page, dashboard.rs's overview KPIs, reports_data.rs's report @@ -117,17 +175,27 @@ impl Es { pub fn connect(url: &str) -> anyhow::Result { // Transport::single_node's own default carries no request timeout at // all -- confirmed live during #1628's preflight: backend-worker's - // /healthz calls this same shared client's ping() (main.rs's healthz - // handler), and a slow/stuck query from any of its four bundled - // worker loops (correlator's aggregation was hitting Elasticsearch's - // own too_many_buckets_exception on every cycle against real data, + // /healthz handler used to share this same client, and a slow/stuck + // query from any of its four bundled worker loops (correlator's + // aggregation was hitting Elasticsearch's own + // too_many_buckets_exception on every cycle against real data, // plausibly saturating the cluster's search queue for a while) could - // therefore hang ping() indefinitely too, since nothing ever timed - // out client-side -- the container never crashed, /healthz just - // never returned, and something external kept restarting it every - // few minutes on the resulting healthcheck failure. 30s bounds the - // worst case without punishing legitimately slower real-data - // queries the way CI's fixture-sized indices never needed to. + // therefore hang the health probe indefinitely too, since nothing ever + // timed out client-side -- the container never crashed, /healthz just + // never returned, and something external kept restarting it every few + // minutes on the resulting healthcheck failure. 30s bounds the worst + // case without punishing legitimately slower real-data queries the + // way CI's fixture-sized indices never needed to. + // + // #3317: that health probe no longer exists. /livez (and its + // /healthz alias) never touches Elasticsearch at all, and /readyz + // runs its two probes under its own 5s READINESS_TIMEOUT rather than + // inheriting this one, so nothing on the probe path can now block for + // 30 seconds. The budget below is therefore sized purely for real + // queries, which is what it was always for -- kept because those + // worker loops are still the callers that can saturate the search + // queue, and deleting the only bound they have would reintroduce the + // hang for them. let conn_pool = SingleNodeConnectionPool::new(url.parse()?); let transport = TransportBuilder::new(conn_pool) .timeout(Duration::from_secs(30)) @@ -138,22 +206,95 @@ impl Es { } /// Cluster color (green/yellow/red), "unreachable" on transport error. + /// + /// Deliberately lossy: this feeds the source-health card, which must + /// render even when Elasticsearch is the thing that is broken, so it + /// has nowhere to put an error string. `/readyz` needs the error + /// itself and uses `cluster_health_status` below instead. pub async fn cluster_status(&self) -> String { - match self + self.cluster_health_status().await.unwrap_or_else(|_| "unreachable".into()) + } + + /// Cluster color as Elasticsearch reports it, or the failure explaining + /// why it could not be asked. The fallible half of `cluster_status`, + /// for callers that have to say which of the two happened (#3317). + pub async fn cluster_health_status(&self) -> anyhow::Result { + let response = self .client .cluster() .health(ClusterHealthParts::None) .send() - .await - { - Ok(response) => response - .json::() - .await - .ok() - .and_then(|value| value["status"].as_str().map(String::from)) - .unwrap_or_else(|| "unknown".into()), - Err(_) => "unreachable".into(), + .await?; + let status = response.status_code(); + let json = response.json::().await?; + if !status.is_success() { + anyhow::bail!("elasticsearch cluster health {}: {}", status, json); } + Ok(json["status"].as_str().unwrap_or("unknown").to_string()) + } + + /// Which of `indexes` Elasticsearch currently has `index.blocks.write` + /// set on, sorted. Empty means "a write to any of them would be + /// accepted" (#3317). + /// + /// The setting is the whole answer to "is this writable", and asking + /// for it by name means the response carries nothing else — this runs + /// on a probe, where a full settings dump per index family would be + /// the most expensive request this tier makes. + /// + /// Two things it deliberately does not report as a block. An index that + /// does not exist is absent from the response, not blocked: every + /// dashboard-owned index is created lazily on its first write (see + /// `search_paginated`), so a fresh cluster correctly reads ready — and + /// the allow_no_indices/ignore_unavailable pair above is what keeps + /// "no such index" from arriving as a 404 error instead. The other is + /// that a set value is compared rather than the key merely existing: + /// Elasticsearch drops the setting entirely when a block is cleared, so + /// `false` is not a shape this endpoint has been observed to return, + /// but reading the block as "the key is present" would be a claim + /// about this Elasticsearch's normalization rather than about the + /// block. + pub async fn write_blocked(&self, indexes: &[&str]) -> anyhow::Result> { + if indexes.is_empty() { + return Ok(Vec::new()); + } + let response = self + .client + .indices() + .get_settings(IndicesGetSettingsParts::IndexName(indexes, &["index.blocks.write"])) + // flat_settings is what puts the block at the literal + // "index.blocks.write" key; without it the response nests it as + // settings.index.blocks.write and the lookup below silently + // matches nothing, i.e. always ready. + .flat_settings(true) + // A family that matches nothing (a worker loop this deployment + // never enabled writes into an index that was never created) is + // normal, not an error. + .allow_no_indices(true) + .ignore_unavailable(true) + .send() + .await?; + let status = response.status_code(); + let json = response.json::().await?; + if !status.is_success() { + anyhow::bail!("elasticsearch get_settings {}: {}", status, json); + } + let Some(indices) = json.as_object() else { + return Ok(Vec::new()); + }; + let mut blocked: Vec = indices + .iter() + .filter(|(_, settings)| { + settings["settings"]["index.blocks.write"] + .as_str() + .is_some_and(|block| block == "true" || block == "1") + }) + .map(|(name, _)| name.clone()) + .collect(); + // Sorted so the same outage always names its indices in the same + // order, which makes two consecutive /readyz bodies diffable. + blocked.sort(); + Ok(blocked) } /// Cluster-wide index stats summary: (index_count, doc_count, bytes). @@ -466,15 +607,6 @@ impl Es { Ok(body["updated"].as_u64().unwrap_or(0)) } - pub async fn ping(&self) -> bool { - self.client - .ping() - .send() - .await - .map(|r| r.status_code().is_success()) - .unwrap_or(false) - } - /// One `_search` against the shared event indices; body is the caller's /// full query DSL. `ignore_unavailable` matches the Go client's posture /// so a not-yet-created index family never fails the whole query. diff --git a/arcane/home/honeypot-dashboard/backend-service/src/main.rs b/arcane/home/honeypot-dashboard/backend-service/src/main.rs index be89218c..182be719 100644 --- a/arcane/home/honeypot-dashboard/backend-service/src/main.rs +++ b/arcane/home/honeypot-dashboard/backend-service/src/main.rs @@ -105,15 +105,183 @@ pub struct AppState { pub observability: Arc, } +/// /livez — the process is up and its HTTP stack is answering. Says nothing +/// about Elasticsearch, and must never ask it. +/// +/// This is the endpoint the container HEALTHCHECK curls on an interval, and +/// the reason it stays dependency-free is the whole point of #3317: a probe +/// that can block on Elasticsearch turns that dependency's outage into a +/// restart loop of a container which was never the thing that broke. The +/// old /healthz answered `{"ok": true, "es": }` from inside this +/// handler, so an ES outage showed up here as a slow or failed probe +/// instead of as an ES outage. +#[derive(Serialize)] +struct Liveness { + live: bool, + /// build.rs's compile stamp, so a probe can answer "is the running + /// binary newer than the merge" without a second round trip. Same + /// field main logs at boot. + built: String, +} + +/// /readyz — Elasticsearch is reachable and this tier's own write targets +/// are not write-blocked, i.e. the backend can actually do its job rather +/// than merely be running. 503 plus a `reason` when it cannot. +/// +/// Unlike liveness this endpoint is allowed to fail, so it is the one +/// diagnostics and the #3315 deploy verifier probe: "the process is up" is +/// the wrong question during an ingest outage, and it is the only question +/// the old endpoint could ask. #[derive(Serialize)] -struct Health { - ok: bool, - es: bool, +struct Readiness { + ready: bool, + /// Present exactly when `ready` is false, and specific enough to act + /// on — "Elasticsearch is unreachable" and "these four indices are + /// write-blocked" send an operator to different pages. + #[serde(skip_serializing_if = "Option::is_none")] + reason: Option, + /// green / yellow / red / unreachable. Reported even when ready, since + /// yellow is the ordinary shape of a replicated cluster and a probe + /// that only ever printed green would be no better than the constant + /// it replaces. + cluster: String, + /// The write-blocked members of `es::WRITE_TARGET_FAMILIES`. Always + /// present so a consumer can read one shape; named rather than counted, + /// because the point is to be able to act on which ones. + write_blocked: Vec, +} + +/// /readyz's own deadline, independent of the shared client's. +/// +/// es::connect gives the transport 30s, sized for real multi-second queries +/// rather than for a probe — and es.rs's own comment on that budget records +/// a /healthz that stopped responding because a worker loop's aggregation +/// saturated the search queue. A readiness answer somebody is waiting on +/// should arrive in seconds, and a probe that blocks for 30 is +/// indistinguishable from the outage it exists to report. +const READINESS_TIMEOUT: std::time::Duration = std::time::Duration::from_secs(5); + +/// The one place readiness is decided, kept pure so its truth table is +/// testable without an Elasticsearch to ask — the same discipline as +/// `resolve_service_token` below, and for the same reason: the interesting +/// cases are the ones where two independent probes disagree about how bad +/// things are, and a test that needs a live cluster to reach them is a test +/// that does not get run. +/// +/// `cluster` and `write_blocked` are two Results rather than one tuple of +/// plain values so a partial failure is a case this function has to answer +/// for, instead of one a caller has to. +fn readiness_verdict( + cluster: anyhow::Result, + write_blocked: anyhow::Result>, +) -> Readiness { + // Unreachable outranks everything else. A write-block reading against a + // cluster we could not reach is not a fact, it is the absence of one, + // and reporting it as the cause would be a guess. + let cluster = match cluster { + Ok(status) => status, + Err(error) => { + return Readiness { + ready: false, + reason: Some(format!("elasticsearch is unreachable: {error}")), + cluster: "unreachable".to_string(), + write_blocked: Vec::new(), + } + } + }; + let blocked = match write_blocked { + Ok(blocked) => blocked, + Err(error) => { + return Readiness { + ready: false, + reason: Some(format!("elasticsearch refused the readiness probe: {error}")), + cluster, + write_blocked: Vec::new(), + } + } + }; + // Red means unassigned primaries, against which both reads and writes + // fail. Yellow means unassigned *replicas*, which is the ordinary shape + // of a replicated cluster during a rolling restart and costs this tier + // nothing — a red-only gate would go not-ready on every deploy. + if cluster == "red" { + return Readiness { + ready: false, + reason: Some("elasticsearch cluster health is red (unassigned primaries)".to_string()), + cluster, + write_blocked: blocked, + }; + } + if !blocked.is_empty() { + return Readiness { + ready: false, + reason: Some(format!( + "elasticsearch has index.blocks.write set on: {}. The flood-stage disk \ + watermark sets this on every index at once, and so does an operator's \ + `PUT //_block/write`; this endpoint cannot tell those apart, so \ + check _cat/allocation free space before concluding which one it is.", + blocked.join(", ") + )), + cluster, + write_blocked: blocked, + }; + } + Readiness { ready: true, reason: None, cluster, write_blocked: blocked } +} + +/// GET /livez, and GET /healthz — the same handler under two names. See +/// `Liveness` for why the response carries no Elasticsearch signal. +async fn livez() -> Json { + Json(Liveness { live: true, built: build_stamp() }) } -async fn healthz(State(state): State) -> Json { - let es_ok = state.es.ping().await; - Json(Health { ok: true, es: es_ok }) +/// `/healthz` is the name the image's HEALTHCHECK, the port-test harness +/// and the ops scripts already use, so it stays as an alias rather than +/// being broken (killing a 2024-era name in a health-probe rename is how a +/// stack ends up reporting permanently unhealthy with nothing wrong). It +/// used to answer `{"ok": true, "es": }` where `ok` was the constant +/// #3317 is about; the `es` half of that answer now lives on /readyz, which +/// can say no. +async fn healthz() -> Json { + livez().await +} + +/// GET /readyz — see `Readiness`. Unauthenticated exactly like the liveness +/// probes and /metrics, because the callers are infrastructure: the deploy +/// verifier, diagnostics, and an operator on a jump host. Authentication +/// would not make it safer here, only less answerable. +async fn readyz(State(state): State) -> (StatusCode, Json) { + // Both probes in flight together: they are independent round trips and + // the endpoint's whole value is being quick to answer. The async block + // is what makes the pair a single future the deadline can wrap -- + // `join!` on its own expands to the values, not to something awaitable. + let probes = tokio::time::timeout(READINESS_TIMEOUT, async { + tokio::join!( + state.es.cluster_health_status(), + state.es.write_blocked(es::WRITE_TARGET_FAMILIES), + ) + }) + .await; + let (cluster, write_blocked) = match probes { + Ok(probes) => probes, + Err(_elapsed) => { + // Both halves report the deadline, because a probe pair that + // timed out established nothing about either question. The + // verdict resolves the cluster half as unreachable; this one + // exists so the pair stays a pair of Results rather than a + // Result of a pair, and its text is never what gets reported. + let expired = || anyhow::anyhow!("no answer within {}s", READINESS_TIMEOUT.as_secs()); + (Err(expired()), Err(expired())) + } + }; + let readiness = readiness_verdict(cluster, write_blocked); + let status = if readiness.ready { + StatusCode::OK + } else { + StatusCode::SERVICE_UNAVAILABLE + }; + tracing::debug!(ready = readiness.ready, cluster = %readiness.cluster, "readyz"); + (status, Json(readiness)) } /// A boot refusal carries the code the cutover doc and dashboards grep @@ -476,7 +644,17 @@ async fn main() -> anyhow::Result<()> { .layer(middleware::from_fn_with_state(state.clone(), require_service_token)); let app = Router::new() + // #3317: liveness and readiness are different questions with + // different consequences, and conflating them is what let a backend + // that could not reach Elasticsearch keep answering `ok: true` to + // its own healthcheck. /livez (and its /healthz alias) is "the + // process is up" and never touches Elasticsearch, so an ES outage + // cannot restart-loop the container; /readyz is "this can do its + // job" and answers 503 with a reason when it cannot. All three are + // unauthenticated, like the /healthz they grew out of. + .route("/livez", get(livez)) .route("/healthz", get(healthz)) + .route("/readyz", get(readyz)) // #1972: same listener, same internal-network posture as /healthz. .route("/metrics", get(obs::metrics_route)) .merge(api) @@ -589,3 +767,112 @@ mod service_token_tests { assert_eq!(resolved, Some("s3cret")); } } + +#[cfg(test)] +mod readiness_tests { + use super::readiness_verdict; + + fn blocked(names: &[&str]) -> Vec { + names.iter().map(|name| name.to_string()).collect() + } + + #[test] + fn a_reachable_unblocked_cluster_is_ready() { + let readiness = readiness_verdict(Ok("green".into()), Ok(vec![])); + assert!(readiness.ready); + assert_eq!(readiness.reason, None, "a ready verdict carries no reason"); + // Still reported when ready: yellow is the ordinary shape of a + // replicated cluster, and a probe that only ever printed green + // would be no better than the constant #3317 replaced. + assert_eq!(readiness.cluster, "green"); + assert!(readiness.write_blocked.is_empty()); + } + + #[test] + fn yellow_stays_ready() { + // Yellow is unassigned *replicas*, which costs this tier nothing. + // Gating on it would take the backend not-ready on every rolling + // restart and every replica relocation. + assert!(readiness_verdict(Ok("yellow".into()), Ok(vec![])).ready); + } + + #[test] + fn an_unreachable_cluster_is_not_ready_and_says_why() { + // The bug in #3317's report, in the shape it took there: a backend + // that cannot reach Elasticsearch still answering healthy. + let readiness = readiness_verdict(Err(anyhow::anyhow!("connection refused")), Ok(vec![])); + assert!(!readiness.ready); + assert_eq!(readiness.cluster, "unreachable"); + let reason = readiness.reason.expect("not-ready must carry a reason"); + assert!(reason.contains("unreachable"), "{reason}"); + assert!(reason.contains("connection refused"), "{reason}"); + } + + #[test] + fn unreachable_outranks_a_write_block_report() { + // A block reading taken against a cluster we could not reach is not + // a fact about that cluster. Reporting it as the cause would be a + // guess, and it would be the wrong one to send an operator after. + let readiness = readiness_verdict( + Err(anyhow::anyhow!("no route to host")), + Ok(blocked(&["dashboard-config-v1"])), + ); + assert!(!readiness.ready); + assert!( + readiness.write_blocked.is_empty(), + "an unreachable cluster has no block state to report, got {:?}", + readiness.write_blocked + ); + assert!(readiness.reason.unwrap().contains("unreachable")); + } + + #[test] + fn a_write_block_is_not_ready_and_names_the_indices() { + // The disk-flood-stage / operator-block case: ES answers health + // fine and the block is the only evidence there is. Names rather + // than counts, because the point is to be actionable. + let readiness = readiness_verdict( + Ok("green".into()), + Ok(blocked(&["dashboard-config-v1", "dashboard-users-v1"])), + ); + assert!(!readiness.ready, "a green cluster is not the same as a writable one"); + assert_eq!(readiness.cluster, "green"); + assert_eq!(readiness.write_blocked, blocked(&["dashboard-config-v1", "dashboard-users-v1"])); + let reason = readiness.reason.expect("not-ready must carry a reason"); + assert!(reason.contains("dashboard-config-v1"), "{reason}"); + assert!(reason.contains("dashboard-users-v1"), "{reason}"); + assert!( + reason.contains("_cat/allocation"), + "the reason has to point at the two causes it cannot tell apart: {reason}" + ); + } + + #[test] + fn red_is_not_ready_even_with_nothing_blocked() { + let readiness = readiness_verdict(Ok("red".into()), Ok(vec![])); + assert!(!readiness.ready); + assert!(readiness.reason.unwrap().contains("red")); + } + + #[test] + fn a_probe_refused_itself_is_not_ready() { + // Reachable enough to answer health, but the settings call itself + // failed. "Ready" would be an answer this endpoint has no basis + // for -- it asked whether writes are permitted and was not told. + let readiness = + readiness_verdict(Ok("green".into()), Err(anyhow::anyhow!("403 forbidden"))); + assert!(!readiness.ready); + assert!(readiness.reason.unwrap().contains("refused the readiness probe")); + } + + #[test] + fn an_unknown_color_is_not_treated_as_red_or_green() { + // Neither branch claims it. The verdict is ready because no + // condition was met -- but `cluster` carries the honest "unknown" + // for the reader, rather than the endpoint inventing a color it + // was not told. + let readiness = readiness_verdict(Ok("unknown".into()), Ok(vec![])); + assert!(readiness.ready); + assert_eq!(readiness.cluster, "unknown"); + } +} diff --git a/arcane/home/honeypot-dashboard/backend-service/src/obs.rs b/arcane/home/honeypot-dashboard/backend-service/src/obs.rs index 0b779f39..8aa22eae 100644 --- a/arcane/home/honeypot-dashboard/backend-service/src/obs.rs +++ b/arcane/home/honeypot-dashboard/backend-service/src/obs.rs @@ -64,7 +64,14 @@ const MAX_SINK_BYTES: u64 = 25 << 20; pub fn family_for_path(path: &str) -> String { let mut segments = path.split('/').filter(|s| !s.is_empty()); match (segments.next(), segments.next(), segments.next()) { + // The three probe routes, each its own family rather than folded + // into "other" (#3317): /readyz answers 503 whenever Elasticsearch + // is not ready, so an operator watching the request-rate series + // needs to be able to tell a readiness probe's failures from an + // unrelated path that happens to be misshapen. (Some("healthz"), None, None) => "healthz".to_string(), + (Some("livez"), None, None) => "livez".to_string(), + (Some("readyz"), None, None) => "readyz".to_string(), (Some("metrics"), None, None) => "metrics".to_string(), (Some("api"), Some("v1"), Some(family)) => { // Only clean single-segment names become labels; anything odd @@ -292,8 +299,9 @@ async fn append_line( } /// Outermost middleware: metrics + request-id echo/span + durable line for -/// everything this tier serves, healthz and metrics included (the probe's -/// own behavior is part of the observability story too). +/// everything this tier serves, the liveness/readiness probes and metrics +/// included (the probes' own behavior is part of the observability story +/// too -- /readyz's 503s are the record of an Elasticsearch outage). pub async fn observe( State(state): State, headers: HeaderMap, @@ -349,10 +357,11 @@ pub async fn observe( response } -/// GET /metrics — deliberately unauthenticated exactly like /healthz: both -/// are reachable only on LISTEN_ADDR, which is the internal docker network -/// (Traefik publishes the BFF tier, not this listener). Adding auth here -/// would just mean operators punt and scrape over SSH tunnels anyway. +/// GET /metrics — deliberately unauthenticated exactly like /healthz, +/// /livez and /readyz: all of them are reachable only on LISTEN_ADDR, +/// which is the internal docker network (Traefik publishes the BFF tier, +/// not this listener). Adding auth here would just mean operators punt and +/// scrape over SSH tunnels anyway. pub async fn metrics_route(State(state): State) -> Response { ( [(axum::http::header::CONTENT_TYPE, "text/plain; version=0.0.4")], @@ -368,6 +377,8 @@ mod tests { #[test] fn families_match_the_tier_path_shapes() { assert_eq!(family_for_path("/healthz"), "healthz"); + assert_eq!(family_for_path("/livez"), "livez"); + assert_eq!(family_for_path("/readyz"), "readyz"); assert_eq!(family_for_path("/metrics"), "metrics"); assert_eq!(family_for_path("/api/v1/store/ml-anomalies?offset=0"), "store"); assert_eq!(family_for_path("/api/v1/live"), "live"); diff --git a/docs/OPERATIONS.md b/docs/OPERATIONS.md index 99a54057..c85a4338 100644 --- a/docs/OPERATIONS.md +++ b/docs/OPERATIONS.md @@ -231,6 +231,78 @@ commands/credentials, payloads, enriched IDS alerts, and ingest failures. - Dionaea/Conpot write their own JSON into the shared volume for jq/ELK; the live dashboard ingests them alongside Cowrie, multipot, HTTP, and Suricata. +## Service health contract (backend-service) + +`apiary-backend` (the Rust tier behind every `/api/v1` route) answers two +different questions on two different endpoints. They used to be one endpoint +with a constant answer: `/healthz` returned `{"ok": true, "es": }`, +where `ok` was literally hardcoded to `true`, so a backend that could not +reach Elasticsearch at all still told its own healthcheck, and anything else +that probed it, that it was healthy. The `#3283` ingest outage would not have +appeared there. Split into: + +| Endpoint | Question | Touches Elasticsearch | Non-200 | +|---|---|---|---| +| `/livez` | Is the process up and serving? | **No** | never | +| `/healthz` | Same handler as `/livez`, historical name | **No** | never | +| `/readyz` | Can it actually do its job? | Yes (2 probes) | 503 + `reason` | + +```console +$ curl -s http://backend-service:8081/livez +{"live":true,"built":"2026-09-27T01:46:34+00:00"} + +$ curl -s http://backend-service:8081/readyz +{"ready":true,"cluster":"green","write_blocked":[]} + +$ curl -s -o /dev/null -w '%{http_code}\n' http://backend-service:8081/readyz # during an ES outage +503 +``` + +**`/livez` is what the container `HEALTHCHECK` curls, and it must stay that +way.** A probe that can block on Elasticsearch converts that dependency's +outage into a restart loop of a container that was never the problem. The +image's own comment on the `HEALTHCHECK` line says this too. `/healthz` is +kept as an alias because the port-test harness +(`arcane/home/honeypot-dashboard/port-tests/lib.sh`) and the ops scripts +already use that name; it is a rename-with-a-twist, not a rename. + +### What `/readyz` actually checks + +1. **Reachable** — `GET /_cluster/health` answers. This is the check the old + `es: ` field gestured at; the difference is that failing it now + produces a 503. +2. **Not red** — red means unassigned primaries, against which reads and + writes both fail. **Yellow stays ready**: it means unassigned *replicas*, + which is the ordinary shape of a replicated cluster during a rolling + restart, and gating on it would mark the backend not-ready on every deploy. +3. **Writable** — no `index.blocks.write` set on any of the index families + this tier writes to (the list is `es::WRITE_TARGET_FAMILIES` in + `backend-service/src/es.rs`, `dashboard-*` plus the bundled worker loops' + own families). This is what catches the flood-stage disk watermark, which + sets the block on every index at once, and an operator's + `PUT //_block/write`. + +The response names the blocked indices rather than counting them, and the +`reason` string points at `_cat/allocation` because the endpoint **cannot +tell those two causes apart** — an honest limit, stated rather than papered +over. A *missing* index is not a block: every dashboard-owned index is created +lazily on its first write, so a fresh cluster correctly reads ready. + +Both probes run concurrently under a 5s deadline +(`READINESS_TIMEOUT`), shorter than the shared client's 30s, because a probe +that blocks for 30s is indistinguishable from the outage it exists to report. + +### What to point at what + +- **Docker/compose healthcheck, Traefik, uptime pings** → `/livez` (or + `/healthz`). Never `/readyz`; see above. +- **Deploy verification, diagnostics, "is ingest actually working?"** → + `/readyz`, and treat 503 as a real answer, not a transport error. This is + the endpoint that would have shown `#3283`. +- **"Why is the dashboard empty?"** → `/api/v1/source-health` (the + per-sensor freshness page). `/readyz` says the backend cannot write; only + source-health says whether events are arriving. + ## Disk space monitoring `hp-disk-space-monitor` (`arcane/home/honeypot-utilities/analysis/disk-space-check.sh`)