Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
14 changes: 14 additions & 0 deletions .github/workflows/quality.yml
Original file line number Diff line number Diff line change
Expand Up @@ -49,6 +49,20 @@ jobs:
- run: python scripts/check-public-leaks.py
- run: python scripts/validate-oidc-redirects.py

design-lab-readonly:
# #1828: the design lab serves variants against the real captured-data
# Elasticsearch, so its read-only guarantee is a safety property, not a
# convenience. The test drives the actual harness against a recording
# stand-in backend and fails if a write ever reaches it.
name: Design lab is read-only
runs-on: ${{ (github.event_name == 'workflow_dispatch' && inputs.use_self_hosted_runner) && 'self-hosted' || 'ubuntu-latest' }}
steps:
- uses: actions/checkout@v7
- uses: actions/setup-node@v7
with:
node-version: "22"
- run: node --test branding/design-lab/lab.test.mjs

go-fmt:
name: Go formatting
runs-on: ${{ (github.event_name == 'workflow_dispatch' && inputs.use_self_hosted_runner) && 'self-hosted' || 'ubuntu-latest' }}
Expand Down
127 changes: 100 additions & 27 deletions arcane/home/honeypot-dashboard/backend-service/src/stores.rs
Original file line number Diff line number Diff line change
Expand Up @@ -51,22 +51,50 @@ fn bad_gateway(error: anyhow::Error) -> (StatusCode, String) {
(StatusCode::BAD_GATEWAY, error.to_string())
}

/// The sort every store list is paged by: the store's own field, then a
/// deterministic tiebreak.
///
/// Both keys are load-bearing.
///
/// `unmapped_type` keeps a store whose index does not exist yet from
/// erroring -- but it also means a *misspelled* field sorts every document
/// as null instead of failing. That is how three lists shipped in no order
/// at all: ml-anomalies asked for "timestamp" where the worker writes
/// "@timestamp", auth-events asked for "last_seen" where Keycloak writes
/// "@timestamp", and static-analysis asked for "Analysis.GeneratedUTC",
/// which its documents do not carry in any spelling. It has to match the
/// field's real type, or a keyword sort reintroduces the same silent no-op.
///
/// The tiebreak is what makes paging correct rather than merely tidy.
/// `from`/`size` over a sort with ties leaves the order within a tie
/// undefined, so the same document can come back on two pages while another
/// is never shown at all -- and an all-null sort is one giant tie. `_doc` is
/// the cheapest total order Elasticsearch offers.
fn sort_spec(sort_field: &str, unmapped_type: &str) -> Value {
json!([
{sort_field: {"order": "desc", "unmapped_type": unmapped_type}},
{"_doc": {"order": "asc"}}
])
}

/// Generic hits page: {"total": N, "rows": [ _source... ]}. The BFF/routes
/// know each store's shape; this tier guarantees ordering + paging.
async fn store_page(
state: &AppState,
indices: &[&str],
sort_field: &str,
unmapped_type: &str,
q: &StoreQuery,
extra_filter: Option<Value>,
) -> anyhow::Result<Value> {
store_page_excluding(state, indices, sort_field, q, extra_filter, &[]).await
store_page_excluding(state, indices, sort_field, unmapped_type, q, extra_filter, &[]).await
}

async fn store_page_excluding(
state: &AppState,
indices: &[&str],
sort_field: &str,
unmapped_type: &str,
q: &StoreQuery,
extra_filter: Option<Value>,
excludes: &[&str],
Expand All @@ -83,7 +111,7 @@ async fn store_page_excluding(
"from": q.offset,
"size": size,
"track_total_hits": true,
"sort": [{sort_field: {"order": "desc", "unmapped_type": "date"}}],
"sort": sort_spec(sort_field, unmapped_type),
"query": query
});
if !excludes.is_empty() {
Expand Down Expand Up @@ -112,7 +140,7 @@ pub async fn campaigns(
State(state): State<AppState>,
Query(q): Query<StoreQuery>,
) -> Result<Json<Value>, (StatusCode, String)> {
store_page(&state, &["campaigns-v1"], "score", &q, None)
store_page(&state, &["campaigns-v1"], "score", "long", &q, None)
.await
.map(Json)
.map_err(bad_gateway)
Expand All @@ -122,7 +150,7 @@ pub async fn clusters(
State(state): State<AppState>,
Query(q): Query<StoreQuery>,
) -> Result<Json<Value>, (StatusCode, String)> {
store_page(&state, &["attacker-clusters-v1"], "events", &q, None)
store_page(&state, &["attacker-clusters-v1"], "events", "long", &q, None)
.await
.map(Json)
.map_err(bad_gateway)
Expand All @@ -132,7 +160,7 @@ pub async fn attackers(
State(state): State<AppState>,
Query(q): Query<StoreQuery>,
) -> Result<Json<Value>, (StatusCode, String)> {
store_page(&state, &["attackers-v1"], "events", &q, None)
store_page(&state, &["attackers-v1"], "events", "long", &q, None)
.await
.map(Json)
.map_err(bad_gateway)
Expand Down Expand Up @@ -224,7 +252,7 @@ pub async fn alerts(
State(state): State<AppState>,
Query(q): Query<StoreQuery>,
) -> Result<Json<Value>, (StatusCode, String)> {
store_page(&state, &["dashboard-alert-state-v1"], "LastSeen", &q, None)
store_page(&state, &["dashboard-alert-state-v1"], "LastSeen", "date", &q, None)
.await
.map(Json)
.map_err(bad_gateway)
Expand All @@ -234,7 +262,7 @@ pub async fn payloads(
State(state): State<AppState>,
Query(q): Query<StoreQuery>,
) -> Result<Json<Value>, (StatusCode, String)> {
store_page(&state, &["dashboard-payload-inventory-v1"], "MtimeUTC", &q, None)
store_page(&state, &["dashboard-payload-inventory-v1"], "MtimeUTC", "date", &q, None)
.await
.map(Json)
.map_err(bad_gateway)
Expand Down Expand Up @@ -270,40 +298,53 @@ pub async fn generic(
Query(q): Query<StoreQuery>,
) -> Result<Json<Value>, (StatusCode, String)> {
// (index, sort field, heavy fields excluded from list responses).
let (index, sort, excludes): (&str, &str, &[&str]) = match name.as_str() {
// (index, sort field, that field's type, heavy fields excluded from
// list responses). The type is not decoration -- see the sort spec in
// store_page_excluding for what a wrong one costs.
let (index, sort, sort_type, excludes): (&str, &str, &str, &[&str]) = match name.as_str() {
// #1611 workstream E.9: `error`, `details.username`, and
// `details.redirect_uri` are already present here (no excludes) —
// the workstream's ask is first-class *columns* for them on the
// auth-events.tsx table, a frontend-only change; this passthrough
// already carries every field they'd need.
"auth-events" => ("auth-failure-events", "last_seen", &[]),
// "last_seen" is not a field these documents have -- Keycloak's
// event stream writes @timestamp -- so this list came back in no
// order at all until #1566.
"auth-events" => ("auth-failure-events", "@timestamp", "date", &[]),
// llm-worker output; index may not exist yet (ignore_unavailable).
"llm-analysis" => ("llm-analysis", "@timestamp", &[]),
"ml-anomalies" => ("ml-anomalies", "timestamp", &[]),
"llm-analysis" => ("llm-analysis", "@timestamp", "date", &[]),
// Likewise "timestamp": the ml-worker writes @timestamp. Measured
// on the live index, the first page mixed 20:58, 20:59 and 20:57
// rows in that order (#1566).
"ml-anomalies" => ("ml-anomalies", "@timestamp", "date", &[]),
// Matches dashboard/agent_campaigns.go's own refreshAgentCampaigns
// sort (`sort=@timestamp:asc`) — the campaign-verdict documents
// this index holds have no last_seen field at all (see the
// agent-intrusion-worker port's build_campaign_verdict, #1610).
"agent-campaigns" => ("agent-intrusion-campaigns", "@timestamp", &[]),
"canarytokens" => ("dashboard-canarytokens-v1", "created_at", &[]),
"problem-reports" => ("dashboard-problem-reports-v1", "submitted_at", &["dom_snapshot"]),
"dead-letters" => ("dead-letter-honeypot", "@timestamp", &[]),
"yara" => ("yara-analysis-v1", "@timestamp", &[]),
"sandbox-runs" => ("sandbox-analysis-v1", "@timestamp", &[]),
"ghidra-runs" => ("ghidra-analysis-v1", "@timestamp", &[]),
"static-analysis" => ("dashboard-static-analysis-v1", "Analysis.GeneratedUTC", &[]),
"agent-campaigns" => ("agent-intrusion-campaigns", "@timestamp", "date", &[]),
"canarytokens" => ("dashboard-canarytokens-v1", "created_at", "date", &[]),
"problem-reports" => ("dashboard-problem-reports-v1", "submitted_at", "date", &["dom_snapshot"]),
"dead-letters" => ("dead-letter-honeypot", "@timestamp", "date", &[]),
"yara" => ("yara-analysis-v1", "@timestamp", "date", &[]),
"sandbox-runs" => ("sandbox-analysis-v1", "@timestamp", "date", &[]),
"ghidra-runs" => ("ghidra-analysis-v1", "@timestamp", "date", &[]),
// These documents carry only Analysis and Fingerprint -- there is
// no GeneratedUTC, and no time field at all, so there is nothing to
// order by chronologically. Fingerprint is the one mapped field, so
// it is what makes paging deterministic instead of arbitrary.
"static-analysis" => ("dashboard-static-analysis-v1", "Fingerprint", "keyword", &[]),
// Result families that may not exist yet on a given deployment
// (ignore_unavailable keeps them safe): revdeck, CAPE, GitHub.
"revdeck" => ("revdeck-analysis-v1", "@timestamp", &[]),
"cape" => ("cape-analysis-v1", "@timestamp", &[]),
"github-analysis" => ("github-analysis-v1", "@timestamp", &[]),
"workbench-runs" => ("dashboard-workbench-runs-v1", "created_at", &[]),
"generated-reports" => ("dashboard-generated-reports-v1", "created_at", &["pdf_base64"]),
"report-definitions" => ("dashboard-reports-definitions-v1", "updated", &[]),
"intelligence" => ("dashboard-intelligence-archive-v1", "generated", &[]),
"revdeck" => ("revdeck-analysis-v1", "@timestamp", "date", &[]),
"cape" => ("cape-analysis-v1", "@timestamp", "date", &[]),
"github-analysis" => ("github-analysis-v1", "@timestamp", "date", &[]),
"workbench-runs" => ("dashboard-workbench-runs-v1", "created_at", "date", &[]),
"generated-reports" => ("dashboard-generated-reports-v1", "created_at", "date", &["pdf_base64"]),
"report-definitions" => ("dashboard-reports-definitions-v1", "updated", "date", &[]),
"intelligence" => ("dashboard-intelligence-archive-v1", "generated", "date", &[]),
_ => return Err((StatusCode::NOT_FOUND, format!("unknown store {name}"))),
};
store_page_excluding(&state, &[index], sort, &q, None, excludes)
store_page_excluding(&state, &[index], sort, sort_type, &q, None, excludes)
.await
.map(Json)
.map_err(bad_gateway)
Expand Down Expand Up @@ -342,3 +383,35 @@ pub async fn generic_delete(
let deleted = state.es.delete_by_query("dead-letter-honeypot", query).await.map_err(bad_gateway)?;
Ok(Json(json!({"deleted": deleted})))
}


#[cfg(test)]
mod sort_tests {
use super::sort_spec;

#[test]
fn the_store_field_leads_and_carries_its_own_type() {
// A date sort and a keyword sort must not both claim "date": an
// unmapped_type that disagrees with the field is the same silent
// no-op as a misspelled field name.
let dated = sort_spec("@timestamp", "date");
assert_eq!(dated[0]["@timestamp"]["order"], "desc");
assert_eq!(dated[0]["@timestamp"]["unmapped_type"], "date");

let keyed = sort_spec("Fingerprint", "keyword");
assert_eq!(keyed[0]["Fingerprint"]["unmapped_type"], "keyword");
}

#[test]
fn every_sort_has_a_tiebreak_so_paging_cannot_repeat_or_skip() {
// The bug this guards. from/size over a sort with ties leaves the
// order inside a tie undefined, so a document can appear on two
// pages while another never appears -- and the measured live data
// ties hard (48 ml-anomaly rows on one IP in a single second).
for (field, kind) in [("@timestamp", "date"), ("score", "long"), ("Fingerprint", "keyword")] {
let spec = sort_spec(field, kind);
assert_eq!(spec.as_array().map(Vec::len), Some(2), "{field} lost its tiebreak");
assert_eq!(spec[1]["_doc"]["order"], "asc", "{field}'s tiebreak is not a total order");
}
}
}
4 changes: 4 additions & 0 deletions arcane/home/honeypot-dashboard/frontend-next/.gitignore
Original file line number Diff line number Diff line change
Expand Up @@ -4,3 +4,7 @@ node_modules/
.tanstack/
dist/
.env

# #1828: the design lab symlinks variant stylesheets in here at start-up
# and clears it on exit. Never content, always disposable.
public/static/lab/
Original file line number Diff line number Diff line change
Expand Up @@ -208,6 +208,9 @@ const MARKER_RADIUS_PX = 6
export function AttackMap({ points }: { points: MapPoint[] | null }) {
const containerRef = useRef<HTMLDivElement>(null)
const mapRef = useRef<import('leaflet').Map | null>(null)
// Torn down alongside the map itself; the ResizeObserver outlives the
// async import that creates it, so it needs its own handle.
const cleanupRef = useRef<(() => void) | null>(null)

useEffect(() => {
const container = containerRef.current
Expand All @@ -217,8 +220,32 @@ export function AttackMap({ points }: { points: MapPoint[] | null }) {
const L = (await import('leaflet')).default
await import('leaflet/dist/leaflet.css')
if (disposed || mapRef.current) return
const map = L.map(container, { worldCopyJump: true, minZoom: 1 }).setView([25, 10], 2)
const map = L.map(container, { worldCopyJump: true, minZoom: 1 })
mapRef.current = map

// #1565: the map is built before the card has settled at its final
// width, so leaflet sizes its tile grid against whatever the container
// measured at construction time and never revisits it. Measured at an
// 1854px viewport, that left six 256px tile columns covering 1536px of
// a 1442px container from the wrong origin -- a 92px strip of bare
// card down the right edge, which reads as a rendering fault rather
// than as ocean.
//
// invalidateSize() re-measures and recomputes the grid, which is the
// whole fix; the setView after it re-centres on the world at the same
// zoom the card has always used, so the full world stays visible
// rather than being cropped to fill the width.
const fitWorld = () => {
map.invalidateSize({ animate: false })
map.setView([25, 10], 2, { animate: false })
}
fitWorld()

// A card that changes width -- a sidebar opening, a window resize,
// the print stylesheet -- puts the strip straight back otherwise.
const observer = new ResizeObserver(fitWorld)
observer.observe(container)
cleanupRef.current = () => observer.disconnect()
L.tileLayer('https://tile.openstreetmap.org/{z}/{x}/{y}.png', {
attribution: '© OpenStreetMap contributors',
}).addTo(map)
Expand Down Expand Up @@ -259,6 +286,8 @@ export function AttackMap({ points }: { points: MapPoint[] | null }) {
})()
return () => {
disposed = true
cleanupRef.current?.()
cleanupRef.current = null
mapRef.current?.remove()
mapRef.current = null
}
Expand Down
Original file line number Diff line number Diff line change
@@ -0,0 +1,95 @@
import { describe, expect, it } from 'vitest'
import { collapseRuns, foldedCount, idsFor } from './mlGrouping'
import type { StoreRow } from '../components/StoreList'

function row(id: string, ip: string, ts: string, score: number): StoreRow {
return { _doc_id: id, src_ip: ip, '@timestamp': ts, composite_score: score }
}

describe('collapseRuns', () => {
it('folds a burst from one address in one second into a single row', () => {
// The measured shape of the bug: 48 rows for one IP inside one second,
// every one scoring exactly 0.8000.
const burst = Array.from({ length: 48 }, (_, i) =>
row(`d${i}`, '153.75.87.176', `2026-08-24T20:59:23.${String(i).padStart(3, '0')}Z`, 0.8),
)
const out = collapseRuns(burst)
expect(out).toHaveLength(1)
expect(foldedCount(out[0])).toBe(48)
expect(idsFor(out[0])).toHaveLength(48)
})

it('keeps different addresses apart even in the same second', () => {
const out = collapseRuns([
row('a', '1.1.1.1', '2026-08-24T20:59:23.100Z', 0.8),
row('b', '2.2.2.2', '2026-08-24T20:59:23.200Z', 0.8),
])
expect(out).toHaveLength(2)
})

it('keeps different seconds apart even from the same address', () => {
const out = collapseRuns([
row('a', '1.1.1.1', '2026-08-24T20:59:23.900Z', 0.8),
row('b', '1.1.1.1', '2026-08-24T20:59:24.000Z', 0.8),
])
expect(out).toHaveLength(2)
})

it('keeps materially different scores apart', () => {
const out = collapseRuns([
row('a', '1.1.1.1', '2026-08-24T20:59:23.100Z', 0.8),
row('b', '1.1.1.1', '2026-08-24T20:59:23.200Z', 0.95),
])
expect(out).toHaveLength(2)
})

it('does not chain a drift into one arbitrarily wide group', () => {
// Each row is within the epsilon of the one before it, but the run spans
// 0.05 end to end. Comparing against the representative rather than the
// predecessor is what stops that becoming a single "group".
const drift = Array.from({ length: 6 }, (_, i) =>
row(`d${i}`, '1.1.1.1', '2026-08-24T20:59:23.100Z', 0.8 + i * 0.009),
)
const out = collapseRuns(drift)
expect(out.length).toBeGreaterThan(1)
})

it('only folds rows that are actually adjacent', () => {
// A lookalike separated by an unrelated row is a different moment, and
// is left alone rather than reached across for.
const out = collapseRuns([
row('a', '1.1.1.1', '2026-08-24T20:59:23.100Z', 0.8),
row('x', '9.9.9.9', '2026-08-24T20:59:23.150Z', 0.4),
row('b', '1.1.1.1', '2026-08-24T20:59:23.200Z', 0.8),
])
expect(out).toHaveLength(3)
})

it('never folds a row that has no score', () => {
const out = collapseRuns([
{ _doc_id: 'a', src_ip: '1.1.1.1', '@timestamp': '2026-08-24T20:59:23.100Z' },
{ _doc_id: 'b', src_ip: '1.1.1.1', '@timestamp': '2026-08-24T20:59:23.200Z' },
])
expect(out).toHaveLength(2)
})

it('leaves an ungrouped row reporting itself, so callers need no special case', () => {
const out = collapseRuns([row('a', '1.1.1.1', '2026-08-24T20:59:23.100Z', 0.8)])
expect(foldedCount(out[0])).toBe(1)
expect(idsFor(out[0])).toEqual(['a'])
})

it('does not mutate the rows it was given', () => {
const input = [
row('a', '1.1.1.1', '2026-08-24T20:59:23.100Z', 0.8),
row('b', '1.1.1.1', '2026-08-24T20:59:23.200Z', 0.8),
]
const snapshot = JSON.parse(JSON.stringify(input))
collapseRuns(input)
expect(input).toEqual(snapshot)
})

it('handles an empty page', () => {
expect(collapseRuns([])).toEqual([])
})
})
Loading
Loading