diff --git a/arcane/home/honeypot-dashboard/backend-service/src/events.rs b/arcane/home/honeypot-dashboard/backend-service/src/events.rs index fdf43898d..e895429d1 100644 --- a/arcane/home/honeypot-dashboard/backend-service/src/events.rs +++ b/arcane/home/honeypot-dashboard/backend-service/src/events.rs @@ -1035,4 +1035,115 @@ mod session_scope_tests { "a null presence must stay null all the way to the response" ); } + + // ── #3212: the capture fields reach the presentation layer ────────────── + // + // The sensor emits them; the normalization pass keeps them. This is the + // last hop of the contract: a document carrying them, as stored in the + // index, still carries them when the read path hands the record to the + // browser -- and the byte-safe evidence field is not a way for a + // credential to get there. + + /// The shape the sensor now writes for a clipped, credential-bearing, + /// Java-marked POST: the redacted body, its bounded base64 head, the + /// prefix hash under its scope label, the declared length, and the + /// marker observation. `body_b64` is the base64 of `body` exactly as the + /// sensor derives the two from one redacted string, so the event's own + /// internal agreement can be asserted rather than assumed. + fn stored_capture_event() -> Value { + json!({ + "@timestamp": "2026-09-27T00:00:00Z", + "event": {"sensor": "http-honeypot"}, + "source": {"ip": "203.0.113.9"}, + "honeypot": { + "sensor": "http-honeypot", + "src_ip": "203.0.113.9", + "method": "POST", + "path": "/api/v1/login", + "username": "admin", + "body": format!("username=admin&password={}&next=%2F", secrets_boundary::REDACT_MARKER), + "headers": {"authorization": format!("Bearer {SECRET}")}, + "body_capture_state": "truncated", + "body_captured_bytes": 65536, + "body_encoding": "base64", + "body_b64": "dXNlcm5hbWU9YWRtaW4mcGFzc3dvcmQ9W3JlZGFjdGVkXSZuZXh0PSUyRg==", + "body_sha256": "05d129712d910c85a45c74ef9f1b825068ddc953e417ba75781e58fa40625eeb", + "body_sha256_scope": "captured-prefix-redacted", + "body_declared_bytes": 4194304, + "java_marker": "stream-magic" + } + }) + } + + #[test] + fn the_capture_fields_survive_into_the_record() { + // row_from_source feeds the explorer list, the SSE live stream and + // both CSV exports through one row shape, so this assertion covers + // every surface the record is rendered on. Serialized rather than + // field-checked for the same reason the #3213 test is: a field + // checked by name is only as good as the name someone remembered. + let row = row_from_source(&stored_capture_event()); + let record = row.record.to_string(); + for fragment in [ + "\"body_capture_state\":\"truncated\"", + "\"body_captured_bytes\":65536", + "\"body_encoding\":\"base64\"", + "\"body_b64\":\"dXNlcm5hbWU9YWRtaW4mcGFzc3dvcmQ9W3JlZGFjdGVkXSZuZXh0PSUyRg==\"", + "\"body_sha256_scope\":\"captured-prefix-redacted\"", + "\"body_declared_bytes\":4194304", + "\"java_marker\":\"stream-magic\"", + ] { + assert!( + record.contains(fragment), + "{fragment} did not survive into the record: {record}" + ); + } + } + + #[test] + fn the_evidence_fields_carry_no_credential_material() { + // The #3213 test proves a stored password never reaches a row. This + // extends the same claim to #3212's evidence: the byte-safe head, the + // prefix hash and the declared length must not become a second, + // differently-encoded copy of a submitted secret -- a base64 blob is + // not a disclosure control. Checked on the serialized row, which is + // what a browser actually receives. + let row = row_from_source(&stored_capture_event()); + let rendered = serde_json::to_string(&row).unwrap(); + assert!( + !rendered.contains(SECRET), + "a captured credential reached the response: {rendered}" + ); + // The header credential has a second encoding -- its own base64, + // which is what "Basic " puts on the wire. The + // #3213 test established that this channel is checked too. + let mut header_encoding = String::new(); + header_encoding.push_str("Basic "); + header_encoding.push_str( + &base64::Engine::encode( + &base64::engine::general_purpose::STANDARD, + format!("admin:{SECRET}"), + ), + ); + assert!( + !rendered.contains(&header_encoding), + "a bearer credential's Basic encoding reached the response: {rendered}" + ); + + // And the boundary must not have destroyed the evidence while + // keeping it secret: body_b64 is not a scrub target (the sensor + // redacts before the field is built), so it has to survive intact + // and still decode to the event's own redacted body. + let hp = &row.record["honeypot"]; + let decoded = + base64::Engine::decode(&base64::engine::general_purpose::STANDARD, hp["body_b64"].as_str().unwrap()) + .unwrap(); + let expected = format!("username=admin&password={}&next=%2F", secrets_boundary::REDACT_MARKER); + assert_eq!(decoded, expected.as_bytes(), "the evidence head was corrupted on the way out"); + assert_eq!( + hp["body_sha256"].as_str().unwrap(), + "05d129712d910c85a45c74ef9f1b825068ddc953e417ba75781e58fa40625eeb", + "the hash covers the redacted captured prefix" + ); + } } diff --git a/arcane/home/honeypot-dashboard/backend-service/src/ip_enrichment/sensors.rs b/arcane/home/honeypot-dashboard/backend-service/src/ip_enrichment/sensors.rs index 780fabf81..2621ffefb 100644 --- a/arcane/home/honeypot-dashboard/backend-service/src/ip_enrichment/sensors.rs +++ b/arcane/home/honeypot-dashboard/backend-service/src/ip_enrichment/sensors.rs @@ -958,6 +958,61 @@ mod tests { use super::*; use serde_json::json; + // ── #3212: the capture fields survive the normalization pass ──────────── + // + // The sensor-side tests prove the fields are emitted and byte-exact. This + // proves they survive the one pass between the sensor and Elasticsearch: + // http-honeypot has no bespoke enrich function, so the generic + // enrich_line is the whole normalization, and filebeat tails the + // enriched copy rather than the raw log (filebeat.yml's honeypot-json + // input: only /logs/enriched/*.json is tailed for this decoy). A field + // this pass dropped would never reach the index, and the backend's + // record inspector -- which renders _source verbatim -- would have + // nothing to show for it. + + #[test] + fn the_capture_fields_survive_the_normalization_pass() { + let line = json!({ + "sensor": "http-honeypot", + "src_ip": "203.0.113.9", + "src_port": 44321, + "time": "2026-09-27T12:00:00Z", + "body_capture_state": "truncated", + "body_captured_bytes": 65536, + "body_encoding": "base64", + "body_b64": "rO0ABQ==", + "body_sha256": "0123456789abcdef0123456789abcdef0123456789abcdef0123456789abcdef", + "body_sha256_scope": "captured-prefix-redacted", + "body_declared_bytes": 4194304, + "body_read_error": "unexpected EOF", + "java_marker": "stream-magic" + }) + .to_string(); + let (out, resolved) = + enrich_line(line.as_bytes(), &ViaMap::new(), &ViaMap::new(), "http-honeypot"); + assert!(resolved, "a complete event is never queued for retry"); + let out: Value = serde_json::from_slice(&out).unwrap(); + for key in [ + "body_capture_state", + "body_captured_bytes", + "body_encoding", + "body_b64", + "body_sha256", + "body_sha256_scope", + "body_declared_bytes", + "body_read_error", + "java_marker", + ] { + assert!( + out.get(key).is_some(), + "{key} did not survive the normalization pass: {out}" + ); + } + assert_eq!(out["body_capture_state"], json!("truncated")); + assert_eq!(out["body_sha256_scope"], json!("captured-prefix-redacted")); + assert_eq!(out["java_marker"], json!("stream-magic")); + } + fn vm_with(port: i64, ip: &str) -> ViaMap { let mut m = ViaMap::new(); // at 0 so these keep exercising the join itself rather than diff --git a/arcane/home/honeypot-http/http-honeypot/binary_evidence_test.go b/arcane/home/honeypot-http/http-honeypot/binary_evidence_test.go new file mode 100644 index 000000000..2fdf18969 --- /dev/null +++ b/arcane/home/honeypot-http/http-honeypot/binary_evidence_test.go @@ -0,0 +1,801 @@ +package main + +import ( + "bytes" + "crypto/sha256" + "encoding/base64" + "encoding/hex" + "encoding/json" + "errors" + "go/parser" + "go/token" + "io" + "net/http" + "net/http/httptest" + "os" + "strings" + "testing" + "unicode/utf8" +) + +// #3212's coverage, in the order the issue states it: a real binary body +// preserved byte-for-byte with its encoding recorded, truncation separated +// from a short body, the hash's scope labelled as what it is, and no +// deserialization of anything that arrived over the socket. +// +// Everything here is offline and inert: the fixtures are bytes this file +// builds, there is no valid object graph among them, and no test reaches the +// network or a running sensor. + +// loggedEvent drives one request through ServeHTTP and returns the decoded +// JSON line the sensor actually wrote, so every assertion below is about the +// data a downstream consumer gets rather than about a helper's return value. +func loggedEvent(t *testing.T, r *http.Request) map[string]any { + t.Helper() + var buf bytes.Buffer + s := &server{log: &logger{out: &buf}, sensor: "http-honeypot"} + s.ServeHTTP(httptest.NewRecorder(), r) + + line := strings.TrimSpace(buf.String()) + if line == "" { + t.Fatal("no event was logged") + } + // The tarpit writes nothing, but a served response does not either -- + // anything but a single line here means the shape under test changed. + if strings.Count(line, "\n") != 0 { + t.Fatalf("expected exactly one JSON line, got:\n%s", line) + } + var e map[string]any + if err := json.Unmarshal([]byte(line), &e); err != nil { + t.Fatalf("logged line is not valid JSON: %v\n%s", err, line) + } + return e +} + +func postRequest(body io.Reader, contentLength string) *http.Request { + r := httptest.NewRequest(http.MethodPost, "/index.php", body) + r.RemoteAddr = "203.0.113.9:54321" + if contentLength != "" { + // httptest derives ContentLength from the reader but does not write + // the header, and declaredLength() reads the claim off the wire. + r.Header.Set("Content-Length", contentLength) + } + return r +} + +func str(t *testing.T, e map[string]any, key string) string { + t.Helper() + v, ok := e[key].(string) + if !ok { + t.Fatalf("event has no string field %q: %v", key, e) + } + return v +} + +func sha256Hex(b []byte) string { + sum := sha256.Sum256(b) + return hex.EncodeToString(sum[:]) +} + +// TestShortBodyAndTruncatedBodyAreDistinguishable is the core of #3212: +// before the capture was described, "the attacker sent 40 bytes" and "we +// kept the first 64 KiB of a 4 MB upload" produced the same record, and +// nothing downstream could tell them apart. Asserted end to end, on the +// logged line, because the defect was in the data rather than in a helper. +func TestShortBodyAndTruncatedBodyAreDistinguishable(t *testing.T) { + short := bytes.Repeat([]byte("A"), 40) + bulk := bytes.Repeat([]byte("B"), 4<<20) // the issue's own example + + shortEvent := loggedEvent(t, postRequest(bytes.NewReader(short), "40")) + bulkEvent := loggedEvent(t, postRequest(bytes.NewReader(bulk), "4194304")) + + if got := str(t, shortEvent, "body_capture_state"); got != captureComplete { + t.Errorf("40-byte body: body_capture_state = %q, want %q", got, captureComplete) + } + if got := shortEvent["body_captured_bytes"]; got != float64(40) { + t.Errorf("40-byte body: body_captured_bytes = %v, want 40", got) + } + if got := shortEvent["body_declared_bytes"]; got != float64(40) { + t.Errorf("40-byte body: body_declared_bytes = %v, want 40", got) + } + + if got := str(t, bulkEvent, "body_capture_state"); got != captureTruncated { + t.Errorf("4 MB body: body_capture_state = %q, want %q", got, captureTruncated) + } + if got := bulkEvent["body_captured_bytes"]; got != float64(bodyCaptureMaxBytes) { + t.Errorf("4 MB body: body_captured_bytes = %v, want %d", got, bodyCaptureMaxBytes) + } + if got := bulkEvent["body_declared_bytes"]; got != float64(4<<20) { + t.Errorf("4 MB body: body_declared_bytes = %v, want %d", got, 4<<20) + } + if bulkEvent["body_read_error"] != nil { + t.Errorf("4 MB body: body_read_error = %v, want none -- it was capped, not failed", bulkEvent["body_read_error"]) + } + + // The two must not merely be labelled differently -- the whole point is + // that a consumer can tell them apart, which means the records differ. + if str(t, shortEvent, "body_sha256") == str(t, bulkEvent, "body_sha256") { + t.Error("a 40-byte body and a clipped 4 MB body hashed the same") + } + if str(t, shortEvent, "body_capture_state") == str(t, bulkEvent, "body_capture_state") { + t.Error("the two captures are indistinguishable in the logged data") + } +} + +// TestCaptureCapBoundaryIsExact pins the boundary the truncation flag depends +// on: exactly at the cap is a complete body, one byte over is a truncated +// one. captureBody reads cap+1 precisely so this can be decided, and a body +// that stops exactly on the limit is the case that cannot be told from a +// complete one without it. +func TestCaptureCapBoundaryIsExact(t *testing.T) { + cases := []struct { + name string + size int + state string + }{ + {"empty", 0, captureComplete}, + {"one byte", 1, captureComplete}, + {"exactly at the cap", bodyCaptureMaxBytes, captureComplete}, + {"one byte over the cap", bodyCaptureMaxBytes + 1, captureTruncated}, + {"far over the cap", bodyCaptureMaxBytes * 4, captureTruncated}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + got := captureBody(bytes.NewReader(bytes.Repeat([]byte("x"), tc.size))) + if state := got.state(); state != tc.state { + t.Fatalf("state() = %q, want %q", state, tc.state) + } + // Bounded means bounded, whatever arrived. + if len(got.Bytes) > bodyCaptureMaxBytes { + t.Fatalf("captured %d bytes, cap is %d", len(got.Bytes), bodyCaptureMaxBytes) + } + }) + } +} + +// TestReadErrorIsRecordedNotSwallowed covers the other half of the discarded +// io.ReadAll error: a read that fails under the cap is neither "all of it" +// nor "we stopped at the limit", and the state has to say which. +func TestReadErrorIsRecordedNotSwallowed(t *testing.T) { + sentinel := errors.New("connection reset by peer") + partial := bytes.Repeat([]byte("Z"), 100) + + e := loggedEvent(t, postRequest(&failingReader{data: partial, err: sentinel}, "")) + + if got := str(t, e, "body_capture_state"); got != captureUnknown { + t.Errorf("body_capture_state = %q, want %q", got, captureUnknown) + } + if got := str(t, e, "body_read_error"); got != sentinel.Error() { + t.Errorf("body_read_error = %q, want %q", got, sentinel) + } + if got := e["body_captured_bytes"]; got != float64(100) { + t.Errorf("body_captured_bytes = %v, want the 100 bytes that did arrive", got) + } + // The bytes that did arrive are still evidence, and still hashed. + if got := str(t, e, "body_sha256"); got != sha256Hex(partial) { + t.Errorf("body_sha256 = %q, want the hash of the 100 captured bytes", got) + } +} + +// failingReader hands over its bytes and then fails, which is what a +// truncated transfer looks like to a server: some data, then a transport +// error rather than a clean end of stream. +type failingReader struct { + data []byte + err error +} + +func (f *failingReader) Read(p []byte) (int, error) { + if len(f.data) > 0 { + n := copy(p, f.data) + f.data = f.data[n:] + return n, nil + } + return 0, f.err +} + +// binaryFixture is a real binary body: the Java stream magic followed by +// bytes that are not text. The 0x80 and 0xff are not valid UTF-8 in any +// position, the 0x00 is a NUL inside a payload, and the ED A0 80 triple is a +// surrogate half that UTF-8 forbids. Everything after the magic is inert +// filler -- there is no object graph here, only bytes. +func binaryFixture() []byte { + return append(append([]byte{}, javaStreamMagic...), + 0x00, 0x80, 0xff, 0xfe, 0xed, 0xa0, 0x80, 0x01, 0x02, 0xc3, 0x28, 0x7f) +} + +// TestBinaryEvidenceIsByteAccurateAndNamesItsEncoding is the first of the +// issue's four required proofs: arbitrary bytes including invalid UTF-8 +// survive the recorded representation exactly, and the encoding is on the +// event rather than implied. +func TestBinaryEvidenceIsByteAccurateAndNamesItsEncoding(t *testing.T) { + fixture := binaryFixture() + if utf8.Valid(fixture) { + t.Fatal("the fixture is supposed to be invalid UTF-8") + } + + e := loggedEvent(t, postRequest(bytes.NewReader(fixture), "")) + + if got := str(t, e, "body_encoding"); got != bodyEvidenceEncoding { + t.Errorf("body_encoding = %q, want %q", got, bodyEvidenceEncoding) + } + encoded := str(t, e, "body_b64") + decoded, err := base64.StdEncoding.DecodeString(encoded) + if err != nil { + t.Fatalf("body_b64 is not valid base64: %v", err) + } + if !bytes.Equal(decoded, fixture) { + t.Errorf("body_b64 decoded to %x, want the received bytes %x", decoded, fixture) + } + + // And the reason the byte-safe half is needed at all: the legacy JSON + // string field cannot carry these bytes, and the event is honest about + // that only because the other field exists. json.Marshal substitutes + // U+FFFD for every byte outside UTF-8, so this assertion failing would + // mean the string field had somehow become lossless -- which is fine, + // and would only mean the two fields now agree. + if body, ok := e["body"].(string); ok && body == string(fixture) { + t.Log("body preserved every byte; body_b64 remains the authoritative copy") + } else if !strings.ContainsRune(body0(e), utf8.RuneError) { + t.Error("expected the JSON string field to have lost the non-UTF-8 bytes it cannot carry") + } +} + +// body0 returns the logged body string, or "" when the event omits it. +func body0(e map[string]any) string { + body, _ := e["body"].(string) + return body +} + +// TestBodySHA256CoversTheCapturedPrefixAndSaysSo is the second required +// proof. A hash whose scope is ambiguous is worse than no hash, because it +// looks comparable across sensors and is not -- so the value and the label +// that describes it are asserted together, including that the clipped case's +// hash is NOT the complete body's hash. +func TestBodySHA256CoversTheCapturedPrefixAndSaysSo(t *testing.T) { + payload := bytes.Repeat([]byte("C"), bodyCaptureMaxBytes+2048) + captured := payload[:bodyCaptureMaxBytes] + + clipped := loggedEvent(t, postRequest(bytes.NewReader(payload), "")) + + if got := str(t, clipped, "body_sha256_scope"); got != bodySHA256Scope { + t.Errorf("body_sha256_scope = %q, want %q", got, bodySHA256Scope) + } + if got := str(t, clipped, "body_sha256"); got != sha256Hex(captured) { + t.Errorf("body_sha256 = %q, want the hash of the captured prefix %q", got, sha256Hex(captured)) + } + if complete := sha256Hex(payload); str(t, clipped, "body_sha256") == complete { + t.Error("body_sha256 is the complete body's hash, which would contradict body_capture_state") + } + + // When nothing was clipped, the captured prefix IS the complete body and + // the same label stays true -- which is what makes it a scope rather + // than a caveat. + whole := []byte(" afterwards. + raw, err := json.Marshal(e[key]) + if err != nil { + t.Fatalf("%s: %v", key, err) + } + if string(raw) == "null" || string(raw) == `""` { + t.Errorf("%s is empty on the %s event, so it carries nothing", key, want.where) + continue + } + if len(raw) > flattenedIgnoreAbove { + t.Errorf("%s is %d characters, past the flattened field's ignore_above of %d -- it would be stored but not indexed", + key, len(raw), flattenedIgnoreAbove) + } + + line, err := json.Marshal(e) + if err != nil { + t.Fatal(err) + } + var nested map[string]any + if err := json.Unmarshal(line, &nested); err != nil { + t.Fatal(err) + } + if _, ok := nested[key]; !ok { + t.Errorf("%s does not survive the filebeat nesting as honeypot.%s", key, key) + } + } +} + +// TestCaptureMetadataCarriesNoCredentialMaterial is the "presentation does +// not expose credentials" half, for the fields this change adds. The +// metadata says how much was captured and what it hashes; none of it repeats +// a submitted secret. +// +// REWRITTEN against #3213, and the original version of this test asserted the +// opposite. It argued that body_b64 "adds no new class of disclosed material" +// because `body` already carried the raw bytes -- a premise #3213 invalidated +// by redacting `body`, which would have left body_b64 as the only place a +// submitted credential survived the event. The assertion that +// decode(body_b64) == the raw form is therefore inverted, and the replacement +// is strictly stronger than the one it replaces: the old test checked only +// that the secret was absent from the base64 TEXT, which is nearly vacuous, +// because a base64 field cannot contain a raw secret as a substring in the +// first place. This one decodes the field and checks the bytes. +// +// The expected redacted form is produced by calling inspectCredentials, the +// same helper the sensor calls. Reimplementing redaction here would give the +// test a second, quietly divergent definition of what redacted means. +func TestCaptureMetadataCarriesNoCredentialMaterial(t *testing.T) { + const submitted = "not-a-real-credential-value" + form := "username=alice&password=" + submitted + + r := postRequest(bytes.NewReader([]byte(form)), "") + r.SetBasicAuth("alice", submitted) + r.Header.Set("Authorization", "Bearer "+submitted) + e := loggedEvent(t, r) + + for _, key := range []string{ + "body_capture_state", "body_capture_state", "body_encoding", + "body_sha256", "body_sha256_scope", "body_read_error", + "body_declared_bytes", "java_marker", + } { + if v, ok := e[key].(string); ok && strings.Contains(v, submitted) { + t.Errorf("%s repeats submitted credential material: %q", key, v) + } + } + + decoded, err := base64.StdEncoding.DecodeString(str(t, e, "body_b64")) + if err != nil { + t.Fatalf("body_b64 is not valid base64: %v", err) + } + + // The assertion that replaces the invalidated one. body_b64 is the + // REDACTED capture, and the redacted capture is exactly what the shared + // helper produced for this request. + want := inspectCredentials(r, form) + if got, want := string(decoded), want.redactedBody; got != want { + t.Errorf("body_b64 decodes to %q, want the redacted body %q", got, want) + } + if bytes.Equal(decoded, []byte(form)) { + t.Error("body_b64 carries the unredacted body -- that is the leak #3213 closed on `body`, reintroduced here") + } + // Checked on the DECODED bytes, which is the check that means something. + // Checking the encoded text could never fail: base64 does not preserve + // substrings, which is exactly why this field was a safe place to put a + // secret and is now not one. + if bytes.Contains(decoded, []byte(submitted)) { + t.Errorf("the decoded evidence carries the submitted secret: %q", decoded) + } + + // Body and BodyB64 are the same string, so the two fields cannot drift + // apart. This is the invariant that makes body_b64 a faithful + // representation of the event's own `body` rather than a second, + // differently-filtered view of it. + if got := body0(e); !strings.HasPrefix(got, string(decoded)) { + t.Errorf("body_b64 (%q) is not a prefix of body (%q) -- the evidence and the body disagree", decoded, got) + } +} + +// TestBinaryEvidenceNeverCarriesCredentialMaterial is the guard that actually +// holds for body_b64, and it exists because the obvious one does not. +// +// credentials_test.go's TestPasswordNeverReachesTheEvent searches the log +// line for four encodings of the secret, one of which is the base64 of the +// secret. That catches a body_b64 leak only when the bytes preceding the +// secret in the body happen to be a multiple of three -- the base64 of a +// whole body only contains the base64 of an interior substring verbatim when +// that substring starts on a 3-byte boundary. Measured on this branch with +// the evidence deliberately built from the raw capture, it caught 2 of its +// own 14 channels: `username=admin&password=` (24 bytes before the +// secret) and `admin:` (6 bytes). The other 12 -- JSON, multipart, +// nested JSON, XML, query-string, header-only, and both read-cap cases -- +// sailed through with the credential sitting in the event, because 3 did not +// divide the offset. +// +// So the check that matters is on the DECODED bytes, which does not care +// where the secret happens to fall. This runs the same canary through the +// shapes that carry a credential in a body, decodes body_b64, and requires +// the secret to be absent from what comes out. It is the property #3213 +// established for `body`, extended to the one field that would otherwise have +// carried the raw bytes around it. +func TestBinaryEvidenceNeverCarriesCredentialMaterial(t *testing.T) { + shapes := []struct { + name string + contentType string + body string + }{ + { + name: "form urlencoded", + contentType: "application/x-www-form-urlencoded", + body: "username=" + canaryUsername + "&password=" + canarySecret, + }, + { + name: "json", + contentType: "application/json", + body: `{"login":"` + canaryUsername + `","password":"` + canarySecret + `"}`, + }, + { + name: "json, nested", + contentType: "application/json", + body: `{"a":{"b":[{"user":"` + canaryUsername + `","secret":"` + canarySecret + `"}]}}`, + }, + { + name: "html form", + contentType: "text/html", + body: `
` + + `
`, + }, + { + name: "multipart", + contentType: "multipart/form-data; boundary=AaB03x", + body: "--AaB03x\r\nContent-Disposition: form-data; name=\"username\"\r\n\r\n" + + canaryUsername + "\r\n--AaB03x\r\nContent-Disposition: form-data; name=\"password\"\r\n\r\n" + + canarySecret + "\r\n--AaB03x--\r\n", + }, + { + name: "bare basic, no field name at all", + contentType: "text/plain", + body: canaryUsername + ":" + canarySecret, + }, + { + name: "a credential-bearing body that hits the read cap", + contentType: "application/x-www-form-urlencoded", + body: "username=" + canaryUsername + "&filler=" + strings.Repeat("x", bodyReadCap) + + "&password=" + canarySecret, + }, + { + // A body no field-name scan can speak for. #3213 documents this as + // an honest limit rather than a bug -- a value with no key cannot + // be told from ordinary data -- and the shapes it covers live in + // credentials_test.go. It is here to pin that body_b64 inherits + // that limit exactly rather than widening it: whatever `body` + // cannot scrub, body_b64 is scrubbed by the same pass on the same + // string, so the two can never disagree about it. + name: "an xml body no parser here reads", + contentType: "text/xml", + body: `` + canaryUsername + `` + + `` + canarySecret + ``, + }, + } + + for _, tc := range shapes { + t.Run(tc.name, func(t *testing.T) { + r := httptest.NewRequest(http.MethodPost, "/index.php", strings.NewReader(tc.body)) + r.RemoteAddr = "198.51.100.7:54321" + if tc.contentType != "" { + r.Header.Set("Content-Type", tc.contentType) + } + e := loggedEvent(t, r) + + decoded, err := base64.StdEncoding.DecodeString(str(t, e, "body_b64")) + if err != nil { + t.Fatalf("body_b64 is not valid base64: %v", err) + } + if bytes.Contains(decoded, []byte(canarySecret)) { + t.Errorf("body_b64 decodes to bytes carrying the submitted secret: %q", decoded) + } + // Not vacuous: the evidence still has to carry the payload it + // exists to preserve. The username half is deliberately allowed + // through -- it is the analytic value and not a secret -- so a + // guard that asserted its absence would be asserting the feature + // away, and a guard that only checked for the secret would pass + // just as happily on a field emptied of evidence entirely. + if strings.Contains(tc.body, canaryUsername) && !bytes.Contains(decoded, []byte(canaryUsername)) { + t.Errorf("body_b64 decodes to %q, which dropped the account name too -- the field is empty of evidence, not just of the secret", decoded) + } + // Whatever else it holds, the evidence is the event's own body. + if got := body0(e); !strings.HasPrefix(got, string(decoded)) { + t.Errorf("body_b64 is not a prefix of body -- the evidence and the body disagree") + } + }) + } +} diff --git a/arcane/home/honeypot-http/http-honeypot/main.go b/arcane/home/honeypot-http/http-honeypot/main.go index 956edc4a8..cafc9d607 100644 --- a/arcane/home/honeypot-http/http-honeypot/main.go +++ b/arcane/home/honeypot-http/http-honeypot/main.go @@ -15,6 +15,15 @@ package main import ( + // #3212: bytes and crypto/sha256 for the byte-safe evidence and its + // hash, encoding/base64 for the former and encoding/hex for the latter. + // #3213 removed base64 from this file when the Basic/Bearer header + // parsing moved into credentials.go; #3212 needs it again for a + // different reason, which is not a reason to move the parsing back. + "bytes" + "crypto/sha256" + "encoding/base64" + "encoding/hex" "encoding/json" "fmt" "io" @@ -56,7 +65,19 @@ type event struct { Query string `json:"query,omitempty"` UserAgent string `json:"user_agent,omitempty"` Headers map[string]string `json:"headers"` - Body string `json:"body,omitempty"` + // Body is the captured prefix as text, for the same reason it has + // always been logged: it is what an analyst greps. It is NOT a + // byte-preserving record of the request on two counts, and both are + // now stated rather than left for a reader to discover. + // + // Lossiness: every byte outside UTF-8 is replaced by U+FFFD on the way + // into JSON, so the stored value is a different byte string from the + // one received. BodyB64/BodyEncoding are the byte-exact half. + // + // Redaction (#3213): this is creds.redactedBody, never the bytes off + // the wire. BodyB64 is built from this same string, so the two cannot + // disagree and neither is a way around the redaction. + Body string `json:"body,omitempty"` // #3213: the account half of a submitted credential, kept because it // is the analytic value (a spray is a spray of accounts) and it is not // a secret. The secret half of the same credential is never written @@ -110,6 +131,108 @@ type event struct { Tarpitted bool `json:"tarpitted,omitempty"` TarpitBytes int `json:"tarpit_bytes,omitempty"` TarpitMS int64 `json:"tarpit_ms,omitempty"` + + // --- #3212: what we captured, how much of it, and the bytes --- + // + // Before this block, the only statement this event made about a body + // was `body`, a Go string of whatever one io.ReadAll behind a 64 KiB + // LimitReader happened to return -- with its error discarded. Two + // defects followed from that, and both are unfixable by a consumer: + // + // 1. A JSON string does not preserve arbitrary bytes. A Java + // serialized object, a PNG, a lone 0x80 -- json.Marshal replaces + // every byte that is not valid UTF-8 with U+FFFD, so the stored + // value is a different byte string from the one received, and + // nothing in the document says so. + // 2. A 40-byte body and the first 64 KiB of a 4 MB one serialized to + // the same shape. "The attacker sent 40 bytes" and "we kept 4 KB + // of something much larger" were the same record. + // + // So the capture is now described rather than implied. BodyCaptureState + // is the authoritative answer to "is this the whole body": "complete", + // "truncated" (we stopped at the cap and more bytes existed), or + // "unknown" (the read failed, so completeness is not knowable). It is + // set on every request event, including zero-byte ones, so an aggregate + // over it has no missing-value bucket -- a genuinely short body answers + // "complete", which is exactly the distinction defect 2 was missing. + // + // BodyReadError is the discarded io error, kept because "truncated" and + // "we never found out" are different failures and the state field can + // only carry one of them. + // + // BodySHA256 covers the CAPTURED PREFIX and says so in + // BodySHA256Scope -- the same field name galah's body_sha256 already + // established (see ip_enrichment/sensors.rs's promotion of + // httpRequest.bodySha256), with the scope its hash could not claim. A + // hash whose scope is ambiguous is worse than no hash: it looks + // comparable across sensors and is not. It is deliberately NOT the + // complete body's hash, because computing that would mean draining + // whatever the client claims to be sending -- unbounded work, on + // time this sensor does not spend, to obtain a fact nobody downstream + // needs. See captureBody. + // + // BodyB64 is the byte-safe half of the answer: a bounded head, base64, + // with BodyEncoding naming that encoding so a consumer never has to + // guess it. BodyDeclaredBytes is the Content-Length the request CLAIMED + // -- attacker-controlled, so it is recorded as a claim and never trusted + // as a fact, and it is the only thing that turns "truncated at 64 KiB" + // into "truncated at 64 KiB of 4 MB". + // + // JavaMarker is #3212's other half: a separate, Java-only observation, + // deliberately not a new payload_class (see javaMarker). + // + // --- #3213 changed one of these premises, and the change is load-bearing + // + // This block was written against a `body` that held the raw capture, and + // on that basis the byte-safe evidence was free: it carried the same + // bytes `body` already published, so it disclosed nothing new. #3213 + // redacted `body`, and with it that argument. The evidence is now built + // from the same redacted string, so the two fields still agree and + // neither reintroduces what #3213 removed -- a base64 blob is not a + // disclosure control, and treating it as one would have made this field + // the single place a submitted credential survived. + // + // The per-field comments below say which of the two questions each field + // answers, because the capture and the published bytes are now different + // things and only one of them is allowed out of the process. + // BodyCaptureState / BodyCapturedBytes / BodyReadError describe the + // CAPTURE: how many bytes came off the wire, whether that was all of + // them, and what ended the read. Nothing about redaction changes any of + // the three, because redaction is not a fact about the wire. + BodyCaptureState string `json:"body_capture_state"` + // BodyCapturedBytes is the raw captured length, so it can exceed + // len(decode(body_b64)) whenever the redactor shortened the body. See + // bodyCapture.apply. + BodyCapturedBytes int `json:"body_captured_bytes"` + BodyReadError string `json:"body_read_error,omitempty"` + // BodyB64 and BodySHA256 describe the PUBLISHED bytes: the redacted + // body, bounded to a head, and the hash of the whole redacted captured + // prefix. They are derived from the same string Body carries, so a + // consumer can decode one, hash it, and get the other -- and neither + // field is a route around the redaction that #3213 put on `body`. + // Before #3213 these were safe to publish only because `body` already + // published the same bytes; that is exactly the premise #3213 + // invalidated, and it is why the evidence is no longer built from the + // raw capture. + BodyEncoding string `json:"body_encoding,omitempty"` + BodyB64 string `json:"body_b64,omitempty"` + // BodySHA256 covers the CAPTURED PREFIX, post-redaction, and says so in + // BodySHA256Scope -- the same field name galah's body_sha256 already + // established (see ip_enrichment/sensors.rs's promotion of + // httpRequest.bodySha256), with a scope label that names both bounds. A + // hash whose scope is ambiguous is worse than no hash: it looks + // comparable across sensors and is not. It is deliberately NOT the + // complete body's hash, because computing that would mean draining + // whatever the client claims to be sending -- unbounded work, on time + // this sensor does not spend, to obtain a fact nobody downstream needs. + // See captureBody and bodyCapture.apply. + BodySHA256 string `json:"body_sha256,omitempty"` + BodySHA256Scope string `json:"body_sha256_scope,omitempty"` + // BodyDeclaredBytes is the Content-Length the client CLAIMED, and + // BodyJavaMarker the byte-pattern Java observation. See declaredLength + // and javaMarker. + BodyDeclaredBytes int64 `json:"body_declared_bytes,omitempty"` + JavaMarker string `json:"java_marker,omitempty"` } type logger struct { @@ -352,8 +475,10 @@ func headerMap(r *http.Request) map[string]string { // payloads nest -- the second most common body here is a 0 && strings.HasPrefix(rest[digits:], `:"`) } +// javaStreamMagic is the four-byte header a Java serialization stream begins +// with -- STREAM_MAGIC 0xACED followed by STREAM_VERSION 5 -- and the same +// four bytes "rO0AB" is the base64 of. Named once so the classifier case +// above and javaMarker below cannot drift apart on the literal. +var javaStreamMagic = []byte{0xac, 0xed, 0x00, 0x05} + +// javaMarker names the Java serialization marker observed in the captured +// bytes, or "" when there is none (#3212). "stream-magic" and +// "stream-magic-base64" say a marker was seen. They do not say an object +// graph was valid, a gadget was named, or anything executed -- nothing here +// can, and nothing here tries. +// +// Byte patterns only. No ObjectInputStream, no class loading, no +// instantiation of anything that arrived over the socket: the sole decode +// performed is base64's, on eight characters, to check whether they decode to +// four known bytes -- a comparison, not a parse. A received stream is never +// read as a stream, only compared against four literals. +// +// This is a SEPARATE field, not a new payload_class, and that is the whole +// point of it. classifyPayload is an ordered first-match-wins classifier, and +// its "serialized-object" value is shared with PHP's serialize(); splitting +// Java out of that value would repurpose a class existing queries already +// mean, and would lose the PHP case the moment a Java-looking body arrived +// first. Kept separate, a request can carry both at once: a body that trips +// "sqli" three cases earlier and also starts with the Java header keeps +// payload_class "sqli" AND java_marker "stream-magic", so neither observation +// erases the other. +func javaMarker(body []byte) string { + // Offset 0 only. A serialization stream starts with this header, so + // that is the only position where seeing it means what it says. The + // classifier's Contains above accepts the four bytes anywhere in the + // body; that looser class label stays exactly as it was, and this field + // reports the narrower observation it is able to stand behind. + if bytes.HasPrefix(body, javaStreamMagic) { + return "stream-magic" + } + // The same header transported as base64 text, which is how it arrives + // when the payload is submitted as a form value. Verified rather than + // assumed: eight characters is the shortest run that can carry four + // bytes, so a body that merely starts with the five-character "rO0AB" + // is a coincidental prefix and is not evidence of anything. That is + // the whole difference between this and the classifier's HasPrefix. + const b64Run = 8 + if bytes.HasPrefix(body, []byte("rO0AB")) && len(body) >= b64Run { + if decoded, err := base64.StdEncoding.DecodeString(string(body[:b64Run])); err == nil && + bytes.Equal(decoded, javaStreamMagic) { + return "stream-magic-base64" + } + } + return "" +} + +// bodyCaptureMaxBytes is how much of a request body this sensor keeps. The +// value is unchanged from the io.LimitReader(64<<10) it replaces -- a bounded +// capture is the point, and an unbounded one is the denial of service an +// attacker picks the size of. What #3212 adds is the record of WHICH bound +// was hit, so a reader can tell a short body from a clipped one. +const bodyCaptureMaxBytes = 64 << 10 + +// bodyEvidenceMaxBytes is how much of the captured prefix is also retained +// as a byte-safe head. Sized for the job the head exists to do: hold any +// serialization header, its class descriptor and the class name, which is +// what the marker evidence is about, in a few hundred bytes. It is not sized +// to reproduce a body -- body_sha256 covers the whole captured prefix, and +// nothing in the fleet reads a 64 KiB body out of a log line to re-hash it. +// +// It also has to stay under the 32000-character ignore_above on the +// flattened `honeypot` field (honeypot-init's elasticsearch-setup.sh), or the +// value stops being indexed and only survives in _source -- so 4 KiB raw, +// 5464 characters of base64, against a 32000-character ceiling. The test +// asserts that margin rather than trusting the arithmetic. +const bodyEvidenceMaxBytes = 4 << 10 + +// The three capture states, and the two fixed strings that make the rest of +// the block self-describing: the encoding of body_b64, and the scope +// body_sha256 covers. Both are recorded on the event rather than left to +// documentation, because a consumer reading a log line three months from now +// does not have this file. +const ( + captureComplete = "complete" + captureTruncated = "truncated" + captureUnknown = "unknown" + + bodyEvidenceEncoding = "base64" + + // bodySHA256Scope names both axes the hash is bounded on, because + // "captured-prefix" alone would now be a half-truth: it says which part + // of the body is covered, and says nothing about the fact that the bytes + // covered are the credential-scrubbed ones (#3213). A reader comparing + // this hash against a body they hold elsewhere would get a mismatch on + // any request that carried a credential, with no field on the event to + // explain it -- which is the "hash whose scope is ambiguous" defect this + // field exists to prevent, reintroduced one axis over. So the value says + // "a prefix, and post-redaction", and a body with no credential in it + // hashes identically under either reading, because redaction is a no-op + // there. + bodySHA256Scope = "captured-prefix-redacted" +) + +// bodyCapture is what came off the wire and how much of it that was. Bytes +// is the captured prefix, never longer than bodyCaptureMaxBytes. +type bodyCapture struct { + Bytes []byte + // Truncated records that more bytes existed than were kept. It is a + // fact about the cap, independent of ReadErr. + Truncated bool + // ReadErr is whatever ended the read early. A non-nil error means + // completeness is not knowable, whatever the length. + ReadErr error +} + +// state is the one authoritative answer to "is this the whole body". The two +// failure facts are separate fields because a single string cannot carry +// both: a body that hit the cap AND whose read then failed is reported as +// truncated (the cap is a fact; the error is in BodyReadError) rather than +// as unknown, which would discard something that is known. +func (c bodyCapture) state() string { + switch { + case c.Truncated: + return captureTruncated + case c.ReadErr != nil: + return captureUnknown + default: + return captureComplete + } +} + +// captureBody reads at most bodyCaptureMaxBytes of r, and says which of the +// three things happened. +// +// One byte past the cap is the whole trick: a read that stops exactly at the +// limit is indistinguishable from a body of precisely that size, so +// truncating to the cap and reporting nothing is the ambiguity this exists +// to remove. Reading one more byte is what makes "there was more" knowable. +// +// The read error is returned rather than dropped, which is the other half of +// it: the old `body, _ := io.ReadAll(...)` threw away the only signal that +// the capture was not what it appeared to be. +func captureBody(r io.Reader) bodyCapture { + data, err := io.ReadAll(io.LimitReader(r, bodyCaptureMaxBytes+1)) + if len(data) > bodyCaptureMaxBytes { + return bodyCapture{Bytes: data[:bodyCaptureMaxBytes], Truncated: true, ReadErr: err} + } + return bodyCapture{Bytes: data, ReadErr: err} +} + +// apply records the capture's completeness, its byte-safe evidence and the +// hash of exactly the bytes it is allowed to publish onto e. The state is +// written even for a zero-byte capture -- a request with no body is genuinely +// a short body, and answering "complete" is what distinguishes it from a +// clipped one. +// +// redacted is the credential-scrubbed body (#3213's creds.redactedBody), +// passed in rather than recomputed here so that this cannot become a second +// redaction path with its own rules. It is deliberately NOT c.Bytes: the +// completeness fields below describe the CAPTURE, which is a fact about what +// came off the wire and is unaffected by redaction, while the two evidence +// fields describe the PUBLISHED bytes, which are the redacted ones. Both are +// recorded because both are true and they answer different questions -- "how +// much arrived" and "what are you allowed to see" are not the same question, +// and answering only the first would still be a leak. +// +// The consequence, stated rather than hidden: when redaction shortened the +// body, the retained head and the hash are of the redacted form, so +// len(decode(body_b64)) can be less than body_captured_bytes. That is the +// redaction working, and BodySHA256Scope is what says so on the event. +func (c bodyCapture) apply(e *event, redacted string) { + e.BodyCaptureState = c.state() + e.BodyCapturedBytes = len(c.Bytes) + if c.ReadErr != nil { + e.BodyReadError = c.ReadErr.Error() + } + if redacted == "" { + // Nothing to preserve and nothing to identify. The sha256 of the + // empty string is a real hash of the empty string, and carrying it + // on every bodyless request would be noise wearing a hash's clothes. + return + } + published := []byte(redacted) + head := published + if len(head) > bodyEvidenceMaxBytes { + head = head[:bodyEvidenceMaxBytes] + } + e.BodyEncoding = bodyEvidenceEncoding + e.BodyB64 = base64.StdEncoding.EncodeToString(head) + sum := sha256.Sum256(published) + e.BodySHA256 = hex.EncodeToString(sum[:]) + e.BodySHA256Scope = bodySHA256Scope +} + +// declaredLength is the Content-Length the request claimed, or 0 when it +// claimed nothing usable. A claim, never a fact: it is written by the +// attacker, it disagrees with the wire as often as not, and it exists here +// only so "we kept 64 KiB" can be read as "we kept 64 KiB of 4 MB" when -- +// and only when -- the client was straight about it. +// +// Read from the header, falling back to the value net/http parsed out of that +// same header, so the field describes the wire rather than one particular +// way of asking for it. A chunked request arrives as -1 and has no declared +// length to record, which is 0 here rather than a made-up number. +func declaredLength(r *http.Request) int64 { + raw := strings.TrimSpace(r.Header.Get("Content-Length")) + if raw == "" { + if r.ContentLength > 0 { + return r.ContentLength + } + return 0 + } + n, err := strconv.ParseInt(raw, 10, 64) + if err != nil || n < 0 { + return 0 + } + return n +} + // classify guesses the intent of a request path so logs are easy to triage. func classify(path string) string { p := strings.ToLower(path) @@ -969,8 +1314,16 @@ func (s *server) ServeHTTP(w http.ResponseWriter, r *http.Request) { return } - body, _ := io.ReadAll(io.LimitReader(r.Body, bodyReadCap)) // cap at 64 KiB + // #3212: read through captureBody instead of a bare LimitReader. The + // cap is the same 64 KiB it always was, and it is bodyReadCap's value: + // #3213's credential pass reads the same bytes, so the two have to + // agree on where "all of it" ends. What the event now carries is which + // of "all of it", "a bounded prefix" and "we never found out" this + // request was, so that a 40-byte POST and the first 64 KiB of a 4 MB + // upload stop being the same record. + capture := captureBody(r.Body) r.Body.Close() + body := capture.Bytes // #3213: one pass over every channel that can carry a credential, // answering three questions the event used to answer badly or not at @@ -1010,7 +1363,24 @@ func (s *server) ServeHTTP(w http.ResponseWriter, r *http.Request) { AuthOutcome: string(authUnknown), Username: creds.username, AuthType: creds.authType, + // Recorded from the raw bytes, independently of PayloadClass: an + // earlier case in that ordered switch must not be able to erase + // the fact that a Java marker was seen, and vice versa. Byte + // patterns only, and detection rather than disclosure, so these read + // the raw body the way classifyPayload does -- what leaves the + // process is a fixed marker name, never a byte of the payload. + JavaMarker: javaMarker(body), + BodyDeclaredBytes: declaredLength(r), } + // The evidence of what was captured is built from the SAME redacted + // string Body was just given, never from string(body). Before #3213 + // these bytes were safe to publish because `body` already published + // them; that stopped being true when `body` was redacted, and had this + // call still read the raw capture it would have made body_b64 the only + // copy of a submitted credential anywhere in the event. One string, two + // fields -- the two cannot drift apart, and neither is a way around + // inspectCredentials. + capture.apply(&e, creds.redactedBody) if s.tarpitEnabled && tarpitCategory(e.Category) { e.Status = http.StatusOK