diff --git a/arcane/home/honeypot-elk/analysis/filebeat.yml b/arcane/home/honeypot-elk/analysis/filebeat.yml index 8aea6240e..978d961be 100644 --- a/arcane/home/honeypot-elk/analysis/filebeat.yml +++ b/arcane/home/honeypot-elk/analysis/filebeat.yml @@ -202,6 +202,17 @@ filebeat.inputs: # content-hash window can. id: zeek-logs file_identity.native: ~ + # #3284: this directory accumulates ~13,700 hourly rotation files (2026-09-03 + # onward) and nothing prunes them, so every restart made filestream open a + # harvester per file until the Go runtime hit its 10000-thread limit. The + # cap is the fix, not GODEBUG: one harvester per log type is plenty, because + # Zeek writes one file per type per hour and only the newest is still being + # appended to. Older rotations are finished by definition. + harvester_limit: 64 + # 72h is ~3 days of hourly rotations per log type, comfortably past any + # backlog while keeping the scanner off the 13,700 accumulated files. A + # rotation older than this is finished; nothing appends to it again. + ignore_older: 72h paths: - /logs/zeek/*.log parsers: @@ -224,6 +235,11 @@ filebeat.inputs: # identity is immune to it because it never hashes file content. id: zeek-proxy-logs file_identity.native: ~ + # #3284: same cap as zeek-logs, same reason -- this glob matches 834 + # accumulated hourly rotations and one harvester per file is what exhausts + # the runtime's thread limit on restart. + harvester_limit: 64 + ignore_older: 72h paths: - /logs/zeek-proxy/*.log parsers: diff --git a/arcane/home/honeypot-utilities/analysis/log-maintenance.sh b/arcane/home/honeypot-utilities/analysis/log-maintenance.sh index 5b764b1c2..a61d58d57 100644 --- a/arcane/home/honeypot-utilities/analysis/log-maintenance.sh +++ b/arcane/home/honeypot-utilities/analysis/log-maintenance.sh @@ -136,6 +136,15 @@ while true; do # its mtime current, so -mmin never fires on it (same shape vps/ # suricata-log-maintenance.sh uses for eve.json). find /logs/zeek-proxy -maxdepth 1 -name '*.log' -mmin "+${json_retention_min}" -print -delete 2>/dev/null || true + # #3284: /logs/zeek here is an sshfs mount of the VPS, deliberately read-only + # (fuse.sshfs ro in /etc/fstab), so a delete attempted from this container + # fails EROFS on every file. Confirmed live: 12,390 paths printed by this + # script's find, file count unchanged, touch and find -delete both returning + # "Read-only file system". The 2>/dev/null above hides exactly that error. + # + # Retention for these files therefore lives on the VPS, in + # vps/zeek-log-maintenance.sh, which runs where they are actually written. + # Do not re-add a find here -- it cannot work, and it fails quietly. # #2323 part 2: extracted-file-importer.py copies carved bytes into ES # and tracks what it has seen in state/extracted-files.json, so the disk # copy only needs to outlive that importer's lag (IMPORT_INTERVAL=60s, diff --git a/vps/zeek-log-maintenance.sh b/vps/zeek-log-maintenance.sh new file mode 100755 index 000000000..503b54630 --- /dev/null +++ b/vps/zeek-log-maintenance.sh @@ -0,0 +1,43 @@ +#!/bin/sh +set -eu + +# Prune rotated Zeek logs on the VPS, where they are actually written. +# +# #3284: this is the root cause of both open ops issues, and it was invisible +# from the homeserver. honeypot-elk's filebeat reads /logs/zeek over an sshfs +# mount that is deliberately read-only (root@10.8.0.1:/opt/stacks/apiary/logs/zeek +# -> /var/dockge/stacks/apiary/logs/zeek, fuse.sshfs ro in /etc/fstab), so +# log-maintenance.sh on the homeserver prints the paths it wants to delete and +# then silently fails every one of them -- its find carries 2>/dev/null || true, +# which is why 12,390 "deleted" zeek paths appeared in that container's log +# with the file count unchanged. +# +# Zeek rotates hourly by rename-and-reopen and never truncates in place, so a +# rotated generation is finished the moment it is closed: -mmin only ever fires +# on files nothing will append to again. The bare-name match also covers the +# live file of a dead sensor, whose mtime freezes (same shape +# suricata-log-maintenance.sh uses for eve.json). +# +# Consequence of not having this: 13,717 accumulated hourly rotations (16 GB), +# and one filestream harvester per file on every hp-filebeat restart, which is +# what drove the Go runtime past its 10000-thread limit. It also fed #3283 -- +# one Elasticsearch index per day per log type, 836 of the cluster's 1091 +# shards, several holding a single document. +# +# #261: default derives from the shared HONEYPOT_RETENTION_DAYS knob, same +# ratio and reasoning as analysis/log-maintenance.sh's json_retention_min and +# suricata-log-maintenance.sh's retention_min. Elasticsearch holds the +# searchable history; this is only the on-disk copy, which needs to outlive +# Filebeat's ingest lag by a comfortable margin, not last forever. + +retention_min="${RETENTION_MINUTES:-$(( ${HONEYPOT_RETENTION_DAYS:-30} * 1440 / 10 ))}" +interval="${CHECK_INTERVAL:-3600}" +start_delay="${START_DELAY:-60}" +log_dir="${LOG_DIR:-/opt/stacks/apiary/logs/zeek}" + +sleep "$start_delay" + +while true; do + find "$log_dir" -maxdepth 1 -name '*.log' -mmin "+${retention_min}" -print -delete 2>/dev/null || true + sleep "$interval" +done