From 219886e618dadc7d92504997cc0d960213ce8fc2 Mon Sep 17 00:00:00 2001 From: devkyato <95612413+devkyato@users.noreply.github.com> Date: Wed, 29 Jul 2026 19:50:35 +0800 Subject: [PATCH] Harden Datary for the 0.2.0 corrective release --- .editorconfig | 18 + .gitattributes | 7 + .github/ISSUE_TEMPLATE/bug_report.yml | 54 + .github/ISSUE_TEMPLATE/config.yml | 5 + .github/ISSUE_TEMPLATE/feature_request.yml | 30 + .github/PULL_REQUEST_TEMPLATE.md | 20 + .github/dependabot.yml | 10 + .github/release.yml | 15 + .github/workflows/ci.yml | 79 +- CHANGELOG.md | 26 +- CONTRIBUTING.md | 20 +- README.md | 242 ++-- SECURITY.md | 32 +- docs/architecture.md | 57 +- docs/input-formats.md | 47 +- docs/metrics.md | 77 +- docs/privacy.md | 21 +- docs/quality-checks.md | 58 +- docs/releases/0.2.0.md | 75 + docs/reproducibility.md | 29 +- docs/session-format.md | 47 +- docs/tutorials.md | 18 +- examples/datasets/noisy-sensor-seed-1.jsonl | 118 +- examples/reports/noisy-sensor-example.md | 71 +- .../sessions/noisy-sensor-example/data.csv | 118 +- .../noisy-sensor-example/manifest.json | 43 +- .../noisy-sensor-example/manifest.sha256 | 1 + .../noisy-sensor-example/metrics.json | 167 ++- .../noisy-sensor-example/plots/plot-value.png | Bin 0 -> 31201 bytes .../noisy-sensor-example/plots/value.png | Bin 26799 -> 0 bytes .../noisy-sensor-example/quality.json | 16 +- .../sessions/noisy-sensor-example/raw.log | 118 +- .../noisy-sensor-example/records.jsonl | 122 +- .../noisy-sensor-example/reports/report.md | 87 ++ examples/simulations/motor_sim.py | 1 + pyproject.toml | 27 +- scripts/build_checksums.py | 14 +- scripts/build_release.py | 2 +- scripts/verify_reproducible.py | 2 +- src/datary/__init__.py | 2 +- src/datary/__main__.py | 1 - src/datary/analysis_store.py | 1235 +++++++++++++++++ src/datary/cli.py | 360 ++++- src/datary/comparison.py | 113 +- src/datary/config.py | 4 +- src/datary/conversion.py | 109 +- src/datary/formats.py | 27 +- src/datary/generators.py | 93 +- src/datary/inspection.py | 202 ++- src/datary/metrics.py | 233 +++- src/datary/models.py | 17 +- src/datary/parsers.py | 278 +++- src/datary/plotting.py | 60 +- src/datary/py.typed | 1 + src/datary/quality.py | 323 ++++- src/datary/recorder.py | 325 +++-- src/datary/replay.py | 27 +- src/datary/reports.py | 127 +- src/datary/sessions.py | 190 ++- src/datary/utils.py | 194 ++- tests/test_cli.py | 14 + tests/test_comparison.py | 38 + tests/test_conversion.py | 41 + tests/test_formats.py | 12 + tests/test_generators.py | 32 + tests/test_metrics.py | 55 +- tests/test_parsers.py | 56 + tests/test_plotting.py | 8 +- tests/test_quality.py | 57 +- tests/test_recorder.py | 186 ++- tests/test_release_scripts.py | 4 +- tests/test_replay.py | 26 +- tests/test_reports.py | 37 +- tests/test_security.py | 62 +- 74 files changed, 5661 insertions(+), 782 deletions(-) create mode 100644 .editorconfig create mode 100644 .gitattributes create mode 100644 .github/ISSUE_TEMPLATE/bug_report.yml create mode 100644 .github/ISSUE_TEMPLATE/config.yml create mode 100644 .github/ISSUE_TEMPLATE/feature_request.yml create mode 100644 .github/PULL_REQUEST_TEMPLATE.md create mode 100644 .github/dependabot.yml create mode 100644 .github/release.yml create mode 100644 docs/releases/0.2.0.md create mode 100644 examples/sessions/noisy-sensor-example/manifest.sha256 create mode 100644 examples/sessions/noisy-sensor-example/plots/plot-value.png delete mode 100644 examples/sessions/noisy-sensor-example/plots/value.png create mode 100644 examples/sessions/noisy-sensor-example/reports/report.md create mode 100644 src/datary/analysis_store.py create mode 100644 src/datary/py.typed create mode 100644 tests/test_conversion.py diff --git a/.editorconfig b/.editorconfig new file mode 100644 index 0000000..436bce9 --- /dev/null +++ b/.editorconfig @@ -0,0 +1,18 @@ +root = true + +[*] +charset = utf-8 +end_of_line = lf +insert_final_newline = true +trim_trailing_whitespace = true + +[*.py] +indent_style = space +indent_size = 4 + +[*.{yml,yaml,json,toml,md}] +indent_style = space +indent_size = 2 + +[Makefile] +indent_style = tab diff --git a/.gitattributes b/.gitattributes new file mode 100644 index 0000000..2bef032 --- /dev/null +++ b/.gitattributes @@ -0,0 +1,7 @@ +* text=auto eol=lf +*.png binary +*.jpg binary +*.jpeg binary +*.gif binary +*.whl binary +*.gz binary diff --git a/.github/ISSUE_TEMPLATE/bug_report.yml b/.github/ISSUE_TEMPLATE/bug_report.yml new file mode 100644 index 0000000..07d99af --- /dev/null +++ b/.github/ISSUE_TEMPLATE/bug_report.yml @@ -0,0 +1,54 @@ +name: Bug report +description: Report a reproducible Datary defect using synthetic or non-sensitive data. +title: "bug: " +labels: ["bug"] +body: + - type: markdown + attributes: + value: | + Thanks for helping me make Datary more dependable. Please do not attach confidential recordings. + - type: input + id: version + attributes: + label: Datary version + placeholder: datary 0.2.0 + validations: + required: true + - type: dropdown + id: platform + attributes: + label: Platform + options: + - Windows + - macOS + - Linux + - Other + validations: + required: true + - type: textarea + id: reproducer + attributes: + label: Minimal reproducer + description: Include commands and synthetic input. + render: shell + validations: + required: true + - type: textarea + id: expected + attributes: + label: Expected behaviour + validations: + required: true + - type: textarea + id: actual + attributes: + label: Actual behaviour + validations: + required: true + - type: checkboxes + id: privacy + attributes: + label: Data safety + options: + - label: I removed secrets and personal data from this report. + required: true diff --git a/.github/ISSUE_TEMPLATE/config.yml b/.github/ISSUE_TEMPLATE/config.yml new file mode 100644 index 0000000..fd122ae --- /dev/null +++ b/.github/ISSUE_TEMPLATE/config.yml @@ -0,0 +1,5 @@ +blank_issues_enabled: false +contact_links: + - name: Security vulnerability + url: https://github.com/devkyato/datary-lab/security/policy + about: Please follow the private reporting guidance instead of opening a public issue. diff --git a/.github/ISSUE_TEMPLATE/feature_request.yml b/.github/ISSUE_TEMPLATE/feature_request.yml new file mode 100644 index 0000000..addb918 --- /dev/null +++ b/.github/ISSUE_TEMPLATE/feature_request.yml @@ -0,0 +1,30 @@ +name: Feature request +description: Propose a focused local-first Datary improvement. +title: "feature: " +labels: ["enhancement"] +body: + - type: textarea + id: problem + attributes: + label: Problem + description: What experiment workflow is difficult today? + validations: + required: true + - type: textarea + id: proposal + attributes: + label: Proposed behaviour + validations: + required: true + - type: textarea + id: evidence + attributes: + label: Reproducibility and safety considerations + description: Mention formats, assumptions, compatibility, privacy, and failure behaviour. + - type: checkboxes + id: boundaries + attributes: + label: Project boundaries + options: + - label: This can remain local-first and does not require telemetry or a hosted service. + required: true diff --git a/.github/PULL_REQUEST_TEMPLATE.md b/.github/PULL_REQUEST_TEMPLATE.md new file mode 100644 index 0000000..d915e0c --- /dev/null +++ b/.github/PULL_REQUEST_TEMPLATE.md @@ -0,0 +1,20 @@ +## What changed + + + +## Evidence + + + +## Assumptions and compatibility + + + +## Checklist + +- [ ] `python -m pytest` +- [ ] `python -m ruff check .` +- [ ] `python -m ruff format --check .` +- [ ] `python -m mypy` +- [ ] Documentation and changelog updated when behaviour changed +- [ ] No secrets, telemetry, background networking, or data execution added diff --git a/.github/dependabot.yml b/.github/dependabot.yml new file mode 100644 index 0000000..abd2e58 --- /dev/null +++ b/.github/dependabot.yml @@ -0,0 +1,10 @@ +version: 2 +updates: + - package-ecosystem: github-actions + directory: / + schedule: + interval: weekly + - package-ecosystem: pip + directory: / + schedule: + interval: weekly diff --git a/.github/release.yml b/.github/release.yml new file mode 100644 index 0000000..6e66d8e --- /dev/null +++ b/.github/release.yml @@ -0,0 +1,15 @@ +changelog: + categories: + - title: Security and correctness + labels: + - security + - bug + - title: New work + labels: + - enhancement + - title: Documentation + labels: + - documentation + - title: Other changes + labels: + - "*" diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 3e73b33..0c67764 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -1,40 +1,91 @@ name: CI + on: push: pull_request: + +permissions: + contents: read + +concurrency: + group: ci-${{ github.workflow }}-${{ github.ref }} + cancel-in-progress: true + jobs: test: + name: Python ${{ matrix.python }} / Ubuntu + timeout-minutes: 20 strategy: fail-fast: false matrix: python: ["3.9", "3.10", "3.11", "3.12", "3.13", "3.14"] runs-on: ubuntu-latest steps: - - uses: actions/checkout@v4 - - uses: actions/setup-python@v5 - with: {python-version: "${{ matrix.python }}"} + - uses: actions/checkout@11bd71901bbe5b1630ceea73d27597364c9af683 # v4.2.2 + - uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0 + with: + python-version: ${{ matrix.python }} + cache: pip + - run: python -m pip install --upgrade pip - run: python -m pip install -e ".[dev]" - - run: pytest - - run: ruff check . - - run: mypy + - run: python -m pip check + - run: python -m pytest + - run: python -m ruff check . + - run: python -m ruff format --check . + - run: python -m mypy - run: python scripts/verify_reproducible.py - - run: datary generate noisy-sensor --seed 1 > sample.jsonl + - run: datary generate noisy-sensor --seed 1 --output sample.jsonl - run: datary record demo --format jsonl --time-field timestamp < sample.jsonl - - run: datary inspect demo + - run: datary inspect demo --quality + - run: datary inspect demo --plot value - run: datary report demo + - run: datary replay demo --no-timing > replay.jsonl - run: python -m build - - uses: actions/upload-artifact@v4 - with: {name: "dist-${{ matrix.python }}", path: dist/*} + - run: python scripts/build_checksums.py + - uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4.6.2 + with: + name: dist-${{ matrix.python }} + path: | + dist/* + demo/reports/* + demo/plots/* + + platform-smoke: + name: Python ${{ matrix.python }} / ${{ matrix.os }} + timeout-minutes: 15 + strategy: + fail-fast: false + matrix: + os: [windows-latest, macos-latest] + python: ["3.9", "3.14"] + runs-on: ${{ matrix.os }} + steps: + - uses: actions/checkout@11bd71901bbe5b1630ceea73d27597364c9af683 # v4.2.2 + - uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0 + with: + python-version: ${{ matrix.python }} + cache: pip + - run: python -m pip install -e ".[dev]" + - run: python -m pytest + - run: datary --version + - run: datary generate sine --seed 1 --duration 1 --output sample.jsonl + - run: datary inspect sample.jsonl --format jsonl --time-field timestamp + wheel-smoke: + name: Clean wheel installation needs: test + timeout-minutes: 15 runs-on: ubuntu-latest steps: - - uses: actions/checkout@v4 - - uses: actions/setup-python@v5 - with: {python-version: "3.12"} + - uses: actions/checkout@11bd71901bbe5b1630ceea73d27597364c9af683 # v4.2.2 + - uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0 + with: + python-version: "3.12" - run: python -m pip install build - run: python -m build - run: python -m venv /tmp/datary-smoke - run: /tmp/datary-smoke/bin/pip install dist/*.whl + - run: /tmp/datary-smoke/bin/pip check - run: /tmp/datary-smoke/bin/datary --version - + - run: /tmp/datary-smoke/bin/datary generate sine --duration 1 --output /tmp/sine.jsonl + - run: /tmp/datary-smoke/bin/datary inspect /tmp/sine.jsonl --format jsonl --time-field timestamp diff --git a/CHANGELOG.md b/CHANGELOG.md index 497f3ac..14750fa 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,5 +1,29 @@ # Changelog +## 0.2.0 - 2026-07-29 + +I treated this release as a corrective audit rather than a feature polish. A hard external review +showed that several green tests were confirming the old implementation instead of independently +checking its scientific and security claims. + +- Preserved observation order for change metrics and added ordered counterexamples. +- Replaced whole-session recorder and inspector lists with exact disk-backed analysis. +- Made overwrite publication recoverable so a failed replacement keeps the prior session. +- Closed plot traversal and session symlink gaps; hardened manifests, record readers, terminal + output, Markdown, CSV formulas, and generated-file publication. +- Added incremental JSON-array parsing, RFC-style multiline CSV, BOM handling, duplicate-key + rejection, conservative scalar coercion, and non-finite-value evidence. +- Expanded integrity to rejected records, notes, structural counts, and the manifest itself while + documenting that checksums are not authentication. +- Corrected frozen-value positions, same-length schema detection, zero-MAD outliers/spikes, and + global duplicate timestamps. +- Added numeric and timezone-aware ISO 8601 timing, explicit engineering field roles, trapezoidal + control integrals, network throughput, unit-aware comparisons, shared time ranges, and honest + non-resampling warnings. +- Expanded Markdown reports, generator option semantics, Python 3.9 strict typing, adversarial + regression coverage, and Windows/macOS CI smoke tests. +- Introduced session format 2 while retaining format 1 reading compatibility. + ## 0.1.3 - 2026-07-29 - Rewrote the project story and workflow in a clearer, more personal voice. @@ -21,5 +45,5 @@ ## 0.1.0 - 2026-07-29 -- Initial release-ready alpha with recording, inspection, comparison, replay, reports, conversion, +- Initial alpha with recording, inspection, comparison, replay, reports, conversion, deterministic generators, headless plots, integrity verification, and typed Python API. diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 61aa451..11dcf62 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -3,17 +3,21 @@ I built Datary around a small promise: input remains inert, evidence stays local, and every result should be explainable. If you contribute, please help me keep that promise. -Use Python 3.9 or newer, install `.[dev]`, and run: +Use a supported Python version and install the development extras: ```bash -pytest -ruff check . -mypy +python -m pip install -e ".[dev]" +python -m pytest +python -m ruff check . +python -m ruff format --check . +python -m mypy python -m build ``` -Add deterministic tests for behavioural changes. If a quality rule needs a threshold, explain the -assumption and show the evidence behind its finding. +Behavioural changes need deterministic tests. Mathematical changes need a definition, an ordered +counterexample where relevant, and a statement of assumptions. Security changes need a regression +test for the boundary: traversal, symlink, failed publication, malicious manifest, control +sequence, or formula-like export data. -Please do not add telemetry, background network dependencies, data execution, or -backwards-incompatible session changes without an explicit design discussion and documentation. +Please do not add telemetry, background networking, data execution, or an always-on database. +Discuss session-format changes explicitly and keep older readers in mind. diff --git a/README.md b/README.md index 79c14a1..8a21a86 100644 --- a/README.md +++ b/README.md @@ -13,8 +13,9 @@ comparing, plotting, and documenting data emitted by programs and simulations. > Run a program, capture its output, validate the data, measure the behaviour, compare > experiments, and generate reproducible evidence. -**Status:** `0.1.3` is an honest alpha. Session format changes will be documented, but may not -remain backward compatible before 1.0. +**Status:** `0.2.0` is an alpha. This release concentrates on analytical correctness, safe +publication, bounded-memory analysis, and adversarial input handling. The session reader remains +compatible with format 1; new recordings use session format 2. ## Why I built it @@ -23,38 +24,30 @@ terminal output, one-off parsing scripts, screenshots, and notes. The program ha evidence around the run was fragile. I thought about that point for a while: what if one command could preserve the original output, -parse what was valid, explain what looked wrong, calculate useful measurements, and leave behind -enough context to reproduce the experiment later? That became Datary. +parse what was valid, explain what looked wrong, calculate useful measurements, and leave enough +context to reproduce the experiment later? That became Datary. -Oh! One important part is that Datary does not try to become the experiment. It does not execute -input files, decide whether your scientific method is sound, or quietly call one result “better.” -It records the evidence and gives you transparent tools for examining it. - -## How I think about a run - -A Datary workflow is deliberately small: - -1. Your program writes data to standard output or an ordinary file. -2. Datary preserves the raw input before interpreting it. -3. Valid records become clean JSON Lines and CSV; malformed input keeps an explanation. -4. Metrics and quality checks describe what happened and state their assumptions. -5. The session keeps the hashes, commands, reports, plots, and notes together. - -That is the whole idea: the next person looking at the result—even when that person is future -you—should be able to follow the trail from raw output to reported evidence. +Oh! One important boundary is that Datary does not try to become the experiment. It never executes +input, silently decides what “better” means, or claims that a clean report proves a sound +experiment. It records evidence and exposes the assumptions used to examine it. ## Install -Datary supports Python 3.9–3.14 and has no required runtime dependency. Matplotlib is optional. - -Install the current GitHub release: +Datary supports CPython 3.9–3.14. Core operation has no required runtime dependency; Matplotlib is +optional and uses the headless `Agg` backend. ```bash -python -m pip install https://github.com/devkyato/datary-lab/releases/download/v0.1.3/datary_lab-0.1.3-py3-none-any.whl +python -m pip install https://github.com/devkyato/datary-lab/releases/download/v0.2.0/datary_lab-0.2.0-py3-none-any.whl datary --version ``` -Or work from a local checkout: +For plotting: + +```bash +python -m pip install "datary-lab[plot]" +``` + +For a local checkout: ```bash python -m venv .venv @@ -63,11 +56,14 @@ python -m pip install -e ".[dev]" datary --version ``` +On macOS or Linux, activate with `source .venv/bin/activate`. + ## Five-minute local demonstration ```bash datary generate noisy-sensor --seed 1 | datary record demo --format jsonl --time-field timestamp datary inspect demo --quality +datary inspect demo --plot value datary report demo datary replay demo --no-timing datary compare demo demo --field value @@ -86,22 +82,31 @@ datary inspect readings.jsonl datary convert readings.jsonl --to csv ``` -Datary reads stdin, CSV, TSV, JSON arrays, JSON Lines, whitespace numeric rows, `key=value` -rows, headerless comma streams (`--format stream`), and existing sessions. Detection is -conservative: ambiguous or empty input requires `--format`. Inputs are inert data—never code. +Datary reads standard input, CSV, TSV, incrementally decoded JSON arrays, JSON Lines, +whitespace-delimited rows, `key=value` rows, headerless comma streams (`--format stream`), and +existing Datary sessions. CSV supports quoted embedded newlines. UTF-8 BOMs are accepted. +Duplicate JSON keys and non-finite JSON numbers are rejected with evidence. + +Detection is deliberately conservative. Empty or ambiguous input requires `--format`; Datary +never guesses merely to keep a pipeline moving. + +For delimiter-based formats, the documented `conservative-scalars-v1` policy converts empty cells, +lowercase JSON booleans, canonical integers, and canonical finite floats. Identifier-like values +such as `00123` and tokens such as `NA` remain strings. The policy name is recorded in the +manifest. ## Session directory -Each recording is self-contained. I chose ordinary files here on purpose: you can inspect, -copy, archive, or version a session without a Datary server. +Each recording is self-contained: ```text demo/ -|-- manifest.json # identity, schema, hashes, commands, privacy choices -|-- raw.log # exact source text -|-- records.jsonl # clean records -|-- invalid.jsonl # malformed-record reasons -|-- data.csv +|-- manifest.json # identity, schema, options, commands, artifact hashes +|-- manifest.sha256 # corruption check for the manifest itself +|-- raw.log # preserved input text +|-- records.jsonl # accepted records +|-- invalid.jsonl # rejected-record reasons +|-- data.csv # spreadsheet-safe convenience export |-- metrics.json |-- quality.json |-- notes.md @@ -109,90 +114,171 @@ demo/ `-- reports/ ``` -Existing names receive a deterministic numeric suffix; `--overwrite` is explicit. Absolute -working paths are redacted unless `--include-path` is used. Environment values are never stored. +Recording uses a sibling staging directory. With `--overwrite`, the previous session is retained +until its complete replacement is ready; a failed recording restores the original. Without +`--overwrite`, an existing name receives a deterministic numeric suffix. + +The ingest and analysis path is disk-backed and bounded by line, JSON-buffer, field, and file +limits rather than the number of records. Temporary SQLite is an implementation detail and is +removed before the session is published. ## Inspection, quality, and plots ```bash datary inspect demo --field value datary inspect demo --quality -datary inspect demo --plot value +datary inspect demo --plot value --plot-kind scatter --plot-format svg datary inspect demo --monotonic-field distance --counter-field packet_count --quality ``` -Checks cover missing/non-finite values, duplicates, changing shapes and types, frozen or -constant signals, spikes, robust outliers, high noise, backwards/duplicate timestamps, -irregular intervals, gaps, sequence loss, counter resets, and explicit monotonicity expectations. -Findings include evidence, thresholds, assumptions, explanations, and suggested investigation. -I treat them as leads to investigate, not verdicts. +Checks cover missing and invalid values, duplicate rows and timestamps, time moving backwards, +irregular intervals, large gaps, shape and type changes, frozen and constant signals, spikes, +robust outliers, high relative noise, sequence loss, counter resets, and explicit monotonicity +expectations. Findings carry the check ID, severity, field, affected records, evidence, threshold, +explanation, assumptions, and a suggested investigation. + +Zero-MAD signals and missing-value index positions are handled explicitly. Schema changes compare +field names, not only record length. + +Plots support PNG and SVG line, scatter, step, and histogram output, with missing-data markers +where an x-position is available. User field names are sanitized before becoming filenames, plot +directories may not be symlinks, and existing plots require `--overwrite-plot`. + +## Metrics and engineering roles -Matplotlib uses the non-interactive `Agg` backend. The Python plotting API creates PNG or SVG -line, scatter, step, and histogram plots without opening windows. +General metrics retain observation order for net change and adjacent differences while using a +separate sorted view for quantiles. They include count, missing count, extrema, mean, median, +sample variance, sample standard deviation, percentiles, sum, RMS, net rate of change, and mean +absolute adjacent difference. -## Metrics and comparison +Timing accepts numeric elapsed seconds or timezone-aware ISO 8601 timestamps. It reports interval +statistics, population jitter, effective sample rate, gaps, backwards time, and duplicate +timestamps—including non-adjacent duplicates. -General metrics include count, missing count, extrema, mean, median, variance, standard -deviation, percentiles, sum, RMS, rate of change, and mean absolute difference. Timing metrics -include intervals, jitter, effective sampling rate, gaps, and duplicate timestamps. The public -module also implements defined control-response and network metrics; see -[docs/metrics.md](docs/metrics.md). +Control and network metrics are opt-in because their field meanings cannot be inferred honestly: + +```bash +python controller.py | datary record controller \ + --format jsonl \ + --time-field timestamp \ + --target-field target \ + --response-field response + +python network.py | datary record network \ + --format jsonl \ + --time-field timestamp \ + --sequence-field sequence \ + --latency-field latency_ms \ + --bytes-field bytes +``` + +The field roles, definitions, assumptions, and warnings flow into inspection and reports. See +[the metric definitions](docs/metrics.md). + +## Comparing experiments ```bash datary compare baseline improved --field error --goal lower:error datary compare run-1 run-2 run-3 --report comparison.md --format markdown ``` -Fields are aligned by name and ordered deterministically. I thought this part deserved a firm -rule: Datary does not call an experiment better without an explicit goal. Incomparable fields -produce warnings instead of a confident-looking guess. +Fields align by name and output ordering is deterministic. Datary reports per-source count, range, +mean, median, standard deviation, declared units, sampling rates, individual time ranges, and the +shared range when available. Unit conflicts and non-overlapping time ranges block confident goal +claims. -## Replay, reports, and generators +I thought this part deserved a firm rule: without `--goal lower:FIELD` or +`--goal higher:FIELD`, Datary does not label an experiment better. Current alpha comparisons do +not interpolate or resample; differing rates are disclosed and block percentage-improvement +claims. + +## Replay, reports, conversion, and generators ```bash datary replay demo --speed 2 +datary replay demo --no-timing --format csv datary report demo --format json --output demo.json +datary convert readings.csv --to jsonl datary generate pid-response --seed 7 --duration 20 --sample-rate 50 ``` -Profiles: `sine`, `noisy-sensor`, `frozen-sensor`, `missing-samples`, -`duplicate-samples`, `pid-response`, `motor-speed`, `battery-drain`, -`network-latency`, and `packet-loss`. Equal profile, seed, and options produce identical output. +Replay preserves relative numeric or ISO 8601 timing unless `--no-timing` is selected. Virtual +replay remains available through the Python API and tests. + +Markdown reports contain identity, reproduction commands, schema, statistics, timing, +engineering metrics, complete quality evidence, integrity status, hashes, assumptions, and links +to local plots. JSON reports contain the same structured evidence. + +CSV convenience exports neutralize formula-like strings and headers with a leading apostrophe; +canonical values remain unchanged in `records.jsonl`. Conversion writes an invalid-record sidecar +rather than silently discarding malformed input. + +Profiles are `sine`, `noisy-sensor`, `frozen-sensor`, `missing-samples`, +`duplicate-samples`, `pid-response`, `motor-speed`, `battery-drain`, `network-latency`, and +`packet-loss`. Equal profile, seed, options, and Datary version produce identical records. Profile +defaults create their named anomaly, while an explicit `--missing-rate 0` or +`--duplicate-rate 0` is honoured exactly. -## Python API +## Typed Python API ```python from datary import Session, compare_sessions, inspect_source session = Session.open("demo") summary = inspect_source(session) -comparison = compare_sessions(["baseline", "improved"], fields=["error", "response"]) +comparison = compare_sessions( + ["baseline", "improved"], + fields=["error", "response"], +) ``` -Only these names are the stable public API in this alpha. +Only `Session`, `inspect_source`, and `compare_sessions` are stable public names during the alpha. +Record values use a recursive JSON type rather than unrestricted Python objects. ## Reproducibility, privacy, and security -Raw input, clean records, invalid reasons, configuration, exact follow-up commands, timezone-aware -timestamps, and SHA-256 hashes remain together. Critical JSON is written atomically. Limits apply -to line size and field count. Session manifests cannot escape through hash paths, symlinks are -rejected for trusted session artifacts, reports never interpret formulas or commands, and there -are no background network calls. See [reproducibility](docs/reproducibility.md), -[privacy](docs/privacy.md), and [SECURITY.md](SECURITY.md). +Absolute working paths are `` unless `--include-path` is supplied. Environment values +are never copied. Datary has no telemetry, background service, or network call. -## Limitations +SHA-256 covers the manifest and every recording-time evidentiary file, including rejected records +and notes. Plots and reports are derived later and are not silently added to the immutable +recording manifest. These checks detect accidental corruption; they are not signatures and do not +prove authenticity against an attacker who can rewrite the entire session. -Datary is not a spreadsheet application, a replacement for statistical expertise, a cloud -observability platform, a guarantee that collected data is scientifically valid, or a substitute -for properly designed experiments. The alpha currently keeps valid records in memory for final -whole-session statistics after streaming them to disk; input ingestion itself is bounded per line. -Timestamp parsing currently expects numeric elapsed time for timing analysis and replay. +Input remains inert. There is no expression evaluation, command expansion, pickle loading, or +formula execution. Session and output paths reject traversal and symlink redirection in trusted +locations; terminal and Markdown output escape untrusted control or markup content. -I would rather state those limits plainly than hide them behind an alpha label. The open issues -track the work needed to remove them. +Read [reproducibility](docs/reproducibility.md), [privacy](docs/privacy.md), and +[security policy](SECURITY.md) before sharing sensitive sessions. + +## Honest limitations + +Datary is not: + +- a spreadsheet application; +- a replacement for statistical or domain expertise; +- a cloud observability platform; +- a guarantee that collected data is scientifically valid; or +- a substitute for properly designed experiments. + +The alpha does not authenticate session authorship, automatically infer physical unit +conversions, interpolate comparison series, or decide domain-specific thresholds. Plotting +materializes the selected records for Matplotlib, so very large plots should be downsampled before +rendering. JSON array decoding is incremental but one pending JSON value is capped at 16 MiB. ## Contributing and licence -If the workflow sounds useful, I would be glad to have another set of eyes on it. Run `pytest`, -`ruff check .`, and `mypy`, then see [CONTRIBUTING.md](CONTRIBUTING.md). Datary is MIT-licensed; -copyright © 2026 devkyato. +```bash +python -m pytest +python -m ruff check . +python -m ruff format --check . +python -m mypy +python -m build +``` + +If the workflow sounds useful, I would be glad to have another set of eyes on it. Please include +deterministic regression tests and explain the mathematical or security assumption behind a +change. See [CONTRIBUTING.md](CONTRIBUTING.md). + +Datary is MIT-licensed; copyright © 2026 devkyato. diff --git a/SECURITY.md b/SECURITY.md index b0321b0..0776171 100644 --- a/SECURITY.md +++ b/SECURITY.md @@ -1,9 +1,31 @@ # Security -Datary treats all input as untrusted inert data. It never evaluates records, executes input files, -expands formulas, or performs background network calls. Report vulnerabilities privately to the -repository owner. Avoid attaching confidential recordings; provide a minimal synthetic reproducer. +Datary treats input and session content as untrusted inert data. It does not evaluate expressions, +execute input files, deserialize Python objects, invoke shells from records, expand spreadsheet +formulas, or perform background network requests. -Session consumers reject traversal paths and symlinked trusted artifacts. Users remain responsible -for filesystem permissions and for reviewing data before sharing reports. +## Defences +- conservative format detection and explicit ambiguity errors; +- line, JSON-buffer, field, manifest, and source-file safety limits; +- duplicate-key and non-finite JSON rejection; +- staging plus recoverable atomic session overwrite; +- traversal and symlink checks at session and generated-output trust boundaries; +- atomic critical metadata and report writes; +- spreadsheet-formula neutralization in CSV convenience output; +- terminal-control and Markdown/HTML escaping; +- SHA-256 coverage of the manifest and all evidentiary files; +- structural parsing and record-count checks during integrity verification. + +Checksums detect corruption; they do not authenticate an author against an attacker who can +rewrite the complete session. Use external signatures and trusted storage when authenticity is +required. + +## Reporting a vulnerability + +Report security problems privately to the repository owner. Do not attach confidential +recordings. A minimal deterministic synthetic reproducer, affected version, operating system, and +expected boundary are enough to begin investigation. + +Users remain responsible for filesystem permissions, trusted installation sources, producer +program safety, and privacy review before sharing sessions. diff --git a/docs/architecture.md b/docs/architecture.md index 3021b18..dd2cbed 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -1,13 +1,52 @@ # Architecture -I wanted the architecture to follow the way I explain a Datary run: capture first, interpret -second, and publish evidence only after the recording is complete. +I wanted Datary's architecture to follow the way I explain a run: capture first, interpret second, +and publish only after every critical artifact is ready. -The CLI delegates to isolated modules for format detection and parsing, session recording and -loading, metrics, quality analysis, comparison, replay, plotting, reports, conversion, and -generators. Input flows line-by-line into `raw.log` and the selected parser. A staging directory is -renamed into place only after critical artifacts and hashes are complete. +```text +stdin/file + | + +--> raw.log + | +parser --> records.jsonl / invalid.jsonl + | +temporary disk-backed analysis + | +metrics + quality + data.csv + manifest + hashes + | +atomic session publication +``` -Oh! This boundary matters: there is no service, central database, or network layer hiding behind -the command. Sessions are ordinary local files, and core recording remains usable without the -optional plotting dependency. +## Boundaries + +- `formats.py` detects only clear input signatures. +- `parsers.py` turns inert text into JSON-compatible records. +- `recorder.py` owns staging, flush/fsync, interruption, overwrite recovery, and publication. +- `analysis_store.py` uses temporary local SQLite for exact bounded-memory distribution and + quality calculations. It is not a catalogue or session dependency. +- `sessions.py` validates manifests, rejects symlinked trust boundaries, reads records, and + verifies integrity. +- `metrics.py` and `quality.py` provide the direct in-memory analytical API. +- `comparison.py`, `replay.py`, `plotting.py`, `reports.py`, and `conversion.py` consume those + boundaries without executing data. +- `cli.py` is orchestration and terminal rendering, not analytical logic. + +The temporary database grows with input on disk, while resident analysis state is bounded by +field count and small batches. Exact percentiles use indexed disk ordering rather than retaining +every measurement in a Python list. + +The recorder regression feeds 50,000 generated records without first storing the source and +enforces a 32 MiB Python-heap ceiling. That catches accidental record-list accumulation; it is not +an operating-system RSS guarantee or a substitute for deployment-specific disk sizing. + +Oh! SQLite here is deliberately disposable. Published sessions remain ordinary portable files; +there is no daemon, account, central database, network layer, or hidden migration service. + +## Failure model + +New recordings are built beside the target. An overwrite renames the old directory to a private +backup only after the replacement is complete, then atomically renames the replacement into place. +If publication fails, the backup is restored. Critical streams flush and `fsync` before analysis. +Critical JSON and text outputs use same-directory temporary files and atomic replacement. + +Core operation uses only the standard library. Matplotlib is imported lazily after forcing `Agg`. diff --git a/docs/input-formats.md b/docs/input-formats.md index 3a9d98d..b5bb36c 100644 --- a/docs/input-formats.md +++ b/docs/input-formats.md @@ -1,7 +1,46 @@ # Input formats -Supported identifiers are `csv`, `tsv`, `json`, `jsonl`, `whitespace`, `keyvalue`, and `stream`. -CSV/TSV require a unique non-empty header. JSON is an array of objects; JSONL is one object per -line. Whitespace and stream fields are named `field_1`, `field_2`, and so on. Detection refuses -empty and ambiguous comma-numeric data. Use `--format` to resolve ambiguity. +Supported format names are `csv`, `tsv`, `json`, `jsonl`, `whitespace`, `keyvalue`, and `stream`. +An existing Datary directory is also a valid inspection, comparison, replay, or report source. +## Parsing rules + +- CSV and TSV use Python's strict CSV parser, require a unique non-empty header, and support quoted + fields containing commas, tabs, quotes, and physical newlines. +- JSON is a top-level array of objects decoded incrementally. One pending value is limited to + 16 MiB. +- JSON Lines requires one object per logical line. +- Whitespace rows receive `field_1`, `field_2`, and subsequent names. +- `key=value` uses shell-like quoting for token boundaries, but never executes tokens. Duplicate + keys are rejected. +- `stream` reads headerless comma rows and assigns generated field names. + +UTF-8 BOMs are stripped at the parser boundary. JSON duplicate keys, non-string keys, unsupported +value types, NaN, and infinity are rejected. Malformed input remains in `raw.log`; recording writes +an explanation to `invalid.jsonl`. + +## Conservative scalar coercion + +Delimiter-based formats use `conservative-scalars-v1`: + +- an empty cell becomes a missing value; +- lowercase `true` and `false` become booleans; +- canonical integers such as `0`, `-3`, and `42` become integers; +- canonical finite decimal or exponent forms become floats; +- `00123`, `NA`, `N/A`, `none`, and other identifier-like/domain tokens remain strings. + +The policy is written to the session manifest. JSON retains the types explicitly represented by +the source document. + +## Detection + +Detection samples at most 262,144 bytes and 20 lines. It recognizes clear JSON arrays, uniform +JSON objects, key-value rows, TSV, header-like CSV, and whitespace numeric rows. Empty input and +headerless comma-numeric input are ambiguous by design: + +```bash +datary inspect numbers.txt --format stream +``` + +Oh! I would rather ask for one explicit flag than produce a plausible-looking parse using the +wrong schema. diff --git a/docs/metrics.md b/docs/metrics.md index 52dfe97..0ac59a0 100644 --- a/docs/metrics.md +++ b/docs/metrics.md @@ -1,16 +1,71 @@ -# Metrics +# Metric definitions -General summaries use sample variance/standard deviation and linearly interpolated percentiles. -RMS is `sqrt(mean(x²))`; mean absolute difference is the mean adjacent absolute step. +Datary keeps ordering and distribution operations separate. Observation order is used for change +metrics; a sorted view is used only for median and percentile calculations. -Timing uses adjacent numeric timestamps: jitter is the population standard deviation of positive -intervals; effective rate is the reciprocal mean interval; gaps exceed twice the mean interval. +## General numeric metrics -Control metrics require numeric time, target, and response. Rise time is 10–90% first crossing; -overshoot is `(peak-final target)/|step| × 100`; settling is the first point after which response -stays within 2% of step amplitude; error integrals use right-rectangle elapsed-time integration. +For finite values \(x_1,\ldots,x_n\): -Network metrics require sequence and latency. Loss is missing IDs inside the observed inclusive -range; duplicate rate counts repeated IDs; out-of-order counts descending adjacent IDs. Latency -uses general summaries. Byte totals require a byte-count field. +- `count`: all records considered; +- `valid_count`: finite numeric values; +- `missing_count`: records without a finite numeric value; +- `minimum`, `maximum`, `sum`, and arithmetic `mean`; +- `median` and linearly interpolated percentiles at 5, 25, 50, 75, 95, and 99%; +- sample `variance` and sample `standard_deviation` (zero for one value); +- `root_mean_square`: \(\sqrt{\sum x_i^2/n}\); +- `rate_of_change`: \(x_n-x_1\), the net ordered change per record series; +- `mean_absolute_difference`: \(\sum_{i=2}^{n}|x_i-x_{i-1}|/(n-1)\). +Missing and non-numeric values are omitted from numeric order. Integer values beyond ±2^53 are +preserved as records but excluded from binary64 analysis and receive an `invalid-values` finding; +this avoids silently rounding identifiers or counters. Their record positions remain available to +quality analysis. + +## Timing metrics + +Time accepts finite numeric seconds or timezone-aware ISO 8601 strings. Naive date-times are not +treated as time because their offset is ambiguous. + +For positive adjacent intervals: + +- mean, median, minimum, and maximum interval; +- population standard deviation as `jitter`; +- effective sample rate \(1/\text{mean interval}\); +- a gap count for intervals at least twice the median; +- global duplicate-timestamp count, including non-adjacent duplicates; +- backwards-timestamp count. + +The start, end, and duration are epoch-second values when ISO 8601 input is used. + +## Control-system metrics + +Required roles are `time`, `target`, and `response`. Supply them with `--time-field`, +`--target-field`, and `--response-field`. + +- Rise time: elapsed time between the first recorded 10% and 90% response crossings. +- Peak: maximum response for an upward step or minimum response for a downward step. +- Percentage overshoot: peak excursion beyond the final target divided by absolute step span. +- Settling time: first recorded point after the last excursion outside a ±2% band. The band uses + the larger of absolute step span and target magnitude. +- Steady-state error: final target minus final response. +- MAE and RMSE: ordinary sample error summaries. +- IAE and ISE: trapezoidal integration over non-decreasing time intervals. + +Rise, overshoot, and settling are reported only when the recorded target is stable. Crossings are +not interpolated. Warnings disclose changing targets and excluded backwards intervals. + +## Network metrics + +Required roles are `sequence` and `latency`; `bytes` and `time` are optional. + +- Packet-loss estimate: missing integer IDs inside the observed inclusive sequence range divided + by the expected range. +- Duplicate-packet rate: repeated valid sequence IDs divided by valid IDs. +- Out-of-order count: descending adjacent valid IDs. +- Mean, median, percentile latency, and latency jitter. Jitter is the mean absolute difference + between consecutive valid latency observations. +- Throughput: total non-negative byte count divided by positive observed duration. + +These definitions assume contiguous integer sequence identifiers and consistent latency/time +units. They are measurement summaries, not transport-protocol truth. diff --git a/docs/privacy.md b/docs/privacy.md index abc291a..d8a549e 100644 --- a/docs/privacy.md +++ b/docs/privacy.md @@ -1,9 +1,18 @@ # Privacy -I wanted a useful session to be shareable without quietly publishing the shape of someone’s -laptop. Working-directory paths are therefore `` by default; enable `--include-path` -only when that provenance is genuinely useful. +I wanted a session to be shareable without quietly publishing the shape of someone's laptop. +Working-directory paths are therefore `` by default. Use `--include-path` only when that +provenance is intentionally part of the record. -Environment variables are never copied into sessions, and Datary has no telemetry or background -network calls. Oh! The raw data can still contain secrets because Datary preserves what it -receives. Check file permissions and review or redact a session before sharing it. +Datary does not copy environment variables, send telemetry, start a background service, or make +network calls. The workspace is an ordinary directory selected by `DATARY_WORKSPACE` or +`--workspace`. + +Oh! Raw capture is intentionally faithful, so it can contain credentials, personal data, device +identifiers, or secrets printed by the producer. Hashes do not anonymize data, and reports can +repeat field names or values. Review permissions and redact a copy—not the original evidence—before +sharing. + +User-supplied commands and parameters are included only when explicitly passed. CSV exports +neutralize spreadsheet formulas, but privacy review is still required before opening or sending +them. diff --git a/docs/quality-checks.md b/docs/quality-checks.md index 542d270..661ecaa 100644 --- a/docs/quality-checks.md +++ b/docs/quality-checks.md @@ -1,18 +1,54 @@ # Quality checks -Findings are deterministic objects containing ID, severity, field, affected range, evidence, -threshold, explanation, assumptions, and investigation advice. Rules cover missing/non-finite -values, duplicate rows/timestamps, backward time, irregularity/gaps, record shape/type changes, -constant/frozen signals, robust spikes/outliers, high relative noise, and sequence gaps. -Thresholds are heuristics and must be reviewed against domain knowledge. +Every finding contains a stable check ID, severity, field, affected record or range, evidence, +threshold, explanation, assumptions, and suggested investigation. Findings are deterministic and +sorted; they are leads for review, not declarations of scientific invalidity. -Use repeatable inspection options for domain expectations: +## Structural and value checks + +- `empty-data` +- `missing-values` +- `non-finite` +- `type-change` +- `record-shape-change` +- `record-length-change` +- `duplicate-rows` + +Shape comparison uses the actual field-name set, so `{a,b}` changing to `{a,c}` is detected even +though both records have length two. Non-finite JSON input is rejected by the parser and preserved +as invalid evidence; the direct quality API also detects non-finite Python floats. + +## Signal checks + +- `constant-signal` +- `frozen-values` +- `sudden-spikes` +- `outliers` +- `high-noise` +- `monotonicity-violation` +- `counter-reset` + +Frozen ranges retain original record positions and missing values break a run. Outliers use a +six-MAD rule; when MAD is zero, values different from the median are still surfaced. Spikes use +ten times the median absolute adjacent step, with a non-zero-step fallback when that median is +zero. High noise uses coefficient of variation and is meaningful only for ratio-scale signals. + +Monotonic and counter expectations are explicit: ```bash -datary inspect session --monotonic-field distance --counter-field packet_count --quality +datary inspect session \ + --monotonic-field distance \ + --counter-field packet_count \ + --quality ``` -`monotonicity-violation` reports decreases in explicitly non-decreasing fields. -`counter-reset` reports decreases in fields explicitly identified as counters. Missing values are -ignored between consecutive valid values; findings document that assumption and recommend checking -record ordering, restarts, rollover, and counter width. +## Timing and sequence checks + +- `duplicate-timestamps`, including duplicates that are not adjacent; +- `timestamps-backwards`; +- `irregular-timing`, outside ±20% of the median positive interval; +- `large-timing-gaps`, over twice the median positive interval; +- `packet-loss`, when a sequence field role is supplied. + +Domain thresholds vary. Treat the raw records, units, system limits, and experimental design as +the final authority. diff --git a/docs/releases/0.2.0.md b/docs/releases/0.2.0.md new file mode 100644 index 0000000..480ed19 --- /dev/null +++ b/docs/releases/0.2.0.md @@ -0,0 +1,75 @@ +# Datary 0.2.0 — corrective audit release + +I treated this release differently from an ordinary feature update. The critique of Datary was +right about the central problem: a tool that calls its output reproducible evidence has to be +especially careful about mathematics, provenance, and failure behavior. So I froze feature work +and followed the evidence from input bytes through parsing, analysis, reports, package builds, and +the release artefacts themselves. + +Oh! On the statistics point, the old implementation really did sort values too early. That made an +order-dependent metric describe sorted spacing instead of observed change. Datary now keeps +observation order for adjacent differences and net change, while using a separate sorted copy only +for percentiles. Independent regression cases lock that behavior down. + +I thought too on the integrity point that the wording matters as much as the implementation. +Sessions now reject symlinked trusted artefacts before resolving them, hash malformed-record +evidence, publish a detached checksum file for the manifest, and perform structural verification. +These hashes detect accidental corruption; they do not authenticate authorship. Datary says that +plainly because a locally editable hash is not a digital signature. + +Recording and conversion analysis now use a temporary SQLite store instead of Python lists that +grow with the recording. Raw lines and parsed records are still streamed immediately. Replacement +sessions are built and verified first, then swapped through a recoverable backup so a failed +overwrite does not destroy the last good session. + +The ingestion boundary is stricter too: + +- CSV and TSV accept quoted multiline records. +- JSON arrays are decoded incrementally under an explicit size ceiling. +- duplicate JSON keys, non-finite numbers, oversized lines, and excessive fields are rejected with + evidence instead of entering accepted records; +- conservative scalar parsing preserves identifiers such as `00123` and strings such as `NA`; +- BOM handling is shared by detection and parsing; and +- ambiguous detection stops and asks for `--format`. + +Analysis received the same pass. Timing no longer repeats median work, ISO 8601 timestamps with +time zones are supported, quality findings preserve original record positions, schema changes +compare field identities, zero-MAD signals use documented fallbacks, and control/network metrics +validate their assumptions. Comparisons now expose counts, missingness, spread, units, and shared +time ranges. Datary refuses to announce an improvement when units or sampling conditions make the +claim dishonest. + +On the output side, formula-like CSV cells are neutralized in convenience exports while canonical +JSONL values remain unchanged. Plot names are sanitized and contained, reports escape untrusted +Markdown and show integrity results prominently, terminal control characters are rendered safely, +and all file outputs use atomic publication where practical. + +This is still an alpha. Datary does not yet sign sessions, convert units, resample comparison +series, or downsample very large plots. Those boundaries are documented and tracked instead of +being hidden behind a confident-looking result. + +## Install + +```bash +python -m pip install "datary-lab[plot]==0.2.0" +``` + +## Five-minute check + +```bash +datary --version +datary generate noisy-sensor --seed 1 | +datary record demo --format jsonl --time-field timestamp +datary inspect demo --quality +datary report demo +datary doctor +``` + +## Release evidence + +The release contains the wheel, source distribution, and `SHA256SUMS`. CI covers Python 3.9–3.14, +strict Mypy, Ruff, Pytest, package installation, the command-line demonstration, deterministic +generation, example reports, and smoke tests on Linux, Windows, and macOS. + +Thank you for the hard critique. It made Datary smaller in its claims and much stronger in the +places where those claims matter. diff --git a/docs/reproducibility.md b/docs/reproducibility.md index 2e6a3b5..4259437 100644 --- a/docs/reproducibility.md +++ b/docs/reproducibility.md @@ -1,16 +1,23 @@ # Reproducibility -I think of a session as a small evidence bundle rather than a convenient export. Keep the -directory intact and the raw bytes, parsed records, invalid reasons, options, hashes, version, and -commands form an auditable chain. +I think of a Datary session as a small evidence bundle, not merely a convenient export. -The important reference points are: +- `raw.log` answers, “What text did the program produce?” +- `records.jsonl` answers, “What did this parser accept?” +- `invalid.jsonl` answers, “What was rejected, where, and why?” +- `manifest.json` answers, “Which Datary version, policy, roles, options, units, and commands + describe the run?” +- metrics, quality, plots, reports, and notes keep interpretation beside the source evidence. -- `raw.log` answers, “What did the program actually produce?” -- `records.jsonl` answers, “What did Datary accept as structured data?” -- `invalid.jsonl` answers, “What was rejected, and why?” -- `manifest.json` answers, “Which version, options, hashes, and follow-up commands describe this?” +Session format 2 hashes the manifest and all evidentiary files. Verification also parses record +files and checks their counts. These are corruption checks, not authorship signatures. If +adversarial provenance matters, sign or version the whole session with an external trusted tool. -Synthetic output is stable for the same profile, seed, and options on a compatible Datary version. -Reports include the version and assumptions so a polished summary never loses its connection to -the recorded evidence. +Synthetic records are deterministic for the same Datary version, profile, seed, and options. +Floating-point results can still vary at the last bit across Python or platform implementations; +the release verifier compares serialized generator output in separate processes on every CI +version. + +Reports state their version, commands, field roles, thresholds, warnings, integrity state, and +assumptions. A reproducible calculation is still not automatically a valid experiment—the raw +input, units, collection method, and domain design remain part of the evidence. diff --git a/docs/session-format.md b/docs/session-format.md index 192644e..7818a1e 100644 --- a/docs/session-format.md +++ b/docs/session-format.md @@ -1,7 +1,46 @@ # Session format -Format version `1` uses UTF-8 JSON/JSONL/CSV and ordinary directories. `manifest.json` is the source -of identity and includes timezone-aware start/end times, counts, fields, units, parser warnings, -privacy-safe provenance, commands, and SHA-256 hashes. `raw.log` is exact text; `records.jsonl` -contains valid objects; `invalid.jsonl` contains reasons. Unknown manifest keys should be ignored. +New Datary 0.2 recordings use session format `2`. The reader also accepts format `1` sessions. +Unknown manifest keys should be ignored by compatible readers. +## Required files + +| Path | Purpose | +|---|---| +| `manifest.json` | Identity, parser policy, schema, field roles, counts, privacy choices, commands, and artifact hashes | +| `manifest.sha256` | Lowercase SHA-256 digest of the exact `manifest.json` bytes | +| `raw.log` | Every input line received as text | +| `records.jsonl` | Valid JSON objects, one per line | +| `invalid.jsonl` | Rejected logical records with record number and reason | +| `data.csv` | Spreadsheet-safe convenience representation of valid records | +| `metrics.json` | General, timing, and requested engineering metrics | +| `quality.json` | Structured quality findings | +| `notes.md` | User-editable local notes | +| `plots/` | Generated PNG or SVG plots | +| `reports/` | Generated Markdown or JSON reports | + +All text is UTF-8. JSON output rejects NaN and infinities. Start and end timestamps are ISO 8601 +with timezone information. + +## Integrity model + +`manifest.json` contains SHA-256 digests for `raw.log`, `records.jsonl`, `invalid.jsonl`, +`data.csv`, `metrics.json`, `quality.json`, and `notes.md`. `manifest.sha256` covers the manifest. +Verification also parses the record files and compares their counts with the manifest. + +Plots and reports are generated after recording and are not automatically added to the immutable +manifest. Editing `notes.md` intentionally changes its digest and will be reported by verification. + +I want the boundary here to be unambiguous: this detects accidental corruption and incomplete +copies. It is not a digital signature. Someone able to rewrite the whole directory can also +rewrite every checksum. + +## Publication and compatibility + +A recording is assembled in a sibling staging directory. With `--overwrite`, the old directory is +renamed to a backup only after the replacement is complete; publication failure restores it. +Temporary analysis storage is never part of the published format. + +Format 1 did not require `manifest.sha256`, did not hash every evidentiary artifact, and performed +less structural validation. Datary continues to open it so existing experiments are not stranded, +but new integrity guarantees apply only to format 2. diff --git a/docs/tutorials.md b/docs/tutorials.md index 365568f..69e2d02 100644 --- a/docs/tutorials.md +++ b/docs/tutorials.md @@ -10,8 +10,9 @@ data so I can repeat the exact run later: ```bash datary generate motor-speed --seed 4 --duration 10 | - datary record motor --format jsonl --time-field timestamp --unit speed_rpm=rpm -datary inspect motor --quality --plot speed_rpm,current_a + datary record motor --format jsonl --time-field timestamp \ + --target-field target_rpm --response-field speed_rpm --unit speed_rpm=rpm +datary inspect motor --quality --plot speed_rpm,current_a --plot-kind line datary report motor ``` @@ -32,6 +33,19 @@ datary compare baseline improved --field latency_ms --goal lower:latency_ms The `lower:` goal is what gives Datary permission to calculate improvement or regression. Without it, Datary shows the measurements but does not invent a winner. +For network metrics, record the field roles explicitly: + +```bash +datary record network-run --format jsonl \ + --time-field timestamp \ + --sequence-field sequence \ + --latency-field latency_ms \ + --bytes-field bytes +``` + +That gives the report enough meaning to calculate loss, duplicate rate, out-of-order packets, +latency summaries, jitter, and throughput without guessing which column means what. + ## Counter and monotonicity expectations Some fields only make sense when I tell Datary how they should behave: diff --git a/examples/datasets/noisy-sensor-seed-1.jsonl b/examples/datasets/noisy-sensor-seed-1.jsonl index d9274d4..a2ed783 100644 --- a/examples/datasets/noisy-sensor-seed-1.jsonl +++ b/examples/datasets/noisy-sensor-seed-1.jsonl @@ -1,21 +1,101 @@ {"timestamp": 0.0, "value": 20.064409237657774} {"timestamp": 0.1, "value": 20.172305697081818} -{"timestamp": 0.2, "value": 20.129687135430082} -{"timestamp": 0.3, "value": 20.318178009797514} -{"timestamp": 0.4, "value": 20.312114475078108} -{"timestamp": 0.5, "value": 20.514173526862105} -{"timestamp": 0.6, "value": 20.66597393908556} -{"timestamp": 0.7, "value": 20.607930682461994} -{"timestamp": 0.8, "value": 20.69170829016585} -{"timestamp": 0.9, "value": 20.80704917255715} -{"timestamp": 1.0, "value": 20.805057052242127} -{"timestamp": 1.1, "value": 20.892165721512733} -{"timestamp": 1.2, "value": 21.026501286564805} -{"timestamp": 1.3, "value": 20.97639091199051} -{"timestamp": 1.4, "value": 21.00162200416873} -{"timestamp": 1.5, "value": 20.97793268730595} -{"timestamp": 1.6, "value": 20.91649199231581} -{"timestamp": 1.7, "value": 21.03591726892986} -{"timestamp": 1.8, "value": 21.00755856169906} -{"timestamp": 1.9, "value": 20.897489961463332} -{"timestamp": 2.0, "value": 20.882838266993875} +{"timestamp": 0.2, "value": 20.201986121241973} +{"timestamp": 0.3, "value": 20.25729302411276} +{"timestamp": 0.4, "value": 20.334809681553445} +{"timestamp": 0.5, "value": 20.48099226444579} +{"timestamp": 0.6, "value": 20.513537314894492} +{"timestamp": 0.7, "value": 20.572376214982565} +{"timestamp": 0.8, "value": 20.72732168972371} +{"timestamp": 0.9, "value": 20.789995639860415} +{"timestamp": 1.0, "value": 20.868794399824807} +{"timestamp": 1.1, "value": 20.84550881287467} +{"timestamp": 1.2, "value": 20.932289350148555} +{"timestamp": 1.3, "value": 20.96032109739856} +{"timestamp": 1.4, "value": 20.910158279925422} +{"timestamp": 1.5, "value": 21.024394845537106} +{"timestamp": 1.6, "value": 21.015609158091447} +{"timestamp": 1.7, "value": 21.111120412614504} +{"timestamp": 1.8, "value": 20.983996089743695} +{"timestamp": 1.9, "value": 20.93906497227917} +{"timestamp": 2.0, "value": 20.970935285577127} +{"timestamp": 2.1, "value": 20.873148929058544} +{"timestamp": 2.2, "value": 20.85394795512725} +{"timestamp": 2.3, "value": 20.727427998865092} +{"timestamp": 2.4, "value": 20.686371771219182} +{"timestamp": 2.5, "value": 20.64968657802831} +{"timestamp": 2.6, "value": 20.550313722944114} +{"timestamp": 2.7, "value": 20.433803492665472} +{"timestamp": 2.8, "value": 20.280872748926104} +{"timestamp": 2.9, "value": 20.261510417832035} +{"timestamp": 3.0, "value": 20.14496318214583} +{"timestamp": 3.1, "value": 20.077604042276253} +{"timestamp": 3.2, "value": 19.952437503462424} +{"timestamp": 3.3, "value": 19.896663576413527} +{"timestamp": 3.4, "value": 19.74188088470197} +{"timestamp": 3.5, "value": 19.659314975314352} +{"timestamp": 3.6, "value": 19.590818378636428} +{"timestamp": 3.7, "value": 19.415819628482783} +{"timestamp": 3.8, "value": 19.368059096053578} +{"timestamp": 3.9, "value": 19.287232412368407} +{"timestamp": 4.0, "value": 19.342228290668235} +{"timestamp": 4.1, "value": 19.177079787652783} +{"timestamp": 4.2, "value": 19.16103523877679} +{"timestamp": 4.3, "value": 19.11480281635923} +{"timestamp": 4.4, "value": 19.034354254601137} +{"timestamp": 4.5, "value": 18.94492804097838} +{"timestamp": 4.6, "value": 19.05455119490034} +{"timestamp": 4.7, "value": 18.979716903952717} +{"timestamp": 4.8, "value": 19.039733224001406} +{"timestamp": 4.9, "value": 18.95228414612017} +{"timestamp": 5.0, "value": 19.01917657523837} +{"timestamp": 5.1, "value": 19.137026384902533} +{"timestamp": 5.2, "value": 19.188095543682472} +{"timestamp": 5.3, "value": 19.10260962671918} +{"timestamp": 5.4, "value": 19.160595138489565} +{"timestamp": 5.5, "value": 19.292246452549517} +{"timestamp": 5.6, "value": 19.405145428293615} +{"timestamp": 5.7, "value": 19.457339691371327} +{"timestamp": 5.8, "value": 19.550575555977595} +{"timestamp": 5.9, "value": 19.576681513921034} +{"timestamp": 6.0, "value": 19.749924087491763} +{"timestamp": 6.1, "value": 19.873680112784747} +{"timestamp": 6.2, "value": 19.895126971168306} +{"timestamp": 6.3, "value": 19.945139497301177} +{"timestamp": 6.4, "value": 20.078608163666306} +{"timestamp": 6.5, "value": 20.25320288837966} +{"timestamp": 6.6, "value": 20.224856503921558} +{"timestamp": 6.7, "value": 20.400256081226317} +{"timestamp": 6.8, "value": 20.444562936881333} +{"timestamp": 6.9, "value": 20.571883042289315} +{"timestamp": 7.0, "value": 20.644760519381386} +{"timestamp": 7.1, "value": 20.72976187928949} +{"timestamp": 7.2, "value": 20.86872823366432} +{"timestamp": 7.3, "value": 20.87147214280896} +{"timestamp": 7.4, "value": 20.96539373307989} +{"timestamp": 7.5, "value": 20.930929341059567} +{"timestamp": 7.6, "value": 20.943940341640136} +{"timestamp": 7.7, "value": 21.0071087629903} +{"timestamp": 7.8, "value": 20.8567538060406} +{"timestamp": 7.9, "value": 20.99694690110517} +{"timestamp": 8.0, "value": 20.997366736077144} +{"timestamp": 8.1, "value": 20.908129373644407} +{"timestamp": 8.2, "value": 20.9639488178084} +{"timestamp": 8.3, "value": 20.874209593539945} +{"timestamp": 8.4, "value": 20.731643898574433} +{"timestamp": 8.5, "value": 20.787821193105092} +{"timestamp": 8.6, "value": 20.685454810659273} +{"timestamp": 8.7, "value": 20.636939436452177} +{"timestamp": 8.8, "value": 20.577302971956495} +{"timestamp": 8.9, "value": 20.563569623919577} +{"timestamp": 9.0, "value": 20.417275894188837} +{"timestamp": 9.1, "value": 20.31767408110017} +{"timestamp": 9.2, "value": 20.242340134908616} +{"timestamp": 9.3, "value": 20.03384980521583} +{"timestamp": 9.4, "value": 20.086781618984535} +{"timestamp": 9.5, "value": 19.87099453352337} +{"timestamp": 9.6, "value": 19.84762797233377} +{"timestamp": 9.7, "value": 19.671900348815626} +{"timestamp": 9.8, "value": 19.584696602517525} +{"timestamp": 9.9, "value": 19.522649712461103} +{"timestamp": 10.0, "value": 19.55076631224155} diff --git a/examples/reports/noisy-sensor-example.md b/examples/reports/noisy-sensor-example.md index a29c100..6995a4d 100644 --- a/examples/reports/noisy-sensor-example.md +++ b/examples/reports/noisy-sensor-example.md @@ -1,13 +1,18 @@ -# Datary report: noisy-sensor-example +# Datary report: noisy\-sensor\-example -- Datary version: `0.1.0` -- Started: `2026-07-29T13:43:20.211495+08:00` -- Ended: `2026-07-29T13:43:20.221953+08:00` +- Datary version: `0.2.0` +- Recorded with Datary: `0.2.0` +- Session format: `2` +- Started: `2026-07-29T19:37:50.019613+08:00` +- Ended: `2026-07-29T19:37:50.055197+08:00` - Input format: `jsonl` -- Records: 21 valid, 0 invalid +- Records: 101 valid, 0 invalid ## Reproduction +- Original command: `not supplied` +- Working directory: `` +- Command context: Run from the session parent directory or set DATARY\_WORKSPACE to that directory\. - compare: `datary compare noisy-sensor-example OTHER` - inspect: `datary inspect noisy-sensor-example` - replay: `datary replay noisy-sensor-example` @@ -24,33 +29,59 @@ ### timestamp -- Mean: 1.0 -- Median: 1.0 -- Range: 0.0 to 2.0 +- Mean: 4.999999999999993 +- Median: 5.0 +- Range: 0.0 to 10.0 ### value -- Mean: 20.700166470541177 -- Median: 20.80704917255715 -- Range: 20.064409237657774 to 21.03591726892986 +- Mean: 20.17767606050936 +- Median: 20.261510417832035 +- Range: 18.94492804097838 to 21.111120412614504 + +## Timing + +- backward\_timestamp\_count: `0` +- duplicate\_timestamp\_count: `0` +- duration: `10.0` +- effective\_sample\_rate: `9.999999999999988` +- end: `10.0` +- gap\_count: `0` +- jitter: `4.3390452126414056e-16` +- maximum\_interval: `0.10000000000000142` +- mean\_interval: `0.10000000000000013` +- median\_interval: `0.09999999999999998` +- minimum\_interval: `0.09999999999999964` +- start: `0.0` + +## Engineering metrics + +No control or network field roles were supplied. ## Quality findings -- **high-noise** (info): Variation is high relative to the mean. Field: `timestamp`; affected: all. +No heuristic findings. ## Input hashes -- `data.csv`: `3943c5fe0029450dc7adbd01ce800f1839760a4df9cc6a8c370bdb4c68a7c187` -- `metrics.json`: `ae371e6a73b6fec4e2c4c538df7845c9c5e731851d5560d009c607e65881563d` -- `quality.json`: `b86c70595fbbafb7b8180847ea53cf369fa14d69d3cc56e2e0c520e419e3e94f` -- `raw.log`: `8f453644d5e733e000bac97a81a7db5e66f158aa67f694499ec91509ac021760` -- `records.jsonl`: `8f453644d5e733e000bac97a81a7db5e66f158aa67f694499ec91509ac021760` +- `data.csv`: `593777e7fbce3416989aba7f20aa1865b5a9535d93456db0f5dd3f600cc7f518` +- `invalid.jsonl`: `e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855` +- `metrics.json`: `70263c17f02120c8f4e04285025aa8d5e43b71bd7034030df7fc808fc211cd65` +- `notes.md`: `a510c5cf6eb3f60d592443e63d03bcda952e157eea94c9cd76a4695426cb51e9` +- `quality.json`: `8aa973100a15bf817985395637131e838cf84e427be4792583def923f020f5fa` +- `raw.log`: `d22ef77cba7e4330d271280e85dc7d1d8a852b6a731b5443ca436d5fd2d65d8e` +- `records.jsonl`: `9ccd2b842825bf7788c36e7e546a04a3b582cae386903dd1171b08816fc3f31e` + +## Integrity verification + +All manifest-listed artefacts passed SHA-256 corruption checks. +These checks detect accidental changes; they do not prove cryptographic authenticity. ## Warnings and assumptions -- Statistics describe recorded data; they do not establish scientific validity. -- Quality checks are heuristics and require domain review. +- Statistics describe recorded data; they do not establish scientific validity\. +- Quality checks are heuristics and require domain review\. ## Plots -Plots are stored in the session `plots/` directory. +- [plot\-value\.png](../plots/plot-value.png) diff --git a/examples/sessions/noisy-sensor-example/data.csv b/examples/sessions/noisy-sensor-example/data.csv index 1cea061..165ab16 100644 --- a/examples/sessions/noisy-sensor-example/data.csv +++ b/examples/sessions/noisy-sensor-example/data.csv @@ -1,22 +1,102 @@ timestamp,value 0.0,20.064409237657774 0.1,20.172305697081818 -0.2,20.129687135430082 -0.3,20.318178009797514 -0.4,20.312114475078108 -0.5,20.514173526862105 -0.6,20.66597393908556 -0.7,20.607930682461994 -0.8,20.69170829016585 -0.9,20.80704917255715 -1.0,20.805057052242127 -1.1,20.892165721512733 -1.2,21.026501286564805 -1.3,20.97639091199051 -1.4,21.00162200416873 -1.5,20.97793268730595 -1.6,20.91649199231581 -1.7,21.03591726892986 -1.8,21.00755856169906 -1.9,20.897489961463332 -2.0,20.882838266993875 +0.2,20.201986121241973 +0.3,20.25729302411276 +0.4,20.334809681553445 +0.5,20.48099226444579 +0.6,20.513537314894492 +0.7,20.572376214982565 +0.8,20.72732168972371 +0.9,20.789995639860415 +1.0,20.868794399824807 +1.1,20.84550881287467 +1.2,20.932289350148555 +1.3,20.96032109739856 +1.4,20.910158279925422 +1.5,21.024394845537106 +1.6,21.015609158091447 +1.7,21.111120412614504 +1.8,20.983996089743695 +1.9,20.93906497227917 +2.0,20.970935285577127 +2.1,20.873148929058544 +2.2,20.85394795512725 +2.3,20.727427998865092 +2.4,20.686371771219182 +2.5,20.64968657802831 +2.6,20.550313722944114 +2.7,20.433803492665472 +2.8,20.280872748926104 +2.9,20.261510417832035 +3.0,20.14496318214583 +3.1,20.077604042276253 +3.2,19.952437503462424 +3.3,19.896663576413527 +3.4,19.74188088470197 +3.5,19.659314975314352 +3.6,19.590818378636428 +3.7,19.415819628482783 +3.8,19.368059096053578 +3.9,19.287232412368407 +4.0,19.342228290668235 +4.1,19.177079787652783 +4.2,19.16103523877679 +4.3,19.11480281635923 +4.4,19.034354254601137 +4.5,18.94492804097838 +4.6,19.05455119490034 +4.7,18.979716903952717 +4.8,19.039733224001406 +4.9,18.95228414612017 +5.0,19.01917657523837 +5.1,19.137026384902533 +5.2,19.188095543682472 +5.3,19.10260962671918 +5.4,19.160595138489565 +5.5,19.292246452549517 +5.6,19.405145428293615 +5.7,19.457339691371327 +5.8,19.550575555977595 +5.9,19.576681513921034 +6.0,19.749924087491763 +6.1,19.873680112784747 +6.2,19.895126971168306 +6.3,19.945139497301177 +6.4,20.078608163666306 +6.5,20.25320288837966 +6.6,20.224856503921558 +6.7,20.400256081226317 +6.8,20.444562936881333 +6.9,20.571883042289315 +7.0,20.644760519381386 +7.1,20.72976187928949 +7.2,20.86872823366432 +7.3,20.87147214280896 +7.4,20.96539373307989 +7.5,20.930929341059567 +7.6,20.943940341640136 +7.7,21.0071087629903 +7.8,20.8567538060406 +7.9,20.99694690110517 +8.0,20.997366736077144 +8.1,20.908129373644407 +8.2,20.9639488178084 +8.3,20.874209593539945 +8.4,20.731643898574433 +8.5,20.787821193105092 +8.6,20.685454810659273 +8.7,20.636939436452177 +8.8,20.577302971956495 +8.9,20.563569623919577 +9.0,20.417275894188837 +9.1,20.31767408110017 +9.2,20.242340134908616 +9.3,20.03384980521583 +9.4,20.086781618984535 +9.5,19.87099453352337 +9.6,19.84762797233377 +9.7,19.671900348815626 +9.8,19.584696602517525 +9.9,19.522649712461103 +10.0,19.55076631224155 diff --git a/examples/sessions/noisy-sensor-example/manifest.json b/examples/sessions/noisy-sensor-example/manifest.json index 1e200c9..422a21d 100644 --- a/examples/sessions/noisy-sensor-example/manifest.json +++ b/examples/sessions/noisy-sensor-example/manifest.json @@ -1,46 +1,55 @@ { + "command_context": "Run from the session parent directory or set DATARY_WORKSPACE to that directory.", "commands": { "compare": "datary compare noisy-sensor-example OTHER", "inspect": "datary inspect noisy-sensor-example", "replay": "datary replay noisy-sensor-example", "report": "datary report noisy-sensor-example" }, - "datary_version": "0.1.0", - "ended_at": "2026-07-29T13:43:20.221953+08:00", + "datary_version": "0.2.0", + "ended_at": "2026-07-29T19:37:50.055197+08:00", + "field_roles": {}, "fields": { "timestamp": "number", "value": "number" }, "hashes": { - "data.csv": "3943c5fe0029450dc7adbd01ce800f1839760a4df9cc6a8c370bdb4c68a7c187", - "metrics.json": "ae371e6a73b6fec4e2c4c538df7845c9c5e731851d5560d009c607e65881563d", - "quality.json": "b86c70595fbbafb7b8180847ea53cf369fa14d69d3cc56e2e0c520e419e3e94f", - "raw.log": "8f453644d5e733e000bac97a81a7db5e66f158aa67f694499ec91509ac021760", - "records.jsonl": "8f453644d5e733e000bac97a81a7db5e66f158aa67f694499ec91509ac021760" + "data.csv": "593777e7fbce3416989aba7f20aa1865b5a9535d93456db0f5dd3f600cc7f518", + "invalid.jsonl": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855", + "metrics.json": "70263c17f02120c8f4e04285025aa8d5e43b71bd7034030df7fc808fc211cd65", + "notes.md": "a510c5cf6eb3f60d592443e63d03bcda952e157eea94c9cd76a4695426cb51e9", + "quality.json": "8aa973100a15bf817985395637131e838cf84e427be4792583def923f020f5fa", + "raw.log": "d22ef77cba7e4330d271280e85dc7d1d8a852b6a731b5443ca436d5fd2d65d8e", + "records.jsonl": "9ccd2b842825bf7788c36e7e546a04a3b582cae386903dd1171b08816fc3f31e" }, "input_format": "jsonl", + "integrity_scope": "SHA-256 corruption detection for manifest and listed artifacts; not cryptographic authenticity", "interrupted": false, "invalid_record_count": 0, "original_command": null, "parameters": {}, + "parser_policy": "conservative-scalars-v1", "parser_warnings": [], - "record_count": 21, + "record_count": 101, "sampling": { "backward_timestamp_count": 0, "duplicate_timestamp_count": 0, - "effective_sample_rate": 10.0, + "duration": 10.0, + "effective_sample_rate": 9.999999999999988, + "end": 10.0, "gap_count": 0, - "jitter": 8.073008690572807e-17, - "maximum_interval": 0.10000000000000009, - "mean_interval": 0.1, - "median_interval": 0.09999999999999999, - "minimum_interval": 0.09999999999999987 + "jitter": 4.3390452126414056e-16, + "maximum_interval": 0.10000000000000142, + "mean_interval": 0.10000000000000013, + "median_interval": 0.09999999999999998, + "minimum_interval": 0.09999999999999964, + "start": 0.0 }, - "session_format_version": "1", + "session_format_version": "2", "session_name": "noisy-sensor-example", - "started_at": "2026-07-29T13:43:20.211495+08:00", + "started_at": "2026-07-29T19:37:50.019613+08:00", "time_field": "timestamp", "units": {}, - "valid_record_count": 21, + "valid_record_count": 101, "working_directory": "" } diff --git a/examples/sessions/noisy-sensor-example/manifest.sha256 b/examples/sessions/noisy-sensor-example/manifest.sha256 new file mode 100644 index 0000000..e006464 --- /dev/null +++ b/examples/sessions/noisy-sensor-example/manifest.sha256 @@ -0,0 +1 @@ +85fc158e4a26e0e7392ea49f502f410e4de8ea4323689c5d31520483ab8dda86 diff --git a/examples/sessions/noisy-sensor-example/metrics.json b/examples/sessions/noisy-sensor-example/metrics.json index ef3897f..6eea5cf 100644 --- a/examples/sessions/noisy-sensor-example/metrics.json +++ b/examples/sessions/noisy-sensor-example/metrics.json @@ -1,61 +1,148 @@ { "numeric": { "timestamp": { - "count": 21, - "maximum": 2.0, - "mean": 1.0, + "count": 101, + "maximum": 10.0, + "mean": 4.999999999999993, "mean_absolute_difference": 0.1, - "median": 1.0, + "median": 5.0, "minimum": 0.0, "missing_count": 0, "percentiles": { - "25": 0.5, - "5": 0.1, - "50": 1.0, - "75": 1.5, - "95": 1.9, - "99": 1.98 + "25": 2.5, + "5": 0.5, + "50": 5.0, + "75": 7.5, + "95": 9.5, + "99": 9.9 }, - "rate_of_change": 2.0, - "root_mean_square": 1.1690451944500122, - "standard_deviation": 0.6204836822995429, - "sum": 21.0, - "valid_count": 21, - "variance": 0.385 + "rate_of_change": 10.0, + "root_mean_square": 5.787918451395113, + "sparkline_values": [ + 0.0, + 0.3, + 0.5, + 0.8, + 1.0, + 1.3, + 1.5, + 1.8, + 2.1, + 2.3, + 2.6, + 2.8, + 3.1, + 3.3, + 3.6, + 3.8, + 4.1, + 4.4, + 4.6, + 4.9, + 5.1, + 5.4, + 5.6, + 5.9, + 6.2, + 6.4, + 6.7, + 6.9, + 7.2, + 7.4, + 7.7, + 7.9, + 8.2, + 8.5, + 8.7, + 9.0, + 9.2, + 9.5, + 9.7, + 10.0 + ], + "standard_deviation": 2.930017064796725, + "sum": 505.0, + "valid_count": 101, + "variance": 8.585000000000017 }, "value": { - "count": 21, - "maximum": 21.03591726892986, - "mean": 20.700166470541177, - "mean_absolute_difference": 0.04857540156360436, - "median": 20.80704917255715, - "minimum": 20.064409237657774, + "count": 101, + "maximum": 21.111120412614504, + "mean": 20.17767606050936, + "mean_absolute_difference": 0.08358103575388814, + "median": 20.261510417832035, + "minimum": 18.94492804097838, "missing_count": 0, "percentiles": { - "25": 20.514173526862105, - "5": 20.129687135430082, - "50": 20.80704917255715, - "75": 20.97639091199051, - "95": 21.026501286564805, - "99": 21.034034072456848 + "25": 19.584696602517525, + "5": 19.039733224001406, + "50": 20.261510417832035, + "75": 20.789995639860415, + "95": 20.99694690110517, + "99": 21.024394845537106 }, - "rate_of_change": 0.9715080312720872, - "root_mean_square": 20.70256514623275, - "standard_deviation": 0.32291997799527944, - "sum": 434.70349588136474, - "valid_count": 21, - "variance": 0.10427731218847178 + "rate_of_change": -0.5136429254162245, + "root_mean_square": 20.188594094511735, + "sparkline_values": [ + 20.064409237657774, + 20.25729302411276, + 20.48099226444579, + 20.72732168972371, + 20.868794399824807, + 20.96032109739856, + 21.024394845537106, + 20.983996089743695, + 20.873148929058544, + 20.727427998865092, + 20.550313722944114, + 20.280872748926104, + 20.077604042276253, + 19.896663576413527, + 19.590818378636428, + 19.368059096053578, + 19.177079787652783, + 19.034354254601137, + 19.05455119490034, + 18.95228414612017, + 19.137026384902533, + 19.160595138489565, + 19.405145428293615, + 19.576681513921034, + 19.895126971168306, + 20.078608163666306, + 20.400256081226317, + 20.571883042289315, + 20.86872823366432, + 20.96539373307989, + 21.0071087629903, + 20.99694690110517, + 20.9639488178084, + 20.787821193105092, + 20.636939436452177, + 20.417275894188837, + 20.242340134908616, + 19.87099453352337, + 19.671900348815626, + 19.55076631224155 + ], + "standard_deviation": 0.6671787715465946, + "sum": 2037.945282111446, + "valid_count": 101, + "variance": 0.445127513202423 } }, "timing": { "backward_timestamp_count": 0, "duplicate_timestamp_count": 0, - "effective_sample_rate": 10.0, + "duration": 10.0, + "effective_sample_rate": 9.999999999999988, + "end": 10.0, "gap_count": 0, - "jitter": 8.073008690572807e-17, - "maximum_interval": 0.10000000000000009, - "mean_interval": 0.1, - "median_interval": 0.09999999999999999, - "minimum_interval": 0.09999999999999987 + "jitter": 4.3390452126414056e-16, + "maximum_interval": 0.10000000000000142, + "mean_interval": 0.10000000000000013, + "median_interval": 0.09999999999999998, + "minimum_interval": 0.09999999999999964, + "start": 0.0 } } diff --git a/examples/sessions/noisy-sensor-example/plots/plot-value.png b/examples/sessions/noisy-sensor-example/plots/plot-value.png new file mode 100644 index 0000000000000000000000000000000000000000..a3455b04b4e0bf487d60b25f10977e3556ddc206 GIT binary patch literal 31201 zcmbTe1yEH{|1Jz7p!5L&=>{dGrRxxq0@5WdCEX=bhwd&Vq+2?r8|en=?(VyI-~YWc z-^_PszWMfy!_4-az1RA!^{eN3)(QG3FNuyqhyn)(hyFndsssmzKn@27-}V9t+`%_2 z?g4*z9mLfgl&wu1ob~OD;pFrkY%HxEEX@qaos8}5&8)w&GqbWXvoex@ad5D)=Vf8B z`ahpwwze~6v71G>0#A8qBc)*v2ZyQu^bbByDAx=Q&Qiz<5T!Qb$8$u)w*qMY z8C6V7YHDa}Ypc^^`UhlkgAK6;`_!~F+!X)b;o-oPlsRK@n31LnKUT!cmD}gX_P(L_ z+L?KIQo3tX=@vB!;48DnYLr(W+2Hc?^P@_K;rzg7m#APFL>;Fu*y8Mi`sHl@(YXF! z4Is;l{GVycYFSw^6%-cg%kz*=OibkE<^9#bfW^h(GBPp!WcaVWdfvDz8W|b=`0)d7 zGDuf z!DeJ^+5f-%T?-tB;#gQ;8Y1Yq25(Z{C3AB03auIU=vNoOwn>L=;NM ziQ;*C;g4PWJR}d7^(|uU$wIL@S;C+Dr`?y}JzA`!e|xd-@_2ulGXL03=KU{>d+}RD zghI@%jEv0jc5X(d>2SJ~oSYRW9K>hv8VYwZD#P*57Vr*FBd!FG+qWWGue#KVAu8;kr(=ziFwhV-1>WFT_zRp&)8E(0N%vXT;|sd#jB^haG?=9oc4 z0|Q=;t_<%7g7oxsI|=dU=Cdp#pO60j6m!EAezGT`j1x82wbx*|!vOrR)2nbPGtY)Ls*Y`7(rVvunaIiKnl9G~=V;tj9uCK32W_P!@x67)k;z-8D@m-@L zD4#{0UA4pTU*$_?0nO~nMbv8aF&#+6%0|-e; zY*B#}@uJ(eg-s;wbaKhmrvgydRW}|#fBvk9j+bJb z(yX&Hj_Dpq=0t69+Hcz!Oi7vdUh9F>4ah9LM-X-J=}Xr1sO0I_aC>9~(P`rSU#p`O3LQIfbY=Go7-1YrQe+dm$MaWCn(YQ7I`5zT}>UaNy-a z;lKK6)CGsQYI*w<7M&eqOVNZUJ52r)G%_(fxpZp6>DwOR;Aut_mH48LCqirmd_)dmNnfK1dQ z;kn78!Ioejmz+$<6O-Bh?`M_g{D{3}Is_-p)LJwhP0+IVQ>~ee&FcN4d^HTiF=5Wa zPy;;#*m=v26^?>;IDrp)ZGQmX$Hx9I$BWVA(K2G0Pwh|fDYGABpF9zzuWujtOY-YH zyJ3GrF8bG`^k7LtVw)(Dx_OVmkf&(!>l^$FMc+~(7Dljr3lFWI*Vnz-CIiKj6C+f-Sl4i_!@lrg1oMi`(C@-p(uSO z!ETkPWMlPAVwR`$^%m=!{?YaW@NcMXTlA*2FsJzWe5 z=6RV_F2@E}x3~QR?juI)Ag1kY+4P1AoXRlLHddlXu@rT4qB{Sp(=$9C^D#;tV(hB$ z-2AAi5>m1l{(54fvU9VRzE5qDRH{$XMA4sAQUOeDP$hwSb*_fw0HMNWS(N|kpD){c zten<#&kK#2H7E!7%a<>sA~c#kctG$dGoPS8+ncGB8)hKAJ!;47|1AFeJ#fG(OC2&| z;!yqe07r*uBa;)E!lT;}jyfG3EtZDlkWy>glHOJ5>YqvJ+3UJ#ec`gg0R#Y?5kWW_n6?3|oG{rxgtrBESMWyW?bJ-vv60(v7;(|%9Ji{d|scG_Qdc6Q^T zfWSb>K&QRMCimkmB5RI1@4e~rtK~p4Uj103ud2k-aW$1lNR&Mbx?Xbot1O3w#f6d{ zPA~b1tv*{AyY_C@)AwPP772#m2f0>={M>>7`%v4VSzjBl$eYIfs4l^%L0e_4V-8|C zU3soCfCT%xypdT_BF_c)Dh$qXVr_PGR4F$MgYPM)wIg8ie`D6E|2&%gA>p3Cv$OLJ z3rm8s`6kYisj+d@m3v%V+{x}8bJ(m_e>q$ z#mu@Gb`4D4bV8;?XQ+06qSi!bVvpi~Cg%iL{aN*Kk;Ph~lbyOHQt1CDHtoWXY9DYn z_P3hTv_ksmA7-H9M^z@4;GolhS+^hZ=|bb}gSuhXhs57MV-A|)>UdNzHV=sF=%k@x z{Oe|A6RN=RzCfrwJ{5S6MeE6u&EcHB+Y&AsH|EA)7W<9Tzo{ntK7S_5Q9{6QxeD7ZL`Kc<`xun+$mktDNX zL%@FZ>Nz(z_j^7*v6r(yK7{(f!?15#NO$FlO@1>iF1cX@Dg}Qu z(GL;kzU2xJ>_#|FByUzoTp2A?i|^{y2+3FUn$Cy~9LSVTefwO((dyn_G8K9t3yYd# z(HTP>`FZeB11=s9?_)lr4Sm*&OHc3Tg^xpjl$9xs73kTl9g%F9DNH`|p53Og!ss$Q zutD;#LlZnqL8y!<8xmDUu!mXt?y5Vwx+J`Gr0#5E`LFRG zNpreCmt4@UQ5OGrCSPqXySG1Q=Ix86APEhu*Y`FMJh-|ck6&U(JHW6i#Qr~9qd$`= zcO5Mt(3kUpj|#iAm?t~~LV5Otp>-93#Lf~uU(^}jDoT#S`!b{)S z-zRH+2O)Pksc=qJE>>$_@aE!iG}ia zSNBG~hK!gO55Rkax=W>b1`$}QjqlY zmbzFJ!_{Qg{J$LHvlC~w1WtSHm3I@8QDK{PR4^3Vu6Ffew`6!;5Zs*a22K>J^_PY- zA5xBKXsW1;xjp;js=_qcVa^*rOpGBPxHAp?($opK&`!86T>xo1=$D1lhbK>lsI4grI85L%SDr_A;~ zt?=467`Acdr=(~ea#vSZw##oaWxRp4nP+<}s5?^QVzTgkCwJzv z^zNyGP8pz-*a2_a?NOM&4{x9<{Jbd%U`D3kwSgpSWg1yv4K zXO9SSQk*_W235@B++-r0*Ja*2eoqwz4bF0=#p}~og3}I~y09Djhc6mKIu>1$e8z}f z+$Q4sP7bFNWhc^w@3!KE6xA^I+}w!oboMnpBBs4co}DI;#q~Ej@pEE0O%4w4gbZg| z;2j8iIK7o{Jdefi>-jxS&{}6KIb5C;D=Z<#XzS)C9;$s1tNAhfcN@z5_4xx{f}okZ zxAvUSN7FxnSh!o?SvD}DB244QVB&f(Z}tKBN9>n>F1sHF_|>?h<(a=jGvV@2;Tt4`}P zJLaiQ9f2BZb5cM1ImhmeX9bT%p>cl#%kCY*iRq$#g6>*E ze=grr*3FqHet)WQdy${x^`bmi#o~>a5OMt2(q=znL2LL}a#aoc+~0-9tN=gv&&PuZ zKU6h7PegGv9%&ITYI!Dj7k#o}&avL~Es zN48%2tHxZaSx@&8m8c{%NWx!eflt;+4?iHFzNdh;GmbN%Y9kQ_<2^8vC{Lxp!dY_; zebGKH_;7W-xLi>SOBBI!rI@9*g9z{OruZg!MYQKVT6L)$T6YJ98r05cEw_wOrFcFx z`!L&-SBff%qlf=`#g!sN$y}mC|G4UrDIK4iRi%x1Bq^&IlX0J>O3mW+QbBfyIl=2? zrarU1NY7tTm;2%-sKR;=t-3L9IVwqO&@nsfB_(=ikbWxsU|4K!&X|8NkbLGf?Obxa zGal@lyspNMyybPV>M?eU*pu(_XK@z$aA_KoWf*3uaeWu;d~`B zIDMd-$W2cy;E}do?b4Hr{-{bR_6N=nE&2sz`ZbNJ?mDEBw0{G=Gi2(lIS=|X@xsf& zfV=MAyM;7HE$=~|rLA|sKwYcg9jCN%{l|=sPY-06VI6uAIur=e8?e@%5ecT!3M-+p zlbg?!lsg_8ow#=S(KeeszZN>1QZ0|}7sPDtTZLUunETd(Q?Xd*Y!_OEJ={RS=lUSc zgUZ6}SGAGB9~puW8HPSl+I4EZ;XZ3YiU|@mZ4TUZj*5-N7j=g3NLCMl>dkYc1{w)3 zL`W=sn+qfxbVH=nKud~bwyc+Y z7nBn4yoS)@OM_3ds}30`m!DdqxOJ{q6hSq*o)jI=hxmFNfCSdXwA~>1&z} zUPI{lazju;Sc*p1Jo$O@`H%u?_0Ftoon`Us?ibRjUse6oXO0ql+W3{+g7~-Cr)xx~ zjRC=ARvP%X+rbU+t&lQ7Y&VJN4_8;TWI%518d)%1Jar6z4CUZ!*iQ@E?<=iOJC1Cj-CJU)oexYpFA6>9`^ zzBCOH@^}%vGu!VampNo`W%|WP6(gfR;hM~VX;&bN(>if~q<@K^MTi62KHy7%EJ3{@ zB>lyuUF4tcU|+^w8ECdlCQnh%woQe~YQPqG)85|JM`sN-!OA=(JSj*WIdb&L+Hlu- z?HwHTbUUdQY0EPUQql?P!z`Yuv(p5I=XTQc<=@wrdf}eTe4^OC*IjjwK-f>2%w}sW z9gC0Gyf_%$2>PzGYp21zZaH6HV^%CrEE&H*u2+&XZfsidYq&fVz?HGKCd-F;%3t-* zT#R@=@NBu`>CwE7j8#|- ztM)%tH8_~PClb2s8gna&`VXCqt|!)OZEdi$_lqNAwF0WTN6&9&q4)qsMq15cN}j7!I0Vi0w}5Q>a%k zh{z8rJbuY7df=?9Db>ZM4uZ*chL7>6dyx|$u1~x%zoNIi!k)lIcdS;CHP32G);pc5YS^;~;{&cZNXh?MFS zEdUBWL7z~s?fKgj`K*dx@?3KCr6ry2L<+07O`Asf^Sek$6r3%@h2hz~-GcAA+AJR_ zFGIH}ck8{I_+H95(^2x25)YRKw>NmFw=uPg+D_$_anTTALz1!#&HO7yV|}Y=bQ^!w z?7e1g*(+Xi&z){gO(#lALZ%eF^?T_y^!PsGg_mmguYS0-a)t&5E(|uXNu+XDDrMFj zYL6Cl#3EHb1x-znB+yMM6vH~B(}IrdDxz?CIwpQLi`=xZ8B2jS_-L!+Ypt~`{++`+ z4OL6Xf}Iek?`XHZ1)U>D#i%LSq%vVa3oIpvakqy^FTE3K>vlh>c=5h_PWM7Q8XB^zu z@GG=!qAmJ_yrAYckA%cU)9O7QtjK;2w6rHo(VcGEI0Dtdm#0zW5gG?vp_E`tY7@d(1 zhp)eV%q_?-ZcicNs3W5sNySZZC_`fq#COe6NxK?=Lc1)+Y^}~2tX+*WW=)`ebVW=m z6^suvd)NnDqYOyoH=`6kH zF2V^SM-b>|QJ_%4z`IQZrpbYet~2D1+$GVIyg6Q1ixvXgBFjt=|7wPtYB>Qjd3))E zBT(-rhb~7PbmkBL!ZBH)rjsfw?-_iFQzE-xSeQc{n#Gg`GkD1;a|_|vAgXFSe~wwH ztpJBgI|lW7R#{_xXLpds<*v}8Kf8u@fq0tlqgK8$u>wn5ptqTcP48C7K073O06B(%dxGH`%?;lZFH>{qm3m-#ghEFW!ck?$kcUuJ(QmL(q;dE@fY* z((Swf^VgrKa&)=N&B`k&PT-s_-QFE;DiBk?`C=jhO#rNweO#(q03bCgBF1qS)V(h4 z!mNAJt_zcl)s0m+UH&zQ<>i&cES2)bhwm9&WM+l~sF|eNrC*~X2lvm!A_}UBLpalxuApBN#>)MTm z;u!PcF0-*gYG+E*_8p=SiL6iGv#Wo4_r_+P%OKD(qkMwRhOm=6%oLb&$dsDz?ARK! zFgGbt5~THmp%LS%d@{85NS{w{y^Fwr^xkGQxN>d>d(jrY9=>;z{VK%vlPDm-{j)il zQ0=q^Mb!^tEroJ(JV!JHta?i_D?wl2q5n**{nbjtRz?{F>gy>mjW&~rU&WTcrGO+Q zSmjo%cb3|)Ap5!Uz7ZhK;x5rAh~Gk1(p3pTZsrR;+0hJz9Yq?SBPwcd-1zv|-OdSY zz9L~E5JDn%&w|;3n7L6mrdrn0o#)B4voRjSH~KsceBjHis?D{^W^HBzJ`~5* zR`GbLA7jS%HAfvs21V(2fHYs0lsJmWKAhdSUjf$_RQep_bfYO5bwreh;v z1B!B1I)GlL-e%Vi5-iiMWH6b z?&~_K?+Rz`8^^TAhRH4_<6qA$pgxi=RRJ%__EwDN&do;IoyJOF#sMQsC`J$RR)l8` zLqC?{Ii+)oQ(o`Cpj*wyWn~PLg3>YxHau@3k&Bsg4n=z!8%;+$d|*ZfjAcWDyZ81> ziTPxhFJ~qB(!qz()fc_7PeTMwB)J%F_GU)1FPV5d@Y3Wm#&K4>$tfvCx&N&#C7T>7 zcZN^mUfSO=q9}M~Y0U1m={%;N<}_D-Z>Ii{6J+nn^t9&>7&L{3$)bV*vX^&At`WVU zYT#<&iaoP-Yk~8)Duq93|Gd8ZX+Up?+z%$J0ALkWl92(g-_=l&o46$p)Q7^)-hxBY}zBmqC?NS zpdIRrdESpmu*~>d0U6B9qF}XOu%23`+03N-*9(KF_dE#JSkMw!HTB`>!iS;`sr@!E zo9xVbyr+u)V(TlKFxt%Y!Qz$OR}R46Oxv*$aeLJxgT7806NK0b(2XwVBD}S`T|S1) zhaj!p_=uD#78)U+*KnnNa(izC1$#SX5A>zzfgWjABc1(G?t`sCl<-RnoUguKsKP$H z?HCYmu%Qi!lZN$NF)rGv#(R@NmVJCE4nP7MTHW{@2K+aAE>rS#Lzc1@E8Ncurk0(Z zm#8By-_g>yu=>J|Z-Y>%TrxyB^{xi$_hFb$mQo3Bqvq zUmIsMjMcrspUFZ$8A|ra$WI-h@Ub`q;?kHwXv84aWbds=6Dv$LCWU@}RzGrV^ZO5` zGiK-7VslH^2vKzR8uyW!h(m%6$?^E;H~VwoZl<3 zc9~%rk}xp{gQ2lrilzKzG!aU4|3VB3{o@HM-{s}cVSxUdc~ui2XPZv|m>Cs=q@i&3 zWWNeUou=9v+17=uHmrZ_^zJ&FF#F=_tFSt-rxe*2h-hy%Q=#9 z_;xCY@w$e(9@HV`6G=tbMtCZH+|eRGzs_sX3R~s3rT$FbP+Z!R2Rj7{l_QoKi0Nd5 z{^{Jzy;#``gb-rw1vr~t#>{0LO0`~$6@`k+f%R9lvdHtIDG?3gf5wJhiHz9$bz4#y z#g@8GczYQO^1;#Z+-3W+g=i_45<$DxG<1AI(W|y@**+&8n>B z0>u8qWL@JTGtayA3+{&ZsG`sS{m!4S08rc3D-qURvR86uljr+j@5quA%dD!z-@q0> zfu9cnlTkF5W4h?r?A?!o3_QN$1Z`-~oRGn2l(JI(nGMJQK_|tJTHZ=ug*XJC>B<+? zf7#nBVmf};2j|tG`K$+>+#aVC%2Y&I|PgkZl2d?>O~eZ z+Se8RoJE5~DGAPcNL-|$oR@W5wR6(6_g;x3)_;D0%HiMQtCFasD*G|Jc@zT_koz&~`Ds5lI)CA8(yd z-8E8xz*9kj{&I^-!)q!#=6NTYnuH?a!yTV8FltniX4yt^-mmc>mtBnTNtW(BrM8TI^mQ#!H#<7l-B#_rgosq zB;S(vOY-;m{o|$WruB=bhUX2#yJSxA_M$XXJ4HQAhf`a>(h#Q0=$8#-Lo5S;Su)z4ogq!D*6&w=+VYN?f;l}W*a zeXF6RVBD*v+QR1P#q0zmAd+7zSt%gfV-d*Fwv<_~nVnQ!jK0$o2 zjEYnap2;})1NiQv`OMx?L5{S}8A*6R^UuiG+}xYjzV=QZMbU$&i?=LCvD zO;!rik_Dld5r-(;$QIy^+dl(qf9pfS?;z^5&8RZ86r=#*#YUr>Q&x zZOH)P>%21?sfc5togm-NHtA=WO$2y4?1h5%$b!s47*j4sQ%?^E=x^Q@{~#W| z!=|b^Y{^6o2(aIdtFtXr>?en+yQ;?^04fL7v{J_Cq)yY=C#Z=-{2k1%e1X&RGgrpn zSM3yIAg~#|Y`hjCjZ#kcM9@(W>Ifa&Dt#$DNzyN@p%OOQPCZ^p>S7#ynDHoCa{QUf6SE&H-R{E0sCE3~8y=rD=W^|#k%xAsTmD4Ga z$nF+0jn0PNfLdneR73c#16k3Js3_2MlAxnZ+w|5Z>0DPEk9{G9{Jf!epNWnK)u4da z<5!*Mt&?mL8zPW`+1S{$_C!&YTTGRNkkGMw5*%_fiA>y z_ILg<>tWL+DUel#C$XEhb#)<~ZVbl!=>#LS>D??XF18vQWj4=HPU=iN9+B5LmU5uh zv0;)qhz>|Z+{Is`xAz4FRM&ziG|k+q{iLvv^1jwIU*jn|t{YJ$ARH?6Z0fR!q%}_q zv9l|0Si87j~PycYS>9v5HLrixV2_K&O#h%%D9R>;Sb9Z-lg85Jab|j!U+ud7E z&QXz6V)a~>?=Veho0`GPSY(5J_%_~A1;EM{QhiQF${qeqyMC+`%{MWjbmu1L4Cd+5 z#uT@{FM!SEc7{CIAeQ_`tq5DAIjFQimUPI#%q-#R%G21$fgBxwCCE^?zLxp#-{&Ir z@@F`7^0C9mfO7skUuz2s3KDT~aj9%k*3fv{@;f*vu}uU3ALzV%_dfgGw?> ziv@g2o#{B`y<3DDSODaLX5g{W8W|oqsZFx&`}lCoZJ>l=%bWE=r;#L%G`g29-thZb zcWguuXL+*51cg3n#c9>s2O_?dw`7>bb(tx2PtElpUFFfa|HNYpZ#@;d8K24exKxuYe}*2q>JmZuv@^E(mqq zwsS%bhKGkG<-*u##5-G*pMP8tR?LqrWR6qO_RsYe3QPQreSN<#c4vDC6ITBeN68W~ z(K)d1Qv2#y$X-1U)u{vj^GteWt+f$AjMy}~>|`S!5>HrRdrzO z+xz-2Wl%@hf4%C)9tFU2)>$N)L>fFX&dNDDC7kQsarf~c_tL*%p+{(jpeNy>vz=YE zOdfOUKGxFhb(QCRu4b_W6#9H+bydqFG?B*jY(z3;;2Th#{uvxZ9~v6^_?_!7w&BMe z@)@Fx5M*zd*jKZ-J%{_DbI*6pbh*ug=Pq9giI4&fB<~MN<=;w=Ve% zu@nTasRbYJ>HfJU3Jna7Kw#T^+8(bWg3i57`{w@qJUFr~2i=xL=c|aoYSngZ_?}5+o?1tj z#L;%vH+U#yIR8W9kK+mm0unzcUBI0yegGG~^BTTyw#uUX%LpYk9o_1Bf}Vu7Hu1)A z#?nL28)BmnHuN}8LWmGTc5IZ*B-PM|Lg=bkOcS(@t!Me3N^D=H4!HU7%}AC`XVhg=F* z@rN;|h8>dI@*o>uon_z_?i->^5-t+jn zA7n51OgBwTJvyL6*B@jZRFS33tX4JfIJy-J>o2`izYm`6otZ>I5SvfdR;O9*hUh{Ta?6 z(u*%fRw`2aLU9rt-Cm zKU?divSOdvykA5}?W24%!ojxtk_KD#_Ge5bS-yNV0_EYlSM4|~ZQr}6!)4P`n{rQ#W$ zqUq}eIeTo1_)oAHG|J4=piBR$sq|K7_u<|E-@}1}z?V%PwJ~|h3C~J-Q-fDfRE;>C zz+y!@T^NEQZ(6hbmd7z11n9ErPrZw5rj;5fQLtpq#&br&89bI6y#_9nDUQ^4*p5iH z@?)L-@Hr|~SN@*)p`-{{ILs~mcbGe}nF+EEi6$i63Hp!`|8=LO`_tHYudmVy^Y7m$ z^1*mL2`1;4f7;#dE(E0paod4+hbG4JMmXm8YO^u3=m@d1x-{}2s|uE|5|4>C%t*-$l3G^-pOMg1q`DkvF5&YE^4ph<21l1DQsF5` zLXMn3A&EJYApJ?3YPAEd!!NbxjRa^^_KFx;6Um-sQ_aHp1#Lm8iDejK(N{o43>2U} zX`;N&yYJ-KZk|Mzii7K|T`({+)fR)NAx5FY_W-f>TKp$FQ=Fu_d-VQNB>(z;X>n*h zS92iuGwuCIJ=HMK+SRp(fi4!OZCx!J6Y!8*+nbCv&Z$o&f_?7bLxw1y*$+N_!DjK(23loW; zvp?-4NTB`c-M4n4v&|(UKq99#aBL@(;v%E=eEn(%iKq%iqciS(oJL!kywS!r8N&C4 zwTjo2p2>5Oq0B=*c8M`xK3#p)e+JM@14m0<=t0I7wsf`yfU#d=A5hc)ac+;tzVRcb z(np~3xK>fo3M9LvJGIsapibg9u;C+X*l!Sj;ODk}SaK!LO+pqmDrw5SpyQ(cQF+Gq z*66F4VVm_O_0M0`;oitpf8aWLf(nN-v7>;QBINgeo!Rt+h93OtMF3g1pLY*x@&>S4 zN#b#36~BcaKdXUG%4BEfjgZTGb*iSaXABrQ0HrmjLL!&Y^@Q@>+V+=W(PqTxI-WcE zfrv|Qu|rt;9XslJdg|y=kiJ!%xu06;`}>Q7hRl$8FZcGhv8t0tnEKCOe|ufN6#k2X zx%h=VmpT7CE>RF{lOX8b5BUL|Ad+pE=!upzhJk>plvw+MXzwcvYL-VmlkWENM$^Y7sE6P6U3pl|hrDt&}W4PVfi-*}?|Sr1C3qSne2*i!vM7-WN3 zQf!@P2v}6-`|0*z(_L3*(hu*#$7Gd$Gm=+rERRgvChViXAt109My+w zd+xx-p`dU;p@jomd83t>@udtbOwuKHny9#ATIdPg#fc(-gvvf?`~xcN>66 zI_j+71$Y@&3axqET&tZ+D@MIvDB04CTf5mg`n8Y(*&cn-WFi{`HZZ0*=(Njj-lyOY z0#Zd{P2HTrYd(XqCZnl7Q2S8I=uHu$N+J(A)-EZ5KTVBze+~hsl#U&Dqrz@+3>N|)Wif#(B8M8ctS`_Cb! zh_)179I3jaA6ndt(C&+)u82#t_Idr}!jCPaAC%odk_->gsT5l9JwCMTG@V;fbpM0) z%jp`|RH)};-L%&7(rSI*zVidWKQ{dGy*9ifA+5+zpjDUsUe5KvYNJGc(?x=8v>7?u? z1>M&SceHwhtGnNg`gFiUi9i>xWrh$*SF#uG7!lf4W`j91zwf`aVl&uA0#H)lgqW}a zcnUTnHHXXZeb0Ir3ydnY5!&H8zTw8IECbtS)mQ6oc+PY2f7+&iRLz}vCBqp_&1h># zB*W0AfN$3DFOsEw$E@#c4!MX?wn#P$)N$VDNC&69mVCt zvC_ssu_ac1v8?#|Rc4ZZOSJL7UMXO`KCS&x5jHdclaJHQt(0#=};lTob z1U@vK^2idXWV1PiY()7(YO#rxS87`y;+U8>&s80mF#iw-W`*>YBKK1~BQE=)=QG#k z5Lq!Gx=(_o>u3iJP;O8V0vR7tW?2y-=7Y`a2xcLAq7B$dK&dWKZzCof!K}GDP$l#c z(^0Z9&hDt)Khx2-Fn+>3dHq2ZUw{3+*J{J(H3oRZ=cyTDSubxQ)$Z(epi=Lqb!xwK z#*CUE0`d*qg(grZO1SAGUQn;m18}dtRo5!;)Z|167TQD?*D9?Z;0q7?4o#Sunk4N` z9lq>~VN={U2PpmV->Z_a2}zi&e{CjjP6SZhoxdy8hY;+bNpT=xv9T(m%b*x4qr{S; zaQ@RM5gkSCw*^YaDW zr+w?1*FNYuPC>HPNZC>wbI~WSzr9)W-j$~W83Pa&S63T86~lqP=n2|UW;k(G)zUHJsn>Absz5d5P|!NN|8PWKwJ3u<^%5%~w|;M)oZg$LwWT=V|@$t$5}IsuLyvQYL8PO;TmQb|!mIgqqjeruRU z&FI%e8IYKY>H;9Uog(w}85Zr2n0x~<(e^g*2EeKH&k-RZmJcu|rIiTE1Cn1tPDenM zd*C^ClwF7aw=oSGyA*bEqZ9O(%F~bqtaorDm4^BLY?opV{*$^dY1yWX&JHi0GPXR3 z1xcJ%50Yxul7LTq0$|f8~|id>>r>J>uijjtZ*Y)K!l_SM3KVWd}xO&KZg@& z+1^iKqP6||=nPJXtbl|7)*woQKEEic|2rZGg892yR#J(50e;ekaG8?*7xyt9%zaVV zN$kho|0XP`^o`AYgdT@FZfem0fNtc3L9?D!U81Y|`qa4!G{#U~nw$AAMA*?4ufJ`5 z5S4!m6!pE#l~d?w<5T_E1Rml=u37Et)rp7@HNzk&pOXeq<85#tL@TvI9sI#?V*o_j zGdHr(=n@yS#g7@b8w}*I+4!pe($I5jtzoZv9zQ=~vLOTI_F9pf*Si{bTNJd{~ZWB)#b)DkyQ*1XL4B5EjBZNk7>v}B$XyQ}$=kzoBbZ`-Vr|Mv>> z>@{wy?+q@y8buNPmUrnqLG|E)U~!u~T7KP3?1?Zo$vE_KQ^z5GC4vh5qq=(nVpD)`7~!g$OaJ~v`8J?I zd)CIsC-ui(%E7VabF+j;U2pinXjFmZBeCWT4yC_v9s3`C>9&6loayx`aB*4a6aC4?;?Y^KI%uZ? zaf5OJm69 zGU-T>*V`Mb)Hov{^yWu%BOG}uc%-z2JBts=8j*(D0XaAxn-rA3x&h~s zXrOm>G_j2pdYpVtu6AKysLiI3NjG=hiT}W!vnjrwM@5ui#)?1k0|D#!my^E0`o!EM2P2ej2KZElu*YgRdd{868u8X5+x(~gs>y}93;~%&>jGw%kALPB>dufy1 z<(E3;XG4EnPU{5+RlkqXW|9rR0mpvQo>6gF{r^PeLx$aXTEw8g#6E=+O;!FR#Cx)o z`L&K}3#C1spdmQ=V9tt3-!n)GT9{80ZS%CG{vzr+|^xifa}{ z#l;c1tY*ir5-%6sF~@Qhy1{8(eM3WE*V7FOUS2{nk0X)){Wjy^YC=4qS#nedN(Xrp zJdm|V5)H+nOx#`jIdV{;Qhf@7FW7D-Ccg$|j*k8o72~*4xL+Q0ZjEH6@P7rT=cX%7 z6)J~q|1)$`OK0mkGzOZM#&Z2v5S?^%Z;N<6badyy!X{2&THqPt)1o$!3{BW?h=9H>{qK1t0~saf51_%AVJR?j>W^ZHR@9_=G3|; z{Mu~NS#QhEOVVW{$@FL|D}pNt{v`^^_mmW()U>!dy8`s&j@RI)10n)1Qh5I&vXq2K zO4#B?_cM;L^h{)9P?O?hM;IEk&lrt!JA6C9ojuunLqOUr`KG&b0=-(*_DyMtm|_}w zTNAPViygFo^Ge043}oN~i37m9N zVRMF3SJa$4GgC)TbWZtqkV*E&cpo1sx}h1pcj5mX(G(`_WchIah44wa@-6b$V5TfQ zFwi;cPbQUXZjvRcdwr-nvc&#@cUuQ<9_K)(!UA0F_ z4cGhPwq`2nq@|_T5?;I+!u&W{q7xz>f=9u_LjZn*fNf=E<(>D#T~{}wdKnx%JpBF* z2_;*lxjO5%rL@$zJV;PLN~!j>@|H*O79?~wNRvGvtgg?-6xLt<`E54#Z+Ca@il*<_ zJdB82K7RKe5M+N#^@NIan>4-BT0Zs9&T29@m=7OfMY%p{x`>OIqEyv=+Itb7VsoCQ zKwMbCEU%wCFL~SCyZA^f0#qw8(I{lmwA^-?GCMoF*|S=WN|WBwNfV_GHFvK-*nscW zH2=>8OjVK(F_q*1l>zWT)X0}2uZg+twcz(y+nR>~1d)dM;wRtv9x)v;Xlby(&qB0g zrbV1a z1B<}_2}D=YK(M9*LNbnG6g%k6|7z{K!?FC|_f;|?duBgGA$yc9k5NRFy?0dh&IsW# zGqOU8M6zYC>=_Y7W=2t=$KHO|?fv$=YCJkQIl zP^X&}JtI`mk2rfo+0cJEwy|^h&^VF$*u7K4I3qmI8~CMO{v)O9`KHcrfGAT8whQt8 zVg&Pxu6r1%SS0w_qLgJ(bTLUqBDT>S-EIFzWUx&VcvN+=Q`SPdipp+BZ}bBr78BJ& znNVvfSHcs)WB>QVcG!G44LMM2dyZ@u1R>)LONhQ;+Q-C^%ug}6u*CF9;7oi6eQZAv zH84>Ti8<$p953MLKH_P6O*LB~+K103F%8KUcDRmA6; z0cSmkBy*2fSr+2u%<_cn32pVBxv5MNc$KfJwGv3RG(Y;1329j9cNv?q$N?Ctq~tXl zV%LBS`i-UKftbY%D5{gwyUTQiiEoCbH6+$^Z#KXA4C$ZBY78gsOy$M>I1{}baqNvA5U>S_b~4!P;Sdo4S2I9Ls*wj(FG}r$QK@1Y6g{QFb|#i& zmVy^0M3{Y{NNM?M1_GtI&3?BIEB%Gx+)0@jr~S>$98G#WlG%(O_hN(O6K^ZUG3kI! zvQv{M4ic0QX4l&9G5wK9h(CDnarw~SAVQxy;C=D{ltuQ2?TE5SprqaC& znOc@QRvC&#uf8(_ZU4K;MIkco#IrW%iQ9g37RQFLJ-O8>oXDJ{X@=ZWVMC4pWV{cJ zr9pYB4M|T%?ScJGmVSRSdK4@Nwbh^eRofTmF9SgqXpR^)7aFThdSFS%$5;6+DKDn?A$}p-VXS^dVR-O;2*yO3t%#EeyM`m}@wSZ0tx5`rQK2G_fTwWiHA=Y(kpzL?~{|GcX30B9XVwZV4RXGLr*EDvh1Tf}B2WZ_lS7W(1e@PR&6?*CpKAvF;bF zBNdxheT#rxMkj(SZI?Z~T7(@2xdyPR0lUcZkW9N1Q$Cgh#T;`BN0t?N6uX(ZdEjEG zK3US-LndTqT(eiD4)wssjziSkaj*5pEtWLF$p$MEN^<(J=2s4D9G$$w?GN zX6#2WZA0~mRf*0aM|0M@kN7L;%wGw<9{wCCuc4x!2xds*6b4G zOfnv-yz@MZu(j35-1cF&w35{m)SN46_+ZqMAYcSST+cFfm=@fB z@S_31%?xKg@+{HS!&Jy9+QIsk5_)Ag$E#v~%~rnaL$YZ|t-{6lWH=BvAvw*Ig(PGx^7+ZwINCpB z0XQtHIt$PAkIHR$x5)%@BKJGPyS&1gWgsCj;^S+rr9-|qvg7tTgN=OSMBL+`GH!q( zhS3j={VdD!E5mTeQ&nWiII~QjAn!ZmMuXqr=K=bQiJ~r-Y2w(frE2opE<%}qaj6&A z4Rwb#>CU+aDqe!fJXI=ZHD0JxQ5xV`;3c4s@guBR^Dx6O!0B<1Qz!=qy>a@xt@<6` zWOh_Ja0N(FXVNuwv?DbtGmuRn8p_OF8`anJvfou%TI#y(0m_RKVxXa+@p8-Jq@7PQ zz5$$|yMS-z6&E`iWeGmnWkGX>9LhaM&icL{APg@u%yaKw@vR5I@pV0{?D=PRo0ST$ z-+dX0P_r^N&U`2C+C>IptUm}*;-@MCqd2@T9qQQ|=SwsV5mazU{$j!C@O6%#iO$nv z$aZC8>!z!b`o|CeF~zh2m?J{2-*Bv)Qo@E5fS&YY)jmwqSQ<3s{k554$P3%#?b_b& zhy17%4t?Q{15(uIt(j2*@BPn#RQ81X{UwMBQop3znw zD*3HOvVDH3?WoN+)s<>xfz@^z)8z{q06|B$>IswjO=8$c%^Z;1Whje0sk3t`nK*Ko zzP7?JMEE0sEVT-78pH|%L4?opyCO02;%TV0_KqJCdgsw5zYUj&yZUCUt^622?pk?y zWt(xfNI5Gowj4Yr2kGHkjb8?o@D%V~iPbg=clEZOkR7(`Uq~`vi9am+_FUjVvzc`a@v5|Y@K20@?9~IOYJnh zZGd`ECzGmGM-A)h#3xew#Iu00&uwS_D5nBaDY>Tb`^0b{7Au1symgqgY7WF`2!#CI zSgK?tP7`?Kbe?>1sgcA^&1Wpx=K;X%<}P}C4OtAs#1CDsI869IC zK&cd17c4qe^NPQ*?mhwRnKgI9ReYR`aDDbAvtJfm7&vk%7@1JOmV=PNqStQj=m9cA zhNDigIg~LvS~$?BQ4Cq{4)j>+RhoWY+UZziJ=h$j>tiG%AaXV%Cl@U#yZe0^0`!U2 z=uTnl$AgT6tlKM&Bs17d%fCILxDcYX=pjd!J;brpK06Y<%YGoc@VME})5;?Fd~?Ed zT9B@I;MW=5NB38hQT9nY_1;$3L9;%)>a4=g>nQXa`}_M{HD6B#G3J5zz@BvZ#x>WT zlW(jYMq>6ml72P9r`$I`RQ2Z9M14jg;cIqB%t_i>RKceS*Fo1T!~7}PL%cz9rcY>cX=cXouny0}UDTVGt;Z1yy~*sCFB87gDYVB5R!z7NkuXbLYU=6j#kNsun0Df;ri^^xlr5+sm<1K8j#+8zvoJRwZOQJW;)$yD^+H4wa=a84FBqKf+zQudQJLXYCVJqy{eKKTb|KU$pdd zpg|G6Z1cA41E@-1T@XF`O_G_zU9r)OZ$Gb2CmmagI31dLssfNgNvE2*jq%g$!+ zh@uY$8^a~9_n=fbPqH3cwNK$vbAK_uCKf8vvEu+VCK|_tWB9}gO2r3f6h7@Xsm=SU zWy&xij*GP(%iOl|Nqoi;*~(8_I&&A^g*P`h>y=uZd=(gJuBj5vy0?eLYZYsRhI9sF>8WP99WCz=X&Pvrm;Bu?{dQi^ zTmm>&&DagpsJ{NJ>0n^;VdH>hke5{Kt0rZ)jfzbQgDd)FxImGt?>l>$m$wa!>72oG zaSj~p9%4Nm8aBUlJTx>k^aV1dUX#Cc2{xq)WP3Ym;KUC=KH>Vx5Oto68hW!v?Jz)aqP+#t)b^j%F670 zqbe#Y3OsTy#(z<44BN>-aH-7jnVcL5A5q3jKbCQ6C3(Oi6UBGy&P0&P{iPKSgc%69 z4PL0*3WYQZb;nARUr0~wS=TA@T`m;XxSNq|kkG^}@FxW?j7HdE8%s@ zo>L)Y%#L6KY`9>_F! zL&2zNAI=M*9Rx~BZ1AhOkHLHcfeT!O4zoWU3&g>^cfw=gmiD%@yFIOM%gjfAmuJmi zAtyJ!0Fi2z7*syv&i^>qlkd!Fc*iZc?`x6pH9ZiqoN(lzc=&cB7K4exJGmB)A9Fci zRPwb7Id*$nd(G@C^pU;`k_}I1db>nZit>R-#{TOyDrzK1O84wu1H^6mw_0? zthqw7j9HePyaGw|vcxV-kUa)%VFF>7l}X$5K{vJF)fXnX8p`*5LHj*wpwBH0j;q;7 zIFT2EY|j&%=)O1uk+8@5!pH~_oZ{?NvS6B` zy3OHi`aESsa0wzH1#uQ-G~QLD28~KUP>8%UMGSNpWiF3|asC2{(CG8Jz1<=OSDy`C z-ER&JMx!%=k*R_&Ey8zNTnWFM`$m?gv7Pn#?YD%O0lsEfOnJPI@9*P7{s3QyV5WJ& zZHd@qTQsWg^C^(3>y-%y?hoUF^_ccl%*v0mj<$wKN^2#Ayk2+E;ZQTDe$sXbD7*6F zP(I4Y7)tESP{UG_;X?f~LZ*sD_j{QmP;)(j@WHqC@IjhC7hfTnvR1^dxPZW;Sy$`a zM_h)Yod2+#AaOz-*S6tRKJNv}q2xbmX!RS`@18yd^TUGnI_H74IdD0gnQnf3FcIr3 zQ}I;xQ2;y!2nFGnsu*2a1>r0{;5ScAk=Uv%a8h!uQmKne8}VBcL&0W!!J^%&CF zs)`x9=3P-){U>9dKsm%xS5`N1qom|D_(f4p@5c3^t`)o ztr-ZI9;DNlu6ZQ&`DK+|>to@RQun}YyZMbU?cu@ahgfhFr2LY^E#;qM)BWYej?@to zanZIi4)Ul=2uO!w>I;I{LYQa`EH6QiCU*3|^g8oV)zRw-e}4o1J|>#es)-?LUh4fk zbJN$&XB}CCWE+U5+aEN`oZt#2HZdpFBGUGINLGIvHH#Azh1@TT=A$8Z(DQf&zIn z@LNWz>uqnBl-FA`W6IDD{j$BqSNG-7uTR{b-8i^EolSVOpVi2mvc7Agztq#;LlOCt zI98a&AUNp4>Dcm`HI9^L&a9VNZq!(pR8_WtJLwTUJTel~$B{_;Sc|7>fIjPDQ-Gfi zRqftYQExSt04_a416D902(RfoquGY9qkWo(=Gc8Y7hG1gV z)7g~Jkmnx&>__0G?4lxOhFPJ7t>H%hbJ0PVE-7SqhDt#(+qMlXC`bnlefvH+k$U$^ zky$gU^TqC7gOVZfYSykH8F`$;$<=~}JgUS`|30~yb3?+hzr@M&{UaQSLzIk=>G zv+lX0vh{SD%5>gYtW7fU1ig%xY4D2ME7 z_&|a-pdJ80U{sn;?uwBKEGpA{wMg&6{Z4P`drHXb39TC3UM=fDLg@y8Zph>G|M~xD zl9rdZ;l&8<3yMrI6>WX-@p}la2?(;*g_CS9NN(J;%L2prYl#|sw2_fH#hJBw%ZwRQ zWKFF-^Qq=^jdUq3GF$x+<8v#ytkXs__p+5>%L3~wEs&-^Je*EltAx}J3fm`l9y__S z+;9>%T6{k6Pxt`fC!7!c0Drw<!wb)`*@S z4)uup&Vxt3X#oM?S_g_>w$Vvu>BDq42^}z>M!+o888_hlMMeEZ`$>b#9SQT^cY#pe z(I<#d6`WjCc*vi8H1r_@n(yL|ISeA(-{zmpcv3%7UAA%fuR*xr` z3b4z@3MI?orC?S-uv0hMGsOz9N6Rp&r`fkKNr>{#Aug{^ZT-U&dy7HB{CBJ{*y)yO zz{ZxRlD_s8z|=rxmRIn${SBqQ%Oe+#G*Sxk%gBKfcZz7Mq%&LBqjM{_<9nJUo+bE5 zfUy~(^-8u0cbNJy3%i=0_f_p5UA(n+Agy7XC+=wK^h}gJ2Bzq_-h4y981b(eXlKRq zh_YQ)<+Ec5__!e%cpNzy*Ce=An18=@reyAIDZyrmGE`R_5YtL-1KB09vuPSu-#6X4 z=C>k0`Z>}L2EnciMcyV?Ns3)1?eev}|>=fM8$);dFeDc(BV4BrF zBpLRhw;PX3D8hOBE54e9{GpnT;L%~Lc`B*v=UB4|DUsLR1B$HJPgg(kV)Q~sjO3Dx zjV`~lA*>=HYJOrO<8wh8BBhi0Pkzp&s#CX0xfLL7FpWWW#NjU^t;2V}e(8SgI!I~l zlp<-8{-tiL4f_t|Hwc;wkb>^m=vpLXg5;uih8=7kfM<+I$~JfSSy5R;CsRqQ;n4W+ zQem)_de>gH+6k_f_wc4idcQnaHv%ue#H$}i?_JvQC`CCdX-s77xJ7EBUQEvBLO<@< zb}OXANF}T=F=wg+ypL&5X7&8x^{J-P%s*k+)<#z^N6Y=XAciv2r>}d)mkv+Qe82bk zmu2o27SSp|*oR$y!Qq{GN&i4eVWsQ%)^do03=o{ruMr3zt+bMVNw!`}TFarSX-OnW zEZvu(Zm8$N15EF;abV+BUDiAf1O;8}KxKRH$|B11b!{cH?$%m!I$N^@i}1fz8RB7lTvtSz zMnXYOPk=lL=j?Z^vGul_bNe(^l0`<-L4F}q?+Bct0+o>y>Q^o$tA3u!m?VXbOoEB$ zUoj?%l1+VxC3D!=p;#&%)%*1VPpPl{{t{!2bT)_=`aK^W%K?w6dl|n|$?RTrn0g1K z3CRc=3g#24iRi}QysrKJ%v`eq0F8RfTKrBMldUX_7EMzpmL*LvM}s40w?3>uAzoyU zQtLywN~%x6n!Jnqc1PpEFBgoxZh3a_LOjH=LcG4VoG2kdps)!v6x)mq88WE1jB@;V zHVf5!h_USa4v~_HdK~65}0)1 zhWJR#?8(c!?0TFG93R z5z{9`=T7t`{ll9%{p1QL)jiAXGlt>%JZ!i6clM_4jT?ma_VxyT6FL4<>A!ycB8iH^ zbk&E?%)Jo&?y{f#(f0P&Q2p42^UWRFagEWFZ6F+?ir_Kum*wLv?y&5ADkq68by`mq z8wdi5ofsc^+F<5LG0X$~urskhgsI;o(NV{Qw<<2AiBbqRq%_`;3}}084(>2x4F?TlhuGrg z=Cvd6->I1&ZHwwlmZY5D+VDTR9nJ+~lQ8lfXCCbOn?V4{Wh?ibJ;}Nduq+MPLy?x2 zmX|v31U*-bHkQX}S>Wj+2Y6n%f?svYHCN6D4;~0OPD!DamFI#CRWBhIgKZ=0ic?@w^3Q8~gLk zM*McML_BA*uAMDE^YKz%-!anLA?(7MB4oA#u8)6g(>`?Yrr~&(ce=P(nym`i(zf`J zsr+W0PXqN+q0g#*t%qmvTgAKdv--7emt&Zv&#|y%+63fQSL^YOtgYD_glhU`CXTph zC-(H*1P9K$cR3t?|N1qbOwsl`;_=OY$YFucWH?tC$s<;`UbF(pin-CRZ1q?*A@kU6 zCcU0%u>H&tU*)Zik=_A9;eC9w_E&f74?3syWE=?^za=8gJgk1MRTZzEQ6>7=bfenV zt@_W!vL!UltQOIy&jY(z*!qn}jKFCjqn-&$3&UM+h+n02JLT^X%f<})*@q49Q;o~4 z2wwyod8ZO->FHs*6S(G%_E$ThfyB3>fb<;`qmcXE|Aw&jzHJ<#Z%a{>yGTNn7-4-tL%j)t<+e#$B5^KZL8s-`3nLzcNhD zu4D#H3B16z9_+mRZRfkU3^QRumiSL$8phb_Vq&ppF%iLtWqbRR$ZI{gN~kk z@RsBSyvBNL5={KChuG7;ng0$J)>c;P=+sh}26Z))jsN^+qKdkDM=yU$zD(JslRfOww(+3iN-#Kw#OmJ-glaMb zH4pwA4qm`#5H)EKyV;)(7V0cu>t^V8aK=)!coQlk`YqxBPMH1yXO+X%jj6R=*i(Jo zuHuX}`bqlw`7I=;G>|n&wryEHmHx7mlIBo*gEJ#is8)Q!DLB1-w8Z>OLBUZ?iw(4f zNH(@Z7<6LHRDr5Z{)Z(#oA$dSx#0Xjl8Wj2bN-yWGI-7aH1T&K23LRV=7a4u*TzeN z4cD@GSAm-HA;sZVWQa&^Kna5q+Uw<;A`_iypRHuQR7V&0XZ#)84;twRkzFcs?@O&F z{k6C=s5K4rY#n%-qb}e17XT?2y1|V78m=~h9XT}?&^jZS6A=xNEZyi#nb;9qA4zpXY#tlxhl<&0@J2+?R8d!RA6Z98AL(=H^IfwEi5)M{o< zNrs(XH9j70otXdeytXu*FBy62WhBh7NpH*hc$S=VYb!7;2^#k4LFs zMR#=d`t94`^Z;8typS5#hqScG(B@0=a3PdWiGuA7bQS6DCIpT_gdzrob5mVCw7k3= zWy#w5UmXhm^_3`JvgWX+reo`mEh!tsm{GRCz(4`qSscjNrI>+H+BR^0)*CmqrDkEF z&-b3-8FD-2*;;(0Q42Mn=|J@&K+m$du{W=%&e5lrz|JWyO@is84 z!BOPNom};$HR~?S1g5j-w7kV%%fOOmv6chx|i^}a_ z#~B`aW4gLj{!BMz22cuDn786R2?;TuZ3{3L1bR zb*sbu?c_|-+1Z7Vo2shLn8(rp9mU1F1x zIHv_x8uuNH>lx_->+7YwQEWw48DPqByd7{HR^$3}779-{Nu#Maikzn1KHplK59&*+ zu9Y@#!G%j@gxf~?Hh+I*XJ=PfdKGsa%2GPsgEx&PX@ptFs~kI(S^S=8zth1wTHt}L z#6k!LXxs01QlNK(!3$-KB^vh$M{>M0C5CN+No5l$e4hP~sU?(a2}l$4a(8nx7G z@DVL5D|5rfTg{mGVikAx_Qd=Hz@w%oVE}rcaZB)NeNp9az$GOmopAqZd~heF1`3V6 z3zW0CpCQbcqGe>{wN>kuN;V09>9+XkRY^%n?Dly&0i3O^E%4pZ@$q_%4;DAA;4))o zXE!!0xQ2L>z?mPGo10s&(HDdnp)vK_Ol&@m9==KXfhwGb8g3g3MmTeR7B`LOaTuk2 z?yBd;dxK%c%XDh+hQIA|>2++OVUgmHBK zXmRo0ZjHqv4CTF=Xs`0CnOl#r!a|O-+_`!&;{FF-#4IvKz&tta;_AAP$X%+VYlP50 zEXK5x`HVm1YOr{(T`qEwl#u92P_2pLgqE}7E^FP=blz-PC$_x$)rtf|g4y)^MtSba^0vh-{!}h3CtDDy z=Ia@V$?T2FZ~W+NZ$AqOf>7XX-v($X7wwIsn#`vSR8O?9`pq&Qvqi? z_XLWMv-9gM7ip~a+IEAs=sr;h|C|*@@56{egSf@7#ij|5?^5xNRz6JAoWI~om_%aa zy^x61m5b_DB&3cH*4NI@Z+~R+9mH5P}H{(B;w3p2v7L9+PJUYA6jbv!dpc{Bl(tn zJ8vFq1YLM!YL!42>G?GJRg&H??FBUz(D)^i2 zLR9zwNn%7Kbzhub?=fsY;vUT_kSaVtk7g^A=7cHv{Ds(PCc|^8X3^i7INg-+)_ipM z9GXg;T)yj$hO{-6YCS^Jflu7j-?DSaAOE{k^fh!~8w1Q5qmqmqxT&w8Q=B;{7jCVL zm)nxARL_yn&5pte?{)$*C?GxQCOxMo=Nn(XWJQ|PgthAY)*Nd{^9PFvZEml?_HTLbs8Imjjx2?>=b` z#&7NIeK&T|IE;iT2pW~)5!2x)D|d3rf>~2%;2C^`ohl!U*WFnmA|@*<`(UD4)87#i zb;$R6_6Iy`=A@6efrl`^8ji&r%0rRqDS{rQc7v(d(c!+PzJ56L64Rc_)p|P*eg7sy zkXf`}=`a}q+bp|?2o>0wAuTQU@A#ss~ftq?`=fZ&0Ix%$b zuq0*>BMu1(!90E1xwgucl|tr7;@stfs@nB%nWNLlKRT2sHl`%mRD9f95mL~ z8}-LWFRcl+^Qy?avyrBX!l)BK}QQ-SIwH!BV%Sp^kl* zM{l2JlDON?eGhmV=zc!XQ=O4L+(<~@CF^|y(ZReL#?00>OxU*n<$KTwum1THg`MlH z+DpzWENlngIP09Eiu+%CaPs*Ca(ppP!hRcrK+hp{uDmxZ`JGkwq-fwwAXZzMux1jmthn3`xBW(7n_;R>1w& z-3iwL>9y_W?59I-Bye^w=m8S&Y?lM(-P*RKgh_L2>lrwPZZXfTuJQ+Sf9zv!7gGn2toear6|z1hoC9-I5@k?zPs1o<_G;XoYEL z9U|$YkcOPle6sqF!qCvDkdV-YgtoKT2{i=(CHo7XUzd^7W^iz@8u%CsQJ6d>cW5A~Z+x>Fjsa)O55*(HefIW zsm0EXNP9tj(kpSQY)QI@(AKZasOah|+Ka9Ortuf!>l8Cd)m%qZ|ISsT`1TXqBZw@}J_SaN{ zs|tOUn2vDx`w`eOoP@*wGy=9MtjRHif0i@~g0@2N@oD{!QS$$nyyHK$CH|CrjDLO* XZ_9I;{w#D`#<`)WjxM@x7W97rhSJBK literal 0 HcmV?d00001 diff --git a/examples/sessions/noisy-sensor-example/plots/value.png b/examples/sessions/noisy-sensor-example/plots/value.png deleted file mode 100644 index 352ed1fa4f56d823b19c50eb5c37dc73a5ba75a3..0000000000000000000000000000000000000000 GIT binary patch literal 0 HcmV?d00001 literal 26799 zcmb5WbwHF~@IJbTf{1{Ef~a(;q@;8RNP~iODJk9EB8_w{-AH$*A`(kTcL*$;H42JgWE(Z97NUx|9 z{Nk|_Q?*mHG_-U4Y-0eE{cLAtW@%?;tV`}-U}I}+X~D+C!otMDK>o$f&dQdDnc4h* zKfz>aW5jGTiGBil!LpK2vxUL%K0`lfKLm4)VX!aHBt?ZkIwft*U^+cg8f)E)N(q

@j1xOMNVuJFSws;1*v4RS0B>a1;#{BG;OD#!GGBds+S}UNX4}y~cS$KDCNbZ> z-OhWT{?*Av+u7Am8k78~o9MfDlMPQmyBC7n?WFW${>QZ-vXw!!fND2=WjP6nfEo4A z3kwTb`dyKCf*uHK5#9-vULP2+l2M2Y4Oonxzx=r1xjCJ$7<+ZRicGZC=jG zq7DoV<*Y%(FjMl!LR1{c3ANo^6!$*YA|HT(PpL9`4KuZZtFDpwHW{ujkn+34Ymz~ucnsWH5M`HeI+`#t7|B=S-u6VssOax6_tsU9Q3pHDI_KUe zdN zafRm9mU9;My2e%+r*gN@(PgnEh0(*8mh=hq54NUMv|YD22tKaatPj6ZtFtliKEi^* zo=@$a?6bP9^ANac>a-+bu`inozH-gz??Sn=7&>nqZz)hwzNCL*tX!An}NlIyg9(e%JQT z@89wIUlS@nXxOrj6ltiekCj+PjopDg;7&}U)nS@o{!Z^JHOBLBN%8AUR+dpG`dm@e z(27E~#pwg*7E1!G-NVRS%O6yjdiVyjM_*1|&n4BW1V?XE*N5m-%1g!WMle-TEv?rZ zWa||T;&1M>H3Vqh6)Ly)8H^2}-1v&za-U#QJyj~#mt31|YQW+%8UFEEo%~N2S?b|6 zID_?dMS6m)P1jfFaWlDLVe+A^Z)vEhm!7z0Ajh_Kj%Qshn-TY5wdHfcxkhG@6N&Qi zQ@sVz(Gg=~5y~UK^4>ja_|Rezh2!5Z@x?1FWM^*rHW@rW`d+m6RsmeB!-rwkzWX*6 zwnIu}Z5BJc?&^n{TO(hVeU-vsk;ON=BE@%7y8;V3x5=*u=gC`U(u%oYd(j9Cauq3~ zoI#;9w^P($nlL&Wm!=v6A-CsMG;Mmk_N&?;8$rzEU*SF@yy z=W;K)2l;b*QK%Bi5oyfMszZcyjSDhMQvZoo)$|jz0D|?2id^FB^hF^%nRm2XCLCY> zk`uG)2Qa9W{|*_5XEO>l=BP_b7xC|o)A2AqDiHR^Ae*$OZ?Inc1835~dB({;wN@Jr z4t4j8$yhymk zCY>jwv!)>4@m*BHKf!?dZw->_%IZyfQ{?>&K{AHmS z%8^Apr2axwyjetPzdp<|1rE>-ZNLC*dH$!5;RDy4pvUwHeAEeI?o-v~3iWoz#N0Mn z@Xy#_7kFBa1C5^{HKpp;91ew6)|W-M4>g{4r{R8JJu68@P30`2_Wkrw=gW(OB}xX) zwWhMK2$J^SK76mrvZA0rUSmqbF{6tSEjXP0F8zo|iMEr$hxeW2W}W`2`wW_zGnXXc zmH-V)U><65lFD%J5EWgJud|w;ttU?R?ar=YJr`cV(LxH!79+7NO_W{%=6xm zf0}8mOzFu)7~+0G=lW$C5k~nA*DFlDNE@?)85}mBhX zcsSGQjo$0EN~e_Hxm7b~Q5j*$J2`e<&!h zkB)m>6uR+whD~2|u^L}4&l%eI)lJEDpMDIrRr#y!>#t8_D%Ni`JlFu_s_?NCCC@DHNJEGF>!`D;DyWxml5ztG)FR`;Ho zt@fzx^iR=rEJpRZCp_5hGBuUlFQ3ghhNbTOvi9#kNm;$U#^C^`&N1u^5B-2LJFdr7 zjN>6RKwiZcdT)v^^d-V8nwx8~?edl0O~B)oa_3kjp;ikJONS z`m#B5_wb0FO~1)a(OnJI!eryRFJ(CB=TBP_Me>3vRTpN#i9H%6FYpT49mA$rO&Cr4 zb4ZAml5%C((&hYQZ!ykIQ%WjGjn$y|!ECR3ic_@Z2gzNSJxAGdp^8rk2Ra=U!B`u8PMV+aw9>JmC|LZX=ocj8LoNWvLA z2(stm3bGW726M!+Jbd3;?29+)E5goi?4d=;-Wc6 zer3xZDi~Uor=$_ms0c7IUiURu?|jUN#ygC|+xcSlN9WOqX_c5(^Jqcy^FZ0i$LD&; ztH1(eF6LxC(##6LY$ojUX-*uWS*JxDxPW_o7uCb(nkQF2L~rQVwh+m`TTYh~S%EFq zJnm?3M|K|+ExxZr9hCIFSBkC_>m)H4E#vGrwrTq{=x%J4-{Doj=}Sbk9_eDozHKA3 zzVYl7VG$@N5t0AzinY@I+nUs8`s1pXjQl2;h!Yyjr5cjG@-Jh?6`C5k(R_c4BSm8w z>ea~h={Z?sl}Uo)Vlex=^L_SgSSdIv=XOX_N@y=1|jMY4?O)5qGxTxC7>QcTk>|bAiC%h~x3luwYtIkJ^ zHK;_L=TLX0*Xq|<{*pRvwA9)=KD}nx+HPHdbz}>8ejeZKIS0NNo8+FyDUyioWz%bR zr-99>1PNx|CU@%s35R~Z{g99+ER^Ed68idClD{3Tui0_#9sEA&fwPGTh!tZ9ZV#WO zmniw*^-^WnYvjlfC%Oq8py%#HHK!IIiM*fw0r@<&2o`R=nTPNe-adWa>)k#>h4jKL z4ccG77*H2D&8SX$57fU4eiY)*B5Ex33A}|~`K&r5`kz+5w5Y`19ZDYCfpzqu8FtF6E?R2+Cq#4lS1g=`?4G|rMIp^6e`VRn zkDX(!4{CGK<{{|290bFuYKmDHF zXN4rjKRFEXOTaR{|GwbKyiei02xcHr&UvidSOJ>1%Vt?emgCMWtL?JP+CZkiOai;4 za)~yh-k*?%8Usnx9Hyf^W0kFUVkYiL8r`#sVJ^AYMWJjEOTjR!yeQPeA=} zs=4d!${%>H^cLKUdL++yJ-5T{{rfs!U>|im(&KjdBo`p~%?k+Yu@UL<)I;@UcKxr0 z#MWgL`sPONrC-s`e^FWdrNHyU>hkEYCR=&pQ^l84d%m{f)!S-iZ{Ef31hX(wtNV#H z$ic|VtejMu-`4Wk7G27KXV>;G?krUFMec#;mAvmqiKH&Gr8O4BEwyw>SNmU zM1FZwu!FWz8x^*tr>gqtiy4_uV4 z`a5`}^qsI8p2wXenv=h$1~+?!gSB?}pqG9};B5jsxR<_8NRq5vztSINo;3daXtj0Zz#$?= zVDW5LklQ_K4dX{e@g-+t)5p8mFRNFW^Xo4@%`%;Snzh)Ub+^@_Cs?iL2}XY&=gZ&0 zWCzPY6Vq6b?R>$Nc!MjnxvYKv8@G;J-%tkFB}O+4lszDycCe$?;qnQ$d8 znzHSWys{coKB{Rg*IqcZ3p@W?VKmGs*ukjglw{QU$5y!F!5$xJU?TW!j|-Uh?~msGVajzDpVPG2P}*R3Y?LW8x5 zAC1IB8I0^aBpc{aWAOQZp1?tiq;Zqo_!G&WYcO*4L{bjH9hzw<8ji(T8I!40rH~UG zl{q+8pfzAeJWSDjn|dnD2!oo1wm$%adXgtAcyN!UC_grOGzJ}+g3B|mk`%kF?2H^M zGQ3=Sw*JU<_d_(hs>epPpA9vWq%v7R)H!W&Y$}OukN18KVs!S2EoCA< zSs;eN&XI-LpCqot)o=Q1rvnOXhqhsUNCUH$90%f5wc*sg@!ucokV6JGuW2O{lyRf- ze5PtQ4;l$EPns)TA;PZH{6n?b%H&}_*VnMG_l}NP*lclX^MgW?Tf}Kg_`c+njxb3A zt5%s1FJVEArR7rjPP#`k?evs_#pwZ=n&)Z=AB%gYo14F@Z2yMLT_G5( zw5;~`N9Xl9vz22_B{80{W;bE2WJY~Rbwy~Vi*#ZnNaobhm5L2`W*|q)ISPMJ;;uQ`g_Sj z^FD)))$x)2$Lh(9ot1&s@Sh`7S}<#-hd({ackD6_h+8zZ6RO1UD{$0na!<0@ ziO;sw*-X_s)D3#9S3^m6lo7XsdC+Gdj-@G=X{pIuj~s>uYsEOoKUoyi>~tt5t_fzF z;tqsd&aN{Tg>j~NU=O)SfYI*uzZg%2Pnq7}) zd~pRlOC0r#mAPdW`g6N32~-l{3Q^#5wvN*<4?<}hztP(=Jg zGjy4yh#KWOs3mcH3&;T>3UdK`%km)lWHCC{#@qJmw3UfNI>14zNX|1h4ejL6Bka=A zkbh*#eKqSbc$<8^SL4~><&IAF*NdKm#snHI<_D|ePl)8aSJ}WYCzZ$q*FB|Fx0G6-oe4pvINh#T_2Qb*Ze?@}Th*`xz!}q9XHSBg1 zqLwF?j0Jv(O>y4%J;6!zow4s7O9{tBiLb1K*YjN!OzQ8SI}}k<=SuI>(I8Jr51k57 zf2-daJ730P7+z+!I3naX!rsd}h(h8(W1J>7#~?3obbS*WiJP?by#39#5CQkq#e3d7 zMW*rlZXO5cIIK>CPq`nanhmWCrWr1U^p@}Ie~rdEijvAb!1W{Ut!;dNhWc{+q2<%2q659i}P=`hvEFZ+cb+b7K)nv0p8Y;Mie~ z+)*JZS2OKYkv>iYIz&C|r})GJ&sp#5-Ifg#>0>`6HAfpIaG><+OVUpsdI4h8_u#ey zK~l1mM8wmF!plK#tXO(!ufNnJpr&?l2Wu66tkE7G^z<55GxQQGK7mNoM9919Fe`PlQDKW>w_C2yWOcYoxwV6>;&-!)`n+T_xhxZXALgy^UQT_C zXnAMH8by?qqp_4M=-Azx$`n=9vA}#r0WHFCL{52aP8dxXVo9wC8>(LU@w zTdiy8iSByo0>yup+H$Cj%FC0(D5tp1ggVG&8cVW6wr9pGznF2mD7$>uqTT78fSbyE zCH3s@pVS={w=1C0G7;Y;9q-gCD(IZY9qFm>GpjiwAKEA6fG!9sX3Nc|%b{4&X%^V} z@c`Q5#v*4LIB!Pg^xw23?mvH+#;N!P)#)~H+u`QyiS@M4hGQcw?5UR0nK*uUK65?u ztG6+4D$fbUlvk(6)C0%iCA4R|WV07~4V%0%FAYR4mJSoU9B`+C?;gcjUnVBHk!Jh2 z88{wr*O^ZXqSh^^mLhh>5M?L&Y##bDbF?|wL}t<*-d}2}{T_?RhNeR19_ZKnx&SB; zW<`wc@D7pXSpn>7we6#_Z5#w*x`Vf?NR~cmnSNN`2XXXU0!xSaMxE^)|Kop2N#oXp zEO1qqIDPMF`}6a-BawRYCs#quFRocl3V$iZ%@%b6?+I9Ps^UDxL;7zD;$De$=lS_IIt}@TS_I*S zeNukEsDN!Df7bi%(g`d_gy-D{L9axYW^sPDLu;veF+Uk~Du2DGO)cs;A)vQf#U}d3 zNJFANl_#TH$mgzH?uOX3B=JI%WImW%G0bGu}__Vg!~K>uK*6N5Xe_?54_L( z8AQTLD})3sPuFJ&ZT4RvWISS*OM0VhD6a68Xy50@VMVT=y?J|$0art?*HJTtLCXy; zPd@&ls%>Gc9^r!aDH;#w_0T?UYW?ieU2bd635M&)bC6`JBp zCclRfLHj+Of!4zJZC%j~?jj&i@{P==GCxr-ZB7&opjZS`08; zI`&ra1Z5`U*i8Vm48!X&sAyjX3cP^W_moxJW!Gh^P3H=RX=l2Q#kUtt74=q^o;Ahf zc?$X8PMqqNt<1qLc})e$NyIj^sMGMuQO8HYwsZ$FOlq5g@3dkWmp(A~8bDt6u6tR+ zyg&`Y!X!^|WbRATO-*rW?gxS6e}a}DBm(t^PrDm8vE*&_8r zJ7RE-VhpGnf^wrc!*3+n*GGWGI$scjpBzpJY&&YEbyOkyV;VHaULX(2Et{W)U2+<- zcH!r*5>uWMfn zhNsqiyl1-|OnbA|-E+XmkK=970P>Z@BsO*rJwX~v#{&Z`t#+>$8)oIKs7xBU;xWsh zTp$tl4vD-Jko`S2B`I!s%PR`H0JRJxo!k}!+RiKe$;%EcqcO-4ZJghTKuqvJ#mxg| zyL-!Iyi5D-=RI2x3VcyqgmfVGM{6B`B%~?l;wJ_ygj!AszpjPgTxRk8^#=w!x@a(H z<;Cv^T=`k;n;<^wi#&kgrqoRuqLX0hUY-NDZUN1-qA+OCnZ*Hs*T z1oVIz)BzI(CyUZ{a;X$2N#Gl`uPwE?zLiBOY~Dn$)+?x8Zf6|64{Z=Q(;43|&$@q~ zt4n1{eLi9$gpuPx63VSU9_A+AMM`lKrZON!E|F8-McMB}!@(B9YHvfQaA?woU3>e~ zCFVDEj?_oE5P{pr``D$9Wk6dAgPU(9MSs-ZnPxc}@4W&zqs}+SRgnt99?%H$f{+N3 z`liiaAk$xMP;<;snrHk%1~+I(l(vLHBXnXyBPhbljsh+$|Il`os%C4{1e_qgY^YKJ(SvopArnc z=`MK|7Wn5`$$<$u`sBwlXarwKzgbDFD6;7TD#Ku)SaD%%jY98?Ga~V+%8eX+Z|2Cg zCuwYR{ieWR>n9i1?p`FYOOh*-41wNu=%UDnI^sFKp4c_wJQic@|^PeVCPyzi!mhc8N-2i^11ReF&(7cq}x`ClTKi zz9yX(kHH#Cor%)VKAYoZBE>CNE_EPSYHLP7MD(Iay($v$DbjLDToe$SX}3MC78w=A zTsP;uzbpyDOtOD%mV1cduRnjSYqFTB|3yd)3zr2TmtmTiG1xIV66va^-|MY9VRaH> z#!txHAV=E?0JaT?0m`Mw#>U&e!uJ&D4e6Ial*X@Fn;Z1!QLnLi_+kmr`NxG5w@u=^ z!Mw2P?65n-DMVymM;OnydJi@_+pf*EwAgKo_D&!Fd1fi0tD9jqSt$vkM@w}pDN&PE zrifbW#ihLtvVPl^oyPr8=L--XHnG?UOTu6w^EgR&19J0aZRc_LSRWsbcORE zzse%SMFRwP9>QwV-eQtZ_~4CH8Y}J{9>!dr9~;fQ0l}=bq1-Smfsdl1-`3;~LG)3Z z6^4a{6)`+KTsQ^aZ}Ru^BdqR(!5&KXmv*)(0?vb~@LL#U$BLf1c&hsy{;F@MGhQQ< ztkA4wZi{SaU;I@~%$YZ27}KWztEq^(@*5sB`pZPTx2fE=IatqGJG@tcjp5m*;%(x0 zJ=&DsHtLCI$>GRWDbZFa(dJjHF#HY@aJr?Lo9wzeva+(`X@)=?x>qN-U6;`H@5;#N zGgtl=`gAB`W^uD)$+X2h%D`1O>sv{j4Ds+}CEwDES7}&MS2%L)^5XSkKC#%x26&oG z7paGHUdtKRY~FIq*K5_p3+A#>9_9to;Vp4Pqj^+TM~j0I@8gEEsmi%pjT%e)&4*z6 z$J;ZkcB}85SJVBoKDYTT3v9W>s362ZDoqlv<96ivpfwrK-6q}kpvZ`b&$oz41S7-4 z$+M+my7yTun@@8>o^sHQUGiYi@bhbJ9%2~74~6y)7!Jap)x3jfzd!J>F15r=g9Vh2 zq5Ck-2mF7Dv}I*wDve|%o7zr(VdwX@aLVB3goa;amN0wMHq2aD`&NGYJpSQVemwUSQ zS$JyQe+N9|Ru`%SR;zE2N$ZG5qVhaG)|>3s`rrpE&3Z?E6N~RP%OtNi!nCwREKIf9 zX2onKL0;1^%}uhBRsmDfwI1yha0nRhr7Lc%t3F8n>bO3fw}3pnR;jn+lc+Kok=c-N zaM8fbSP|3UtG1X)Fm;nSKiWcqBi5|XeQ|NoW5eQZyLsx9gBjR_n5P5bguwx>DHi9= z9%tuID3_tFH9rjGQbd7iVD5@9i^I{fnf+QP^wdLZ&3ZX47ebUXw${3cKQ4S;M1#H= zM8^53GNvpVWOHN_V@Hru`!1`S$}XNczSu+w#x5}!uQ^PH!)c^rm&W@Cv!%0@i`0D? zSz)!=OO^K_`hwe?sXgS_36Kl^SOPHEcIID3@-a=@?@0pg>R@G5B~3@_qnA&k_4~pP zH!J$P$viJ8%%`e3v?}d(ok4mE62xD79(uX${=`Hb49O=>)z3CKTC#0te#m@RuG@ZZ z0$h8V1|2jQlkEAuu%t3cPO|TDfI3`)j7(Z;?+OJPFhh^e#*A9M(5|^#HO_w&X*W&w zgl99-@~B=c$gDmsJUPFtC*Dg0GZT-LZ+x1FlvI1165VTxxWy^8XVR)? zY5nA&cp0Bq``4`4Em&=m=&S*I>yU6cVx0osSEvWyN+7MND~a zzusV;QI#*wPdS$P3WGhue0PTCg*`>!E=LNl*Dh3!;MGuyy)w@qZ;*h&-iQNJ!$f!G zmT(_l`3_%U5X2%PS=V#U43{6lV4H!!P-(vRYd29BI7*g_2Gl%jv8V0c;W9lL?33rk zQ_!)?N(Pb5dY{TH66P8~{I*YI1eeGk(u93@w8k8~7cy z;otrHJ3EOJd(T9)@)6&^)89;I39@`ll)!4;KXv#HJj^VOnY+#l36p>=gTd;6areTO zkVAy0&lSZ)87XD~*w8pv6X@i^y7(B@x(BsKRj3!CF8Vd}FVorZ7j_tRrf1%My>#CM z5E+Jo%%+^huX1!K+aUvoRPUT43i!%fTGRm9K#NTpADqcW9!^a4o>k0jNC6@Y_#2=t z7-}2#2DHuJXnWeVdGlVnuGXWGo5m{wupqAVZ%gdNe8%5@r^-jsMtnw{G2;Ag^ATQw z!8#xr%l*+U>s}Nz3!fQSp{5*{{{vXKD*MvqT1w8aXIhKZ(zWu$#kCs>-IP;Y z-K_fb9n`v1Bio_B`IX=B5^#tK+bV$8CICpr3Liw45K%*K5)OZouSi z-OBCH?hKbSrQzDl+DXOs5k zC&Y&c<%+9e`>Jfy2DRSk4k)^Y7fMujGPrFyp)g`=2;vi7x)eG%B`l-_u zDZJ?v_%qI(Mwf}wzNIi%B9OS2NC^NLCX3Mq7BQP+oSJ8q3cB$Jj%jE;t;^7Q&^GxZ z$f-QWKaK9xrFrf&D?BorWmWdILQkK>=@zZ>)B$cwG)_)_3PH0S$Y$Mer zfHaqYa}{_WFC6e;v;x|09Oy7jq81z2rw-y0h}XHAct!<0_!}M3C8RTp2lJBNuXY4f zDoM71>e3on51_}FoE zG&r5Zg0R*iFn=KyRmYcd7;>77){@2w+>PI4>RVzyGLlHlpu?td!Lfc+<_vOE0A7%^ zNf3LgbT6i4+xQiLct0FrUK&8GUX<(xNDAkxm0sPlAiH9%cLHl;GPf>I)&Z7h_PB$V z7GUGnC}>#&cwcUDs=K=a3MbCH^}Hk+PLu@4;jL% zm*0kliPw*Yca-EIZAu7WUSGj>yx$bisudTAcT6`(lip%QNtIwg>XBwPFl*S(`3s6i zt#m)y$t9rC!B%8#tRCzmI2~=fl|aNI&k=bcwUZLKkUD56jNYh7|c^9hVQG2*U{xi!%5a>xOubB z#ijL;%};XO#Br14qNn~`Qaua%E!~ALz-nOitS*NX3j+-L1nsVuE}Ll^%c+rN$ceLS zcR|xriSGgAaMs!?x4$)k(YUA>{(+UOofZ0}uC-@?9}XI7WP28t#PYM_XFN4@rXUK^ zFh&}3(qLTF;3BkB3cQllK6g)4^ITd_P^cUBWg+8veZHLY7`gK>8Cg6UBi3PjU3(Ou z$!{2OQF<1xYImiMS#38*yms797bqzcgYj5IX7?dT2AgdMci=N1yIvAnUe<{KNpKXA z`tXV(&#nRMZkV9X?_HuHAlgTr-363&Hb?<1U;M4+?-mExAO`u1O2xih1CRuljXbAN zkrXX$xskUBa8utejgp4MUuSHvenKbIWW8G08ppe;_#gltdOQwSAU7=(acWcGPc#1dRg zRI{H)%(~f7#cDtDnN8VXSoUx!9XEpd32@@>L8pXOd?B(NAJ=5}A?d0{+NGveO`$Q4 zbDFLF-qFG9BJ!Dc%#r;>I~8VDuQwWFqihQkL-6yuaz zK+iOns&6 znE7HXVu+1XQd2j7d~%s;&t{|6p#0h{V-16I(Yqi^Y{V*25D%azK<9h)CQD1kf%H|5 z4ZzJweh>JRxUJ&@pJk*pxFSf-whWfAojWeTOPr=i5E0eTG5ZYs*1_X_q~Y18Od>K0r%c)`?9WK$;qw!BQ7&VTf-OK~-u1y;d47n0qFn0gNs1wjR;wow3)MEr^088=NTJ`NRfzpXl%63pC^X8rth> z{Y)Ish_&A~H|F{RjicG^e}?!szHc-4djvJVRAoaca2gpV4Ui6Ivj+WK5tlwtqi%Zq z&dYxF3v0%SKleeJ{Viyk58XM{W>5FuB7Q24jm6&O4UPLNybhB7jBNbxZ@-sq;VDz`8zcj`>ZC5(p30|4YYrBtDDiU#J{PU@w~a7X&7VS$=C z?9q1>&p9TnS8Ibm6n;#POtRn~+&Obb*z&J;S+-3{S9E`lj8C8d6p5_G(}td#7M|KKvJA{h zSxyiBt>oguPkwvc@!1>+0(A+ZU#CJ-V>AS`Xys++Ue<)3iytLRLBJKYh;A-B2)1Ls z6|Hl1Ak*M)nsB-2wFkGuX8dr8j!rS~3qX~K)hK7n)mZ`G!Cw*7tm}^C+qYjqcGcVM z+Qm>l^cb03`P5j4tx5PH0PdZ6in&>`iJawM1~LFA;f1p{9=PH;t2?@5t|9yM<9++i zvyrmrj@9`Lo`M6C7Us$yDnHntO;Xv=QHZTaW5>ers@WuQ!5JPvl}^@LDw>$&-^Re` zUP<+2I-YS%tt&AdE5;`!P01V)@d*Tp+Xz6w>`#YbyaL(8QBAJLL8+-a-rX5RH0kN- z_+(_v6ciM?c#}v5PR`Lr4|#dN>GyG=KVv(Pt1S-Q9s9=;bq_q@$D%K=Q2T^NyZy66 zs3S7o9@Uru_ovblXgL$eeRJOb@e6N?SiEczzf$z3E@Xr?=vY}23!Y->IW?^ky!C1Z0=Nux%QqR-i+8L%!^bNX{&=I-r3o9 z7_4<`{J>1u=lj1le`!GLAfb|gV`GqNm__|cF-IoUoQjfi2`n;hddwRglbVZbyfOvK z0T*?yne%AC(m%Lryx|cPmz^Hse7eVT(o}ljT&2Bvcv1YV-mb8IxI^%Gh%m}B(=?gY zMt0_I4R0SeEm+HMYe_90%24UCd}(Ql)Ttb_PAIQ(f2BVT_>@vYLahea`8D?WJ<lX)>0g8$ zNJ|Ylc*vs1Yzj)%KWN!@+2X`a@qd*PJ&!9^iUW*H=tD2Q%Yz{lNGs}|uCvXSjG|RL zHZ}DP#R?4Vg9oZw#?h+)PJ&*ixsly+R@0(>jn=jSuokFJs<6%*G6l8-H%ejA6ECmv z0t8FK_NSxg=o#LzvEDEsg-5>pOpB2U$jtVQo-@4}6Dc2)ZEkbMRLdKG`=+pIuxfp9 zYHI2Vt+S$EfPV4M_b6`LWr+giBAbg;Xe*@PHUl;@{uxxQ+E6K{YV*PMc3 zKmyoQDmNZYnkrPAO*{`9*w&1qm1U*6w13`-N#;h5j&c8`Sp?dnU-_41tmeHBE3W<& z;r#AExcdOJWre)wzPvjiP6m&~BC#VSB#f=nPd&`nsIf4t*TVS-(E)6RJ+CV?IV`5R zOrM!g>>eNY*UtcIjbPXf)%61Z@naehzxynKcjuf}lEEDpLdynG&$?93#6w;UF{ynz z>Me8_%%Kl`L823VLH+sXSyRUn1V{mkfYJdKyqjs$byTr*tc{*W0ZMA2Hj9bmVMn&< z2y)`>GTJ`|T%qRYz=(%k&6J$gZj6Z zufbb!9LptIP}D?8Iwl;GS0%R^1`|pL@TE3OR=ltC4FWIzaLmHuC2&fIflvDK5qQU0 z3Xp{rzz8P)0Ew#i2Ad6yP3{3o;VA&Me2@?iPdVg}g`nN%oX^}7g9Slwh#93Ph$YX#{d3(+RK+@;`@+WJ7JX?F41Y8>+XTKgP zvINo_wzQrD+p++9ds7YaHhVN3Kl1U3&empoJGt(~fc*&~9GZq%&;hp_zy!n59TP1z zC|q|_miML=Dc341tM0A|(JX_)9;jh^`^}pkA;G<=1dzb@!7v;eU+EA&+6*%u%n7B^||pZ^Pu2sn(^cCi=K zh!A^_30x{S@}GzykDchv=Ls}P*%}FeLI(5IL5LC>@;Mmt7jyv}5qO4G9^r>zN5-E) zA3|kH0Lg#?HHrEWM$?@Iupi)X71Po1i!^K3&lTdWJ)ucSA-&-uVBn2>4|T_n83YPh zngK1-(^(le1AO{>2ir=Rn6LouqJp#jUe>MY^amKXqg4%S{_C;@F9nFTN&)2aAAtZ` z!a@0?6$_}5{lhqII=a)GpAQNrrXYd_rUtE|!c8Z{j1Qj~SPUD>hWr~T3E=`M1%r$) z-!d|Hbx|o0FumCTauB#`^@ZXaAmo9cZ0%^}|MI^EBB&wcy5=xbJo6rizheELkHsMM zGRH~6wjshhcm{64bHG*2Q#PxX%QGlPgjRJ+_ync@H*5+tA*(aZAQ=tY@j|=}(G9Q% zR+?Zk*eA=W=#h~Y5r1OdB7rixdA^kf0B^Z%z)r}hd;A5@f7IgRPiUCoIo=Ed`2%i{858SKL_gKQ1^R;s zH7<&R5xPafZI+G8l6*>vD}kGrgqp#&%Y8ZY=2CN94S}dO_pSWN|6T$F6<2yc;TDId zrKz$c<=4xb^XBCSyxF84DD3PhUiqc@Ck097&CT?phem<#T~~WRStgPKcftLaX*4g3 z0_1dOhFCB)BcrT(wb?5G6?>{=H8qpm&bMl8cISoj)GBhO-H&G!GDHL2wmdsRNmv~= zRgEVr^X@fw`|u~&NZ`|PqLo^(*B-?W^11D z=sqYYog8^d^PSFj;%52nbGK024LeexD6`xGGIt6!7Fq-h2de{Z)5nvhC4nt1Ej@{E z#JYR8bIf+bb}e!Z)F>~PIA9`XX$ra4_Du%4pzxFRO1+GfY@Fp4_oq>R z`w-T{G-=^FTdhfaP>cjY#{Z%e%3A98PNDJ?PP1`xJi&`nzWp9Hg$4(6+qu*ImEO7e z`T6Hle&4=HC@3hToDaa>HP{#*(~~%*==`tvQ9s&ly}WCaFaUtdAbO-f$g*DksaePI z?Ly>WmO^}LW&S)p2pSD%qtEZl{=Y`1rKaY32P?PVP_6?YNOpdAy`aQ!hV03cm!zbm zATA2m6!ylTaE5~_$wIY?cTe9~nB7X~xpZ}QhEJ6=qJqBp`>(hpVF)~})VD+n3<-&j zq>2M|*@2Q631`igw$|jiD=_!}0xa1zH8q#$;UG{V zIaaI%#rRk~E}Rr!!QI^4dTPVBrfOE4LS~D>=PBny=gg$X>3YI;I#8*j=YEEs3~L8m zKi4vpU;e5}U{ec2l2qQRMx3*Q|#JpW@AkzPYr*fJ2u}>R) zK_R?2DEWLdV;2S2e&f0uA(=aH3ChBeAQH>QNR)uR1pI?2P~ZK+G!b}-iR2kE9MD|SUxa40@PLY*rh^f4nyEHd56uf}NHb-F5e@62d$Rnz8dP0h%K=jU zF^D18wP^Jj7lOx=9%ssA|quot~VzWq;c!60<-{4c1CgExR&59W4Ij`E*& zY$AI5zx#jlY5d~e3UUQUycd|;N)ND}F-XV7m&9(yRo*YCEGA@< zrH`w;S*aOdn5@V_1_EHkC@KUB*@{4_ed{4mueeK7kAYA*L?9=(uc~|eM!!J#iz@&G z>ENZCUTU6cwujFHdBFTN?*NNLo>Rij0E$o*^7Oq9zNi9k5wyI-2f80!0zl9eX(pK_ z-dd4p0Hp>1wkZy*c;1TEGi&n}s)^qP)wn~Hga5t53f@UpD^fl+`=?#uVL6qLKfhsc zgRve#*rtzgM}QGx2te(r&9Rcwf0~>UFoNT1h8rCujP%j~uiqHZGznng94hyAiPb>L zTq8ff{(FCuMk26CB-C(lBk`t$kJ0x&qy3+cp>p9<+YKxr5yeM400Tfgy;rhZvs0>S z5#`MKJBi-%pN&8(wlV@yIXnRLjbSmb2~a-yeSoxwO=Faq0#>S<+i~6Pq9h!UzQ4YOG$;O5PwM zQ*;Sny(=DlYYAJS@rmu3GOXk39GGuMC^nzfJQv|rCX)Z|2Td5@%;{gCutx$U5I_k} z15M8o=S3Rc!dq5`t}hc~k1wxhTELanx@~lXnO?UX zwxxjk-vhA!eBUbitvJI1NPw_$Ao%Hw!^MGuttkj8d12obnlBg7I=L<)(|-nGq5EKx z6Qf~kre7|>^H%KOPsy80!bJ;uf36N(7T9wlw7Z(shnem;f1vBb^%S&3Q&~^^@fhdc z0iD2Xs9@gGAn-6-I2snK)O8Lo6uiF;4j4* z56#Ez4$m} zQ&Rs4?;a4I+~C>gh*40pDux81HfUEuA9zu}cpli*sd!+!ZrPg0zxvRQ zYtN02%#a&Nx&T|`alZA&c&s=%nn_E&JNgwB&QtbqP)H1D%-%P1(Yw1=dCElz6A-6a z+U4<20GytinR|d&K{^!;C@f3)-O+&nh6*{VVHu;{j+2wqO_Uy1ie_1s9ujh& zn8)tN|Kd(nHDan?lgn$V7lqSXzV-x@ne5z_xRtQ-u~HAs zN7erW6l?|OAl426(=GkyCH*NJ~s}7tNxl}EYT~+tL!DCKcO~PcgSRgy*i#uTc7$n zu>FnUyYtT;&92V|9|a0-v$sVv+AR%IyHh-f55@ABFd^!}TP z5ucY@iP+QvXZcqob=3x2ozpsTf&d!3jWr!8w0T`~y+e&fSYxjY&jZsfhG@VYGBsR- zxvVZ_M~>mqxP*ntC++ex3D<~y+G^MnyUGEptLJd}HkrdDR5R(KNSU+iiX$u)QkB=+mLVZg4C@(#gwquSXCpKhv2`o1V_lM01J^2%lJ}URbzu?C?=h(J%tj9Chvp{y6QnSI3ZPFboiG6g zEeQmp(KN%WEkG=&>+U|)D0k%3*0H`BPinG#d5b$RTtH6%jH!!=fklzO6JX9xp@12P z%ol9J1wXJm)y=OUdZL8VOGrnw=jx6YF7J}kdnF(C&08v@fxRfZD|8$Y!K8z{EH^$=QAN^X3+` zqJbJYS*Es0u>jQ{;3%2YG{?zW(V;Lq{`X)eqf?KnV|a9z^ts^&vz(2S_5PwC>mwp9(=xb-gpg^!J{zm?(h6qSj}BfI=ClAEM>$X3sH{ zQr3cksLtgZKCwNoG`kB-^M(P$uKU30&s)6jz|IDw;e+uYcX|(*Ve4CV@brs(LWT#GE z9rwg3F4;X^rk5k7LQku$IZk3@^JlUN7>`SQF_pRq)m8%tO4DtkTDz=fd2;#NM1r+~ z-@=jyd?}cjy?PxMyDSCq%4(zyMyBmaYNN=yX;H?GqEm2kwh^uFZ zLko-JTu%bdd=?l`_9SG98|4>KAUf2}eE}aZDKR6TsoGUQ)Z*h#VH-af`j!OiETR?G8EQ!(N$|vS(<`EtS7vr1{1QxN7!O z9cv8@k!ClwaJ>zxUhRJZl)R~WLfuna7L{SkYGe9AI}H?FBNHCxElYb^;$aymXkBeP zhjjW-KdN`CC)6h73Xeq&7sqnNC;nEc_<+l_SoE0D1(0w}>x!I=1 zQMVIQ|EX0G>J`tozhg9yEw*t>dpZMFIu;pEm-aghUrgZM67u8)FqK%R2~jb120*sc z1hJ=GI0~jCs{N(3>5hnms1qQeJB5IhF!5HINl^zU8Xxd(o_6d09oG3bsm~{m+X6PW zYk4JV*9s{jWIqn!a0A}H3;tnYIe~W`)D0t`+Z$!cBgMleaSl~$2Upevx_*R!L1qq= z$c7h>k>P6rp(jwT`xm?uBDn0GB^jXo5h{_(vDqxg5I}weCCC3fIyj5K1zmO^2O%=n zO&of&_Ve-rt@O->9Z4|5Zj)mIzDMB#Q4_gBGGQ-0uf0| z5p-S|u;HeUAh`q=g&XL)VUY7p&06OnurI1U5LLGqVebQX6PyIzA`Ca?P}B`InOeS! z@QhF|j%*3xMY#P~Q=9>`B(zI%e2<);`WIr$mjXJ$!-a^@4LYhZOY-;cp5b+hJUep| zIkMLU)Qy*jNTg$}sE@!fsPH$v#5vxLT(Q?>)D7)oevIfoT}xp=kiT~orE19mVu&g| ze&3I)mN!BB%G<~K@=awC9tgzLCMV`271J`%f3o0D!Yk$3&t+Pe?2F76#8aB@?h|iT41!!Y)pPi3{6ICLNtyS&opH zbwR_vc%~%_xhftoZ+h|N)vK1)=HCl98ahty5e?#`>LY`o`xuqJ)92yCCSW-V%E=X< zNk;rCA8>p8jZCQPn_3X)1RaMy4E)q1LP6*!x5;7E0UyZzhdom6CRq-s&4@-hhetw3DY!O|Mre_Igq+tJ$N5~)SEZ^DTMnv#27Yv|Q4R}OF_ot-^b-FlDo}! zU6Y8dNkQ{&YZO|VZ~$A8|LRU-xD-k}CN(CIZ}IFF7eN00A1;}t9(-?7-y|f6flC-Z zB}L%9VNO&VpZ*)RlCsXNLicgqvlit=2DzrPR$+>+BTr3IQ&ZjA{BxiCfU+ZkAqj|Z zWNodwtF5G@WWoJmHGE9Ad6`@Epq!GDDUcc>8S5YjphIrI7?j-l9$r;ngXP&_iC4$v<#yd(7zJB1 z#5?f4Ma8lZKR(&nybWp`DztQIrJbYWHR+j|E#R?8Ua^T3F?16e9fb5AilXsSy_!%4 zbY+RUIH!&TMeFK?^;W>O$+PZWaaL0!!(zO;IV>G0(3ne%(2^3r1j}5jnxV>ND6A~J zl-ZT^S%1qonY(w|{l5Y?+?|bEqW5HgE#63rZnBy))_bv3#H7HcnlG(J)t7u1{hC|W z@$kdr6`dLrP03a`b`Linp9`?H%VovH$o(G-MH$Zz33ISx0;hWW{O9VqDBzfR6IdC} zd;a>XXz0P`Jrwuxb_+KafWT&E#rN)t8xs>q;7Ugc2tYB?!qgl|2sjTE22Ejuk1)2B zexgMM!4~ugM#jq{p%SL-wKW1Dn=IIB;do|tfOg5&2x_n>CXS`FGNJeU+lZ=U?)R^jAF;i4$SK(WVS4er#4pP!!r z<3OH<>MTYg4$7!s1zSSa)%$&G+Gn2ZQ#*&r^p$$+qvGS|XJ+3P7dh+ikV0tu0X-nM zKLwbl=(>wrpRE6O9@} z>0}T$ROF=d&p6O*v&G+|VZd}4bbS2|d}@6rzvZWqu0^92XS%J;!@RvEedcW#pPub5 zbO9kkuPOLKxf3i~fSRL|g4BJ4(7nF5i@^!Vxad{YfdF%jqP)lFmn9-*Yc_@wuGp_1 zfGSr}jd zmpodn@nn0fL|p4O0cvY>iOW!BU%72-luBL^@{ov7$5UY5QI#4NuVS%-p1l?nQ<5XK zRPT@d&OKl8HhWyUy1EE6=gs&R4h2j+&$o=E%s!rSq!8|!DlJsJ679wpAh+9jh{vCh z^xbBk`eJIB<>Maw5pTDQEg^jFImUVIh4!DyJsH}Aj<@hy4E;Ywld;TZI-P?%zy>RC zV;MqDU(tz^X*PfG^7-@p^0aj}Yn+2z;0O0a_P)J591cMtZnqtffL4bZhK@ePZe#P1 ztg?^=>JM7BgQ4{&kITD={j%mLu(Hp}D+3i940-c$aWB ze`X9eLQi!m!p6oXB&XcK!f`M?Bg10ewLf~Tc)$gpr0N$p#{r{Wt+UJ^EXq0$mdkeZ z5NUq4h>Z--%=AmVX#NBaOKB_~=Ts_@`PKpl9ceb@kBJG&knwpe2b zoSdAo)*?oPn9+5K>ffJu@(}Rq(kt#}7>?{uDwRNck^;0vn)97?<7-81mQ+~Pl(m4z z%Mfh4B3fb^?2n{sR6(ofMBMO$;6Vi!%V>@xB29pN-D?Od-3{6EKyLXSc<$Y|)(gp`}GphED$TGMhl4R6=)&8Nl{W{vTs3{?P z;#LVnRAd>RM0}@=J>QpqO}Vy%I*#ZOPrv!1)V1XkL~D`?pd^S?@rYwhyY>?+e`lX6 z+8Q68+h8J^rMmJRKZ@m*AuJKLw1dHy%~;c0H)tKkp>*;u>xM3Ma0qFU<@pGd@M}`6+$b z7o&m|mQqTaAwe3$yX zR*gA2Y1b+D<>OXX!&@<#0xVCjXZ~%76D|aEUF`bmIFGdT#FV(Nr-*)q*EE^yJA?rV z;s*iG_e-g{F0nCJ<$Tgmv~N=tXFRi0&McaEoROGLBl#vb|a0zj&iV<;W! zL|ASR;_x7G>a?SJ2Qy3R{l;7tHEGa0lW+V1h+)j=M$tvIyH_o2Y%xaH_qV#a1FteV%j!ca0x6%&FhQAtrlpUr4+mRY4wz73SLJqph*bnBrK7vjU4@A9x~E zxWQ;JyxbUfj}$kP>vb=?U=j4}nMLPfI#(|~cbOsP@#6eH&-T&YC3<#J#8?koMB>1K z189G9%d@IOU#WBqWdjSK?hUncSkD)}%IWN_d^k74$spQtgByz~tSb6w`mTw$-`mA#3p?H((mH!3bU{>q#$r?11)jD4i#$)L!Vka{fg&e&tnBvzi6tvmb^ z&KSLyN2Vhjn@97^r`=e}<*MSLrk9<01b*D}KnIPF3Nv#l6YAZtU%$jqxRB8pxgKQW zwTrywPm#2YZM4M}=PUb#1$uMF-9JBmr`M+k__QvWieJq+`o>d+!RtKs?-TB{&^49Q zn1NF6#JOwZ8NFwJdz2fuk$zCAwzP_lBIw5IN5^(M(HHZ4S)^V|=|T%zv{J_X%o7)a zVoWSN+8_Au^H@IFcIZdTLO#E;7n!Vz1Kx!h4zW*Ur&Ze}{uE)olrZ&H+LKHNi|5!f z&4mtx%4+xG=3pZvS44*z;-xSl|5r+fqN-^iL~Q%2v1}4tGz1w=B;z5a@6zHo@-=7< zx5*3?+IJLVLFh|?d=w0fcx+FHr>1yjChH}3>V>JY)w&1z?9IB4eJ*m5nv^M2x)7?1 zc|W{c=^DPBh*!nvV2fR4m2!!c#C16b3wqHPUuD*#b>=b3sI0r`LU#6<1>T(Dw7wdN z(b3Vu%pOWPzr0IAbH_Q~MF+fn-P_CwF%D)qiSN>Dp1W9kX$BP-4kiN=9`$-EbD9*g zbzIs{2-pP>Qj#fPoZP4F%>1gaNBx$^7pC;A1lvn5`T*H?!4NP}jEpl2bsPsL83ZT@ zjNno_uG(95jUqbY_Bn;be`znAq{?0c$}0A56|CBhG}U9W!p~v`zQqTxTh9tKnK*c;!F@nB z{=B(g@m6gBkF zxKE&HcnzGB-KH;_>8P#$9wmrUls--a&0NO#x6C2Z%msTvS6ijp)kN!O^o zdXKK1ynZQ`{!51Lp1OJKwq4e4(9*Hn|H1Au7J4!6uiiIz$PqO$IZ8C~y!;1XJxem_ z9o4G2OwuLE3bMv{`tDZA-uw22;1v3Nh*ASSu=tt5ifhBw{`HYnYxAv{hHH0}ewpAT zBG#83tJHw8PFn_#Z|&$4&PRagA>--c+8UH}Y4Uy#WYgr?FOVYG6CK8dt*Z)=G%@z( z=$-z9>4j!@cgHUq9F@GI`Fi14;-H9#$kim=`l`x`vVA*G_ra8Ledb`!07KiGNQYed zT-6?QrbTuUBfV=l4a;W6E@-mZ*ZV(Z4(B6&{&<0jF<~z6!IloWWkhU=ASUOw|8KE@ z09WWx#45UVcwxO3(8>zQ${K>m*w5Vx#J&g_qbu!l7-1SIUgPEO zN5Kfk*C3z(5Rc>3LGnE~H45vh3yML6qn>0@4u+z_w{PF>BA7_;;^1hR+kix{z^;vZ z?$@t3oZ_Z2(`dtDI8SS(&jQ&Qhr=OlkTOg~V3a&-6zXsrzAS1d6Z*}3hzvFsI*rZ1 z?H3Xfx(J)Uy?b(2R+h>3J=}sYMRUsTccxyPhddsk%L~cK7(nVv9)+21eGDQ2Fnge5 z89vodYP~n`@%CrJ_(No70;wCHLoB}vZr!#pR}kLUB((BYH$c~Ljk!2^>J_(aIKTjf zVPR^_D<1&PuK+T(KBtRtIy@`h5{%OoD$-r_D#!q0u`_OAoA`Dn~o&aa43;of}cKZ>VbvW z=M50j1tX&QT{}7xf}nOUEG(lV<&{?}9;+g~xe}NnE~chZ75-aL_kz*jj&iNb^_Z_) zQgcTKyk$Tv*7|srgrcG%V;PhT+5ztd(+aw0v^hiT31NJy{T?eXpfJ`r52v*S${fwr zr!(LuqJ|jEr&DsB8t^=-zF$?UKT-|Bg&=&BT?Po$-+$_$Py>jT^;<---GurqEQH1I z0f+th_xhyZH6Kw((u_EiuCLFxO8B^J!0-6SQ^0Usu zO9`Hs>-TduG`&4Vg;g2>6-f(BSnbe|_DcXVY?rQxpo7~6yr4x3j5X}-a+t6mXF%~8 zEj9q9g_uO;ufWMWhN+snq;m}IywPS~Bt`bA{0g`?Sl&YpT!Ob{#G4BO1+btG-hYGM zj0Wj{l#Y{Qt;=NqxoGuJ?K=CQQ1~PPI`fP49?`~8p-8>=G#A5bIRt~4wUiOJN?^z6#$QZ3#GBXP&joGcB0Ps0_O{Y;y?JQ zi+CQ<3>A=ifeW 0.5" - } - ] + "findings": [] } diff --git a/examples/sessions/noisy-sensor-example/raw.log b/examples/sessions/noisy-sensor-example/raw.log index d9274d4..a2ed783 100644 --- a/examples/sessions/noisy-sensor-example/raw.log +++ b/examples/sessions/noisy-sensor-example/raw.log @@ -1,21 +1,101 @@ {"timestamp": 0.0, "value": 20.064409237657774} {"timestamp": 0.1, "value": 20.172305697081818} -{"timestamp": 0.2, "value": 20.129687135430082} -{"timestamp": 0.3, "value": 20.318178009797514} -{"timestamp": 0.4, "value": 20.312114475078108} -{"timestamp": 0.5, "value": 20.514173526862105} -{"timestamp": 0.6, "value": 20.66597393908556} -{"timestamp": 0.7, "value": 20.607930682461994} -{"timestamp": 0.8, "value": 20.69170829016585} -{"timestamp": 0.9, "value": 20.80704917255715} -{"timestamp": 1.0, "value": 20.805057052242127} -{"timestamp": 1.1, "value": 20.892165721512733} -{"timestamp": 1.2, "value": 21.026501286564805} -{"timestamp": 1.3, "value": 20.97639091199051} -{"timestamp": 1.4, "value": 21.00162200416873} -{"timestamp": 1.5, "value": 20.97793268730595} -{"timestamp": 1.6, "value": 20.91649199231581} -{"timestamp": 1.7, "value": 21.03591726892986} -{"timestamp": 1.8, "value": 21.00755856169906} -{"timestamp": 1.9, "value": 20.897489961463332} -{"timestamp": 2.0, "value": 20.882838266993875} +{"timestamp": 0.2, "value": 20.201986121241973} +{"timestamp": 0.3, "value": 20.25729302411276} +{"timestamp": 0.4, "value": 20.334809681553445} +{"timestamp": 0.5, "value": 20.48099226444579} +{"timestamp": 0.6, "value": 20.513537314894492} +{"timestamp": 0.7, "value": 20.572376214982565} +{"timestamp": 0.8, "value": 20.72732168972371} +{"timestamp": 0.9, "value": 20.789995639860415} +{"timestamp": 1.0, "value": 20.868794399824807} +{"timestamp": 1.1, "value": 20.84550881287467} +{"timestamp": 1.2, "value": 20.932289350148555} +{"timestamp": 1.3, "value": 20.96032109739856} +{"timestamp": 1.4, "value": 20.910158279925422} +{"timestamp": 1.5, "value": 21.024394845537106} +{"timestamp": 1.6, "value": 21.015609158091447} +{"timestamp": 1.7, "value": 21.111120412614504} +{"timestamp": 1.8, "value": 20.983996089743695} +{"timestamp": 1.9, "value": 20.93906497227917} +{"timestamp": 2.0, "value": 20.970935285577127} +{"timestamp": 2.1, "value": 20.873148929058544} +{"timestamp": 2.2, "value": 20.85394795512725} +{"timestamp": 2.3, "value": 20.727427998865092} +{"timestamp": 2.4, "value": 20.686371771219182} +{"timestamp": 2.5, "value": 20.64968657802831} +{"timestamp": 2.6, "value": 20.550313722944114} +{"timestamp": 2.7, "value": 20.433803492665472} +{"timestamp": 2.8, "value": 20.280872748926104} +{"timestamp": 2.9, "value": 20.261510417832035} +{"timestamp": 3.0, "value": 20.14496318214583} +{"timestamp": 3.1, "value": 20.077604042276253} +{"timestamp": 3.2, "value": 19.952437503462424} +{"timestamp": 3.3, "value": 19.896663576413527} +{"timestamp": 3.4, "value": 19.74188088470197} +{"timestamp": 3.5, "value": 19.659314975314352} +{"timestamp": 3.6, "value": 19.590818378636428} +{"timestamp": 3.7, "value": 19.415819628482783} +{"timestamp": 3.8, "value": 19.368059096053578} +{"timestamp": 3.9, "value": 19.287232412368407} +{"timestamp": 4.0, "value": 19.342228290668235} +{"timestamp": 4.1, "value": 19.177079787652783} +{"timestamp": 4.2, "value": 19.16103523877679} +{"timestamp": 4.3, "value": 19.11480281635923} +{"timestamp": 4.4, "value": 19.034354254601137} +{"timestamp": 4.5, "value": 18.94492804097838} +{"timestamp": 4.6, "value": 19.05455119490034} +{"timestamp": 4.7, "value": 18.979716903952717} +{"timestamp": 4.8, "value": 19.039733224001406} +{"timestamp": 4.9, "value": 18.95228414612017} +{"timestamp": 5.0, "value": 19.01917657523837} +{"timestamp": 5.1, "value": 19.137026384902533} +{"timestamp": 5.2, "value": 19.188095543682472} +{"timestamp": 5.3, "value": 19.10260962671918} +{"timestamp": 5.4, "value": 19.160595138489565} +{"timestamp": 5.5, "value": 19.292246452549517} +{"timestamp": 5.6, "value": 19.405145428293615} +{"timestamp": 5.7, "value": 19.457339691371327} +{"timestamp": 5.8, "value": 19.550575555977595} +{"timestamp": 5.9, "value": 19.576681513921034} +{"timestamp": 6.0, "value": 19.749924087491763} +{"timestamp": 6.1, "value": 19.873680112784747} +{"timestamp": 6.2, "value": 19.895126971168306} +{"timestamp": 6.3, "value": 19.945139497301177} +{"timestamp": 6.4, "value": 20.078608163666306} +{"timestamp": 6.5, "value": 20.25320288837966} +{"timestamp": 6.6, "value": 20.224856503921558} +{"timestamp": 6.7, "value": 20.400256081226317} +{"timestamp": 6.8, "value": 20.444562936881333} +{"timestamp": 6.9, "value": 20.571883042289315} +{"timestamp": 7.0, "value": 20.644760519381386} +{"timestamp": 7.1, "value": 20.72976187928949} +{"timestamp": 7.2, "value": 20.86872823366432} +{"timestamp": 7.3, "value": 20.87147214280896} +{"timestamp": 7.4, "value": 20.96539373307989} +{"timestamp": 7.5, "value": 20.930929341059567} +{"timestamp": 7.6, "value": 20.943940341640136} +{"timestamp": 7.7, "value": 21.0071087629903} +{"timestamp": 7.8, "value": 20.8567538060406} +{"timestamp": 7.9, "value": 20.99694690110517} +{"timestamp": 8.0, "value": 20.997366736077144} +{"timestamp": 8.1, "value": 20.908129373644407} +{"timestamp": 8.2, "value": 20.9639488178084} +{"timestamp": 8.3, "value": 20.874209593539945} +{"timestamp": 8.4, "value": 20.731643898574433} +{"timestamp": 8.5, "value": 20.787821193105092} +{"timestamp": 8.6, "value": 20.685454810659273} +{"timestamp": 8.7, "value": 20.636939436452177} +{"timestamp": 8.8, "value": 20.577302971956495} +{"timestamp": 8.9, "value": 20.563569623919577} +{"timestamp": 9.0, "value": 20.417275894188837} +{"timestamp": 9.1, "value": 20.31767408110017} +{"timestamp": 9.2, "value": 20.242340134908616} +{"timestamp": 9.3, "value": 20.03384980521583} +{"timestamp": 9.4, "value": 20.086781618984535} +{"timestamp": 9.5, "value": 19.87099453352337} +{"timestamp": 9.6, "value": 19.84762797233377} +{"timestamp": 9.7, "value": 19.671900348815626} +{"timestamp": 9.8, "value": 19.584696602517525} +{"timestamp": 9.9, "value": 19.522649712461103} +{"timestamp": 10.0, "value": 19.55076631224155} diff --git a/examples/sessions/noisy-sensor-example/records.jsonl b/examples/sessions/noisy-sensor-example/records.jsonl index d9274d4..c76c4de 100644 --- a/examples/sessions/noisy-sensor-example/records.jsonl +++ b/examples/sessions/noisy-sensor-example/records.jsonl @@ -1,21 +1,101 @@ -{"timestamp": 0.0, "value": 20.064409237657774} -{"timestamp": 0.1, "value": 20.172305697081818} -{"timestamp": 0.2, "value": 20.129687135430082} -{"timestamp": 0.3, "value": 20.318178009797514} -{"timestamp": 0.4, "value": 20.312114475078108} -{"timestamp": 0.5, "value": 20.514173526862105} -{"timestamp": 0.6, "value": 20.66597393908556} -{"timestamp": 0.7, "value": 20.607930682461994} -{"timestamp": 0.8, "value": 20.69170829016585} -{"timestamp": 0.9, "value": 20.80704917255715} -{"timestamp": 1.0, "value": 20.805057052242127} -{"timestamp": 1.1, "value": 20.892165721512733} -{"timestamp": 1.2, "value": 21.026501286564805} -{"timestamp": 1.3, "value": 20.97639091199051} -{"timestamp": 1.4, "value": 21.00162200416873} -{"timestamp": 1.5, "value": 20.97793268730595} -{"timestamp": 1.6, "value": 20.91649199231581} -{"timestamp": 1.7, "value": 21.03591726892986} -{"timestamp": 1.8, "value": 21.00755856169906} -{"timestamp": 1.9, "value": 20.897489961463332} -{"timestamp": 2.0, "value": 20.882838266993875} +{"timestamp":0.0,"value":20.064409237657774} +{"timestamp":0.1,"value":20.172305697081818} +{"timestamp":0.2,"value":20.201986121241973} +{"timestamp":0.3,"value":20.25729302411276} +{"timestamp":0.4,"value":20.334809681553445} +{"timestamp":0.5,"value":20.48099226444579} +{"timestamp":0.6,"value":20.513537314894492} +{"timestamp":0.7,"value":20.572376214982565} +{"timestamp":0.8,"value":20.72732168972371} +{"timestamp":0.9,"value":20.789995639860415} +{"timestamp":1.0,"value":20.868794399824807} +{"timestamp":1.1,"value":20.84550881287467} +{"timestamp":1.2,"value":20.932289350148555} +{"timestamp":1.3,"value":20.96032109739856} +{"timestamp":1.4,"value":20.910158279925422} +{"timestamp":1.5,"value":21.024394845537106} +{"timestamp":1.6,"value":21.015609158091447} +{"timestamp":1.7,"value":21.111120412614504} +{"timestamp":1.8,"value":20.983996089743695} +{"timestamp":1.9,"value":20.93906497227917} +{"timestamp":2.0,"value":20.970935285577127} +{"timestamp":2.1,"value":20.873148929058544} +{"timestamp":2.2,"value":20.85394795512725} +{"timestamp":2.3,"value":20.727427998865092} +{"timestamp":2.4,"value":20.686371771219182} +{"timestamp":2.5,"value":20.64968657802831} +{"timestamp":2.6,"value":20.550313722944114} +{"timestamp":2.7,"value":20.433803492665472} +{"timestamp":2.8,"value":20.280872748926104} +{"timestamp":2.9,"value":20.261510417832035} +{"timestamp":3.0,"value":20.14496318214583} +{"timestamp":3.1,"value":20.077604042276253} +{"timestamp":3.2,"value":19.952437503462424} +{"timestamp":3.3,"value":19.896663576413527} +{"timestamp":3.4,"value":19.74188088470197} +{"timestamp":3.5,"value":19.659314975314352} +{"timestamp":3.6,"value":19.590818378636428} +{"timestamp":3.7,"value":19.415819628482783} +{"timestamp":3.8,"value":19.368059096053578} +{"timestamp":3.9,"value":19.287232412368407} +{"timestamp":4.0,"value":19.342228290668235} +{"timestamp":4.1,"value":19.177079787652783} +{"timestamp":4.2,"value":19.16103523877679} +{"timestamp":4.3,"value":19.11480281635923} +{"timestamp":4.4,"value":19.034354254601137} +{"timestamp":4.5,"value":18.94492804097838} +{"timestamp":4.6,"value":19.05455119490034} +{"timestamp":4.7,"value":18.979716903952717} +{"timestamp":4.8,"value":19.039733224001406} +{"timestamp":4.9,"value":18.95228414612017} +{"timestamp":5.0,"value":19.01917657523837} +{"timestamp":5.1,"value":19.137026384902533} +{"timestamp":5.2,"value":19.188095543682472} +{"timestamp":5.3,"value":19.10260962671918} +{"timestamp":5.4,"value":19.160595138489565} +{"timestamp":5.5,"value":19.292246452549517} +{"timestamp":5.6,"value":19.405145428293615} +{"timestamp":5.7,"value":19.457339691371327} +{"timestamp":5.8,"value":19.550575555977595} +{"timestamp":5.9,"value":19.576681513921034} +{"timestamp":6.0,"value":19.749924087491763} +{"timestamp":6.1,"value":19.873680112784747} +{"timestamp":6.2,"value":19.895126971168306} +{"timestamp":6.3,"value":19.945139497301177} +{"timestamp":6.4,"value":20.078608163666306} +{"timestamp":6.5,"value":20.25320288837966} +{"timestamp":6.6,"value":20.224856503921558} +{"timestamp":6.7,"value":20.400256081226317} +{"timestamp":6.8,"value":20.444562936881333} +{"timestamp":6.9,"value":20.571883042289315} +{"timestamp":7.0,"value":20.644760519381386} +{"timestamp":7.1,"value":20.72976187928949} +{"timestamp":7.2,"value":20.86872823366432} +{"timestamp":7.3,"value":20.87147214280896} +{"timestamp":7.4,"value":20.96539373307989} +{"timestamp":7.5,"value":20.930929341059567} +{"timestamp":7.6,"value":20.943940341640136} +{"timestamp":7.7,"value":21.0071087629903} +{"timestamp":7.8,"value":20.8567538060406} +{"timestamp":7.9,"value":20.99694690110517} +{"timestamp":8.0,"value":20.997366736077144} +{"timestamp":8.1,"value":20.908129373644407} +{"timestamp":8.2,"value":20.9639488178084} +{"timestamp":8.3,"value":20.874209593539945} +{"timestamp":8.4,"value":20.731643898574433} +{"timestamp":8.5,"value":20.787821193105092} +{"timestamp":8.6,"value":20.685454810659273} +{"timestamp":8.7,"value":20.636939436452177} +{"timestamp":8.8,"value":20.577302971956495} +{"timestamp":8.9,"value":20.563569623919577} +{"timestamp":9.0,"value":20.417275894188837} +{"timestamp":9.1,"value":20.31767408110017} +{"timestamp":9.2,"value":20.242340134908616} +{"timestamp":9.3,"value":20.03384980521583} +{"timestamp":9.4,"value":20.086781618984535} +{"timestamp":9.5,"value":19.87099453352337} +{"timestamp":9.6,"value":19.84762797233377} +{"timestamp":9.7,"value":19.671900348815626} +{"timestamp":9.8,"value":19.584696602517525} +{"timestamp":9.9,"value":19.522649712461103} +{"timestamp":10.0,"value":19.55076631224155} diff --git a/examples/sessions/noisy-sensor-example/reports/report.md b/examples/sessions/noisy-sensor-example/reports/report.md new file mode 100644 index 0000000..6995a4d --- /dev/null +++ b/examples/sessions/noisy-sensor-example/reports/report.md @@ -0,0 +1,87 @@ +# Datary report: noisy\-sensor\-example + +- Datary version: `0.2.0` +- Recorded with Datary: `0.2.0` +- Session format: `2` +- Started: `2026-07-29T19:37:50.019613+08:00` +- Ended: `2026-07-29T19:37:50.055197+08:00` +- Input format: `jsonl` +- Records: 101 valid, 0 invalid + +## Reproduction + +- Original command: `not supplied` +- Working directory: `` +- Command context: Run from the session parent directory or set DATARY\_WORKSPACE to that directory\. +- compare: `datary compare noisy-sensor-example OTHER` +- inspect: `datary inspect noisy-sensor-example` +- replay: `datary replay noisy-sensor-example` +- report: `datary report noisy-sensor-example` + +## Data schema + +| Field | Type | Unit | +|---|---|---| +| timestamp | number | | +| value | number | | + +## Descriptive statistics + +### timestamp + +- Mean: 4.999999999999993 +- Median: 5.0 +- Range: 0.0 to 10.0 + +### value + +- Mean: 20.17767606050936 +- Median: 20.261510417832035 +- Range: 18.94492804097838 to 21.111120412614504 + +## Timing + +- backward\_timestamp\_count: `0` +- duplicate\_timestamp\_count: `0` +- duration: `10.0` +- effective\_sample\_rate: `9.999999999999988` +- end: `10.0` +- gap\_count: `0` +- jitter: `4.3390452126414056e-16` +- maximum\_interval: `0.10000000000000142` +- mean\_interval: `0.10000000000000013` +- median\_interval: `0.09999999999999998` +- minimum\_interval: `0.09999999999999964` +- start: `0.0` + +## Engineering metrics + +No control or network field roles were supplied. + +## Quality findings + +No heuristic findings. + +## Input hashes + +- `data.csv`: `593777e7fbce3416989aba7f20aa1865b5a9535d93456db0f5dd3f600cc7f518` +- `invalid.jsonl`: `e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855` +- `metrics.json`: `70263c17f02120c8f4e04285025aa8d5e43b71bd7034030df7fc808fc211cd65` +- `notes.md`: `a510c5cf6eb3f60d592443e63d03bcda952e157eea94c9cd76a4695426cb51e9` +- `quality.json`: `8aa973100a15bf817985395637131e838cf84e427be4792583def923f020f5fa` +- `raw.log`: `d22ef77cba7e4330d271280e85dc7d1d8a852b6a731b5443ca436d5fd2d65d8e` +- `records.jsonl`: `9ccd2b842825bf7788c36e7e546a04a3b582cae386903dd1171b08816fc3f31e` + +## Integrity verification + +All manifest-listed artefacts passed SHA-256 corruption checks. +These checks detect accidental changes; they do not prove cryptographic authenticity. + +## Warnings and assumptions + +- Statistics describe recorded data; they do not establish scientific validity\. +- Quality checks are heuristics and require domain review\. + +## Plots + +- [plot\-value\.png](../plots/plot-value.png) diff --git a/examples/simulations/motor_sim.py b/examples/simulations/motor_sim.py index 3ee9012..b5f2465 100644 --- a/examples/simulations/motor_sim.py +++ b/examples/simulations/motor_sim.py @@ -1,4 +1,5 @@ """Tiny deterministic motor simulation for Datary examples.""" + import json import math diff --git a/pyproject.toml b/pyproject.toml index aaedae3..f482e29 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -4,7 +4,7 @@ build-backend = "hatchling.build" [project] name = "datary-lab" -version = "0.1.3" +version = "0.2.0" description = "A local-first terminal laboratory for reproducible program data." readme = "README.md" requires-python = ">=3.9" @@ -13,8 +13,18 @@ authors = [{name = "devkyato"}] keywords = ["data", "terminal", "experiments", "reproducibility"] classifiers = [ "Development Status :: 3 - Alpha", + "Environment :: Console", + "Intended Audience :: Developers", "License :: OSI Approved :: MIT License", + "Operating System :: OS Independent", "Programming Language :: Python :: 3", + "Programming Language :: Python :: 3.9", + "Programming Language :: Python :: 3.10", + "Programming Language :: Python :: 3.11", + "Programming Language :: Python :: 3.12", + "Programming Language :: Python :: 3.13", + "Programming Language :: Python :: 3.14", + "Typing :: Typed", ] dependencies = [] @@ -45,7 +55,7 @@ packages = ["src/datary"] include = ["/src", "/tests", "/docs", "/examples", "/scripts", "/README.md", "/LICENSE", "/CHANGELOG.md", "/CONTRIBUTING.md", "/SECURITY.md", "/pyproject.toml"] [tool.pytest.ini_options] -addopts = "-q" +addopts = "-q --cov=datary --cov-report=term-missing --cov-fail-under=75" testpaths = ["tests"] [tool.ruff] @@ -58,10 +68,17 @@ select = ["E", "F", "I", "B"] ignore = ["E501"] [tool.mypy] -python_version = "3.10" +python_version = "3.9" strict = true -files = ["src/datary", "scripts"] +files = ["src/datary", "scripts", "tests"] warn_unreachable = true +follow_imports = "normal" + +[[tool.mypy.overrides]] +module = ["pytest", "_pytest", "_pytest.*"] follow_imports = "skip" -no_site_packages = true ignore_missing_imports = true + +[[tool.mypy.overrides]] +module = ["test_formats", "test_parsers"] +disallow_untyped_decorators = false diff --git a/scripts/build_checksums.py b/scripts/build_checksums.py index 6dbfb2f..3cb934f 100644 --- a/scripts/build_checksums.py +++ b/scripts/build_checksums.py @@ -1,4 +1,5 @@ """Write SHA-256 checksums for the current version's distribution artifacts.""" + import hashlib import re from pathlib import Path @@ -19,10 +20,7 @@ def main() -> int: artifacts = [path for path in artifacts if path.is_file()] if not artifacts: raise FileNotFoundError(f"no distribution artifacts found for version {version}") - lines = [ - f"{hashlib.sha256(path.read_bytes()).hexdigest()} {path.name}" - for path in artifacts - ] + lines = [f"{sha256_file(path)} {path.name}" for path in artifacts] (distribution / "SHA256SUMS").write_text( "\n".join(lines) + "\n", encoding="utf-8", @@ -30,5 +28,13 @@ def main() -> int: return 0 +def sha256_file(path: Path) -> str: + digest = hashlib.sha256() + with path.open("rb") as stream: + for chunk in iter(lambda: stream.read(1024 * 1024), b""): + digest.update(chunk) + return digest.hexdigest() + + if __name__ == "__main__": raise SystemExit(main()) diff --git a/scripts/build_release.py b/scripts/build_release.py index c7ffefe..4751e5e 100644 --- a/scripts/build_release.py +++ b/scripts/build_release.py @@ -1,6 +1,6 @@ """Build wheel and source distribution.""" + import subprocess import sys raise SystemExit(subprocess.call([sys.executable, "-m", "build"])) - diff --git a/scripts/verify_reproducible.py b/scripts/verify_reproducible.py index 77fda83..c357169 100644 --- a/scripts/verify_reproducible.py +++ b/scripts/verify_reproducible.py @@ -1,4 +1,5 @@ """Verify deterministic generator output in separate calls.""" + import subprocess import sys @@ -8,4 +9,3 @@ if first != second: raise SystemExit("generator output differs") print("deterministic") - diff --git a/src/datary/__init__.py b/src/datary/__init__.py index e53282b..4eaeee9 100644 --- a/src/datary/__init__.py +++ b/src/datary/__init__.py @@ -5,4 +5,4 @@ from datary.sessions import Session __all__ = ["Session", "compare_sessions", "inspect_source"] -__version__ = "0.1.3" +__version__ = "0.2.0" diff --git a/src/datary/__main__.py b/src/datary/__main__.py index b6a5df9..bdcaf2a 100644 --- a/src/datary/__main__.py +++ b/src/datary/__main__.py @@ -1,4 +1,3 @@ from datary.cli import main raise SystemExit(main()) - diff --git a/src/datary/analysis_store.py b/src/datary/analysis_store.py new file mode 100644 index 0000000..fc1c709 --- /dev/null +++ b/src/datary/analysis_store.py @@ -0,0 +1,1235 @@ +"""Disk-backed analysis used to keep recording and inspection memory bounded.""" + +from __future__ import annotations + +import json +import math +import sqlite3 +from pathlib import Path +from typing import Any, Dict, Iterator, List, Optional, Sequence, Tuple + +from datary.models import Finding, Record +from datary.utils import temporal_number + +_PERCENTILES = (5, 25, 50, 75, 95, 99) + + +class AnalysisStore: + """A temporary SQLite spool for exact, bounded-memory analysis.""" + + def __init__(self, path: Path, max_fields: int = 1000) -> None: + self.path = path + self.max_fields = max_fields + self.connection = sqlite3.connect(str(path)) + self.connection.executescript( + """ + PRAGMA journal_mode=OFF; + PRAGMA synchronous=OFF; + PRAGMA temp_store=FILE; + CREATE TABLE records ( + record_index INTEGER PRIMARY KEY, + payload TEXT NOT NULL, + schema_key TEXT NOT NULL, + field_count INTEGER NOT NULL + ); + CREATE TABLE cells ( + record_index INTEGER NOT NULL, + field TEXT NOT NULL, + value_kind TEXT NOT NULL, + text_value TEXT, + numeric_value REAL, + PRIMARY KEY (record_index, field) + ); + CREATE INDEX cells_field_record ON cells(field, record_index); + CREATE INDEX cells_field_number ON cells(field, numeric_value); + CREATE TEMP TABLE scratch (position INTEGER, value REAL); + """ + ) + self._fields: set[str] = set() + self._count = 0 + + def __enter__(self) -> "AnalysisStore": + return self + + def __exit__(self, *_args: object) -> None: + self.close() + + @property + def count(self) -> int: + return self._count + + @property + def fields(self) -> List[str]: + return sorted(self._fields) + + def add(self, record: Record) -> None: + new_fields = set(record) - self._fields + if len(self._fields | new_fields) > self.max_fields: + raise ValueError( + f"dataset exceeds unique field limit ({self.max_fields}); " + "use a larger --max-fields only for trusted input" + ) + self._fields.update(new_fields) + index = self._count + payload = json.dumps( + record, + ensure_ascii=False, + sort_keys=True, + separators=(",", ":"), + allow_nan=False, + ) + schema_key = json.dumps(sorted(record), ensure_ascii=False, separators=(",", ":")) + self.connection.execute( + "INSERT INTO records VALUES (?, ?, ?, ?)", + (index, payload, schema_key, len(record)), + ) + for field, value in record.items(): + kind, text_value, numeric_value = _cell_parts(value) + self.connection.execute( + "INSERT INTO cells VALUES (?, ?, ?, ?, ?)", + (index, field, kind, text_value, numeric_value), + ) + self._count += 1 + if self._count % 1000 == 0: + self.connection.commit() + + def finish(self) -> None: + self.connection.commit() + + def records(self) -> Iterator[Record]: + cursor = self.connection.execute("SELECT payload FROM records ORDER BY record_index") + for (payload,) in cursor: + value = json.loads(str(payload)) + if isinstance(value, dict): + yield value + + def field_definitions(self) -> Dict[str, str]: + result: Dict[str, str] = {} + for field in self.fields: + kinds = { + str(row[0]) + for row in self.connection.execute( + "SELECT DISTINCT value_kind FROM cells " + "WHERE field = ? AND value_kind != 'null'", + (field,), + ) + } + if not kinds: + result[field] = "null" + elif kinds <= {"int"}: + result[field] = "integer" + elif kinds <= {"int", "float"}: + result[field] = "number" + elif len(kinds) == 1: + result[field] = next(iter(kinds)) + else: + result[field] = "mixed" + return result + + def metrics(self, time_field: Optional[str] = None) -> Dict[str, Any]: + numeric: Dict[str, Any] = {} + for field in self.fields: + summary = self._numeric_summary(field) + if summary["valid_count"]: + numeric[field] = summary + timing = self._timing_metrics(time_field) if time_field in self._fields else {} + return {"numeric": numeric, "timing": timing} + + def quality( + self, + time_field: Optional[str] = None, + sequence_field: Optional[str] = None, + monotonic_fields: Optional[Sequence[str]] = None, + counter_fields: Optional[Sequence[str]] = None, + ) -> List[Finding]: + if not self._count: + return [ + _finding( + "empty-data", + "warning", + None, + "all", + 0, + "> 0 records", + "No valid records were parsed.", + ) + ] + findings: List[Finding] = [] + duplicate_row = self.connection.execute( + "SELECT COALESCE(SUM(amount - 1), 0) FROM " + "(SELECT COUNT(*) AS amount FROM records GROUP BY payload HAVING amount > 1)" + ).fetchone() + duplicates = int(duplicate_row[0]) if duplicate_row else 0 + if duplicates: + findings.append( + _finding( + "duplicate-rows", + "warning", + None, + "multiple", + duplicates, + 0, + "Identical records recur.", + ) + ) + schemas = [ + json.loads(str(row[0])) + for row in self.connection.execute( + "SELECT DISTINCT schema_key FROM records ORDER BY schema_key" + ) + ] + if len(schemas) > 1: + findings.append( + _finding( + "record-shape-change", + "warning", + None, + "multiple", + schemas, + "one stable field set", + "Record field names change across the dataset.", + ) + ) + lengths = [ + int(row[0]) + for row in self.connection.execute( + "SELECT DISTINCT field_count FROM records ORDER BY field_count" + ) + ] + if len(lengths) > 1: + findings.append( + _finding( + "record-length-change", + "warning", + None, + "multiple", + lengths, + 1, + "Record field counts vary.", + ) + ) + monotonic = set(monotonic_fields or ()) + counters = set(counter_fields or ()) + for field in self.fields: + findings.extend( + self._field_quality( + field, + field in monotonic, + field in counters, + field in {time_field, sequence_field}, + ) + ) + if time_field and time_field in self._fields: + findings.extend(self._timing_quality(time_field)) + if sequence_field and sequence_field in self._fields: + findings.extend(self._sequence_quality(sequence_field)) + return sorted(findings, key=lambda item: (item.check_id, item.field or "", item.affected)) + + def control_metrics( + self, + time_field: str, + target_field: str, + response_field: str, + ) -> Dict[str, Any]: + """Calculate step-response metrics in bounded memory.""" + + initial_response: Optional[float] = None + final_target: Optional[float] = None + target_min = math.inf + target_max = -math.inf + valid_count = 0 + for record in self.records(): + time_value = temporal_number(record.get(time_field)) + target = _finite(record.get(target_field)) + response = _finite(record.get(response_field)) + if time_value is None or target is None or response is None: + continue + if initial_response is None: + initial_response = response + final_target = target + target_min = min(target_min, target) + target_max = max(target_max, target) + valid_count += 1 + if valid_count < 2 or initial_response is None or final_target is None: + return {} + span = final_target - initial_response + stable_target = target_max - target_min <= max(abs(final_target), 1.0) * 1e-9 + tolerance = max(abs(span), abs(final_target), 1e-12) * 0.02 + rise_10: Optional[float] = None + rise_90: Optional[float] = None + peak: Optional[float] = None + first_time: Optional[float] = None + settling_candidate: Optional[float] = None + previous_time: Optional[float] = None + previous_error: Optional[float] = None + absolute_error_sum = 0.0 + squared_error_sum = 0.0 + integral_absolute = 0.0 + integral_squared = 0.0 + final_error = 0.0 + backwards = 0 + for record in self.records(): + time_value = temporal_number(record.get(time_field)) + target = _finite(record.get(target_field)) + response = _finite(record.get(response_field)) + if time_value is None or target is None or response is None: + continue + if first_time is None: + first_time = time_value + error = target - response + final_error = error + absolute_error_sum += abs(error) + squared_error_sum += error * error + peak = ( + response + if peak is None + else (max(peak, response) if span >= 0 else min(peak, response)) + ) + if stable_target: + if rise_10 is None and _crossed(response, initial_response + span * 0.1, span): + rise_10 = time_value + if rise_90 is None and _crossed(response, initial_response + span * 0.9, span): + rise_90 = time_value + if abs(response - final_target) <= tolerance: + if settling_candidate is None: + settling_candidate = time_value + else: + settling_candidate = None + if previous_time is not None and previous_error is not None: + delta_time = time_value - previous_time + if delta_time >= 0: + integral_absolute += delta_time * (abs(previous_error) + abs(error)) / 2 + integral_squared += ( + delta_time * (previous_error * previous_error + error * error) / 2 + ) + else: + backwards += 1 + previous_time = time_value + previous_error = error + assert peak is not None + overshoot: Optional[float] + if span and stable_target: + overshoot = ( + (peak - final_target) / abs(span) * 100 + if span >= 0 + else (final_target - peak) / abs(span) * 100 + ) + else: + overshoot = None + warnings: List[str] = [] + if not stable_target: + warnings.append( + "The target changes during the record; step-response metrics are not reported." + ) + if backwards: + warnings.append( + f"{backwards} backwards time interval(s) were excluded from integration." + ) + return { + "rise_time": ( + rise_90 - rise_10 if rise_10 is not None and rise_90 is not None else None + ), + "peak_value": peak, + "percentage_overshoot": overshoot, + "settling_time": ( + settling_candidate - first_time + if settling_candidate is not None and first_time is not None + else None + ), + "steady_state_error": final_error, + "mean_absolute_error": absolute_error_sum / valid_count, + "root_mean_squared_error": math.sqrt(squared_error_sum / valid_count), + "integral_absolute_error": integral_absolute, + "integral_squared_error": integral_squared, + "required_fields": { + "time": time_field, + "target": target_field, + "response": response_field, + }, + "assumptions": [ + "Rise time, overshoot, and settling time require a stable step target.", + "Rise-time crossings use recorded samples without interpolation.", + "Settling uses a ±2% band based on the larger of step span or target magnitude.", + "Error integrals use trapezoidal integration over non-decreasing timestamps.", + ], + "warnings": warnings, + } + + def network_metrics( + self, + sequence_field: str, + latency_field: str, + bytes_field: Optional[str] = None, + time_field: Optional[str] = None, + ) -> Dict[str, Any]: + self.connection.execute("DELETE FROM scratch") + sequence_count = 0 + sequence_min: Optional[int] = None + sequence_max: Optional[int] = None + previous: Optional[int] = None + out_of_order = 0 + invalid_sequences = 0 + sequence_batch: List[Tuple[int, float]] = [] + for index, value in self._numeric_rows(sequence_field): + if not value.is_integer(): + invalid_sequences += 1 + continue + sequence = int(value) + sequence_count += 1 + sequence_batch.append((index, float(sequence))) + if len(sequence_batch) >= 1000: + self.connection.executemany( + "INSERT INTO scratch(position, value) VALUES (?, ?)", + sequence_batch, + ) + sequence_batch.clear() + sequence_min = sequence if sequence_min is None else min(sequence_min, sequence) + sequence_max = sequence if sequence_max is None else max(sequence_max, sequence) + if previous is not None and sequence < previous: + out_of_order += 1 + previous = sequence + if sequence_batch: + self.connection.executemany( + "INSERT INTO scratch(position, value) VALUES (?, ?)", + sequence_batch, + ) + distinct_row = self.connection.execute( + "SELECT COUNT(DISTINCT value) FROM scratch" + ).fetchone() + distinct = int(distinct_row[0]) if distinct_row else 0 + expected = ( + sequence_max - sequence_min + 1 + if sequence_min is not None and sequence_max is not None + else 0 + ) + self.connection.execute("DELETE FROM scratch") + latency_count = 0 + latency_mean = 0.0 + previous_latency: Optional[float] = None + latency_difference_total = 0.0 + latency_difference_count = 0 + negative_latencies = 0 + latency_batch: List[Tuple[int, float]] = [] + for index, value in self._numeric_rows(latency_field): + if value < 0: + negative_latencies += 1 + continue + latency_count += 1 + latency_mean += (value - latency_mean) / latency_count + if previous_latency is not None: + latency_difference_total += abs(value - previous_latency) + latency_difference_count += 1 + previous_latency = value + latency_batch.append((index, value)) + if len(latency_batch) >= 1000: + self.connection.executemany( + "INSERT INTO scratch(position, value) VALUES (?, ?)", + latency_batch, + ) + latency_batch.clear() + if latency_batch: + self.connection.executemany( + "INSERT INTO scratch(position, value) VALUES (?, ?)", + latency_batch, + ) + median_latency = self._scratch_percentile(latency_count, 50) if latency_count else None + percentile_latency = ( + { + str(percentile): self._scratch_percentile(latency_count, percentile) + for percentile in _PERCENTILES + } + if latency_count + else {} + ) + warnings = [ + warning + for warning in ( + ( + f"{invalid_sequences} non-integer sequence value(s) were excluded." + if invalid_sequences + else "" + ), + ( + f"{negative_latencies} negative latency value(s) were excluded." + if negative_latencies + else "" + ), + ) + if warning + ] + result: Dict[str, Any] = { + "packet_loss_estimate": ((expected - distinct) / expected if expected else None), + "duplicate_packet_rate": ( + (sequence_count - distinct) / sequence_count if sequence_count else None + ), + "out_of_order_count": out_of_order, + "mean_latency": latency_mean if latency_count else None, + "median_latency": median_latency, + "latency_jitter": ( + latency_difference_total / latency_difference_count + if latency_difference_count + else 0.0 + ), + "percentile_latency": percentile_latency, + "required_fields": { + "sequence": sequence_field, + "latency": latency_field, + "bytes": bytes_field, + "time": time_field, + }, + "assumptions": [ + "Sequence identifiers are expected to be contiguous integers.", + "Latency jitter is the mean absolute difference between consecutive valid samples.", + ], + "warnings": warnings, + } + if bytes_field: + total_bytes = 0.0 + negative_bytes = 0 + for _, value in self._numeric_rows(bytes_field): + if value < 0: + negative_bytes += 1 + else: + total_bytes += value + if negative_bytes: + warnings.append(f"{negative_bytes} negative byte count(s) were excluded.") + result["total_bytes"] = total_bytes + if time_field: + first_time: Optional[float] = None + last_time: Optional[float] = None + for _, value in self._time_rows(time_field): + if first_time is None: + first_time = value + last_time = value + duration = ( + last_time - first_time + if first_time is not None and last_time is not None + else 0.0 + ) + result["throughput_bytes_per_second"] = ( + float(total_bytes) / duration + if total_bytes is not None and duration > 0 + else None + ) + return result + + def close(self) -> None: + self.connection.close() + + def _numeric_summary(self, field: str) -> Dict[str, Any]: + values = self._numeric_rows(field) + count = 0 + mean = 0.0 + m2 = 0.0 + total = 0.0 + total_compensation = 0.0 + square_total = 0.0 + square_compensation = 0.0 + first: Optional[float] = None + last: Optional[float] = None + previous: Optional[float] = None + difference_total = 0.0 + difference_compensation = 0.0 + difference_count = 0 + for _, value in values: + count += 1 + if first is None: + first = value + last = value + delta = value - mean + mean += delta / count + m2 += delta * (value - mean) + total, total_compensation = _compensated_add(total, total_compensation, value) + square_total, square_compensation = _compensated_add( + square_total, square_compensation, value * value + ) + if previous is not None: + difference_total, difference_compensation = _compensated_add( + difference_total, + difference_compensation, + abs(value - previous), + ) + difference_count += 1 + previous = value + if not count: + return { + "count": self._count, + "valid_count": 0, + "missing_count": self._count, + } + minimum, maximum = self.connection.execute( + "SELECT MIN(numeric_value), MAX(numeric_value) FROM cells " + "WHERE field = ? AND numeric_value IS NOT NULL", + (field,), + ).fetchone() + return { + "count": self._count, + "valid_count": count, + "missing_count": self._count - count, + "minimum": float(minimum), + "maximum": float(maximum), + "mean": mean, + "median": self._percentile(field, count, 50), + "standard_deviation": math.sqrt(m2 / (count - 1)) if count > 1 else 0.0, + "variance": m2 / (count - 1) if count > 1 else 0.0, + "percentiles": { + str(percentile): self._percentile(field, count, percentile) + for percentile in _PERCENTILES + }, + "sum": total + total_compensation, + "rate_of_change": (last - first) if first is not None and last is not None else 0.0, + "root_mean_square": math.sqrt((square_total + square_compensation) / count), + "mean_absolute_difference": ( + (difference_total + difference_compensation) / difference_count + if difference_count + else 0.0 + ), + "sparkline_values": self._ordered_samples(field, count, 40), + } + + def _percentile(self, field: str, count: int, percentile: int) -> float: + index = (count - 1) * percentile / 100 + low = math.floor(index) + high = math.ceil(index) + low_value = self._ordered_value(field, low) + if low == high: + return low_value + high_value = self._ordered_value(field, high) + return low_value * (high - index) + high_value * (index - low) + + def _ordered_value(self, field: str, offset: int) -> float: + row = self.connection.execute( + "SELECT numeric_value FROM cells " + "WHERE field = ? AND numeric_value IS NOT NULL " + "ORDER BY numeric_value LIMIT 1 OFFSET ?", + (field, offset), + ).fetchone() + if row is None: + raise ValueError("numeric percentile requested for empty field") + return float(row[0]) + + def _ordered_samples(self, field: str, count: int, maximum: int) -> List[float]: + sample_count = min(count, maximum) + offsets = ( + [0] + if sample_count == 1 + else [round(index * (count - 1) / (sample_count - 1)) for index in range(sample_count)] + ) + values: List[float] = [] + for offset in offsets: + row = self.connection.execute( + "SELECT numeric_value FROM cells " + "WHERE field = ? AND numeric_value IS NOT NULL " + "ORDER BY record_index LIMIT 1 OFFSET ?", + (field, offset), + ).fetchone() + if row is not None: + values.append(float(row[0])) + return values + + def _numeric_rows(self, field: str) -> Iterator[Tuple[int, float]]: + cursor = self.connection.execute( + "SELECT record_index, numeric_value FROM cells " + "WHERE field = ? AND numeric_value IS NOT NULL ORDER BY record_index", + (field,), + ) + for record_index, value in cursor: + yield int(record_index), float(value) + + def _timing_metrics(self, field: str) -> Dict[str, Any]: + self.connection.execute("DELETE FROM scratch") + self.connection.executemany( + "INSERT INTO scratch(position, value) VALUES (?, ?)", + self._time_rows(field), + ) + time_counts = self.connection.execute( + "SELECT COUNT(*), COUNT(DISTINCT value) FROM scratch" + ).fetchone() + duplicate_count = int(time_counts[0]) - int(time_counts[1]) + self.connection.execute("DELETE FROM scratch") + interval_count = 0 + positive_count = 0 + positive_mean = 0.0 + positive_m2 = 0.0 + minimum = math.inf + maximum = -math.inf + backward_count = 0 + batch: List[Tuple[int, float]] = [] + for index, interval in self._intervals(field): + interval_count += 1 + minimum = min(minimum, interval) + maximum = max(maximum, interval) + if interval > 0: + positive_count += 1 + delta = interval - positive_mean + positive_mean += delta / positive_count + positive_m2 += delta * (interval - positive_mean) + batch.append((index, interval)) + if len(batch) >= 1000: + self.connection.executemany( + "INSERT INTO scratch(position, value) VALUES (?, ?)", batch + ) + batch.clear() + elif interval < 0: + backward_count += 1 + if batch: + self.connection.executemany("INSERT INTO scratch(position, value) VALUES (?, ?)", batch) + start: Optional[float] = None + end: Optional[float] = None + for _, value in self._time_rows(field): + if start is None: + start = value + end = value + if start is None or end is None: + return {} + if not interval_count: + return { + "start": start, + "end": end, + "duration": 0.0, + "mean_interval": None, + "median_interval": None, + "minimum_interval": None, + "maximum_interval": None, + "jitter": 0.0, + "effective_sample_rate": None, + "gap_count": 0, + "duplicate_timestamp_count": 0, + "backward_timestamp_count": 0, + } + median = self._scratch_percentile(positive_count, 50) + return { + "start": start, + "end": end, + "duration": end - start, + "mean_interval": positive_mean, + "median_interval": median, + "minimum_interval": minimum, + "maximum_interval": maximum, + "jitter": (math.sqrt(positive_m2 / positive_count) if positive_count > 1 else 0.0), + "effective_sample_rate": (1.0 / positive_mean if positive_mean > 0 else None), + "gap_count": ( + int( + self.connection.execute( + "SELECT COUNT(*) FROM scratch WHERE value >= ?", + (median * 2,), + ).fetchone()[0] + ) + if median > 0 + else 0 + ), + "duplicate_timestamp_count": duplicate_count, + "backward_timestamp_count": backward_count, + } + + def _field_quality( + self, + field: str, + monotonic: bool, + counter: bool, + is_time: bool, + ) -> List[Finding]: + findings: List[Finding] = [] + missing = _Affected() + cursor = self.connection.execute( + "SELECT records.record_index, cells.value_kind, cells.text_value " + "FROM records LEFT JOIN cells ON records.record_index = cells.record_index " + "AND cells.field = ? ORDER BY records.record_index", + (field,), + ) + for index, kind, text_value in cursor: + if kind is None or kind == "null" or (kind == "str" and text_value == ""): + missing.add(int(index)) + if missing.count: + findings.append( + _finding( + "missing-values", + "warning", + field, + missing.range, + missing.count, + 0, + "Values are absent.", + ) + ) + kinds = { + str(row[0]) + for row in self.connection.execute( + "SELECT DISTINCT value_kind FROM cells WHERE field = ? AND value_kind != 'null'", + (field,), + ) + } + if len(kinds) > 1 and not kinds <= {"int", "float"}: + findings.append( + _finding( + "type-change", + "warning", + field, + "multiple", + sorted(kinds), + 1, + "The field changes type.", + ) + ) + unsafe_integer_row = self.connection.execute( + "SELECT COUNT(*), MIN(record_index), MAX(record_index) FROM cells " + "WHERE field = ? AND value_kind = 'int' AND numeric_value IS NULL", + (field,), + ).fetchone() + unsafe_integer_count = int(unsafe_integer_row[0]) + if unsafe_integer_count: + first = int(unsafe_integer_row[1]) + last = int(unsafe_integer_row[2]) + findings.append( + _finding( + "invalid-values", + "warning", + field, + str(first) if first == last else f"{first}-{last}", + unsafe_integer_count, + "integer magnitude ≤ 2^53 for exact binary64 analysis", + "Large integers were preserved but excluded from numeric analysis.", + ) + ) + if is_time: + return findings + rows = self._numeric_rows(field) + first_row: Optional[Tuple[int, float]] = None + previous: Optional[Tuple[int, float]] = None + minimum = math.inf + maximum = -math.inf + count = 0 + decreases = _Affected() + best_start = best_end = current_start = 0 + best_length = current_length = 0 + for index, value in rows: + count += 1 + if first_row is None: + first_row = (index, value) + current_start = index + current_length = 1 + elif previous is not None: + if value < previous[1]: + decreases.add(index) + if value == previous[1] and index == previous[0] + 1: + current_length += 1 + else: + current_start = index + current_length = 1 + if current_length > best_length: + best_start, best_end, best_length = current_start, index, current_length + minimum = min(minimum, value) + maximum = max(maximum, value) + previous = (index, value) + if monotonic and decreases.count: + findings.append( + _finding( + "monotonicity-violation", + "warning", + field, + decreases.range, + decreases.count, + "no decreases", + "The field decreases despite an explicit monotonicity expectation.", + ["Missing values are ignored between consecutive valid values."], + "Confirm ordering and whether resets or wraparound are expected.", + ) + ) + if counter and decreases.count: + findings.append( + _finding( + "counter-reset", + "warning", + field, + decreases.range, + decreases.count, + "no decreases", + "The counter decreases, indicating a reset, rollover, or reordered record.", + ["The selected field is expected to be a non-decreasing counter."], + "Check device restarts, counter width, wraparound, and record ordering.", + ) + ) + if count >= 2 and minimum == maximum and first_row and previous: + findings.append( + _finding( + "constant-signal", + "info", + field, + f"{first_row[0]}-{previous[0]}", + minimum, + "no variation", + "The signal is constant.", + ) + ) + if best_length >= 5 and best_length < count: + findings.append( + _finding( + "frozen-values", + "warning", + field, + f"{best_start}-{best_end}", + best_length, + 5, + "The same value repeats for an extended run.", + ["Missing values break a frozen run."], + ) + ) + if count >= 5: + findings.extend(self._distribution_quality(field, count)) + return findings + + def _distribution_quality(self, field: str, count: int) -> List[Finding]: + findings: List[Finding] = [] + median = self._percentile(field, count, 50) + self.connection.execute("DELETE FROM scratch") + self.connection.executemany( + "INSERT INTO scratch(position, value) VALUES (?, ?)", + ((index, abs(value - median)) for index, value in self._numeric_rows(field)), + ) + deviation_count = int(self.connection.execute("SELECT COUNT(*) FROM scratch").fetchone()[0]) + mad = self._scratch_percentile(deviation_count, 50) + outliers = _Affected() + for index, value in self._numeric_rows(field): + if abs(value - median) > 6 * mad if mad > 0 else value != median: + outliers.add(index) + if outliers.count: + threshold = "6 × MAD" if mad > 0 else "different from zero-MAD median" + findings.append( + _finding( + "outliers", + "warning", + field, + outliers.range, + outliers.count, + threshold, + "Values are far from the median.", + ["Robust median absolute deviation rule."], + ) + ) + self.connection.execute("DELETE FROM scratch") + previous: Optional[float] = None + differences: List[Tuple[int, float]] = [] + for index, value in self._numeric_rows(field): + if previous is not None: + differences.append((index, abs(value - previous))) + previous = value + if len(differences) >= 1000: + self.connection.executemany( + "INSERT INTO scratch(position, value) VALUES (?, ?)", + differences, + ) + differences.clear() + if differences: + self.connection.executemany( + "INSERT INTO scratch(position, value) VALUES (?, ?)", + differences, + ) + difference_count = int( + self.connection.execute("SELECT COUNT(*) FROM scratch").fetchone()[0] + ) + base = self._scratch_percentile(difference_count, 50) + spikes = _Affected() + for index, value in self.connection.execute( + "SELECT position, value FROM scratch ORDER BY position" + ): + if value > base * 10 if base > 0 else value > 0: + spikes.add(int(index)) + if spikes.count: + threshold = ( + "10 × median absolute step" if base > 0 else "non-zero step after zero baseline" + ) + findings.append( + _finding( + "sudden-spikes", + "warning", + field, + spikes.range, + spikes.count, + threshold, + "Abrupt changes exceed the configured robust threshold.", + ) + ) + summary = self._numeric_summary(field) + mean = float(summary["mean"]) + deviation = float(summary["standard_deviation"]) + if mean and deviation / abs(mean) > 0.5: + findings.append( + _finding( + "high-noise", + "info", + field, + "all", + deviation, + "coefficient of variation > 0.5", + "Variation is high relative to the mean.", + ["Only meaningful for ratio-scale signals."], + ) + ) + return findings + + def _scratch_percentile(self, count: int, percentile: int) -> float: + if not count: + return 0.0 + index = (count - 1) * percentile / 100 + low = math.floor(index) + high = math.ceil(index) + low_value = float( + self.connection.execute( + "SELECT value FROM scratch ORDER BY value LIMIT 1 OFFSET ?", + (low,), + ).fetchone()[0] + ) + if low == high: + return low_value + high_value = float( + self.connection.execute( + "SELECT value FROM scratch ORDER BY value LIMIT 1 OFFSET ?", + (high,), + ).fetchone()[0] + ) + return low_value * (high - index) + high_value * (index - low) + + def _intervals(self, field: str) -> Iterator[Tuple[int, float]]: + previous: Optional[float] = None + for index, value in self._time_rows(field): + if previous is not None: + yield index, value - previous + previous = value + + def _time_rows(self, field: str) -> Iterator[Tuple[int, float]]: + cursor = self.connection.execute( + "SELECT record_index, value_kind, text_value, numeric_value FROM cells " + "WHERE field = ? ORDER BY record_index", + (field,), + ) + for record_index, kind, text_value, numeric_value in cursor: + raw: Any = numeric_value if kind in {"int", "float"} else text_value + value = temporal_number(raw) + if value is not None: + yield int(record_index), value + + def _timing_quality(self, field: str) -> List[Finding]: + findings: List[Finding] = [] + duplicate = _Affected() + backward = _Affected() + self.connection.execute("DELETE FROM scratch") + self.connection.executemany( + "INSERT INTO scratch(position, value) VALUES (?, ?)", + self._time_rows(field), + ) + for (index,) in self.connection.execute( + "SELECT position FROM scratch WHERE position NOT IN " + "(SELECT MIN(position) FROM scratch GROUP BY value) ORDER BY position" + ): + duplicate.add(int(index)) + self.connection.execute("DELETE FROM scratch") + batch: List[Tuple[int, float]] = [] + for index, interval in self._intervals(field): + if interval < 0: + backward.add(index) + elif interval > 0: + batch.append((index, interval)) + if len(batch) >= 1000: + self.connection.executemany( + "INSERT INTO scratch(position, value) VALUES (?, ?)", batch + ) + batch.clear() + if batch: + self.connection.executemany("INSERT INTO scratch(position, value) VALUES (?, ?)", batch) + if duplicate.count: + findings.append( + _finding( + "duplicate-timestamps", + "warning", + field, + duplicate.range, + duplicate.count, + 0, + "Adjacent timestamps are equal.", + ) + ) + if backward.count: + findings.append( + _finding( + "timestamps-backwards", + "error", + field, + backward.range, + backward.count, + 0, + "Time moves backwards.", + ) + ) + positive_count = int(self.connection.execute("SELECT COUNT(*) FROM scratch").fetchone()[0]) + if positive_count > 2: + median = self._scratch_percentile(positive_count, 50) + irregular = _Affected() + gaps = _Affected() + for index, interval in self.connection.execute( + "SELECT position, value FROM scratch ORDER BY position" + ): + if abs(float(interval) - median) > median * 0.2: + irregular.add(int(index)) + if float(interval) > median * 2: + gaps.add(int(index)) + if irregular.count: + findings.append( + _finding( + "irregular-timing", + "warning", + field, + irregular.range, + irregular.count, + "±20% of median interval", + "Sampling intervals vary.", + ) + ) + if gaps.count: + findings.append( + _finding( + "large-timing-gaps", + "warning", + field, + gaps.range, + gaps.count, + "2 × median interval", + "Large gaps occur in sampling.", + ) + ) + return findings + + def _sequence_quality(self, field: str) -> List[Finding]: + self.connection.execute("DELETE FROM scratch") + invalid = _Affected() + batch: List[Tuple[int, float]] = [] + for index, value in self._numeric_rows(field): + if not value.is_integer(): + invalid.add(index) + continue + batch.append((index, value)) + if len(batch) >= 1000: + self.connection.executemany( + "INSERT INTO scratch(position, value) VALUES (?, ?)", batch + ) + batch.clear() + if batch: + self.connection.executemany("INSERT INTO scratch(position, value) VALUES (?, ?)", batch) + findings: List[Finding] = [] + if invalid.count: + findings.append( + _finding( + "invalid-values", + "warning", + field, + invalid.range, + invalid.count, + "integer sequence identifiers", + "Sequence identifiers contain non-integer numeric values.", + ) + ) + row = self.connection.execute( + "SELECT MIN(value), MAX(value), COUNT(DISTINCT value) FROM scratch" + ).fetchone() + if row is None or row[0] is None: + return findings + expected = int(float(row[1])) - int(float(row[0])) + 1 + lost = expected - int(row[2]) + if lost <= 0: + return findings + findings.append( + _finding( + "packet-loss", + "warning", + field, + "range", + lost, + 0, + "Sequence identifiers contain gaps.", + ["Identifiers are expected to increase by one."], + ) + ) + return findings + + +class _Affected: + def __init__(self) -> None: + self.count = 0 + self.first: Optional[int] = None + self.last: Optional[int] = None + + def add(self, index: int) -> None: + if self.first is None: + self.first = index + self.last = index + self.count += 1 + + @property + def range(self) -> str: + if self.first is None or self.last is None: + return "" + return str(self.first) if self.first == self.last else f"{self.first}-{self.last}" + + +def _cell_parts(value: Any) -> Tuple[str, Optional[str], Optional[float]]: + if value is None: + return "null", None, None + if isinstance(value, bool): + return "bool", "true" if value else "false", None + if isinstance(value, int): + numeric = float(value) if abs(value) <= 2**53 else None + return "int", str(value), numeric + if isinstance(value, float): + if not math.isfinite(value): + raise ValueError("non-finite numbers cannot be analysed") + return "float", repr(value), value + if isinstance(value, str): + return "str", value, None + if isinstance(value, list): + return "list", json.dumps(value, ensure_ascii=False, sort_keys=True), None + if isinstance(value, dict): + return "dict", json.dumps(value, ensure_ascii=False, sort_keys=True), None + raise ValueError(f"unsupported record value: {type(value).__name__}") + + +def _finite(value: Any) -> Optional[float]: + if isinstance(value, bool) or not isinstance(value, (int, float)): + return None + number = float(value) + return number if math.isfinite(number) else None + + +def _crossed(value: float, level: float, direction: float) -> bool: + return value >= level if direction >= 0 else value <= level + + +def _compensated_add(total: float, compensation: float, value: float) -> Tuple[float, float]: + updated = total + value + if abs(total) >= abs(value): + compensation += (total - updated) + value + else: + compensation += (value - updated) + total + return updated, compensation + + +def _finding( + check_id: str, + severity: str, + field: Optional[str], + affected: str, + evidence: Any, + threshold: Any, + explanation: str, + assumptions: Optional[List[str]] = None, + suggestion: str = "Inspect the affected raw records and confirm domain expectations.", +) -> Finding: + return Finding( + check_id, + severity, + field, + affected, + evidence, + threshold, + explanation, + assumptions or [], + suggestion, + ) diff --git a/src/datary/cli.py b/src/datary/cli.py index 15a124c..59afa7d 100644 --- a/src/datary/cli.py +++ b/src/datary/cli.py @@ -4,12 +4,14 @@ import argparse import csv +import importlib import json import os import sys +import tempfile import time from pathlib import Path -from typing import Any, Dict, List, Optional, Sequence +from typing import IO, Any, Dict, List, Optional, Sequence from datary import __version__ from datary.comparison import compare_sessions @@ -24,18 +26,39 @@ from datary.replay import replay_session from datary.reports import write_report from datary.sessions import Session, list_sessions -from datary.utils import parse_key_values, sparkline +from datary.utils import ( + atomic_text, + csv_safe_cell, + markdown_safe, + parse_key_values, + safe_filename_component, + safe_output, + sparkline, + terminal_safe, +) def parser() -> argparse.ArgumentParser: - root = argparse.ArgumentParser(prog="datary", description="Local-first terminal laboratory for reproducible data.") + root = argparse.ArgumentParser( + prog="datary", description="Local-first terminal laboratory for reproducible data." + ) root.add_argument("--version", action="version", version=f"%(prog)s {__version__}") - root.add_argument("--workspace", type=Path, default=default_workspace(), help="session workspace (or DATARY_WORKSPACE)") + root.add_argument( + "--workspace", + type=Path, + default=default_workspace(), + help="session workspace (or DATARY_WORKSPACE)", + ) commands = root.add_subparsers(dest="command", required=True) record = commands.add_parser("record", help="record a named session from standard input") record.add_argument("name") record.add_argument("--format", choices=SUPPORTED_FORMATS) record.add_argument("--time-field") + record.add_argument("--target-field") + record.add_argument("--response-field") + record.add_argument("--sequence-field") + record.add_argument("--latency-field") + record.add_argument("--bytes-field") record.add_argument("--param", action="append", default=[], metavar="KEY=VALUE") record.add_argument("--unit", action="append", default=[], metavar="FIELD=UNIT") record.add_argument("--command", dest="original_command") @@ -47,6 +70,11 @@ def parser() -> argparse.ArgumentParser: inspect.add_argument("source") inspect.add_argument("--format", choices=SUPPORTED_FORMATS) inspect.add_argument("--time-field") + inspect.add_argument("--target-field") + inspect.add_argument("--response-field") + inspect.add_argument("--sequence-field") + inspect.add_argument("--latency-field") + inspect.add_argument("--bytes-field") inspect.add_argument("--field") inspect.add_argument("--quality", action="store_true") inspect.add_argument( @@ -62,13 +90,23 @@ def parser() -> argparse.ArgumentParser: help="counter field checked for resets; repeat for multiple fields", ) inspect.add_argument("--plot", help="comma-separated fields") + inspect.add_argument("--plot-format", choices=("png", "svg"), default="png") + inspect.add_argument( + "--plot-kind", + choices=("line", "scatter", "step", "histogram"), + default="line", + ) + inspect.add_argument("--overwrite-plot", action="store_true") inspect.add_argument("--json", action="store_true") compare = commands.add_parser("compare", help="compare two or more sessions/files") compare.add_argument("sources", nargs="+") compare.add_argument("--field", action="append") compare.add_argument("--goal") compare.add_argument("--report", type=Path) - compare.add_argument("--format", choices=("terminal", "json", "markdown", "csv"), default="terminal") + compare.add_argument("--overwrite", action="store_true") + compare.add_argument( + "--format", choices=("terminal", "json", "markdown", "csv"), default="terminal" + ) replay = commands.add_parser("replay", help="replay a recorded session") replay.add_argument("session") replay.add_argument("--speed", type=float, default=1.0) @@ -80,16 +118,18 @@ def parser() -> argparse.ArgumentParser: report.add_argument("session") report.add_argument("--format", choices=("markdown", "json"), default="markdown") report.add_argument("--output", type=Path) + report.add_argument("--overwrite", action="store_true") generate = commands.add_parser("generate", help="generate deterministic synthetic data") generate.add_argument("profile", choices=PROFILES) generate.add_argument("--seed", type=int, default=0) generate.add_argument("--duration", type=float, default=10.0) generate.add_argument("--sample-rate", type=float, default=10.0) generate.add_argument("--noise", type=float, default=0.05) - generate.add_argument("--missing-rate", type=float, default=0.0) - generate.add_argument("--duplicate-rate", type=float, default=0.0) + generate.add_argument("--missing-rate", type=float) + generate.add_argument("--duplicate-rate", type=float) generate.add_argument("--format", choices=("jsonl", "csv"), default="jsonl") generate.add_argument("--output", type=Path) + generate.add_argument("--overwrite", action="store_true") generate.add_argument("--real-time", action="store_true") convert = commands.add_parser("convert", help="convert a supported local file") convert.add_argument("source", type=Path) @@ -116,8 +156,8 @@ def main(argv: Optional[Sequence[str]] = None) -> int: except KeyboardInterrupt: print("datary: interrupted", file=sys.stderr) return 130 - except (OSError, ValueError, json.JSONDecodeError) as error: - print(f"datary: error: {error}", file=sys.stderr) + except (OSError, ValueError, json.JSONDecodeError, ImportError) as error: + print(f"datary: error: {terminal_safe(error)}", file=sys.stderr) return 2 @@ -127,13 +167,30 @@ def _dispatch(args: argparse.Namespace) -> int: path = record_stream( sys.stdin, RecordOptions( - args.name, workspace, args.format, args.time_field, args.original_command, - parse_key_values(args.param), parse_key_values(args.unit), args.overwrite, - args.include_path, args.max_line_bytes, args.max_fields, + name=args.name, + workspace=workspace, + input_format=args.format, + time_field=args.time_field, + command=args.original_command, + parameters=parse_key_values(args.param), + units=parse_key_values(args.unit), + overwrite=args.overwrite, + include_path=args.include_path, + max_line_bytes=args.max_line_bytes, + max_fields=args.max_fields, + target_field=args.target_field, + response_field=args.response_field, + sequence_field=args.sequence_field, + latency_field=args.latency_field, + bytes_field=args.bytes_field, ), ) session = Session.open(path) - print(f"Recorded {session.manifest['valid_record_count']} valid and {session.manifest['invalid_record_count']} invalid records in {path}") + print( + f"Recorded {session.manifest['valid_record_count']} valid and " + f"{session.manifest['invalid_record_count']} invalid records in " + f"{terminal_safe(path)}" + ) elif args.command == "inspect": inspection = inspect_source( _resolve(args.source, workspace), @@ -141,6 +198,11 @@ def _dispatch(args: argparse.Namespace) -> int: time_field=args.time_field, monotonic_fields=args.monotonic_field, counter_fields=args.counter_field, + target_field=args.target_field, + response_field=args.response_field, + sequence_field=args.sequence_field, + latency_field=args.latency_field, + bytes_field=args.bytes_field, ) if args.json: print(json.dumps(inspection.to_dict(), indent=2, ensure_ascii=False)) @@ -150,41 +212,93 @@ def _dispatch(args: argparse.Namespace) -> int: source = _resolve(args.source, workspace) records, _, _, _, manifest_time = load_source(source, args.format) session_path = Path(source) if Path(source).is_dir() else workspace - output = session_path / "plots" / f"{'-'.join(args.plot.split(','))}.png" - create_plot(records, args.plot.split(","), output, time_field=args.time_field or manifest_time) - print(f"Plot: {output}") + fields = [field.strip() for field in args.plot.split(",") if field.strip()] + if not fields: + raise ValueError("--plot requires at least one non-empty field") + plot_root = session_path / "plots" + if plot_root.is_symlink(): + raise ValueError("session plot directory may not be a symbolic link") + filename = "plot-" + safe_filename_component("-".join(fields), "data") + output = safe_output( + plot_root, + Path(f"{filename}.{args.plot_format}"), + ) + create_plot( + records, + fields, + output, + time_field=args.time_field or manifest_time, + kind=args.plot_kind, + overwrite=args.overwrite_plot, + ) + print(f"Plot: {terminal_safe(output)}") elif args.command == "compare": - result = compare_sessions([_resolve(source, workspace) for source in args.sources], args.field, args.goal) + result = compare_sessions( + [_resolve(source, workspace) for source in args.sources], args.field, args.goal + ) rendered = _comparison_output(result.to_dict(), args.format) if args.report: - args.report.write_text(rendered, encoding="utf-8") + if args.report.exists() and not args.overwrite: + raise FileExistsError(args.report) + atomic_text(args.report, rendered) else: print(rendered) elif args.command == "replay": session = Session.open(_resolve(args.session, workspace)) while True: - replay_session(session, sys.stdout, speed=args.speed, no_timing=args.no_timing, virtual=args.virtual, output_format=args.format) + replay_session( + session, + sys.stdout, + speed=args.speed, + no_timing=args.no_timing, + virtual=args.virtual, + output_format=args.format, + ) if not args.loop: break elif args.command == "report": session = Session.open(_resolve(args.session, workspace)) suffix = ".json" if args.format == "json" else ".md" - output = args.output or session.path / "reports" / f"report{suffix}" + report_root = session.path / "reports" + if args.output is None and report_root.is_symlink(): + raise ValueError("session report directory may not be a symbolic link") + output = args.output or report_root / f"report{suffix}" + if output.exists() and not args.overwrite: + raise FileExistsError(output) write_report(session, output, args.format) - print(output) + print(terminal_safe(output)) elif args.command == "generate": _generate(args) elif args.command == "convert": output = args.output or args.source.with_suffix(".csv" if args.to == "csv" else ".jsonl") valid, invalid = convert_source(args.source, output, args.to, args.format, args.overwrite) - print(f"Converted {valid} records ({invalid} invalid) to {output}", file=sys.stderr) + print( + f"Converted {valid} records ({invalid} invalid) to {terminal_safe(output)}", + file=sys.stderr, + ) elif args.command == "sessions": for session in list_sessions(workspace): - print(f"{session.name}\t{session.manifest.get('valid_record_count', 0)}\t{session.manifest.get('ended_at', '')}") + print( + f"{terminal_safe(session.name)}\t" + f"{session.manifest.get('valid_record_count', 0)}\t" + f"{terminal_safe(session.manifest.get('ended_at', ''))}" + ) elif args.command == "doctor": checks = _doctor(workspace) - print(json.dumps(checks, indent=2) if args.json else "\n".join(f"{'OK' if value else 'FAIL'} {name}" for name, value in checks.items())) - return 0 if all(checks.values()) else 1 + if args.json: + print(json.dumps(checks, indent=2)) + else: + doctor_lines: List[str] = [] + for name, value in checks.items(): + status = ( + "OK" + if value + else ("OPTIONAL-MISSING" if name.startswith("optional_") else "FAIL") + ) + doctor_lines.append(f"{status} {name}") + print("\n".join(doctor_lines)) + required = [value for name, value in checks.items() if not name.startswith("optional_")] + return 0 if all(required) else 1 return 0 @@ -194,16 +308,16 @@ def _resolve(source: str, workspace: Path) -> Path: def _print_inspection(data: Dict[str, Any], selected: Optional[str], show_quality: bool) -> None: - print(f"Source: {data['source']} ({data['format']})") + print(f"Source: {terminal_safe(data['source'])} ({terminal_safe(data['format'])})") print(f"Records: {data['record_count']} valid, {data['invalid_count']} invalid") for field in sorted(data["fields"]): if selected and field != selected: continue metric = data["metrics"].get(field) - unit = f" [{data['units'][field]}]" if field in data["units"] else "" - line = f"{field}{unit}: {data['fields'][field]}" + unit = f" [{terminal_safe(data['units'][field])}]" if field in data["units"] else "" + line = f"{terminal_safe(field)}{unit}: {terminal_safe(data['fields'][field])}" if metric and metric.get("valid_count"): - values = [metric[key] for key in ("minimum", "mean", "maximum") if metric.get(key) is not None] + values = metric.get("sparkline_values", []) line += ( f" min={metric['minimum']:.6g} mean={metric['mean']:.6g} " f"max={metric['maximum']:.6g} {sparkline(values, _ascii_terminal())}" @@ -211,78 +325,230 @@ def _print_inspection(data: Dict[str, Any], selected: Optional[str], show_qualit print(line) if data["timing"]: print(f"Timing: {json.dumps(data['timing'], sort_keys=True)}") + for category, metrics in sorted(data.get("engineering", {}).items()): + printable = { + key: value + for key, value in metrics.items() + if key not in {"assumptions", "warnings", "required_fields"} + } + print( + f"{terminal_safe(str(category).title())}: " + f"{terminal_safe(json.dumps(printable, sort_keys=True))}" + ) if show_quality or data["quality"]: print(f"Quality findings: {len(data['quality'])}") for item in data["quality"]: separator = "-" if _ascii_terminal() else "—" - print(f" {item['severity']}: {item['check_id']} {item.get('field') or ''} {separator} {item['explanation']}") + print( + f" {terminal_safe(item['severity'])}: " + f"{terminal_safe(item['check_id'])} " + f"{terminal_safe(item.get('field') or '')} {separator} " + f"{terminal_safe(item['explanation'])}" + ) def _comparison_output(data: Dict[str, Any], fmt: str) -> str: if fmt == "json": return json.dumps(data, indent=2, ensure_ascii=False) if fmt == "csv": - lines = ["field,source,mean"] + lines = [ + "record_type,field,source,valid_count,minimum,maximum,mean,median," + "standard_deviation,unit,comparable" + ] for field, detail in data["fields"].items(): - for source, value in detail["means"].items(): - lines.append(f"{_csv(field)},{_csv(source)},{value}") + for source, statistics in detail["statistics_by_source"].items(): + lines.append( + ",".join( + [ + "metric", + _csv(field), + _csv(source), + str(statistics.get("valid_count")), + str(statistics.get("minimum")), + str(statistics.get("maximum")), + str(statistics.get("mean")), + str(statistics.get("median")), + str(statistics.get("standard_deviation")), + _csv(str(detail["units"].get(source) or "")), + str(detail["comparable"]).lower(), + ] + ) + ) + for warning in data["warnings"]: + lines.append( + ",".join( + [ + "warning", + _csv(str(warning)), + "", + "", + "", + "", + "", + "", + "", + "", + "", + ] + ) + ) return "\n".join(lines) + "\n" lines = ["# Datary comparison", ""] if fmt == "markdown" else ["Datary comparison"] for field, detail in data["fields"].items(): - lines.append(f"{'## ' if fmt == 'markdown' else ''}{field}") + display_field = markdown_safe(field) if fmt == "markdown" else terminal_safe(field) + lines.append(f"{'## ' if fmt == 'markdown' else ''}{display_field}") for source, value in detail["means"].items(): - lines.append(f"- {source}: {value}" if fmt == "markdown" else f" {source}: {value}") - lines.extend(f"Warning: {warning}" for warning in data["warnings"]) + statistics = detail["statistics_by_source"][source] + unit = detail["units"].get(source) or "" + lines.append( + ( + f"- {markdown_safe(source)}: mean `{value}`, median " + f"`{statistics.get('median')}`, standard deviation " + f"`{statistics.get('standard_deviation')}`, count " + f"`{statistics.get('valid_count')}`, unit " + f"{markdown_safe(unit or '')}" + ) + if fmt == "markdown" + else ( + f" {terminal_safe(source)}: mean={value} " + f"median={statistics.get('median')} " + f"std={statistics.get('standard_deviation')} " + f"n={statistics.get('valid_count')} " + f"unit={terminal_safe(unit or '')}" + ) + ) + lines.extend( + f"Warning: {markdown_safe(warning)}" + if fmt == "markdown" + else f"Warning: {terminal_safe(warning)}" + for warning in data["warnings"] + ) return "\n".join(lines) def _csv(value: str) -> str: import io + buffer = io.StringIO() - csv.writer(buffer, lineterminator="").writerow([value]) + csv.writer(buffer, lineterminator="").writerow([csv_safe_cell(value)]) return buffer.getvalue() def _generate(args: argparse.Namespace) -> None: - records = generate_records(args.profile, seed=args.seed, duration=args.duration, sample_rate=args.sample_rate, noise=args.noise, missing_rate=args.missing_rate, duplicate_rate=args.duplicate_rate) - output = args.output.open("w", encoding="utf-8", newline="") if args.output else sys.stdout + records = generate_records( + args.profile, + seed=args.seed, + duration=args.duration, + sample_rate=args.sample_rate, + noise=args.noise, + missing_rate=args.missing_rate, + duplicate_rate=args.duplicate_rate, + ) + requested = args.output + if requested and requested.exists() and not args.overwrite: + raise FileExistsError(requested) + if requested and requested.is_symlink(): + raise ValueError("generator output may not be a symbolic link") + temporary: Optional[Path] = None + output: IO[str] + if requested: + requested.parent.mkdir(parents=True, exist_ok=True) + descriptor, temporary_name = tempfile.mkstemp( + prefix=f".{requested.name}.generating-", + dir=str(requested.parent), + ) + temporary = Path(temporary_name) + output = os.fdopen(descriptor, "w", encoding="utf-8", newline="") + else: + output = sys.stdout close = args.output is not None + completed = False try: first = True fields: List[str] = [] writer: Optional[csv.DictWriter[str]] = None for record in records: if args.format == "jsonl": - output.write(json.dumps(record, sort_keys=True, ensure_ascii=False, allow_nan=False) + "\n") + output.write( + json.dumps(record, sort_keys=True, ensure_ascii=False, allow_nan=False) + "\n" + ) else: if first: fields = list(record) writer = csv.DictWriter(output, fieldnames=fields) - writer.writeheader() + csv.writer(output).writerow([csv_safe_cell(field) for field in fields]) assert writer is not None writer.writerow(record) output.flush() if args.real_time and not first: time.sleep(1 / args.sample_rate) first = False + completed = True finally: if close: output.close() + if temporary and requested: + if completed: + try: + os.replace(temporary, requested) + except OSError: + temporary.unlink(missing_ok=True) + raise + else: + temporary.unlink(missing_ok=True) def _doctor(workspace: Path) -> Dict[str, bool]: - python_ok = sys.version_info >= (3, 9) + python_ok = (3, 9) <= sys.version_info[:2] <= (3, 14) workspace.mkdir(parents=True, exist_ok=True) - writable = os.access(workspace, os.W_OK) + writable = _probe_writable(workspace) plotting = False try: - import matplotlib + matplotlib: Any = importlib.import_module("matplotlib") matplotlib.use("Agg", force=True) plotting = True - except ImportError: + except (ImportError, RuntimeError): pass - integrity = all(not session.verify() for session in list_sessions(workspace)) - return {"python_3_9_or_newer": python_ok, "workspace_writable": writable, "plotting_available": plotting, "session_integrity": integrity, f"datary_{__version__}": True} + sessions = list_sessions(workspace) + broken_sessions = False + try: + candidates = [ + child + for child in workspace.iterdir() + if child.is_dir() and (child / "manifest.json").exists() + ] + for candidate in candidates: + try: + Session.open(candidate) + except ValueError: + broken_sessions = True + except OSError: + broken_sessions = True + integrity = not broken_sessions and all(not session.verify() for session in sessions) + outputs = all( + _probe_writable(session.path / directory) + for session in sessions + for directory in ("plots", "reports") + ) + return { + "python_3_9_through_3_14": python_ok, + "workspace_writable": writable, + "session_output_directories_writable": outputs, + "optional_plotting_available": plotting, + "session_integrity": integrity, + f"datary_{__version__}": True, + } + + +def _probe_writable(directory: Path) -> bool: + try: + directory.mkdir(parents=True, exist_ok=True) + descriptor, name = tempfile.mkstemp(prefix=".datary-doctor-", dir=str(directory)) + os.close(descriptor) + Path(name).unlink() + return True + except OSError: + return False def _ascii_terminal() -> bool: diff --git a/src/datary/comparison.py b/src/datary/comparison.py index 23fb1e4..6a7d17d 100644 --- a/src/datary/comparison.py +++ b/src/datary/comparison.py @@ -6,7 +6,7 @@ from typing import Any, Dict, List, Optional, Sequence, Union from datary.inspection import inspect_source -from datary.models import Comparison +from datary.models import Comparison, Inspection def compare_sessions( @@ -17,6 +17,7 @@ def compare_sessions( if len(sources) < 2: raise ValueError("comparison requires at least two sources") inspections = [inspect_source(source) for source in sources] + labels = _unique_labels([str(source) for source in sources]) common = set(inspections[0].metrics) for inspection in inspections[1:]: common &= set(inspection.metrics) @@ -34,15 +35,55 @@ def compare_sessions( raise ValueError("goal must be lower:FIELD or higher:FIELD") from error if direction not in {"lower", "higher"}: raise ValueError("goal direction must be lower or higher") + timing = _timing_context(inspections, labels, warnings) + timing_comparable = bool(timing.get("sampling_rates_comparable", True)) result: Dict[str, Dict[str, Any]] = {} for field in selected: means = { - str(source): inspections[index].metrics[field].get("mean") - for index, source in enumerate(sources) + labels[index]: inspections[index].metrics[field].get("mean") + for index in range(len(sources)) } - detail: Dict[str, Any] = {"means": means} + units = { + labels[index]: inspections[index].units.get(field) for index in range(len(sources)) + } + unit_values = set(units.values()) + units_comparable = len(unit_values) <= 1 + comparable = units_comparable and timing_comparable + if not units_comparable: + warnings.append( + f"field {field!r} has incompatible declared units or missing unit declarations: " + + ", ".join(f"{label}={unit or ''}" for label, unit in units.items()) + ) + statistics_by_source = { + labels[index]: { + key: inspections[index].metrics[field].get(key) + for key in ( + "valid_count", + "minimum", + "maximum", + "mean", + "median", + "standard_deviation", + ) + } + for index in range(len(sources)) + } + detail: Dict[str, Any] = { + "means": means, + "statistics_by_source": statistics_by_source, + "units": units, + "comparable": comparable, + } + valid_counts = { + label: statistics["valid_count"] for label, statistics in statistics_by_source.items() + } + if len(set(valid_counts.values())) > 1: + warnings.append( + f"field {field!r} has different valid sample counts: " + + ", ".join(f"{label}={count}" for label, count in valid_counts.items()) + ) baseline = next(iter(means.values())) - if field == goal_field and baseline not in (None, 0): + if field == goal_field and comparable and baseline not in (None, 0): improvements: Dict[str, Optional[float]] = {} for name, value in means.items(): if value is None: @@ -52,9 +93,69 @@ def compare_sessions( else: improvements[name] = (value - baseline) / abs(baseline) * 100 detail["improvement_percent_vs_first"] = improvements + elif field == goal_field and baseline in (None, 0): + warnings.append( + f"goal field {field!r} has a missing or zero baseline mean; " + "percentage improvement is undefined" + ) result[field] = detail if goal_field and goal_field not in result: warnings.append(f"goal field {goal_field!r} is not comparable") if not goal: warnings.append("no comparison goal supplied; no experiment is labelled better") - return Comparison([str(source) for source in sources], result, warnings, goal) + return Comparison(labels, result, sorted(set(warnings)), goal, timing) + + +def _unique_labels(sources: Sequence[str]) -> List[str]: + counts: Dict[str, int] = {} + labels: List[str] = [] + for source in sources: + counts[source] = counts.get(source, 0) + 1 + labels.append(source if counts[source] == 1 else f"{source}#{counts[source]}") + return labels + + +def _timing_context( + inspections: Sequence[Inspection], + labels: Sequence[str], + warnings: List[str], +) -> Dict[str, Any]: + timed = [ + (labels[index], inspection.timing) + for index, inspection in enumerate(inspections) + if inspection.timing + ] + if len(timed) != len(inspections): + warnings.append("one or more sources have no comparable numeric time field") + return {} + starts = [float(timing["start"]) for _, timing in timed if "start" in timing] + ends = [float(timing["end"]) for _, timing in timed if "end" in timing] + rates = {label: timing.get("effective_sample_rate") for label, timing in timed} + positive_rates = [float(rate) for rate in rates.values() if rate] + rates_comparable = not positive_rates or max(positive_rates) / min(positive_rates) <= 1.01 + if not rates_comparable: + warnings.append( + "sampling rates differ; descriptive statistics are shown without resampling " + "and goal improvement is not calculated" + ) + result: Dict[str, Any] = { + "sample_rates": rates, + "sampling_rates_comparable": rates_comparable, + } + if len(starts) == len(timed) and len(ends) == len(timed): + shared_start = max(starts) + shared_end = min(ends) + result["ranges"] = {label: [timing["start"], timing["end"]] for label, timing in timed} + result["shared_range"] = [shared_start, shared_end] if shared_start <= shared_end else None + durations = {label: max(0.0, float(timing.get("duration", 0.0))) for label, timing in timed} + result["relative_ranges"] = { + label: [0.0, duration] for label, duration in durations.items() + } + result["shared_relative_range"] = [0.0, min(durations.values())] + if shared_start > shared_end: + warnings.append("sources have no shared time range") + else: + warnings.append( + "time ranges are reported, but values are not interpolated or resampled" + ) + return result diff --git a/src/datary/config.py b/src/datary/config.py index 9190891..9d4ac81 100644 --- a/src/datary/config.py +++ b/src/datary/config.py @@ -13,7 +13,7 @@ def default_workspace() -> Path: return Path.cwd() -SESSION_FORMAT_VERSION = "1" +SESSION_FORMAT_VERSION = "2" +SUPPORTED_SESSION_FORMAT_VERSIONS = ("1", "2") MAX_AUTO_DETECT_BYTES = 262_144 LARGE_FILE_WARNING_BYTES = 1_000_000_000 - diff --git a/src/datary/conversion.py b/src/datary/conversion.py index 746eb87..ab174c1 100644 --- a/src/datary/conversion.py +++ b/src/datary/conversion.py @@ -1,44 +1,95 @@ -"""Safe format conversion.""" +"""Safe, streaming format conversion.""" from __future__ import annotations import csv import json +import os +import tempfile from pathlib import Path -from typing import Any, Optional +from typing import Optional -from datary.inspection import load_source +from datary.analysis_store import AnalysisStore +from datary.inspection import iter_source +from datary.utils import atomic_json, csv_safe_cell def convert_source( - source: Path, output: Path, to_format: str, input_format: Optional[str] = None, overwrite: bool = False + source: Path, + output: Path, + to_format: str, + input_format: Optional[str] = None, + overwrite: bool = False, ) -> tuple[int, int]: + if to_format not in {"csv", "jsonl"}: + raise ValueError("conversion target must be csv or jsonl") if output.exists() and not overwrite: raise FileExistsError(output) - records, invalid, _, _, _ = load_source(source, input_format) + if output.is_symlink(): + raise ValueError("conversion output may not be a symbolic link") output.parent.mkdir(parents=True, exist_ok=True) - if to_format == "jsonl": - with output.open("w", encoding="utf-8", newline="\n") as stream: - for record in records: - stream.write(json.dumps(record, ensure_ascii=False, allow_nan=False) + "\n") - elif to_format == "csv": - fields = sorted({key for record in records for key in record}) - with output.open("w", encoding="utf-8", newline="") as stream: - writer = csv.DictWriter(stream, fieldnames=fields) - writer.writeheader() - for record in records: - writer.writerow({key: _cell(record.get(key)) for key in fields}) - else: - raise ValueError("conversion target must be csv or jsonl") - if invalid: - sidecar = output.with_suffix(output.suffix + ".invalid.json") - sidecar.write_text( - json.dumps({"invalid_record_count": invalid, "source": str(source)}, indent=2) + "\n", - encoding="utf-8", + iterator, known_invalid, _, _, _ = iter_source(source, input_format) + output_descriptor, temporary_output_name = tempfile.mkstemp( + prefix=f".{output.name}.converting-", + suffix=output.suffix, + dir=str(output.parent), + ) + temporary_output = Path(temporary_output_name) + valid = 0 + invalid = known_invalid + fd, database_name = tempfile.mkstemp(prefix="datary-convert-", suffix=".sqlite3") + os.close(fd) + database = Path(database_name) + try: + with ( + os.fdopen( + output_descriptor, + "w", + encoding="utf-8", + newline="\n" if to_format == "jsonl" else "", + ) as stream, + AnalysisStore(database) as analysis, + ): + for record in iterator: + if record is None: + invalid += 1 + continue + valid += 1 + if to_format == "jsonl": + stream.write( + json.dumps( + record, + ensure_ascii=False, + allow_nan=False, + separators=(",", ":"), + ) + + "\n" + ) + else: + analysis.add(record) + if to_format == "csv": + analysis.finish() + fields = analysis.fields + writer = csv.DictWriter(stream, fieldnames=fields) + csv.writer(stream).writerow([csv_safe_cell(field) for field in fields]) + for record in analysis.records(): + writer.writerow({field: csv_safe_cell(record.get(field)) for field in fields}) + stream.flush() + os.fsync(stream.fileno()) + os.replace(temporary_output, output) + atomic_json( + output.with_suffix(output.suffix + ".invalid.json"), + { + "invalid_record_count": invalid, + "source": str(source), + "explanation": ( + "Malformed source records were omitted from the converted data; " + "inspect the original source or a Datary session invalid.jsonl " + "for record-level evidence." + ), + }, ) - return len(records), invalid - - -def _cell(value: Any) -> Any: - return json.dumps(value, ensure_ascii=False, sort_keys=True) if isinstance(value, (list, dict)) else value - + return valid, invalid + finally: + database.unlink(missing_ok=True) + temporary_output.unlink(missing_ok=True) diff --git a/src/datary/formats.py b/src/datary/formats.py index 4de7471..3337f5a 100644 --- a/src/datary/formats.py +++ b/src/datary/formats.py @@ -28,7 +28,11 @@ def detect_format(sample: str, filename: Optional[str] = None) -> Tuple[str, Lis if isinstance(value, list): return "json", warnings except json.JSONDecodeError: - pass + warnings.append( + "JSON array inferred from its opening delimiter; " + "the complete stream will be validated while parsing" + ) + return "json", warnings lines = [line for line in text.splitlines() if line.strip()] if lines: json_objects = 0 @@ -40,14 +44,16 @@ def detect_format(sample: str, filename: Optional[str] = None) -> Tuple[str, Lis break if json_objects == min(20, len(lines)): return "jsonl", warnings - if lines and all( - len(re.findall(r"(?:^|\s)[^=\s]+=[^\s]+", line)) >= 1 for line in lines[:10] - ): + if lines and all(len(re.findall(r"(?:^|\s)[^=\s]+=[^\s]+", line)) >= 1 for line in lines[:10]): return "keyvalue", warnings - if suffix == ".tsv" or (lines and "\t" in lines[0]): - return "tsv", warnings - if suffix == ".csv": - return "csv", warnings + if lines and "\t" in lines[0]: + rows = list(csv.reader(io.StringIO("\n".join(lines[:10])), delimiter="\t")) + if len({len(row) for row in rows}) == 1: + if any(not _number(cell) for cell in rows[0]): + return "tsv", warnings + raise AmbiguousFormatError( + "tab-separated numeric input is ambiguous; use --format tsv or whitespace" + ) if lines and "," in lines[0]: rows = list(csv.reader(io.StringIO("\n".join(lines[:10])))) if len({len(row) for row in rows}) == 1: @@ -58,6 +64,11 @@ def detect_format(sample: str, filename: Optional[str] = None) -> Tuple[str, Lis raise AmbiguousFormatError( "comma-separated numeric input is ambiguous; use --format stream or csv" ) + if suffix in {".csv", ".tsv"} and lines: + delimiter = "," if suffix == ".csv" else "\t" + first = next(csv.reader([lines[0]], delimiter=delimiter)) + if any(not _number(cell) for cell in first): + return suffix[1:], warnings if lines and all(len(line.split()) > 1 for line in lines[:10]): if all(all(_number(cell) for cell in line.split()) for line in lines[:10]): return "whitespace", warnings diff --git a/src/datary/generators.py b/src/datary/generators.py index 098f327..96ac883 100644 --- a/src/datary/generators.py +++ b/src/datary/generators.py @@ -7,9 +7,18 @@ from typing import Any, Dict, Iterator, Optional PROFILES = ( - "sine", "noisy-sensor", "frozen-sensor", "missing-samples", "duplicate-samples", - "pid-response", "motor-speed", "battery-drain", "network-latency", "packet-loss", + "sine", + "noisy-sensor", + "frozen-sensor", + "missing-samples", + "duplicate-samples", + "pid-response", + "motor-speed", + "battery-drain", + "network-latency", + "packet-loss", ) +MAX_GENERATED_RECORDS = 10_000_000 def generate_records( @@ -19,27 +28,59 @@ def generate_records( duration: float = 10.0, sample_rate: float = 10.0, noise: float = 0.05, - missing_rate: float = 0.0, - duplicate_rate: float = 0.0, + missing_rate: Optional[float] = None, + duplicate_rate: Optional[float] = None, ) -> Iterator[Dict[str, Any]]: if profile not in PROFILES: raise ValueError(f"unknown profile {profile!r}") - if duration < 0 or sample_rate <= 0 or not 0 <= missing_rate <= 1 or not 0 <= duplicate_rate <= 1: + if ( + not math.isfinite(duration) + or not math.isfinite(sample_rate) + or not math.isfinite(noise) + or duration < 0 + or sample_rate <= 0 + or noise < 0 + or (missing_rate is not None and not 0 <= missing_rate <= 1) + or (duplicate_rate is not None and not 0 <= duplicate_rate <= 1) + or (missing_rate is not None and not math.isfinite(missing_rate)) + or (duplicate_rate is not None and not math.isfinite(duplicate_rate)) + ): raise ValueError("duration/sample-rate/rates are out of range") + default_missing = missing_rate is None and profile in {"missing-samples", "packet-loss"} + default_duplicate = duplicate_rate is None and profile == "duplicate-samples" + effective_missing = 0.1 if default_missing else (missing_rate or 0.0) + effective_duplicate = 0.1 if default_duplicate else (duplicate_rate or 0.0) rng = random.Random(seed) - count = int(duration * sample_rate) + 1 + requested_count = duration * sample_rate + if not math.isfinite(requested_count) or requested_count + 1 > MAX_GENERATED_RECORDS: + raise ValueError( + f"generated dataset exceeds safety limit ({MAX_GENERATED_RECORDS} records)" + ) + count = int(requested_count) + 1 previous: Optional[Dict[str, Any]] = None for index in range(count): time = index / sample_rate - if profile in {"missing-samples", "packet-loss"} and index and rng.random() < max(missing_rate, 0.1): + if ( + profile in {"missing-samples", "packet-loss"} + and index + and effective_missing > 0 + and ( + (default_missing and index == max(1, count // 2)) + or rng.random() < effective_missing + ) + ): continue record = _profile(profile, index, time, duration, rng, noise) - if profile == "noisy-sensor" and rng.random() < missing_rate: + if profile == "noisy-sensor" and effective_missing > 0 and rng.random() < effective_missing: record["value"] = None - if profile in {"duplicate-samples"} and previous and rng.random() < max(duplicate_rate, 0.1): - yield dict(previous) - elif previous and rng.random() < duplicate_rate: - yield dict(previous) + if previous: + if profile == "duplicate-samples": + if ( + default_duplicate and index == max(1, count // 2) + ) or rng.random() < effective_duplicate: + yield dict(previous) + elif effective_duplicate > 0 and rng.random() < effective_duplicate: + yield dict(previous) yield record previous = record @@ -58,13 +99,33 @@ def _profile( if profile == "pid-response": target = 1.0 response = 1 - math.exp(-time) * (math.cos(2 * time) + 0.2 * math.sin(2 * time)) - return {"timestamp": time, "target": target, "response": response + perturb, "error": target - response} + measured_response = response + perturb + return { + "timestamp": time, + "target": target, + "response": measured_response, + "error": target - measured_response, + } if profile == "motor-speed": target = 1500.0 speed = target * (1 - math.exp(-time / 1.5)) + perturb * 20 - return {"timestamp": time, "target_rpm": target, "speed_rpm": speed, "current_a": 2 + 8 * math.exp(-time)} + return { + "timestamp": time, + "target_rpm": target, + "speed_rpm": speed, + "current_a": 2 + 8 * math.exp(-time), + } if profile == "battery-drain": - return {"timestamp": time, "voltage_v": 4.2 - 1.2 * time / max(duration, 1e-12) + perturb, "current_a": 0.5 + perturb / 10} + return { + "timestamp": time, + "voltage_v": 4.2 - 1.2 * time / max(duration, 1e-12) + perturb, + "current_a": 0.5 + perturb / 10, + } if profile in {"network-latency", "packet-loss"}: - return {"timestamp": time, "sequence": index, "latency_ms": max(0.0, 20 + rng.gauss(0, noise * 20)), "bytes": 1024} + return { + "timestamp": time, + "sequence": index, + "latency_ms": max(0.0, 20 + rng.gauss(0, noise * 20)), + "bytes": 1024, + } raise AssertionError(profile) diff --git a/src/datary/inspection.py b/src/datary/inspection.py index 3e196ad..91480b3 100644 --- a/src/datary/inspection.py +++ b/src/datary/inspection.py @@ -1,38 +1,75 @@ -"""Source inspection for sessions and ordinary files.""" +"""Source inspection for sessions and ordinary local files.""" from __future__ import annotations +import codecs +import os +import tempfile from pathlib import Path -from typing import Dict, List, Optional, Sequence, Union +from typing import Any, Dict, Iterator, List, Optional, Sequence, Tuple, Union +from datary.analysis_store import AnalysisStore +from datary.config import LARGE_FILE_WARNING_BYTES, MAX_AUTO_DETECT_BYTES from datary.formats import detect_format -from datary.metrics import summarize_records from datary.models import Inspection, Record from datary.parsers import parse_lines -from datary.quality import analyze_quality from datary.sessions import Session -from datary.utils import infer_type +from datary.utils import bounded_text_lines def load_source( - source: Union[str, Path, Session], input_format: Optional[str] = None -) -> tuple[List[Record], int, str, Dict[str, str], Optional[str]]: + source: Union[str, Path, Session], + input_format: Optional[str] = None, +) -> Tuple[List[Record], int, str, Dict[str, str], Optional[str]]: + """Load records for callers that explicitly need materialized data, such as plots.""" + + iterator, invalid, selected, units, time_field = iter_source(source, input_format) + records: List[Record] = [] + invalid_count = invalid + for result in iterator: + if result is None: + invalid_count += 1 + else: + records.append(result) + return records, invalid_count, selected, units, time_field + + +def iter_source( + source: Union[str, Path, Session], + input_format: Optional[str] = None, +) -> Tuple[Iterator[Optional[Record]], int, str, Dict[str, str], Optional[str]]: + """Return a lazy record iterator and source metadata.""" + if isinstance(source, Session): - return list(source.records()), int(source.manifest.get("invalid_record_count", 0)), str(source.manifest["input_format"]), dict(source.manifest.get("units", {})), source.manifest.get("time_field") + return ( + (record for record in source.records()), + int(source.manifest.get("invalid_record_count", 0)), + str(source.manifest["input_format"]), + dict(source.manifest.get("units", {})), + _optional_string(source.manifest.get("time_field")), + ) path = Path(source) if path.is_dir(): - return load_source(Session.open(path), input_format) - sample = path.read_text(encoding="utf-8")[:262_144] + return iter_source(Session.open(path), input_format) + if not path.exists(): + return iter_source(Session.open(path), input_format) + if not path.is_file() or path.is_symlink(): + raise ValueError(f"source is missing or unsafe: {source}") + if path.stat().st_size > LARGE_FILE_WARNING_BYTES: + raise ValueError( + f"source exceeds the default safety limit ({LARGE_FILE_WARNING_BYTES} bytes)" + ) + with path.open("rb") as sample_stream: + sample_bytes = sample_stream.read(MAX_AUTO_DETECT_BYTES) + sample = codecs.getincrementaldecoder("utf-8-sig")().decode(sample_bytes, final=False) selected = input_format or detect_format(sample, path.name)[0] - records: List[Record] = [] - invalid = 0 - with path.open("r", encoding="utf-8", newline="") as stream: - for result in parse_lines(stream, selected): - if result.record is None: - invalid += 1 - else: - records.append(result.record) - return records, invalid, selected, {}, None + + def records() -> Iterator[Optional[Record]]: + with path.open("r", encoding="utf-8-sig", newline="") as stream: + for result in parse_lines(bounded_text_lines(stream, 1_048_576), selected): + yield result.record + + return records(), 0, selected, {}, None def inspect_source( @@ -42,30 +79,105 @@ def inspect_source( time_field: Optional[str] = None, monotonic_fields: Optional[Sequence[str]] = None, counter_fields: Optional[Sequence[str]] = None, + target_field: Optional[str] = None, + response_field: Optional[str] = None, + sequence_field: Optional[str] = None, + latency_field: Optional[str] = None, + bytes_field: Optional[str] = None, ) -> Inspection: - records, invalid, selected, units, manifest_time = load_source(source, input_format) - chosen_time = time_field or manifest_time - fields = { - field: infer_type(record.get(field) for record in records) - for field in sorted({key for record in records for key in record}) - } - metrics = summarize_records(records, chosen_time) - return Inspection( - source=str(source.path if isinstance(source, Session) else source), - format=selected, - record_count=len(records), - invalid_count=invalid, - fields=fields, - metrics=metrics["numeric"], - timing=metrics["timing"], - quality=[ - finding.to_dict() - for finding in analyze_quality( - records, - chosen_time, - monotonic_fields=monotonic_fields, - counter_fields=counter_fields, - ) - ], - units=units, - ) + if not isinstance(source, Session) and Path(source).is_dir(): + source = Session.open(Path(source)) + roles = dict(source.manifest.get("field_roles", {})) if isinstance(source, Session) else {} + target_field = target_field or roles.get("target") + response_field = response_field or roles.get("response") + sequence_field = sequence_field or roles.get("sequence") + latency_field = latency_field or roles.get("latency") + bytes_field = bytes_field or roles.get("bytes") + if (target_field is None) != (response_field is None): + raise ValueError("target_field and response_field must be supplied together") + if target_field and time_field is None and roles.get("time") is None: + # A session time field is resolved below; ordinary files need it explicitly. + if not isinstance(source, Session) or source.manifest.get("time_field") is None: + raise ValueError("control metrics require a time field") + if (sequence_field is None) != (latency_field is None): + raise ValueError("sequence_field and latency_field must be supplied together") + if bytes_field and not sequence_field: + raise ValueError("bytes_field requires sequence and latency fields") + iterator, known_invalid, selected, units, manifest_time = iter_source(source, input_format) + fd, temporary_name = tempfile.mkstemp(prefix="datary-analysis-", suffix=".sqlite3") + os.close(fd) + temporary = Path(temporary_name) + invalid = known_invalid + try: + with AnalysisStore(temporary) as analysis: + for record in iterator: + if record is None: + invalid += 1 + else: + analysis.add(record) + analysis.finish() + chosen_time = time_field or manifest_time + metrics = analysis.metrics(chosen_time) + fields = analysis.field_definitions() + configured_fields = { + "time": chosen_time, + "target": target_field, + "response": response_field, + "sequence": sequence_field, + "latency": latency_field, + "bytes": bytes_field, + } + for role, field in configured_fields.items(): + if field is not None and field not in fields: + raise ValueError(f"{role} field {field!r} does not exist in valid records") + for expectation, names in ( + ("monotonic", monotonic_fields or ()), + ("counter", counter_fields or ()), + ): + missing = sorted(set(names) - set(fields)) + if missing: + raise ValueError( + f"{expectation} fields do not exist in valid records: " + ", ".join(missing) + ) + quality = [ + finding.to_dict() + for finding in analysis.quality( + chosen_time, + sequence_field=sequence_field, + monotonic_fields=monotonic_fields, + counter_fields=counter_fields, + ) + ] + engineering: Dict[str, Any] = {} + if chosen_time and target_field and response_field: + control = analysis.control_metrics(chosen_time, target_field, response_field) + if control: + engineering["control"] = control + if sequence_field and latency_field: + network = analysis.network_metrics( + sequence_field, + latency_field, + bytes_field, + chosen_time, + ) + if network: + engineering["network"] = network + count = analysis.count + return Inspection( + source=str(source.path if isinstance(source, Session) else source), + format=selected, + record_count=count, + invalid_count=invalid, + fields=fields, + metrics=metrics["numeric"], + timing=metrics["timing"], + quality=quality, + units=units, + engineering=engineering, + ) + finally: + temporary.unlink(missing_ok=True) + + +def _optional_string(value: object) -> Optional[str]: + return value if isinstance(value, str) else None diff --git a/src/datary/metrics.py b/src/datary/metrics.py index d157d84..572ebdc 100644 --- a/src/datary/metrics.py +++ b/src/datary/metrics.py @@ -7,65 +7,83 @@ from typing import Any, Dict, Optional, Sequence from datary.models import Record -from datary.utils import finite_number +from datary.utils import finite_number, temporal_number def numeric_summary(values: Sequence[Optional[float]]) -> Dict[str, Any]: - clean = sorted(value for value in values if value is not None and math.isfinite(value)) - missing = len(values) - len(clean) - if not clean: + ordered = [value for value in values if value is not None and math.isfinite(value)] + sorted_values = sorted(ordered) + missing = len(values) - len(ordered) + if not ordered: return {"count": len(values), "valid_count": 0, "missing_count": missing} - differences = [abs(b - a) for a, b in zip(clean, clean[1:])] + differences = [abs(b - a) for a, b in zip(ordered, ordered[1:])] return { "count": len(values), - "valid_count": len(clean), + "valid_count": len(ordered), "missing_count": missing, - "minimum": clean[0], - "maximum": clean[-1], - "mean": statistics.fmean(clean), - "median": statistics.median(clean), - "standard_deviation": statistics.stdev(clean) if len(clean) > 1 else 0.0, - "variance": statistics.variance(clean) if len(clean) > 1 else 0.0, - "percentiles": {str(p): _percentile(clean, p) for p in (5, 25, 50, 75, 95, 99)}, - "sum": math.fsum(clean), - "rate_of_change": clean[-1] - clean[0] if len(clean) > 1 else 0.0, - "root_mean_square": math.sqrt(statistics.fmean([value * value for value in clean])), + "minimum": sorted_values[0], + "maximum": sorted_values[-1], + "mean": statistics.fmean(ordered), + "median": statistics.median(sorted_values), + "standard_deviation": statistics.stdev(ordered) if len(ordered) > 1 else 0.0, + "variance": statistics.variance(ordered) if len(ordered) > 1 else 0.0, + "percentiles": {str(p): _percentile(sorted_values, p) for p in (5, 25, 50, 75, 95, 99)}, + "sum": math.fsum(ordered), + "rate_of_change": ordered[-1] - ordered[0] if len(ordered) > 1 else 0.0, "mean_absolute_difference": statistics.fmean(differences) if differences else 0.0, + "root_mean_square": math.sqrt(statistics.fmean([value * value for value in ordered])), + "sparkline_values": _downsample(ordered, 40), } def timing_metrics(times: Sequence[float]) -> Dict[str, Any]: + if not times: + return {} intervals = [b - a for a, b in zip(times, times[1:])] positive = [value for value in intervals if value > 0] if not intervals: - return {} + return { + "start": times[0], + "end": times[0], + "duration": 0.0, + "mean_interval": None, + "median_interval": None, + "minimum_interval": None, + "maximum_interval": None, + "jitter": 0.0, + "effective_sample_rate": None, + "gap_count": 0, + "duplicate_timestamp_count": 0, + "backward_timestamp_count": 0, + } mean = statistics.fmean(positive) if positive else 0.0 + median = statistics.median(positive) if positive else 0.0 return { + "start": times[0], + "end": times[-1], + "duration": times[-1] - times[0], "mean_interval": mean, - "median_interval": statistics.median(positive) if positive else 0.0, + "median_interval": median, "minimum_interval": min(intervals), "maximum_interval": max(intervals), "jitter": statistics.pstdev(positive) if len(positive) > 1 else 0.0, "effective_sample_rate": 1.0 / mean if mean > 0 else None, - "gap_count": sum( - 1 - for value in positive - if statistics.median(positive) > 0 - and value >= statistics.median(positive) * 2 - ), - "duplicate_timestamp_count": sum(1 for value in intervals if value == 0), + "gap_count": sum(1 for value in positive if median > 0 and value >= median * 2), + "duplicate_timestamp_count": len(times) - len(set(times)), "backward_timestamp_count": sum(1 for value in intervals if value < 0), } -def summarize_records(records: Sequence[Record], time_field: Optional[str] = None) -> Dict[str, Any]: +def summarize_records( + records: Sequence[Record], time_field: Optional[str] = None +) -> Dict[str, Any]: fields = sorted({key for record in records for key in record}) summaries: Dict[str, Any] = {} for field in fields: values = [finite_number(record.get(field)) for record in records] if any(value is not None for value in values): summaries[field] = numeric_summary(values) - times = [finite_number(record.get(time_field)) for record in records] if time_field else [] + times = [temporal_number(record.get(time_field)) for record in records] if time_field else [] return { "numeric": summaries, "timing": timing_metrics([value for value in times if value is not None]), @@ -76,10 +94,16 @@ def control_metrics( records: Sequence[Record], time_field: str, target_field: str, response_field: str ) -> Dict[str, Any]: points = [ - (finite_number(r.get(time_field)), finite_number(r.get(target_field)), finite_number(r.get(response_field))) + ( + temporal_number(r.get(time_field)), + finite_number(r.get(target_field)), + finite_number(r.get(response_field)), + ) for r in records ] - clean = [(t, target, response) for t, target, response in points if None not in (t, target, response)] + clean = [ + (t, target, response) for t, target, response in points if None not in (t, target, response) + ] if len(clean) < 2: return {} times = [item[0] for item in clean if item[0] is not None] @@ -89,50 +113,147 @@ def control_metrics( initial = responses[0] span = final_target - initial errors = [target - response for target, response in zip(targets, responses)] - rise_start = _first_crossing(times, responses, initial + span * 0.1, span) - rise_end = _first_crossing(times, responses, initial + span * 0.9, span) + target_span = max(targets) - min(targets) + stable_target = target_span <= max(abs(final_target), 1.0) * 1e-9 + rise_start = ( + _first_crossing(times, responses, initial + span * 0.1, span) if stable_target else None + ) + rise_end = ( + _first_crossing(times, responses, initial + span * 0.9, span) if stable_target else None + ) peak = max(responses) if span >= 0 else min(responses) - overshoot = ((peak - final_target) / abs(span) * 100) if span else 0.0 - tolerance = abs(span) * 0.02 - settling = None - for index in range(len(responses)): - if all(abs(value - final_target) <= tolerance for value in responses[index:]): - settling = times[index] - times[0] - break + overshoot = ( + ( + (peak - final_target) / abs(span) * 100 + if span >= 0 + else (final_target - peak) / abs(span) * 100 + ) + if span and stable_target + else None + ) + tolerance = max(abs(span), abs(final_target), 1e-12) * 0.02 + settling: Optional[float] = None + if stable_target: + for index in range(len(responses)): + if all(abs(value - final_target) <= tolerance for value in responses[index:]): + settling = times[index] - times[0] + break dt_errors = [ - (times[i] - times[i - 1], errors[i]) for i in range(1, len(times)) if times[i] >= times[i - 1] + ( + times[i] - times[i - 1], + errors[i - 1], + errors[i], + ) + for i in range(1, len(times)) + if times[i] >= times[i - 1] ] return { - "rise_time": rise_end - rise_start if rise_start is not None and rise_end is not None else None, + "rise_time": rise_end - rise_start + if rise_start is not None and rise_end is not None + else None, "peak_value": peak, "percentage_overshoot": overshoot, "settling_time": settling, "steady_state_error": errors[-1], "mean_absolute_error": statistics.fmean(abs(value) for value in errors), "root_mean_squared_error": math.sqrt(statistics.fmean(value * value for value in errors)), - "integral_absolute_error": math.fsum(dt * abs(error) for dt, error in dt_errors), - "integral_squared_error": math.fsum(dt * error * error for dt, error in dt_errors), + "integral_absolute_error": math.fsum( + dt * (abs(previous) + abs(current)) / 2 for dt, previous, current in dt_errors + ), + "integral_squared_error": math.fsum( + dt * (previous * previous + current * current) / 2 + for dt, previous, current in dt_errors + ), + "assumptions": [ + "Rise time, overshoot, and settling time require a stable step target.", + "Rise-time crossings use recorded samples without interpolation.", + "Settling uses a ±2% band based on the larger of step span or target magnitude.", + "Error integrals use trapezoidal integration over non-decreasing timestamps.", + ], + "warnings": ( + [] + if stable_target + else ["The target changes during the record; step-response metrics are not reported."] + ), } def network_metrics( - records: Sequence[Record], sequence_field: str, latency_field: str, bytes_field: Optional[str] = None + records: Sequence[Record], + sequence_field: str, + latency_field: str, + bytes_field: Optional[str] = None, + time_field: Optional[str] = None, ) -> Dict[str, Any]: sequences = [finite_number(record.get(sequence_field)) for record in records] - valid_sequences = [int(value) for value in sequences if value is not None] - latencies = [finite_number(record.get(latency_field)) for record in records] + valid_sequences = [ + int(value) for value in sequences if value is not None and float(value).is_integer() + ] + invalid_sequence_count = sum( + 1 for value in sequences if value is not None and not float(value).is_integer() + ) + raw_latencies = [finite_number(record.get(latency_field)) for record in records] + negative_latency_count = sum(1 for value in raw_latencies if value is not None and value < 0) + latencies = [value if value is None or value >= 0 else None for value in raw_latencies] unique = set(valid_sequences) expected = max(unique) - min(unique) + 1 if unique else 0 duplicates = len(valid_sequences) - len(unique) - result = { + latency_summary = numeric_summary(latencies) + result: Dict[str, Any] = { "packet_loss_estimate": (expected - len(unique)) / expected if expected else None, "duplicate_packet_rate": duplicates / len(valid_sequences) if valid_sequences else None, "out_of_order_count": sum(1 for a, b in zip(valid_sequences, valid_sequences[1:]) if b < a), - "latency": numeric_summary(latencies), + "latency": latency_summary, + "mean_latency": latency_summary.get("mean"), + "median_latency": latency_summary.get("median"), + "latency_jitter": latency_summary.get("mean_absolute_difference"), + "percentile_latency": latency_summary.get("percentiles", {}), + "assumptions": [ + "Sequence identifiers are expected to be contiguous integers.", + "Latency jitter is the mean absolute difference between consecutive valid samples.", + ], + "warnings": [ + warning + for warning in ( + ( + f"{invalid_sequence_count} non-integer sequence value(s) were excluded." + if invalid_sequence_count + else "" + ), + ( + f"{negative_latency_count} negative latency value(s) were excluded." + if negative_latency_count + else "" + ), + ) + if warning + ], } if bytes_field: - byte_values = [finite_number(record.get(bytes_field)) for record in records] - result["total_bytes"] = math.fsum(value for value in byte_values if value is not None) + raw_bytes = [finite_number(record.get(bytes_field)) for record in records] + negative_bytes = sum(1 for value in raw_bytes if value is not None and value < 0) + total_bytes = math.fsum(value for value in raw_bytes if value is not None and value >= 0) + if negative_bytes: + result["warnings"].append(f"{negative_bytes} negative byte count(s) were excluded.") + result["total_bytes"] = total_bytes + selected_time = time_field + if selected_time is None: + selected_time = next( + ( + candidate + for candidate in ("timestamp", "time", "t") + if any(candidate in record for record in records) + ), + None, + ) + times = ( + [temporal_number(record.get(selected_time)) for record in records] + if selected_time + else [] + ) + valid_times = [value for value in times if value is not None] + duration = valid_times[-1] - valid_times[0] if len(valid_times) > 1 else 0.0 + result["throughput_bytes_per_second"] = total_bytes / duration if duration > 0 else None return result @@ -141,10 +262,20 @@ def _percentile(values: Sequence[float], percentile: int) -> float: return values[0] index = (len(values) - 1) * percentile / 100 low, high = math.floor(index), math.ceil(index) - return values[low] if low == high else values[low] * (high - index) + values[high] * (index - low) + return ( + values[low] if low == high else values[low] * (high - index) + values[high] * (index - low) + ) + + +def _downsample(values: Sequence[float], maximum: int) -> list[float]: + if len(values) <= maximum: + return list(values) + return [values[round(index * (len(values) - 1) / (maximum - 1))] for index in range(maximum)] -def _first_crossing(times: Sequence[float], values: Sequence[float], level: float, direction: float) -> Optional[float]: +def _first_crossing( + times: Sequence[float], values: Sequence[float], level: float, direction: float +) -> Optional[float]: for time, value in zip(times, values): if (value >= level) if direction >= 0 else (value <= level): return time diff --git a/src/datary/models.py b/src/datary/models.py index 5d30acd..023e6d7 100644 --- a/src/datary/models.py +++ b/src/datary/models.py @@ -5,9 +5,11 @@ from dataclasses import asdict, dataclass from dataclasses import field as dataclass_field from pathlib import Path -from typing import Any, Dict, List, Optional +from typing import Any, Dict, List, Optional, Union -Record = Dict[str, Any] +JsonScalar = Union[None, bool, int, float, str] +JsonValue = Union[JsonScalar, List["JsonValue"], Dict[str, "JsonValue"]] +Record = Dict[str, JsonValue] @dataclass @@ -16,8 +18,8 @@ class Finding: severity: str field: Optional[str] affected: str - evidence: Any - threshold: Any + evidence: JsonValue + threshold: JsonValue explanation: str assumptions: List[str] = dataclass_field(default_factory=list) suggested_investigation: str = "" @@ -43,6 +45,7 @@ class Inspection: timing: Dict[str, Any] quality: List[Dict[str, Any]] units: Dict[str, str] = dataclass_field(default_factory=dict) + engineering: Dict[str, Any] = dataclass_field(default_factory=dict) def to_dict(self) -> Dict[str, Any]: return asdict(self) @@ -54,6 +57,7 @@ class Comparison: fields: Dict[str, Dict[str, Any]] warnings: List[str] goal: Optional[str] = None + timing: Dict[str, Any] = dataclass_field(default_factory=dict) def to_dict(self) -> Dict[str, Any]: return asdict(self) @@ -72,3 +76,8 @@ class RecordOptions: include_path: bool = False max_line_bytes: int = 1_048_576 max_fields: int = 1_000 + target_field: Optional[str] = None + response_field: Optional[str] = None + sequence_field: Optional[str] = None + latency_field: Optional[str] = None + bytes_field: Optional[str] = None diff --git a/src/datary/parsers.py b/src/datary/parsers.py index 86fd02a..c1002d3 100644 --- a/src/datary/parsers.py +++ b/src/datary/parsers.py @@ -1,57 +1,75 @@ -"""Streaming parsers. Input is treated only as inert text.""" +"""Streaming parsers. Input is always treated as inert text.""" from __future__ import annotations import csv import json +import re import shlex -from typing import Any, Dict, Iterable, Iterator +import threading +from typing import Any, Dict, Iterable, Iterator, Optional from datary.models import ParseResult, Record +_CANONICAL_INTEGER = re.compile(r"-?(?:0|[1-9]\d*)\Z") +_CANONICAL_FLOAT = re.compile( + r"-?(?:(?:0|[1-9]\d*)\.\d+|(?:0|[1-9]\d*)(?:[eE][+-]?\d+)|" + r"(?:0|[1-9]\d*)\.\d+(?:[eE][+-]?\d+))\Z" +) +_MAX_JSON_BUFFER_BYTES = 16 * 1024 * 1024 +_MAX_CSV_FIELD_CHARACTERS = 16 * 1024 * 1024 +_CSV_LIMIT_LOCK = threading.Lock() + def scalar(value: str) -> Any: + """Apply Datary's conservative, documented scalar-coercion policy. + + Empty cells become missing values. Canonical JSON booleans and canonical + numbers are converted. Identifier-like values such as ``00123`` and + domain tokens such as ``NA`` remain strings. + """ + stripped = value.strip() if stripped == "": return None - lowered = stripped.lower() - if lowered in {"true", "false"}: - return lowered == "true" - if lowered in {"null", "none", "na", "n/a"}: - return None - try: + if stripped == "true": + return True + if stripped == "false": + return False + if _CANONICAL_INTEGER.fullmatch(stripped): return int(stripped) - except ValueError: - try: - return float(stripped) - except ValueError: + if _CANONICAL_FLOAT.fullmatch(stripped): + number = float(stripped) + if not _is_finite(number): return stripped + return number + return stripped -def parse_lines(lines: Iterable[str], input_format: str, max_fields: int = 1000) -> Iterator[ParseResult]: +def parse_lines( + lines: Iterable[str], + input_format: str, + max_fields: int = 1000, +) -> Iterator[ParseResult]: + """Parse records without executing input or silently accepting non-finite JSON.""" + + if max_fields <= 0: + raise ValueError("max_fields must be positive") + clean_lines = _without_bom(lines) if input_format in {"csv", "tsv"}: - yield from _delimited(lines, "\t" if input_format == "tsv" else ",", max_fields) + yield from _delimited(clean_lines, "\t" if input_format == "tsv" else ",", max_fields) return if input_format == "json": - joined = "".join(lines) - try: - value = json.loads(joined) - if not isinstance(value, list): - yield ParseResult(None, "JSON input must be an array") - return - for item in value: - yield _json_record(item, max_fields) - except json.JSONDecodeError as error: - yield ParseResult(None, f"invalid JSON: {error.msg}") + yield from _json_array(clean_lines, max_fields) return - for line_number, line in enumerate(lines, 1): + for line_number, line in enumerate(clean_lines, 1): text = line.rstrip("\r\n") if not text.strip(): yield ParseResult(None, f"line {line_number}: empty line") continue try: if input_format == "jsonl": - result = _json_record(json.loads(text), max_fields) + result = _json_record(_json_loads(text), max_fields) elif input_format == "keyvalue": pairs = shlex.split(text) record: Record = {} @@ -59,32 +77,41 @@ def parse_lines(lines: Iterable[str], input_format: str, max_fields: int = 1000) if "=" not in pair: raise ValueError(f"token lacks '=': {pair}") key, value = pair.split("=", 1) + if not key: + raise ValueError("field name may not be empty") + if key in record: + raise ValueError(f"duplicate field name: {key}") record[key] = scalar(value) result = _checked(record, max_fields) elif input_format == "whitespace": result = _checked( - {f"field_{index + 1}": scalar(value) for index, value in enumerate(text.split())}, + { + f"field_{index + 1}": scalar(value) + for index, value in enumerate(text.split()) + }, max_fields, ) elif input_format == "stream": result = _checked( { f"field_{index + 1}": scalar(value) - for index, value in enumerate(next(csv.reader([text]))) + for index, value in enumerate(next(csv.reader([text], strict=True))) }, max_fields, ) else: result = ParseResult(None, f"unsupported format: {input_format}") - except (ValueError, json.JSONDecodeError, csv.Error) as error: + except (ValueError, json.JSONDecodeError, csv.Error, RecursionError) as error: result = ParseResult(None, f"line {line_number}: {error}") yield result def _delimited(lines: Iterable[str], delimiter: str, max_fields: int) -> Iterator[ParseResult]: - iterator = iter(lines) + """Parse RFC-style CSV records, including quoted embedded newlines.""" + + reader = csv.reader(lines, delimiter=delimiter, strict=True) try: - header = next(csv.reader([next(iterator)], delimiter=delimiter)) + header = _next_csv_row(reader) except StopIteration: return except csv.Error as error: @@ -93,15 +120,141 @@ def _delimited(lines: Iterable[str], delimiter: str, max_fields: int) -> Iterato if not header or len(set(header)) != len(header) or any(not name.strip() for name in header): yield ParseResult(None, "header fields must be non-empty and unique") return - for line_number, line in enumerate(iterator, 2): + if len(header) > max_fields: + yield ParseResult(None, f"header exceeds field limit ({max_fields})") + return + while True: try: - row = next(csv.reader([line], delimiter=delimiter)) - if len(row) != len(header): - yield ParseResult(None, f"line {line_number}: expected {len(header)} fields, got {len(row)}") - else: - yield _checked(dict(zip(header, map(scalar, row))), max_fields) + row = _next_csv_row(reader) + except StopIteration: + return except csv.Error as error: - yield ParseResult(None, f"line {line_number}: {error}") + yield ParseResult(None, f"line {reader.line_num}: {error}") + return + if len(row) != len(header): + yield ParseResult( + None, + f"line {reader.line_num}: expected {len(header)} fields, got {len(row)}", + ) + else: + yield _checked(dict(zip(header, map(scalar, row))), max_fields) + + +def _next_csv_row(reader: Any) -> list[str]: + with _CSV_LIMIT_LOCK: + previous_limit = csv.field_size_limit() + csv.field_size_limit(_MAX_CSV_FIELD_CHARACTERS) + try: + row = next(reader) + return list(row) + finally: + csv.field_size_limit(previous_limit) + + +def _json_array(lines: Iterable[str], max_fields: int) -> Iterator[ParseResult]: + """Incrementally decode a top-level JSON array with a bounded pending buffer.""" + + decoder = json.JSONDecoder( + parse_constant=_reject_json_constant, + object_pairs_hook=_object_without_duplicates, + ) + iterator = iter(lines) + buffer = "" + position = 0 + finished = False + + def fill() -> bool: + nonlocal buffer, position, finished + if finished: + return False + try: + chunk = next(iterator) + except StopIteration: + finished = True + return False + if position: + buffer = buffer[position:] + position = 0 + buffer += chunk + if len(buffer.encode("utf-8")) > _MAX_JSON_BUFFER_BYTES: + raise ValueError( + f"JSON record or pending document exceeds {_MAX_JSON_BUFFER_BYTES} bytes" + ) + return True + + try: + while not buffer and fill(): + pass + position = _skip_space(buffer, position) + while position >= len(buffer) and fill(): + position = _skip_space(buffer, position) + if position >= len(buffer) or buffer[position] != "[": + yield ParseResult(None, "JSON input must be an array") + return + position += 1 + item_number = 0 + expecting_item = True + while True: + position = _skip_space(buffer, position) + while position >= len(buffer) and fill(): + position = _skip_space(buffer, position) + if position >= len(buffer): + yield ParseResult(None, "invalid JSON: unterminated array") + return + if buffer[position] == "]": + if expecting_item and item_number: + yield ParseResult(None, "invalid JSON: trailing comma") + return + position += 1 + break + if not expecting_item: + if buffer[position] != ",": + yield ParseResult(None, "invalid JSON: expected ',' between records") + return + position += 1 + expecting_item = True + continue + while True: + try: + value, end = decoder.raw_decode(buffer, position) + break + except json.JSONDecodeError as error: + if fill(): + continue + yield ParseResult(None, f"invalid JSON: {error.msg}") + return + position = end + item_number += 1 + yield _json_record(value, max_fields) + expecting_item = False + position = _skip_space(buffer, position) + while fill(): + position = _skip_space(buffer, position) + if buffer[position:].strip(): + yield ParseResult(None, "invalid JSON: trailing content after array") + except (ValueError, RecursionError) as error: + yield ParseResult(None, f"invalid JSON: {error}") + + +def _json_loads(text: str) -> Any: + return json.loads( + text, + parse_constant=_reject_json_constant, + object_pairs_hook=_object_without_duplicates, + ) + + +def _reject_json_constant(value: str) -> None: + raise ValueError(f"non-finite JSON number is not allowed: {value}") + + +def _object_without_duplicates(pairs: list[tuple[str, Any]]) -> Dict[str, Any]: + result: Dict[str, Any] = {} + for key, value in pairs: + if key in result: + raise ValueError(f"duplicate JSON object key: {key}") + result[key] = value + return result def _json_record(value: Any, max_fields: int) -> ParseResult: @@ -115,4 +268,53 @@ def _checked(record: Dict[str, Any], max_fields: int) -> ParseResult: return ParseResult(None, f"record exceeds field limit ({max_fields})") if any(not isinstance(key, str) for key in record): return ParseResult(None, "field names must be strings") + encoded_names = [len(key.encode("utf-8")) for key in record] + if any(length > 1024 for length in encoded_names): + return ParseResult(None, "field name exceeds 1024-byte limit") + if sum(encoded_names) > 65_536: + return ParseResult(None, "record field names exceed 65536-byte total limit") + problem = _invalid_json_value(record) + if problem: + return ParseResult(None, problem) return ParseResult(record) + + +def _invalid_json_value(value: Any, path: str = "$") -> Optional[str]: + if value is None or isinstance(value, (str, bool, int)): + return None + if isinstance(value, float): + return None if _is_finite(value) else f"non-finite number at {path}" + if isinstance(value, list): + for index, item in enumerate(value): + problem = _invalid_json_value(item, f"{path}[{index}]") + if problem: + return problem + return None + if isinstance(value, dict): + for key, item in value.items(): + if not isinstance(key, str): + return f"non-string object key at {path}" + problem = _invalid_json_value(item, f"{path}.{key}") + if problem: + return problem + return None + return f"unsupported value type at {path}: {type(value).__name__}" + + +def _without_bom(lines: Iterable[str]) -> Iterator[str]: + first = True + for line in lines: + if first: + line = line.lstrip("\ufeff") + first = False + yield line + + +def _skip_space(value: str, position: int) -> int: + while position < len(value) and value[position] in " \t\r\n": + position += 1 + return position + + +def _is_finite(value: float) -> bool: + return value == value and value not in (float("inf"), float("-inf")) diff --git a/src/datary/plotting.py b/src/datary/plotting.py index bf64149..609ae50 100644 --- a/src/datary/plotting.py +++ b/src/datary/plotting.py @@ -2,11 +2,14 @@ from __future__ import annotations +import importlib +import os +import tempfile from pathlib import Path -from typing import List, Optional, Sequence +from typing import Any, List, Optional, Sequence from datary.models import Record -from datary.utils import finite_number +from datary.utils import finite_number, temporal_number def create_plot( @@ -18,25 +21,34 @@ def create_plot( kind: str = "line", overwrite: bool = False, ) -> Path: + if output.is_symlink(): + raise ValueError("plot output may not be a symbolic link") if output.exists() and not overwrite: raise FileExistsError(output) if output.suffix.lower() not in {".png", ".svg"}: raise ValueError("plot output must use .png or .svg") - import matplotlib - + if kind not in {"line", "scatter", "step", "histogram"}: + raise ValueError("plot kind must be line, scatter, step, or histogram") + matplotlib: Any = importlib.import_module("matplotlib") matplotlib.use("Agg", force=True) - from matplotlib import pyplot as plt + plt: Any = importlib.import_module("matplotlib.pyplot") figure, axis = plt.subplots(figsize=(8, 4.5)) + plotted = 0 for field in fields: + if not any(field in record for record in records): + continue points = [(index, finite_number(record.get(field))) for index, record in enumerate(records)] x_values: List[float] = [] y_values: List[float] = [] + missing_x: List[float] = [] for index, value in points: - x = finite_number(records[index].get(time_field)) if time_field else float(index) + x = temporal_number(records[index].get(time_field)) if time_field else float(index) if x is not None and value is not None: x_values.append(x) y_values.append(value) + elif x is not None: + missing_x.append(x) if kind == "scatter": axis.scatter(x_values, y_values, s=10, label=field) elif kind == "step": @@ -45,12 +57,40 @@ def create_plot( axis.hist(y_values, alpha=0.5, label=field) else: axis.plot(x_values, y_values, label=field) + if missing_x and kind != "histogram": + marker_y = min(y_values) if y_values else 0.0 + axis.scatter( + missing_x, + [marker_y] * len(missing_x), + marker="x", + s=24, + label=f"{field} missing", + ) + plotted += len(y_values) + (len(missing_x) if kind != "histogram" else 0) + if not fields: + plt.close(figure) + raise ValueError("at least one plot field is required") + if plotted == 0: + plt.close(figure) + raise ValueError("selected plot fields contain no numeric values") axis.set_xlabel(time_field or "record") axis.grid(True, alpha=0.25) axis.legend() - figure.tight_layout() output.parent.mkdir(parents=True, exist_ok=True) - figure.savefig(output) - plt.close(figure) + descriptor, temporary_name = tempfile.mkstemp( + prefix=f".{output.stem}.rendering-", + suffix=output.suffix, + dir=str(output.parent), + ) + os.close(descriptor) + temporary = Path(temporary_name) + try: + figure.tight_layout() + figure.savefig(temporary) + if output.exists() and not overwrite: + raise FileExistsError(output) + os.replace(temporary, output) + finally: + plt.close(figure) + temporary.unlink(missing_ok=True) return output - diff --git a/src/datary/py.typed b/src/datary/py.typed new file mode 100644 index 0000000..8b13789 --- /dev/null +++ b/src/datary/py.typed @@ -0,0 +1 @@ + diff --git a/src/datary/quality.py b/src/datary/quality.py index 8edacb0..94c16d8 100644 --- a/src/datary/quality.py +++ b/src/datary/quality.py @@ -9,7 +9,7 @@ from typing import Any, List, Optional, Sequence, Tuple from datary.models import Finding, Record -from datary.utils import finite_number +from datary.utils import finite_number, temporal_number def _finding( @@ -24,7 +24,15 @@ def _finding( suggestion: str = "Inspect the affected raw records and confirm domain expectations.", ) -> Finding: return Finding( - check_id, severity, field, affected, evidence, threshold, explanation, assumptions or [], suggestion + check_id, + severity, + field, + affected, + evidence, + threshold, + explanation, + assumptions or [], + suggestion, ) @@ -39,32 +47,117 @@ def analyze_quality( monotonic = set(monotonic_fields or ()) counters = set(counter_fields or ()) if not records: - return [_finding("empty-data", "warning", None, "all", 0, "> 0 records", "No valid records were parsed.")] + return [ + _finding( + "empty-data", + "warning", + None, + "all", + 0, + "> 0 records", + "No valid records were parsed.", + ) + ] fields = sorted({key for record in records for key in record}) serialized = [json.dumps(record, sort_keys=True, ensure_ascii=False) for record in records] duplicates = sum(count - 1 for count in Counter(serialized).values() if count > 1) if duplicates: - findings.append(_finding("duplicate-rows", "warning", None, "multiple", duplicates, 0, "Identical records recur.")) - lengths = {len(record) for record in records} + findings.append( + _finding( + "duplicate-rows", + "warning", + None, + "multiple", + duplicates, + 0, + "Identical records recur.", + ) + ) + schemas = {tuple(sorted(record)) for record in records} + if len(schemas) > 1: + findings.append( + _finding( + "record-shape-change", + "warning", + None, + "multiple", + [list(schema) for schema in sorted(schemas)], + "one stable field set", + "Record field names change across the dataset.", + ) + ) + lengths = {len(schema) for schema in schemas} if len(lengths) > 1: - findings.append(_finding("record-length-change", "warning", None, "multiple", sorted(lengths), 1, "Record field counts vary.")) + findings.append( + _finding( + "record-length-change", + "warning", + None, + "multiple", + sorted(lengths), + 1, + "Record field counts vary.", + ) + ) for field in fields: values = [record.get(field) for record in records] missing = [index for index, value in enumerate(values) if value is None or value == ""] if missing: - findings.append(_finding("missing-values", "warning", field, _range(missing), len(missing), 0, "Values are absent.")) + findings.append( + _finding( + "missing-values", + "warning", + field, + _range(missing), + len(missing), + 0, + "Values are absent.", + ) + ) types = sorted({type(value).__name__ for value in values if value is not None}) if len(types) > 1 and not set(types) <= {"int", "float"}: - findings.append(_finding("type-change", "warning", field, "multiple", types, 1, "The field changes type.")) + findings.append( + _finding( + "type-change", "warning", field, "multiple", types, 1, "The field changes type." + ) + ) nonfinite = [ - index for index, value in enumerate(values) + index + for index, value in enumerate(values) if isinstance(value, float) and (math.isnan(value) or math.isinf(value)) ] if nonfinite: - findings.append(_finding("non-finite", "error", field, _range(nonfinite), len(nonfinite), 0, "NaN or infinity is not a finite measurement.")) + findings.append( + _finding( + "non-finite", + "error", + field, + _range(nonfinite), + len(nonfinite), + 0, + "NaN or infinity is not a finite measurement.", + ) + ) + large_integers = [ + index + for index, value in enumerate(values) + if isinstance(value, int) and not isinstance(value, bool) and abs(value) > 2**53 + ] + if large_integers: + findings.append( + _finding( + "invalid-values", + "warning", + field, + _range(large_integers), + len(large_integers), + "integer magnitude ≤ 2^53 for exact binary64 analysis", + "Large integers were preserved but excluded from numeric analysis.", + ) + ) numbers = [(index, finite_number(value)) for index, value in enumerate(values)] clean = [(index, value) for index, value in numbers if value is not None] - if len(clean) >= 2: + if len(clean) >= 2 and field not in {time_field, sequence_field}: numeric = [float(value) for _, value in clean] decreases = [ clean[index][0] @@ -100,71 +193,229 @@ def analyze_quality( ) ) if max(numeric) == min(numeric): - findings.append(_finding("constant-signal", "info", field, f"{clean[0][0]}-{clean[-1][0]}", numeric[0], "no variation", "The signal is constant.")) - frozen = _longest_run(numeric) + findings.append( + _finding( + "constant-signal", + "info", + field, + f"{clean[0][0]}-{clean[-1][0]}", + numeric[0], + "no variation", + "The signal is constant.", + ) + ) + frozen = _longest_run(clean) if frozen[1] >= 5 and frozen[1] < len(numeric): - findings.append(_finding("frozen-values", "warning", field, f"{frozen[0]}-{frozen[0] + frozen[1] - 1}", frozen[1], 5, "The same value repeats for an extended run.")) + findings.append( + _finding( + "frozen-values", + "warning", + field, + f"{frozen[0]}-{frozen[2]}", + frozen[1], + 5, + "The same value repeats for an extended run.", + ["Missing values break a frozen run."], + ) + ) if len(numeric) >= 5: median = statistics.median(numeric) deviations = [abs(value - median) for value in numeric] mad = statistics.median(deviations) - if mad > 0: - outliers = [clean[i][0] for i, value in enumerate(numeric) if abs(value - median) > 6 * mad] - if outliers: - findings.append(_finding("outliers", "warning", field, _range(outliers), len(outliers), "6 × MAD", "Values are far from the median.", ["Robust median absolute deviation rule."])) + outliers = [ + clean[i][0] + for i, value in enumerate(numeric) + if (abs(value - median) > 6 * mad if mad > 0 else value != median) + ] + if outliers: + threshold = "6 × MAD" if mad > 0 else "different from zero-MAD median" + findings.append( + _finding( + "outliers", + "warning", + field, + _range(outliers), + len(outliers), + threshold, + "Values are far from the median.", + ["Robust median absolute deviation rule."], + ) + ) diffs = [abs(b - a) for a, b in zip(numeric, numeric[1:])] base = statistics.median(diffs) - spikes = [clean[i + 1][0] for i, value in enumerate(diffs) if base > 0 and value > base * 10] + spikes = [ + clean[i + 1][0] + for i, value in enumerate(diffs) + if (value > base * 10 if base > 0 else value > 0) + ] if spikes: - findings.append(_finding("sudden-spikes", "warning", field, _range(spikes), len(spikes), "10 × median absolute step", "Abrupt changes exceed the configured robust threshold.")) + threshold = ( + "10 × median absolute step" + if base > 0 + else "non-zero step after zero baseline" + ) + findings.append( + _finding( + "sudden-spikes", + "warning", + field, + _range(spikes), + len(spikes), + threshold, + "Abrupt changes exceed the configured robust threshold.", + ) + ) mean = statistics.fmean(numeric) if mean and statistics.pstdev(numeric) / abs(mean) > 0.5: - findings.append(_finding("high-noise", "info", field, "all", statistics.pstdev(numeric), "coefficient of variation > 0.5", "Variation is high relative to the mean.", ["Only meaningful for ratio-scale signals."])) + findings.append( + _finding( + "high-noise", + "info", + field, + "all", + statistics.pstdev(numeric), + "coefficient of variation > 0.5", + "Variation is high relative to the mean.", + ["Only meaningful for ratio-scale signals."], + ) + ) if time_field and time_field in fields: - times = [(index, finite_number(record.get(time_field))) for index, record in enumerate(records)] + times = [ + (index, temporal_number(record.get(time_field))) for index, record in enumerate(records) + ] clean_times = [(index, float(value)) for index, value in times if value is not None] - intervals = [(clean_times[i][0], clean_times[i][1] - clean_times[i - 1][1]) for i in range(1, len(clean_times))] - _timing_findings(findings, time_field, intervals) + intervals = [ + (clean_times[i][0], clean_times[i][1] - clean_times[i - 1][1]) + for i in range(1, len(clean_times)) + ] + seen_times: set[float] = set() + duplicate_times: List[int] = [] + for index, value in clean_times: + if value in seen_times: + duplicate_times.append(index) + else: + seen_times.add(value) + _timing_findings(findings, time_field, intervals, duplicate_times) if sequence_field and sequence_field in fields: sequence = [finite_number(record.get(sequence_field)) for record in records] - clean_seq = [int(value) for value in sequence if value is not None] + invalid_sequence = [ + index + for index, value in enumerate(sequence) + if value is not None and not float(value).is_integer() + ] + if invalid_sequence: + findings.append( + _finding( + "invalid-values", + "warning", + sequence_field, + _range(invalid_sequence), + len(invalid_sequence), + "integer sequence identifiers", + "Sequence identifiers contain non-integer numeric values.", + ) + ) + clean_seq = [ + int(value) for value in sequence if value is not None and float(value).is_integer() + ] if clean_seq: expected = max(clean_seq) - min(clean_seq) + 1 lost = expected - len(set(clean_seq)) if lost > 0: - findings.append(_finding("packet-loss", "warning", sequence_field, "range", lost, 0, "Sequence identifiers contain gaps.", ["Identifiers are expected to increase by one."])) + findings.append( + _finding( + "packet-loss", + "warning", + sequence_field, + "range", + lost, + 0, + "Sequence identifiers contain gaps.", + ["Identifiers are expected to increase by one."], + ) + ) return sorted(findings, key=lambda item: (item.check_id, item.field or "", item.affected)) -def _timing_findings(findings: List[Finding], field: str, intervals: List[Tuple[int, float]]) -> None: - duplicate = [index for index, value in intervals if value == 0] +def _timing_findings( + findings: List[Finding], + field: str, + intervals: List[Tuple[int, float]], + duplicate: Optional[List[int]] = None, +) -> None: + duplicate = duplicate or [index for index, value in intervals if value == 0] backward = [index for index, value in intervals if value < 0] positive = [value for _, value in intervals if value > 0] if duplicate: - findings.append(_finding("duplicate-timestamps", "warning", field, _range(duplicate), len(duplicate), 0, "Adjacent timestamps are equal.")) + findings.append( + _finding( + "duplicate-timestamps", + "warning", + field, + _range(duplicate), + len(duplicate), + 0, + "Adjacent timestamps are equal.", + ) + ) if backward: - findings.append(_finding("timestamps-backwards", "error", field, _range(backward), len(backward), 0, "Time moves backwards.")) + findings.append( + _finding( + "timestamps-backwards", + "error", + field, + _range(backward), + len(backward), + 0, + "Time moves backwards.", + ) + ) if len(positive) > 2: median = statistics.median(positive) - irregular = [index for index, value in intervals if value > 0 and abs(value - median) > median * 0.2] + irregular = [ + index for index, value in intervals if value > 0 and abs(value - median) > median * 0.2 + ] gaps = [index for index, value in intervals if value > median * 2] if irregular: - findings.append(_finding("irregular-timing", "warning", field, _range(irregular), len(irregular), "±20% of median interval", "Sampling intervals vary.")) + findings.append( + _finding( + "irregular-timing", + "warning", + field, + _range(irregular), + len(irregular), + "±20% of median interval", + "Sampling intervals vary.", + ) + ) if gaps: - findings.append(_finding("large-timing-gaps", "warning", field, _range(gaps), len(gaps), "2 × median interval", "Large gaps occur in sampling.")) + findings.append( + _finding( + "large-timing-gaps", + "warning", + field, + _range(gaps), + len(gaps), + "2 × median interval", + "Large gaps occur in sampling.", + ) + ) -def _longest_run(values: List[float]) -> Tuple[int, int]: +def _longest_run(values: List[Tuple[int, float]]) -> Tuple[int, int, int]: best_start = current_start = 0 best_length = current_length = 1 for index in range(1, len(values)): - if values[index] == values[index - 1]: + if ( + values[index][1] == values[index - 1][1] + and values[index][0] == values[index - 1][0] + 1 + ): current_length += 1 else: current_start, current_length = index, 1 if current_length > best_length: best_start, best_length = current_start, current_length - return best_start, best_length + return values[best_start][0], best_length, values[best_start + best_length - 1][0] def _range(indices: List[int]) -> str: diff --git a/src/datary/recorder.py b/src/datary/recorder.py index 9762ab9..d207ffc 100644 --- a/src/datary/recorder.py +++ b/src/datary/recorder.py @@ -10,13 +10,20 @@ from typing import IO, Any, Dict, Iterable, List from datary import __version__ -from datary.config import SESSION_FORMAT_VERSION -from datary.formats import detect_format -from datary.metrics import summarize_records -from datary.models import Record, RecordOptions +from datary.analysis_store import AnalysisStore +from datary.config import MAX_AUTO_DETECT_BYTES, SESSION_FORMAT_VERSION +from datary.formats import SUPPORTED_FORMATS, AmbiguousFormatError, detect_format +from datary.models import RecordOptions from datary.parsers import parse_lines -from datary.quality import analyze_quality -from datary.utils import atomic_json, infer_type, safe_name, sha256_file, utc_now +from datary.utils import ( + atomic_json, + atomic_text, + bounded_text_lines, + csv_safe_cell, + safe_name, + sha256_file, + utc_now, +) def _session_path(options: RecordOptions) -> Path: @@ -33,16 +40,19 @@ def _session_path(options: RecordOptions) -> Path: def record_stream(stream: IO[str], options: RecordOptions) -> Path: + """Capture a stream and publish a complete session with recoverable overwrite.""" + + _validate_options(options) final_path = _session_path(options) staging = final_path.with_name(f".{final_path.name}.recording-{os.getpid()}") - if staging.exists(): - raise ValueError(f"staging path already exists: {staging}") + backup = final_path.with_name(f".{final_path.name}.backup-{os.getpid()}") + if staging.exists() or backup.exists(): + raise ValueError("a recorder staging or backup path already exists") if final_path.exists(): if not options.overwrite: raise FileExistsError(final_path) if final_path.is_symlink() or not final_path.is_dir(): raise ValueError("refusing to overwrite unsafe session path") - shutil.rmtree(final_path) staging.mkdir(parents=True) (staging / "plots").mkdir() (staging / "reports").mkdir() @@ -50,122 +60,281 @@ def record_stream(stream: IO[str], options: RecordOptions) -> Path: raw_path = staging / "raw.log" records_path = staging / "records.jsonl" invalid_path = staging / "invalid.jsonl" - records: List[Record] = [] - errors: List[Dict[str, Any]] = [] + analysis_path = staging / ".analysis.sqlite3" interrupted = False input_format = options.input_format + valid_count = 0 + invalid_count = 0 + parser_warnings: List[str] = [] + published = False try: - with raw_path.open("w", encoding="utf-8", newline="") as raw: + with ( + AnalysisStore(analysis_path, options.max_fields) as analysis, + raw_path.open("w", encoding="utf-8", newline="") as raw, + records_path.open("w", encoding="utf-8", newline="\n") as clean, + invalid_path.open("w", encoding="utf-8", newline="\n") as invalid, + ): sample_lines: List[str] = [] + bounded_input = iter(bounded_text_lines(stream, options.max_line_bytes)) if input_format is None: - while sum(len(item.encode("utf-8")) for item in sample_lines) < 262_144: - line = stream.readline() - if not line: - break - _validate_line(line, options.max_line_bytes) - raw.write(line) - sample_lines.append(line) - if len(sample_lines) >= 20: - break - input_format, _ = detect_format("".join(sample_lines)) - with records_path.open("w", encoding="utf-8", newline="\n") as clean, invalid_path.open( - "w", encoding="utf-8", newline="\n" - ) as invalid: - line_number = 0 - - def source() -> Iterable[str]: - nonlocal line_number, interrupted - for item in sample_lines: - line_number += 1 + try: + sample_bytes = 0 + while sample_bytes < MAX_AUTO_DETECT_BYTES and len(sample_lines) < 20: + try: + line = next(bounded_input) + except StopIteration: + break + raw.write(line) + sample_lines.append(line) + sample_bytes += len(line.encode("utf-8")) + except KeyboardInterrupt: + interrupted = True + try: + input_format, detection_warnings = detect_format("".join(sample_lines)) + parser_warnings.extend(detection_warnings) + except AmbiguousFormatError: + if not interrupted: + raise + input_format = "jsonl" + parser_warnings.append( + "recording was interrupted before format detection completed; " + "captured input was conservatively parsed as JSON Lines" + ) + + def source() -> Iterable[str]: + nonlocal interrupted + yield from sample_lines + if interrupted: + return + try: + for item in bounded_input: + raw.write(item) yield item - try: - for item in stream: - _validate_line(item, options.max_line_bytes) - raw.write(item) - line_number += 1 - yield item - except KeyboardInterrupt: - interrupted = True + except KeyboardInterrupt: + interrupted = True + try: for result in parse_lines(source(), input_format, options.max_fields): if result.record is not None: - records.append(result.record) - clean.write(json.dumps(result.record, ensure_ascii=False, allow_nan=False) + "\n") + analysis.add(result.record) + clean.write( + json.dumps( + result.record, + ensure_ascii=False, + allow_nan=False, + separators=(",", ":"), + ) + + "\n" + ) + valid_count += 1 else: - error = {"record": len(records) + len(errors) + 1, "reason": result.error} - errors.append(error) + invalid_count += 1 + reason = result.error or "unknown parser error" + error = { + "record": valid_count + invalid_count, + "reason": reason, + } + if len(parser_warnings) < 100: + parser_warnings.append(reason) invalid.write(json.dumps(error, ensure_ascii=False) + "\n") - if (len(records) + len(errors)) % 1000 == 0: + if (valid_count + invalid_count) % 1000 == 0: raw.flush() clean.flush() invalid.flush() - _write_csv(staging / "data.csv", records) - metrics = summarize_records(records, options.time_field) - quality = [finding.to_dict() for finding in analyze_quality(records, options.time_field)] + except KeyboardInterrupt: + interrupted = True + if not interrupted: + try: + for unparsed_line in bounded_input: + raw.write(unparsed_line) + except KeyboardInterrupt: + interrupted = True + raw.flush() + clean.flush() + invalid.flush() + os.fsync(raw.fileno()) + os.fsync(clean.fileno()) + os.fsync(invalid.fileno()) + analysis.finish() + fields = analysis.field_definitions() + requested_roles = { + "time": options.time_field, + "target": options.target_field, + "response": options.response_field, + "sequence": options.sequence_field, + "latency": options.latency_field, + "bytes": options.bytes_field, + } + for role, field in requested_roles.items(): + if field is not None and field not in fields: + raise ValueError(f"{role} field {field!r} does not exist in valid records") + unknown_units = sorted(set(options.units) - set(fields)) + if unknown_units: + raise ValueError("units reference unknown fields: " + ", ".join(unknown_units)) + _write_csv(staging / "data.csv", analysis.records(), sorted(fields)) + metrics = analysis.metrics(options.time_field) + if options.time_field and options.target_field and options.response_field: + metrics["control"] = analysis.control_metrics( + options.time_field, + options.target_field, + options.response_field, + ) + if options.sequence_field and options.latency_field: + metrics["network"] = analysis.network_metrics( + options.sequence_field, + options.latency_field, + options.bytes_field, + options.time_field, + ) + quality = [ + finding.to_dict() + for finding in analysis.quality( + options.time_field, + sequence_field=options.sequence_field, + ) + ] + analysis_path.unlink(missing_ok=True) atomic_json(staging / "metrics.json", metrics) atomic_json(staging / "quality.json", {"findings": quality}) - (staging / "notes.md").write_text("# Notes\n\n", encoding="utf-8") - fields = { - key: infer_type(record.get(key) for record in records) - for key in sorted({key for record in records for key in record}) - } + atomic_text(staging / "notes.md", "# Notes\n\n") ended = utc_now() - hashes = { - name: sha256_file(staging / name) - for name in ("raw.log", "records.jsonl", "data.csv", "metrics.json", "quality.json") - } + hashed_names = ( + "raw.log", + "records.jsonl", + "invalid.jsonl", + "data.csv", + "metrics.json", + "quality.json", + "notes.md", + ) + hashes = {name: sha256_file(staging / name) for name in hashed_names} session_name = final_path.name - manifest = { + manifest: Dict[str, Any] = { "datary_version": __version__, "session_format_version": SESSION_FORMAT_VERSION, "session_name": session_name, "started_at": started, "ended_at": ended, "original_command": options.command, - "working_directory": str(Path.cwd().resolve()) if options.include_path else "", + "working_directory": ( + str(Path.cwd().resolve()) if options.include_path else "" + ), "parameters": options.parameters, "input_format": input_format, + "parser_policy": "conservative-scalars-v1", "fields": fields, - "record_count": len(records) + len(errors), - "valid_record_count": len(records), - "invalid_record_count": len(errors), + "record_count": valid_count + invalid_count, + "valid_record_count": valid_count, + "invalid_record_count": invalid_count, "time_field": options.time_field, + "field_roles": { + name: value + for name, value in { + "target": options.target_field, + "response": options.response_field, + "sequence": options.sequence_field, + "latency": options.latency_field, + "bytes": options.bytes_field, + }.items() + if value is not None + }, "sampling": metrics.get("timing", {}), "units": options.units, - "parser_warnings": [item["reason"] for item in errors[:100]], + "parser_warnings": parser_warnings, "interrupted": interrupted, "hashes": hashes, + "integrity_scope": ( + "SHA-256 corruption detection for manifest and listed artifacts; " + "not cryptographic authenticity" + ), "commands": { "inspect": f"datary inspect {session_name}", "compare": f"datary compare {session_name} OTHER", "replay": f"datary replay {session_name}", "report": f"datary report {session_name}", }, + "command_context": ( + "Run from the session parent directory or set DATARY_WORKSPACE to that directory." + ), } atomic_json(staging / "manifest.json", manifest) - os.replace(staging, final_path) + atomic_text( + staging / "manifest.sha256", + sha256_file(staging / "manifest.json") + "\n", + encoding="ascii", + ) + _publish(staging, final_path, backup, options.overwrite) + published = True return final_path - except BaseException: - if staging.exists(): + finally: + if not published and staging.exists(): shutil.rmtree(staging, ignore_errors=True) + if backup.exists() and not final_path.exists(): + os.replace(backup, final_path) + + +def _publish(staging: Path, final: Path, backup: Path, overwrite: bool) -> None: + moved_original = False + try: + if final.exists(): + if not overwrite: + raise FileExistsError(final) + os.replace(final, backup) + moved_original = True + os.replace(staging, final) + except BaseException: + if moved_original and backup.exists() and not final.exists(): + os.replace(backup, final) raise + if backup.exists(): + shutil.rmtree(backup, ignore_errors=True) -def _validate_line(line: str, maximum: int) -> None: - if len(line.encode("utf-8")) > maximum: - raise ValueError(f"input line exceeds byte limit ({maximum})") +def _validate_options(options: RecordOptions) -> None: + if options.input_format is not None and options.input_format not in SUPPORTED_FORMATS: + raise ValueError(f"unsupported input format: {options.input_format}") + if options.max_line_bytes <= 0: + raise ValueError("max-line-bytes must be positive") + if options.max_fields <= 0: + raise ValueError("max-fields must be positive") + if (options.target_field is None) != (options.response_field is None): + raise ValueError("target-field and response-field must be supplied together") + if options.target_field and not options.time_field: + raise ValueError("control metrics require --time-field") + if (options.sequence_field is None) != (options.latency_field is None): + raise ValueError("sequence-field and latency-field must be supplied together") + if options.bytes_field and not options.sequence_field: + raise ValueError("bytes-field requires sequence and latency fields") + for name, mapping in (("parameters", options.parameters), ("units", options.units)): + if len(mapping) > 1000: + raise ValueError(f"{name} exceed the 1000-entry limit") + if any( + not isinstance(key, str) + or not isinstance(value, str) + or len(key) > 120 + or len(value) > 4096 + for key, value in mapping.items() + ): + raise ValueError(f"{name} entries must be bounded strings") + if options.command is not None and len(options.command) > 16_384: + raise ValueError("original command exceeds 16384 characters") + for field in ( + options.time_field, + options.target_field, + options.response_field, + options.sequence_field, + options.latency_field, + options.bytes_field, + ): + if field is not None and (not field or len(field) > 1024): + raise ValueError("field-role names must contain 1 to 1024 characters") -def _write_csv(path: Path, records: List[Record]) -> None: - fields = sorted({key for record in records for key in record}) +def _write_csv(path: Path, records: Iterable[Dict[str, Any]], fields: List[str]) -> None: with path.open("w", encoding="utf-8", newline="") as stream: writer = csv.DictWriter(stream, fieldnames=fields, extrasaction="ignore") - writer.writeheader() + csv.writer(stream).writerow([csv_safe_cell(field) for field in fields]) for record in records: - writer.writerow({key: _cell(record.get(key)) for key in fields}) - - -def _cell(value: Any) -> Any: - if isinstance(value, (dict, list)): - return json.dumps(value, sort_keys=True, ensure_ascii=False) - return value + writer.writerow({key: csv_safe_cell(record.get(key)) for key in fields}) + stream.flush() + os.fsync(stream.fileno()) diff --git a/src/datary/replay.py b/src/datary/replay.py index 1b02ab7..817909d 100644 --- a/src/datary/replay.py +++ b/src/datary/replay.py @@ -2,12 +2,14 @@ from __future__ import annotations +import csv import json +import math import time from typing import IO, Callable, Optional from datary.sessions import Session -from datary.utils import finite_number +from datary.utils import csv_safe_cell, temporal_number def replay_session( @@ -20,15 +22,17 @@ def replay_session( output_format: str = "jsonl", sleep: Callable[[float], None] = time.sleep, ) -> None: - if speed <= 0: - raise ValueError("speed must be positive") + if speed <= 0 or not math.isfinite(speed): + raise ValueError("speed must be finite and positive") time_field = session.manifest.get("time_field") previous: Optional[float] = None fields = sorted(session.manifest.get("fields", {})) + writer = csv.writer(output, lineterminator="\n") if output_format == "csv" else None if output_format == "csv": - output.write(",".join(fields) + "\n") + assert writer is not None + writer.writerow([csv_safe_cell(field) for field in fields]) for record in session.records(): - current = finite_number(record.get(time_field)) if time_field else None + current = temporal_number(record.get(time_field)) if time_field else None if not no_timing and not virtual and current is not None and previous is not None: delay = max(0.0, (current - previous) / speed) if delay: @@ -37,17 +41,8 @@ def replay_session( if output_format == "jsonl": output.write(json.dumps(record, ensure_ascii=False, allow_nan=False) + "\n") elif output_format == "csv": - output.write(",".join(_csv_cell(record.get(field)) for field in fields) + "\n") + assert writer is not None + writer.writerow([csv_safe_cell(record.get(field)) for field in fields]) else: raise ValueError("replay format must be jsonl or csv") output.flush() - - -def _csv_cell(value: object) -> str: - import csv - import io - - buffer = io.StringIO() - csv.writer(buffer, lineterminator="").writerow([value]) - return buffer.getvalue() - diff --git a/src/datary/reports.py b/src/datary/reports.py index 790031f..f2d008b 100644 --- a/src/datary/reports.py +++ b/src/datary/reports.py @@ -4,20 +4,32 @@ from pathlib import Path from typing import Any, Dict +from urllib.parse import quote from datary import __version__ from datary.inspection import inspect_source from datary.sessions import Session -from datary.utils import atomic_json +from datary.utils import atomic_json, atomic_text, markdown_safe def report_data(session: Session) -> Dict[str, Any]: inspection = inspect_source(session) + plots_directory = session.path / "plots" + plots = ( + [ + path.name + for path in sorted(plots_directory.iterdir(), key=lambda item: item.name) + if path.is_file() and not path.is_symlink() and path.suffix.lower() in {".png", ".svg"} + ] + if plots_directory.is_dir() and not plots_directory.is_symlink() + else [] + ) return { "datary_version": __version__, "session": session.manifest, "inspection": inspection.to_dict(), "integrity_errors": session.verify(), + "plots": plots, "warnings_and_assumptions": [ "Statistics describe recorded data; they do not establish scientific validity.", "Quality checks are heuristics and require domain review.", @@ -31,8 +43,7 @@ def write_report(session: Session, output: Path, report_format: str = "markdown" if report_format == "json": atomic_json(output, data) elif report_format == "markdown": - with output.open("w", encoding="utf-8", newline="\n") as stream: - stream.write(_markdown(data)) + atomic_text(output, _markdown(data)) else: raise ValueError("report format must be markdown or json") return output @@ -44,17 +55,31 @@ def _markdown(data: Dict[str, Any]) -> str: lines = [ f"# Datary report: {_md(str(manifest['session_name']))}", "", - f"- Datary version: `{data['datary_version']}`", - f"- Started: `{manifest.get('started_at')}`", - f"- Ended: `{manifest.get('ended_at')}`", - f"- Input format: `{manifest.get('input_format')}`", + f"- Datary version: {_code(data['datary_version'])}", + f"- Recorded with Datary: {_code(manifest.get('datary_version'))}", + f"- Session format: {_code(manifest.get('session_format_version'))}", + f"- Started: {_code(manifest.get('started_at'))}", + f"- Ended: {_code(manifest.get('ended_at'))}", + f"- Input format: {_code(manifest.get('input_format'))}", f"- Records: {manifest.get('valid_record_count')} valid, {manifest.get('invalid_record_count')} invalid", "", "## Reproduction", "", ] + lines.append(f"- Original command: {_code(manifest.get('original_command') or 'not supplied')}") + lines.append(f"- Working directory: {_code(manifest.get('working_directory', ''))}") + lines.append(f"- Command context: {_md(str(manifest.get('command_context', '')))}") + parameters = manifest.get("parameters", {}) + if parameters: + lines.append("- Parameters:") + for name, value in sorted(parameters.items()): + lines.append(f" - {_code(name)} = {_code(value)}") for name, command in sorted(manifest.get("commands", {}).items()): - lines.append(f"- {name}: `{_md(str(command))}`") + lines.append(f"- {_md(str(name))}: {_code(command)}") + parser_warnings = manifest.get("parser_warnings", []) + if parser_warnings: + lines += ["", "### Parser warnings", ""] + lines.extend(f"- {_md(str(warning))}" for warning in parser_warnings) lines += ["", "## Data schema", "", "| Field | Type | Unit |", "|---|---|---|"] for field, kind in inspection["fields"].items(): lines.append(f"| {_md(field)} | {_md(kind)} | {_md(inspection['units'].get(field, ''))} |") @@ -66,19 +91,97 @@ def _markdown(data: Dict[str, Any]) -> str: lines.append(f"- Median: {metric.get('median')}") lines.append(f"- Range: {metric.get('minimum')} to {metric.get('maximum')}") lines.append("") + lines += ["## Timing", ""] + if inspection["timing"]: + for name, value in sorted(inspection["timing"].items()): + lines.append(f"- {_md(str(name))}: {_code(value)}") + else: + lines.append("No usable time field was supplied.") + lines.append("") + lines += ["## Engineering metrics", ""] + if inspection.get("engineering"): + for category, metrics in sorted(inspection["engineering"].items()): + lines.append(f"### {_md(str(category).title())}") + lines.append("") + for name, value in sorted(metrics.items()): + if name in {"assumptions", "warnings", "required_fields"}: + continue + lines.append(f"- {_md(str(name))}: {_code(value)}") + required = metrics.get("required_fields", {}) + if required: + lines.append( + "- Required fields: " + + ", ".join( + f"{_md(str(role))}={_code(field)}" + for role, field in sorted(required.items()) + if field is not None + ) + ) + for assumption in metrics.get("assumptions", []): + lines.append(f"- Assumption: {_md(str(assumption))}") + for warning in metrics.get("warnings", []): + lines.append(f"- Warning: {_md(str(warning))}") + lines.append("") + else: + lines.append("No control or network field roles were supplied.") + lines.append("") lines += ["## Quality findings", ""] if not inspection["quality"]: lines.append("No heuristic findings.") for finding in inspection["quality"]: - lines.append(f"- **{_md(finding['check_id'])}** ({finding['severity']}): {_md(finding['explanation'])} Field: `{_md(str(finding.get('field') or '-'))}`; affected: {_md(finding['affected'])}.") + lines.append( + f"- **{_md(finding['check_id'])}** ({finding['severity']}): {_md(finding['explanation'])} Field: {_code(finding.get('field') or '-')}; affected: {_md(finding['affected'])}." + ) + lines.append( + f" - Evidence: {_code(finding.get('evidence'))}; " + f"threshold: {_code(finding.get('threshold'))}" + ) + assumptions = finding.get("assumptions") or [] + if assumptions: + lines.append(" - Assumptions: " + "; ".join(_md(str(item)) for item in assumptions)) + lines.append( + " - Suggested investigation: " + + _md(str(finding.get("suggested_investigation") or "Review the raw input.")) + ) lines += ["", "## Input hashes", ""] for name, digest in sorted(manifest.get("hashes", {}).items()): - lines.append(f"- `{_md(name)}`: `{digest}`") + lines.append(f"- {_code(name)}: {_code(digest)}") + lines += ["", "## Integrity verification", ""] + integrity_errors = data["integrity_errors"] + if integrity_errors: + lines.append("**FAILED:** this session did not pass integrity verification.") + lines.extend(f"- {_md(str(item))}" for item in integrity_errors) + else: + lines.append("All manifest-listed artefacts passed SHA-256 corruption checks.") + lines.append( + "These checks detect accidental changes; they do not prove cryptographic authenticity." + ) lines += ["", "## Warnings and assumptions", ""] lines.extend(f"- {_md(item)}" for item in data["warnings_and_assumptions"]) - lines += ["", "## Plots", "", "Plots are stored in the session `plots/` directory.", ""] + lines += ["", "## Plots", ""] + if data["plots"]: + for filename in data["plots"]: + lines.append(f"- [{_md(str(filename))}](../plots/{quote(str(filename))})") + else: + lines.append("No plots have been generated for this session.") + lines.append("") return "\n".join(lines) def _md(value: str) -> str: - return value.replace("\\", "\\\\").replace("|", "\\|").replace("`", "\\`").replace("\n", " ") + return markdown_safe(value) + + +def _code(value: object) -> str: + text = str(value).replace("\r", " ").replace("\n", " ") + longest = 0 + current = 0 + for character in text: + if character == "`": + current += 1 + longest = max(longest, current) + else: + current = 0 + fence = "`" * (longest + 1) + content = f" {text} " if text.startswith(("`", " ")) or text.endswith(("`", " ")) else text + return f"{fence}{content}{fence}" diff --git a/src/datary/sessions.py b/src/datary/sessions.py index ec6cc27..86f3694 100644 --- a/src/datary/sessions.py +++ b/src/datary/sessions.py @@ -4,10 +4,12 @@ import json from dataclasses import dataclass +from datetime import datetime from pathlib import Path from typing import Any, Dict, Iterator, List, Union -from datary.config import SESSION_FORMAT_VERSION, default_workspace +from datary.config import SUPPORTED_SESSION_FORMAT_VERSIONS, default_workspace +from datary.formats import SUPPORTED_FORMATS from datary.models import Record from datary.utils import json_lines, safe_name, sha256_file @@ -20,26 +22,60 @@ class Session: @classmethod def open(cls, source: Union[str, Path], workspace: Union[str, Path, None] = None) -> "Session": candidate = Path(source) - root = Path(workspace) if workspace else default_workspace() + root = (Path(workspace) if workspace else default_workspace()).resolve() + if candidate.is_symlink(): + raise ValueError("session directories may not be symbolic links") if not candidate.exists(): candidate = root / safe_name(str(source)) + try: + candidate.relative_to(root) + except ValueError as error: + raise ValueError("named session escapes the configured workspace") from error + if _contains_symlink(candidate): + raise ValueError("session paths may not contain symbolic links") candidate = candidate.resolve() - if candidate.is_symlink(): - raise ValueError("session directories may not be symbolic links") manifest_path = candidate / "manifest.json" if not candidate.is_dir() or not manifest_path.is_file() or manifest_path.is_symlink(): raise ValueError(f"not a Datary session: {source}") + if manifest_path.stat().st_size > 1_048_576: + raise ValueError("session manifest exceeds the 1 MiB safety limit") try: value = json.loads(manifest_path.read_text(encoding="utf-8")) except (OSError, json.JSONDecodeError) as error: raise ValueError(f"broken session manifest: {error}") from error if not isinstance(value, dict): raise ValueError("session manifest must be an object") - if value.get("session_format_version") != SESSION_FORMAT_VERSION: + if value.get("session_format_version") not in SUPPORTED_SESSION_FORMAT_VERSIONS: raise ValueError("unsupported session format version") for required in ("session_name", "record_count", "hashes"): if required not in value: raise ValueError(f"manifest is missing {required}") + if value.get("session_format_version") == "2": + for required in ("valid_record_count", "invalid_record_count"): + if required not in value: + raise ValueError(f"manifest is missing {required}") + if safe_name(str(value["session_name"])) != value["session_name"]: + raise ValueError("manifest has an unsafe session name") + count_names = ["record_count"] + if "valid_record_count" in value: + count_names.append("valid_record_count") + if "invalid_record_count" in value: + count_names.append("invalid_record_count") + for count_name in count_names: + count = value[count_name] + if not isinstance(count, int) or isinstance(count, bool) or count < 0: + raise ValueError(f"manifest {count_name} must be a non-negative integer") + if ( + "valid_record_count" in value + and "invalid_record_count" in value + and value["record_count"] != value["valid_record_count"] + value["invalid_record_count"] + ): + raise ValueError("manifest record counts are inconsistent") + hashes = value["hashes"] + if not isinstance(hashes, dict) or len(hashes) > 100: + raise ValueError("manifest hashes must be a bounded object") + if value.get("session_format_version") == "2": + _validate_v2_manifest(value) return cls(candidate, value) @property @@ -54,11 +90,29 @@ def records(self) -> Iterator[Record]: def verify(self) -> List[str]: errors: List[str] = [] + manifest_digest_path = self.path / "manifest.sha256" + if self.manifest.get("session_format_version") == "2": + if not manifest_digest_path.is_file() or manifest_digest_path.is_symlink(): + errors.append("missing or unsafe file: manifest.sha256") + else: + expected_manifest = manifest_digest_path.read_text(encoding="ascii").strip() + if ( + len(expected_manifest) != 64 + or any(character not in "0123456789abcdef" for character in expected_manifest) + or expected_manifest != sha256_file(self.path / "manifest.json") + ): + errors.append("hash mismatch: manifest.json") hashes = self.manifest.get("hashes", {}) if not isinstance(hashes, dict): return ["manifest hashes must be an object"] for relative, expected in hashes.items(): - if not isinstance(relative, str) or Path(relative).name != relative: + if ( + not isinstance(relative, str) + or Path(relative).name != relative + or not isinstance(expected, str) + or len(expected) != 64 + or any(character not in "0123456789abcdef" for character in expected) + ): errors.append(f"unsafe hash path: {relative!r}") continue target = self.path / relative @@ -66,9 +120,132 @@ def verify(self) -> List[str]: errors.append(f"missing or unsafe file: {relative}") elif sha256_file(target) != expected: errors.append(f"hash mismatch: {relative}") + if self.manifest.get("session_format_version") == "2": + required = { + "raw.log", + "records.jsonl", + "invalid.jsonl", + "data.csv", + "metrics.json", + "quality.json", + "notes.md", + } + for missing in sorted(required - set(hashes)): + errors.append(f"unhashed required file: {missing}") + if "records.jsonl" not in { + error.removeprefix("hash mismatch: ") + for error in errors + if error.startswith("hash mismatch: ") + }: + try: + valid_count = sum(1 for _ in self.records()) + if valid_count != self.manifest["valid_record_count"]: + errors.append( + "record count mismatch: records.jsonl " + f"has {valid_count}, manifest has " + f"{self.manifest['valid_record_count']}" + ) + except (OSError, ValueError, json.JSONDecodeError) as error: + errors.append(f"invalid records.jsonl: {error}") + try: + invalid_count = _count_invalid_records(self.path / "invalid.jsonl") + if invalid_count != self.manifest["invalid_record_count"]: + errors.append( + "record count mismatch: invalid.jsonl " + f"has {invalid_count}, manifest has " + f"{self.manifest['invalid_record_count']}" + ) + except (OSError, ValueError, json.JSONDecodeError) as error: + errors.append(f"invalid invalid.jsonl: {error}") return errors +def _contains_symlink(path: Path) -> bool: + """Check each existing path component before resolution changes its identity.""" + + absolute = path.absolute() + current = Path(absolute.anchor) + for part in absolute.parts[1:]: + current /= part + try: + if current.is_symlink(): + return True + except OSError: + return True + return False + + +def _validate_v2_manifest(value: Dict[str, Any]) -> None: + for name in ( + "datary_version", + "parser_policy", + "working_directory", + "integrity_scope", + "command_context", + ): + if not isinstance(value.get(name), str): + raise ValueError(f"manifest {name} must be a string") + if value["parser_policy"] != "conservative-scalars-v1": + raise ValueError("manifest uses an unsupported parser policy") + if value.get("original_command") is not None and not isinstance( + value.get("original_command"), str + ): + raise ValueError("manifest original command must be a string or null") + if not isinstance(value.get("interrupted"), bool): + raise ValueError("manifest interrupted flag must be boolean") + sampling = value.get("sampling") + if not isinstance(sampling, dict) or len(sampling) > 100: + raise ValueError("manifest sampling must be a bounded object") + if value.get("input_format") not in SUPPORTED_FORMATS: + raise ValueError("manifest has an unsupported input format") + for name in ("fields", "units", "parameters", "commands", "field_roles"): + mapping = value.get(name) + if not isinstance(mapping, dict) or len(mapping) > 1000: + raise ValueError(f"manifest {name} must be a bounded object") + if any( + not isinstance(key, str) or not isinstance(item, str) for key, item in mapping.items() + ): + raise ValueError(f"manifest {name} entries must be strings") + parser_warnings = value.get("parser_warnings", []) + if not isinstance(parser_warnings, list) or any( + not isinstance(item, str) for item in parser_warnings + ): + raise ValueError("manifest parser warnings must be strings") + if len(parser_warnings) > 100: + raise ValueError("manifest contains too many parser warnings") + if value.get("time_field") is not None and not isinstance(value.get("time_field"), str): + raise ValueError("manifest time field must be a string or null") + for timestamp_name in ("started_at", "ended_at"): + timestamp = value.get(timestamp_name) + if not isinstance(timestamp, str): + raise ValueError(f"manifest {timestamp_name} must be an ISO 8601 string") + try: + parsed = datetime.fromisoformat(timestamp.replace("Z", "+00:00")) + except ValueError as error: + raise ValueError(f"manifest {timestamp_name} is not valid ISO 8601") from error + if parsed.tzinfo is None or parsed.utcoffset() is None: + raise ValueError(f"manifest {timestamp_name} must include a timezone") + + +def _count_invalid_records(path: Path) -> int: + count = 0 + with path.open("r", encoding="utf-8") as stream: + for line_number, line in enumerate(stream, 1): + if not line.strip(): + continue + if len(line.encode("utf-8")) > 1_048_576: + raise ValueError(f"line {line_number} exceeds the safety limit") + value = json.loads(line) + if ( + not isinstance(value, dict) + or not isinstance(value.get("record"), int) + or not isinstance(value.get("reason"), str) + ): + raise ValueError(f"line {line_number} has an invalid error record") + count += 1 + return count + + def list_sessions(workspace: Union[str, Path, None] = None) -> List[Session]: root = Path(workspace) if workspace else default_workspace() if not root.exists(): @@ -81,4 +258,3 @@ def list_sessions(workspace: Union[str, Path, None] = None) -> List[Session]: except ValueError: pass return sessions - diff --git a/src/datary/utils.py b/src/datary/utils.py index c7318ad..e08446f 100644 --- a/src/datary/utils.py +++ b/src/datary/utils.py @@ -3,6 +3,7 @@ from __future__ import annotations import hashlib +import html import json import math import os @@ -10,7 +11,7 @@ import tempfile from datetime import datetime, timezone from pathlib import Path -from typing import Any, Dict, Iterable, Iterator, List, Optional +from typing import IO, Any, Callable, Dict, Iterable, Iterator, List, NoReturn, Optional def utc_now() -> str: @@ -32,6 +33,24 @@ def atomic_json(path: Path, value: Any) -> None: os.unlink(temporary) +def atomic_text(path: Path, value: str, encoding: str = "utf-8") -> None: + """Atomically replace a regular text file without following an output symlink.""" + + path.parent.mkdir(parents=True, exist_ok=True) + if path.is_symlink(): + raise ValueError("output path may not be a symbolic link") + fd, temporary = tempfile.mkstemp(prefix=f".{path.name}.", dir=str(path.parent)) + try: + with os.fdopen(fd, "w", encoding=encoding, newline="\n") as stream: + stream.write(value) + stream.flush() + os.fsync(stream.fileno()) + os.replace(temporary, path) + finally: + if os.path.exists(temporary): + os.unlink(temporary) + + def sha256_file(path: Path) -> str: digest = hashlib.sha256() with path.open("rb") as stream: @@ -40,12 +59,49 @@ def sha256_file(path: Path) -> str: return digest.hexdigest() +def bounded_text_lines(stream: IO[str], maximum: int) -> Iterator[str]: + """Read physical lines without first allocating an unbounded line.""" + + if maximum <= 0: + raise ValueError("line limit must be positive") + while True: + line = stream.readline(maximum + 1) + if not line: + return + _validate_text_line(line, maximum) + if line.endswith(("\n", "\r")): + yield line + continue + continuation = stream.readline(1) + if not continuation: + yield line + return + combined = line + continuation + _validate_text_line(combined, maximum) + if continuation in {"\n", "\r"}: + yield combined + continue + raise ValueError(f"input line exceeds byte limit ({maximum})") + + +def _validate_text_line(line: str, maximum: int) -> None: + if len(line.encode("utf-8")) > maximum: + raise ValueError(f"input line exceeds byte limit ({maximum})") + + def safe_name(value: str) -> str: if not value or value in {".", ".."} or Path(value).name != value: raise ValueError("session name must be a single, non-empty path component") cleaned = re.sub(r"[^\w.-]", "-", value, flags=re.UNICODE).strip(".-") if not cleaned: raise ValueError("session name contains no safe characters") + if len(cleaned) > 120: + raise ValueError("session name exceeds 120 characters") + reserved = {"CON", "PRN", "AUX", "NUL"} + reserved.update(f"COM{number}" for number in range(1, 10)) + reserved.update(f"LPT{number}" for number in range(1, 10)) + if cleaned.split(".", 1)[0].upper() in reserved: + raise ValueError("session name is reserved by Windows filesystems") return cleaned @@ -58,15 +114,109 @@ def safe_output(base: Path, requested: Path) -> Path: return resolved +def safe_filename_component(value: str, fallback: str = "output") -> str: + """Return a display-derived filename component with no path semantics.""" + + cleaned = re.sub(r"[^\w.-]+", "-", value, flags=re.UNICODE).strip(".-") + if not cleaned: + cleaned = fallback + return cleaned[:120] + + +def csv_safe_cell(value: Any) -> Any: + """Render nested values and neutralize spreadsheet formula prefixes. + + The canonical value remains unchanged in ``records.jsonl``. CSV is a + convenience export and receives a leading apostrophe for formula-like + strings so opening it in a spreadsheet does not execute cell formulas. + """ + + if isinstance(value, (dict, list)): + rendered = json.dumps(value, sort_keys=True, ensure_ascii=False) + else: + rendered = value + if isinstance(rendered, str): + stripped = rendered.lstrip() + if stripped.startswith(("=", "+", "-", "@")): + return "'" + rendered + return rendered + + def finite_number(value: Any) -> Optional[float]: if isinstance(value, bool): return None if isinstance(value, (int, float)): + if isinstance(value, int) and abs(value) > 2**53: + return None number = float(value) return number if math.isfinite(number) else None return None +def temporal_number(value: Any) -> Optional[float]: + """Return numeric seconds for a finite number or timezone-aware ISO 8601 value.""" + + number = finite_number(value) + if number is not None: + return number + if not isinstance(value, str): + return None + candidate = value.strip() + if not candidate: + return None + if candidate.endswith(("Z", "z")): + candidate = candidate[:-1] + "+00:00" + try: + parsed = datetime.fromisoformat(candidate) + except ValueError: + return None + if parsed.tzinfo is None or parsed.utcoffset() is None: + return None + return parsed.timestamp() + + +def terminal_safe(value: object, maximum: int = 4096) -> str: + """Render untrusted text without emitting terminal control sequences.""" + + text = str(value) + rendered: List[str] = [] + for character in text: + code = ord(character) + if code < 32 or 127 <= code <= 159: + rendered.append(f"\\x{code:02x}") + else: + rendered.append(character) + result = "".join(rendered) + return result if len(result) <= maximum else result[: maximum - 1] + "…" + + +def markdown_safe(value: object) -> str: + """Escape untrusted text for ordinary Markdown and Markdown tables.""" + + escaped = html.escape(str(value), quote=True) + for character in ( + "\\", + "`", + "*", + "_", + "{", + "}", + "[", + "]", + "(", + ")", + "#", + "+", + "-", + ".", + "!", + "|", + ">", + ): + escaped = escaped.replace(character, f"\\{character}") + return escaped.replace("\r", " ").replace("\n", " ") + + def infer_type(values: Iterable[Any]) -> str: names = {type(value).__name__ for value in values if value is not None} if not names: @@ -81,6 +231,8 @@ def infer_type(values: Iterable[Any]) -> str: def parse_key_values(items: List[str]) -> Dict[str, str]: + if len(items) > 1000: + raise ValueError("too many KEY=VALUE options (maximum 1000)") result: Dict[str, str] = {} for item in items: if "=" not in item: @@ -88,6 +240,8 @@ def parse_key_values(items: List[str]) -> Dict[str, str]: key, value = item.split("=", 1) if not key: raise ValueError("key may not be empty") + if len(key) > 120 or len(value) > 4096: + raise ValueError("KEY=VALUE option exceeds the safety limit") result[key] = value return result @@ -99,13 +253,39 @@ def sparkline(values: List[float], ascii_only: bool = False) -> str: low, high = min(values), max(values) if high == low: return chars[len(chars) // 2] * len(values) - return "".join(chars[min(len(chars) - 1, int((v - low) / (high - low) * len(chars)))] for v in values) + return "".join( + chars[min(len(chars) - 1, int((v - low) / (high - low) * len(chars)))] for v in values + ) -def json_lines(path: Path) -> Iterator[Dict[str, Any]]: +def json_lines(path: Path, max_line_bytes: int = 16 * 1024 * 1024) -> Iterator[Dict[str, Any]]: with path.open("r", encoding="utf-8") as stream: - for line in stream: + for line_number, line in enumerate(stream, 1): if line.strip(): - value = json.loads(line) - if isinstance(value, dict): - yield value + if len(line.encode("utf-8")) > max_line_bytes: + raise ValueError( + f"records line {line_number} exceeds byte limit ({max_line_bytes})" + ) + try: + value = json.loads( + line, + parse_constant=_constant_rejector(line_number), + ) + except json.JSONDecodeError as error: + raise ValueError( + f"invalid records JSON at line {line_number}: {error.msg}" + ) from error + if not isinstance(value, dict): + raise ValueError(f"records line {line_number} must contain a JSON object") + yield value + + +def _constant_rejector(line_number: int) -> Callable[[str], NoReturn]: + def reject(constant: str) -> NoReturn: + _reject_constant(constant, line_number) + + return reject + + +def _reject_constant(constant: str, line_number: int) -> NoReturn: + raise ValueError(f"non-finite JSON number {constant!r} in records line {line_number}") diff --git a/tests/test_cli.py b/tests/test_cli.py index d5bf83f..fae8036 100644 --- a/tests/test_cli.py +++ b/tests/test_cli.py @@ -1,5 +1,8 @@ +import importlib from pathlib import Path +import pytest + from datary.cli import main, parser @@ -33,3 +36,14 @@ def test_quality_field_options_are_repeatable() -> None: ) assert arguments.monotonic_field == ["distance"] assert arguments.counter_field == ["packets"] + + +def test_doctor_treats_plotting_as_optional( + tmp_path: Path, capsys: pytest.CaptureFixture[str], monkeypatch: pytest.MonkeyPatch +) -> None: + def unavailable(_name: str) -> object: + raise ImportError + + monkeypatch.setattr(importlib, "import_module", unavailable) + assert main(["--workspace", str(tmp_path), "doctor"]) == 0 + assert "OPTIONAL-MISSING optional_plotting_available" in capsys.readouterr().out diff --git a/tests/test_comparison.py b/tests/test_comparison.py index 374b87e..8986c08 100644 --- a/tests/test_comparison.py +++ b/tests/test_comparison.py @@ -19,3 +19,41 @@ def test_no_goal_is_honest(tmp_path: Path) -> None: result = compare_sessions([tmp_path / "a", tmp_path / "b"]) assert "no comparison goal" in result.warnings[0] + +def test_duplicate_source_labels_do_not_overwrite(tmp_path: Path) -> None: + session = record_stream(io.StringIO('{"x":1}\n'), RecordOptions("same", tmp_path, "jsonl")) + result = compare_sessions([session, session], ["x"]) + assert list(result.fields["x"]["means"]) == [str(session), f"{session}#2"] + + +def test_unit_mismatch_blocks_goal_claim(tmp_path: Path) -> None: + first = record_stream( + io.StringIO('{"x":1}\n'), + RecordOptions("metres", tmp_path, "jsonl", units={"x": "m"}), + ) + second = record_stream( + io.StringIO('{"x":2}\n'), + RecordOptions("seconds", tmp_path, "jsonl", units={"x": "s"}), + ) + result = compare_sessions([first, second], ["x"], "higher:x") + assert result.fields["x"]["comparable"] is False + assert "improvement_percent_vs_first" not in result.fields["x"] + assert any("incompatible declared units" in warning for warning in result.warnings) + + +def test_different_sampling_rates_are_not_claimed_comparable(tmp_path: Path) -> None: + first = record_stream( + io.StringIO('{"t":0,"x":2}\n{"t":1,"x":2}\n{"t":2,"x":2}\n'), + RecordOptions("slow", tmp_path, "jsonl", time_field="t"), + ) + second = record_stream( + io.StringIO( + '{"t":0,"x":1}\n{"t":0.5,"x":1}\n{"t":1,"x":1}\n{"t":1.5,"x":1}\n{"t":2,"x":1}\n' + ), + RecordOptions("fast", tmp_path, "jsonl", time_field="t"), + ) + result = compare_sessions([first, second], ["x"], "lower:x") + assert result.fields["x"]["comparable"] is False + assert "improvement_percent_vs_first" not in result.fields["x"] + assert result.timing["shared_relative_range"] == [0.0, 2.0] + assert any("sampling rates differ" in warning for warning in result.warnings) diff --git a/tests/test_conversion.py b/tests/test_conversion.py new file mode 100644 index 0000000..c1e5c17 --- /dev/null +++ b/tests/test_conversion.py @@ -0,0 +1,41 @@ +import json +import os +from pathlib import Path + +import pytest + +from datary.conversion import convert_source + + +def test_conversion_streams_multiline_csv_and_writes_sidecar(tmp_path: Path) -> None: + source = tmp_path / "source.csv" + source.write_text('name,value\n"two\nlines",1\nbroken\n', encoding="utf-8") + output = tmp_path / "output.jsonl" + valid, invalid = convert_source(source, output, "jsonl", "csv") + assert (valid, invalid) == (1, 1) + assert json.loads(output.read_text(encoding="utf-8")) == { + "name": f"two{os.linesep}lines", + "value": 1, + } + sidecar = json.loads(output.with_suffix(".jsonl.invalid.json").read_text(encoding="utf-8")) + assert sidecar["invalid_record_count"] == 1 + + +def test_conversion_never_silently_overwrites(tmp_path: Path) -> None: + source = tmp_path / "source.jsonl" + source.write_text('{"x":1}\n', encoding="utf-8") + output = tmp_path / "output.csv" + output.write_text("keep\n", encoding="utf-8") + with pytest.raises(FileExistsError): + convert_source(source, output, "csv", "jsonl") + assert output.read_text(encoding="utf-8") == "keep\n" + + +def test_clean_conversion_replaces_stale_invalid_count(tmp_path: Path) -> None: + source = tmp_path / "source.jsonl" + source.write_text('{"x":1}\n', encoding="utf-8") + output = tmp_path / "output.csv" + sidecar = output.with_suffix(".csv.invalid.json") + sidecar.write_text('{"invalid_record_count": 99}\n', encoding="utf-8") + convert_source(source, output, "csv", "jsonl") + assert json.loads(sidecar.read_text(encoding="utf-8"))["invalid_record_count"] == 0 diff --git a/tests/test_formats.py b/tests/test_formats.py index 358807c..3e395c6 100644 --- a/tests/test_formats.py +++ b/tests/test_formats.py @@ -23,3 +23,15 @@ def test_ambiguous(text: str) -> None: with pytest.raises(AmbiguousFormatError): detect_format(text) + +def test_numeric_extension_does_not_override_ambiguity() -> None: + with pytest.raises(AmbiguousFormatError): + detect_format("1,2\n3,4\n", "numbers.csv") + with pytest.raises(AmbiguousFormatError): + detect_format("1\t2\n3\t4\n", "numbers.tsv") + + +def test_incomplete_json_array_sample_is_still_detected() -> None: + detected, warnings = detect_format('[{"x":1},\n') + assert detected == "json" + assert warnings diff --git a/tests/test_generators.py b/tests/test_generators.py index 81f830c..babdaa4 100644 --- a/tests/test_generators.py +++ b/tests/test_generators.py @@ -1,3 +1,5 @@ +import pytest + from datary.generators import PROFILES, generate_records @@ -8,3 +10,33 @@ def test_all_profiles_deterministic() -> None: assert first == second assert first + +def test_explicit_zero_anomaly_rates_are_honoured() -> None: + missing = list(generate_records("missing-samples", duration=2, sample_rate=10, missing_rate=0)) + assert len(missing) == 21 + duplicated = list( + generate_records("duplicate-samples", duration=2, sample_rate=10, duplicate_rate=0) + ) + assert len(duplicated) == 21 + + +def test_pid_error_matches_emitted_response() -> None: + record = next(generate_records("pid-response", seed=7, noise=0.2)) + assert record["error"] == record["target"] - record["response"] + + +def test_named_anomaly_profiles_include_their_default_anomaly() -> None: + missing = list(generate_records("missing-samples", seed=999, duration=1, sample_rate=5)) + assert len(missing) < 6 + duplicated = list(generate_records("duplicate-samples", seed=999, duration=1, sample_rate=5)) + timestamps = [record["timestamp"] for record in duplicated] + assert len(timestamps) > len(set(timestamps)) + + +def test_generator_rejects_nonfinite_and_unbounded_options() -> None: + with pytest.raises(ValueError): + list(generate_records("sine", duration=float("nan"))) + with pytest.raises(ValueError): + list(generate_records("sine", noise=-1)) + with pytest.raises(ValueError, match="safety limit"): + list(generate_records("sine", duration=1_000_000, sample_rate=100)) diff --git a/tests/test_metrics.py b/tests/test_metrics.py index 2ba93fd..736b9d5 100644 --- a/tests/test_metrics.py +++ b/tests/test_metrics.py @@ -1,6 +1,13 @@ import pytest -from datary.metrics import control_metrics, network_metrics, numeric_summary, timing_metrics +from datary.metrics import ( + control_metrics, + network_metrics, + numeric_summary, + summarize_records, + timing_metrics, +) +from datary.models import Record def test_numeric_and_timing() -> None: @@ -12,15 +19,26 @@ def test_numeric_and_timing() -> None: assert timing["gap_count"] == 1 +def test_order_sensitive_metrics_keep_observation_order() -> None: + summary = numeric_summary([3.0, 1.0, 2.0]) + assert summary["minimum"] == 1.0 + assert summary["median"] == 2.0 + assert summary["rate_of_change"] == -1.0 + assert summary["mean_absolute_difference"] == 1.5 + assert summary["sparkline_values"] == [3.0, 1.0, 2.0] + + def test_control_metrics() -> None: - records = [{"t": i / 10, "target": 1.0, "response": 1 - 2.71828 ** (-i / 10)} for i in range(51)] + records: list[Record] = [ + {"t": i / 10, "target": 1.0, "response": 1 - 2.71828 ** (-i / 10)} for i in range(51) + ] result = control_metrics(records, "t", "target", "response") assert result["mean_absolute_error"] > 0 assert result["rise_time"] == pytest.approx(2.2, abs=0.2) def test_network_metrics() -> None: - records = [ + records: list[Record] = [ {"seq": 1, "latency": 10, "bytes": 100}, {"seq": 2, "latency": 20, "bytes": 100}, {"seq": 2, "latency": 20, "bytes": 100}, @@ -30,3 +48,34 @@ def test_network_metrics() -> None: assert result["packet_loss_estimate"] == 0.25 assert result["duplicate_packet_rate"] == 0.25 + +def test_iso8601_timing_and_network_throughput() -> None: + records: list[Record] = [ + { + "timestamp": "2026-01-01T00:00:00+00:00", + "seq": 1, + "latency": 10, + "bytes": 100, + }, + { + "timestamp": "2026-01-01T00:00:02Z", + "seq": 2, + "latency": 20, + "bytes": 100, + }, + ] + timing = summarize_records(records, "timestamp")["timing"] + assert timing["mean_interval"] == 2 + network = network_metrics(records, "seq", "latency", "bytes", "timestamp") + assert network["throughput_bytes_per_second"] == 100 + + +def test_non_adjacent_duplicate_timestamps_are_counted() -> None: + assert timing_metrics([0, 1, 0])["duplicate_timestamp_count"] == 1 + + +def test_single_timestamp_still_has_a_time_range() -> None: + timing = timing_metrics([5]) + assert timing["start"] == timing["end"] == 5 + assert timing["duration"] == 0 + assert timing["mean_interval"] is None diff --git a/tests/test_parsers.py b/tests/test_parsers.py index e7bac40..ca9c4bf 100644 --- a/tests/test_parsers.py +++ b/tests/test_parsers.py @@ -1,3 +1,8 @@ +import random +import string +from collections.abc import Iterator +from pathlib import Path + import pytest from datary.parsers import parse_lines @@ -28,3 +33,54 @@ def test_csv_record_length_change() -> None: results = list(parse_lines(["a,b\n", "1\n"], "csv")) assert results[0].error and "expected 2" in results[0].error + +def test_multiline_csv_and_bom() -> None: + results = list(parse_lines(["\ufeffa,b\n", '1,"two\n', 'lines"\n'], "csv")) + assert len(results) == 1 + assert results[0].record == {"a": 1, "b": "two\nlines"} + + +def test_bom_jsonl_conservative_scalars_and_nonfinite() -> None: + assert list(parse_lines(['\ufeff{"a":1}\n'], "jsonl"))[0].record == {"a": 1} + record = list(parse_lines(["00123,NA,1e3\n"], "stream"))[0].record + assert record == {"field_1": "00123", "field_2": "NA", "field_3": 1000.0} + assert "non-finite" in (list(parse_lines(['{"a":NaN}\n'], "jsonl"))[0].error or "") + + +def test_json_array_is_incremental() -> None: + consumed: list[int] = [] + + def chunks() -> Iterator[str]: + consumed.append(1) + yield '[{"a":1},' + consumed.append(2) + yield '{"a":2}]' + + parsed = parse_lines(chunks(), "json") + assert next(parsed).record == {"a": 1} + assert consumed == [1] + assert [item.record for item in parsed] == [{"a": 2}] + + +def test_duplicate_json_keys_are_rejected() -> None: + result = list(parse_lines(['{"a":1,"a":2}\n'], "jsonl"))[0] + assert result.record is None + assert "duplicate JSON object key" in (result.error or "") + + +def test_deterministic_hostile_text_corpus_never_executes_or_crashes(tmp_path: Path) -> None: + rng = random.Random(20260729) + alphabet = string.printable + "\x00\x1b\u202e\u2603" + corpus = [ + "".join(rng.choice(alphabet) for _ in range(rng.randrange(0, 160))) + "\n" + for _ in range(250) + ] + marker = tmp_path / "must-not-exist" + payloads = corpus + [ + f'__import__("pathlib").Path("{marker.as_posix()}").touch()\n', + "=cmd|' /C calc'!A0\n", + '{"__class__":{"__mro__":"ignored"}}\n', + ] + for kind in ("csv", "tsv", "jsonl", "whitespace", "keyvalue", "stream"): + list(parse_lines(payloads, kind)) + assert not marker.exists() diff --git a/tests/test_plotting.py b/tests/test_plotting.py index acc5706..44a4adb 100644 --- a/tests/test_plotting.py +++ b/tests/test_plotting.py @@ -1,15 +1,19 @@ +import importlib from pathlib import Path +from typing import Any import pytest +from datary.models import Record from datary.plotting import create_plot def test_png_and_svg(tmp_path: Path) -> None: - records = [{"t": 0, "x": 1}, {"t": 1, "x": 2}] + records: list[Record] = [{"t": 0, "x": 1}, {"t": 1, "x": 2}] for suffix in (".png", ".svg"): output = create_plot(records, ["x"], tmp_path / f"plot{suffix}", time_field="t") assert output.stat().st_size > 100 with pytest.raises(FileExistsError): create_plot(records, ["x"], output) - + matplotlib: Any = importlib.import_module("matplotlib") + assert str(matplotlib.get_backend()).lower() == "agg" diff --git a/tests/test_quality.py b/tests/test_quality.py index 9459343..d0b4fde 100644 --- a/tests/test_quality.py +++ b/tests/test_quality.py @@ -1,12 +1,13 @@ +from datary.models import Record from datary.quality import analyze_quality -def ids(records: list[dict[str, object]], time_field: str = "t") -> set[str]: +def ids(records: list[Record], time_field: str = "t") -> set[str]: return {item.check_id for item in analyze_quality(records, time_field)} def test_quality_checks() -> None: - records = [ + records: list[Record] = [ {"t": 0, "x": 1}, {"t": 1, "x": 1}, {"t": 1, "x": 1}, @@ -15,17 +16,61 @@ def test_quality_checks() -> None: {"t": 11, "x": 100}, ] result = ids(records) - assert {"missing-values", "duplicate-timestamps", "timestamps-backwards", "large-timing-gaps"} <= result + assert { + "missing-values", + "duplicate-timestamps", + "timestamps-backwards", + "large-timing-gaps", + } <= result def test_frozen_and_constant() -> None: - frozen = [{"t": i, "x": 2 if i < 6 else i} for i in range(10)] + frozen: list[Record] = [{"t": i, "x": 2 if i < 6 else i} for i in range(10)] assert "frozen-values" in ids(frozen) - assert "constant-signal" in ids([{"t": i, "x": 2} for i in range(10)]) + constant: list[Record] = [{"t": i, "x": 2} for i in range(10)] + assert "constant-signal" in ids(constant) + + +def test_quality_preserves_original_indices_and_detects_schema_changes() -> None: + records: list[Record] = [ + {"a": 1, "x": 0}, + {"a": 2, "x": None}, + {"b": 3, "x": 2}, + {"b": 4, "x": 2}, + {"b": 5, "x": 2}, + {"b": 6, "x": 2}, + {"b": 7, "x": 2}, + {"b": 8, "x": 3}, + ] + findings = analyze_quality(records) + by_id = {finding.check_id: finding for finding in findings} + assert by_id["frozen-values"].affected == "2-6" + assert "record-shape-change" in by_id + + +def test_zero_mad_still_finds_isolated_anomaly() -> None: + records: list[Record] = [{"x": value} for value in [0, 0, 0, 0, 100, 0]] + findings = analyze_quality(records) + assert {"outliers", "sudden-spikes"} <= {finding.check_id for finding in findings} + + +def test_non_adjacent_duplicate_timestamp_is_reported() -> None: + records: list[Record] = [{"t": 0}, {"t": 1}, {"t": 0}] + findings = analyze_quality(records, "t") + duplicate = next(finding for finding in findings if finding.check_id == "duplicate-timestamps") + assert duplicate.affected == "2" + + +def test_time_field_is_not_misclassified_as_a_noisy_signal() -> None: + records: list[Record] = [{"t": index} for index in range(20)] + findings = analyze_quality(records, "t") + assert not any( + finding.check_id == "high-noise" and finding.field == "t" for finding in findings + ) def test_configurable_monotonicity_and_counter_resets() -> None: - records = [ + records: list[Record] = [ {"distance": 0, "packets": 10}, {"distance": 2, "packets": 11}, {"distance": 1, "packets": 1}, diff --git a/tests/test_recorder.py b/tests/test_recorder.py index 15ac3f9..5ebdd85 100644 --- a/tests/test_recorder.py +++ b/tests/test_recorder.py @@ -1,13 +1,24 @@ import io +import json +import os +import tracemalloc +from collections.abc import Iterator from pathlib import Path +from typing import IO, cast +import pytest + +import datary.recorder from datary.models import RecordOptions from datary.recorder import record_stream from datary.sessions import Session def test_record_session_and_integrity(tmp_path: Path) -> None: - path = record_stream(io.StringIO('{"t":0,"x":1}\nbad\n{"t":1,"x":2}\n'), RecordOptions("demo", tmp_path, "jsonl", "t")) + path = record_stream( + io.StringIO('{"t":0,"x":1}\nbad\n{"t":1,"x":2}\n'), + RecordOptions("demo", tmp_path, "jsonl", "t"), + ) session = Session.open(path) assert session.manifest["valid_record_count"] == 2 assert session.manifest["invalid_record_count"] == 1 @@ -21,7 +32,9 @@ def test_duplicate_names_and_overwrite(tmp_path: Path) -> None: second = record_stream(io.StringIO('{"x":2}\n'), RecordOptions("demo", tmp_path, "jsonl")) assert first.name == "demo" assert second.name == "demo-2" - replaced = record_stream(io.StringIO('{"x":3}\n'), RecordOptions("demo", tmp_path, "jsonl", overwrite=True)) + replaced = record_stream( + io.StringIO('{"x":3}\n'), RecordOptions("demo", tmp_path, "jsonl", overwrite=True) + ) assert list(Session.open(replaced).records())[0]["x"] == 3 @@ -30,10 +43,30 @@ def test_empty_explicit_format_and_unicode(tmp_path: Path) -> None: assert Session.open(path).manifest["record_count"] == 0 -def test_large_stream(tmp_path: Path) -> None: - stream = io.StringIO("".join(f'{{"x":{i}}}\n' for i in range(10_000))) +def test_large_stream_has_a_bounded_python_heap(tmp_path: Path) -> None: + class GeneratedStream: + def __init__(self, count: int) -> None: + self.index = 0 + self.count = count + + def readline(self, _size: int = -1) -> str: + if self.index >= self.count: + return "" + line = f'{{"x":{self.index}}}\n' + self.index += 1 + return line + + count = 50_000 + stream = cast(IO[str], GeneratedStream(count)) + tracemalloc.start() + tracemalloc.reset_peak() session = Session.open(record_stream(stream, RecordOptions("large", tmp_path, "jsonl"))) - assert session.manifest["valid_record_count"] == 10_000 + _, peak = tracemalloc.get_traced_memory() + tracemalloc.stop() + assert session.manifest["valid_record_count"] == count + # This catches accidental record/error list accumulation. It is a Python + # heap ceiling, not an operating-system RSS guarantee. + assert peak < 32 * 1024 * 1024 def test_hash_tamper(tmp_path: Path) -> None: @@ -41,3 +74,146 @@ def test_hash_tamper(tmp_path: Path) -> None: (path / "raw.log").write_text("changed", encoding="utf-8") assert Session.open(path).verify() == ["hash mismatch: raw.log"] + +def test_nonfinite_input_is_preserved_as_invalid(tmp_path: Path) -> None: + path = record_stream( + io.StringIO('{"x":NaN}\n{"x":2}\n'), + RecordOptions("nonfinite", tmp_path, "jsonl"), + ) + session = Session.open(path) + assert session.manifest["valid_record_count"] == 1 + assert session.manifest["invalid_record_count"] == 1 + assert "non-finite" in (path / "invalid.jsonl").read_text(encoding="utf-8") + assert "invalid.jsonl" in session.manifest["hashes"] + assert not (path / ".analysis.sqlite3").exists() + + +def test_failed_overwrite_keeps_original_session( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + path = record_stream(io.StringIO('{"x":1}\n'), RecordOptions("demo", tmp_path, "jsonl")) + + def fail_publish(*_args: object) -> None: + raise OSError("simulated publish failure") + + monkeypatch.setattr(datary.recorder, "_publish", fail_publish) + with pytest.raises(OSError, match="simulated"): + record_stream( + io.StringIO('{"x":2}\n'), + RecordOptions("demo", tmp_path, "jsonl", overwrite=True), + ) + assert list(Session.open(path).records()) == [{"x": 1}] + + +def test_publish_restores_backup_when_final_rename_fails( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + final = tmp_path / "final" + staging = tmp_path / "staging" + backup = tmp_path / "backup" + final.mkdir() + staging.mkdir() + (final / "old").write_text("old", encoding="utf-8") + (staging / "new").write_text("new", encoding="utf-8") + real_replace = os.replace + + def fail_new(source: Path, destination: Path) -> None: + if source == staging and destination == final: + raise OSError("simulated final rename failure") + real_replace(source, destination) + + monkeypatch.setattr(os, "replace", fail_new) + with pytest.raises(OSError, match="final rename"): + datary.recorder._publish(staging, final, backup, True) + assert (final / "old").read_text(encoding="utf-8") == "old" + assert not backup.exists() + + +def test_manifest_and_invalid_artifact_integrity(tmp_path: Path) -> None: + path = record_stream(io.StringIO("bad\n"), RecordOptions("demo", tmp_path, "jsonl")) + (path / "invalid.jsonl").write_text("changed\n", encoding="utf-8") + assert "hash mismatch: invalid.jsonl" in Session.open(path).verify() + manifest = json.loads((path / "manifest.json").read_text(encoding="utf-8")) + manifest["record_count"] = 99 + (path / "manifest.json").write_text(json.dumps(manifest), encoding="utf-8") + with pytest.raises(ValueError, match="inconsistent"): + Session.open(path) + + other = record_stream(io.StringIO('{"x":1}\n'), RecordOptions("manifest", tmp_path, "jsonl")) + other_manifest = json.loads((other / "manifest.json").read_text(encoding="utf-8")) + other_manifest["original_command"] = "changed" + (other / "manifest.json").write_text(json.dumps(other_manifest), encoding="utf-8") + assert "hash mismatch: manifest.json" in Session.open(other).verify() + + +def test_interrupted_auto_detection_finishes_a_session(tmp_path: Path) -> None: + class InterruptedStream: + def readline(self, size: int = -1) -> str: + raise KeyboardInterrupt + + def __iter__(self) -> Iterator[str]: + return iter(()) + + stream = cast(IO[str], InterruptedStream()) + path = record_stream(stream, RecordOptions("stopped", tmp_path)) + manifest = Session.open(path).manifest + assert manifest["interrupted"] is True + assert manifest["record_count"] == 0 + + +def test_engineering_field_roles_are_recorded_and_analysed(tmp_path: Path) -> None: + source = "".join( + json.dumps( + { + "t": index / 10, + "target": 1, + "response": 1 - 2.71828 ** (-index / 10), + } + ) + + "\n" + for index in range(31) + ) + path = record_stream( + io.StringIO(source), + RecordOptions( + "control", + tmp_path, + "jsonl", + time_field="t", + target_field="target", + response_field="response", + ), + ) + session = Session.open(path) + assert session.manifest["field_roles"]["target"] == "target" + metrics = json.loads((path / "metrics.json").read_text(encoding="utf-8")) + assert metrics["control"]["mean_absolute_error"] > 0 + + +def test_line_limit_rejects_before_session_publication(tmp_path: Path) -> None: + with pytest.raises(ValueError, match="line exceeds"): + record_stream( + io.StringIO('{"x":"' + ("a" * 100) + '"}\n'), + RecordOptions("limited", tmp_path, "jsonl", max_line_bytes=32), + ) + assert not (tmp_path / "limited").exists() + + +def test_unknown_field_roles_and_units_are_rejected(tmp_path: Path) -> None: + with pytest.raises(ValueError, match="time field"): + record_stream( + io.StringIO('{"x":1}\n'), + RecordOptions("role", tmp_path, "jsonl", time_field="missing"), + ) + with pytest.raises(ValueError, match="unknown fields"): + record_stream( + io.StringIO('{"x":1}\n'), + RecordOptions("unit", tmp_path, "jsonl", units={"missing": "m"}), + ) + + +def test_parser_failure_still_drains_raw_input(tmp_path: Path) -> None: + source = '[{"x":1}, broken]\nthis tail must remain\n' + path = record_stream(io.StringIO(source), RecordOptions("raw-tail", tmp_path, "json")) + assert (path / "raw.log").read_text(encoding="utf-8") == source + assert Session.open(path).manifest["invalid_record_count"] == 1 diff --git a/tests/test_release_scripts.py b/tests/test_release_scripts.py index c263b60..4def928 100644 --- a/tests/test_release_scripts.py +++ b/tests/test_release_scripts.py @@ -6,9 +6,7 @@ from datary import __version__ SCRIPT = Path(__file__).parents[1] / "scripts" / "build_checksums.py" -current_version = runpy.run_path(str(SCRIPT), run_name="datary_checksum_test")[ - "current_version" -] +current_version = runpy.run_path(str(SCRIPT), run_name="datary_checksum_test")["current_version"] def test_current_release_version() -> None: diff --git a/tests/test_replay.py b/tests/test_replay.py index 2c1f843..88c42f8 100644 --- a/tests/test_replay.py +++ b/tests/test_replay.py @@ -1,6 +1,8 @@ import io from pathlib import Path +import pytest + from datary.models import RecordOptions from datary.recorder import record_stream from datary.replay import replay_session @@ -8,7 +10,9 @@ def test_replay_virtual_and_timing(tmp_path: Path) -> None: - session = Session.open(record_stream(io.StringIO('{"t":0}\n{"t":2}\n'), RecordOptions("a", tmp_path, "jsonl", "t"))) + session = Session.open( + record_stream(io.StringIO('{"t":0}\n{"t":2}\n'), RecordOptions("a", tmp_path, "jsonl", "t")) + ) output = io.StringIO() sleeps: list[float] = [] replay_session(session, output, speed=2, sleep=sleeps.append) @@ -17,3 +21,23 @@ def test_replay_virtual_and_timing(tmp_path: Path) -> None: replay_session(session, output, virtual=True) assert len(output.getvalue().splitlines()) == 2 + +def test_csv_replay_quotes_and_neutralizes_formulas(tmp_path: Path) -> None: + session = Session.open( + record_stream( + io.StringIO('{"text":"a,b","formula":"=2+2"}\n'), + RecordOptions("csv", tmp_path, "jsonl"), + ) + ) + output = io.StringIO() + replay_session(session, output, output_format="csv", no_timing=True) + assert '"a,b"' in output.getvalue() + assert "'=2+2" in output.getvalue() + + +def test_replay_rejects_nonfinite_speed(tmp_path: Path) -> None: + session = Session.open( + record_stream(io.StringIO('{"x":1}\n'), RecordOptions("speed", tmp_path, "jsonl")) + ) + with pytest.raises(ValueError, match="finite"): + replay_session(session, io.StringIO(), speed=float("nan")) diff --git a/tests/test_reports.py b/tests/test_reports.py index ad2ed14..e49a2de 100644 --- a/tests/test_reports.py +++ b/tests/test_reports.py @@ -9,9 +9,44 @@ def test_reports(tmp_path: Path) -> None: - session = Session.open(record_stream(io.StringIO('{"x":1}\n'), RecordOptions("demo", tmp_path, "jsonl"))) + session = Session.open( + record_stream(io.StringIO('{"x":1}\n'), RecordOptions("demo", tmp_path, "jsonl")) + ) markdown = write_report(session, tmp_path / "report.md") assert "# Datary report: demo" in markdown.read_text(encoding="utf-8") json_path = write_report(session, tmp_path / "report.json", "json") assert json.loads(json_path.read_text(encoding="utf-8"))["session"]["session_name"] == "demo" + +def test_markdown_escapes_html_and_displays_integrity_failures(tmp_path: Path) -> None: + session = Session.open( + record_stream( + io.StringIO('{"":1}\n'), + RecordOptions("safe-report", tmp_path, "jsonl"), + ) + ) + (session.path / "raw.log").write_text("tampered", encoding="utf-8") + text = write_report(session, tmp_path / "safe.md").read_text(encoding="utf-8") + assert "