From 757a668e0662de288bb423848692dfed6d77dee4 Mon Sep 17 00:00:00 2001 From: triasha72 Date: Fri, 11 Sep 2026 20:33:26 -0400 Subject: [PATCH 1/4] Implement and document MLE portfolio improvements for EdgeGenBench --- README.md | 5 +++ docs/ios-implementation-update.md | 32 ++++++++++++++++++ ios/EdgeGenBenchDemo/BenchmarkEvidence.swift | 8 ++++- ios/EdgeGenBenchDemo/ContentView.swift | 33 +++++++++++-------- ios/EdgeGenBenchDemo/SurrogatePredictor.swift | 30 +++++++++++++++-- .../CoreMLIntegrationTests.swift | 22 +++++++++++++ scripts/validate_ios_evidence.py | 27 ++++++++++++--- tests/test_ios_evidence.py | 25 ++++++++++++++ 8 files changed, 161 insertions(+), 21 deletions(-) create mode 100644 docs/ios-implementation-update.md create mode 100644 ios/EdgeGenBenchDemoTests/CoreMLIntegrationTests.swift mode change 100644 => 100755 scripts/validate_ios_evidence.py diff --git a/README.md b/README.md index 3940ff4..2e11741 100644 --- a/README.md +++ b/README.md @@ -1189,3 +1189,8 @@ git diff --check end-to-end Android application latency or a cross-hardware speed ranking. - Distribution-shift and extrapolation robustness were not measured in the reported experiments. + +## iOS implementation update + +See [iOS implementation and validation boundary](docs/ios-implementation-update.md) +for background execution, raw evidence export, integration tests and device steps. diff --git a/docs/ios-implementation-update.md b/docs/ios-implementation-update.md new file mode 100644 index 0000000..bdbf9a6 --- /dev/null +++ b/docs/ios-implementation-update.md @@ -0,0 +1,32 @@ +# iOS implementation and validation boundary + +The SwiftUI app executes the generated aircraft-design Core ML surrogate. It does +not execute the separate DASHlink real-flight model. Benchmarking now runs off the +UI thread and saves raw warm latencies and inputs alongside model provenance, +thermal state and simulator identity. The contract rejects invalid normalization +and nonfinite inputs. The Python validator rejects nonfinite summaries and checks +raw sample consistency when samples are supplied. + +From the repository root: + +```sh +pip install -e '.[neural,coreml,dev]' +python scripts/prepare_ios_resources.py +cd ios +xcodegen generate +xcodebuild -project EdgeGenBenchDemo.xcodeproj -scheme EdgeGenBenchDemo -destination 'platform=iOS Simulator,name=iPhone 16' test +``` + +Choose an installed simulator from `xcrun simctl list devices available`. The new +CoreMLIntegrationTests loads the bundled model, executes inference and attaches +100-run evidence to the Xcode result. A simulator test checks integration only. +For hardware evidence, choose a physical iPhone and signing team in Xcode, run the +app and export its JSON through Share. Use `scripts/validate_ios_evidence.py --help` +to validate against the source artifact hashes. No physical iPhone timings have +been collected in this update. The local Xcode build was blocked by the execution +sandbox; the integration test has not yet been confirmed passing. + +Core ML `.all` requests available compute units; it does not prove ANE placement. +Use retained Instruments measurements for placement or power claims. Cold latency +includes model loading plus first inference. This is a model benchmark, not a +validated aircraft-design recommendation system. diff --git a/ios/EdgeGenBenchDemo/BenchmarkEvidence.swift b/ios/EdgeGenBenchDemo/BenchmarkEvidence.swift index 54d0120..a6de9ed 100644 --- a/ios/EdgeGenBenchDemo/BenchmarkEvidence.swift +++ b/ios/EdgeGenBenchDemo/BenchmarkEvidence.swift @@ -33,6 +33,9 @@ struct IOSBenchmarkEvidence: Codable { let latency: LatencySummary let outputMaxAbsDrift: Double let outputs: [PredictionValue] + var warmLatencySamplesMs: [Double]? = nil + var inputValues: [Double]? = nil + var inputCategory: String? = nil } struct PredictionValue: Codable { @@ -100,7 +103,10 @@ enum IOSBenchmarkRunner { warmRuns: warmRuns ), outputMaxAbsDrift: maxDrift, - outputs: coldOutput.map { PredictionValue(name: $0.name, value: $0.value) } + outputs: coldOutput.map { PredictionValue(name: $0.name, value: $0.value) }, + warmLatencySamplesMs: latencies, + inputValues: numericValues, + inputCategory: category ) } diff --git a/ios/EdgeGenBenchDemo/ContentView.swift b/ios/EdgeGenBenchDemo/ContentView.swift index 321e12a..8b7e001 100644 --- a/ios/EdgeGenBenchDemo/ContentView.swift +++ b/ios/EdgeGenBenchDemo/ContentView.swift @@ -17,7 +17,7 @@ struct ContentView: View { var body: some View { NavigationStack { Form { - Section("Aircraft design") { + Section("Aircraft design — generated-data model") { ForEach(featureNames.indices, id: \.self) { index in TextField(featureNames[index], value: $values[index], format: .number) .keyboardType(.decimalPad) @@ -50,23 +50,30 @@ struct ContentView: View { } } .navigationTitle("EdgeGenBench") + .disabled(isRunning) } } private func runBenchmark() { isRunning = true - do { - let result = try IOSBenchmarkRunner.run(numericValues: values, category: category) - evidence = result - predictions = result.outputs.map { Prediction(name: $0.name, value: $0.value) } - evidenceURL = try result.writeTemporaryJSON() - message = "Core ML benchmark completed (1 cold + \(result.latency.warmRuns) warm runs)." - } catch { - predictions = [] - evidence = nil - evidenceURL = nil - message = error.localizedDescription + evidence = nil + evidenceURL = nil + let inputValues = values + let inputCategory = category + Task { + do { + let result = try await Task.detached(priority: .userInitiated) { + try IOSBenchmarkRunner.run(numericValues: inputValues, category: inputCategory) + }.value + evidence = result + predictions = result.outputs.map { Prediction(name: $0.name, value: $0.value) } + evidenceURL = try result.writeTemporaryJSON() + message = "Completed 1 model-load + first-prediction and \(result.latency.warmRuns) warm runs." + } catch { + predictions = [] + message = error.localizedDescription + } + isRunning = false } - isRunning = false } } diff --git a/ios/EdgeGenBenchDemo/SurrogatePredictor.swift b/ios/EdgeGenBenchDemo/SurrogatePredictor.swift index 5aee0ec..19199eb 100644 --- a/ios/EdgeGenBenchDemo/SurrogatePredictor.swift +++ b/ios/EdgeGenBenchDemo/SurrogatePredictor.swift @@ -52,6 +52,7 @@ final class SurrogatePredictor { let contractData = try Data(contentsOf: contractURL) contract = try JSONDecoder().decode(ModelContract.self, from: contractData) contractSHA256 = SHA256.hash(data: contractData).map { String(format: "%02x", $0) }.joined() + try contract.validate() guard contract.featureMean.count + contract.categories.count == contract.inputDimension, contract.featureScale.count == contract.featureMean.count, contract.targets.count == contract.outputDimension, @@ -72,7 +73,8 @@ final class SurrogatePredictor { } func predict(numericValues: [Double], category: String) throws -> [Prediction] { - guard numericValues.count == contract.featureMean.count, + guard numericValues.allSatisfy({ $0.isFinite }), + numericValues.count == contract.featureMean.count, let categoryIndex = contract.categories.firstIndex(of: category) else { throw SurrogateError.invalidContract("input values do not agree") } @@ -89,8 +91,32 @@ final class SurrogatePredictor { normalized.count == contract.outputDimension else { throw SurrogateError.invalidOutput } - return contract.targets.indices.map { index in + guard (0.. 0 }), + targetMean.allSatisfy({ $0.isFinite }), + targetScale.allSatisfy({ $0.isFinite && $0 > 0 }) else { + throw SurrogateError.invalidContract("nonfinite, duplicate, or invalid preprocessing values") + } } } diff --git a/ios/EdgeGenBenchDemoTests/CoreMLIntegrationTests.swift b/ios/EdgeGenBenchDemoTests/CoreMLIntegrationTests.swift new file mode 100644 index 0000000..7f32996 --- /dev/null +++ b/ios/EdgeGenBenchDemoTests/CoreMLIntegrationTests.swift @@ -0,0 +1,22 @@ +import XCTest +@testable import EdgeGenBenchDemo + +final class CoreMLIntegrationTests: XCTestCase { + func testBundledModelRunsAndRejectsInvalidInputs() throws { + let predictor = try SurrogatePredictor() + let values = predictor.contract.featureMean + let category = try XCTUnwrap(predictor.contract.categories.first) + let result = try predictor.predict(numericValues: values, category: category) + XCTAssertEqual(result.count, predictor.contract.outputDimension) + XCTAssertTrue(result.allSatisfy { $0.value.isFinite }) + XCTAssertThrowsError(try predictor.predict(numericValues: [.nan], category: category)) + XCTAssertThrowsError(try predictor.predict(numericValues: values, category: "unknown")) + let evidence = try IOSBenchmarkRunner.run(numericValues: values, category: category) + XCTAssertEqual(evidence.warmLatencySamplesMs?.count, 100) + XCTAssertLessThanOrEqual(evidence.outputMaxAbsDrift, 1e-6) + let attachment = XCTAttachment(data: try JSONEncoder().encode(evidence), uniformTypeIdentifier: "public.json") + attachment.name = "CoreML execution evidence" + attachment.lifetime = .keepAlways + add(attachment) + } +} diff --git a/scripts/validate_ios_evidence.py b/scripts/validate_ios_evidence.py old mode 100644 new mode 100755 index 8674fff..872e9ee --- a/scripts/validate_ios_evidence.py +++ b/scripts/validate_ios_evidence.py @@ -6,6 +6,7 @@ import argparse import hashlib import json +import math from pathlib import Path from typing import Any @@ -39,22 +40,37 @@ def validate_ios_evidence( latency = evidence.get("latency") if not isinstance(device, dict) or not isinstance(latency, dict): raise ValueError("device identity and latency summary are required") - if bool(device.get("simulator")) and not allow_simulator: + if not isinstance(device.get("simulator"), bool) or not device.get("model"): + raise ValueError("explicit simulator boolean and device model required") + if device["simulator"] and not allow_simulator: raise ValueError("physical-iPhone evidence cannot come from a simulator") if device.get("systemName") != "iOS" and not allow_simulator: raise ValueError("physical evidence must identify iOS") if int(latency.get("warmRuns", 0)) < 100: raise ValueError("at least 100 warm inference runs are required") for name in ("coldMs", "warmMeanMs", "warmP95Ms"): - if float(latency.get(name, 0)) <= 0: + if not math.isfinite(float(latency.get(name, 0))) or float(latency.get(name, 0)) <= 0: raise ValueError(f"{name} must be positive") - if float(evidence.get("outputMaxAbsDrift", 1.0)) > 1e-6: + drift = float(evidence.get("outputMaxAbsDrift", 1.0)) + if not math.isfinite(drift) or not 0 <= drift <= 1e-6: raise ValueError("iOS repeated-output drift exceeds tolerance") if evidence.get("sourceModelSha256") != _sha256(model_path): raise ValueError("iOS source-model provenance does not match the repository") if evidence.get("preprocessingSha256") != _sha256(preprocessing_path): raise ValueError("iOS preprocessing provenance does not match the repository") + samples = evidence.get("warmLatencySamplesMs") + if samples is not None: + if len(samples) != latency["warmRuns"] or not all( + math.isfinite(x) and x > 0 for x in samples + ): + raise ValueError("invalid warm latency samples") + mean = sum(samples) / len(samples) + p95 = sorted(samples)[math.ceil(0.95 * len(samples)) - 1] + if not math.isclose(mean, latency["warmMeanMs"], rel_tol=1e-8) or not math.isclose( + p95, latency["warmP95Ms"], rel_tol=1e-8 + ): + raise ValueError("summary does not match latency samples") return { "status": "validated_physical_iphone_coreml" if not device["simulator"] @@ -71,7 +87,8 @@ def validate_ios_evidence( "power_measurement": "not_measured", "neural_engine_placement": "not_measured", "claim_boundary": ( - "Physical iPhone Core ML application latency; not proof of Apple Neural Engine " + ("Simulator integration only; " if device["simulator"] else "Physical iPhone latency; ") + + "not proof of Apple Neural Engine " "placement and not a power measurement." ), } @@ -83,7 +100,7 @@ def write_report(summary: dict[str, Any], output_json: Path, output_markdown: Pa latency = summary["latency"] device = summary["device"] lines = [ - "# EdgeGenBench physical iPhone Core ML report", + "# EdgeGenBench Core ML execution report", "", f"- Status: `{summary['status']}`", f"- Device: `{device['model']}`", diff --git a/tests/test_ios_evidence.py b/tests/test_ios_evidence.py index d033a13..062d304 100644 --- a/tests/test_ios_evidence.py +++ b/tests/test_ios_evidence.py @@ -81,3 +81,28 @@ def test_rejects_unproven_ane_claim(tmp_path: Path) -> None: evidence.write_text(json.dumps(payload)) with pytest.raises(ValueError, match="ANE placement"): validate_ios_evidence(evidence, model_path=model, preprocessing_path=preprocessing) + + +@pytest.mark.parametrize("field", ["coldMs", "warmMeanMs", "warmP95Ms"]) +def test_rejects_nonfinite_latency(tmp_path, field): + model, prep = tmp_path / "model", tmp_path / "prep" + model.write_bytes(b"model") + prep.write_bytes(b"prep") + payload = _evidence(model, prep) + payload["latency"][field] = float("nan") + evidence = tmp_path / "evidence.json" + evidence.write_text(json.dumps(payload)) + with pytest.raises(ValueError, match="positive"): + validate_ios_evidence(evidence, model_path=model, preprocessing_path=prep) + + +def test_rejects_inconsistent_raw_samples(tmp_path): + model, prep = tmp_path / "model", tmp_path / "prep" + model.write_bytes(b"model") + prep.write_bytes(b"prep") + payload = _evidence(model, prep) + payload["warmLatencySamplesMs"] = [2.0] * 100 + evidence = tmp_path / "evidence.json" + evidence.write_text(json.dumps(payload)) + with pytest.raises(ValueError, match="summary"): + validate_ios_evidence(evidence, model_path=model, preprocessing_path=prep) From d900d086f11a25a057a8bbb79e62bc21ce893a7f Mon Sep 17 00:00:00 2001 From: triasha72 Date: Sat, 12 Sep 2026 20:36:56 -0400 Subject: [PATCH 2/4] Fix iOS model contract compatibility and default category --- ios/EdgeGenBenchDemo/ContentView.swift | 2 +- ios/EdgeGenBenchDemo/SurrogatePredictor.swift | 2 +- tests/test_ios_evidence.py | 8 ++++++++ 3 files changed, 10 insertions(+), 2 deletions(-) diff --git a/ios/EdgeGenBenchDemo/ContentView.swift b/ios/EdgeGenBenchDemo/ContentView.swift index 8b7e001..10c54c4 100644 --- a/ios/EdgeGenBenchDemo/ContentView.swift +++ b/ios/EdgeGenBenchDemo/ContentView.swift @@ -7,7 +7,7 @@ struct ContentView: View { "hybridization_ratio" ] @State private var values = [4.0, 250.0, 180.0, 300.0, 0.65, 0.5] - @State private var category = "battery_electric" + @State private var category = "conventional_turboprop" @State private var predictions: [Prediction] = [] @State private var message = "Run the bundled Core ML model and capture cold + warm evidence." @State private var evidence: IOSBenchmarkEvidence? diff --git a/ios/EdgeGenBenchDemo/SurrogatePredictor.swift b/ios/EdgeGenBenchDemo/SurrogatePredictor.swift index 19199eb..f93c839 100644 --- a/ios/EdgeGenBenchDemo/SurrogatePredictor.swift +++ b/ios/EdgeGenBenchDemo/SurrogatePredictor.swift @@ -107,7 +107,7 @@ final class SurrogatePredictor { extension ModelContract { func validate() throws { - guard schemaVersion == "1.0", !numericFeatures.isEmpty, + guard schemaVersion == "1.0" || schemaVersion == "1.1", !numericFeatures.isEmpty, numericFeatures.count == featureMean.count, featureMean.count == featureScale.count, Set(numericFeatures).count == numericFeatures.count, diff --git a/tests/test_ios_evidence.py b/tests/test_ios_evidence.py index 062d304..edbba28 100644 --- a/tests/test_ios_evidence.py +++ b/tests/test_ios_evidence.py @@ -83,6 +83,14 @@ def test_rejects_unproven_ane_claim(tmp_path: Path) -> None: validate_ios_evidence(evidence, model_path=model, preprocessing_path=preprocessing) +def test_bundled_contract_schema_and_default_category_are_current() -> None: + contract = json.loads( + (Path(__file__).parents[1] / "ios/EdgeGenBenchDemo/Resources/ModelContract.json").read_text() + ) + assert contract["schemaVersion"] in {"1.0", "1.1"} + assert "conventional_turboprop" in contract["categories"] + + @pytest.mark.parametrize("field", ["coldMs", "warmMeanMs", "warmP95Ms"]) def test_rejects_nonfinite_latency(tmp_path, field): model, prep = tmp_path / "model", tmp_path / "prep" From cb009f4d9aafa26d7080f3fe9a60b85b8bef005c Mon Sep 17 00:00:00 2001 From: triasha72 Date: Sat, 12 Sep 2026 20:48:17 -0400 Subject: [PATCH 3/4] Add validated physical iPhone benchmark evidence --- reports/iphone/run-01-report.md | 16 +++ reports/iphone/run-01-summary.json | 25 +++++ reports/iphone/run-01.json | 165 +++++++++++++++++++++++++++++ reports/iphone/run-02-report.md | 16 +++ reports/iphone/run-02-summary.json | 25 +++++ reports/iphone/run-02.json | 165 +++++++++++++++++++++++++++++ 6 files changed, 412 insertions(+) create mode 100644 reports/iphone/run-01-report.md create mode 100644 reports/iphone/run-01-summary.json create mode 100644 reports/iphone/run-01.json create mode 100644 reports/iphone/run-02-report.md create mode 100644 reports/iphone/run-02-summary.json create mode 100644 reports/iphone/run-02.json diff --git a/reports/iphone/run-01-report.md b/reports/iphone/run-01-report.md new file mode 100644 index 0000000..741c2f8 --- /dev/null +++ b/reports/iphone/run-01-report.md @@ -0,0 +1,16 @@ +# EdgeGenBench Core ML execution report + +- Status: `validated_physical_iphone_coreml` +- Device: `iPhone17,1` +- OS: `iOS 26.6.2` +- Backend: `CoreML` (requested compute units: `all`) +- Cold latency: `154.524125 ms` +- Warm mean latency: `0.036253 ms` +- Warm p95 latency: `0.042750 ms` +- Warm runs: `100` +- Output max absolute drift: `0` +- Thermal state: `nominal` → `nominal` +- Power: `not measured` +- Apple Neural Engine placement: `not measured` + +> Physical iPhone latency; not proof of Apple Neural Engine placement and not a power measurement. diff --git a/reports/iphone/run-01-summary.json b/reports/iphone/run-01-summary.json new file mode 100644 index 0000000..a82aa70 --- /dev/null +++ b/reports/iphone/run-01-summary.json @@ -0,0 +1,25 @@ +{ + "status": "validated_physical_iphone_coreml", + "captured_at_utc": "2026-09-13T00:42:55Z", + "app_version": "0.1.0", + "device": { + "model": "iPhone17,1", + "simulator": false, + "systemName": "iOS", + "systemVersion": "26.6.2" + }, + "backend": "CoreML", + "requested_compute_units": "all", + "latency": { + "coldMs": 154.524125, + "warmMeanMs": 0.03625252, + "warmP95Ms": 0.04275, + "warmRuns": 100 + }, + "output_max_abs_drift": 0, + "thermal_state_before": "nominal", + "thermal_state_after": "nominal", + "power_measurement": "not_measured", + "neural_engine_placement": "not_measured", + "claim_boundary": "Physical iPhone latency; not proof of Apple Neural Engine placement and not a power measurement." +} diff --git a/reports/iphone/run-01.json b/reports/iphone/run-01.json new file mode 100644 index 0000000..5f8cf3b --- /dev/null +++ b/reports/iphone/run-01.json @@ -0,0 +1,165 @@ +{ + "appVersion" : "0.1.0", + "backend" : "CoreML", + "capturedAtUTC" : "2026-09-13T00:42:55Z", + "contractSha256" : "b876d26ebe67b5949e6f00c77723e64929e0ffcea90c41bff5227414bd087481", + "device" : { + "model" : "iPhone17,1", + "simulator" : false, + "systemName" : "iOS", + "systemVersion" : "26.6.2" + }, + "inputCategory" : "conventional_turboprop", + "inputValues" : [ + 4, + 250, + 180, + 300, + 0.65, + 0.5 + ], + "latency" : { + "coldMs" : 154.524125, + "warmMeanMs" : 0.03625252, + "warmP95Ms" : 0.04275, + "warmRuns" : 100 + }, + "lowPowerMode" : false, + "neuralEnginePlacement" : "not_measured", + "outputMaxAbsDrift" : 0, + "outputs" : [ + { + "name" : "estimated_takeoff_mass_kg", + "value" : 15796.306063175201 + }, + { + "name" : "mission_energy_kwh", + "value" : 1287.0568504333496 + }, + { + "name" : "energy_per_passenger_km_kwh", + "value" : 0.37213412301207427 + }, + { + "name" : "lifecycle_emissions_proxy_kgco2e", + "value" : 530.1332678794861 + }, + { + "name" : "operating_cost_proxy_usd", + "value" : 324.0722385644913 + }, + { + "name" : "noise_proxy_db", + "value" : 80.77662195730954 + } + ], + "powerMeasurement" : "not_measured", + "preprocessingSha256" : "c53831dd106a26b668d586b4eb83f73a1e43483c52ec8ee36c7ba35c95cdb08e", + "requestedComputeUnits" : "all", + "schemaVersion" : "1.0", + "sourceModelSha256" : "55d6db9f19f2e361c6066b639920cfd1aad54ea5544b463b205857ebb7ceb657", + "thermalStateAfter" : "nominal", + "thermalStateBefore" : "nominal", + "warmLatencySamplesMs" : [ + 0.082625, + 0.05075, + 0.04275, + 0.040209, + 0.03875, + 0.038167, + 0.036792, + 0.036375, + 0.036334, + 0.036875, + 0.036916, + 0.0355, + 0.035458, + 0.035292, + 0.036792, + 0.036, + 0.036, + 0.035792, + 0.035042, + 0.036042, + 0.035375, + 0.035083, + 0.0345, + 0.034792, + 0.036042, + 0.035708, + 0.035292, + 0.034959, + 0.034333, + 0.035709, + 0.0345, + 0.063833, + 0.034958, + 0.034875, + 0.035583, + 0.035041, + 0.035042, + 0.034667, + 0.034625, + 0.035542, + 0.034709, + 0.034333, + 0.034375, + 0.03425, + 0.035416, + 0.037917, + 0.034667, + 0.034667, + 0.034791, + 0.03475, + 0.034, + 0.043792, + 0.03475, + 0.034375, + 0.034375, + 0.034833, + 0.034667, + 0.034458, + 0.034125, + 0.034834, + 0.035, + 0.034667, + 0.034583, + 0.034375, + 0.03425, + 0.034459, + 0.034292, + 0.034583, + 0.034083, + 0.034958, + 0.034708, + 0.034125, + 0.034292, + 0.034416, + 0.034292, + 0.034584, + 0.034583, + 0.034375, + 0.034458, + 0.034667, + 0.034583, + 0.031458, + 0.035417, + 0.034375, + 0.036167, + 0.036167, + 0.034416, + 0.034583, + 0.034416, + 0.03475, + 0.034583, + 0.034333, + 0.034291, + 0.034375, + 0.034625, + 0.034542, + 0.034375, + 0.046583, + 0.033584, + 0.034875 + ] +} \ No newline at end of file diff --git a/reports/iphone/run-02-report.md b/reports/iphone/run-02-report.md new file mode 100644 index 0000000..3920ff6 --- /dev/null +++ b/reports/iphone/run-02-report.md @@ -0,0 +1,16 @@ +# EdgeGenBench Core ML execution report + +- Status: `validated_physical_iphone_coreml` +- Device: `iPhone17,1` +- OS: `iOS 26.6.2` +- Backend: `CoreML` (requested compute units: `all`) +- Cold latency: `14.942959 ms` +- Warm mean latency: `0.044276 ms` +- Warm p95 latency: `0.047708 ms` +- Warm runs: `100` +- Output max absolute drift: `0` +- Thermal state: `nominal` → `nominal` +- Power: `not measured` +- Apple Neural Engine placement: `not measured` + +> Physical iPhone latency; not proof of Apple Neural Engine placement and not a power measurement. diff --git a/reports/iphone/run-02-summary.json b/reports/iphone/run-02-summary.json new file mode 100644 index 0000000..7b2661c --- /dev/null +++ b/reports/iphone/run-02-summary.json @@ -0,0 +1,25 @@ +{ + "status": "validated_physical_iphone_coreml", + "captured_at_utc": "2026-09-13T00:45:47Z", + "app_version": "0.1.0", + "device": { + "model": "iPhone17,1", + "simulator": false, + "systemName": "iOS", + "systemVersion": "26.6.2" + }, + "backend": "CoreML", + "requested_compute_units": "all", + "latency": { + "coldMs": 14.942959, + "warmMeanMs": 0.044276289999999996, + "warmP95Ms": 0.047708, + "warmRuns": 100 + }, + "output_max_abs_drift": 0, + "thermal_state_before": "nominal", + "thermal_state_after": "nominal", + "power_measurement": "not_measured", + "neural_engine_placement": "not_measured", + "claim_boundary": "Physical iPhone latency; not proof of Apple Neural Engine placement and not a power measurement." +} diff --git a/reports/iphone/run-02.json b/reports/iphone/run-02.json new file mode 100644 index 0000000..79104d0 --- /dev/null +++ b/reports/iphone/run-02.json @@ -0,0 +1,165 @@ +{ + "appVersion" : "0.1.0", + "backend" : "CoreML", + "capturedAtUTC" : "2026-09-13T00:45:47Z", + "contractSha256" : "b876d26ebe67b5949e6f00c77723e64929e0ffcea90c41bff5227414bd087481", + "device" : { + "model" : "iPhone17,1", + "simulator" : false, + "systemName" : "iOS", + "systemVersion" : "26.6.2" + }, + "inputCategory" : "conventional_turboprop", + "inputValues" : [ + 65, + 950, + 535, + 527, + 0.57, + 0.24 + ], + "latency" : { + "coldMs" : 14.942959, + "warmMeanMs" : 0.044276289999999996, + "warmP95Ms" : 0.047708, + "warmRuns" : 100 + }, + "lowPowerMode" : false, + "neuralEnginePlacement" : "not_measured", + "outputMaxAbsDrift" : 0, + "outputs" : [ + { + "name" : "estimated_takeoff_mass_kg", + "value" : 27896.445220708847 + }, + { + "name" : "mission_energy_kwh", + "value" : 14179.826229214668 + }, + { + "name" : "energy_per_passenger_km_kwh", + "value" : 0.22973118861409603 + }, + { + "name" : "lifecycle_emissions_proxy_kgco2e", + "value" : 3414.5965099334717 + }, + { + "name" : "operating_cost_proxy_usd", + "value" : 1866.5452010631561 + }, + { + "name" : "noise_proxy_db", + "value" : 89.17093080934137 + } + ], + "powerMeasurement" : "not_measured", + "preprocessingSha256" : "c53831dd106a26b668d586b4eb83f73a1e43483c52ec8ee36c7ba35c95cdb08e", + "requestedComputeUnits" : "all", + "schemaVersion" : "1.0", + "sourceModelSha256" : "55d6db9f19f2e361c6066b639920cfd1aad54ea5544b463b205857ebb7ceb657", + "thermalStateAfter" : "nominal", + "thermalStateBefore" : "nominal", + "warmLatencySamplesMs" : [ + 0.094917, + 0.059042, + 0.042083, + 0.039542, + 0.061666, + 0.036083, + 0.040458, + 0.038875, + 0.038917, + 0.036542, + 0.036666, + 0.037708, + 0.036708, + 0.034625, + 0.038417, + 0.035584, + 0.037958, + 0.034667, + 0.036542, + 0.033375, + 0.037833, + 0.037708, + 0.037, + 0.036709, + 0.03825, + 0.037416, + 0.041417, + 0.045375, + 0.041416, + 0.046292, + 0.045875, + 0.045208, + 0.041, + 0.046375, + 0.043667, + 0.046041, + 0.045875, + 0.045792, + 0.044792, + 0.0475, + 0.045834, + 0.044875, + 0.045875, + 0.044917, + 0.04775, + 0.044917, + 0.045542, + 0.044792, + 0.044833, + 0.047708, + 0.0455, + 0.046, + 0.045541, + 0.044666, + 0.045083, + 0.0455, + 0.045792, + 0.044292, + 0.044625, + 0.047167, + 0.045542, + 0.0455, + 0.045042, + 0.044417, + 0.048708, + 0.045917, + 0.044584, + 0.044959, + 0.045167, + 0.045833, + 0.046209, + 0.045542, + 0.045125, + 0.045375, + 0.045791, + 0.045209, + 0.045041, + 0.045083, + 0.044917, + 0.047584, + 0.045834, + 0.045375, + 0.045625, + 0.045792, + 0.045458, + 0.044208, + 0.044625, + 0.045125, + 0.045334, + 0.046708, + 0.0455, + 0.045417, + 0.044417, + 0.045291, + 0.046583, + 0.044667, + 0.045042, + 0.045291, + 0.043083, + 0.043584 + ] +} \ No newline at end of file From 9565d1ad5d25bcef2159fd61063c37e025e4744c Mon Sep 17 00:00:00 2001 From: triasha72 Date: Sat, 12 Sep 2026 21:33:53 -0400 Subject: [PATCH 4/4] Improve iOS inputs and out-of-distribution guidance --- ios/EdgeGenBenchDemo/ContentView.swift | 42 ++++++++++++++++++++++++-- 1 file changed, 40 insertions(+), 2 deletions(-) diff --git a/ios/EdgeGenBenchDemo/ContentView.swift b/ios/EdgeGenBenchDemo/ContentView.swift index 10c54c4..7845f88 100644 --- a/ios/EdgeGenBenchDemo/ContentView.swift +++ b/ios/EdgeGenBenchDemo/ContentView.swift @@ -10,6 +10,7 @@ struct ContentView: View { @State private var category = "conventional_turboprop" @State private var predictions: [Prediction] = [] @State private var message = "Run the bundled Core ML model and capture cold + warm evidence." + @State private var inputWarning: String? @State private var evidence: IOSBenchmarkEvidence? @State private var evidenceURL: URL? @State private var isRunning = false @@ -19,10 +20,20 @@ struct ContentView: View { Form { Section("Aircraft design — generated-data model") { ForEach(featureNames.indices, id: \.self) { index in - TextField(featureNames[index], value: $values[index], format: .number) + TextField(featureNames[index].replacingOccurrences(of: "_", with: " "), value: $values[index], format: .number) .keyboardType(.decimalPad) } - TextField("propulsion_architecture", text: $category) + Picker("Propulsion architecture", selection: $category) { + ForEach(["conventional_turboprop", "fuel_cell_electric", "parallel_hybrid", "series_hybrid"], id: \.self) { value in + Text(value.replacingOccurrences(of: "_", with: " ")).tag(value) + } + } + Button("Reset to reference inputs", action: resetInputs) + if let inputWarning { + Label(inputWarning, systemImage: "exclamationmark.triangle") + .font(.footnote) + .foregroundStyle(.orange) + } } Section { Button(isRunning ? "Benchmarking…" : "Run cold + warm benchmark", action: runBenchmark) @@ -51,10 +62,17 @@ struct ContentView: View { } .navigationTitle("EdgeGenBench") .disabled(isRunning) + .onChange(of: values) { _, _ in updateInputWarning() } } + .task { updateInputWarning() } } private func runBenchmark() { + guard values.count == featureNames.count, values.allSatisfy(\.isFinite) else { + message = "Enter finite numeric values for every feature." + return + } + updateInputWarning() isRunning = true evidence = nil evidenceURL = nil @@ -76,4 +94,24 @@ struct ContentView: View { isRunning = false } } + + private func resetInputs() { + values = [65.0, 950.0, 535.0, 527.0, 0.57, 0.24] + category = "conventional_turboprop" + message = "Reference inputs restored." + updateInputWarning() + } + + private func updateInputWarning() { + let means = [64.93524, 950.4023, 534.7646, 526.7014, 0.57418, 0.24301] + let scales = [14.40655, 318.0289, 66.45373, 130.1412, 0.072085, 0.215283] + guard values.count == means.count, values.allSatisfy(\.isFinite) else { + inputWarning = "Some inputs are not finite." + return + } + let maximumZ = zip(values, zip(means, scales)).map { pair in + abs((pair.0 - pair.1.0) / pair.1.1) + }.max() ?? 0 + inputWarning = maximumZ > 3 ? "Inputs are outside the training distribution (max |z| = \(maximumZ.formatted(.number.precision(.fractionLength(1))))." : nil + } }