diff --git a/.github/workflows/validate.yml b/.github/workflows/validate.yml new file mode 100644 index 0000000..653d5c8 --- /dev/null +++ b/.github/workflows/validate.yml @@ -0,0 +1,26 @@ +name: Validate analysis release + +on: + push: + pull_request: + +permissions: + contents: read + +jobs: + validate: + runs-on: ubuntu-latest + steps: + - name: Check out repository + uses: actions/checkout@v4 + + - name: Set up Python + uses: actions/setup-python@v5 + with: + python-version: "3.12" + + - name: Check Python syntax + run: python -m py_compile scripts/build_scirep_analysis.py scripts/validate_release.py + + - name: Validate aggregate results + run: python scripts/validate_release.py diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..334ff13 --- /dev/null +++ b/.gitignore @@ -0,0 +1,40 @@ +.DS_Store +.venv/ +__pycache__/ +*.py[cod] +paper/build/ +paper/tables/ + +# The full analysis creates participant-level intermediate files. Keep them +# local until the complete governed dataset release is public. +data/derived/* +!data/derived/analysis_manifest.json +!data/derived/analysis_summary.json +!data/derived/all_trial_feature_statistics.csv +!data/derived/classification_summary.json +!data/derived/cohort_adjusted_models.csv +!data/derived/cohort_specific_effects.csv +!data/derived/continuous_performance_validation.csv +!data/derived/coordination_control_time_correlation.csv +!data/derived/cycle_duration_ci.csv +!data/derived/feature_dictionary.csv +!data/derived/kmeans_cluster_summary.csv +!data/derived/kmeans_thresholds.json +!data/derived/kmeans_validation.csv +!data/derived/metric_band_classification_comparison.csv +!data/derived/metric_band_classification_summary.json +!data/derived/phase_fraction_ci.csv +!data/derived/phase_kinematics_summary.csv +!data/derived/phase_segment_summary.csv +!data/derived/position_smoothing_sensitivity.csv +!data/derived/processing_sensitivity.csv +!data/derived/trimming_rule_validation.csv + +# Keep a compact figure gallery in Git. The full figure set can be regenerated. +paper/figures/* +!paper/figures/fig1_cohort_structure.png +!paper/figures/fig2_feature_effects.png +!paper/figures/fig3_phase_timing.png +!paper/figures/fig4_experience_links.png +!paper/figures/fig21_cohort_effect_forest.png +!paper/figures/fig23_continuous_performance_validation.png diff --git a/CITATION.cff b/CITATION.cff index a45cd0e..3449205 100644 --- a/CITATION.cff +++ b/CITATION.cff @@ -14,9 +14,12 @@ authors: affiliation: University of Leeds identifiers: - type: doi - value: 10.5281/zenodo.20752650 + value: 10.5281/zenodo.20752651 description: Zenodo record for version 1.0 -repository: 'https://doi.org/10.5281/zenodo.20752650' + - type: doi + value: 10.5281/zenodo.20752650 + description: Zenodo concept DOI for all versions +repository: 'https://github.com/omariosc/LASK' repository-code: 'https://github.com/omariosc/LASK' url: 'https://doi.org/10.5281/zenodo.20752650' abstract: >- @@ -28,8 +31,9 @@ abstract: >- validation set (10 trials) and a 6-DoF out-of-distribution test set (8 trials), spanning surgeons from novice to expert. Each trial provides per-frame electromagnetic tool poses across roughly 91,000 frames, with - positions in millimetres, orientations as unit quaternions and calibrated jaw - angles for the 7-DoF cohorts, time-aligned to the video. Labelled keyframes + positions in millimetres, orientations as unit quaternions and relative + voltage-derived jaw-opening signals for the 7-DoF cohorts, time-aligned to + the video. Labelled keyframes supply instrument segmentation masks, tooltip and jaw keypoints, shaft-joint keypoints and per-component visibility flags for both instruments, alongside surgeon handedness and procedure experience. @@ -68,3 +72,18 @@ references: name: Frontiers Media SA year: 2025 doi: 10.3389/978-2-8325-5137-0 + - type: conference-paper + title: Bayesian Temporal Pose Networks for Uncertainty-Calibrated Laparoscopic Tool Pose Tracking + authors: + - given-names: Omar + family-names: Choudhry + - given-names: Sharib + family-names: Ali + - given-names: Chandra Shekhar + family-names: Biyani + - given-names: Dominic + family-names: Jones + conference: + name: Medical Image Computing and Computer Assisted Intervention (MICCAI) + year: 2026 + repository-code: https://github.com/omariosc/BTPN diff --git a/README.md b/README.md index 96ccd3b..c42bac9 100644 --- a/README.md +++ b/README.md @@ -1,89 +1,186 @@ # LASK -**LA**paroscopic **S**kill & **K**inematics: a peg-transfer box-trainer dataset -pairing endoscopic video with dense 7-DoF instrument kinematics and manual -instrument annotations. +**LA**paroscopic **S**kill and **K**inematics is a laparoscopic peg-transfer +dataset that pairs endoscopic video with electromagnetic measurements from two +instruments, manual instrument annotations, and participant experience +metadata. [![DOI](https://img.shields.io/badge/DOI-10.5281%2Fzenodo.20752650-1682D4?style=flat-square)](https://doi.org/10.5281/zenodo.20752650) -[![Download](https://img.shields.io/badge/Download-Zenodo-0F62FE?style=flat-square)](https://doi.org/10.5281/zenodo.20752650) -[![Licence](https://img.shields.io/badge/Licence-CC%20BY%204.0-orange?style=flat-square)](https://creativecommons.org/licenses/by/4.0/) -[![Version](https://img.shields.io/badge/Version-1.0-informational?style=flat-square)](https://doi.org/10.5281/zenodo.20752650) -[![Trials](https://img.shields.io/badge/Trials-37%20annotated-1f4e79?style=flat-square)](#what-is-in-the-dataset) -[![Frames](https://img.shields.io/badge/Frames-~91,000-1f4e79?style=flat-square)](#what-is-in-the-dataset) -[![Paper](https://img.shields.io/badge/Paper-MIUA%202025-548235?style=flat-square)](https://doi.org/10.3389/978-2-8325-5137-0) - -> **The dataset is hosted on Zenodo:** -> **[https://doi.org/10.5281/zenodo.20752650](https://doi.org/10.5281/zenodo.20752650)** -> -> This repository is its landing page. Zenodo holds the archived, versioned, -> citable release. - ---- - -## Why it exists - -Accurate perception of surgical instruments underpins automated skill assessment -and computer-assisted training in minimally invasive surgery, but public -video-kinematic datasets are scarce, especially for non-in-vivo tasks. LASK -synchronises endoscopic video with ground-truth kinematics for **two graspers -tracked continuously**, so instrument detection, pose estimation and skill -classification can all be benchmarked against the same trials. - -## What is in the dataset - -**37 annotated trials**, arranged as a cross-cohort benchmark: - -| Split | Cohort | Trials | Purpose | -|:--|:--|--:|:--| -| Train | 7-DoF | 19 | In-distribution training | -| Validation | 7-DoF | 10 | In-distribution validation | -| Test | 6-DoF | 8 | Out-of-distribution generalisation | - -Spanning surgeons from novice to expert. - -**Per-frame kinematics**, roughly 91,000 frames, time-aligned to the video: - -- Electromagnetic tool poses: positions in millimetres, orientations as unit quaternions -- Calibrated jaw angles for the 7-DoF cohorts -- Both instruments tracked throughout - -**Manual annotations** on labelled keyframes, for both instruments: - -- Instrument segmentation masks -- Tooltip and jaw (left/right) keypoints -- Shaft-joint keypoints -- Per-component visibility flags - -**Surgeon metadata**: handedness (left, right or both) and procedure experience, -both lifetime and over the last 12 months. +[![Release](https://img.shields.io/badge/Public_release-37_recordings-1f4e79?style=flat-square)](https://zenodo.org/records/20752651) +[![Analysis](https://img.shields.io/badge/Analysis-115_recordings-548235?style=flat-square)](docs/ANALYSIS.md) +[![Validation](https://github.com/omariosc/LASK/actions/workflows/validate.yml/badge.svg)](https://github.com/omariosc/LASK/actions/workflows/validate.yml) +[![Licence](https://img.shields.io/badge/Licence-CC_BY_4.0-orange?style=flat-square)](LICENSE) -## What it supports - -- Multi-class instrument detection and segmentation -- Instrument tracking through occlusion and clutter -- 6-DoF and 7-DoF pose estimation from video alone -- Surgical skill classification -- Cross-configuration generalisation, via the 6-DoF versus 7-DoF split +> **Release status** +> +> [Zenodo version 1.0](https://zenodo.org/records/20752651) contains 37 +> recordings. The complete 115-recording collection used by the quantitative +> analysis will be added to the same +> [concept DOI](https://doi.org/10.5281/zenodo.20752650). This repository +> already provides the analysis code, aggregate results, feature definitions, +> and representative figures. It does not expose participant-level data from +> recordings that have not yet been released. + +## Start here + +- [Dataset guide](docs/DATASET.md) +- [Quantitative analysis](docs/ANALYSIS.md) +- [Motion feature definitions](docs/FEATURES.md) +- [Phase annotation protocol](docs/PHASES.md) +- [Analysis script](scripts/build_scirep_analysis.py) +- [Aggregate result tables](data/derived) +- [Representative figures](paper/figures) + +## Collection at a glance + +The three cohorts used different collection settings. Raw measurements should +therefore be interpreted within cohort or with cohort included explicitly in +the statistical model. + +| Cohort | Collection | Instrument channels | Frame rate | Public v1.0 | Complete collection | +|:--|:--|:--|--:|--:|--:| +| Paediatric | British Association of Paediatric Endoscopic Surgeons meeting, November 2024 | Position, orientation, relative jaw opening | 13 frames/s | 10 | 30 | +| Urology 1 | Urology boot camp, October 2023 | Position and orientation | 26 frames/s | 8 | 24 | +| Urology 2 | Urology boot camp, October 2024 | Position, orientation, relative jaw opening | 13 frames/s | 19 | 61 | +| **Total** | | | | **37** | **115** | + +The public release is organised as Urology 2 training data, Paediatric +validation data, and Urology 1 testing data. The names `7DOF2024`, +`BAPES2024`, and `6DOF2023` are retained in files for compatibility. + +## What each recording contains + +- Endoscopic video of a complete peg-transfer attempt. +- Three-dimensional tool position in millimetres for both instruments. +- Tool orientation as a unit quaternion for both instruments. +- A relative jaw-opening signal in the two seven-channel cohorts. This is a + per-recording voltage-derived opening fraction, not an absolute jaw angle. +- Manual masks and instrument landmarks on selected video frames. +- Self-reported handedness and laparoscopic procedure experience where + collected. +- Dense action-phase labels for the currently annotated subset. + +The electromagnetic sensor is mounted near the instrument base and calibrated +to estimate tool-tip position. The complete data structure and coordinate +conventions are described in the [dataset guide](docs/DATASET.md). + +## Quantitative analysis + +The companion analysis keeps four denominators separate. + +| Analysis unit | Recordings | Study identifiers | Purpose | +|:--|--:|--:|:--| +| Complete inventory | 115 | 111 | Describe every available recording | +| Primary motion set | 107 | 107 | Motion inference without repeated identifiers | +| Phase inventory | 38 | 37 | Describe every densely annotated recording | +| Primary phase set | 34 | 34 | Phase comparisons without repeated identifiers | + +The 38 phase-labelled recordings contain 425 placement-complete transfer +cycles. The primary phase set contains 383 cycles. + +The main findings are: + +- No novice, intermediate, and expert procedure-count comparison remained + significant after correction across 29 motion features. +- Eight features had modest within-cohort associations with lifetime procedure + volume. Longer experience was associated with shorter analysed duration, + lower normalised jerk, fewer speed peaks, shorter Tool 1 path length, faster + Tool 2 movement, and stronger bimanual correlation. +- Cohort accounted for 88% and 92% of the variation in Tool 1 and Tool 2 speed, + respectively. This shows why unadjusted pooling across collection settings is + misleading. +- Data-derived lower, middle, and upper motion-score bands had negligible + agreement with procedure-count groups (adjusted Rand index 0.019). These are + descriptive motion strata, not clinical skill grades. +- Models recovered the motion-score rule with high accuracy because the target + was calculated from the same motion domains. This is a software consistency + check and must not be interpreted as independent skill prediction. + +Full methods, confidence intervals, corrected probability values, sensitivity +analyses, and limitations are provided in the +[analysis guide](docs/ANALYSIS.md). + +## Representative results + +### Procedure experience and motion + +![Within-cohort experience links](paper/figures/fig4_experience_links.png) + +The same experience measure can have different motion relationships in each +cohort. The top row shows bimanual correlation across all 107 primary +recordings. The lower row uses the phase-labelled subset to show mean cycle +duration. Corrected all-recording results are available in +[`all_trial_feature_statistics.csv`](data/derived/all_trial_feature_statistics.csv). + +### Why cohort adjustment matters + +![Cohort and procedure effects](paper/figures/fig21_cohort_effect_forest.png) + +Each colour shows the expert-minus-novice effect estimated within one cohort. +The wide and sometimes inconsistent intervals show why a pooled effect can +hide acquisition-specific uncertainty. The analysis therefore avoids treating +the three cohorts as if their coordinate systems and equipment were +interchangeable. + +### Continuous score checks + +![Continuous motion-score validation](paper/figures/fig23_continuous_performance_validation.png) + +The coordination and control score is interpreted continuously. Duration and +task efficiency are shown as contextual measurements rather than external +clinical validation. -## Getting the data +Additional examples include the +[cohort inventory](paper/figures/fig1_cohort_structure.png) and +[phase timing analysis](paper/figures/fig3_phase_timing.png). + +## Reproducing the analysis -Download from the Zenodo record: **[10.5281/zenodo.20752650](https://doi.org/10.5281/zenodo.20752650)** +Create an environment and install the recorded package versions: + +```bash +python3 -m venv .venv +source .venv/bin/activate +python -m pip install -r requirements.txt +``` -Released under [CC BY 4.0](https://creativecommons.org/licenses/by/4.0/), so it -may be used commercially and redistributed provided the dataset is credited. +The script uses the sibling `BTPN-MT` and `AI-ELT` paths by default. Other +locations can be supplied explicitly: -## Work using LASK +```bash +export LASK_PHASE_CACHE=/path/to/phase_cache +export LASK_AI_ELT_ROOT=/path/to/AI-ELT +export LASK_ORIGIN_MOTION=/path/to/per_recording_kinematic_json +python scripts/build_scirep_analysis.py +``` -| Work | Venue | What it did | -|:--|:--|:--| -| [7-DoF Laparoscopic Peg Transfer Dataset for Surgical Skill Assessment](https://doi.org/10.3389/978-2-8325-5137-0) | MIUA 2025 | Introduced the dataset. Won Best Presentation at the Doctoral Consortium | -| [Real-Time Tool Detection in Laparoscopic Datasets for Surgical Training in Low-Resource Settings](https://doi.org/10.1049/htl2.70045) | *Healthcare Technology Letters* **12**(1) | Real-time detection on low-cost embedded devices | -| [Bayesian Temporal Pose Networks](https://github.com/omariosc/BTPN) | MICCAI 2026 | Uncertainty-calibrated 7-DoF vision-only pose tracking | +Expected inputs and their schemas are listed in +[`data/README.md`](data/README.md). The current Zenodo release does not yet +contain every cache needed to regenerate the 115-recording paper analysis. +Until the staged release is expanded, the repository provides the aggregate +outputs and file hashes produced by the complete internal collection. + +## Interpretation boundaries + +Procedure volume is an experience measure, not a direct assessment of +competence. The motion-score bands are relative partitions of two calculated +motion domains. They are not scores from the Objective Structured Assessment +of Technical Skills, Global Operative Assessment of Laparoscopic Skills, +Global Evaluative Assessment of Robotic Skills, McGill Inanimate System for +Training and Evaluation of Laparoscopic Skills, or Fundamentals of +Laparoscopic Surgery. + +One trained researcher annotated the action phases. Participant experience +metadata were not visible during annotation. Differences in apparent +confidence and fluency could still be perceived from the videos, but informal +impressions of skill were not recorded and were not used to define phase +labels. Independent phase annotation and expert rating remain planned +validation work. ## Citation -Please cite the dataset by its DOI. Machine-readable metadata is in -[`CITATION.cff`](CITATION.cff). +Please cite the dataset using the concept DOI so that the citation resolves to +the latest release: ```bibtex @dataset{Choudhry2026LASK, @@ -91,32 +188,18 @@ Please cite the dataset by its DOI. Machine-readable metadata is in author = {Choudhry, Omar and Jones, Dominic}, year = {2026}, publisher = {Zenodo}, - version = {1.0}, doi = {10.5281/zenodo.20752650}, url = {https://doi.org/10.5281/zenodo.20752650} } ``` -If you use the dataset for skill assessment, please also cite the paper that -introduced it: - -```bibtex -@inproceedings{Choudhry2025PegTransfer, - title = {7-DoF Laparoscopic Peg Transfer Dataset for Surgical Skill Assessment}, - author = {Choudhry, Omar and Ali, Sharib and Rajasundaram, Ramanan and - Biyani, Chandra Shekhar and Jones, Dominic}, - booktitle = {Medical Image Understanding and Analysis (MIUA)}, - publisher = {Frontiers Media SA}, - year = {2025}, - doi = {10.3389/978-2-8325-5137-0} -} -``` - -## Licence - -Dataset and the contents of this repository: [CC BY 4.0](LICENSE). +The original dataset paper and related work are listed in +[`CITATION.cff`](CITATION.cff). -## Contact +## Licence and contact -[Omar Choudhry](https://omarchoudhry.co.uk), School of Computing, University of -Leeds. Open an issue here with questions about the dataset. +The dataset and repository contents are released under the +[Creative Commons Attribution 4.0 licence](LICENSE). For questions, open a +GitHub issue or contact +[Omar Choudhry](https://omarchoudhry.co.uk), School of Computing, +University of Leeds. diff --git a/data/README.md b/data/README.md new file mode 100644 index 0000000..fbb2351 --- /dev/null +++ b/data/README.md @@ -0,0 +1,80 @@ +# Analysis data + +## Public contents + +`derived/` contains aggregate tables and machine-readable summaries generated +from the complete collection. These files support review of the reported +statistics without exposing participant-level rows that are not yet available +in the staged Zenodo release. + +Important files: + +| File | Contents | +|:--|:--| +| `analysis_manifest.json` | Software versions, random seeds, input hashes, and portable input locations | +| `analysis_summary.json` | Main denominators, model summaries, clustering diagnostics, and sensitivity checks | +| `all_trial_feature_statistics.csv` | Effect sizes, confidence intervals, continuous correlations, and corrected probabilities for 29 features | +| `cohort_specific_effects.csv` | Cohort-specific novice-versus-expert estimates | +| `cohort_adjusted_models.csv` | Cohort-adjusted procedure-volume models | +| `feature_dictionary.csv` | Feature labels, interpretation, direction, domain, and availability | +| `classification_summary.json` | Procedure-group classification checks | +| `metric_band_classification_summary.json` | Procedure-based targets and circular motion-band reconstruction checks | +| `kmeans_cluster_summary.csv` | Descriptive motion-score band sizes and centres | +| `kmeans_validation.csv` | Internal indices for alternative values of `k` | +| `processing_sensitivity.csv` | Position and orientation processing sensitivity | +| `trimming_rule_validation.csv` | Automatic versus labelled task-boundary agreement | +| `phase_kinematics_summary.csv` | Aggregate phase-specific duration, speed, and path-rate summaries | + +## Inputs required by the full script + +The complete analysis entry point expects two roots. + +### Phase cache + +Set `LASK_PHASE_CACHE` to a directory containing: + +| File | Minimum fields | +|:--|:--| +| `canonical_trials.csv` | `dataset`, `trial_name`, `trial_number`, `total_procedures`, `skill_category`, inclusion flags | +| `trials.parquet` | Recording identifiers, frame count, duration, cycle count, phase fractions, procedure metadata | +| `cycles.parquet` | Recording identifier, cycle index, duration, transitions, and phase fractions | +| `frames.parquet` | Recording identifier, frame index, phase labels, cycle index, and tool-specific labels | + +### AI-ELT root + +Set `LASK_AI_ELT_ROOT` to the AI-ELT project root containing: + +```text +outputs/ +├── motion/ +│ └── combined_motion_metrics.csv +├── papers/paper1/results/ +│ └── extended_features_cache_trimmed.csv +└── ssl/origin/origin_data/ORIGIN_ALL/ + └── .json +``` + +The per-recording JSON files contain a dense `features` matrix. Its first +14 columns are: + +```text +tool1 x, y, z, qw, qx, qy, qz, +tool2 x, y, z, qw, qx, qy, qz +``` + +`LASK_ORIGIN_MOTION` can override that JSON directory. + +## Output policy + +The full script also writes participant-level intermediates. These remain +ignored by Git until the corresponding recordings and metadata are released +through the governed Zenodo record. Aggregate tables are tracked. + +When the Zenodo release is expanded, the release process should include: + +1. an analysis-ready manifest mapping public trial names to analysis keys; +2. the complete phase cache or a deterministic builder; +3. a converter from released kinematic CSV files to the per-recording analysis + input; +4. checksums matching `analysis_manifest.json`; +5. a rerun of the complete analysis from a clean environment. diff --git a/data/derived/all_trial_feature_statistics.csv b/data/derived/all_trial_feature_statistics.csv new file mode 100644 index 0000000..ce1656b --- /dev/null +++ b/data/derived/all_trial_feature_statistics.csv @@ -0,0 +1,30 @@ +feature,label,cliffs_delta_novice_expert,delta_ci_low,delta_ci_high,kruskal_h,kruskal_p,spearman_rho_procedures,spearman_p,spearman_ci_low,spearman_ci_high,spearman_meta_i2,n_cohorts,n_nonmissing,kruskal_q,spearman_q +bimanual_correlation,Bimanual correlation,-0.44337811900191937,-0.6813819577735125,-0.1746641074856046,11.583733754844957,0.07192611507002791,0.26519610813065986,0.007154035304095336,0.07356633757928907,0.43794014383705293,59.69385918068588,3,107,0.3170324957516792,0.025933377977345594 +tool2_avg_speed,Tool 2 speed,-0.43953934740882916,-0.6852207293666027,-0.16314779270633398,9.446343678317177,0.14998807706400527,0.26545918253608347,0.007094210430320777,0.07384779717900485,0.4381688360883283,0.0,3,107,0.3170324957516792,0.025933377977345594 +tool1_avg_speed,Tool 1 speed,-0.3666026871401152,-0.6314779270633397,-0.08253358925143954,10.070936294222903,0.12169634914428706,0.20642377607265003,0.0381460925868868,0.011442656610917675,0.3862823505464995,0.0,3,107,0.3170324957516792,0.11062366850197172 +tool2_range_z,Tool 2 depth range,-0.362763915547025,-0.6353166986564299,-0.05566218809980806,8.038309445980472,0.23531007622449016,0.23875854482084943,0.015948005812675872,0.04543599067993563,0.4148432170028269,0.0,3,107,0.45493281403401425,0.051388018729733365 +tool2_working_volume,Tool 2 working volume,-0.2629558541266795,-0.5470249520153551,0.02111324376199616,11.752049475787643,0.06773444765157813,0.19246671234055343,0.05368284045721828,-0.0030923734558430975,0.3738467701508323,0.0,3,107,0.3170324957516792,0.12973353110494418 +tool2_active_time_ratio,Tool 2 active time,-0.236084452975048,-0.5124760076775432,0.0710172744721689,3.852667833233512,0.6966064736925249,0.17384877612433292,0.08209198970146667,-0.022353502034510433,0.3571562667897536,0.0,3,107,0.8417328223784676,0.18312828471865641 +tool1_range_z,Tool 1 depth range,-0.2053742802303263,-0.4779270633397313,0.0671785028790787,11.39277866522508,0.07696966614726616,0.0024388652860445384,0.9807380796990711,-0.19309593792515908,0.19778735062844746,78.79217874757137,3,107,0.3170324957516792,0.9807380796990711 +tool2_working_area_xy,Tool 2 working area,-0.16314779270633398,-0.4473128598848368,0.1324376199616123,12.920231023058719,0.044320318949920945,0.15332898025471015,0.1260303123915256,-0.04341479499768882,0.3386241785150563,54.01057825144936,3,107,0.3170324957516792,0.24365860395694947 +simultaneous_motion_ratio,Simultaneous motion,-0.15547024952015356,-0.43570057581573896,0.1325335892514399,2.352682271610704,0.8845768427286301,0.14830662457394425,0.13912187580250768,-0.04854318415091862,0.33406632214142973,0.0,3,107,0.9866434015050106,0.25215839989204514 +tool1_angular_velocity_cv,Tool 1 angular velocity variability,-0.11708253358925144,-0.4165067178502879,0.1746641074856046,5.405935780460767,0.4928977444594195,-0.007374101334269107,0.941805160971722,-0.20252498172427333,0.1883400995702248,49.41660025033508,3,107,0.6806683137772935,0.9754410595778549 +tool1_working_volume,Tool 1 working volume,-0.09788867562380038,-0.381957773512476,0.21689059500959693,9.442730244602913,0.15016726399115785,-0.05283186103628268,0.6006290186190759,-0.24573723779662934,0.1440988087013587,70.2484119675861,3,107,0.3170324957516792,0.6451200570353037 +bimanual_symmetry,Bimanual symmetry,-0.09404990403071017,-0.38589251439539346,0.2092130518234165,1.4775509440425625,0.9609718010133059,0.07096090270086099,0.4816446805886372,-0.12623255305805667,0.2627595073773897,0.0,3,107,0.9877753958666414,0.6072911190030643 +tool_distance_cv,Tool distance variability,-0.07869481765834933,-0.381957773512476,0.22082533589251474,0.9402734096993726,0.9877753958666414,0.0967073250888628,0.33687656619412654,-0.10063756236961648,0.2867306638025451,0.0,3,107,0.9877753958666414,0.48526714094510986 +tool1_active_time_ratio,Tool 1 active time,-0.07869481765834933,-0.381957773512476,0.22840690978886757,1.8615186143753135,0.9319837998880092,0.0619234676460434,0.5393498793878274,-0.1351550073394433,0.25428870263461834,0.0,3,107,0.9877753958666414,0.6288042457022069 +tool1_rotation_per_path,Tool 1 rotation per path,-0.04798464491362764,-0.3550863723608445,0.2476007677543186,6.11896892398272,0.4099960344698175,0.0615094119510837,0.5420726256053509,-0.1355630308075032,0.25389989907560206,69.18693626791988,3,107,0.6605491666458172,0.6288042457022069 +tool2_rotation_per_path,Tool 2 rotation per path,-0.04030710172744722,-0.33589251439539347,0.2591170825335892,4.939842744995474,0.5515516316447138,0.0897623157386279,0.37292929730177227,-0.10756754659238395,0.2802878771885398,26.67760421067217,3,107,0.7270453326225772,0.491588619170518 +tool1_working_area_xy,Tool 1 working area,0.036468330134357005,-0.2514395393474088,0.3435700575815739,5.820839351524582,0.44355627473497816,-0.0938566809851888,0.35140034344301063,-0.2840882397713419,0.10348435799322389,27.571098955890267,3,107,0.6770069456481246,0.48526714094510986 +bimanual_lag_s,Bimanual lag,0.12284069097888675,-0.0671785028790787,0.30906909788867576,3.4167085736608502,0.49065545885155004,-0.12756724230420405,0.26036445587208346,-0.33781924920756906,0.0948110353497303,0.0,2,107,0.6806683137772935,0.39739838001528527 +bimanual_concurrent_efficiency,Concurrent efficiency,0.1324376199616123,-0.16698656429942418,0.42034548944337813,6.255847554442752,0.3951487351725862,0.05434034789577619,0.5902509335709227,-0.14261706874968966,0.24715818464282197,0.0,3,107,0.6605491666458172,0.6451200570353037 +combined_idle_ratio,Combined idle ratio,0.15930902111324377,-0.1362763915547025,0.43570057581573896,2.582682735526149,0.8591018094151637,-0.12851294408927272,0.20079592076353536,-0.3160182986269308,0.0686542006839411,0.0,3,107,0.9866434015050106,0.3235045390079181 +tool2_path_length,Tool 2 path length,0.16698656429942418,-0.12859884836852206,0.45873320537428025,4.3259270970116,0.6326606280766702,-0.1626993657295065,0.10414478195750654,-0.34710474781701767,0.033818780204501575,0.0,3,107,0.7977025310531929,0.21572847691197783 +bimanual_dist_speed_corr,Distance-speed coupling,0.21305182341650672,-0.09788867562380038,0.4971209213051823,6.457136135650996,0.37397349378076755,-0.13287305287754708,0.18576878872865032,-0.32000556622565396,0.06423790866459668,19.20870804646566,3,107,0.6605491666458172,0.3168996984194623 +tool1_path_length,Tool 1 path length,0.2783109404990403,-0.028790786948176585,0.5547024952015355,9.38510505560636,0.15305017036287963,-0.27769187125220074,0.00475570678189519,-0.44877815668567483,-0.08696917530738506,0.0,3,107,0.3170324957516792,0.022985916112493423 +tool2_angular_velocity_cv,Tool 2 angular velocity variability,0.30902111324376197,0.028790786948176585,0.5585412667946257,10.774072184447647,0.0956150561686818,-0.1949345551877209,0.050610254353440826,-0.3760503454483363,0.0005283360255154149,0.0,3,107,0.3170324957516792,0.12973353110494418 +tool2_normalized_jerk,Tool 2 normalised jerk,0.4126679462571977,0.1362763915547025,0.6660268714011516,10.234331232795597,0.11512439345342916,-0.30095850521529366,0.0021084284496400976,-0.4688249105810884,-0.11211007560374386,0.0,3,107,0.3170324957516792,0.02038147501318761 +tool2_num_speed_peaks,Tool 2 speed peaks,0.4165067178502879,0.1457773512476008,0.6776391554702499,9.957316556589404,0.1264610585640119,-0.28802921047107855,0.0033422853252693006,-0.4577062141617639,-0.0981093391515713,0.0,3,107,0.3170324957516792,0.022985916112493423 +tool1_num_speed_peaks,Tool 1 speed peaks,0.418426103646833,0.12471209213051825,0.6890595009596929,9.917667997223782,0.1281622922367285,-0.282473804927565,0.0040466778158233735,-0.452912422522839,-0.09211656414791568,0.0,3,107,0.3170324957516792,0.022985916112493423 +total_time,Analysed duration,0.4472168905950096,0.1785028790786948,0.6929942418426107,11.451832434664146,0.07537628840171491,-0.32716976238946716,0.0007726322842619239,-0.49120351688973213,-0.14072539886146385,0.0,3,107,0.3170324957516792,0.011203168121797896 +tool1_normalized_jerk,Tool 1 normalised jerk,0.45489443378119004,0.1708253358925144,0.708253358925144,12.231543642480347,0.05699825695363934,-0.33800207579813035,0.0004958456978932772,-0.500389104235357,-0.15264291333999488,0.0,3,107,0.3170324957516792,0.011203168121797896 diff --git a/data/derived/analysis_manifest.json b/data/derived/analysis_manifest.json new file mode 100644 index 0000000..61c8b63 --- /dev/null +++ b/data/derived/analysis_manifest.json @@ -0,0 +1,64 @@ +{ + "analysis_script": { + "path": "scripts/build_scirep_analysis.py", + "sha256": "2a147245ae25e03f5c998ea95fa94efb905e5b05e78dcdcbb942dbf5d2cafe82" + }, + "python": { + "executable": "python3.14", + "version": "3.14.6 (main, Jun 10 2026, 10:03:53) [Clang 21.0.0 (clang-2100.0.123.102)]", + "platform": "macOS-26.5.2-arm64-arm-64bit-Mach-O" + }, + "packages": { + "numpy": "2.4.4", + "pandas": "3.0.1", + "matplotlib": "3.10.8", + "seaborn": "0.13.2", + "scipy": "1.17.1", + "scikit-learn": "1.8.0", + "pyarrow": "23.0.1" + }, + "inputs": { + "canonical_trials": { + "path": "$LASK_PHASE_CACHE/canonical_trials.csv", + "exists": true, + "size_bytes": 10880, + "sha256": "21fdd897bd4fadce9654339159b12e5519efb14d42dfc04cddb2e931bbc49914" + }, + "phase_trials": { + "path": "$LASK_PHASE_CACHE/trials.parquet", + "exists": true, + "size_bytes": 18820, + "sha256": "33a59d1db3d2202c5f82fa0f7d32ecdf8d451fdff779af0182b04935df8ff30f" + }, + "phase_cycles": { + "path": "$LASK_PHASE_CACHE/cycles.parquet", + "exists": true, + "size_bytes": 33995, + "sha256": "60f96435ee9ca525661c83c7c0a3a83483131487a54b333232d12ed3b7ac70bb" + }, + "phase_frames": { + "path": "$LASK_PHASE_CACHE/frames.parquet", + "exists": true, + "size_bytes": 384362, + "sha256": "f59217f43a631b618a138e6b7c37d9fbecfb10b283cd47ab0c648cf0b7b2ffd5" + }, + "combined_motion_metrics": { + "path": "$LASK_AI_ELT_ROOT/outputs/motion/combined_motion_metrics.csv", + "exists": true, + "size_bytes": 115865, + "sha256": "dadbc6c7d4629c8058bf3bf34df7e9c73f273861d43ef8fe11ef1681fb4f1265" + }, + "trimmed_extended_features": { + "path": "$LASK_AI_ELT_ROOT/outputs/papers/paper1/results/extended_features_cache_trimmed.csv", + "exists": true, + "size_bytes": 297728, + "sha256": "1a72effd894355ef4a8f1f625845cac274b456f4759db8c4e33a7e095002ded2" + } + }, + "random_seeds": { + "bootstrap_and_sampling": 20260618, + "classification_cv": 13, + "pca": 20260618, + "kmeans": 20260618 + } +} \ No newline at end of file diff --git a/data/derived/analysis_summary.json b/data/derived/analysis_summary.json new file mode 100644 index 0000000..804a956 --- /dev/null +++ b/data/derived/analysis_summary.json @@ -0,0 +1,940 @@ +{ + "all_trials": 115, + "phase_trials": 38, + "phase_cycles": 425, + "feature_dictionary_rows": 30, + "trimming_rule_validation": [ + { + "cohort": "All cohorts", + "n": 34, + "median_abs_start_error_s": 2.576923076923077, + "median_abs_end_error_s": 1.5, + "median_abs_duration_error_s": 4.653846153846146, + "median_interval_iou": 0.9678538255991654 + }, + { + "cohort": "Paediatric", + "n": 9, + "median_abs_start_error_s": 3.6923076923076925, + "median_abs_end_error_s": 1.6153846153846154, + "median_abs_duration_error_s": 5.307692307692321, + "median_interval_iou": 0.9521149241819633 + }, + { + "cohort": "Urology 1 (2023)", + "n": 10, + "median_abs_start_error_s": 1.3653846153846154, + "median_abs_end_error_s": 1.3653846153846154, + "median_abs_duration_error_s": 2.5576923076923137, + "median_interval_iou": 0.9803009547741184 + }, + { + "cohort": "Urology 2 (2024)", + "n": 15, + "median_abs_start_error_s": 4.230769230769231, + "median_abs_end_error_s": 1.5384615384615385, + "median_abs_duration_error_s": 6.307692307692292, + "median_interval_iou": 0.955458989679522 + } + ], + "processing_sensitivity": [ + { + "processing": "No position smoothing", + "position_window_s": 0.0, + "orientation_window_s": 0.0, + "n": 107, + "score_spearman_rho": 0.9833552945234894, + "median_absolute_score_change": 0.054545454545454675, + "fixed_cut_assignment_ari": 0.7461892039309828, + "refit_assignment_ari": 0.3175844249017427 + }, + { + "processing": "Primary processing", + "position_window_s": 0.4, + "orientation_window_s": 0.0, + "n": 107, + "score_spearman_rho": 1.0, + "median_absolute_score_change": 0.0, + "fixed_cut_assignment_ari": 1.0, + "refit_assignment_ari": 1.0 + }, + { + "processing": "0.8 s position window", + "position_window_s": 0.8, + "orientation_window_s": 0.0, + "n": 107, + "score_spearman_rho": 0.9836590342619369, + "median_absolute_score_change": 0.054545454545454675, + "fixed_cut_assignment_ari": 0.85387290419947, + "refit_assignment_ari": 0.8053423741454286 + }, + { + "processing": "0.4 s orientation sensitivity", + "position_window_s": 0.4, + "orientation_window_s": 0.4, + "n": 107, + "score_spearman_rho": 0.9880088170937018, + "median_absolute_score_change": 0.042424242424242475, + "fixed_cut_assignment_ari": 0.8286069303742338, + "refit_assignment_ari": 0.42755379100614965 + } + ], + "classification": { + "available": true, + "features": [ + "total_time", + "bimanual_correlation", + "bimanual_lag_s", + "bimanual_concurrent_efficiency", + "bimanual_dist_speed_corr", + "simultaneous_motion_ratio", + "bimanual_symmetry", + "combined_idle_ratio", + "tool1_normalized_jerk", + "tool2_normalized_jerk", + "tool1_active_time_ratio", + "tool2_active_time_ratio", + "tool1_path_length", + "tool2_path_length", + "tool1_avg_speed", + "tool2_avg_speed", + "tool_distance_cv", + "tool_close_proximity_ratio", + "tool1_working_volume", + "tool2_working_volume", + "tool1_working_area_xy", + "tool2_working_area_xy", + "tool1_range_z", + "tool2_range_z", + "tool1_rotation_per_path", + "tool2_rotation_per_path", + "tool1_angular_velocity_cv", + "tool2_angular_velocity_cv", + "tool1_num_speed_peaks", + "tool2_num_speed_peaks" + ], + "time_only_features": [ + "total_time" + ], + "stratified_group_5fold": { + "time_only_logistic_regression": { + "mean_balanced_accuracy": 0.44841269841269843, + "sd": 0.16861577600976474, + "features": [ + "total_time" + ] + }, + "time_only_random_forest": { + "mean_balanced_accuracy": 0.39550264550264547, + "sd": 0.05432467376806009, + "features": [ + "total_time" + ] + }, + "all_features_logistic_regression": { + "mean_balanced_accuracy": 0.3603174603174603, + "sd": 0.06844059019686591, + "features": [ + "total_time", + "bimanual_correlation", + "bimanual_lag_s", + "bimanual_concurrent_efficiency", + "bimanual_dist_speed_corr", + "simultaneous_motion_ratio", + "bimanual_symmetry", + "combined_idle_ratio", + "tool1_normalized_jerk", + "tool2_normalized_jerk", + "tool1_active_time_ratio", + "tool2_active_time_ratio", + "tool1_path_length", + "tool2_path_length", + "tool1_avg_speed", + "tool2_avg_speed", + "tool_distance_cv", + "tool_close_proximity_ratio", + "tool1_working_volume", + "tool2_working_volume", + "tool1_working_area_xy", + "tool2_working_area_xy", + "tool1_range_z", + "tool2_range_z", + "tool1_rotation_per_path", + "tool2_rotation_per_path", + "tool1_angular_velocity_cv", + "tool2_angular_velocity_cv", + "tool1_num_speed_peaks", + "tool2_num_speed_peaks" + ] + }, + "all_features_random_forest": { + "mean_balanced_accuracy": 0.34325396825396826, + "sd": 0.10826701395943916, + "features": [ + "total_time", + "bimanual_correlation", + "bimanual_lag_s", + "bimanual_concurrent_efficiency", + "bimanual_dist_speed_corr", + "simultaneous_motion_ratio", + "bimanual_symmetry", + "combined_idle_ratio", + "tool1_normalized_jerk", + "tool2_normalized_jerk", + "tool1_active_time_ratio", + "tool2_active_time_ratio", + "tool1_path_length", + "tool2_path_length", + "tool1_avg_speed", + "tool2_avg_speed", + "tool_distance_cv", + "tool_close_proximity_ratio", + "tool1_working_volume", + "tool2_working_volume", + "tool1_working_area_xy", + "tool2_working_area_xy", + "tool1_range_z", + "tool2_range_z", + "tool1_rotation_per_path", + "tool2_rotation_per_path", + "tool1_angular_velocity_cv", + "tool2_angular_velocity_cv", + "tool1_num_speed_peaks", + "tool2_num_speed_peaks" + ] + } + }, + "leave_one_dataset_out": { + "time_only_logistic_regression": [ + { + "held_out_dataset": "6DOF2023", + "balanced_accuracy": 0.3333333333333333, + "n_test": 24 + }, + { + "held_out_dataset": "7DOF2024", + "balanced_accuracy": 0.5279866332497911, + "n_test": 55 + }, + { + "held_out_dataset": "BAPES2024", + "balanced_accuracy": 0.5413165266106442, + "n_test": 28 + } + ], + "time_only_random_forest": [ + { + "held_out_dataset": "6DOF2023", + "balanced_accuracy": 0.5370370370370371, + "n_test": 24 + }, + { + "held_out_dataset": "7DOF2024", + "balanced_accuracy": 0.4357560568086884, + "n_test": 55 + }, + { + "held_out_dataset": "BAPES2024", + "balanced_accuracy": 0.4747899159663866, + "n_test": 28 + } + ], + "all_features_logistic_regression": [ + { + "held_out_dataset": "6DOF2023", + "balanced_accuracy": 0.46296296296296297, + "n_test": 24 + }, + { + "held_out_dataset": "7DOF2024", + "balanced_accuracy": 0.32497911445279865, + "n_test": 55 + }, + { + "held_out_dataset": "BAPES2024", + "balanced_accuracy": 0.36274509803921573, + "n_test": 28 + } + ], + "all_features_random_forest": [ + { + "held_out_dataset": "6DOF2023", + "balanced_accuracy": 0.27777777777777773, + "n_test": 24 + }, + { + "held_out_dataset": "7DOF2024", + "balanced_accuracy": 0.3505430242272347, + "n_test": 55 + }, + { + "held_out_dataset": "BAPES2024", + "balanced_accuracy": 0.30392156862745096, + "n_test": 28 + } + ] + }, + "top_random_forest_features": [] + }, + "metric_band_classification": { + "available": true, + "features": [ + "Bimanual coordination", + "Instrument motion control", + "Coordination-control composite" + ], + "targets": { + "motion_defined": { + "available": true, + "label": "Motion-score bands", + "class_counts": { + "metric_high": 50, + "metric_middle": 41, + "metric_low": 16 + }, + "models": { + "logistic_regression": { + "stratified_cv": { + "mean_balanced_accuracy": 0.9481481481481481, + "sd": 0.0863844725162267, + "n_splits": 5 + }, + "leave_one_dataset_out": [ + { + "held_out_dataset": "6DOF2023", + "balanced_accuracy": 1.0, + "n_test": 24 + }, + { + "held_out_dataset": "7DOF2024", + "balanced_accuracy": 0.9722222222222222, + "n_test": 55 + }, + { + "held_out_dataset": "BAPES2024", + "balanced_accuracy": 1.0, + "n_test": 28 + } + ] + }, + "random_forest": { + "stratified_cv": { + "mean_balanced_accuracy": 0.9555555555555555, + "sd": 0.054433105395181765, + "n_splits": 5 + }, + "leave_one_dataset_out": [ + { + "held_out_dataset": "6DOF2023", + "balanced_accuracy": 1.0, + "n_test": 24 + }, + { + "held_out_dataset": "7DOF2024", + "balanced_accuracy": 0.9583333333333334, + "n_test": 55 + }, + { + "held_out_dataset": "BAPES2024", + "balanced_accuracy": 1.0, + "n_test": 28 + } + ] + } + } + }, + "procedure_fixed": { + "available": true, + "label": "Procedure thresholds", + "class_counts": { + "procedure_expert": 42, + "procedure_novice": 34, + "procedure_intermediate": 31 + }, + "models": { + "logistic_regression": { + "stratified_cv": { + "mean_balanced_accuracy": 0.3843915343915344, + "sd": 0.11729933197995121, + "n_splits": 5 + }, + "leave_one_dataset_out": [ + { + "held_out_dataset": "6DOF2023", + "balanced_accuracy": 0.4074074074074074, + "n_test": 24 + }, + { + "held_out_dataset": "7DOF2024", + "balanced_accuracy": 0.4484544695071011, + "n_test": 55 + }, + { + "held_out_dataset": "BAPES2024", + "balanced_accuracy": 0.26820728291316526, + "n_test": 28 + } + ] + }, + "random_forest": { + "stratified_cv": { + "mean_balanced_accuracy": 0.362962962962963, + "sd": 0.06755414162236494, + "n_splits": 5 + }, + "leave_one_dataset_out": [ + { + "held_out_dataset": "6DOF2023", + "balanced_accuracy": 0.3333333333333333, + "n_test": 24 + }, + { + "held_out_dataset": "7DOF2024", + "balanced_accuracy": 0.33266499582289055, + "n_test": 55 + }, + { + "held_out_dataset": "BAPES2024", + "balanced_accuracy": 0.3991596638655462, + "n_test": 28 + } + ] + } + } + }, + "procedure_tertiles": { + "available": true, + "label": "Procedure tertiles", + "class_counts": { + "procedure_tertile_low": 36, + "procedure_tertile_high": 36, + "procedure_tertile_mid": 35 + }, + "models": { + "logistic_regression": { + "stratified_cv": { + "mean_balanced_accuracy": 0.33333333333333337, + "sd": 0.12192940599113683, + "n_splits": 5 + }, + "leave_one_dataset_out": [ + { + "held_out_dataset": "6DOF2023", + "balanced_accuracy": 0.4598484848484849, + "n_test": 24 + }, + { + "held_out_dataset": "7DOF2024", + "balanced_accuracy": 0.41319966583124473, + "n_test": 55 + }, + { + "held_out_dataset": "BAPES2024", + "balanced_accuracy": 0.3333333333333333, + "n_test": 28 + } + ] + }, + "random_forest": { + "stratified_cv": { + "mean_balanced_accuracy": 0.2702380952380952, + "sd": 0.028071014577503817, + "n_splits": 5 + }, + "leave_one_dataset_out": [ + { + "held_out_dataset": "6DOF2023", + "balanced_accuracy": 0.36287878787878786, + "n_test": 24 + }, + { + "held_out_dataset": "7DOF2024", + "balanced_accuracy": 0.454469507101086, + "n_test": 55 + }, + { + "held_out_dataset": "BAPES2024", + "balanced_accuracy": 0.2916666666666667, + "n_test": 28 + } + ] + } + } + }, + "procedure_kmeans": { + "available": true, + "label": "Procedure K-means", + "class_counts": { + "procedure_kmeans_mid": 53, + "procedure_kmeans_high": 32, + "procedure_kmeans_low": 22 + }, + "models": { + "logistic_regression": { + "stratified_cv": { + "mean_balanced_accuracy": 0.5048340548340549, + "sd": 0.16704300258582674, + "n_splits": 5 + }, + "leave_one_dataset_out": [ + { + "held_out_dataset": "6DOF2023", + "balanced_accuracy": 0.5222222222222223, + "n_test": 24 + }, + { + "held_out_dataset": "7DOF2024", + "balanced_accuracy": 0.5847222222222223, + "n_test": 55 + }, + { + "held_out_dataset": "BAPES2024", + "balanced_accuracy": 0.28571428571428575, + "n_test": 28 + } + ] + }, + "random_forest": { + "stratified_cv": { + "mean_balanced_accuracy": 0.26275613275613274, + "sd": 0.04702399204348823, + "n_splits": 5 + }, + "leave_one_dataset_out": [ + { + "held_out_dataset": "6DOF2023", + "balanced_accuracy": 0.3277777777777778, + "n_test": 24 + }, + { + "held_out_dataset": "7DOF2024", + "balanced_accuracy": 0.3194444444444444, + "n_test": 55 + }, + { + "held_out_dataset": "BAPES2024", + "balanced_accuracy": 0.1984126984126984, + "n_test": 28 + } + ] + } + } + } + }, + "comparison_rows": [ + { + "target": "motion_defined", + "target_label": "Motion-score bands", + "model": "logistic_regression", + "model_label": "Logistic regression", + "stratified_cv_balanced_accuracy": 0.9481481481481481, + "stratified_cv_sd": 0.0863844725162267, + "leave_one_dataset_out_balanced_accuracy": 0.9907407407407408, + "leave_one_dataset_out_min": 0.9722222222222222, + "leave_one_dataset_out_max": 1.0, + "n_splits": 5 + }, + { + "target": "motion_defined", + "target_label": "Motion-score bands", + "model": "random_forest", + "model_label": "Random forest", + "stratified_cv_balanced_accuracy": 0.9555555555555555, + "stratified_cv_sd": 0.054433105395181765, + "leave_one_dataset_out_balanced_accuracy": 0.9861111111111112, + "leave_one_dataset_out_min": 0.9583333333333334, + "leave_one_dataset_out_max": 1.0, + "n_splits": 5 + }, + { + "target": "procedure_fixed", + "target_label": "Procedure thresholds", + "model": "logistic_regression", + "model_label": "Logistic regression", + "stratified_cv_balanced_accuracy": 0.3843915343915344, + "stratified_cv_sd": 0.11729933197995121, + "leave_one_dataset_out_balanced_accuracy": 0.3746897199425579, + "leave_one_dataset_out_min": 0.26820728291316526, + "leave_one_dataset_out_max": 0.4484544695071011, + "n_splits": 5 + }, + { + "target": "procedure_fixed", + "target_label": "Procedure thresholds", + "model": "random_forest", + "model_label": "Random forest", + "stratified_cv_balanced_accuracy": 0.362962962962963, + "stratified_cv_sd": 0.06755414162236494, + "leave_one_dataset_out_balanced_accuracy": 0.35505266434059, + "leave_one_dataset_out_min": 0.33266499582289055, + "leave_one_dataset_out_max": 0.3991596638655462, + "n_splits": 5 + }, + { + "target": "procedure_tertiles", + "target_label": "Procedure tertiles", + "model": "logistic_regression", + "model_label": "Logistic regression", + "stratified_cv_balanced_accuracy": 0.33333333333333337, + "stratified_cv_sd": 0.12192940599113683, + "leave_one_dataset_out_balanced_accuracy": 0.4021271613376876, + "leave_one_dataset_out_min": 0.3333333333333333, + "leave_one_dataset_out_max": 0.4598484848484849, + "n_splits": 5 + }, + { + "target": "procedure_tertiles", + "target_label": "Procedure tertiles", + "model": "random_forest", + "model_label": "Random forest", + "stratified_cv_balanced_accuracy": 0.2702380952380952, + "stratified_cv_sd": 0.028071014577503817, + "leave_one_dataset_out_balanced_accuracy": 0.3696716538821802, + "leave_one_dataset_out_min": 0.2916666666666667, + "leave_one_dataset_out_max": 0.454469507101086, + "n_splits": 5 + }, + { + "target": "procedure_kmeans", + "target_label": "Procedure K-means", + "model": "logistic_regression", + "model_label": "Logistic regression", + "stratified_cv_balanced_accuracy": 0.5048340548340549, + "stratified_cv_sd": 0.16704300258582674, + "leave_one_dataset_out_balanced_accuracy": 0.46421957671957675, + "leave_one_dataset_out_min": 0.28571428571428575, + "leave_one_dataset_out_max": 0.5847222222222223, + "n_splits": 5 + }, + { + "target": "procedure_kmeans", + "target_label": "Procedure K-means", + "model": "random_forest", + "model_label": "Random forest", + "stratified_cv_balanced_accuracy": 0.26275613275613274, + "stratified_cv_sd": 0.04702399204348823, + "leave_one_dataset_out_balanced_accuracy": 0.28187830687830684, + "leave_one_dataset_out_min": 0.1984126984126984, + "leave_one_dataset_out_max": 0.3277777777777778, + "n_splits": 5 + } + ], + "stratified_cv": { + "metric_band_logistic_regression": { + "mean_balanced_accuracy": 0.9481481481481481, + "sd": 0.0863844725162267, + "n_splits": 5 + }, + "metric_band_random_forest": { + "mean_balanced_accuracy": 0.9555555555555555, + "sd": 0.054433105395181765, + "n_splits": 5 + } + }, + "leave_one_dataset_out": { + "metric_band_logistic_regression": [ + { + "held_out_dataset": "6DOF2023", + "balanced_accuracy": 1.0, + "n_test": 24 + }, + { + "held_out_dataset": "7DOF2024", + "balanced_accuracy": 0.9722222222222222, + "n_test": 55 + }, + { + "held_out_dataset": "BAPES2024", + "balanced_accuracy": 1.0, + "n_test": 28 + } + ], + "metric_band_random_forest": [ + { + "held_out_dataset": "6DOF2023", + "balanced_accuracy": 1.0, + "n_test": 24 + }, + { + "held_out_dataset": "7DOF2024", + "balanced_accuracy": 0.9583333333333334, + "n_test": 55 + }, + { + "held_out_dataset": "BAPES2024", + "balanced_accuracy": 1.0, + "n_test": 28 + } + ] + } + }, + "kmeans": { + "available": true, + "features": [ + "Coordination-control composite" + ], + "n_recordings_assigned": 107, + "n_independent_identifiers_fitted": 107, + "metric_band_silhouette": 0.5340644126720104, + "metric_band_davies_bouldin": 0.5574945498396043, + "metric_band_calinski_harabasz": 205.40036661359844, + "metric_band_seed_ari_mean": 0.6881185433201465, + "metric_band_seed_ari_min": 0.2457480281764961, + "metric_band_bootstrap_assignment_ari_median": 0.7557894702759759, + "metric_band_bootstrap_assignment_ari_ci": [ + 0.25019440016739297, + 1.0 + ], + "metric_band_centres": [ + 2.1605203823953825, + 2.8661563368880443, + 3.498385281385281 + ], + "metric_band_thresholds": [ + 2.5133383596417134, + 3.1822708091366625 + ], + "metric_band_threshold_bootstrap_ci": [ + { + "threshold": 1, + "ci_low": 2.2926625593349086, + "median": 2.549881584464918, + "ci_high": 2.9269077055625816 + }, + { + "threshold": 2, + "ci_low": 3.0849795438675742, + "median": 3.198773797443426, + "ci_high": 3.5601637509174164 + } + ], + "sensitivity": { + "complete_case": { + "n_trials": 107, + "thresholds": [ + 2.5133383596417134, + 3.1822708091366625 + ], + "assignment_adjusted_rand_index": 1.0 + }, + "domain_weight_perturbation": { + "n_repetitions": 500, + "adjusted_rand_index_median": 0.9591035280410379, + "adjusted_rand_index_ci_low": 0.7199041122076191, + "adjusted_rand_index_ci_high": 1.0 + }, + "leave_one_domain_out": [ + { + "omitted_domain": "Bimanual coordination", + "adjusted_rand_index": 0.13060915248255114, + "thresholds": [ + 2.7302769243898277, + 3.630029209142112 + ] + }, + { + "omitted_domain": "Instrument motion control", + "adjusted_rand_index": 0.35444949343374316, + "thresholds": [ + 2.334584187210237, + 3.2698409477427335 + ] + } + ], + "leave_one_cohort_out": [ + { + "held_out_dataset": "BAPES2024", + "thresholds": [ + 2.3904748376623375, + 3.1671951659451656 + ], + "held_out_adjusted_rand_index": 0.865164531375405 + }, + { + "held_out_dataset": "6DOF2023", + "thresholds": [ + 2.5767321004648194, + 3.1817356808443398 + ], + "held_out_adjusted_rand_index": 1.0 + }, + { + "held_out_dataset": "7DOF2024", + "thresholds": [ + 2.535273368606702, + 3.215837742504409 + ], + "held_out_adjusted_rand_index": 0.8914523798531924 + } + ] + }, + "k_validation": [ + { + "k": 2, + "silhouette": 0.5662174552602269, + "davies_bouldin": 0.5878701558520018, + "calinski_harabasz": 183.48140711375248, + "inertia": 10.989325724233584 + }, + { + "k": 3, + "silhouette": 0.5340644126720104, + "davies_bouldin": 0.5574945498396043, + "calinski_harabasz": 205.40036661359844, + "inertia": 6.099493281079126 + }, + { + "k": 4, + "silhouette": 0.5526414765421528, + "davies_bouldin": 0.5339018410670197, + "calinski_harabasz": 274.1045156576356, + "inertia": 3.3608403214706066 + }, + { + "k": 5, + "silhouette": 0.5435087697321636, + "davies_bouldin": 0.5047211727805098, + "calinski_harabasz": 309.8046641567337, + "inertia": 2.296149497107527 + }, + { + "k": 6, + "silhouette": 0.5354954704018147, + "davies_bouldin": 0.4844300101979635, + "calinski_harabasz": 340.0575213267391, + "inertia": 1.692925659373024 + } + ], + "pca_variance": [ + 0.6118434437413051, + 0.38815655625869483 + ], + "adjusted_rand_vs_original_skill": 0.018846337788122808, + "cluster_summary": [ + { + "metric_cluster": "metric_low", + "cluster_label": "Lower motion-score band", + "n": 16, + "median_total_procedures": 40.0, + "procedure_iqr_low": 13.0, + "procedure_iqr_high": 85.0, + "mean_coordination_control_composite": 2.1605203823953825 + }, + { + "metric_cluster": "metric_middle", + "cluster_label": "Middle motion-score band", + "n": 41, + "median_total_procedures": 20.0, + "procedure_iqr_low": 5.0, + "procedure_iqr_high": 100.0, + "mean_coordination_control_composite": 2.8661563368880443 + }, + { + "metric_cluster": "metric_high", + "cluster_label": "Upper motion-score band", + "n": 50, + "median_total_procedures": 80.0, + "procedure_iqr_low": 16.25, + "procedure_iqr_high": 300.0, + "mean_coordination_control_composite": 3.4983852813852807 + } + ] + }, + "core_motion_score_mean_by_procedure_group": { + "expert": 3.180458496529925, + "intermediate": 2.9892845505748733, + "novice": 2.9633212375859435 + }, + "top_feature_effects": [ + { + "feature": "bimanual_correlation", + "label": "Bimanual correlation", + "cliffs_delta_novice_expert": -0.44337811900191937, + "delta_ci_low": -0.6813819577735125, + "delta_ci_high": -0.1746641074856046, + "kruskal_h": 11.583733754844957, + "kruskal_p": 0.07192611507002791, + "spearman_rho_procedures": 0.26519610813065986, + "spearman_p": 0.007154035304095336, + "spearman_ci_low": 0.07356633757928907, + "spearman_ci_high": 0.43794014383705293, + "spearman_meta_i2": 59.69385918068588, + "n_cohorts": 3, + "n_nonmissing": 107, + "kruskal_q": 0.3170324957516792, + "spearman_q": 0.025933377977345594 + }, + { + "feature": "tool2_avg_speed", + "label": "Tool 2 speed", + "cliffs_delta_novice_expert": -0.43953934740882916, + "delta_ci_low": -0.6852207293666027, + "delta_ci_high": -0.16314779270633398, + "kruskal_h": 9.446343678317177, + "kruskal_p": 0.14998807706400527, + "spearman_rho_procedures": 0.26545918253608347, + "spearman_p": 0.007094210430320777, + "spearman_ci_low": 0.07384779717900485, + "spearman_ci_high": 0.4381688360883283, + "spearman_meta_i2": 0.0, + "n_cohorts": 3, + "n_nonmissing": 107, + "kruskal_q": 0.3170324957516792, + "spearman_q": 0.025933377977345594 + }, + { + "feature": "tool1_avg_speed", + "label": "Tool 1 speed", + "cliffs_delta_novice_expert": -0.3666026871401152, + "delta_ci_low": -0.6314779270633397, + "delta_ci_high": -0.08253358925143954, + "kruskal_h": 10.070936294222903, + "kruskal_p": 0.12169634914428706, + "spearman_rho_procedures": 0.20642377607265003, + "spearman_p": 0.0381460925868868, + "spearman_ci_low": 0.011442656610917675, + "spearman_ci_high": 0.3862823505464995, + "spearman_meta_i2": 0.0, + "n_cohorts": 3, + "n_nonmissing": 107, + "kruskal_q": 0.3170324957516792, + "spearman_q": 0.11062366850197172 + }, + { + "feature": "tool2_range_z", + "label": "Tool 2 depth range", + "cliffs_delta_novice_expert": -0.362763915547025, + "delta_ci_low": -0.6353166986564299, + "delta_ci_high": -0.05566218809980806, + "kruskal_h": 8.038309445980472, + "kruskal_p": 0.23531007622449016, + "spearman_rho_procedures": 0.23875854482084943, + "spearman_p": 0.015948005812675872, + "spearman_ci_low": 0.04543599067993563, + "spearman_ci_high": 0.4148432170028269, + "spearman_meta_i2": 0.0, + "n_cohorts": 3, + "n_nonmissing": 107, + "kruskal_q": 0.45493281403401425, + "spearman_q": 0.051388018729733365 + }, + { + "feature": "tool2_working_volume", + "label": "Tool 2 working volume", + "cliffs_delta_novice_expert": -0.2629558541266795, + "delta_ci_low": -0.5470249520153551, + "delta_ci_high": 0.02111324376199616, + "kruskal_h": 11.752049475787643, + "kruskal_p": 0.06773444765157813, + "spearman_rho_procedures": 0.19246671234055343, + "spearman_p": 0.05368284045721828, + "spearman_ci_low": -0.0030923734558430975, + "spearman_ci_high": 0.3738467701508323, + "spearman_meta_i2": 0.0, + "n_cohorts": 3, + "n_nonmissing": 107, + "kruskal_q": 0.3170324957516792, + "spearman_q": 0.12973353110494418 + } + ] +} \ No newline at end of file diff --git a/data/derived/classification_summary.json b/data/derived/classification_summary.json new file mode 100644 index 0000000..4113477 --- /dev/null +++ b/data/derived/classification_summary.json @@ -0,0 +1,197 @@ +{ + "available": true, + "features": [ + "total_time", + "bimanual_correlation", + "bimanual_lag_s", + "bimanual_concurrent_efficiency", + "bimanual_dist_speed_corr", + "simultaneous_motion_ratio", + "bimanual_symmetry", + "combined_idle_ratio", + "tool1_normalized_jerk", + "tool2_normalized_jerk", + "tool1_active_time_ratio", + "tool2_active_time_ratio", + "tool1_path_length", + "tool2_path_length", + "tool1_avg_speed", + "tool2_avg_speed", + "tool_distance_cv", + "tool_close_proximity_ratio", + "tool1_working_volume", + "tool2_working_volume", + "tool1_working_area_xy", + "tool2_working_area_xy", + "tool1_range_z", + "tool2_range_z", + "tool1_rotation_per_path", + "tool2_rotation_per_path", + "tool1_angular_velocity_cv", + "tool2_angular_velocity_cv", + "tool1_num_speed_peaks", + "tool2_num_speed_peaks" + ], + "time_only_features": [ + "total_time" + ], + "stratified_group_5fold": { + "time_only_logistic_regression": { + "mean_balanced_accuracy": 0.44841269841269843, + "sd": 0.16861577600976474, + "features": [ + "total_time" + ] + }, + "time_only_random_forest": { + "mean_balanced_accuracy": 0.39550264550264547, + "sd": 0.05432467376806009, + "features": [ + "total_time" + ] + }, + "all_features_logistic_regression": { + "mean_balanced_accuracy": 0.3603174603174603, + "sd": 0.06844059019686591, + "features": [ + "total_time", + "bimanual_correlation", + "bimanual_lag_s", + "bimanual_concurrent_efficiency", + "bimanual_dist_speed_corr", + "simultaneous_motion_ratio", + "bimanual_symmetry", + "combined_idle_ratio", + "tool1_normalized_jerk", + "tool2_normalized_jerk", + "tool1_active_time_ratio", + "tool2_active_time_ratio", + "tool1_path_length", + "tool2_path_length", + "tool1_avg_speed", + "tool2_avg_speed", + "tool_distance_cv", + "tool_close_proximity_ratio", + "tool1_working_volume", + "tool2_working_volume", + "tool1_working_area_xy", + "tool2_working_area_xy", + "tool1_range_z", + "tool2_range_z", + "tool1_rotation_per_path", + "tool2_rotation_per_path", + "tool1_angular_velocity_cv", + "tool2_angular_velocity_cv", + "tool1_num_speed_peaks", + "tool2_num_speed_peaks" + ] + }, + "all_features_random_forest": { + "mean_balanced_accuracy": 0.34325396825396826, + "sd": 0.10826701395943916, + "features": [ + "total_time", + "bimanual_correlation", + "bimanual_lag_s", + "bimanual_concurrent_efficiency", + "bimanual_dist_speed_corr", + "simultaneous_motion_ratio", + "bimanual_symmetry", + "combined_idle_ratio", + "tool1_normalized_jerk", + "tool2_normalized_jerk", + "tool1_active_time_ratio", + "tool2_active_time_ratio", + "tool1_path_length", + "tool2_path_length", + "tool1_avg_speed", + "tool2_avg_speed", + "tool_distance_cv", + "tool_close_proximity_ratio", + "tool1_working_volume", + "tool2_working_volume", + "tool1_working_area_xy", + "tool2_working_area_xy", + "tool1_range_z", + "tool2_range_z", + "tool1_rotation_per_path", + "tool2_rotation_per_path", + "tool1_angular_velocity_cv", + "tool2_angular_velocity_cv", + "tool1_num_speed_peaks", + "tool2_num_speed_peaks" + ] + } + }, + "leave_one_dataset_out": { + "time_only_logistic_regression": [ + { + "held_out_dataset": "6DOF2023", + "balanced_accuracy": 0.3333333333333333, + "n_test": 24 + }, + { + "held_out_dataset": "7DOF2024", + "balanced_accuracy": 0.5279866332497911, + "n_test": 55 + }, + { + "held_out_dataset": "BAPES2024", + "balanced_accuracy": 0.5413165266106442, + "n_test": 28 + } + ], + "time_only_random_forest": [ + { + "held_out_dataset": "6DOF2023", + "balanced_accuracy": 0.5370370370370371, + "n_test": 24 + }, + { + "held_out_dataset": "7DOF2024", + "balanced_accuracy": 0.4357560568086884, + "n_test": 55 + }, + { + "held_out_dataset": "BAPES2024", + "balanced_accuracy": 0.4747899159663866, + "n_test": 28 + } + ], + "all_features_logistic_regression": [ + { + "held_out_dataset": "6DOF2023", + "balanced_accuracy": 0.46296296296296297, + "n_test": 24 + }, + { + "held_out_dataset": "7DOF2024", + "balanced_accuracy": 0.32497911445279865, + "n_test": 55 + }, + { + "held_out_dataset": "BAPES2024", + "balanced_accuracy": 0.36274509803921573, + "n_test": 28 + } + ], + "all_features_random_forest": [ + { + "held_out_dataset": "6DOF2023", + "balanced_accuracy": 0.27777777777777773, + "n_test": 24 + }, + { + "held_out_dataset": "7DOF2024", + "balanced_accuracy": 0.3505430242272347, + "n_test": 55 + }, + { + "held_out_dataset": "BAPES2024", + "balanced_accuracy": 0.30392156862745096, + "n_test": 28 + } + ] + }, + "top_random_forest_features": [] +} \ No newline at end of file diff --git a/data/derived/cohort_adjusted_models.csv b/data/derived/cohort_adjusted_models.csv new file mode 100644 index 0000000..74d6516 --- /dev/null +++ b/data/derived/cohort_adjusted_models.csv @@ -0,0 +1,9 @@ +outcome,label,n,beta_log10_procedures,ci_low,ci_high,procedure_p,r2_procedure_only,r2_cohort_only,r2_full,cohort_delta_r2_after_procedure,procedure_q +total_time,Analysed duration,107,-0.42733493177412707,-0.7210666407293955,-0.1867809146394167,0.0001999600079984003,0.1329750713357727,0.018294307226203954,0.1488697330305392,0.0158946616947665,0.0015996800639872025 +bimanual_correlation,Bimanual correlation,107,0.29244008987169356,0.09152485904450995,0.5145501031277906,0.007398520295940812,0.05784230200288021,0.11128877039829821,0.17243905438252594,0.11459675237964573,0.02959408118376325 +tool1_avg_speed,Tool 1 speed,107,0.010798900046193131,-0.06338028617402766,0.0772080729996927,0.8016396720655868,0.009102444209678429,0.8753372340442597,0.8754206182322058,0.8663181740225274,0.8016396720655868 +tool2_avg_speed,Tool 2 speed,107,0.06356029050664681,-0.0023985628506424193,0.133968472852232,0.04799040191961608,0.00308542926382005,0.9249964885062787,0.9278851483746007,0.9247997191107806,0.07678464307138573 +combined_idle_ratio,Combined idle ratio,107,-0.07971270303197589,-0.20172280757710917,0.026271543507389142,0.17556488702259548,0.0006216419971403209,0.7603364482992011,0.7648798353216057,0.7642581933244654,0.2340865160301273 +tool1_range_z,Tool 1 depth range,107,-0.042217666962035216,-0.2449249058902264,0.15589781666696684,0.6914617076584683,0.011659559009017606,0.17780657175161974,0.17908099334106864,0.16742143433205103,0.790241951609678 +tool2_range_z,Tool 2 depth range,107,0.23582273751353064,0.05054699263230346,0.42192360858637795,0.02979404119176165,0.014836525031217462,0.13513978127316173,0.17490432594371252,0.16006780091249506,0.0595880823835233 +Coordination-control composite,Coordination and control score,107,0.28042398109730016,0.06831070397570142,0.5080588734377782,0.012397520495900819,0.05433644447282082,0.0015170135435146337,0.0577453146695327,0.0034088701967118773,0.03306005465573552 diff --git a/data/derived/cohort_specific_effects.csv b/data/derived/cohort_specific_effects.csv new file mode 100644 index 0000000..80c4ce8 --- /dev/null +++ b/data/derived/cohort_specific_effects.csv @@ -0,0 +1,22 @@ +feature,dataset,delta,lo,hi +Tool 2 speed,BAPES2024,0.47058823529411764,-0.23529411764705882,1.0 +Tool 2 speed,6DOF2023,0.2222222222222222,-0.37037037037037035,0.7777777777777778 +Tool 2 speed,7DOF2024,0.46365914786967416,0.10263157894736842,0.7694235588972431 +Tool 1 speed,BAPES2024,0.5,-0.2654411764705882,1.0 +Tool 1 speed,6DOF2023,0.07407407407407407,-0.5194444444444444,0.7416666666666658 +Tool 1 speed,7DOF2024,0.38345864661654133,0.01741854636591479,0.6942355889724311 +Bimanual correlation,BAPES2024,0.4117647058823529,-0.29411764705882354,1.0 +Bimanual correlation,6DOF2023,-0.14814814814814814,-0.7777777777777778,0.5555555555555556 +Bimanual correlation,7DOF2024,0.5288220551378446,0.18796992481203006,0.7997493734335838 +Tool 2 stop-start peaks,BAPES2024,0.3235294117647059,-0.4419117647058823,0.9117647058823529 +Tool 2 stop-start peaks,6DOF2023,0.2222222222222222,-0.4074074074074074,0.7777777777777778 +Tool 2 stop-start peaks,7DOF2024,0.45864661654135336,0.13283208020050125,0.7520050125313281 +Tool 1 stop-start peaks,BAPES2024,0.3235294117647059,-0.5007352941176471,0.9411764705882353 +Tool 1 stop-start peaks,6DOF2023,0.37037037037037035,-0.2222222222222222,0.889814814814814 +Tool 1 stop-start peaks,7DOF2024,0.44110275689223055,0.07268170426065163,0.7494987468671678 +Analysed duration,BAPES2024,0.38235294117647056,-0.47058823529411764,0.9705882352941176 +Analysed duration,6DOF2023,0.2962962962962963,-0.3333333333333333,0.8148148148148148 +Analysed duration,7DOF2024,0.47869674185463656,0.14786967418546365,0.7493734335839599 +Tool 2 depth range,BAPES2024,-0.20588235294117646,-0.8823529411764706,0.5294117647058824 +Tool 2 depth range,6DOF2023,-0.48148148148148145,-1.0,0.18518518518518517 +Tool 2 depth range,7DOF2024,-0.37343358395989973,-0.7045112781954888,-0.022556390977443608 diff --git a/data/derived/continuous_performance_validation.csv b/data/derived/continuous_performance_validation.csv new file mode 100644 index 0000000..ee50195 --- /dev/null +++ b/data/derived/continuous_performance_validation.csv @@ -0,0 +1,7 @@ +dataset,outcome,n,spearman_rho,ci_low,ci_high,p,q +BAPES2024,Task efficiency,28,0.4738571037503423,0.10120811268230709,0.7593587564884463,0.010856998437097476,0.021713996874194952 +6DOF2023,Task efficiency,24,0.3756521739130434,-0.04782799473453269,0.7072115384615385,0.07045361689375705,0.0951498225567684 +7DOF2024,Task efficiency,55,0.7467349736633234,0.6046869747002657,0.8353758657774063,5.919503255200333e-11,3.5517019531202e-10 +BAPES2024,Analysed duration (s),28,-0.33721089682147404,-0.666205289903589,0.046433734184466384,0.079291518797307,0.0951498225567684 +6DOF2023,Analysed duration (s),24,-0.31478260869565217,-0.6462501377781588,0.09987353003433055,0.13408706839947865,0.13408706839947865 +7DOF2024,Analysed duration (s),55,-0.7188339918553194,-0.813942675630546,-0.5675621268336416,6.386065820756053e-10,1.915819746226816e-09 diff --git a/data/derived/coordination_control_time_correlation.csv b/data/derived/coordination_control_time_correlation.csv new file mode 100644 index 0000000..91b5d16 --- /dev/null +++ b/data/derived/coordination_control_time_correlation.csv @@ -0,0 +1,4 @@ +dataset,n,spearman_rho,ci_low,ci_high,p +BAPES2024,28,-0.38368910782703886,-0.7465771840836282,0.05147759017516738,0.04383971743892305 +6DOF2023,24,-0.041739130434782605,-0.41095636278631287,0.3395114421575783,0.8464521133997217 +7DOF2024,55,-0.47616299003038437,-0.6566871273281804,-0.24345247671185027,0.00023823151622729487 diff --git a/data/derived/cycle_duration_ci.csv b/data/derived/cycle_duration_ci.csv new file mode 100644 index 0000000..79b46e3 --- /dev/null +++ b/data/derived/cycle_duration_ci.csv @@ -0,0 +1,10 @@ +dataset,skill_category,n_cycles,n_trials,mean_duration_s,ci_low,ci_high +6DOF2023,expert,23,2,10.618444055944057,8.586538461538462,12.650349650349652 +6DOF2023,intermediate,44,4,12.394070512820512,10.692307692307693,13.353685897435897 +6DOF2023,novice,48,4,12.45673076923077,10.665064102564102,14.248397435897436 +7DOF2024,expert,58,5,12.94025641025641,11.612564102564104,15.010256410256408 +7DOF2024,intermediate,36,3,15.34188034188034,9.512820512820513,19.01923076923077 +7DOF2024,novice,84,7,13.602564102564102,9.774725274725274,18.555860805860807 +BAPES2024,expert,60,6,15.152777777777779,12.998824786324786,17.550213675213676 +BAPES2024,intermediate,24,2,11.067307692307693,10.833333333333334,11.301282051282051 +BAPES2024,novice,6,1,15.294871794871796,15.294871794871796,15.294871794871796 diff --git a/data/derived/feature_dictionary.csv b/data/derived/feature_dictionary.csv new file mode 100644 index 0000000..2a394a6 --- /dev/null +++ b/data/derived/feature_dictionary.csv @@ -0,0 +1,31 @@ +feature,label,favourable_direction,interpretation,domain,available_cohorts,nonmissing_trials,missing_trials +total_time,Analysed duration,downarrow,shorter analysed task interval,Task efficiency,Paediatric; Urology 1 (2023); Urology 2 (2024),107,8 +bimanual_correlation,Bimanual correlation,uparrow,more synchronous tool motion,Bimanual coordination,Paediatric; Urology 1 (2023); Urology 2 (2024),107,8 +bimanual_lag_s,Bimanual lag,downarrow,less timing delay between tools,,Paediatric; Urology 1 (2023); Urology 2 (2024),107,8 +bimanual_concurrent_efficiency,Concurrent efficiency,uparrow,more efficient simultaneous tool use,Bimanual coordination,Paediatric; Urology 1 (2023); Urology 2 (2024),107,8 +bimanual_dist_speed_corr,Distance-speed coupling,uparrow,more coupled spacing and speed control,,Paediatric; Urology 1 (2023); Urology 2 (2024),107,8 +simultaneous_motion_ratio,Simultaneous motion,uparrow,larger fraction of active frames with both tools moving,Bimanual coordination,Paediatric; Urology 1 (2023); Urology 2 (2024),107,8 +bimanual_symmetry,Bimanual symmetry,uparrow,more balanced left/right tool use,,Paediatric; Urology 1 (2023); Urology 2 (2024),107,8 +combined_idle_ratio,Combined idle ratio,downarrow,less inactive task time,Task efficiency,Paediatric; Urology 1 (2023); Urology 2 (2024),107,8 +tool1_normalized_jerk,Tool 1 normalised jerk,downarrow,smoother Tool 1 motion,Instrument motion control,Paediatric; Urology 1 (2023); Urology 2 (2024),107,8 +tool2_normalized_jerk,Tool 2 normalised jerk,downarrow,smoother Tool 2 motion,Instrument motion control,Paediatric; Urology 1 (2023); Urology 2 (2024),107,8 +tool1_active_time_ratio,Tool 1 active time,uparrow,more active Tool 1 use,Task efficiency,Paediatric; Urology 1 (2023); Urology 2 (2024),107,8 +tool2_active_time_ratio,Tool 2 active time,uparrow,more active Tool 2 use,Task efficiency,Paediatric; Urology 1 (2023); Urology 2 (2024),107,8 +tool1_path_length,Tool 1 path length,downarrow,shorter Tool 1 travel distance,,Paediatric; Urology 1 (2023); Urology 2 (2024),107,8 +tool2_path_length,Tool 2 path length,downarrow,shorter Tool 2 travel distance,,Paediatric; Urology 1 (2023); Urology 2 (2024),107,8 +tool1_avg_speed,Tool 1 speed,uparrow,faster Tool 1 movement,,Paediatric; Urology 1 (2023); Urology 2 (2024),107,8 +tool2_avg_speed,Tool 2 speed,uparrow,faster Tool 2 movement,,Paediatric; Urology 1 (2023); Urology 2 (2024),107,8 +tool_distance_cv,Tool distance variability,downarrow,steadier distance between tools,Bimanual coordination,Paediatric; Urology 1 (2023); Urology 2 (2024),107,8 +tool_close_proximity_ratio,Close proximity,downarrow,less time with tools very close together,,Paediatric; Urology 1 (2023); Urology 2 (2024),107,8 +tool1_working_volume,Tool 1 working volume,downarrow,more compact Tool 1 three-dimensional workspace,Workspace excursion,Paediatric; Urology 1 (2023); Urology 2 (2024),107,8 +tool2_working_volume,Tool 2 working volume,downarrow,more compact Tool 2 three-dimensional workspace,Workspace excursion,Paediatric; Urology 1 (2023); Urology 2 (2024),107,8 +tool1_working_area_xy,Tool 1 working area,downarrow,more compact Tool 1 image-plane workspace,Workspace excursion,Paediatric; Urology 1 (2023); Urology 2 (2024),107,8 +tool2_working_area_xy,Tool 2 working area,downarrow,more compact Tool 2 image-plane workspace,Workspace excursion,Paediatric; Urology 1 (2023); Urology 2 (2024),107,8 +tool1_range_z,Tool 1 depth range,downarrow,less Tool 1 camera-axis excursion,Workspace excursion,Paediatric; Urology 1 (2023); Urology 2 (2024),107,8 +tool2_range_z,Tool 2 depth range,downarrow,less Tool 2 camera-axis excursion,Workspace excursion,Paediatric; Urology 1 (2023); Urology 2 (2024),107,8 +tool1_rotation_per_path,Tool 1 rotation per path,downarrow,less Tool 1 rotation per millimetre travelled,Instrument motion control,Paediatric; Urology 1 (2023); Urology 2 (2024),107,8 +tool2_rotation_per_path,Tool 2 rotation per path,downarrow,less Tool 2 rotation per millimetre travelled,Instrument motion control,Paediatric; Urology 1 (2023); Urology 2 (2024),107,8 +tool1_angular_velocity_cv,Tool 1 angular velocity variability,downarrow,steadier Tool 1 rotational speed,Instrument motion control,Paediatric; Urology 1 (2023); Urology 2 (2024),107,8 +tool2_angular_velocity_cv,Tool 2 angular velocity variability,downarrow,steadier Tool 2 rotational speed,Instrument motion control,Paediatric; Urology 1 (2023); Urology 2 (2024),107,8 +tool1_num_speed_peaks,Tool 1 speed peaks,downarrow,fewer Tool 1 stop-start speed peaks,Task efficiency,Paediatric; Urology 1 (2023); Urology 2 (2024),107,8 +tool2_num_speed_peaks,Tool 2 speed peaks,downarrow,fewer Tool 2 stop-start speed peaks,Task efficiency,Paediatric; Urology 1 (2023); Urology 2 (2024),107,8 diff --git a/data/derived/kmeans_cluster_summary.csv b/data/derived/kmeans_cluster_summary.csv new file mode 100644 index 0000000..be348ca --- /dev/null +++ b/data/derived/kmeans_cluster_summary.csv @@ -0,0 +1,4 @@ +metric_cluster,cluster_label,n,median_total_procedures,procedure_iqr_low,procedure_iqr_high,mean_coordination_control_composite +metric_low,Lower motion-score band,16,40.0,13.0,85.0,2.1605203823953825 +metric_middle,Middle motion-score band,41,20.0,5.0,100.0,2.8661563368880443 +metric_high,Upper motion-score band,50,80.0,16.25,300.0,3.4983852813852807 diff --git a/data/derived/kmeans_thresholds.json b/data/derived/kmeans_thresholds.json new file mode 100644 index 0000000..50db8d2 --- /dev/null +++ b/data/derived/kmeans_thresholds.json @@ -0,0 +1,172 @@ +{ + "available": true, + "features": [ + "Coordination-control composite" + ], + "n_recordings_assigned": 107, + "n_independent_identifiers_fitted": 107, + "metric_band_silhouette": 0.5340644126720104, + "metric_band_davies_bouldin": 0.5574945498396043, + "metric_band_calinski_harabasz": 205.40036661359844, + "metric_band_seed_ari_mean": 0.6881185433201465, + "metric_band_seed_ari_min": 0.2457480281764961, + "metric_band_bootstrap_assignment_ari_median": 0.7557894702759759, + "metric_band_bootstrap_assignment_ari_ci": [ + 0.25019440016739297, + 1.0 + ], + "metric_band_centres": [ + 2.1605203823953825, + 2.8661563368880443, + 3.498385281385281 + ], + "metric_band_thresholds": [ + 2.5133383596417134, + 3.1822708091366625 + ], + "metric_band_threshold_bootstrap_ci": [ + { + "threshold": 1, + "ci_low": 2.2926625593349086, + "median": 2.549881584464918, + "ci_high": 2.9269077055625816 + }, + { + "threshold": 2, + "ci_low": 3.0849795438675742, + "median": 3.198773797443426, + "ci_high": 3.5601637509174164 + } + ], + "sensitivity": { + "complete_case": { + "n_trials": 107, + "thresholds": [ + 2.5133383596417134, + 3.1822708091366625 + ], + "assignment_adjusted_rand_index": 1.0 + }, + "domain_weight_perturbation": { + "n_repetitions": 500, + "adjusted_rand_index_median": 0.9591035280410379, + "adjusted_rand_index_ci_low": 0.7199041122076191, + "adjusted_rand_index_ci_high": 1.0 + }, + "leave_one_domain_out": [ + { + "omitted_domain": "Bimanual coordination", + "adjusted_rand_index": 0.13060915248255114, + "thresholds": [ + 2.7302769243898277, + 3.630029209142112 + ] + }, + { + "omitted_domain": "Instrument motion control", + "adjusted_rand_index": 0.35444949343374316, + "thresholds": [ + 2.334584187210237, + 3.2698409477427335 + ] + } + ], + "leave_one_cohort_out": [ + { + "held_out_dataset": "BAPES2024", + "thresholds": [ + 2.3904748376623375, + 3.1671951659451656 + ], + "held_out_adjusted_rand_index": 0.865164531375405 + }, + { + "held_out_dataset": "6DOF2023", + "thresholds": [ + 2.5767321004648194, + 3.1817356808443398 + ], + "held_out_adjusted_rand_index": 1.0 + }, + { + "held_out_dataset": "7DOF2024", + "thresholds": [ + 2.535273368606702, + 3.215837742504409 + ], + "held_out_adjusted_rand_index": 0.8914523798531924 + } + ] + }, + "k_validation": [ + { + "k": 2, + "silhouette": 0.5662174552602269, + "davies_bouldin": 0.5878701558520018, + "calinski_harabasz": 183.48140711375248, + "inertia": 10.989325724233584 + }, + { + "k": 3, + "silhouette": 0.5340644126720104, + "davies_bouldin": 0.5574945498396043, + "calinski_harabasz": 205.40036661359844, + "inertia": 6.099493281079126 + }, + { + "k": 4, + "silhouette": 0.5526414765421528, + "davies_bouldin": 0.5339018410670197, + "calinski_harabasz": 274.1045156576356, + "inertia": 3.3608403214706066 + }, + { + "k": 5, + "silhouette": 0.5435087697321636, + "davies_bouldin": 0.5047211727805098, + "calinski_harabasz": 309.8046641567337, + "inertia": 2.296149497107527 + }, + { + "k": 6, + "silhouette": 0.5354954704018147, + "davies_bouldin": 0.4844300101979635, + "calinski_harabasz": 340.0575213267391, + "inertia": 1.692925659373024 + } + ], + "pca_variance": [ + 0.6118434437413051, + 0.38815655625869483 + ], + "adjusted_rand_vs_original_skill": 0.018846337788122808, + "cluster_summary": [ + { + "metric_cluster": "metric_low", + "cluster_label": "Lower motion-score band", + "n": 16, + "median_total_procedures": 40.0, + "procedure_iqr_low": 13.0, + "procedure_iqr_high": 85.0, + "mean_coordination_control_composite": 2.1605203823953825 + }, + { + "metric_cluster": "metric_middle", + "cluster_label": "Middle motion-score band", + "n": 41, + "median_total_procedures": 20.0, + "procedure_iqr_low": 5.0, + "procedure_iqr_high": 100.0, + "mean_coordination_control_composite": 2.8661563368880443 + }, + { + "metric_cluster": "metric_high", + "cluster_label": "Upper motion-score band", + "n": 50, + "median_total_procedures": 80.0, + "procedure_iqr_low": 16.25, + "procedure_iqr_high": 300.0, + "mean_coordination_control_composite": 3.4983852813852807 + } + ] +} \ No newline at end of file diff --git a/data/derived/kmeans_validation.csv b/data/derived/kmeans_validation.csv new file mode 100644 index 0000000..e7c2d33 --- /dev/null +++ b/data/derived/kmeans_validation.csv @@ -0,0 +1,6 @@ +k,silhouette,davies_bouldin,calinski_harabasz,inertia +2,0.5662174552602269,0.5878701558520018,183.48140711375248,10.989325724233584 +3,0.5340644126720104,0.5574945498396043,205.40036661359844,6.099493281079126 +4,0.5526414765421528,0.5339018410670197,274.1045156576356,3.3608403214706066 +5,0.5435087697321636,0.5047211727805098,309.8046641567337,2.296149497107527 +6,0.5354954704018147,0.4844300101979635,340.0575213267391,1.692925659373024 diff --git a/data/derived/metric_band_classification_comparison.csv b/data/derived/metric_band_classification_comparison.csv new file mode 100644 index 0000000..d05c26b --- /dev/null +++ b/data/derived/metric_band_classification_comparison.csv @@ -0,0 +1,9 @@ +target,target_label,model,model_label,stratified_cv_balanced_accuracy,stratified_cv_sd,leave_one_dataset_out_balanced_accuracy,leave_one_dataset_out_min,leave_one_dataset_out_max,n_splits +motion_defined,Motion-score bands,logistic_regression,Logistic regression,0.9481481481481481,0.0863844725162267,0.9907407407407408,0.9722222222222222,1.0,5 +motion_defined,Motion-score bands,random_forest,Random forest,0.9555555555555555,0.054433105395181765,0.9861111111111112,0.9583333333333334,1.0,5 +procedure_fixed,Procedure thresholds,logistic_regression,Logistic regression,0.3843915343915344,0.11729933197995121,0.3746897199425579,0.26820728291316526,0.4484544695071011,5 +procedure_fixed,Procedure thresholds,random_forest,Random forest,0.362962962962963,0.06755414162236494,0.35505266434059,0.33266499582289055,0.3991596638655462,5 +procedure_tertiles,Procedure tertiles,logistic_regression,Logistic regression,0.33333333333333337,0.12192940599113683,0.4021271613376876,0.3333333333333333,0.4598484848484849,5 +procedure_tertiles,Procedure tertiles,random_forest,Random forest,0.2702380952380952,0.028071014577503817,0.3696716538821802,0.2916666666666667,0.454469507101086,5 +procedure_kmeans,Procedure K-means,logistic_regression,Logistic regression,0.5048340548340549,0.16704300258582674,0.46421957671957675,0.28571428571428575,0.5847222222222223,5 +procedure_kmeans,Procedure K-means,random_forest,Random forest,0.26275613275613274,0.04702399204348823,0.28187830687830684,0.1984126984126984,0.3277777777777778,5 diff --git a/data/derived/metric_band_classification_summary.json b/data/derived/metric_band_classification_summary.json new file mode 100644 index 0000000..927bccb --- /dev/null +++ b/data/derived/metric_band_classification_summary.json @@ -0,0 +1,392 @@ +{ + "available": true, + "features": [ + "Bimanual coordination", + "Instrument motion control", + "Coordination-control composite" + ], + "targets": { + "motion_defined": { + "available": true, + "label": "Motion-score bands", + "class_counts": { + "metric_high": 50, + "metric_middle": 41, + "metric_low": 16 + }, + "models": { + "logistic_regression": { + "stratified_cv": { + "mean_balanced_accuracy": 0.9481481481481481, + "sd": 0.0863844725162267, + "n_splits": 5 + }, + "leave_one_dataset_out": [ + { + "held_out_dataset": "6DOF2023", + "balanced_accuracy": 1.0, + "n_test": 24 + }, + { + "held_out_dataset": "7DOF2024", + "balanced_accuracy": 0.9722222222222222, + "n_test": 55 + }, + { + "held_out_dataset": "BAPES2024", + "balanced_accuracy": 1.0, + "n_test": 28 + } + ] + }, + "random_forest": { + "stratified_cv": { + "mean_balanced_accuracy": 0.9555555555555555, + "sd": 0.054433105395181765, + "n_splits": 5 + }, + "leave_one_dataset_out": [ + { + "held_out_dataset": "6DOF2023", + "balanced_accuracy": 1.0, + "n_test": 24 + }, + { + "held_out_dataset": "7DOF2024", + "balanced_accuracy": 0.9583333333333334, + "n_test": 55 + }, + { + "held_out_dataset": "BAPES2024", + "balanced_accuracy": 1.0, + "n_test": 28 + } + ] + } + } + }, + "procedure_fixed": { + "available": true, + "label": "Procedure thresholds", + "class_counts": { + "procedure_expert": 42, + "procedure_novice": 34, + "procedure_intermediate": 31 + }, + "models": { + "logistic_regression": { + "stratified_cv": { + "mean_balanced_accuracy": 0.3843915343915344, + "sd": 0.11729933197995121, + "n_splits": 5 + }, + "leave_one_dataset_out": [ + { + "held_out_dataset": "6DOF2023", + "balanced_accuracy": 0.4074074074074074, + "n_test": 24 + }, + { + "held_out_dataset": "7DOF2024", + "balanced_accuracy": 0.4484544695071011, + "n_test": 55 + }, + { + "held_out_dataset": "BAPES2024", + "balanced_accuracy": 0.26820728291316526, + "n_test": 28 + } + ] + }, + "random_forest": { + "stratified_cv": { + "mean_balanced_accuracy": 0.362962962962963, + "sd": 0.06755414162236494, + "n_splits": 5 + }, + "leave_one_dataset_out": [ + { + "held_out_dataset": "6DOF2023", + "balanced_accuracy": 0.3333333333333333, + "n_test": 24 + }, + { + "held_out_dataset": "7DOF2024", + "balanced_accuracy": 0.33266499582289055, + "n_test": 55 + }, + { + "held_out_dataset": "BAPES2024", + "balanced_accuracy": 0.3991596638655462, + "n_test": 28 + } + ] + } + } + }, + "procedure_tertiles": { + "available": true, + "label": "Procedure tertiles", + "class_counts": { + "procedure_tertile_low": 36, + "procedure_tertile_high": 36, + "procedure_tertile_mid": 35 + }, + "models": { + "logistic_regression": { + "stratified_cv": { + "mean_balanced_accuracy": 0.33333333333333337, + "sd": 0.12192940599113683, + "n_splits": 5 + }, + "leave_one_dataset_out": [ + { + "held_out_dataset": "6DOF2023", + "balanced_accuracy": 0.4598484848484849, + "n_test": 24 + }, + { + "held_out_dataset": "7DOF2024", + "balanced_accuracy": 0.41319966583124473, + "n_test": 55 + }, + { + "held_out_dataset": "BAPES2024", + "balanced_accuracy": 0.3333333333333333, + "n_test": 28 + } + ] + }, + "random_forest": { + "stratified_cv": { + "mean_balanced_accuracy": 0.2702380952380952, + "sd": 0.028071014577503817, + "n_splits": 5 + }, + "leave_one_dataset_out": [ + { + "held_out_dataset": "6DOF2023", + "balanced_accuracy": 0.36287878787878786, + "n_test": 24 + }, + { + "held_out_dataset": "7DOF2024", + "balanced_accuracy": 0.454469507101086, + "n_test": 55 + }, + { + "held_out_dataset": "BAPES2024", + "balanced_accuracy": 0.2916666666666667, + "n_test": 28 + } + ] + } + } + }, + "procedure_kmeans": { + "available": true, + "label": "Procedure K-means", + "class_counts": { + "procedure_kmeans_mid": 53, + "procedure_kmeans_high": 32, + "procedure_kmeans_low": 22 + }, + "models": { + "logistic_regression": { + "stratified_cv": { + "mean_balanced_accuracy": 0.5048340548340549, + "sd": 0.16704300258582674, + "n_splits": 5 + }, + "leave_one_dataset_out": [ + { + "held_out_dataset": "6DOF2023", + "balanced_accuracy": 0.5222222222222223, + "n_test": 24 + }, + { + "held_out_dataset": "7DOF2024", + "balanced_accuracy": 0.5847222222222223, + "n_test": 55 + }, + { + "held_out_dataset": "BAPES2024", + "balanced_accuracy": 0.28571428571428575, + "n_test": 28 + } + ] + }, + "random_forest": { + "stratified_cv": { + "mean_balanced_accuracy": 0.26275613275613274, + "sd": 0.04702399204348823, + "n_splits": 5 + }, + "leave_one_dataset_out": [ + { + "held_out_dataset": "6DOF2023", + "balanced_accuracy": 0.3277777777777778, + "n_test": 24 + }, + { + "held_out_dataset": "7DOF2024", + "balanced_accuracy": 0.3194444444444444, + "n_test": 55 + }, + { + "held_out_dataset": "BAPES2024", + "balanced_accuracy": 0.1984126984126984, + "n_test": 28 + } + ] + } + } + } + }, + "comparison_rows": [ + { + "target": "motion_defined", + "target_label": "Motion-score bands", + "model": "logistic_regression", + "model_label": "Logistic regression", + "stratified_cv_balanced_accuracy": 0.9481481481481481, + "stratified_cv_sd": 0.0863844725162267, + "leave_one_dataset_out_balanced_accuracy": 0.9907407407407408, + "leave_one_dataset_out_min": 0.9722222222222222, + "leave_one_dataset_out_max": 1.0, + "n_splits": 5 + }, + { + "target": "motion_defined", + "target_label": "Motion-score bands", + "model": "random_forest", + "model_label": "Random forest", + "stratified_cv_balanced_accuracy": 0.9555555555555555, + "stratified_cv_sd": 0.054433105395181765, + "leave_one_dataset_out_balanced_accuracy": 0.9861111111111112, + "leave_one_dataset_out_min": 0.9583333333333334, + "leave_one_dataset_out_max": 1.0, + "n_splits": 5 + }, + { + "target": "procedure_fixed", + "target_label": "Procedure thresholds", + "model": "logistic_regression", + "model_label": "Logistic regression", + "stratified_cv_balanced_accuracy": 0.3843915343915344, + "stratified_cv_sd": 0.11729933197995121, + "leave_one_dataset_out_balanced_accuracy": 0.3746897199425579, + "leave_one_dataset_out_min": 0.26820728291316526, + "leave_one_dataset_out_max": 0.4484544695071011, + "n_splits": 5 + }, + { + "target": "procedure_fixed", + "target_label": "Procedure thresholds", + "model": "random_forest", + "model_label": "Random forest", + "stratified_cv_balanced_accuracy": 0.362962962962963, + "stratified_cv_sd": 0.06755414162236494, + "leave_one_dataset_out_balanced_accuracy": 0.35505266434059, + "leave_one_dataset_out_min": 0.33266499582289055, + "leave_one_dataset_out_max": 0.3991596638655462, + "n_splits": 5 + }, + { + "target": "procedure_tertiles", + "target_label": "Procedure tertiles", + "model": "logistic_regression", + "model_label": "Logistic regression", + "stratified_cv_balanced_accuracy": 0.33333333333333337, + "stratified_cv_sd": 0.12192940599113683, + "leave_one_dataset_out_balanced_accuracy": 0.4021271613376876, + "leave_one_dataset_out_min": 0.3333333333333333, + "leave_one_dataset_out_max": 0.4598484848484849, + "n_splits": 5 + }, + { + "target": "procedure_tertiles", + "target_label": "Procedure tertiles", + "model": "random_forest", + "model_label": "Random forest", + "stratified_cv_balanced_accuracy": 0.2702380952380952, + "stratified_cv_sd": 0.028071014577503817, + "leave_one_dataset_out_balanced_accuracy": 0.3696716538821802, + "leave_one_dataset_out_min": 0.2916666666666667, + "leave_one_dataset_out_max": 0.454469507101086, + "n_splits": 5 + }, + { + "target": "procedure_kmeans", + "target_label": "Procedure K-means", + "model": "logistic_regression", + "model_label": "Logistic regression", + "stratified_cv_balanced_accuracy": 0.5048340548340549, + "stratified_cv_sd": 0.16704300258582674, + "leave_one_dataset_out_balanced_accuracy": 0.46421957671957675, + "leave_one_dataset_out_min": 0.28571428571428575, + "leave_one_dataset_out_max": 0.5847222222222223, + "n_splits": 5 + }, + { + "target": "procedure_kmeans", + "target_label": "Procedure K-means", + "model": "random_forest", + "model_label": "Random forest", + "stratified_cv_balanced_accuracy": 0.26275613275613274, + "stratified_cv_sd": 0.04702399204348823, + "leave_one_dataset_out_balanced_accuracy": 0.28187830687830684, + "leave_one_dataset_out_min": 0.1984126984126984, + "leave_one_dataset_out_max": 0.3277777777777778, + "n_splits": 5 + } + ], + "stratified_cv": { + "metric_band_logistic_regression": { + "mean_balanced_accuracy": 0.9481481481481481, + "sd": 0.0863844725162267, + "n_splits": 5 + }, + "metric_band_random_forest": { + "mean_balanced_accuracy": 0.9555555555555555, + "sd": 0.054433105395181765, + "n_splits": 5 + } + }, + "leave_one_dataset_out": { + "metric_band_logistic_regression": [ + { + "held_out_dataset": "6DOF2023", + "balanced_accuracy": 1.0, + "n_test": 24 + }, + { + "held_out_dataset": "7DOF2024", + "balanced_accuracy": 0.9722222222222222, + "n_test": 55 + }, + { + "held_out_dataset": "BAPES2024", + "balanced_accuracy": 1.0, + "n_test": 28 + } + ], + "metric_band_random_forest": [ + { + "held_out_dataset": "6DOF2023", + "balanced_accuracy": 1.0, + "n_test": 24 + }, + { + "held_out_dataset": "7DOF2024", + "balanced_accuracy": 0.9583333333333334, + "n_test": 55 + }, + { + "held_out_dataset": "BAPES2024", + "balanced_accuracy": 1.0, + "n_test": 28 + } + ] + } +} \ No newline at end of file diff --git a/data/derived/phase_fraction_ci.csv b/data/derived/phase_fraction_ci.csv new file mode 100644 index 0000000..7f4c2ed --- /dev/null +++ b/data/derived/phase_fraction_ci.csv @@ -0,0 +1,28 @@ +skill_category,phase,n_trials,mean_fraction,ci_low,ci_high +expert,dropped,13,0.0032817445805510023,0.0,0.007169367647976989 +expert,idle,13,0.03470974466434072,0.021452542600950437,0.049264162164491775 +expert,active1,13,0.0,0.0,0.0 +expert,active2,13,0.0,0.0,0.0 +expert,transfer,13,0.286860806331103,0.21271746031310187,0.3500414969360202 +expert,place,13,0.3315762725924012,0.29056076161408195,0.3756211675337971 +expert,grasp,13,0.13664443229267087,0.11697067552680984,0.1580611768405519 +expert,reach,13,0.17887888507026897,0.16172897787677873,0.19625276708345085 +expert,nudge,13,0.02804811446866428,0.018417527012453392,0.0385352844311838 +intermediate,dropped,9,0.0020357287263343036,0.0005140066820868671,0.003772565709507457 +intermediate,idle,9,0.04690651359583262,0.035785545137860995,0.057908686010548965 +intermediate,active1,9,0.0,0.0,0.0 +intermediate,active2,9,0.0,0.0,0.0 +intermediate,transfer,9,0.2916955450368333,0.21176998348410092,0.3420442979604823 +intermediate,place,9,0.3228310284306484,0.28248570711722043,0.36699889724984497 +intermediate,grasp,9,0.14027477448808673,0.11291882445431675,0.17480857196349225 +intermediate,reach,9,0.16174108770157322,0.13532074562555016,0.185591989939403 +intermediate,nudge,9,0.034515322020691346,0.02428755041242684,0.04502943912146738 +novice,dropped,12,0.0017370348657187458,0.0002821670428893905,0.0032989161642641165 +novice,idle,12,0.032488503311202115,0.02471049050225703,0.041689781566996674 +novice,active1,12,0.0,0.0,0.0 +novice,active2,12,0.0,0.0,0.0 +novice,transfer,12,0.32562627065469996,0.3089926171556426,0.342253663901653 +novice,place,12,0.3325447097930767,0.3141083832884708,0.3536370969958957 +novice,grasp,12,0.13125037260593608,0.11878101120890072,0.1435458540047876 +novice,reach,12,0.1554831274418424,0.12965532306806185,0.17818568165798385 +novice,nudge,12,0.020869981327523962,0.0134675390962613,0.028250321998084943 diff --git a/data/derived/phase_kinematics_summary.csv b/data/derived/phase_kinematics_summary.csv new file mode 100644 index 0000000..2d9308d --- /dev/null +++ b/data/derived/phase_kinematics_summary.csv @@ -0,0 +1,8 @@ +phase,n_trials,duration_s,tool1_speed_mm_s,tool2_speed_mm_s,combined_path_rate_mm_s +reach,34,874.1923076923075,43.64497380933804,38.63477297139324,82.27974678073127 +grasp,34,693.6153846153845,42.33294602684745,39.058830625868666,81.39177665271612 +transfer,31,1617.4999999999998,38.377615719404794,32.292785148904876,70.67040086830967 +place,34,1728.5769230769229,42.69559837613292,37.25730176261506,79.95290013874798 +nudge,32,139.65384615384613,36.003999240375705,30.559871943533597,66.5638711839093 +idle,34,200.88461538461536,57.38604534269178,50.430112252669474,107.81615759536125 +dropped,11,14.230769230769234,54.876743361268126,37.98917245393057,92.86591581519869 diff --git a/data/derived/phase_segment_summary.csv b/data/derived/phase_segment_summary.csv new file mode 100644 index 0000000..bb448fc --- /dev/null +++ b/data/derived/phase_segment_summary.csv @@ -0,0 +1,8 @@ +phase,n_segments,mean_duration_s,median_duration_s +reach,620,1.5572580645161305,1.3846153846153868 +grasp,485,1.5296590007930229,1.4615384615384386 +transfer,394,4.41419367434596,4.153846153846157 +place,438,4.157709870038639,3.69230769230769 +nudge,179,0.8594757198109167,0.6923076923076934 +idle,309,0.711102813044562,0.461538461538467 +dropped,22,0.6958041958041984,0.6538461538461587 diff --git a/data/derived/position_smoothing_sensitivity.csv b/data/derived/position_smoothing_sensitivity.csv new file mode 100644 index 0000000..7cbc84e --- /dev/null +++ b/data/derived/position_smoothing_sensitivity.csv @@ -0,0 +1,4 @@ +position_smoothing,window_s,n,score_spearman_rho,median_absolute_score_change,fixed_cut_assignment_ari,refit_assignment_ari +No position smoothing,0.0,107,0.9833552945234894,0.054545454545454675,0.7461892039309828,0.3175844249017427 +0.4 s primary window,0.4,107,1.0,0.0,1.0,1.0 +0.8 s window,0.8,107,0.9836590342619369,0.054545454545454675,0.85387290419947,0.8053423741454286 diff --git a/data/derived/processing_sensitivity.csv b/data/derived/processing_sensitivity.csv new file mode 100644 index 0000000..f4c0dc1 --- /dev/null +++ b/data/derived/processing_sensitivity.csv @@ -0,0 +1,5 @@ +processing,position_window_s,orientation_window_s,n,score_spearman_rho,median_absolute_score_change,fixed_cut_assignment_ari,refit_assignment_ari +No position smoothing,0.0,0.0,107,0.9833552945234894,0.054545454545454675,0.7461892039309828,0.3175844249017427 +Primary processing,0.4,0.0,107,1.0,0.0,1.0,1.0 +0.8 s position window,0.8,0.0,107,0.9836590342619369,0.054545454545454675,0.85387290419947,0.8053423741454286 +0.4 s orientation sensitivity,0.4,0.4,107,0.9880088170937018,0.042424242424242475,0.8286069303742338,0.42755379100614965 diff --git a/data/derived/trimming_rule_validation.csv b/data/derived/trimming_rule_validation.csv new file mode 100644 index 0000000..e273bb0 --- /dev/null +++ b/data/derived/trimming_rule_validation.csv @@ -0,0 +1,5 @@ +cohort,n,median_abs_start_error_s,median_abs_end_error_s,median_abs_duration_error_s,median_interval_iou +All cohorts,34,2.576923076923077,1.5,4.653846153846146,0.9678538255991654 +Paediatric,9,3.6923076923076925,1.6153846153846154,5.307692307692321,0.9521149241819633 +Urology 1 (2023),10,1.3653846153846154,1.3653846153846154,2.5576923076923137,0.9803009547741184 +Urology 2 (2024),15,4.230769230769231,1.5384615384615385,6.307692307692292,0.955458989679522 diff --git a/docs/ANALYSIS.md b/docs/ANALYSIS.md new file mode 100644 index 0000000..cb500c2 --- /dev/null +++ b/docs/ANALYSIS.md @@ -0,0 +1,285 @@ +# Quantitative peg-transfer analysis + +## Research question + +The analysis asks which measurable aspects of instrument motion are associated +with laparoscopic experience during peg transfer, and whether they can support +transparent quantitative feedback. + +The intended clinical use is not to replace an educator with an unvalidated +score. It is to identify reproducible measurements that could complement +expert observation, make feedback more specific, and support later automated +assessment after external validation. + +## Analysis populations + +Four populations are kept separate: + +| Population | Recordings | Study identifiers | Cycles | Use | +|:--|--:|--:|--:|:--| +| Complete inventory | 115 | 111 | Not applicable | Dataset description | +| Primary motion set | 107 | 107 | Not applicable | Motion inference and modelling | +| Phase inventory | 38 | 37 | 425 | All available dense phase labels | +| Primary phase set | 34 | 34 | 383 | Recording-level phase comparisons | + +The study identifiers cannot prove that a participant who attended two +different events received the same identifier. The paper therefore refers to +recordings rather than claiming 107 confirmed unique people. + +## Cohort handling + +The collection settings differ in tools, cameras, trainers, frame rate, and +coordinate distributions. Paediatric, Urology 1, and Urology 2 are not treated +as interchangeable samples from one acquisition system. + +The analysis therefore uses: + +- descriptive plots separated by cohort; +- feature correlations calculated within cohort and combined afterwards; +- cohort-adjusted linear models; +- within-cohort percentiles when constructing relative motion domains; +- leave-one-cohort-out checks for exploratory models. + +This approach retains common directional evidence without interpreting raw +millimetre, speed, or workspace values as directly comparable across rigs. + +## Task interval and motion extraction + +Blank start and end frames are excluded using one automatic sustained-motion +rule across all primary recordings. The interval begins shortly before the +first sustained instrument motion and ends shortly after the last. Dense phase +labels are used only to check this boundary rule. + +Position is smoothed with an approximately 0.4-second Savitzky-Golay window. +Velocity, acceleration, jerk, path length, workspace range, rotation, active +time, and bimanual measurements are then calculated from the retained frames. +The complete feature definitions are in [`FEATURES.md`](FEATURES.md). + +## Statistical analysis + +### Feature screen + +One constant feature was removed before inference, leaving 29 motion features. + +For each feature: + +1. novice, intermediate, and expert procedure groups were compared separately + within each cohort using the Kruskal-Wallis rank test, which does not assume + a normal distribution; +2. the three cohort probabilities were combined using Fisher's method, which + tests whether the cohort results collectively depart from the null + hypothesis; +3. associations with lifetime procedure count were estimated using Spearman + rank correlation within each cohort; +4. correlations were combined using fixed-effect meta-analysis, which weights + more precise cohort estimates more strongly; +5. false discovery rate correction was applied separately across the 29 group + tests and 29 continuous correlations. + +The corrected probability is reported as `q`. Cliff's delta compares novice +and expert procedure groups without assuming a normal distribution. It +estimates how often a value from one group exceeds a value from the other after +accounting for ties. Confidence intervals use 3,000 bootstrap resamples within +cohort. + +### Cohort-adjusted models + +Eight outcomes were selected to represent task duration, bimanual +coordination, speed, idle time, depth excursion, and the continuous +coordination and control score. + +Each standardised outcome was regressed on: + +- `log10(lifetime procedures + 1)`; and +- indicators for acquisition cohort. + +A one-unit change in the procedure term represents an approximately tenfold +increase in procedures plus one. Confidence intervals use 2,000 bootstrap +resamples within cohort. Probabilities use 5,000 permutations of procedure +volume within cohort and are corrected across the eight models. + +### Exploratory classification + +Logistic regression and random forest models compare: + +- duration alone; +- all 30 retained classifier inputs; +- fixed procedure-count groups; +- procedure tertiles; +- procedure clusters; +- data-derived motion-score bands. + +Five-fold evaluation preserves approximate class proportions. A separate +leave-one-cohort-out evaluation trains on two cohorts and tests on the third. +Balanced accuracy is the mean sensitivity across classes, so each class +contributes equally despite unequal sample sizes. + +## Main results + +### Procedure groups did not define clear motion classes + +No novice, intermediate, and expert feature comparison remained significant +after false discovery rate correction. This is an important result. It means +that the analysis cannot assume every high-volume participant performs the +recorded task better than every low-volume participant. + +Procedure-group classification was also weak: + +| Input and model | Five-fold balanced accuracy | +|:--|--:| +| Duration only, logistic regression | 0.45 +/- 0.17 | +| Duration only, random forest | 0.40 +/- 0.05 | +| All motion features, logistic regression | 0.36 +/- 0.07 | +| All motion features, random forest | 0.34 +/- 0.11 | + +Adding more kinematic features did not rescue the noisy procedure-count target. +This guards against presenting case volume as ground truth skill. + +### Continuous procedure volume retained modest associations + +Eight of 29 within-cohort meta-analytic correlations remained significant: + +| Feature | Spearman correlation | 95% confidence interval | Corrected `q` | +|:--|--:|:--|--:| +| Tool 1 normalised jerk | -0.338 | -0.500 to -0.153 | 0.011 | +| Analysed duration | -0.327 | -0.491 to -0.141 | 0.011 | +| Tool 2 normalised jerk | -0.301 | -0.469 to -0.112 | 0.020 | +| Tool 2 speed peaks | -0.288 | -0.458 to -0.098 | 0.023 | +| Tool 1 speed peaks | -0.282 | -0.453 to -0.092 | 0.023 | +| Tool 1 path length | -0.278 | -0.449 to -0.087 | 0.023 | +| Tool 2 speed | 0.265 | 0.074 to 0.438 | 0.026 | +| Bimanual correlation | 0.265 | 0.074 to 0.438 | 0.026 | + +The effects are modest and distributions overlap. Bimanual correlation also +showed moderate between-cohort heterogeneity (`I2 = 60%`). The `I2` statistic +estimates the percentage of observed variation attributable to differences +between cohorts rather than sampling uncertainty. This result should not be +treated as a universal threshold. + +### Cohort-adjusted estimates + +Three prespecified outcomes remained significant after correction: + +| Outcome | Standardised change per tenfold procedure increase | 95% confidence interval | Corrected `q` | +|:--|--:|:--|--:| +| Analysed duration | -0.43 | -0.72 to -0.19 | 0.002 | +| Bimanual correlation | 0.29 | 0.09 to 0.51 | 0.030 | +| Coordination and control score | 0.28 | 0.07 to 0.51 | 0.033 | + +These estimates support faster completion and stronger temporal coordination +as the clearest repeated signals. They do not establish causal effects of +training. + +### Acquisition setting can dominate raw movement + +Cohort alone explained 88% of the variance in Tool 1 speed and 92% in Tool 2 +speed. Procedure volume added little after cohort for these outcomes. + +This result supports separate cohort plots and warns against a single raw-speed +threshold across equipment configurations. + +## Data-derived motion-score bands + +The coordination and control score combines bimanual coordination and +instrument motion control. Within-cohort percentiles reduce the effect of +acquisition scale. K-means with three centres gives: + +| Band | Recordings | Centre | Procedure-count median | +|:--|--:|--:|--:| +| Lower | 16 | 2.16 | 40 | +| Middle | 41 | 2.87 | 20 | +| Upper | 50 | 3.50 | 80 | + +Observed cut points are 2.51 and 3.18. Agreement with fixed procedure-count +groups is negligible (adjusted Rand index 0.019). The adjusted Rand index +measures agreement between two group assignments after allowing for chance. A +value of one indicates identical assignments, while a value near zero +indicates no greater agreement than expected by chance. + +This mismatch is clinically informative. Defining skill from procedure count +alone can label a strong low-volume performance as novice and an inefficient +high-volume performance as expert. Motion and experience should therefore be +reported as related but distinct evidence. + +The three-band solution has silhouette 0.534 and Davies-Bouldin index 0.557. +The silhouette score is higher when recordings are closer to their assigned +group than neighbouring groups. The Davies-Bouldin index is lower when groups +are compact and well separated. Two- and four-band solutions also have +plausible internal indices. Three bands are retained to aid interpretation, +not because the data prove three natural competence classes. + +Logistic regression and random forest recover the band rule with balanced +accuracy around 0.95. This is expected because the target bands are calculated +from the same motion domains supplied to the classifier. It confirms software +consistency but is not independent prediction and is not evidence of clinical +validity. + +## Phase analysis + +The phase subset describes where time and corrections occur within a transfer +cycle. Labels include reach, grasp, transfer, place, nudge, idle, and dropped +object. + +Only four lower-band recordings enter the primary phase comparison. Drop +episodes are sparse, and several phase measures are non-monotonic across bands. +The phase results are therefore exploratory. Expanded annotation will improve +coverage, while an independently annotated sample is needed to quantify label +reliability. + +The current phase analysis should be used to: + +- identify actions responsible for long cycles; +- examine phase-specific speed or path rate; +- count visible drop episodes; +- develop candidate feedback for corrective nudges or repeated grasp attempts. + +It should not yet be used to assign clinical grades. + +## Potential value to surgeons and trainers + +With further validation, the framework could support: + +1. objective evidence alongside expert observation; +2. feedback that distinguishes timing, bimanual coordination, motion + smoothness, and workspace use; +3. longitudinal tracking within the same simulator setup; +4. identification of a specific phase responsible for inefficient completion; +5. transparent comparison of training curricula without assuming procedure + volume equals competence; +6. development of video-based systems trained against high-fidelity motion + measurements. + +The strongest near-term application is formative feedback. Summative +assessment would require blinded expert ratings, independent cohorts, +predefined thresholds, test-retest reliability, and evidence that score +changes correspond to meaningful training or clinical outcomes. + +## Reproducibility + +The analysis entry point is +[`scripts/build_scirep_analysis.py`](../scripts/build_scirep_analysis.py). +Package versions are pinned in [`requirements.txt`](../requirements.txt). + +Random seeds: + +| Operation | Seed | +|:--|--:| +| Bootstrap and sampling | 20260618 | +| Classification | 13 | +| Principal component analysis | 20260618 | +| K-means | 20260618 | + +Aggregate outputs include: + +- [`all_trial_feature_statistics.csv`](../data/derived/all_trial_feature_statistics.csv); +- [`cohort_adjusted_models.csv`](../data/derived/cohort_adjusted_models.csv); +- [`cohort_specific_effects.csv`](../data/derived/cohort_specific_effects.csv); +- [`continuous_performance_validation.csv`](../data/derived/continuous_performance_validation.csv); +- [`kmeans_cluster_summary.csv`](../data/derived/kmeans_cluster_summary.csv); +- [`kmeans_validation.csv`](../data/derived/kmeans_validation.csv); +- [`processing_sensitivity.csv`](../data/derived/processing_sensitivity.csv); +- [`phase_kinematics_summary.csv`](../data/derived/phase_kinematics_summary.csv); +- [`analysis_manifest.json`](../data/derived/analysis_manifest.json). + +Participant-level intermediate tables are deliberately omitted until the +complete governed dataset release is available. diff --git a/docs/DATASET.md b/docs/DATASET.md new file mode 100644 index 0000000..07b6d89 --- /dev/null +++ b/docs/DATASET.md @@ -0,0 +1,214 @@ +# Dataset guide + +## Scope + +LASK contains video and instrument measurements from a laparoscopic +peg-transfer task performed in box trainers. A participant uses two graspers to +move objects between pegs. One recording contains one participant attempt. + +The collection was designed to support: + +- instrument detection, segmentation, and landmark localisation; +- six-channel and seven-channel pose estimation from video; +- instrument tracking through occlusion; +- analysis of motion, coordination, and task timing; +- study of acquisition shift between training settings. + +It was not designed as a validated clinical examination. Procedure counts, +career stage, and calculated motion scores should not be interpreted as +interchangeable measures of competence. + +## Release stages + +The permanent concept DOI is +[10.5281/zenodo.20752650](https://doi.org/10.5281/zenodo.20752650). +[Version 1.0](https://zenodo.org/records/20752651) contains 37 recordings. +Further recordings and phase labels will be added as new versions. + +### Public version 1.0 + +| Cohort | Archive | Role | Recordings | +|:--|:--|:--|--:| +| Paediatric | `DatasetB_BAPES_val.zip` | Validation | 10 | +| Urology 1 | `DatasetC_6DOF_test.zip` | Testing | 8 | +| Urology 2 | `DatasetA_7DOF_train.zip` | Training | 19 | +| **Total** | | | **37** | + +### Complete collection used in the quantitative analysis + +| Cohort | Available | Primary motion set | Phase labelled | Primary phase set | +|:--|--:|--:|--:|--:| +| Paediatric | 30 | 28 | 9 | 9 | +| Urology 1 | 24 | 24 | 10 | 10 | +| Urology 2 | 61 | 55 | 19 | 15 | +| **Total** | **115** | **107** | **38** | **34** | + +The primary sets retain study identifiers represented by one recording. +Repeated or multipart recordings remain in the dataset inventory but are not +treated as independent observations in the primary inferential analyses. + +## Cohorts + +### Paediatric + +Data were collected at the British Association of Paediatric Endoscopic +Surgeons meeting in November 2024. Participants were paediatric consultants or +trainees between specialty training years 3 and 7. The questionnaire recorded +lifetime and previous-year skin-to-skin procedure counts. Position, +orientation, and relative jaw-opening channels were recorded at 13 frames per +second. + +### Urology 1 + +Data were collected at a urology simulation boot camp in October 2023. +Fenestrated and curved instruments were tracked at approximately 26 frames per +second. Position and orientation were available, but no jaw-opening channel was +recorded. + +### Urology 2 + +Data were collected at a urology simulation boot camp in October 2024. +Position, orientation, and relative jaw-opening channels were recorded at +13 frames per second. + +The trainers, tools, camera settings, coordinate distributions, and insertion +angles differed between cohorts. Raw kinematic values should not be pooled +without cohort adjustment. + +Degrees of freedom describe the independently measured motion channels. Six +degrees of freedom comprise three-dimensional position and three-dimensional +orientation. The seventh channel is the relative jaw-opening measurement. + +## Archive structure + +Each public archive follows this layout: + +```text +DatasetX_.../ +├── videos/ +│ └── .mp4 +├── annotations/ +│ └── .npz +├── DatasetX_..._kinematics.csv +├── participants.csv +└── README.md +``` + +The videos are H.264 working encodes. Dense kinematics are aligned by frame +number and time. Instrument annotations are sparse, usually sampled at +approximately every 100th frame. + +## Kinematic files + +The Aurora electromagnetic tracker from Northern Digital Inc. records tool +pose on every sampled frame. + +### Position + +Position columns ending in `px`, `py`, and `pz` are expressed in millimetres. +The sensor coils are mounted near the instrument base. A fixed calibration +transform estimates the instrument tip, and positions can then be represented +relative to the camera frame. + +### Orientation + +Orientations are stored as unit quaternions. The seven-channel archives use +`qw`, `qx`, `qy`, and `qz`. The six-channel archive retains its original +`q0`, `q1`, `q2`, and `q3` naming. Check the archive README before converting +between quaternion conventions. + +Angular movement in the companion analysis is calculated from the relative +rotation between adjacent unit quaternions. Quaternion signs are aligned before +the optional orientation-smoothing sensitivity analysis because `q` and `-q` +represent the same rotation. + +### Jaw opening + +The two seven-channel cohorts contain voltage-derived jaw measurements. The +current files may retain `tool1_angle` and `tool2_angle` as historical column +names. These channels should be treated as relative aperture signals unless an +independent physical angle calibration is supplied. + +For descriptive analysis, the voltage is low-pass filtered and mapped within +each recording so that the 10th percentile represents closed and the 90th +percentile represents open. Values are clipped to an opening fraction from +zero to one. This removes between-recording voltage offsets but does not +produce an angle in radians or degrees. + +### Time + +The seven-channel cohorts use seconds. The six-channel archive retains a +millisecond time column. The paper analysis uses fixed cohort rates of +13 frames per second for Paediatric and Urology 2, and 26 frames per second for +Urology 1. + +## Instrument annotation files + +The `.npz` files are compressed NumPy array archives and can be loaded without +pickle: + +```python +import numpy as np + +annotation = np.load("annotations/Trial01.npz") +print(annotation.files) +``` + +Important arrays include: + +| Array | Shape | Meaning | +|:--|:--|:--| +| `frame_idx` | `(F,)` | Frame number in the source video and kinematic file | +| `toolN_mask_points` | `(M, 2)` | Concatenated polygon coordinates in pixels | +| `toolN_mask_offsets` | `(F+1,)` | Start and end offsets for each frame polygon | +| `toolN_ee_tip` | `(F, 2)` | Instrument-tip landmark | +| `toolN_ee_left` | `(F, 2)` | Left jaw landmark | +| `toolN_ee_right` | `(F, 2)` | Right jaw landmark | +| `toolN_joint` | `(F, 2)` | End-effector or shaft-joint landmark | +| `toolN_vis_mask` | `(F,)` | Whether a segmentation mask is present | + +`N` is 1 or 2. Landmark coordinates are stored as image `(x, y)` positions. +Unavailable landmarks are represented by `NaN`, meaning not a number. + +## Participant metadata + +`participants.csv` contains one row per released recording: + +- dataset and split; +- release trial identifier; +- handedness; +- lifetime laparoscopic procedure count; +- procedure count during the previous year, where collected. + +The handedness questionnaire was based on the Edinburgh Handedness Inventory. +The analysis uses the lifetime procedure count as a continuous experience +measure. The historical categories are: + +- novice: fewer than 20 procedures; +- intermediate: 20 to 99 procedures; +- expert: at least 100 procedures. + +These labels describe procedure-volume strata. They are not expert ratings of +the recorded peg-transfer attempt. + +## Dense phase labels + +The phase subset includes reach, grasp, transfer, place, nudge, idle, and +dropped-object labels. It is documented separately in +[`PHASES.md`](PHASES.md). + +## Known limitations + +- The current public release is smaller than the complete collection. +- Cohort and acquisition setting are confounded. +- Jaw opening is absent from Urology 1. +- The relative jaw signal does not provide an absolute physical aperture. +- Fine object possession and grasp quality cannot always be inferred from the + kinematic channels alone. +- Experience metadata were self-reported. +- Phase labels were produced by one trained annotator. +- Study identifiers cannot exclude the possibility that a participant + attended more than one event under different identifiers. + +These limitations are retained explicitly so that future benchmark results can +be interpreted against the evidence the dataset actually provides. diff --git a/docs/FEATURES.md b/docs/FEATURES.md new file mode 100644 index 0000000..339ab0f --- /dev/null +++ b/docs/FEATURES.md @@ -0,0 +1,241 @@ +# Motion feature definitions + +## Overview + +The analysis converts dense three-dimensional position and orientation +measurements into interpretable summaries of instrument use. This page defines +the main quantities in plain language and gives the calculation where it +matters. + +The machine-readable dictionary is +[`data/derived/feature_dictionary.csv`](../data/derived/feature_dictionary.csv). +Its display directions are analytical hypotheses, not validated clinical +preferences. + +## Preprocessing + +For Paediatric and Urology 2, time is calculated at 13 frames per second. +Urology 1 uses 26 frames per second. + +Position is smoothed separately on each axis using a third-order +Savitzky-Golay filter over approximately 0.4 seconds. This corresponds to +5 frames at 13 frames per second and 11 frames at 26 frames per second. The +filter fits a short polynomial within each moving window, reducing +frame-to-frame noise while retaining the overall path. + +The analysed task interval starts 0.5 seconds before the first period of +sustained motion above 2 mm/s and ends 0.5 seconds after the final sustained +period. The sustained interval must last approximately 0.4 seconds. The same +automatic rule is applied to all primary recordings. + +Dense phase labels are used to validate, but not define, the primary automatic +interval. In 34 phase-labelled primary recordings, median interval overlap was +0.968 and the median absolute duration difference was 4.65 seconds. + +## Single-instrument features + +Let `p(t)` be tool-tip position at time `t`. + +### Speed + +Velocity is the rate of change of position: + +```text +v(t) = dp(t) / dt +speed(t) = ||v(t)|| +``` + +Average speed is the mean speed across the analysed interval. + +### Path length + +Path length is the sum of the three-dimensional distances between consecutive +positions: + +```text +path length = sum ||p(i) - p(i-1)|| +``` + +It measures total instrument travel in millimetres. A shorter path is not +automatically better because the required path depends on object and peg +locations. + +### Acceleration and jerk + +Acceleration is the rate of change of velocity. Jerk is the rate of change of +acceleration. Abrupt starts, stops, or corrections increase jerk. + +The dimensionless normalised jerk used here is: + +```text +normalised jerk = +sqrt(0.5 * integral(||jerk(t)||^2 dt) * duration^5 / path_length^2) +``` + +Duration and path length appear in the normalisation. Normalised jerk is +therefore not independent of task time. + +### Active and idle fractions + +A tool is considered active at a frame when speed is at least 5 mm/s. Its +active fraction is the proportion of active frames. Idle episodes are +contiguous periods below 5 mm/s lasting at least 0.25 seconds. + +### Speed peaks + +Speed peaks are local maxima above 5 mm/s with at least 2 mm/s prominence and +at least 0.25 seconds separation. They provide a simple description of +stop-start movement, not a direct count of corrective actions. + +### Workspace + +- Image-plane working area is the `x` range multiplied by the `y` range. +- Working volume is the product of the `x`, `y`, and `z` ranges. +- Depth range is the range along the camera axis. + +These are axis-aligned bounding measures. They do not reconstruct the exact +shape or occupied volume of the path. + +### Rotation + +The change in orientation between two frames is obtained from the relative +rotation between their unit quaternions. Total rotation is the sum of the +frame-to-frame rotation magnitudes in degrees. + +Rotation per path is: + +```text +total rotation in degrees / path length in millimetres +``` + +Angular velocity variability is the standard deviation of non-zero angular +speed divided by its mean. + +## Bimanual features + +### Bimanual correlation + +Bimanual correlation is the Pearson correlation between the simultaneous Tool +1 and Tool 2 speed traces. A positive value means the tools tend to speed up +and slow down together. It does not show that the instruments follow the same +spatial path. + +### Bimanual lag + +Cross-correlation compares one speed trace with shifted copies of the other. +The shift with the greatest similarity, searched within plus or minus +5 seconds, defines the bimanual lag. A lag of zero means the strongest +similarity occurs without shifting either trace. A larger lag indicates a +greater timing offset, but it does not identify which hand caused it. + +### Simultaneous motion + +Frames above 10 mm/s are treated as clearly moving: + +```text +simultaneous motion = +frames where both tools move / frames where either tool moves +``` + +### Concurrent efficiency + +When both tools move faster than 5 mm/s, their speed balance is calculated as: + +```text +minimum(tool speeds) / maximum(tool speeds) +``` + +The reported value is the mean across those frames. A value near one indicates +similar concurrent speeds. This is called concurrent efficiency rather than +concurrent tool use because it measures speed balance during concurrent +movement. + +### Bimanual symmetry + +Bimanual symmetry is the shorter tool path divided by the longer tool path. +It approaches one when path lengths are similar. Equal path lengths are not +necessarily optimal for every transfer. + +### Tool-distance variability + +The distance between tool tips is calculated at each frame. Its coefficient of +variation is the standard deviation divided by the mean. This describes the +stability of inter-tool spacing. + +## Motion domains + +The paper uses four descriptive domains. Closely related or mirrored tool +measurements are averaged into constructs first so that duplicate channels do +not receive extra weight. + +### Bimanual coordination + +Equal-weighted constructs: + +1. bimanual speed correlation; +2. simultaneous motion and concurrent efficiency; +3. stability of the distance between tools. + +### Instrument motion control + +Equal-weighted constructs: + +1. normalised jerk for both tools; +2. rotation per path for both tools; +3. angular velocity variability for both tools. + +### Task efficiency + +Equal-weighted constructs: + +1. analysed duration; +2. idle and active fractions; +3. speed-peak counts. + +### Descriptive workspace compactness + +Equal-weighted constructs: + +1. working volumes; +2. image-plane working areas; +3. depth ranges. + +No clinical preference for a smaller workspace is assumed because appropriate +excursion depends on task phase and trainer geometry. + +## Coordination and control score + +Each feature is first oriented according to its prespecified display direction +and converted to a percentile within its acquisition cohort. Let `r` be the +mean construct percentile. A domain is displayed on a relative 1 to 5 scale: + +```text +domain score = 1 + 4 * r +``` + +The coordination and control score is the equal-weighted mean of bimanual +coordination and instrument motion control. Task efficiency and workspace are +not entered as separate score features. + +This score is relative to the observed cohort. It is not a validated Objective +Structured Assessment of Technical Skills, Global Operative Assessment of +Laparoscopic Skills, Global Evaluative Assessment of Robotic Skills, McGill +Inanimate System for Training and Evaluation of Laparoscopic Skills, or +Fundamentals of Laparoscopic Surgery rating. + +## Motion-score bands + +K-means groups similar score values around a chosen number of centres. With +three centres, it partitions the continuous coordination and control score +into descriptive bands: + +| Band | Recordings | Centre | Observed cut point | +|:--|--:|--:|--:| +| Lower | 16 | 2.16 | Below 2.51 | +| Middle | 41 | 2.87 | 2.51 to 3.18 | +| Upper | 50 | 3.50 | Above 3.18 | + +The cut points are conditional on the current score construction and cohort +percentiles. They should be recalculated when the dataset expands. Two- and +four-band solutions also had plausible internal indices, so three bands are +retained for interpretability rather than claimed as natural clinical classes. diff --git a/docs/PHASES.md b/docs/PHASES.md new file mode 100644 index 0000000..61cf67c --- /dev/null +++ b/docs/PHASES.md @@ -0,0 +1,74 @@ +# Phase annotation protocol + +## Purpose + +Dense phase labels divide a peg-transfer recording into observable actions. +They support cycle timing, action composition, and phase-specific kinematic +analysis. They are not ratings of how well an action was performed. + +## Annotation process + +One researcher, O.C., annotated the phases. The definitions were developed +through direct instruction and discussion with surgeons at the urology and +paediatric boot camps, together with observation of peg-transfer performance +at those events. + +Participant seniority and procedure-count metadata were not visible during +annotation. Differences in apparent confidence and fluency could still be +perceived from the videos, and the annotator sometimes formed informal +impressions of skill. Those impressions were not recorded and were not used to +define the phase labels. + +No second annotator or adjudication process has yet been completed. Inter-rater +agreement therefore cannot currently be reported. + +## Labels + +| Phase | Operational definition | +|:--|:--| +| Reach | Movement towards an object before attempting control | +| Grasp | Attempt to obtain control of an object | +| Transfer | Movement of an object between instruments | +| Place | Positioning or releasing an object at a destination peg | +| Nudge | Corrective contact intended to reposition an object | +| Idle | No task-directed instrument action | +| Dropped | Visible loss of object control | + +Per-frame exports can also contain tool-specific labels, cycle index, active +tool, and events. The `active_tool` field in an earlier consolidated export is +zero for every frame and is not used by the paper analysis. Tool activity is +instead derived from speed or from the phase label, depending on the analysis. + +## From frames to cycles + +A candidate cycle is retained only when: + +1. it contains a `place` label; and +2. it contains at least one `reach`, `grasp`, `transfer`, or `nudge` label. + +An exporter could occasionally place a boundary segment containing only +placement frames into the next cycle index. Such a segment is attached to the +immediately preceding interval only when that interval has no placement. +Other placement-only residuals and attempts without placement are excluded. + +This rule produced: + +| Analysis set | Recordings | Placement-complete cycles | +|:--|--:|--:| +| Full phase inventory | 38 | 425 | +| Primary phase set | 34 | 383 | + +Cycle measurements are first averaged within recording and then across +recordings. A participant with more completed transfers therefore does not +receive more weight in recording-level comparisons. + +## Current interpretation + +Phase results are descriptive. The annotated subset is incomplete and its +motion-score composition is imbalanced. In particular, only four recordings in +the lower motion-score band enter the primary phase comparison. Expanded +annotation and an independent reliability sample are planned. + +Phase labels may be used to locate where time, corrections, or dropped-object +episodes occur. They should not be converted directly into clinical competence +grades without expert validation. diff --git a/paper/figures/fig1_cohort_structure.png b/paper/figures/fig1_cohort_structure.png new file mode 100644 index 0000000..e678110 Binary files /dev/null and b/paper/figures/fig1_cohort_structure.png differ diff --git a/paper/figures/fig21_cohort_effect_forest.png b/paper/figures/fig21_cohort_effect_forest.png new file mode 100644 index 0000000..f2f4413 Binary files /dev/null and b/paper/figures/fig21_cohort_effect_forest.png differ diff --git a/paper/figures/fig23_continuous_performance_validation.png b/paper/figures/fig23_continuous_performance_validation.png new file mode 100644 index 0000000..d2dd303 Binary files /dev/null and b/paper/figures/fig23_continuous_performance_validation.png differ diff --git a/paper/figures/fig2_feature_effects.png b/paper/figures/fig2_feature_effects.png new file mode 100644 index 0000000..87ac966 Binary files /dev/null and b/paper/figures/fig2_feature_effects.png differ diff --git a/paper/figures/fig3_phase_timing.png b/paper/figures/fig3_phase_timing.png new file mode 100644 index 0000000..939a410 Binary files /dev/null and b/paper/figures/fig3_phase_timing.png differ diff --git a/paper/figures/fig4_experience_links.png b/paper/figures/fig4_experience_links.png new file mode 100644 index 0000000..83e4889 Binary files /dev/null and b/paper/figures/fig4_experience_links.png differ diff --git a/requirements.txt b/requirements.txt new file mode 100644 index 0000000..d02d375 --- /dev/null +++ b/requirements.txt @@ -0,0 +1,7 @@ +matplotlib==3.10.8 +numpy==2.4.4 +pandas==3.0.1 +pyarrow==23.0.1 +scikit-learn==1.8.0 +scipy==1.17.1 +seaborn==0.13.2 diff --git a/scripts/build_scirep_analysis.py b/scripts/build_scirep_analysis.py new file mode 100644 index 0000000..b418801 --- /dev/null +++ b/scripts/build_scirep_analysis.py @@ -0,0 +1,4987 @@ +#!/usr/bin/env python3 +"""Build analysis artefacts for the Scientific Reports peg-transfer paper. + +This script intentionally separates: + 1. the 115-recording dataset inventory, + 2. consistently processed kinematics for the 107 recordings whose study + identifier occurs once, and + 3. the 38-recording phase-annotated subset for phase/cycle analyses. + +The input locations can be configured with: + LASK_PHASE_CACHE Directory containing canonical_trials.csv and the phase + parquet files. + LASK_AI_ELT_ROOT AI-ELT project root containing motion and feature caches. + LASK_ORIGIN_MOTION + Optional override for the per-recording kinematic JSON root. + LASK_PROJECT_ROOT Optional output project root. +""" +from __future__ import annotations + +import json +import math +import os +import platform +import re +import hashlib +import sys +from importlib import metadata +from pathlib import Path +from typing import Iterable + + +def configured_path(name: str, default: Path) -> Path: + """Return an environment-configured path with a deterministic default.""" + return Path(os.environ.get(name, str(default))).expanduser().resolve() + + +SCRIPT_PROJECT_ROOT = Path(__file__).resolve().parents[1] +ROOT = configured_path("LASK_PROJECT_ROOT", SCRIPT_PROJECT_ROOT) +PHD = SCRIPT_PROJECT_ROOT.parents[1] +BTPN_CACHE = configured_path( + "LASK_PHASE_CACHE", + PHD / "Code" / "BTPN-MT" / "data" / "cache", +) +AI_ELT = configured_path("LASK_AI_ELT_ROOT", PHD / "Code" / "AI-ELT") +ORIGIN_MOTION = configured_path( + "LASK_ORIGIN_MOTION", + AI_ELT / "outputs" / "ssl" / "origin" / "origin_data" / "ORIGIN_ALL", +) + +os.environ.setdefault( + "MPLCONFIGDIR", + str(ROOT / "paper" / "build" / "mplconfig"), +) + +import matplotlib + +matplotlib.use("Agg") +import matplotlib.pyplot as plt +from matplotlib.patches import FancyArrowPatch, Rectangle +import numpy as np +import pandas as pd +from scipy import stats +from scipy.signal import correlate, correlation_lags, find_peaks, savgol_filter +from scipy.spatial.transform import Rotation + + +OUT = ROOT / "data" / "derived" +FIG = ROOT / "paper" / "figures" +TAB = ROOT / "paper" / "tables" +for d in (OUT, FIG, TAB): + d.mkdir(parents=True, exist_ok=True) + + +def file_sha256(path: Path) -> str | None: + if not path.exists() or not path.is_file(): + return None + h = hashlib.sha256() + with path.open("rb") as fh: + for chunk in iter(lambda: fh.read(1024 * 1024), b""): + h.update(chunk) + return h.hexdigest() + + +def package_version(name: str) -> str | None: + try: + return metadata.version(name) + except metadata.PackageNotFoundError: + return None + + +def analysis_input_paths() -> dict[str, Path]: + """Return the core files required to regenerate the reported analysis.""" + return { + "canonical_trials": BTPN_CACHE / "canonical_trials.csv", + "phase_trials": BTPN_CACHE / "trials.parquet", + "phase_cycles": BTPN_CACHE / "cycles.parquet", + "phase_frames": BTPN_CACHE / "frames.parquet", + "combined_motion_metrics": AI_ELT + / "outputs" + / "motion" + / "combined_motion_metrics.csv", + "trimmed_extended_features": AI_ELT + / "outputs" + / "papers" + / "paper1" + / "results" + / "extended_features_cache_trimmed.csv", + } + + +def validate_analysis_inputs() -> None: + """Fail early with actionable configuration guidance.""" + missing = [ + f"{name}: {path}" + for name, path in analysis_input_paths().items() + if not path.is_file() + ] + if not ORIGIN_MOTION.is_dir(): + missing.append(f"origin_motion_directory: {ORIGIN_MOTION}") + if missing: + details = "\n ".join(missing) + raise FileNotFoundError( + "The analysis inputs are incomplete. Configure LASK_PHASE_CACHE, " + "LASK_AI_ELT_ROOT, and optionally LASK_ORIGIN_MOTION.\n " + f"{details}" + ) + + +def portable_input_path(path: Path) -> str: + """Describe an input without writing a machine-specific absolute path.""" + try: + return f"$LASK_PHASE_CACHE/{path.relative_to(BTPN_CACHE)}" + except ValueError: + pass + try: + return f"$LASK_AI_ELT_ROOT/{path.relative_to(AI_ELT)}" + except ValueError: + return path.name + +DATASET_ORDER = ["BAPES2024", "6DOF2023", "7DOF2024"] +DATASET_LABELS = { + "7DOF2024": "Urology 2 (2024)", + "BAPES2024": "Paediatric", + "6DOF2023": "Urology 1 (2023)", +} +SKILL_ORDER = ["novice", "intermediate", "expert"] +PHASE_ORDER = ["reach", "grasp", "transfer", "place", "nudge", "idle", "dropped"] +DATASET_COLORS = {"BAPES2024": "#0072B2", "6DOF2023": "#009E73", "7DOF2024": "#D55E00"} +BAND_ORDER = ["metric_low", "metric_middle", "metric_high"] +BAND_LABELS = { + "metric_low": "Lower motion-score band", + "metric_middle": "Middle motion-score band", + "metric_high": "Upper motion-score band", +} +BAND_SHORT_LABELS = { + "metric_low": "Lower", + "metric_middle": "Middle", + "metric_high": "Upper", +} +CLUSTER_COLORS = { + "metric_low": "#009E73", + "metric_middle": "#E69F00", + "metric_high": "#CC3311", +} +SKILL_COLORS = {"novice": "#009E73", "intermediate": "#E69F00", "expert": "#CC3311"} +PHASE_COLORS = { + "reach": "#4FC3F7", + "grasp": "#66BB6A", + "transfer": "#FFA726", + "place": "#AB47BC", + "nudge": "#FFD54F", + "idle": "#B0BEC5", + "dropped": "#E53935", +} + +FEATURES = { + "total_time": "Analysed duration", + "bimanual_correlation": "Bimanual correlation", + "bimanual_lag_s": "Bimanual lag", + "bimanual_concurrent_efficiency": "Concurrent efficiency", + "bimanual_dist_speed_corr": "Distance-speed coupling", + "simultaneous_motion_ratio": "Simultaneous motion", + "bimanual_symmetry": "Bimanual symmetry", + "combined_idle_ratio": "Combined idle ratio", + "tool1_normalized_jerk": "Tool 1 normalised jerk", + "tool2_normalized_jerk": "Tool 2 normalised jerk", + "tool1_active_time_ratio": "Tool 1 active time", + "tool2_active_time_ratio": "Tool 2 active time", + "tool1_path_length": "Tool 1 path length", + "tool2_path_length": "Tool 2 path length", + "tool1_avg_speed": "Tool 1 speed", + "tool2_avg_speed": "Tool 2 speed", + "tool_distance_cv": "Tool distance variability", + "tool_close_proximity_ratio": "Close proximity", + "tool1_working_volume": "Tool 1 working volume", + "tool2_working_volume": "Tool 2 working volume", + "tool1_working_area_xy": "Tool 1 working area", + "tool2_working_area_xy": "Tool 2 working area", + "tool1_range_z": "Tool 1 depth range", + "tool2_range_z": "Tool 2 depth range", + "tool1_rotation_per_path": "Tool 1 rotation per path", + "tool2_rotation_per_path": "Tool 2 rotation per path", + "tool1_angular_velocity_cv": "Tool 1 angular velocity variability", + "tool2_angular_velocity_cv": "Tool 2 angular velocity variability", + "tool1_num_speed_peaks": "Tool 1 speed peaks", + "tool2_num_speed_peaks": "Tool 2 speed peaks", + "Coordination-control composite": "Coordination and control score", +} + +FEATURE_DIRECTIONS = { + "total_time": ("$\\downarrow$", "shorter analysed task interval"), + "bimanual_correlation": ("$\\uparrow$", "more synchronous tool motion"), + "bimanual_lag_s": ("$\\downarrow$", "less timing delay between tools"), + "bimanual_concurrent_efficiency": ("$\\uparrow$", "more efficient simultaneous tool use"), + "bimanual_dist_speed_corr": ("$\\uparrow$", "more coupled spacing and speed control"), + "simultaneous_motion_ratio": ("$\\uparrow$", "larger fraction of active frames with both tools moving"), + "bimanual_symmetry": ("$\\uparrow$", "more balanced left/right tool use"), + "combined_idle_ratio": ("$\\downarrow$", "less inactive task time"), + "tool1_normalized_jerk": ("$\\downarrow$", "smoother Tool 1 motion"), + "tool2_normalized_jerk": ("$\\downarrow$", "smoother Tool 2 motion"), + "tool1_active_time_ratio": ("$\\uparrow$", "more active Tool 1 use"), + "tool2_active_time_ratio": ("$\\uparrow$", "more active Tool 2 use"), + "tool1_path_length": ("$\\downarrow$", "shorter Tool 1 travel distance"), + "tool2_path_length": ("$\\downarrow$", "shorter Tool 2 travel distance"), + "tool1_avg_speed": ("$\\uparrow$", "faster Tool 1 movement"), + "tool2_avg_speed": ("$\\uparrow$", "faster Tool 2 movement"), + "tool_distance_cv": ("$\\downarrow$", "steadier distance between tools"), + "tool_close_proximity_ratio": ("$\\downarrow$", "less time with tools very close together"), + "tool1_working_volume": ("$\\downarrow$", "more compact Tool 1 three-dimensional workspace"), + "tool2_working_volume": ("$\\downarrow$", "more compact Tool 2 three-dimensional workspace"), + "tool1_working_area_xy": ("$\\downarrow$", "more compact Tool 1 image-plane workspace"), + "tool2_working_area_xy": ("$\\downarrow$", "more compact Tool 2 image-plane workspace"), + "tool1_range_z": ("$\\downarrow$", "less Tool 1 camera-axis excursion"), + "tool2_range_z": ("$\\downarrow$", "less Tool 2 camera-axis excursion"), + "tool1_rotation_per_path": ("$\\downarrow$", "less Tool 1 rotation per millimetre travelled"), + "tool2_rotation_per_path": ("$\\downarrow$", "less Tool 2 rotation per millimetre travelled"), + "tool1_angular_velocity_cv": ("$\\downarrow$", "steadier Tool 1 rotational speed"), + "tool2_angular_velocity_cv": ("$\\downarrow$", "steadier Tool 2 rotational speed"), + "tool1_num_speed_peaks": ("$\\downarrow$", "fewer Tool 1 stop-start speed peaks"), + "tool2_num_speed_peaks": ("$\\downarrow$", "fewer Tool 2 stop-start speed peaks"), +} + +KMEANS_FEATURES = [ + "bimanual_correlation", + "bimanual_lag_s", + "simultaneous_motion_ratio", + "bimanual_symmetry", + "combined_idle_ratio", + "tool_distance_cv", + "tool1_normalized_jerk", + "tool2_normalized_jerk", + "tool1_active_time_ratio", + "tool2_active_time_ratio", + "bimanual_concurrent_efficiency", + "tool1_working_volume", + "tool2_working_volume", + "tool1_working_area_xy", + "tool2_working_area_xy", + "tool1_range_z", + "tool2_range_z", + "tool1_rotation_per_path", + "tool2_rotation_per_path", + "tool1_angular_velocity_cv", + "tool2_angular_velocity_cv", + "tool_close_proximity_ratio", +] + +OSATS_DOMAINS = { + "Bimanual coordination": [ + ("bimanual_correlation", 1), + ("simultaneous_motion_ratio", 1), + ("bimanual_concurrent_efficiency", 1), + ("tool_distance_cv", -1), + ], + "Task efficiency": [ + ("total_time", -1), + ("combined_idle_ratio", -1), + ("tool1_active_time_ratio", 1), + ("tool2_active_time_ratio", 1), + ("tool1_num_speed_peaks", -1), + ("tool2_num_speed_peaks", -1), + ], + "Instrument motion control": [ + ("tool1_normalized_jerk", -1), + ("tool2_normalized_jerk", -1), + ("tool1_rotation_per_path", -1), + ("tool2_rotation_per_path", -1), + ("tool1_angular_velocity_cv", -1), + ("tool2_angular_velocity_cv", -1), + ], + "Workspace excursion": [ + ("tool1_working_volume", -1), + ("tool2_working_volume", -1), + ("tool1_working_area_xy", -1), + ("tool2_working_area_xy", -1), + ("tool1_range_z", -1), + ("tool2_range_z", -1), + ], +} + +# Mirrored tools and closely related measurements are first averaged into +# motor constructs so that duplicated channels do not receive extra weight. +OSATS_DOMAIN_CONSTRUCTS = { + "Bimanual coordination": [ + [("bimanual_correlation", 1)], + [("simultaneous_motion_ratio", 1), ("bimanual_concurrent_efficiency", 1)], + [("tool_distance_cv", -1)], + ], + "Task efficiency": [ + [("total_time", -1)], + [("combined_idle_ratio", -1), ("tool1_active_time_ratio", 1), ("tool2_active_time_ratio", 1)], + [("tool1_num_speed_peaks", -1), ("tool2_num_speed_peaks", -1)], + ], + "Instrument motion control": [ + [("tool1_normalized_jerk", -1), ("tool2_normalized_jerk", -1)], + [("tool1_rotation_per_path", -1), ("tool2_rotation_per_path", -1)], + [("tool1_angular_velocity_cv", -1), ("tool2_angular_velocity_cv", -1)], + ], + "Workspace excursion": [ + [("tool1_working_volume", -1), ("tool2_working_volume", -1)], + [("tool1_working_area_xy", -1), ("tool2_working_area_xy", -1)], + [("tool1_range_z", -1), ("tool2_range_z", -1)], + ], +} + +# These two domains are available in every cohort and avoid direct timing +# measures. Task efficiency and workspace excursion are retained as +# descriptive checks that do not define the bands. +CORE_PERFORMANCE_DOMAINS = [ + "Bimanual coordination", + "Instrument motion control", +] + + +def setup_style() -> None: + plt.rcParams.update( + { + "figure.dpi": 300, + "savefig.dpi": 300, + "font.family": "serif", + "font.serif": ["Times New Roman", "Times", "DejaVu Serif"], + "mathtext.fontset": "stix", + "pdf.fonttype": 42, + "ps.fonttype": 42, + "axes.linewidth": 0.6, + "font.size": 8, + "axes.labelsize": 8, + "axes.titlesize": 9, + "xtick.labelsize": 7, + "ytick.labelsize": 7, + "legend.fontsize": 7, + "figure.facecolor": "white", + "savefig.facecolor": "white", + "savefig.bbox": "tight", + } + ) + + +def save_fig(fig: plt.Figure, name: str) -> None: + fig.savefig(FIG / f"{name}.png") + fig.savefig(FIG / f"{name}.pdf") + plt.close(fig) + + +def nice(s: str) -> str: + return str(s).replace("_", " ").title() + + +def dataset_label(dataset: str) -> str: + return DATASET_LABELS.get(str(dataset), str(dataset)) + + +def dataset_slug(dataset: str) -> str: + return re.sub(r"[^A-Za-z0-9]+", "", str(dataset)) + + +def panel_label(ax: plt.Axes, label: str) -> None: + ax.text(-0.12, 1.03, label, transform=ax.transAxes, fontsize=10, fontweight="bold") + + +def trial_key(dataset: str, trial_number: int) -> str: + if dataset == "6DOF2023": + return f"6Test {int(trial_number)}" + if dataset == "7DOF2024": + return f"7Trial{int(trial_number)}" + if dataset == "BAPES2024": + return f"BTrial{int(trial_number)}" + return f"{dataset}_{trial_number}" + + +def phase_trial_key(dataset: str, trial_short: str) -> str: + text = str(trial_short) + match = re.search(r"Trial\s*(\d+)", text) + if match: + return trial_key(dataset, int(match.group(1))) + match = re.search(r"Test\s*(\d+)", text) + if match: + return trial_key(dataset, int(match.group(1))) + return f"{dataset}_{text}" + + +def load_handedness() -> pd.DataFrame: + rows: list[dict[str, object]] = [] + paths = { + "6DOF2023": AI_ELT / "outputs" / "ssl" / "raw" / "6DOF2023" / "ssl_results" / "trial_embeddings.json", + "7DOF2024": AI_ELT / "outputs" / "ssl" / "raw" / "7DOF2024" / "ssl_results" / "trial_embeddings.json", + "BAPES2024": AI_ELT / "outputs" / "ssl" / "raw" / "BAPES2024" / "ssl_results" / "trial_embeddings.json", + } + for dataset, path in paths.items(): + if not path.exists(): + continue + data = json.loads(path.read_text()) + for value in data.values(): + hand = str(value.get("handedness", "unknown")).lower().strip() + if "right" in hand: + compact = "right" + elif "left" in hand: + compact = "left" + elif "mixed" in hand or "ambi" in hand: + compact = "mixed" + else: + compact = "unknown" + rows.append( + { + "dataset": dataset, + "trial_number": int(value.get("trial_number", 0)), + "handedness": compact, + "procedures_last_12_months": value.get("procedures_last_12_months", np.nan), + } + ) + return pd.DataFrame(rows) + + +def _safe_corr(x: np.ndarray, y: np.ndarray) -> float: + if len(x) < 3 or len(y) < 3 or np.std(x) <= 1e-12 or np.std(y) <= 1e-12: + return np.nan + return float(np.corrcoef(x, y)[0, 1]) + + +def _savgol_positions( + positions: np.ndarray, + fps: int, + window_s: float = 0.4, +) -> np.ndarray: + """Smooth position using a short polynomial window.""" + if window_s <= 0: + return positions.copy() + window = max(5, int(round(window_s * fps))) + if window % 2 == 0: + window += 1 + if len(positions) <= window: + return positions.copy() + return savgol_filter(positions, window_length=window, polyorder=3, axis=0, mode="interp") + + +def _savgol_quaternions( + quaternions: np.ndarray, + fps: int, + window_s: float = 0.0, +) -> np.ndarray: + """Smooth sign-aligned quaternion components for sensitivity analysis.""" + q = np.asarray(quaternions, dtype=float).copy() + if window_s <= 0 or len(q) < 3: + return q + norms = np.linalg.norm(q, axis=1) + if not np.isfinite(q).all() or np.any(norms <= 1e-12): + return q + q /= norms[:, None] + for idx in range(1, len(q)): + if np.dot(q[idx - 1], q[idx]) < 0: + q[idx] *= -1 + window = max(5, int(round(window_s * fps))) + if window % 2 == 0: + window += 1 + if len(q) <= window: + return q + q = savgol_filter(q, window_length=window, polyorder=3, axis=0, mode="interp") + norms = np.linalg.norm(q, axis=1) + if np.any(norms <= 1e-12): + return np.asarray(quaternions, dtype=float).copy() + return q / norms[:, None] + + +def _active_motion_bounds( + positions1: np.ndarray, + positions2: np.ndarray, + fps: int, + speed_threshold: float = 2.0, +) -> tuple[int, int]: + """Find the first and last sustained instrument movement.""" + n = min(len(positions1), len(positions2)) + if n < 2 * fps: + return 0, max(0, n - 1) + dt = 1.0 / fps + p1 = _savgol_positions(positions1[:n], fps) + p2 = _savgol_positions(positions2[:n], fps) + speed = np.maximum( + np.linalg.norm(np.gradient(p1, dt, axis=0), axis=1), + np.linalg.norm(np.gradient(p2, dt, axis=0), axis=1), + ) + sustained_frames = max(3, int(round(0.4 * fps))) + active = speed > speed_threshold + runs = np.convolve(active.astype(int), np.ones(sustained_frames, dtype=int), mode="valid") + candidates = np.flatnonzero(runs == sustained_frames) + if not len(candidates): + return 0, n - 1 + buffer_frames = int(round(0.5 * fps)) + start = max(0, int(candidates[0]) - buffer_frames) + end = min(n - 1, int(candidates[-1] + sustained_frames - 1) + buffer_frames) + return start, end + + +def _movement_episodes(speed: np.ndarray, fps: int, threshold: float = 5.0) -> list[tuple[int, int]]: + moving = speed > threshold + minimum = max(2, int(round(0.25 * fps))) + starts = np.flatnonzero(moving & ~np.r_[False, moving[:-1]]) + ends = np.flatnonzero(moving & ~np.r_[moving[1:], False]) + 1 + return [(int(start), int(end)) for start, end in zip(starts, ends) if end - start >= minimum] + + +def _idle_episodes(speed: np.ndarray, fps: int, threshold: float = 5.0) -> list[tuple[int, int]]: + idle = speed < threshold + minimum = max(2, int(round(0.25 * fps))) + starts = np.flatnonzero(idle & ~np.r_[False, idle[:-1]]) + ends = np.flatnonzero(idle & ~np.r_[idle[1:], False]) + 1 + return [(int(start), int(end)) for start, end in zip(starts, ends) if end - start >= minimum] + + +def _angular_metrics(quaternions_wxyz: np.ndarray, fps: int) -> tuple[float, float, float]: + """Return total rotation, mean angular speed, and angular-speed CV.""" + q = np.asarray(quaternions_wxyz, dtype=float) + if len(q) < 3 or q.shape[1] != 4: + return np.nan, np.nan, np.nan + norms = np.linalg.norm(q, axis=1) + if not np.isfinite(q).all() or np.any(norms <= 1e-12): + return np.nan, np.nan, np.nan + q = q / norms[:, None] + rotations = Rotation.from_quat(q[:, [1, 2, 3, 0]]) + relative = rotations[:-1].inv() * rotations[1:] + increments_deg = np.degrees(relative.magnitude()) + angular_speed = increments_deg * fps + active = angular_speed[angular_speed > 0.1] + total_rotation = float(np.sum(increments_deg)) + mean_speed = float(np.mean(angular_speed)) + cv = float(np.std(active, ddof=0) / np.mean(active)) if len(active) > 5 else np.nan + return total_rotation, mean_speed, cv + + +def _tool_motion_features( + positions: np.ndarray, + quaternions: np.ndarray, + fps: int, + prefix: str, + position_window_s: float = 0.4, + orientation_window_s: float = 0.0, +) -> tuple[dict[str, float], np.ndarray]: + dt = 1.0 / fps + pos = _savgol_positions( + np.asarray(positions, dtype=float), + fps, + window_s=position_window_s, + ) + velocity = np.gradient(pos, dt, axis=0) + acceleration = np.gradient(velocity, dt, axis=0) + jerk = np.gradient(acceleration, dt, axis=0) + speed = np.linalg.norm(velocity, axis=1) + acceleration_magnitude = np.linalg.norm(acceleration, axis=1) + jerk_magnitude = np.linalg.norm(jerk, axis=1) + path_length = float(np.linalg.norm(np.diff(pos, axis=0), axis=1).sum()) + duration = float(len(pos) / fps) + elapsed = np.arange(len(pos), dtype=float) * dt + if path_length > 0 and len(pos) > 3: + jerk_integral = float(np.trapezoid(jerk_magnitude**2, elapsed)) + normalised_jerk = float( + np.sqrt(0.5 * jerk_integral * ((len(pos) - 1) * dt) ** 5 / path_length**2) + ) + else: + normalised_jerk = np.nan + ranges = np.ptp(pos, axis=0) + episodes = _movement_episodes(speed, fps) + idle_episodes = _idle_episodes(speed, fps) + movement_durations = [(end - start) / fps for start, end in episodes] + peaks, _ = find_peaks( + speed, + height=5.0, + prominence=2.0, + distance=max(1, int(round(0.25 * fps))), + ) + processed_quaternions = _savgol_quaternions( + quaternions, + fps, + window_s=orientation_window_s, + ) + total_rotation, mean_angular_speed, angular_cv = _angular_metrics( + processed_quaternions, + fps, + ) + features = { + f"{prefix}_total_time": duration, + f"{prefix}_path_length": path_length, + f"{prefix}_path_length_3d": path_length, + f"{prefix}_avg_speed": float(np.mean(speed)), + f"{prefix}_max_speed": float(np.max(speed)), + f"{prefix}_std_speed": float(np.std(speed, ddof=0)), + f"{prefix}_speed_cv": ( + float(np.std(speed, ddof=0) / np.mean(speed)) if np.mean(speed) > 0 else np.nan + ), + f"{prefix}_avg_acceleration": float(np.mean(acceleration_magnitude)), + f"{prefix}_max_acceleration": float(np.max(acceleration_magnitude)), + f"{prefix}_std_acceleration": float(np.std(acceleration_magnitude, ddof=0)), + f"{prefix}_accel_cv": ( + float(np.std(acceleration_magnitude, ddof=0) / np.mean(acceleration_magnitude)) + if np.mean(acceleration_magnitude) > 0 + else np.nan + ), + f"{prefix}_avg_jerk": float(np.mean(jerk_magnitude)), + f"{prefix}_max_jerk": float(np.max(jerk_magnitude)), + f"{prefix}_normalized_jerk": normalised_jerk, + f"{prefix}_working_area_xy": float(ranges[0] * ranges[1]), + f"{prefix}_working_volume": float(np.prod(ranges)), + f"{prefix}_range_x": float(ranges[0]), + f"{prefix}_range_y": float(ranges[1]), + f"{prefix}_range_z": float(ranges[2]), + f"{prefix}_economy_of_motion": ( + float(np.linalg.norm(pos[-1] - pos[0]) / path_length) if path_length > 0 else np.nan + ), + f"{prefix}_idle_time_ratio": float(np.mean(speed < 5.0)), + f"{prefix}_active_time_ratio": float(np.mean(speed >= 5.0)), + f"{prefix}_num_movements": float(len(episodes)), + f"{prefix}_idle_episode_count": float(len(idle_episodes)), + f"{prefix}_avg_movement_duration": ( + float(np.mean(movement_durations)) if movement_durations else 0.0 + ), + f"{prefix}_num_speed_peaks": float(len(peaks)), + f"{prefix}_total_rotation": total_rotation, + f"{prefix}_avg_angular_velocity": mean_angular_speed, + f"{prefix}_angular_velocity_cv": angular_cv, + f"{prefix}_rotation_per_path": ( + float(total_rotation / path_length) + if np.isfinite(total_rotation) and path_length > 0 + else np.nan + ), + f"{prefix}_rotation_economy": ( + float(total_rotation / len(episodes)) + if np.isfinite(total_rotation) and episodes + else np.nan + ), + } + return features, speed + + +def _trial_motion_features( + trial_key_value: str, + dataset: str, + phase_bounds: dict[str, tuple[int, int]], + position_window_s: float = 0.4, + orientation_window_s: float = 0.0, +) -> dict[str, object] | None: + path = ORIGIN_MOTION / f"{trial_key_value}.json" + if not path.exists(): + return None + payload = json.loads(path.read_text()) + raw = np.asarray(payload.get("features", []), dtype=float) + if raw.ndim != 2 or raw.shape[1] < 14 or len(raw) < 20: + return None + fps = _phase_frame_rate(dataset) + pos1, quat1 = raw[:, 0:3], raw[:, 3:7] + pos2, quat2 = raw[:, 7:10], raw[:, 10:14] + if trial_key_value in phase_bounds: + start, end = phase_bounds[trial_key_value] + trim_method = "phase annotation" + else: + start, end = _active_motion_bounds(pos1, pos2, fps) + trim_method = "sustained motion" + start = max(0, min(int(start), len(raw) - 1)) + end = max(start, min(int(end), len(raw) - 1)) + selection = slice(start, end + 1) + pos1, quat1 = pos1[selection], quat1[selection] + pos2, quat2 = pos2[selection], quat2[selection] + if len(pos1) < 20: + return None + + tool1, speed1 = _tool_motion_features( + pos1, + quat1, + fps, + "tool1", + position_window_s=position_window_s, + orientation_window_s=orientation_window_s, + ) + tool2, speed2 = _tool_motion_features( + pos2, + quat2, + fps, + "tool2", + position_window_s=position_window_s, + orientation_window_s=orientation_window_s, + ) + distances = np.linalg.norm( + _savgol_positions(pos1, fps, window_s=position_window_s) + - _savgol_positions(pos2, fps, window_s=position_window_s), + axis=1, + ) + moving1_10 = speed1 > 10.0 + moving2_10 = speed2 > 10.0 + either_10 = moving1_10 | moving2_10 + both_10 = moving1_10 & moving2_10 + moving1_5 = speed1 > 5.0 + moving2_5 = speed2 > 5.0 + both_5 = moving1_5 & moving2_5 + if np.any(both_5): + concurrent_efficiency = float( + np.mean( + np.minimum(speed1[both_5], speed2[both_5]) + / np.maximum(speed1[both_5], speed2[both_5]) + ) + ) + else: + concurrent_efficiency = np.nan + + speed1_standard = (speed1 - np.mean(speed1)) / (np.std(speed1) + 1e-12) + speed2_standard = (speed2 - np.mean(speed2)) / (np.std(speed2) + 1e-12) + cross = correlate(speed1_standard, speed2_standard, mode="full", method="fft") + lags = correlation_lags(len(speed1_standard), len(speed2_standard), mode="full") + lag_limit = 5 * fps + allowed = np.abs(lags) <= lag_limit + best_lag_frames = int(lags[allowed][np.argmax(cross[allowed])]) + path1 = tool1["tool1_path_length"] + path2 = tool2["tool2_path_length"] + result: dict[str, object] = { + "trial_key": trial_key_value, + "trim_start_frame": start, + "trim_end_frame": end, + "trim_method": trim_method, + "n_analysed_frames": int(len(pos1)), + "fps_used": fps, + "total_time": float(len(pos1) / fps), + **tool1, + **tool2, + "bimanual_correlation": _safe_corr(speed1, speed2), + "bimanual_lag_s": float(abs(best_lag_frames) / fps), + "bimanual_symmetry": ( + float(min(path1, path2) / max(path1, path2)) if max(path1, path2) > 0 else np.nan + ), + "tool_distance_avg": float(np.mean(distances)), + "tool_distance_std": float(np.std(distances, ddof=0)), + "tool_distance_cv": ( + float(np.std(distances, ddof=0) / np.mean(distances)) + if np.mean(distances) > 0 + else np.nan + ), + "tool_close_proximity_ratio": float(np.mean(distances < 20.0)), + "simultaneous_motion_ratio": ( + float(np.sum(both_10) / np.sum(either_10)) if np.any(either_10) else 0.0 + ), + "bimanual_concurrent_efficiency": concurrent_efficiency, + "bimanual_dist_speed_corr": _safe_corr(distances, speed1 + speed2), + "combined_idle_ratio": float( + (tool1["tool1_idle_time_ratio"] + tool2["tool2_idle_time_ratio"]) / 2 + ), + "combined_jerk_index": float( + (tool1["tool1_normalized_jerk"] + tool2["tool2_normalized_jerk"]) / 2 + ), + } + return result + + +def extract_primary_kinematics( + canonical: pd.DataFrame, + phase_frames: pd.DataFrame | None, + position_window_s: float = 0.4, + orientation_window_s: float = 0.0, + output_name: str | None = "primary_kinematics.csv", +) -> pd.DataFrame: + counts = canonical.groupby("trial_key")["trial_key"].transform("size") + independent = canonical.loc[counts.eq(1), ["trial_key", "dataset"]].drop_duplicates() + # Use one endpoint rule for every primary recording. Dense labels are + # retained only to validate this automatic interval and for phase analyses. + phase_bounds: dict[str, tuple[int, int]] = {} + rows = [] + for row in independent.itertuples(index=False): + features = _trial_motion_features( + str(row.trial_key), + str(row.dataset), + phase_bounds, + position_window_s=position_window_s, + orientation_window_s=orientation_window_s, + ) + if features is not None: + rows.append(features) + out = pd.DataFrame(rows) + if output_name is not None: + out.to_csv(OUT / output_name, index=False) + return out + + +def load_all_trial_data(phase_frames: pd.DataFrame | None = None) -> pd.DataFrame: + canon = pd.read_csv(BTPN_CACHE / "canonical_trials.csv") + motion = pd.read_csv(AI_ELT / "outputs" / "motion" / "combined_motion_metrics.csv") + trimmed = pd.read_csv(AI_ELT / "outputs" / "papers" / "paper1" / "results" / "extended_features_cache_trimmed.csv") + hand = load_handedness() + + df = canon.merge(motion, on=["dataset", "trial_name"], how="left", validate="one_to_one") + df = df.merge(trimmed, on="trial_name", how="left", validate="one_to_one", suffixes=("", "_trimmed")) + if not hand.empty: + df = df.merge(hand, on=["dataset", "trial_number"], how="left") + else: + df["handedness"] = "unknown" + df["handedness"] = df["handedness"].fillna("unknown") + df["cohort"] = df["dataset"].map( + {"7DOF2024": "urology", "6DOF2023": "urology", "BAPES2024": "paediatric"} + ) + df["cohort_label"] = df["dataset"].map(DATASET_LABELS) + df["trial_key"] = [trial_key(ds, tn) for ds, tn in zip(df["dataset"], df["trial_number"])] + primary = extract_primary_kinematics(df, phase_frames) + df = df.merge(primary, on="trial_key", how="left", suffixes=("", "_primary"), validate="many_to_one") + for col in primary.columns: + if col == "trial_key": + continue + primary_col = f"{col}_primary" if col in canon.columns or col in motion.columns or col in trimmed.columns else col + if primary_col in df.columns: + df[col] = df[primary_col] + if primary_col != col: + df = df.drop(columns=primary_col) + df["dof"] = df["dataset"].map({"7DOF2024": "7 DoF", "BAPES2024": "7 DoF", "6DOF2023": "6 DoF"}) + df["jaw"] = df["dataset"].map({"7DOF2024": "yes", "BAPES2024": "yes", "6DOF2023": "no"}) + df["fps"] = df["dataset"].map({"7DOF2024": 13, "BAPES2024": 13, "6DOF2023": 26}) + df.to_csv(OUT / "all_trial_analysis.csv", index=False) + return df + + +def _phase_frame_rate(dataset: str) -> int: + return 26 if dataset == "6DOF2023" else 13 + + +def _trim_phase_frames_to_task(frames: pd.DataFrame) -> pd.DataFrame: + """Retain the labelled task interval from first to last non-idle frame.""" + kept = [] + group_cols = ["dataset", "trial_id", "ann_dir"] + for _, group in frames.sort_values(group_cols + ["frame_idx"]).groupby(group_cols, sort=False): + active = ~group["coarse_derived"].isin(["idle", "nan", "None"]) + if not active.any(): + continue + active_idx = np.flatnonzero(active.to_numpy()) + trimmed = group.iloc[active_idx[0] : active_idx[-1] + 1].copy() + fps = _phase_frame_rate(str(trimmed["dataset"].iloc[0])) + trimmed["time_s"] = np.arange(len(trimmed), dtype=float) / fps + kept.append(trimmed) + return pd.concat(kept, ignore_index=True) if kept else frames.iloc[0:0].copy() + + +def _rebuild_completed_cycles(frames: pd.DataFrame) -> pd.DataFrame: + """Retain phase-defined cycles that reach a placement endpoint. + + A cycle is complete only when it contains at least one placement frame and + at least one task-action frame (reach, grasp, transfer, or nudge). The + exporter can put placement-only boundary frames into the following cycle + index. Such a segment is attached to the immediately preceding interval + only when that interval has no placement. Other placement-only residuals + and attempts without placement are excluded. + """ + rows: list[dict[str, object]] = [] + excluded_rows: list[dict[str, object]] = [] + group_cols = ["dataset", "trial_id", "ann_dir"] + for _, group in frames.sort_values(group_cols + ["frame_idx"]).groupby(group_cols, sort=False): + dataset = str(group["dataset"].iloc[0]) + fps = _phase_frame_rate(dataset) + cycle_indices = sorted( + pd.to_numeric(group["cycle_index"], errors="coerce") + .dropna() + .astype(int) + .unique() + ) + if not cycle_indices: + continue + cycle_frames = { + cycle_index: group[group["cycle_index"] == cycle_index].copy() + for cycle_index in cycle_indices + } + for position, cycle_index in enumerate(cycle_indices): + cycle = cycle_frames[cycle_index] + labels = set(cycle["coarse_derived"].astype(str)) + substantive = labels - {"idle", "place"} + if "place" not in labels or substantive: + continue + previous = [ + index + for index in cycle_indices[:position] + if not cycle_frames[index].empty + ] + if previous: + previous_index = previous[-1] + if not cycle_frames[previous_index]["coarse_derived"].eq("place").any(): + cycle_frames[previous_index] = ( + pd.concat( + [cycle_frames[previous_index], cycle], + ignore_index=True, + ) + .sort_values("frame_idx") + .reset_index(drop=True) + ) + cycle_frames[cycle_index] = cycle.iloc[0:0].copy() + + for cycle_index in cycle_indices: + cycle = cycle_frames[cycle_index] + if cycle.empty: + continue + labels = cycle["coarse_derived"].astype(str) + has_placement = bool(labels.eq("place").any()) + has_task_action = bool( + labels.isin(["reach", "grasp", "transfer", "nudge"]).any() + ) + if not (has_placement and has_task_action): + excluded_rows.append( + { + "dataset": dataset, + "trial_id": group["trial_id"].iloc[0], + "ann_dir": group["ann_dir"].iloc[0], + "cycle_index": cycle_index, + "n_frames": int(len(cycle)), + "has_placement": has_placement, + "has_task_action": has_task_action, + "phase_sequence": ",".join(labels.drop_duplicates()), + } + ) + continue + row: dict[str, object] = { + "dataset": dataset, + "trial_id": group["trial_id"].iloc[0], + "ann_dir": group["ann_dir"].iloc[0], + "cycle_index": cycle_index, + "duration_s": float(len(cycle) / fps), + "dominant_phase": labels.value_counts().idxmax(), + "n_transitions": int(labels.ne(labels.shift()).sum() - 1), + "n_dropped_frames": int(labels.eq("dropped").sum()), + "n_drop_events": int( + (labels.eq("dropped") & ~labels.shift(fill_value="").eq("dropped")).sum() + ), + } + for phase in PHASE_ORDER: + row[f"frac_{phase}"] = float(labels.eq(phase).mean()) + rows.append(row) + pd.DataFrame(excluded_rows).to_csv( + OUT / "phase_cycle_exclusions.csv", + index=False, + ) + return pd.DataFrame(rows) + + +def _rebuild_phase_trials( + source_trials: pd.DataFrame, + frames: pd.DataFrame, + cycles: pd.DataFrame, +) -> pd.DataFrame: + rebuilt = [] + phase_cols = [f"frac_{phase}" for phase in PHASE_ORDER] + for (dataset, trial_id), group in frames.groupby(["dataset", "trial_id"], sort=False): + source = source_trials[ + (source_trials["dataset"] == dataset) & (source_trials["trial_id"] == trial_id) + ].iloc[0].to_dict() + fps = _phase_frame_rate(str(dataset)) + labels = group["coarse_derived"].astype(str) + trial_cycles = cycles[ + (cycles["dataset"] == dataset) & (cycles["trial_id"] == trial_id) + ] + durations = trial_cycles["duration_s"].to_numpy(float) + source.update( + { + "n_frames": int(len(group)), + "total_s": float(len(group) / fps), + "n_cycles": int(len(trial_cycles)), + "min_cycle_s": float(np.min(durations)) if len(durations) else np.nan, + "max_cycle_s": float(np.max(durations)) if len(durations) else np.nan, + "mean_cycle_s": float(np.mean(durations)) if len(durations) else np.nan, + "std_cycle_s": float(np.std(durations, ddof=0)) if len(durations) else np.nan, + } + ) + for col in phase_cols: + source[col] = float(labels.eq(col.replace("frac_", "")).mean()) + rebuilt.append(source) + return pd.DataFrame(rebuilt) + + +def load_phase_data() -> tuple[pd.DataFrame, pd.DataFrame, pd.DataFrame]: + source_trials = pd.read_parquet(BTPN_CACHE / "trials.parquet") + frames = pd.read_parquet(BTPN_CACHE / "frames.parquet") + meta = source_trials[["dataset", "trial_id", "trial_short", "skill_category", "total_procedures"]] + frames = frames.merge(meta, on=["dataset", "trial_id", "trial_short"], how="left") + frames["exported_max_cycle_index"] = frames.groupby( + ["dataset", "trial_id", "ann_dir"] + )["cycle_index"].transform("max") + frames = _trim_phase_frames_to_task(frames) + cycles = _rebuild_completed_cycles(frames) + trials = _rebuild_phase_trials(source_trials, frames, cycles) + cycles = cycles.merge(meta, on=["dataset", "trial_id"], how="left") + trials["trial_key"] = [ + phase_trial_key(dataset, short) + for dataset, short in zip(trials["dataset"], trials["trial_short"]) + ] + cycles["trial_key"] = [ + phase_trial_key(dataset, short) + for dataset, short in zip(cycles["dataset"], cycles["trial_short"]) + ] + frames["trial_key"] = [ + phase_trial_key(dataset, short) + for dataset, short in zip(frames["dataset"], frames["trial_short"]) + ] + trials.to_csv(OUT / "phase_trials_current.csv", index=False) + cycles.to_csv(OUT / "phase_cycles_current.csv", index=False) + return trials, cycles, frames + + +def bootstrap_ci(values: Iterable[float], func=np.mean, n_boot: int = 3000) -> tuple[float, float]: + x = np.asarray([v for v in values if np.isfinite(v)], dtype=float) + if len(x) == 0: + return (np.nan, np.nan) + if len(x) == 1: + return (float(x[0]), float(x[0])) + rng = np.random.default_rng(20260618) + boots = [func(rng.choice(x, size=len(x), replace=True)) for _ in range(n_boot)] + return tuple(np.percentile(boots, [2.5, 97.5])) + + +def bootstrap_spearman_ci( + x_values: Iterable[float], + y_values: Iterable[float], + n_boot: int = 3000, +) -> tuple[float, float]: + """Paired recording-level percentile interval for Spearman correlation.""" + pairs = pd.DataFrame({"x": x_values, "y": y_values}).replace( + [np.inf, -np.inf], + np.nan, + ).dropna() + if len(pairs) < 3: + return (np.nan, np.nan) + x = pairs["x"].to_numpy(float) + y = pairs["y"].to_numpy(float) + rng = np.random.default_rng(20260618) + estimates = [] + for _ in range(n_boot): + idx = rng.choice(np.arange(len(pairs)), size=len(pairs), replace=True) + if np.unique(x[idx]).size < 2 or np.unique(y[idx]).size < 2: + continue + estimates.append(stats.spearmanr(x[idx], y[idx]).statistic) + if not estimates: + return (np.nan, np.nan) + return tuple(np.nanpercentile(estimates, [2.5, 97.5])) + + +def cliffs_delta(a: Iterable[float], b: Iterable[float]) -> float: + x = np.asarray([v for v in a if np.isfinite(v)], dtype=float) + y = np.asarray([v for v in b if np.isfinite(v)], dtype=float) + if len(x) == 0 or len(y) == 0: + return np.nan + gt = sum((v > y).sum() for v in x) + lt = sum((v < y).sum() for v in x) + return (gt - lt) / (len(x) * len(y)) + + +def benjamini_hochberg(p_values: Iterable[float]) -> np.ndarray: + """Benjamini-Hochberg false-discovery-rate correction.""" + p = np.asarray(list(p_values), dtype=float) + q = np.full_like(p, np.nan, dtype=float) + finite = np.isfinite(p) + if not finite.any(): + return q + idx = np.where(finite)[0] + order = idx[np.argsort(p[idx])] + ranked = p[order] + m = len(ranked) + adjusted = ranked * m / np.arange(1, m + 1) + adjusted = np.minimum.accumulate(adjusted[::-1])[::-1] + q[order] = np.minimum(adjusted, 1.0) + return q + + +def zscore_within_dataset(df: pd.DataFrame, columns: list[str]) -> pd.DataFrame: + out = df.copy() + for col in columns: + vals = [] + for _, g in df.groupby("dataset"): + x = g[col].astype(float) + sd = x.std(ddof=0) + vals.append((x - x.mean()) / (sd if sd and np.isfinite(sd) else 1.0)) + out[col + "_z"] = pd.concat(vals).sort_index() + return out + + +def feature_statistics(df: pd.DataFrame) -> pd.DataFrame: + cols = [ + c + for c in FEATURES + if c in df.columns + and pd.to_numeric(df[c], errors="coerce").dropna().nunique() > 1 + ] + identifier_counts = df.groupby("trial_key")["trial_key"].transform("size") + independent_df = df.loc[identifier_counts.eq(1)].copy() + records: list[dict[str, object]] = [] + rng = np.random.default_rng(20260618) + for col in cols: + cohort_deltas = [] + cohort_delta_weights = [] + cohort_kruskal_p = [] + cohort_rho_z = [] + cohort_rho_weights = [] + for _, cohort in independent_df.groupby("dataset"): + cohort = cohort[ + ["skill_category", "total_procedures", col] + ].replace([np.inf, -np.inf], np.nan).dropna() + novice = cohort.loc[ + cohort["skill_category"] == "novice", col + ].to_numpy(float) + expert = cohort.loc[ + cohort["skill_category"] == "expert", col + ].to_numpy(float) + if len(novice) and len(expert): + cohort_deltas.append(cliffs_delta(novice, expert)) + cohort_delta_weights.append(len(novice) * len(expert)) + groups = [ + cohort.loc[cohort["skill_category"] == skill, col].to_numpy(float) + for skill in SKILL_ORDER + ] + groups = [group for group in groups if len(group)] + if len(groups) >= 2 and np.unique(np.concatenate(groups)).size > 1: + cohort_kruskal_p.append(float(stats.kruskal(*groups).pvalue)) + if ( + len(cohort) >= 4 + and cohort[col].nunique() > 1 + and cohort["total_procedures"].nunique() > 1 + ): + rho = float( + stats.spearmanr( + cohort[col], + cohort["total_procedures"], + ).statistic + ) + cohort_rho_z.append(np.arctanh(np.clip(rho, -0.999999, 0.999999))) + cohort_rho_weights.append(max(len(cohort) - 3, 1)) + + delta = ( + float(np.average(cohort_deltas, weights=cohort_delta_weights)) + if cohort_deltas + else np.nan + ) + boot = [] + if cohort_deltas: + for _ in range(3000): + sampled_deltas = [] + sampled_weights = [] + for _, cohort in independent_df.groupby("dataset"): + novice = cohort.loc[ + cohort["skill_category"] == "novice", col + ].dropna().to_numpy(float) + expert = cohort.loc[ + cohort["skill_category"] == "expert", col + ].dropna().to_numpy(float) + if len(novice) and len(expert): + sampled_deltas.append( + cliffs_delta( + rng.choice(novice, size=len(novice), replace=True), + rng.choice(expert, size=len(expert), replace=True), + ) + ) + sampled_weights.append(len(novice) * len(expert)) + boot.append(np.average(sampled_deltas, weights=sampled_weights)) + lo, hi = np.nanpercentile(boot, [2.5, 97.5]) + else: + lo = hi = np.nan + + if cohort_kruskal_p: + h = float( + -2 + * np.sum( + np.log(np.clip(cohort_kruskal_p, np.finfo(float).tiny, 1.0)) + ) + ) + p = float( + stats.chi2.sf(h, 2 * len(cohort_kruskal_p)) + ) + else: + h = p = np.nan + + if cohort_rho_z: + weights = np.asarray(cohort_rho_weights, dtype=float) + z_values = np.asarray(cohort_rho_z, dtype=float) + z_meta = float(np.average(z_values, weights=weights)) + z_se = float(1.0 / np.sqrt(weights.sum())) + rho = float(np.tanh(z_meta)) + rho_p = float(2 * stats.norm.sf(abs(z_meta / z_se))) + rho_lo, rho_hi = np.tanh( + [z_meta - 1.96 * z_se, z_meta + 1.96 * z_se] + ) + heterogeneity_q = float(np.sum(weights * (z_values - z_meta) ** 2)) + heterogeneity_df = max(len(z_values) - 1, 0) + heterogeneity_i2 = ( + float( + max( + 0.0, + 100.0 + * (heterogeneity_q - heterogeneity_df) + / heterogeneity_q, + ) + ) + if heterogeneity_q > 0 and heterogeneity_df > 0 + else 0.0 + ) + else: + rho = rho_p = rho_lo = rho_hi = np.nan + heterogeneity_q = heterogeneity_i2 = np.nan + records.append( + { + "feature": col, + "label": FEATURES[col], + "cliffs_delta_novice_expert": delta, + "delta_ci_low": lo, + "delta_ci_high": hi, + "kruskal_h": h, + "kruskal_p": p, + "spearman_rho_procedures": rho, + "spearman_p": rho_p, + "spearman_ci_low": float(rho_lo), + "spearman_ci_high": float(rho_hi), + "spearman_meta_i2": heterogeneity_i2, + "n_cohorts": int(len(cohort_rho_z)), + "n_nonmissing": int(independent_df[col].notna().sum()), + } + ) + res = pd.DataFrame(records) + res["kruskal_q"] = benjamini_hochberg(res["kruskal_p"]) + res["spearman_q"] = benjamini_hochberg(res["spearman_p"]) + res = res.sort_values("cliffs_delta_novice_expert") + res.to_csv(OUT / "all_trial_feature_statistics.csv", index=False) + return res + + +def ols_fit(X: np.ndarray, y: np.ndarray) -> tuple[np.ndarray, float]: + beta = np.linalg.lstsq(X, y, rcond=None)[0] + pred = X @ beta + ss_res = float(np.sum((y - pred) ** 2)) + ss_tot = float(np.sum((y - y.mean()) ** 2)) + r2 = 1.0 - ss_res / ss_tot if ss_tot > 0 else np.nan + return beta, r2 + + +def cohort_adjusted_models(df: pd.DataFrame, proxy: pd.DataFrame) -> pd.DataFrame: + """Cohort-adjusted association between experience volume and key outcomes.""" + use_df = df.copy() + use_df["Coordination-control composite"] = ( + proxy["Coordination-control composite"].to_numpy() if len(proxy) == len(df) else np.nan + ) + identifier_counts = use_df.groupby("trial_key")["trial_key"].transform("size") + use_df = use_df.loc[identifier_counts.eq(1)].copy() + outcomes = [ + "total_time", + "bimanual_correlation", + "tool1_avg_speed", + "tool2_avg_speed", + "combined_idle_ratio", + "tool1_range_z", + "tool2_range_z", + "Coordination-control composite", + ] + rows = [] + rng = np.random.default_rng(20260618) + for outcome in outcomes: + if outcome not in use_df.columns: + continue + d = use_df[["dataset", "total_procedures", outcome]].replace([np.inf, -np.inf], np.nan).dropna() + if len(d) < 20: + continue + y_raw = d[outcome].to_numpy(float) + y_sd = y_raw.std(ddof=0) + if not np.isfinite(y_sd) or y_sd == 0: + continue + y = (y_raw - y_raw.mean()) / y_sd + log_proc = np.log10(d["total_procedures"].to_numpy(float) + 1) + cohort = pd.get_dummies(d["dataset"], drop_first=True).to_numpy(float) + X_full = np.column_stack([np.ones(len(d)), log_proc, cohort]) + X_proc = np.column_stack([np.ones(len(d)), log_proc]) + X_cohort = np.column_stack([np.ones(len(d)), cohort]) + beta, r2_full = ols_fit(X_full, y) + _, r2_proc = ols_fit(X_proc, y) + _, r2_cohort = ols_fit(X_cohort, y) + boot = [] + for _ in range(2000): + idx = np.concatenate( + [ + rng.choice( + cohort_idx, + size=len(cohort_idx), + replace=True, + ) + for dataset in d["dataset"].unique() + for cohort_idx in [ + np.flatnonzero(d["dataset"].to_numpy() == dataset) + ] + ] + ) + b, _ = ols_fit(X_full[idx], y[idx]) + boot.append(b[1]) + lo, hi = np.nanpercentile(boot, [2.5, 97.5]) + permuted = [] + dataset_values = d["dataset"].to_numpy() + for _ in range(5000): + permuted_log_proc = log_proc.copy() + for dataset in d["dataset"].unique(): + idx = np.flatnonzero(dataset_values == dataset) + permuted_log_proc[idx] = rng.permutation(log_proc[idx]) + X_permuted = np.column_stack( + [np.ones(len(d)), permuted_log_proc, cohort] + ) + permuted.append(ols_fit(X_permuted, y)[0][1]) + procedure_p = float( + ( + 1 + + np.sum( + np.abs(np.asarray(permuted, dtype=float)) + >= abs(float(beta[1])) + ) + ) + / (len(permuted) + 1) + ) + rows.append( + { + "outcome": outcome, + "label": FEATURES.get(outcome, outcome), + "n": int(len(d)), + "beta_log10_procedures": float(beta[1]), + "ci_low": float(lo), + "ci_high": float(hi), + "procedure_p": procedure_p, + "r2_procedure_only": float(r2_proc), + "r2_cohort_only": float(r2_cohort), + "r2_full": float(r2_full), + "cohort_delta_r2_after_procedure": float(r2_full - r2_proc), + } + ) + out = pd.DataFrame(rows) + if not out.empty: + out["procedure_q"] = benjamini_hochberg(out["procedure_p"]) + out.to_csv(OUT / "cohort_adjusted_models.csv", index=False) + return out + + +def write_feature_dictionary(df: pd.DataFrame) -> pd.DataFrame: + domain_by_feature: dict[str, list[str]] = {} + for domain, specs in OSATS_DOMAINS.items(): + for feature, _ in specs: + domain_by_feature.setdefault(feature, []).append(domain) + rows = [] + for feature, label in FEATURES.items(): + if feature not in df.columns: + continue + direction, meaning = FEATURE_DIRECTIONS.get(feature, ("", "")) + nonmissing = int(df[feature].notna().sum()) + cohorts = [ + dataset_label(ds) + for ds in DATASET_ORDER + if ds in set(df.loc[df[feature].notna(), "dataset"].astype(str)) + ] + rows.append( + { + "feature": feature, + "label": label, + "favourable_direction": direction.replace("$", "").replace("\\", ""), + "interpretation": meaning, + "domain": "; ".join(domain_by_feature.get(feature, [])), + "available_cohorts": "; ".join(cohorts), + "nonmissing_trials": nonmissing, + "missing_trials": int(len(df) - nonmissing), + } + ) + out = pd.DataFrame(rows) + out.to_csv(OUT / "feature_dictionary.csv", index=False) + return out + + +def run_classification(df: pd.DataFrame) -> dict[str, object]: + try: + from sklearn.ensemble import RandomForestClassifier + from sklearn.impute import SimpleImputer + from sklearn.linear_model import LogisticRegression + from sklearn.metrics import balanced_accuracy_score + from sklearn.model_selection import LeaveOneGroupOut, StratifiedGroupKFold, cross_val_score + from sklearn.pipeline import make_pipeline + from sklearn.preprocessing import StandardScaler + except Exception as exc: # pragma: no cover + return {"available": False, "error": str(exc)} + + all_cols = [c for c in FEATURES if c in df.columns] + time_cols = ["total_time"] if "total_time" in df.columns else [] + identifier_counts = df.groupby("trial_key")["trial_key"].transform("size") + use = df.loc[identifier_counts.eq(1)].dropna( + subset=["skill_category", "dataset", "total_time"] + ).copy() + y = use["skill_category"].astype(str).to_numpy() + groups = use["dataset"].astype(str).to_numpy() + participant_groups = use["trial_key"].astype(str).to_numpy() + out: dict[str, object] = {"available": True, "features": all_cols, "time_only_features": time_cols} + + base_models = { + "logistic_regression": make_pipeline( + SimpleImputer(strategy="median"), + StandardScaler(), + LogisticRegression(max_iter=500, class_weight="balanced", random_state=13), + ), + "random_forest": make_pipeline( + SimpleImputer(strategy="median"), + RandomForestClassifier(n_estimators=400, class_weight="balanced", random_state=13), + ), + } + model_specs = { + "time_only_logistic_regression": (base_models["logistic_regression"], time_cols), + "time_only_random_forest": (base_models["random_forest"], time_cols), + "all_features_logistic_regression": (base_models["logistic_regression"], all_cols), + "all_features_random_forest": (base_models["random_forest"], all_cols), + } + skf = StratifiedGroupKFold(n_splits=5, shuffle=True, random_state=13) + cv = {} + for name, (model, cols) in model_specs.items(): + if not cols: + continue + X = use[cols].replace([np.inf, -np.inf], np.nan) + scores = cross_val_score( + model, + X, + y, + cv=skf, + groups=participant_groups, + scoring="balanced_accuracy", + ) + cv[name] = {"mean_balanced_accuracy": float(scores.mean()), "sd": float(scores.std()), "features": cols} + out["stratified_group_5fold"] = cv + + logo = LeaveOneGroupOut() + lodo: dict[str, list[dict[str, object]]] = {} + for name, (model, cols) in model_specs.items(): + if not cols: + continue + X = use[cols].replace([np.inf, -np.inf], np.nan) + rows = [] + for train, test in logo.split(X, y, groups): + model.fit(X.iloc[train], y[train]) + pred = model.predict(X.iloc[test]) + rows.append( + { + "held_out_dataset": str(groups[test][0]), + "balanced_accuracy": float(balanced_accuracy_score(y[test], pred)), + "n_test": int(len(test)), + } + ) + lodo[name] = rows + out["leave_one_dataset_out"] = lodo + + pd.DataFrame(columns=["feature", "label", "importance"]).to_csv( + OUT / "skill_feature_importance.csv", + index=False, + ) + out["top_random_forest_features"] = [] + (OUT / "classification_summary.json").write_text(json.dumps(out, indent=2)) + return out + + +def run_metric_band_classification(clusters: pd.DataFrame) -> dict[str, object]: + try: + from sklearn.cluster import KMeans + from sklearn.ensemble import RandomForestClassifier + from sklearn.impute import SimpleImputer + from sklearn.linear_model import LogisticRegression + from sklearn.metrics import balanced_accuracy_score + from sklearn.model_selection import LeaveOneGroupOut, StratifiedGroupKFold, cross_val_score + from sklearn.pipeline import make_pipeline + from sklearn.preprocessing import StandardScaler + except Exception as exc: # pragma: no cover + return {"available": False, "error": str(exc)} + + if clusters.empty or "metric_cluster" not in clusters: + return {"available": False, "error": "metric clusters unavailable"} + feature_cols = [ + c + for c in [*CORE_PERFORMANCE_DOMAINS, "Coordination-control composite"] + if c in clusters.columns + ] + use = clusters.dropna(subset=["metric_cluster", "dataset", "total_procedures"]).copy() + if not feature_cols or use["metric_cluster"].nunique() < 2: + return {"available": False, "error": "insufficient metric bands"} + + X = use[feature_cols].replace([np.inf, -np.inf], np.nan) + groups = use["dataset"].astype(str).to_numpy() + participant_groups = use["trial_key"].astype(str).to_numpy() + + proc = use["total_procedures"].astype(float) + use["procedure_fixed_band"] = np.select( + [proc < 20, proc < 100], + ["procedure_novice", "procedure_intermediate"], + default="procedure_expert", + ) + ranked_proc = proc.rank(method="first") + use["procedure_tertile_band"] = pd.qcut( + ranked_proc, + q=3, + labels=["procedure_tertile_low", "procedure_tertile_mid", "procedure_tertile_high"], + ).astype(str) + proc_x = np.log10(proc.to_numpy().reshape(-1, 1) + 1) + if len(use) >= 6: + km = KMeans(n_clusters=3, n_init=200, random_state=29) + raw = km.fit_predict(proc_x) + proc_tmp = pd.DataFrame({"raw": raw, "proc": proc.to_numpy()}) + ordered = proc_tmp.groupby("raw")["proc"].median().sort_values().index.to_list() + proc_map = { + ordered[0]: "procedure_kmeans_low", + ordered[1]: "procedure_kmeans_mid", + ordered[2]: "procedure_kmeans_high", + } + use["procedure_kmeans_band"] = [proc_map[v] for v in raw] + else: + use["procedure_kmeans_band"] = use["procedure_tertile_band"] + + model_factories = { + "logistic_regression": lambda: make_pipeline( + SimpleImputer(strategy="median"), + StandardScaler(), + LogisticRegression(max_iter=500, class_weight="balanced", random_state=29), + ), + "random_forest": lambda: make_pipeline( + SimpleImputer(strategy="median"), + RandomForestClassifier(n_estimators=500, class_weight="balanced", random_state=29), + ), + } + + targets = { + "motion_defined": { + "label": "Motion-score bands", + "column": "metric_cluster", + }, + "procedure_fixed": { + "label": "Procedure thresholds", + "column": "procedure_fixed_band", + }, + "procedure_tertiles": { + "label": "Procedure tertiles", + "column": "procedure_tertile_band", + }, + "procedure_kmeans": { + "label": "Procedure K-means", + "column": "procedure_kmeans_band", + }, + } + + results: list[dict[str, object]] = [] + details: dict[str, object] = {} + logo = LeaveOneGroupOut() + for target_key, target in targets.items(): + y = use[target["column"]].astype(str).to_numpy() + class_counts = pd.Series(y).value_counts() + if len(class_counts) < 2 or class_counts.min() < 2: + details[target_key] = {"available": False, "class_counts": class_counts.to_dict()} + continue + n_splits = max(2, min(5, int(class_counts.min()))) + skf = StratifiedGroupKFold(n_splits=n_splits, shuffle=True, random_state=29) + details[target_key] = { + "available": True, + "label": target["label"], + "class_counts": class_counts.to_dict(), + "models": {}, + } + for model_key, factory in model_factories.items(): + model = factory() + scores = cross_val_score( + model, + X, + y, + cv=skf, + groups=participant_groups, + scoring="balanced_accuracy", + ) + lodo_rows = [] + for train, test in logo.split(X, y, groups): + model = factory() + model.fit(X.iloc[train], y[train]) + pred = model.predict(X.iloc[test]) + lodo_rows.append( + { + "held_out_dataset": str(groups[test][0]), + "balanced_accuracy": float(balanced_accuracy_score(y[test], pred)), + "n_test": int(len(test)), + } + ) + mean_lodo = float(np.mean([r["balanced_accuracy"] for r in lodo_rows])) + min_lodo = float(np.min([r["balanced_accuracy"] for r in lodo_rows])) + max_lodo = float(np.max([r["balanced_accuracy"] for r in lodo_rows])) + model_label = "Logistic regression" if model_key == "logistic_regression" else "Random forest" + results.append( + { + "target": target_key, + "target_label": target["label"], + "model": model_key, + "model_label": model_label, + "stratified_cv_balanced_accuracy": float(scores.mean()), + "stratified_cv_sd": float(scores.std()), + "leave_one_dataset_out_balanced_accuracy": mean_lodo, + "leave_one_dataset_out_min": min_lodo, + "leave_one_dataset_out_max": max_lodo, + "n_splits": int(n_splits), + } + ) + details[target_key]["models"][model_key] = { + "stratified_cv": { + "mean_balanced_accuracy": float(scores.mean()), + "sd": float(scores.std()), + "n_splits": int(n_splits), + }, + "leave_one_dataset_out": lodo_rows, + } + + result_df = pd.DataFrame(results) + result_df.to_csv(OUT / "metric_band_classification_comparison.csv", index=False) + legacy = {} + if not result_df.empty: + motion = result_df[result_df["target"] == "motion_defined"] + legacy["stratified_cv"] = { + f"metric_band_{r['model']}": { + "mean_balanced_accuracy": float(r["stratified_cv_balanced_accuracy"]), + "sd": float(r["stratified_cv_sd"]), + "n_splits": int(r["n_splits"]), + } + for _, r in motion.iterrows() + } + legacy["leave_one_dataset_out"] = { + f"metric_band_{model_key}": details["motion_defined"]["models"][model_key]["leave_one_dataset_out"] + for model_key in ["logistic_regression", "random_forest"] + if "motion_defined" in details and details["motion_defined"].get("available") and model_key in details["motion_defined"]["models"] + } + + out = {"available": True, "features": feature_cols, "targets": details, "comparison_rows": results, **legacy} + (OUT / "metric_band_classification_summary.json").write_text(json.dumps(out, indent=2)) + return out + + +def oriented_percentile_scores(df: pd.DataFrame, specs: list[tuple[str, int]]) -> pd.Series: + parts = [] + for col, direction in specs: + if col not in df.columns: + continue + x = pd.to_numeric(df[col], errors="coerce") * direction + scored = [] + for _, g in df.assign(_x=x).groupby("dataset"): + ranks = g["_x"].rank(pct=True) + scored.append(ranks) + parts.append(pd.concat(scored).sort_index()) + if not parts: + return pd.Series(np.nan, index=df.index) + score01 = pd.concat(parts, axis=1).mean(axis=1, skipna=True) + return 1.0 + 4.0 * score01 + + +def construct_balanced_domain_score( + df: pd.DataFrame, + constructs: list[list[tuple[str, int]]], + reference_mask: pd.Series, +) -> pd.Series: + construct_scores = [] + for specs in constructs: + feature_scores = [] + for col, direction in specs: + if col not in df.columns: + continue + x = pd.to_numeric(df[col], errors="coerce") * direction + ranked = pd.Series(np.nan, index=df.index, dtype=float) + reference = df.loc[reference_mask] + for _, idx in reference.groupby("dataset").groups.items(): + ranked.loc[idx] = x.loc[idx].rank(pct=True) + feature_scores.append(ranked) + if len(feature_scores) == len(specs): + construct_scores.append(pd.concat(feature_scores, axis=1).mean(axis=1, skipna=False)) + if not construct_scores: + return pd.Series(np.nan, index=df.index) + score01 = pd.concat(construct_scores, axis=1).mean(axis=1, skipna=False) + return 1.0 + 4.0 * score01 + + +def osats_proxy_scores(df: pd.DataFrame) -> pd.DataFrame: + out = df[["dataset", "cohort_label", "trial_name", "trial_number", "trial_key", "skill_category", "total_procedures"]].copy() + identifier_counts = df.groupby("trial_key")["trial_key"].transform("size") + reference_mask = identifier_counts.eq(1) & df["total_time"].notna() + out["Analysed duration (s)"] = pd.to_numeric(df["total_time"], errors="coerce").where(reference_mask) + for domain, constructs in OSATS_DOMAIN_CONSTRUCTS.items(): + out[domain] = construct_balanced_domain_score(df, constructs, reference_mask) + domain_cols = list(OSATS_DOMAINS) + out["Coordination-control composite"] = out[CORE_PERFORMANCE_DOMAINS].mean( + axis=1, + skipna=False, + ) + out.to_csv(OUT / "osats_aligned_proxy_scores.csv", index=False) + + rows = [] + for skill, g in out.groupby("skill_category"): + for col in domain_cols + ["Coordination-control composite"]: + lo, hi = bootstrap_ci(g[col].dropna()) + rows.append( + { + "skill_category": skill, + "domain": col, + "n": int(g[col].notna().sum()), + "mean_score": float(g[col].mean()), + "ci_low": lo, + "ci_high": hi, + } + ) + pd.DataFrame(rows).to_csv(OUT / "osats_proxy_by_skill.csv", index=False) + return out + + +def processing_sensitivity( + df: pd.DataFrame, + phase_frames: pd.DataFrame, + baseline_proxy: pd.DataFrame, + baseline_clusters: pd.DataFrame, + cluster_summary: dict[str, object], +) -> pd.DataFrame: + """Compare the primary score under alternative position and orientation processing.""" + from sklearn.cluster import KMeans + from sklearn.metrics import adjusted_rand_score + + baseline = baseline_proxy[ + ["trial_key", "Coordination-control composite"] + ].dropna( + subset=["Coordination-control composite"] + ).merge( + baseline_clusters[ + ["trial_key", "metric_cluster"] + ].dropna(subset=["metric_cluster"]), + on="trial_key", + how="left", + validate="one_to_one", + ) + baseline = baseline.dropna( + subset=["Coordination-control composite", "metric_cluster"] + ).copy() + baseline["baseline_label"] = baseline["metric_cluster"].map( + {band: i for i, band in enumerate(BAND_ORDER)} + ) + thresholds = np.asarray( + cluster_summary.get("metric_band_thresholds", []), + dtype=float, + ) + if len(baseline) == 0 or len(thresholds) != 2: + return pd.DataFrame() + + rows = [] + for position_window_s, orientation_window_s, label in [ + (0.0, 0.0, "No position smoothing"), + (0.4, 0.0, "Primary processing"), + (0.8, 0.0, "0.8 s position window"), + (0.4, 0.4, "0.4 s orientation sensitivity"), + ]: + if position_window_s == 0.4 and orientation_window_s == 0.0: + alternative = baseline[ + ["trial_key", "Coordination-control composite"] + ].copy() + else: + motion = extract_primary_kinematics( + df, + phase_frames, + position_window_s=position_window_s, + orientation_window_s=orientation_window_s, + output_name=None, + ) + replace_cols = [c for c in motion.columns if c != "trial_key"] + alternative_df = df.drop(columns=replace_cols, errors="ignore").merge( + motion, + on="trial_key", + how="left", + validate="many_to_one", + ) + identifier_counts = alternative_df.groupby("trial_key")[ + "trial_key" + ].transform("size") + reference_mask = identifier_counts.eq(1) & alternative_df[ + "total_time" + ].notna() + alternative = alternative_df[["trial_key"]].copy() + for domain in CORE_PERFORMANCE_DOMAINS: + alternative[domain] = construct_balanced_domain_score( + alternative_df, + OSATS_DOMAIN_CONSTRUCTS[domain], + reference_mask, + ) + alternative["Coordination-control composite"] = alternative[ + CORE_PERFORMANCE_DOMAINS + ].mean(axis=1, skipna=False) + + compared = baseline.merge( + alternative[ + ["trial_key", "Coordination-control composite"] + ].dropna( + subset=["Coordination-control composite"] + ).rename( + columns={ + "Coordination-control composite": "alternative_score" + } + ), + on="trial_key", + how="inner", + validate="one_to_one", + ).dropna(subset=["alternative_score"]) + if compared.empty: + continue + rho = stats.spearmanr( + compared["Coordination-control composite"], + compared["alternative_score"], + ).statistic + fixed_label = np.digitize( + compared["alternative_score"].to_numpy(float), + thresholds, + ) + fixed_ari = adjusted_rand_score( + compared["baseline_label"].to_numpy(int), + fixed_label, + ) + values = compared["alternative_score"].to_numpy(float).reshape(-1, 1) + fitted = KMeans( + n_clusters=3, + n_init=300, + random_state=20260618, + ).fit(values) + centres = np.sort(fitted.cluster_centers_.ravel()) + refit_thresholds = (centres[:-1] + centres[1:]) / 2 + refit_label = np.digitize(values.ravel(), refit_thresholds) + refit_ari = adjusted_rand_score( + compared["baseline_label"].to_numpy(int), + refit_label, + ) + rows.append( + { + "processing": label, + "position_window_s": position_window_s, + "orientation_window_s": orientation_window_s, + "n": int(len(compared)), + "score_spearman_rho": float(rho), + "median_absolute_score_change": float( + np.median( + np.abs( + compared["alternative_score"] + - compared["Coordination-control composite"] + ) + ) + ), + "fixed_cut_assignment_ari": float(fixed_ari), + "refit_assignment_ari": float(refit_ari), + } + ) + out = pd.DataFrame(rows) + out.to_csv(OUT / "processing_sensitivity.csv", index=False) + if not out.empty: + table_rows = [ + f"{row['processing']} & {int(row['n'])} & " + f"{row['score_spearman_rho']:.3f} & " + f"{row['median_absolute_score_change']:.3f} & " + f"{row['fixed_cut_assignment_ari']:.3f} & " + f"{row['refit_assignment_ari']:.3f} \\\\" + for _, row in out.iterrows() + ] + (TAB / "processing_sensitivity.tex").write_text( + "\n".join( + [ + "\\begin{tabular}{lrrrrr}", + "\\toprule", + "Processing alternative & $n$ & Score $\\rho$ & Median $|\\Delta|$ & Fixed-cut ARI & Refit ARI \\\\", + "\\midrule", + *table_rows, + "\\bottomrule", + "\\end{tabular}", + ] + ) + ) + return out + + +def kmeans_skill_clusters(df: pd.DataFrame, proxy: pd.DataFrame) -> tuple[pd.DataFrame, dict[str, object]]: + try: + from sklearn.cluster import KMeans + from sklearn.decomposition import PCA + from sklearn.metrics import adjusted_rand_score, calinski_harabasz_score, davies_bouldin_score, silhouette_score + from sklearn.preprocessing import StandardScaler + except Exception as exc: # pragma: no cover + empty = df[["dataset", "trial_name", "trial_number", "skill_category", "total_procedures"]].copy() + summary = {"available": False, "error": str(exc)} + (OUT / "kmeans_thresholds.json").write_text(json.dumps(summary, indent=2)) + return empty, summary + + cols = [c for c in OSATS_DOMAINS if c in proxy.columns] + core_cols = [c for c in CORE_PERFORMANCE_DOMAINS if c in proxy.columns] + tmp = proxy[ + [ + "dataset", + "cohort_label", + "trial_name", + "trial_number", + "trial_key", + "skill_category", + "total_procedures", + *cols, + "Analysed duration (s)", + "Coordination-control composite", + ] + ].copy() + tmp["log_total_procedures"] = np.log10(tmp["total_procedures"] + 1) + + pca_cols = core_cols + X = tmp[pca_cols].replace([np.inf, -np.inf], np.nan) + cluster_cols = ["Coordination-control composite"] + Xc = tmp[cluster_cols].replace([np.inf, -np.inf], np.nan).to_numpy(float) + identifier_counts = tmp.groupby("trial_key")["trial_key"].transform("size") + fit_mask = ( + identifier_counts.eq(1).to_numpy() + & np.isfinite(Xc).all(axis=1) + & np.isfinite(X.to_numpy(float)).all(axis=1) + ) + Xc_fit = Xc[fit_mask] + tmp["metric_pc1"] = np.nan + tmp["metric_pc2"] = np.nan + Xs = StandardScaler().fit_transform(X.loc[fit_mask].to_numpy(float)) + pca = PCA(n_components=2, random_state=20260618) + pcs = pca.fit_transform(Xs) + tmp.loc[fit_mask, "metric_pc1"] = pcs[:, 0] + tmp.loc[fit_mask, "metric_pc2"] = pcs[:, 1] + validation_rows = [] + if len(Xc_fit) >= 6: + for k in range(2, 7): + km_k = KMeans(n_clusters=k, n_init=200, random_state=20260618) + labels_k = km_k.fit_predict(Xc_fit) + validation_rows.append( + { + "k": k, + "silhouette": float(silhouette_score(Xc_fit, labels_k)), + "davies_bouldin": float(davies_bouldin_score(Xc_fit, labels_k)), + "calinski_harabasz": float(calinski_harabasz_score(Xc_fit, labels_k)), + "inertia": float(km_k.inertia_), + } + ) + + km = KMeans(n_clusters=3, n_init=300, random_state=20260618) + raw_cluster_fit = km.fit_predict(Xc_fit) + metric_silhouette = float(silhouette_score(Xc_fit, raw_cluster_fit)) + metric_davies = float(davies_bouldin_score(Xc_fit, raw_cluster_fit)) + metric_calinski = float(calinski_harabasz_score(Xc_fit, raw_cluster_fit)) + ordered_centres = np.sort(km.cluster_centers_.ravel()) + thresholds = (ordered_centres[:-1] + ordered_centres[1:]) / 2 + reference_ordered = np.full(len(tmp), -1, dtype=int) + reference_ordered[fit_mask] = np.digitize(Xc_fit.ravel(), thresholds) + tmp["metric_cluster"] = pd.Series(np.nan, index=tmp.index, dtype=object) + tmp.loc[fit_mask, "metric_cluster"] = [ + BAND_ORDER[i] for i in reference_ordered[fit_mask] + ] + rng = np.random.default_rng(20260618) + bootstrap_thresholds = [] + bootstrap_ari = [] + for seed in range(1000): + sample = rng.choice( + Xc_fit.ravel(), + size=len(Xc_fit), + replace=True, + ).reshape(-1, 1) + alt = KMeans(n_clusters=3, n_init=30, random_state=seed).fit(sample) + alt_centres = np.sort(alt.cluster_centers_.ravel()) + alt_thresholds = (alt_centres[:-1] + alt_centres[1:]) / 2 + bootstrap_thresholds.append(alt_thresholds) + bootstrap_ari.append( + adjusted_rand_score( + reference_ordered[fit_mask], + np.digitize(Xc_fit.ravel(), alt_thresholds), + ) + ) + bootstrap_thresholds = np.asarray(bootstrap_thresholds) + metric_seed_ari_mean = float(np.mean(bootstrap_ari)) + metric_seed_ari_min = float(np.min(bootstrap_ari)) + threshold_ci = np.percentile(bootstrap_thresholds, [2.5, 50, 97.5], axis=0) + + required_core_features = sorted( + { + col + for domain in CORE_PERFORMANCE_DOMAINS + for construct in OSATS_DOMAIN_CONSTRUCTS[domain] + for col, _ in construct + if col in df.columns + } + ) + complete_keys = set( + df.loc[df[required_core_features].notna().all(axis=1), "trial_key"].astype(str) + ) + complete_mask = ( + tmp["trial_key"].astype(str).isin(complete_keys).to_numpy() + & fit_mask + ) + complete_values = tmp.loc[complete_mask, ["Coordination-control composite"]].to_numpy() + complete_model = KMeans(n_clusters=3, n_init=300, random_state=20260618).fit(complete_values) + complete_centres = np.sort(complete_model.cluster_centers_.ravel()) + complete_thresholds = (complete_centres[:-1] + complete_centres[1:]) / 2 + complete_assignment_ari = float( + adjusted_rand_score( + reference_ordered[fit_mask], + np.digitize(Xc_fit.ravel(), complete_thresholds), + ) + ) + + perturbation_ari = [] + domain_matrix = tmp[core_cols].to_numpy(float) + for seed in range(500): + weights = rng.uniform(0.75, 1.25, size=len(core_cols)) + perturbed_score = np.average(domain_matrix, axis=1, weights=weights) + perturbed_model = KMeans(n_clusters=3, n_init=30, random_state=seed).fit( + perturbed_score[fit_mask].reshape(-1, 1) + ) + perturbed_centres = np.sort(perturbed_model.cluster_centers_.ravel()) + perturbed_thresholds = (perturbed_centres[:-1] + perturbed_centres[1:]) / 2 + perturbation_ari.append( + adjusted_rand_score( + reference_ordered[fit_mask], + np.digitize(perturbed_score[fit_mask], perturbed_thresholds), + ) + ) + + leave_one_domain_out = [] + for omitted in core_cols: + retained = [c for c in core_cols if c != omitted] + score = tmp[retained].mean(axis=1).to_numpy(float) + model = KMeans(n_clusters=3, n_init=300, random_state=20260618).fit( + score[fit_mask].reshape(-1, 1) + ) + centres = np.sort(model.cluster_centers_.ravel()) + cuts = (centres[:-1] + centres[1:]) / 2 + omitted_slug = re.sub(r"[^a-z0-9]+", "_", omitted.lower()).strip("_") + omitted_clusters = pd.Series(np.nan, index=tmp.index, dtype=object) + omitted_clusters.loc[fit_mask] = [ + BAND_ORDER[i] for i in np.digitize(score[fit_mask], cuts) + ] + tmp[f"metric_cluster_without_{omitted_slug}"] = omitted_clusters + leave_one_domain_out.append( + { + "omitted_domain": omitted, + "adjusted_rand_index": float( + adjusted_rand_score( + reference_ordered[fit_mask], + np.digitize(score[fit_mask], cuts), + ) + ), + "thresholds": [float(v) for v in cuts], + } + ) + + leave_one_cohort_out = [] + for held_out in DATASET_ORDER: + train_mask = (tmp["dataset"] != held_out).to_numpy() & fit_mask + train_values = Xc[train_mask] + model = KMeans(n_clusters=3, n_init=300, random_state=20260618).fit(train_values) + centres = np.sort(model.cluster_centers_.ravel()) + cuts = (centres[:-1] + centres[1:]) / 2 + held_mask = (tmp["dataset"] == held_out).to_numpy() & fit_mask + leave_one_cohort_out.append( + { + "held_out_dataset": held_out, + "thresholds": [float(v) for v in cuts], + "held_out_adjusted_rand_index": float( + adjusted_rand_score( + reference_ordered[held_mask], + np.digitize(Xc.ravel()[held_mask], cuts), + ) + ), + } + ) + else: + tmp["metric_cluster"] = np.nan + metric_silhouette = np.nan + metric_davies = np.nan + metric_calinski = np.nan + metric_seed_ari_mean = np.nan + metric_seed_ari_min = np.nan + ordered_centres = np.array([np.nan, np.nan, np.nan]) + thresholds = np.array([np.nan, np.nan]) + threshold_ci = np.full((3, 2), np.nan) + complete_mask = np.zeros(len(tmp), dtype=bool) + complete_thresholds = np.array([np.nan, np.nan]) + complete_assignment_ari = np.nan + bootstrap_ari = [np.nan] + perturbation_ari = [np.nan] + leave_one_domain_out = [] + leave_one_cohort_out = [] + tmp["metric_cluster_order"] = tmp["metric_cluster"].map({name: i for i, name in enumerate(BAND_ORDER)}) + tmp["data_driven_skill_band"] = tmp["metric_cluster"] + tmp.to_csv(OUT / "kmeans_skill_clusters.csv", index=False) + + cluster_rows = [] + for cl in BAND_ORDER: + g = tmp[tmp["metric_cluster"] == cl] + if g.empty: + continue + cluster_rows.append( + { + "metric_cluster": cl, + "cluster_label": BAND_LABELS[cl], + "n": int(len(g)), + "median_total_procedures": float(g["total_procedures"].median()), + "procedure_iqr_low": float(g["total_procedures"].quantile(0.25)), + "procedure_iqr_high": float(g["total_procedures"].quantile(0.75)), + "mean_coordination_control_composite": float( + g["Coordination-control composite"].mean() + ), + } + ) + cluster_summary = pd.DataFrame(cluster_rows) + cluster_summary.to_csv(OUT / "kmeans_cluster_summary.csv", index=False) + validation = pd.DataFrame(validation_rows) + validation.to_csv(OUT / "kmeans_validation.csv", index=False) + + label_codes = tmp["skill_category"].map({"novice": 0, "intermediate": 1, "expert": 2}) + valid_ari = ( + label_codes.notna().to_numpy() + & tmp["metric_cluster_order"].notna().to_numpy() + & fit_mask + ) + summary = { + "available": True, + "features": cluster_cols, + "n_recordings_assigned": int(tmp["metric_cluster"].notna().sum()), + "n_independent_identifiers_fitted": int(fit_mask.sum()), + "metric_band_silhouette": metric_silhouette, + "metric_band_davies_bouldin": metric_davies, + "metric_band_calinski_harabasz": metric_calinski, + "metric_band_seed_ari_mean": metric_seed_ari_mean, + "metric_band_seed_ari_min": metric_seed_ari_min, + "metric_band_bootstrap_assignment_ari_median": float(np.nanmedian(bootstrap_ari)), + "metric_band_bootstrap_assignment_ari_ci": [ + float(v) for v in np.nanpercentile(bootstrap_ari, [2.5, 97.5]) + ], + "metric_band_centres": [float(v) for v in ordered_centres], + "metric_band_thresholds": [float(v) for v in thresholds], + "metric_band_threshold_bootstrap_ci": [ + { + "threshold": i + 1, + "ci_low": float(threshold_ci[0, i]), + "median": float(threshold_ci[1, i]), + "ci_high": float(threshold_ci[2, i]), + } + for i in range(2) + ], + "sensitivity": { + "complete_case": { + "n_trials": int(complete_mask.sum()), + "thresholds": [float(v) for v in complete_thresholds], + "assignment_adjusted_rand_index": float(complete_assignment_ari), + }, + "domain_weight_perturbation": { + "n_repetitions": int(len(perturbation_ari)), + "adjusted_rand_index_median": float(np.nanmedian(perturbation_ari)), + "adjusted_rand_index_ci_low": float(np.nanpercentile(perturbation_ari, 2.5)), + "adjusted_rand_index_ci_high": float(np.nanpercentile(perturbation_ari, 97.5)), + }, + "leave_one_domain_out": leave_one_domain_out, + "leave_one_cohort_out": leave_one_cohort_out, + }, + "k_validation": validation.to_dict(orient="records"), + "pca_variance": [float(v) for v in pca.explained_variance_ratio_], + "adjusted_rand_vs_original_skill": float( + adjusted_rand_score( + label_codes.to_numpy()[valid_ari], + tmp.loc[valid_ari, "metric_cluster_order"], + ) + ), + "cluster_summary": cluster_summary.to_dict(orient="records"), + } + (OUT / "kmeans_thresholds.json").write_text(json.dumps(summary, indent=2)) + return tmp, summary + + +def held_out_band_statistics(clusters: pd.DataFrame) -> pd.DataFrame: + """Evaluate band gradients using outcomes that did not define the bands.""" + if clusters.empty: + out = pd.DataFrame() + out.to_csv(OUT / "held_out_band_statistics.csv", index=False) + return out + + identifier_counts = clusters.groupby("trial_key")["trial_key"].transform("size") + d = clusters.loc[identifier_counts.eq(1)].copy() + outcomes = [ + ("Task efficiency", 1), + ("Analysed duration (s)", -1), + ("Workspace excursion", 1), + ] + rows = [] + rng = np.random.default_rng(20260618) + for outcome, direction in outcomes: + if outcome not in d: + continue + oriented = pd.Series(np.nan, index=d.index, dtype=float) + for _, idx in d.groupby("dataset").groups.items(): + values = pd.to_numeric(d.loc[idx, outcome], errors="coerce") * direction + oriented.loc[idx] = values.rank(pct=True) + groups = [ + oriented.loc[d["metric_cluster"] == band].dropna().to_numpy() + for band in BAND_ORDER + ] + finite_groups = [values for values in groups if len(values)] + h, p = stats.kruskal(*finite_groups) if len(finite_groups) >= 2 else (np.nan, np.nan) + low, middle, high = groups + delta = cliffs_delta(high, low) + upper_lower_p = ( + stats.mannwhitneyu(high, low, alternative="two-sided").pvalue + if len(high) and len(low) + else np.nan + ) + boot = [] + if len(high) and len(low): + for _ in range(3000): + boot.append( + cliffs_delta( + rng.choice(high, size=len(high), replace=True), + rng.choice(low, size=len(low), replace=True), + ) + ) + lo, hi = ( + np.nanpercentile(boot, [2.5, 97.5]) + if boot + else (np.nan, np.nan) + ) + raw_groups = [ + pd.to_numeric( + d.loc[d["metric_cluster"] == band, outcome], + errors="coerce", + ).dropna() + for band in BAND_ORDER + ] + rows.append( + { + "outcome": outcome, + "direction": "higher" if direction > 0 else "lower", + "n_lower": int(len(raw_groups[0])), + "n_middle": int(len(raw_groups[1])), + "n_upper": int(len(raw_groups[2])), + "median_lower": float(raw_groups[0].median()), + "median_middle": float(raw_groups[1].median()), + "median_upper": float(raw_groups[2].median()), + "oriented_cliffs_delta_upper_vs_lower": float(delta), + "delta_ci_low": float(lo), + "delta_ci_high": float(hi), + "kruskal_h": float(h), + "kruskal_p": float(p), + "upper_lower_p": float(upper_lower_p), + } + ) + out = pd.DataFrame(rows) + if not out.empty: + out["kruskal_q"] = benjamini_hochberg(out["kruskal_p"]) + out["upper_lower_q"] = benjamini_hochberg(out["upper_lower_p"]) + out.to_csv(OUT / "held_out_band_statistics.csv", index=False) + + time_rows = [] + for cohort, g in [("All cohorts", d), *[(dataset_label(ds), d[d["dataset"] == ds]) for ds in DATASET_ORDER]]: + g = g.dropna(subset=["Coordination-control composite", "Analysed duration (s)"]) + rho, p = ( + stats.spearmanr( + g["Coordination-control composite"], + g["Analysed duration (s)"], + ) + if len(g) >= 3 + else (np.nan, np.nan) + ) + time_rows.append( + { + "cohort": cohort, + "n": int(len(g)), + "spearman_rho": float(rho), + "p": float(p), + } + ) + time_corr = pd.DataFrame(time_rows) + time_corr["q"] = benjamini_hochberg(time_corr["p"]) + time_corr.to_csv(OUT / "coordination_control_time_correlation.csv", index=False) + return out + + +def load_ssl_embedding_table(df: pd.DataFrame) -> pd.DataFrame: + path = AI_ELT / "outputs" / "ssl" / "origin" / "ORIGIN_ALL" / "ssl_results" / "trial_embeddings.json" + if not path.exists(): + return pd.DataFrame() + data = json.loads(path.read_text()) + rows = [] + meta = df[["trial_key", "dataset", "cohort_label", "trial_number", "skill_category", "total_procedures"]] + identifier_counts = meta.groupby("trial_key")["trial_key"].transform("size") + meta = meta.loc[identifier_counts.eq(1)].copy() + meta_map = meta.set_index("trial_key").to_dict(orient="index") + for key, value in data.items(): + if key not in meta_map: + continue + emb = value.get("mean_embedding") + if not isinstance(emb, list): + continue + row = {"trial_key": key, **meta_map[key]} + for i, v in enumerate(emb): + row[f"emb_{i:02d}"] = v + rows.append(row) + out = pd.DataFrame(rows) + if out.empty: + return out + try: + from sklearn.decomposition import PCA + from sklearn.preprocessing import StandardScaler + + emb_cols = [c for c in out.columns if c.startswith("emb_")] + X = StandardScaler().fit_transform(out[emb_cols].to_numpy(float)) + pca = PCA(n_components=2, random_state=20260618) + pcs = pca.fit_transform(X) + out["ssl_pc1"] = pcs[:, 0] + out["ssl_pc2"] = pcs[:, 1] + out.attrs["pca_variance"] = [float(v) for v in pca.explained_variance_ratio_] + except Exception: + pass + out.to_csv(OUT / "ssl_embedding_pca.csv", index=False) + return out + + +def phase_statistics(trials: pd.DataFrame, cycles: pd.DataFrame, frames: pd.DataFrame) -> dict[str, object]: + # Segment durations in the corrected phase cache. + seg_rows = [] + for (trial_id, ann_dir), g in frames.sort_values(["trial_id", "ann_dir", "frame_idx"]).groupby(["trial_id", "ann_dir"]): + gg = g.reset_index(drop=True) + if len(gg) < 2: + continue + dt = float(np.median(np.diff(gg["time_s"]))) if len(gg) > 2 else 0.0 + start = 0 + labels = gg["coarse_derived"].astype(str).to_numpy() + times = gg["time_s"].to_numpy(float) + for i in range(1, len(gg) + 1): + if i == len(gg) or labels[i] != labels[start]: + duration = max(0.0, (times[i - 1] - times[start]) + dt) + seg_rows.append( + { + "dataset": gg.loc[start, "dataset"], + "trial_id": trial_id, + "skill_category": gg.loc[start, "skill_category"], + "phase": labels[start], + "duration_s": duration, + "n_frames": i - start, + } + ) + start = i + seg = pd.DataFrame(seg_rows) + seg.to_csv(OUT / "phase_segments_current.csv", index=False) + if not seg.empty: + seg_summary = ( + seg.groupby("phase")["duration_s"] + .agg(n_segments="count", mean_duration_s="mean", median_duration_s="median") + .reindex(PHASE_ORDER) + .dropna(how="all") + .reset_index() + ) + seg_summary.to_csv(OUT / "phase_segment_summary.csv", index=False) + + canonical = pd.read_csv(BTPN_CACHE / "canonical_trials.csv") + canonical["trial_key"] = [ + trial_key(dataset, trial_number) + for dataset, trial_number in zip(canonical["dataset"], canonical["trial_number"]) + ] + global_counts = canonical.groupby("trial_key")["trial_key"].transform("size") + unique_keys = set(canonical.loc[global_counts.eq(1), "trial_key"]) + inferential_trials = trials[trials["trial_key"].isin(unique_keys)].copy() + inferential_cycles = cycles[cycles["trial_key"].isin(unique_keys)].copy() + + stats_out: dict[str, object] = {} + cycle_records = [] + trial_cycles = ( + inferential_cycles.groupby(["dataset", "skill_category", "trial_id"], as_index=False) + .agg(mean_cycle_duration_s=("duration_s", "mean"), n_cycles=("cycle_index", "count")) + ) + for (dataset, skill), g in trial_cycles.groupby(["dataset", "skill_category"]): + x = g["mean_cycle_duration_s"].dropna() + lo, hi = bootstrap_ci(x) + cycle_records.append( + { + "dataset": dataset, + "skill_category": skill, + "n_cycles": int(g["n_cycles"].sum()), + "n_trials": int(len(x)), + "mean_duration_s": float(x.mean()), + "ci_low": lo, + "ci_high": hi, + } + ) + cycle_summary = pd.DataFrame(cycle_records) + cycle_summary.to_csv(OUT / "cycle_duration_ci.csv", index=False) + + phase_records = [] + frac_cols = [c for c in inferential_trials.columns if c.startswith("frac_")] + for skill, g in inferential_trials.groupby("skill_category"): + for c in frac_cols: + phase = c.replace("frac_", "") + x = g[c].dropna() + lo, hi = bootstrap_ci(x) + phase_records.append( + { + "skill_category": skill, + "phase": phase, + "n_trials": int(len(x)), + "mean_fraction": float(x.mean()), + "ci_low": lo, + "ci_high": hi, + } + ) + phase_summary = pd.DataFrame(phase_records) + phase_summary.to_csv(OUT / "phase_fraction_ci.csv", index=False) + + stats_out["n_segments"] = int(len(seg)) + stats_out["n_inferential_phase_recordings"] = int(len(inferential_trials)) + stats_out["n_unique_phase_identifiers"] = int(inferential_trials["trial_key"].nunique()) + stats_out["mean_cycle_duration_by_skill"] = ( + trial_cycles.groupby("skill_category")["mean_cycle_duration_s"].mean().to_dict() + ) + return stats_out + + +def phase_specific_kinematics( + frames: pd.DataFrame, + clusters: pd.DataFrame, +) -> pd.DataFrame: + """Exploratory phase-specific kinematics for phase-labelled primary recordings.""" + base = AI_ELT / "outputs" / "ssl" / "origin" / "origin_data" / "ORIGIN_ALL" + if not base.exists(): + out = pd.DataFrame() + out.to_csv(OUT / "phase_kinematics_summary.csv", index=False) + return out + rows = [] + for (dataset, trial_id, trial_short), g in frames.groupby(["dataset", "trial_id", "trial_short"]): + key = phase_trial_key(dataset, trial_short) + path = base / f"{key}.json" + if not path.exists(): + continue + data = json.loads(path.read_text()) + feats = np.asarray(data.get("features", []), dtype=float) + if feats.ndim != 2 or feats.shape[1] < 14 or len(feats) < 3: + continue + phase_g = g.sort_values("frame_idx").reset_index(drop=True) + phase_frame_idx = phase_g["frame_idx"].to_numpy(int) + phase_labels = phase_g["coarse_derived"].astype(str).to_numpy() + if len(phase_frame_idx) < 3: + continue + valid_frames = (phase_frame_idx >= 0) & (phase_frame_idx < len(feats)) + phase_frame_idx = phase_frame_idx[valid_frames] + labels = phase_labels[valid_frames] + if len(phase_frame_idx) < 3: + continue + fps = _phase_frame_rate(str(dataset)) + p1 = _savgol_positions(feats[:, 0:3], fps)[phase_frame_idx] + p2 = _savgol_positions(feats[:, 7:10], fps)[phase_frame_idx] + frame_steps = np.diff(phase_frame_idx) + dt = frame_steps / fps + valid = frame_steps > 0 + d1 = np.linalg.norm(np.diff(p1, axis=0), axis=1) + d2 = np.linalg.norm(np.diff(p2, axis=0), axis=1) + interval_phase = labels[1:] + for phase in PHASE_ORDER: + mask = valid & (interval_phase == phase) + if not mask.any(): + continue + rows.append( + { + "dataset": dataset, + "trial_id": trial_id, + "trial_key": key, + "phase": phase, + "duration_s": float(dt[mask].sum()), + "tool1_path_mm": float(d1[mask].sum()), + "tool2_path_mm": float(d2[mask].sum()), + "tool1_mean_speed_mm_s": float(d1[mask].sum() / dt[mask].sum()), + "tool2_mean_speed_mm_s": float(d2[mask].sum() / dt[mask].sum()), + "combined_path_rate_mm_s": float((d1[mask].sum() + d2[mask].sum()) / dt[mask].sum()), + "n_intervals": int(mask.sum()), + } + ) + phase_df = pd.DataFrame(rows) + if not phase_df.empty: + valid_keys = set(clusters.loc[clusters["metric_cluster"].notna(), "trial_key"]) + phase_df = phase_df[phase_df["trial_key"].isin(valid_keys)].copy() + phase_df.to_csv(OUT / "phase_kinematics_by_trial.csv", index=False) + if phase_df.empty: + out = pd.DataFrame() + else: + out = ( + phase_df.groupby("phase") + .agg( + n_trials=("trial_key", "nunique"), + duration_s=("duration_s", "sum"), + tool1_speed_mm_s=("tool1_mean_speed_mm_s", "mean"), + tool2_speed_mm_s=("tool2_mean_speed_mm_s", "mean"), + combined_path_rate_mm_s=("combined_path_rate_mm_s", "mean"), + ) + .reindex(PHASE_ORDER) + .dropna(how="all") + .reset_index() + ) + out.to_csv(OUT / "phase_kinematics_summary.csv", index=False) + return out + + +def trimming_rule_validation( + frames: pd.DataFrame, + clusters: pd.DataFrame, +) -> pd.DataFrame: + """Compare automatic motion bounds with dense non-idle annotation bounds.""" + valid_keys = set( + clusters.loc[clusters["metric_cluster"].notna(), "trial_key"].astype(str) + ) + rows = [] + for (dataset, key), group in frames.groupby(["dataset", "trial_key"]): + key = str(key) + if key not in valid_keys: + continue + path = ORIGIN_MOTION / f"{key}.json" + if not path.exists(): + continue + payload = json.loads(path.read_text()) + raw = np.asarray(payload.get("features", []), dtype=float) + if raw.ndim != 2 or raw.shape[1] < 14 or len(raw) < 20: + continue + fps = _phase_frame_rate(str(dataset)) + manual_start = int(group["frame_idx"].min()) + manual_end = int(group["frame_idx"].max()) + automatic_start, automatic_end = _active_motion_bounds( + raw[:, 0:3], + raw[:, 7:10], + fps, + ) + manual_start = max(0, min(manual_start, len(raw) - 1)) + manual_end = max(manual_start, min(manual_end, len(raw) - 1)) + overlap = max( + 0, + min(manual_end, automatic_end) + - max(manual_start, automatic_start) + + 1, + ) + union = ( + max(manual_end, automatic_end) + - min(manual_start, automatic_start) + + 1 + ) + manual_duration = (manual_end - manual_start + 1) / fps + automatic_duration = (automatic_end - automatic_start + 1) / fps + rows.append( + { + "dataset": str(dataset), + "trial_key": key, + "manual_start_frame": manual_start, + "manual_end_frame": manual_end, + "automatic_start_frame": automatic_start, + "automatic_end_frame": automatic_end, + "start_error_s": (automatic_start - manual_start) / fps, + "end_error_s": (automatic_end - manual_end) / fps, + "duration_error_s": automatic_duration - manual_duration, + "absolute_duration_error_s": abs( + automatic_duration - manual_duration + ), + "interval_iou": overlap / union if union > 0 else np.nan, + } + ) + detail = pd.DataFrame(rows) + detail.to_csv(OUT / "trimming_rule_validation_by_recording.csv", index=False) + if detail.empty: + return detail + summary_rows = [] + for label, group in [ + ("All cohorts", detail), + *[ + (dataset_label(dataset), detail[detail["dataset"] == dataset]) + for dataset in DATASET_ORDER + ], + ]: + if group.empty: + continue + summary_rows.append( + { + "cohort": label, + "n": int(len(group)), + "median_abs_start_error_s": float( + np.median(np.abs(group["start_error_s"])) + ), + "median_abs_end_error_s": float( + np.median(np.abs(group["end_error_s"])) + ), + "median_abs_duration_error_s": float( + np.median(group["absolute_duration_error_s"]) + ), + "median_interval_iou": float( + np.median(group["interval_iou"]) + ), + } + ) + summary = pd.DataFrame(summary_rows) + summary.to_csv(OUT / "trimming_rule_validation.csv", index=False) + table_rows = [ + f"{row['cohort']} & {int(row['n'])} & " + f"{row['median_abs_start_error_s']:.2f} & " + f"{row['median_abs_end_error_s']:.2f} & " + f"{row['median_abs_duration_error_s']:.2f} & " + f"{row['median_interval_iou']:.3f} \\\\" + for _, row in summary.iterrows() + ] + (TAB / "trimming_rule_validation.tex").write_text( + "\n".join( + [ + "\\begin{tabular}{lrrrrr}", + "\\toprule", + "Cohort & $n$ & Start error (s) & End error (s) & Duration error (s) & Interval overlap \\\\", + "\\midrule", + *table_rows, + "\\bottomrule", + "\\end{tabular}", + ] + ) + ) + return summary + + +def load_origin_motion(df: pd.DataFrame, max_points_per_trial: int = 350) -> pd.DataFrame: + base = AI_ELT / "outputs" / "ssl" / "origin" / "origin_data" / "ORIGIN_ALL" + if not base.exists(): + return pd.DataFrame() + meta = df[["trial_key", "dataset", "cohort_label", "trial_number", "skill_category", "total_procedures"]] + identifier_counts = meta.groupby("trial_key")["trial_key"].transform("size") + meta = ( + meta.loc[identifier_counts.eq(1)] + .set_index("trial_key") + .to_dict(orient="index") + ) + rows = [] + rng = np.random.default_rng(20260618) + for path in sorted(base.glob("*.json")): + if path.name == "metadata.json": + continue + data = json.loads(path.read_text()) + key = data.get("name") + if key not in meta: + continue + feats = np.asarray(data.get("features", []), dtype=float) + times = np.asarray(data.get("times", []), dtype=float) + if feats.ndim != 2 or feats.shape[1] < 14: + continue + n = len(feats) + if n > max_points_per_trial: + idx = np.sort(rng.choice(np.arange(n), size=max_points_per_trial, replace=False)) + else: + idx = np.arange(n) + for i in idx: + row = { + "trial_key": key, + "frame_sample": int(i), + "time_s": float(times[i]) if i < len(times) else np.nan, + "tool1_x": feats[i, 0], + "tool1_y": feats[i, 1], + "tool1_z": feats[i, 2], + "tool1_qw": feats[i, 3], + "tool1_qx": feats[i, 4], + "tool1_qy": feats[i, 5], + "tool1_qz": feats[i, 6], + "tool2_x": feats[i, 7], + "tool2_y": feats[i, 8], + "tool2_z": feats[i, 9], + "tool2_qw": feats[i, 10], + "tool2_qx": feats[i, 11], + "tool2_qy": feats[i, 12], + "tool2_qz": feats[i, 13], + **meta[key], + } + rows.append(row) + out = pd.DataFrame(rows) + if not out.empty: + out.to_csv(OUT / "origin_motion_sample.csv", index=False) + return out + + +def plot_cohort(df: pd.DataFrame, phase_trials: pd.DataFrame) -> None: + fig, axes = plt.subplots(1, 2, figsize=(7.2, 3.0), gridspec_kw={"width_ratios": [1.2, 1]}) + ax = axes[0] + counts = pd.crosstab(df["dataset"], df["skill_category"]).reindex(DATASET_ORDER).fillna(0) + bottom = np.zeros(len(counts)) + x = np.arange(len(counts)) + for skill in SKILL_ORDER: + vals = counts.get(skill, pd.Series(0, index=counts.index)).to_numpy() + ax.bar(x, vals, bottom=bottom, color=SKILL_COLORS[skill], label=nice(skill), width=0.7) + bottom += vals + ax.set_xticks(x, [dataset_label(ds) for ds in counts.index], rotation=20, ha="right") + ax.set_ylabel("Trials") + panel_label(ax, "a") + ax.legend(frameon=False) + + ax = axes[1] + phase_counts = pd.crosstab(phase_trials["dataset"], phase_trials["skill_category"]).reindex(DATASET_ORDER).fillna(0) + bottom = np.zeros(len(phase_counts)) + x = np.arange(len(phase_counts)) + for skill in SKILL_ORDER: + vals = phase_counts.get(skill, pd.Series(0, index=phase_counts.index)).to_numpy() + ax.bar(x, vals, bottom=bottom, color=SKILL_COLORS[skill], width=0.7) + bottom += vals + ax.set_xticks(x, [dataset_label(ds) for ds in phase_counts.index], rotation=20, ha="right") + ax.set_ylabel("Trials") + panel_label(ax, "b") + fig.tight_layout() + save_fig(fig, "fig1_cohort_structure") + + +def plot_task_setup_overview() -> None: + sample_paths = [ + (AI_ELT / "docs" / "paper" / "figures" / "dataset_b.png", "Paediatric"), + (AI_ELT / "docs" / "paper" / "figures" / "dataset_c.png", "Urology 1"), + (AI_ELT / "docs" / "paper" / "figures" / "dataset_a.png", "Urology 2"), + ] + if not all(path.exists() for path, _ in sample_paths): + return + fig, axes = plt.subplots(1, 3, figsize=(7.2, 2.8)) + for ax, (path, label), panel in zip(axes, sample_paths, "abc"): + img = plt.imread(path) + ax.imshow(img) + ax.set_axis_off() + ax.text( + 0.02, + 0.08, + label, + transform=ax.transAxes, + fontsize=8, + weight="bold", + color="white", + bbox=dict(facecolor="black", alpha=0.55, edgecolor="none", pad=2), + ) + panel_label(ax, panel) + fig.tight_layout(w_pad=0.2) + save_fig(fig, "fig0_task_setup_overview") + + +def plot_training_systems_placeholder() -> None: + systems = [ + { + "name": "MISTELS", + "subtitle": "Bench-station laparoscopic skills", + "items": ["Peg transfer", "Pattern cutting", "Ligating loop", "Suturing tasks"], + "note": "Insert representative station or official task photograph", + }, + { + "name": "FLS", + "subtitle": "Fundamentals of Laparoscopic Surgery", + "items": ["Standard timed tasks", "Error penalties", "Proficiency targets", "Certification context"], + "note": "Insert FLS box trainer or task-board image", + }, + { + "name": "E-BLUS", + "subtitle": "European Basic Laparoscopic Urological Skills", + "items": ["Urology-focused dry-lab tasks", "Peg transfer", "Cutting and suturing", "Exam pathway"], + "note": "Insert E-BLUS exam or task-station image", + }, + ] + fig, axes = plt.subplots(1, 3, figsize=(7.4, 2.8)) + for ax, system, panel in zip(axes, systems, "abc"): + ax.set_axis_off() + ax.add_patch(Rectangle((0.04, 0.24), 0.92, 0.58, facecolor="#F7F7F7", edgecolor="#555555", lw=0.9)) + ax.text(0.5, 0.72, system["name"], ha="center", va="center", fontsize=13, weight="bold") + ax.text(0.5, 0.61, system["subtitle"], ha="center", va="center", fontsize=7.5, color="#333333", wrap=True) + y = 0.50 + for item in system["items"]: + ax.text(0.12, y, f"- {item}", ha="left", va="center", fontsize=6.9) + y -= 0.085 + ax.text( + 0.5, + 0.14, + system["note"], + ha="center", + va="center", + fontsize=6.4, + color="#555555", + style="italic", + wrap=True, + ) + panel_label(ax, panel) + fig.tight_layout(w_pad=0.4) + save_fig(fig, "fig0_training_systems_placeholder") + + +def plot_collection_setup_placeholder() -> None: + setups = [ + ("Paediatric", "7 DoF + jaw", "Paediatric dry-lab course", "Insert photograph of BAPES collection station"), + ("Urology 1", "6 DoF", "2023 urology cohort", "Insert photograph of original 6 DoF rig"), + ("Urology 2", "7 DoF + jaw", "2024 urology cohort", "Insert photograph of updated 7 DoF rig"), + ] + fig, axes = plt.subplots(1, 3, figsize=(7.4, 2.8)) + for ax, (cohort, sensing, context, note), panel in zip(axes, setups, "abc"): + ax.set_axis_off() + ax.add_patch(Rectangle((0.05, 0.22), 0.90, 0.60, facecolor="#F4F8FB", edgecolor="#4A4A4A", lw=0.9)) + ax.plot([0.18, 0.45], [0.40, 0.56], color="#0072B2", lw=2.2) + ax.plot([0.82, 0.55], [0.40, 0.56], color="#009E73", lw=2.2) + ax.add_patch(Rectangle((0.36, 0.50), 0.28, 0.10, facecolor="#DDDDDD", edgecolor="#777777", lw=0.6)) + ax.add_patch(Rectangle((0.43, 0.57), 0.14, 0.08, facecolor="#BBBBBB", edgecolor="#777777", lw=0.6)) + ax.text(0.5, 0.76, cohort, ha="center", va="center", fontsize=11, weight="bold") + ax.text(0.5, 0.66, sensing, ha="center", va="center", fontsize=8) + ax.text(0.5, 0.30, context, ha="center", va="center", fontsize=7.2) + ax.text(0.5, 0.13, note, ha="center", va="center", fontsize=6.3, style="italic", color="#555555", wrap=True) + panel_label(ax, panel) + fig.tight_layout(w_pad=0.35) + save_fig(fig, "fig0_collection_setups_placeholder") + + +def plot_peg_transfer_cycle_placeholder() -> None: + stages = [ + ("Reach", "Approach the source peg", "left", (0.24, 0.54), (0.22, 0.45)), + ("Grasp", "Close jaws around object", "left", (0.28, 0.55), (0.28, 0.55)), + ("Lift", "Lift clear of the peg", "left", (0.30, 0.68), (0.30, 0.68)), + ("Transfer", "Meet the opposite tool", "both", (0.50, 0.65), (0.50, 0.65)), + ("Place", "Move to the target peg", "right", (0.72, 0.58), (0.72, 0.58)), + ("Release", "Release and reset", "right", (0.76, 0.52), (0.78, 0.45)), + ] + + def draw_scene(ax: plt.Axes, stage: tuple[str, str, str, tuple[float, float], tuple[float, float]], panel: str) -> None: + name, subtitle, active, tool_tip, object_xy = stage + ax.set_xlim(0, 1) + ax.set_ylim(0, 1) + ax.set_aspect("equal") + ax.set_axis_off() + ax.add_patch(Rectangle((0.10, 0.20), 0.80, 0.50, facecolor="#F8F8F8", edgecolor="#777777", lw=0.8)) + for x in (0.25, 0.75): + ax.plot([x, x], [0.34, 0.56], color="#555555", lw=2.0, solid_capstyle="round") + ax.add_patch(plt.Circle((x, 0.56), 0.025, facecolor="#D0D0D0", edgecolor="#777777", lw=0.6)) + + left_tip = tool_tip if active in {"left", "both"} else (0.18, 0.40) + right_tip = tool_tip if active in {"right", "both"} else (0.82, 0.40) + ax.plot([0.05, left_tip[0]], [0.22, left_tip[1]], color="#0072B2", lw=2.2, solid_capstyle="round") + ax.plot([0.95, right_tip[0]], [0.22, right_tip[1]], color="#009E73", lw=2.2, solid_capstyle="round") + + if active == "both": + ax.add_patch(FancyArrowPatch((0.35, 0.63), (0.65, 0.63), arrowstyle="<->", mutation_scale=10, lw=1.2, color="#555555")) + elif active == "left": + ax.add_patch(FancyArrowPatch((0.18, 0.45), tool_tip, arrowstyle="->", mutation_scale=10, lw=1.2, color="#0072B2")) + elif active == "right": + ax.add_patch(FancyArrowPatch((0.58, 0.60), tool_tip, arrowstyle="->", mutation_scale=10, lw=1.2, color="#009E73")) + + ax.add_patch(plt.Circle(object_xy, 0.040, facecolor="#E69F00", edgecolor="#8A5A00", lw=0.8)) + if name == "Release": + ax.add_patch(FancyArrowPatch((0.70, 0.52), (0.80, 0.52), arrowstyle="->", mutation_scale=9, lw=1.0, color="#CC3311")) + ax.text(0.50, 0.16, "Optional nudge or correction", ha="center", va="center", fontsize=6.4, color="#CC3311") + + ax.text(0.50, 0.91, name, ha="center", va="center", fontsize=10, weight="bold") + ax.text(0.50, 0.82, subtitle, ha="center", va="center", fontsize=6.9, color="#333333", wrap=True) + panel_label(ax, panel) + + fig, axes = plt.subplots(2, 3, figsize=(7.4, 4.8)) + for ax, stage, panel in zip(axes.flat, stages, "abcdef"): + draw_scene(ax, stage, panel) + fig.tight_layout(w_pad=0.25, h_pad=0.35) + save_fig(fig, "fig0_peg_transfer_cycle_placeholder") + + +def plot_feature_effects(stats_df: pd.DataFrame) -> None: + d = stats_df.dropna(subset=["cliffs_delta_novice_expert"]).copy() + d = d[d["n_nonmissing"] >= 20] + orientation = [] + for feature in d["feature"]: + arrow, _ = FEATURE_DIRECTIONS.get(feature, ("$\\uparrow$", "")) + orientation.append(-1 if "\\uparrow" in arrow else 1) + d["_orientation"] = orientation + d["oriented_delta"] = d["cliffs_delta_novice_expert"] * d["_orientation"] + d["oriented_ci_low"] = np.where( + d["_orientation"] > 0, + d["delta_ci_low"], + -d["delta_ci_high"], + ) + d["oriented_ci_high"] = np.where( + d["_orientation"] > 0, + d["delta_ci_high"], + -d["delta_ci_low"], + ) + d = d.loc[d["oriented_delta"].abs().sort_values(ascending=False).index].head(18) + d = d.sort_values("oriented_delta") + y = np.arange(len(d)) + fig, ax = plt.subplots(figsize=(7.2, max(5.2, 0.22 * len(d)))) + ax.axvline(0, color="#444444", lw=0.8) + colors = ["#0072B2" if q < 0.05 else "#8A8A8A" for q in d["kruskal_q"]] + ax.errorbar( + d["oriented_delta"], + y, + xerr=[ + d["oriented_delta"] - d["oriented_ci_low"], + d["oriented_ci_high"] - d["oriented_delta"], + ], + fmt="none", + ecolor="#666666", + elinewidth=0.9, + capsize=2, + ) + ax.scatter(d["oriented_delta"], y, s=24, c=colors, zorder=3) + ax.set_yticks(y, d["label"]) + ax.set_xlabel("Oriented Cliff's delta for expert versus novice") + ax.text( + 0.01, + 1.01, + "Novice-favouring", + transform=ax.transAxes, + ha="left", + va="bottom", + fontsize=7, + ) + ax.text( + 0.99, + 1.01, + "Expert-favouring", + transform=ax.transAxes, + ha="right", + va="bottom", + fontsize=7, + ) + fig.tight_layout() + save_fig(fig, "fig2_feature_effects") + + +def plot_phase_timing(cycles: pd.DataFrame, trials: pd.DataFrame) -> None: + fig, axes = plt.subplots(1, 2, figsize=(7.2, 3.35)) + recording_counts = ( + trials[["trial_key", "trial_id"]] + .drop_duplicates() + .groupby("trial_key")["trial_id"] + .transform("size") + ) + unique_keys = set( + trials[["trial_key", "trial_id"]] + .drop_duplicates() + .loc[recording_counts.eq(1), "trial_key"] + ) + inferential_trials = trials[trials["trial_key"].isin(unique_keys)].copy() + inferential_cycles = cycles[cycles["trial_key"].isin(unique_keys)].copy() + trial_cycles = ( + inferential_cycles.groupby(["dataset", "skill_category", "trial_id"], as_index=False) + .agg(mean_cycle_duration_s=("duration_s", "mean")) + ) + ax = axes[0] + x_base = np.arange(len(DATASET_ORDER)) + offsets = {"novice": -0.22, "intermediate": 0.0, "expert": 0.22} + for skill in SKILL_ORDER: + xs, ys, los, his = [], [], [], [] + for i, ds in enumerate(DATASET_ORDER): + g = trial_cycles[ + (trial_cycles["dataset"] == ds) + & (trial_cycles["skill_category"] == skill) + ]["mean_cycle_duration_s"].dropna() + if len(g) == 0: + continue + lo, hi = bootstrap_ci(g) + xs.append(i + offsets[skill]) + ys.append(g.mean()) + los.append(g.mean() - lo) + his.append(hi - g.mean()) + ax.errorbar(xs, ys, yerr=[los, his], fmt="o", capsize=2, color=SKILL_COLORS[skill], label=nice(skill), ms=4) + ax.set_xticks(x_base, [dataset_label(ds) for ds in DATASET_ORDER], rotation=20, ha="right") + ax.set_ylabel("Cycle duration (s)") + panel_label(ax, "a") + ax.legend(frameon=False) + + ax = axes[1] + phase_cols = ["frac_reach", "frac_grasp", "frac_transfer", "frac_place", "frac_nudge", "frac_idle"] + means = inferential_trials.groupby("skill_category")[phase_cols].mean().reindex(SKILL_ORDER) + bottom = np.zeros(len(means)) + x = np.arange(len(means)) + for c in phase_cols: + phase = c.replace("frac_", "") + vals = means[c].fillna(0).to_numpy() + ax.bar(x, vals, bottom=bottom, color=PHASE_COLORS[phase], label=nice(phase), width=0.7) + bottom += vals + ax.set_xticks(x, [nice(s) for s in SKILL_ORDER], rotation=20, ha="right") + ax.set_ylabel("Mean fraction of corrected trial time") + panel_label(ax, "b") + handles, labels = ax.get_legend_handles_labels() + fig.legend(handles, labels, frameon=False, ncol=len(labels), loc="lower center", bbox_to_anchor=(0.5, 0.02)) + fig.subplots_adjust(left=0.10, right=0.98, top=0.91, bottom=0.25, wspace=0.34) + save_fig(fig, "fig3_phase_timing") + + +def plot_experience_links(df: pd.DataFrame, phase_trials: pd.DataFrame) -> None: + fig, axes = plt.subplots(2, 3, figsize=(9.4, 5.2), sharex=False) + for c, ds in enumerate(DATASET_ORDER): + ax = axes[0, c] + g = df[df["dataset"] == ds] + ax.scatter( + np.log10(g["total_procedures"] + 1), + g["bimanual_correlation"], + s=18, + alpha=0.75, + color=DATASET_COLORS[ds], + ) + ax.set_title(dataset_label(ds), fontsize=8) + ax.set_xlabel("log10(total procedures + 1)") + if c == 0: + ax.set_ylabel("Bimanual correlation") + panel_label(ax, chr(ord("a") + c)) + + ax = axes[1, c] + g = phase_trials[phase_trials["dataset"] == ds] + ax.scatter( + np.log10(g["total_procedures"] + 1), + g["mean_cycle_s"], + s=22, + alpha=0.8, + color=DATASET_COLORS[ds], + ) + ax.set_xlabel("log10(total procedures + 1)") + if c == 0: + ax.set_ylabel("Mean cycle duration (s)") + panel_label(ax, chr(ord("d") + c)) + fig.tight_layout() + save_fig(fig, "fig4_experience_links") + + +def plot_osats_proxy(proxy: pd.DataFrame, clusters: pd.DataFrame, cluster_summary: dict[str, object]) -> None: + band_proxy = proxy.merge( + clusters[["dataset", "trial_name", "metric_cluster"]].drop_duplicates( + ["dataset", "trial_name"] + ), + on=["dataset", "trial_name"], + how="left", + validate="one_to_one", + ) + domain_cols = [ + "Coordination-control composite", + *CORE_PERFORMANCE_DOMAINS, + "Task efficiency", + "Workspace excursion", + ] + display_labels = { + "Bimanual coordination": "Bimanual\ncoordination", + "Task efficiency": "Task\nefficiency", + "Instrument motion control": "Instrument\nmotion control", + "Workspace excursion": "Workspace\ncompactness\n(descriptive)", + "Coordination-control composite": "Coordination and control\nscore", + "Analysed duration (s)": "Analysed\nduration", + } + + fig, axes = plt.subplots(len(DATASET_ORDER), 1, figsize=(7.2, 6.8), sharex=True) + ims = [] + for idx, (ax, ds, label) in enumerate(zip(axes, DATASET_ORDER, "abc")): + cohort = band_proxy[band_proxy["dataset"] == ds] + summary = cohort.groupby("metric_cluster")[domain_cols].mean().reindex(BAND_ORDER) + im = ax.imshow(summary[domain_cols].to_numpy(float), vmin=1, vmax=5, cmap="YlGnBu", aspect="auto") + ims.append(im) + counts = cohort["metric_cluster"].value_counts() + ax.set_title(dataset_label(ds), fontsize=8, loc="left", pad=3) + ax.set_yticks( + np.arange(len(summary)), + [f"{BAND_SHORT_LABELS[b]} (n={int(counts.get(b, 0))})" for b in summary.index], + ) + ax.set_xticks(np.arange(len(domain_cols))) + if idx == len(DATASET_ORDER) - 1: + ax.set_xticklabels([display_labels[c] for c in domain_cols], rotation=24, ha="right") + else: + ax.set_xticklabels([]) + ax.set_ylabel("Measured\nband") + for r in range(summary.shape[0]): + for c in range(summary.shape[1]): + value = summary.iloc[r, c] + text = "Not recorded" if not np.isfinite(value) else f"{value:.2f}" + ax.text(c, r, text, ha="center", va="center", fontsize=6.1) + panel_label(ax, label) + fig.subplots_adjust(left=0.20, right=0.98, top=0.95, bottom=0.24, hspace=0.38) + cax = fig.add_axes([0.30, 0.055, 0.42, 0.025]) + cbar = fig.colorbar(ims[0], cax=cax, orientation="horizontal") + cbar.set_label("Relative domain score (1-5)") + save_fig(fig, "fig5_osats_proxy_heatmap") + + held_out_cols = [ + "Task efficiency", + "Analysed duration (s)", + "Workspace excursion", + ] + identifier_counts = band_proxy.groupby("trial_key")["trial_key"].transform("size") + held_out_proxy = band_proxy[identifier_counts == 1].copy() + fig, axes = plt.subplots( + len(held_out_cols), + len(DATASET_ORDER), + figsize=(9.4, 6.8), + sharey="row", + ) + for r, domain in enumerate(held_out_cols): + for c, ds in enumerate(DATASET_ORDER): + ax = axes[r, c] + cohort = held_out_proxy[held_out_proxy["dataset"] == ds] + data = [cohort.loc[cohort["metric_cluster"] == band, domain].dropna() for band in BAND_ORDER] + if sum(len(values) for values in data) == 0: + ax.set_axis_off() + ax.text(0.5, 0.5, "Jaw aperture not recorded", ha="center", va="center", transform=ax.transAxes, fontsize=7) + continue + bp = ax.boxplot( + data, + patch_artist=True, + tick_labels=[BAND_SHORT_LABELS[band] for band in BAND_ORDER], + showfliers=False, + widths=0.58, + ) + for patch, band in zip(bp["boxes"], BAND_ORDER): + patch.set_facecolor(CLUSTER_COLORS[band]) + patch.set_alpha(0.68) + for i, values in enumerate(data): + if len(values) == 0: + continue + xs = np.full(len(values), i + 1) + np.linspace(-0.08, 0.08, len(values)) + ax.scatter(xs, values, s=6.5, color="#222222", alpha=0.34, linewidths=0) + if domain != "Analysed duration (s)": + ax.set_ylim(1, 5) + if r == 0: + ax.set_title(dataset_label(ds), fontsize=8) + if c == 0: + unit = "score (1-5)" if domain != "Analysed duration (s)" else "seconds" + ax.set_ylabel(f"{display_labels[domain].replace(chr(10), ' ')}\n{unit}") + ax.tick_params(axis="x", rotation=0, labelsize=6.5) + ax.text( + 0.08, + 1.03, + chr(ord("a") + r * len(DATASET_ORDER) + c), + transform=ax.transAxes, + fontsize=10, + fontweight="bold", + ) + fig.tight_layout(h_pad=1.12, w_pad=0.52) + save_fig(fig, "fig5_osats_proxy_boxplots") + + experience_cols = [ + *CORE_PERFORMANCE_DOMAINS, + "Task efficiency", + "Workspace excursion", + "Coordination-control composite", + ] + fig, axes = plt.subplots(len(DATASET_ORDER), 1, figsize=(7.0, 6.8), sharex=True) + ims = [] + for idx, (ax, ds, label) in enumerate(zip(axes, DATASET_ORDER, "abc")): + cohort = proxy[proxy["dataset"] == ds] + summary = cohort.groupby("skill_category")[experience_cols].mean().reindex(SKILL_ORDER) + im = ax.imshow(summary[experience_cols].to_numpy(float), vmin=1, vmax=5, cmap="YlGnBu", aspect="auto") + ims.append(im) + ax.set_title(dataset_label(ds), fontsize=8, loc="left", pad=3) + ax.set_yticks(np.arange(len(summary)), [nice(s) for s in summary.index]) + ax.set_xticks(np.arange(len(experience_cols))) + if idx == len(DATASET_ORDER) - 1: + ax.set_xticklabels([display_labels.get(c, c) for c in experience_cols], rotation=24, ha="right") + else: + ax.set_xticklabels([]) + ax.set_ylabel("Procedure-count\ngroup") + for rr in range(summary.shape[0]): + for cc in range(summary.shape[1]): + value = summary.iloc[rr, cc] + ax.text(cc, rr, "Not recorded" if not np.isfinite(value) else f"{value:.2f}", ha="center", va="center", fontsize=6.1) + panel_label(ax, label) + fig.subplots_adjust(left=0.20, right=0.98, top=0.95, bottom=0.24, hspace=0.38) + cax = fig.add_axes([0.30, 0.055, 0.42, 0.025]) + cbar = fig.colorbar(ims[0], cax=cax, orientation="horizontal") + cbar.set_label("Relative domain score (1-5)") + save_fig(fig, "figS_experience_domain_heatmap") + + fig, axes = plt.subplots(len(experience_cols), len(DATASET_ORDER), figsize=(9.4, 11.2), sharey=True) + for r, domain in enumerate(experience_cols): + for c, ds in enumerate(DATASET_ORDER): + ax = axes[r, c] + cohort = proxy[proxy["dataset"] == ds] + data = [cohort.loc[cohort["skill_category"] == skill, domain].dropna() for skill in SKILL_ORDER] + if sum(len(values) for values in data) == 0: + ax.set_axis_off() + ax.text(0.5, 0.5, "Jaw aperture not recorded", ha="center", va="center", transform=ax.transAxes, fontsize=7) + continue + bp = ax.boxplot(data, patch_artist=True, tick_labels=[nice(s) for s in SKILL_ORDER], showfliers=False, widths=0.58) + for patch, skill in zip(bp["boxes"], SKILL_ORDER): + patch.set_facecolor(SKILL_COLORS[skill]) + patch.set_alpha(0.68) + for i, values in enumerate(data): + if len(values) == 0: + continue + xs = np.full(len(values), i + 1) + np.linspace(-0.08, 0.08, len(values)) + ax.scatter(xs, values, s=6.5, color="#222222", alpha=0.34, linewidths=0) + ax.set_ylim(1, 5) + if r == 0: + ax.set_title(dataset_label(ds), fontsize=8) + if c == 0: + ax.set_ylabel(f"{display_labels.get(domain, domain).replace(chr(10), ' ')}\nscore (1-5)") + ax.tick_params(axis="x", rotation=20, labelsize=6.2) + panel_label(ax, chr(ord("a") + r * len(DATASET_ORDER) + c)) + fig.tight_layout(h_pad=1.08, w_pad=0.50) + save_fig(fig, "figS_experience_domain_boxplots") + + +def plot_kmeans_clusters(clusters: pd.DataFrame, summary: dict[str, object]) -> None: + if clusters.empty or not summary.get("available"): + return + fig, axes = plt.subplots(1, 3, figsize=(7.8, 3.35), gridspec_kw={"width_ratios": [1.25, 1.0, 1.0]}) + ax = axes[0] + for i, cl in enumerate(BAND_ORDER): + g = clusters[clusters["metric_cluster"] == cl].sort_values( + "Coordination-control composite" + ) + if g.empty: + continue + y = np.linspace(i - 0.16, i + 0.16, len(g)) + ax.scatter( + g["Coordination-control composite"], + y, + s=21, + alpha=0.82, + color=CLUSTER_COLORS[cl], + label=BAND_SHORT_LABELS[cl], + ) + for threshold in summary.get("metric_band_thresholds", []): + ax.axvline(threshold, color="#555555", lw=0.9, ls="--") + ax.set_xlim(1, 5) + ax.set_yticks(range(len(BAND_ORDER)), [BAND_SHORT_LABELS[b] for b in BAND_ORDER]) + ax.set_xlabel("Coordination and control score (1-5)") + ax.set_ylabel("Relative band") + panel_label(ax, "a") + + ax = axes[1] + validation = pd.DataFrame(summary.get("k_validation", [])) + if not validation.empty: + ax.plot(validation["k"], validation["silhouette"], color="#555555", marker="o", lw=1.1, ms=4) + selected = validation[validation["k"] == 3] + if not selected.empty: + ax.scatter(selected["k"], selected["silhouette"], s=42, color="#CC3311", zorder=3) + ax.set_xticks(range(2, 7)) + ax.set_xlabel("Number of bands, k") + ax.set_ylabel("Silhouette score") + panel_label(ax, "b") + + ax = axes[2] + for i, cl in enumerate(BAND_ORDER): + g = clusters[clusters["metric_cluster"] == cl] + if g.empty: + continue + values = np.log10(g["total_procedures"] + 1) + xs = np.full(len(g), i) + np.linspace(-0.10, 0.10, len(g)) + ax.scatter(xs, values, s=18, alpha=0.78, color=CLUSTER_COLORS[cl]) + ax.hlines(values.median(), i - 0.24, i + 0.24, colors="#222222", lw=1.2) + ax.set_xticks(range(len(BAND_ORDER)), [BAND_SHORT_LABELS[cl] for cl in BAND_ORDER]) + ax.set_ylabel("log10(total procedures + 1)") + ax.set_xlabel("Relative band") + panel_label(ax, "c") + fig.tight_layout() + save_fig(fig, "fig6_kmeans_procedure_thresholds") + + +def plot_motion_domains(df: pd.DataFrame) -> None: + panels = [ + ("Path length", "tool1_path_length", "tool2_path_length", "mm"), + ("Average speed", "tool1_avg_speed", "tool2_avg_speed", "mm/s"), + ("Speed variability", "tool1_speed_cv", "tool2_speed_cv", "CV"), + ("Acceleration variability", "tool1_accel_cv", "tool2_accel_cv", "CV"), + ("Normalised jerk", "tool1_normalized_jerk", "tool2_normalized_jerk", "a.u."), + ("Rotation per path", "tool1_rotation_per_path", "tool2_rotation_per_path", "degrees/mm"), + ("Angular velocity variability", "tool1_angular_velocity_cv", "tool2_angular_velocity_cv", "CV"), + ("Workspace depth range", "tool1_range_z", "tool2_range_z", "mm"), + ("Working area", "tool1_working_area_xy", "tool2_working_area_xy", "mm$^2$"), + ("Movement count", "tool1_num_movements", "tool2_num_movements", "count"), + ("Idle episodes", "tool1_idle_episode_count", "tool2_idle_episode_count", "count"), + ("Speed peaks", "tool1_num_speed_peaks", "tool2_num_speed_peaks", "count"), + ] + for ds in DATASET_ORDER: + cohort = df[df["dataset"] == ds] + if cohort.empty: + continue + fig, axes = plt.subplots(3, 4, figsize=(9.4, 6.7)) + axes = axes.ravel() + for ax, (title, c1, c2, unit), label in zip(axes, panels, list("abcdefghijkl")): + plot_df = cohort[["skill_category", c1, c2]].copy() + plot_df["metric"] = plot_df[[c1, c2]].mean(axis=1, skipna=True) + data = [plot_df.loc[plot_df["skill_category"] == s, "metric"].dropna() for s in SKILL_ORDER] + bp = ax.boxplot(data, patch_artist=True, tick_labels=[nice(s) for s in SKILL_ORDER], showfliers=False) + for patch, skill in zip(bp["boxes"], SKILL_ORDER): + patch.set_facecolor(SKILL_COLORS[skill]) + patch.set_alpha(0.65) + ax.set_ylabel(unit) + ax.set_title(title, fontsize=8) + ax.tick_params(axis="x", rotation=20) + panel_label(ax, label) + for ax in axes[len(panels) :]: + ax.axis("off") + fig.tight_layout() + save_fig(fig, f"fig7_motion_component_domains_{dataset_slug(ds)}") + + +def plot_jaw_and_rotation(df: pd.DataFrame) -> None: + fig, axes = plt.subplots(1, 3, figsize=(7.4, 2.9)) + jaw_df = df[df["dataset"].isin(["7DOF2024", "BAPES2024"])].copy() + panels = [ + ("Jaw signal range", "tool1_jaw_angle_range", "tool2_jaw_angle_range", "Voltage range"), + ("Detected aperture cycles", "tool1_jaw_open_close_count", "tool2_jaw_open_close_count", "Cycles"), + ("Angular velocity variability", "tool1_angular_velocity_cv", "tool2_angular_velocity_cv", "CV"), + ] + for ax, (title, c1, c2, ylabel), label in zip(axes, panels, ["a", "b", "c"]): + source = jaw_df if "jaw" in c1 else df + order = ["BAPES2024", "7DOF2024"] if "jaw" in c1 else DATASET_ORDER + source = source.copy() + source["metric"] = source[[c1, c2]].mean(axis=1, skipna=True) + for i, ds in enumerate(order): + g = source[source["dataset"] == ds]["metric"].dropna() + if len(g) == 0: + continue + x = np.full(len(g), i) + np.linspace(-0.12, 0.12, len(g)) + ax.scatter(x, g, s=15, alpha=0.75, color=DATASET_COLORS[ds]) + lo, hi = bootstrap_ci(g) + ax.errorbar(i, g.mean(), yerr=[[g.mean() - lo], [hi - g.mean()]], color="#333333", capsize=3, fmt="o", ms=3) + ax.set_xticks(range(len(order)), [dataset_label(ds) for ds in order], rotation=25, ha="right") + ax.set_ylabel(ylabel) + ax.set_title(title, fontsize=8) + panel_label(ax, label) + fig.tight_layout() + save_fig(fig, "fig8_jaw_rotation_components") + + +def plot_workspace_heatmaps(motion: pd.DataFrame) -> None: + if motion.empty: + return + long_rows = [] + for tool in ["tool1", "tool2"]: + sub = motion[["dataset", "cohort_label", "skill_category", f"{tool}_x", f"{tool}_y", f"{tool}_z"]].copy() + sub.columns = ["dataset", "cohort_label", "skill_category", "x", "y", "z"] + sub["tool"] = tool + long_rows.append(sub) + pos = pd.concat(long_rows, ignore_index=True).dropna(subset=["x", "y", "z"]) + planes = [("x", "y", "Camera X (mm)", "Camera Y (mm)"), ("x", "z", "Camera X (mm)", "Depth Z (mm)"), ("y", "z", "Camera Y (mm)", "Depth Z (mm)")] + fig, axes = plt.subplots(len(planes), len(DATASET_ORDER), figsize=(7.4, 6.4), sharex=False, sharey=False) + mesh = None + for r, (a, b, xlabel, ylabel) in enumerate(planes): + for c, ds in enumerate(DATASET_ORDER): + ax = axes[r, c] + g = pos[pos["dataset"] == ds] + density, a_edges, b_edges = np.histogram2d(g[a], g[b], bins=48) + peak = density.max() + relative_density = density / peak if peak > 0 else density + relative_density = np.ma.masked_where(relative_density <= 0, relative_density) + mesh = ax.pcolormesh( + a_edges, + b_edges, + relative_density.T, + cmap="viridis", + vmin=0, + vmax=1, + shading="auto", + ) + if r == 0: + ax.set_title(dataset_label(ds), fontsize=8) + if c == 0: + ax.set_ylabel(ylabel) + panel_label(ax, chr(ord("a") + r)) + ax.set_xlabel(xlabel if r == len(planes) - 1 else "") + ax.tick_params(labelsize=6) + fig.subplots_adjust(left=0.10, right=0.98, top=0.96, bottom=0.16, wspace=0.24, hspace=0.28) + if mesh is not None: + cax = fig.add_axes([0.30, 0.055, 0.40, 0.018]) + colourbar = fig.colorbar(mesh, cax=cax, orientation="horizontal") + colourbar.set_label("Relative sampling density within each panel", fontsize=7) + colourbar.ax.tick_params(labelsize=6) + save_fig(fig, "fig9_workspace_heatmaps_camera_planes") + + +def plot_skill_workspace_xy(motion: pd.DataFrame) -> None: + if motion.empty: + return + rows = [] + for tool in ["tool1", "tool2"]: + sub = motion[["dataset", "skill_category", f"{tool}_x", f"{tool}_y"]].copy() + sub.columns = ["dataset", "skill_category", "x", "y"] + rows.append(sub) + pos = pd.concat(rows, ignore_index=True).dropna() + fig, axes = plt.subplots(len(SKILL_ORDER), len(DATASET_ORDER), figsize=(7.4, 6.4), sharex=False, sharey=False) + mesh = None + for r, skill in enumerate(SKILL_ORDER): + for c, ds in enumerate(DATASET_ORDER): + ax = axes[r, c] + g = pos[(pos["dataset"] == ds) & (pos["skill_category"] == skill)] + if len(g): + density, x_edges, y_edges = np.histogram2d(g["x"], g["y"], bins=42) + peak = density.max() + relative_density = density / peak if peak > 0 else density + relative_density = np.ma.masked_where(relative_density <= 0, relative_density) + mesh = ax.pcolormesh( + x_edges, + y_edges, + relative_density.T, + cmap="magma", + vmin=0, + vmax=1, + shading="auto", + ) + if r == 0: + ax.set_title(dataset_label(ds), fontsize=8) + if c == 0: + ax.set_ylabel(f"{nice(skill)}\nCamera Y (mm)") + if r == len(SKILL_ORDER) - 1: + ax.set_xlabel("Camera X (mm)") + ax.tick_params(labelsize=6) + fig.subplots_adjust(left=0.11, right=0.98, top=0.96, bottom=0.16, wspace=0.24, hspace=0.28) + if mesh is not None: + cax = fig.add_axes([0.30, 0.055, 0.40, 0.018]) + colourbar = fig.colorbar(mesh, cax=cax, orientation="horizontal") + colourbar.set_label("Relative sampling density within each panel", fontsize=7) + colourbar.ax.tick_params(labelsize=6) + save_fig(fig, "fig10_skill_workspace_xy_heatmaps") + + +def plot_phase_by_cluster(trials: pd.DataFrame, clusters: pd.DataFrame) -> None: + if clusters.empty or "metric_cluster" not in clusters: + return + phase_cols = ["frac_reach", "frac_grasp", "frac_transfer", "frac_place", "frac_nudge", "frac_idle"] + phase = trials.copy() + phase["trial_key"] = [phase_trial_key(ds, ts) for ds, ts in zip(phase["dataset"], phase["trial_short"])] + cluster_lookup = clusters[["trial_key", "metric_cluster"]].drop_duplicates("trial_key") + merged = phase.merge(cluster_lookup, on="trial_key", how="left") + fig, axes = plt.subplots(1, 2, figsize=(6.9, 2.65)) + for ax, group_col, labels, label in [ + (axes[0], "skill_category", SKILL_ORDER, "a"), + (axes[1], "metric_cluster", BAND_ORDER, "b"), + ]: + means = merged.groupby(group_col)[phase_cols].mean().reindex(labels) + left = np.zeros(len(means)) + y = np.arange(len(means)) + for col in phase_cols: + phase = col.replace("frac_", "") + vals = means[col].fillna(0).to_numpy() + ax.barh(y, vals, left=left, color=PHASE_COLORS[phase], label=nice(phase), height=0.68) + left += vals + tick_labels = [BAND_SHORT_LABELS.get(s, nice(s)) for s in labels] + ax.set_yticks(y, tick_labels, fontsize=7) + ax.set_xlabel("Mean fraction of trial time", fontsize=8) + ax.set_xlim(0, 1) + ax.invert_yaxis() + panel_label(ax, label) + handles, labels = axes[1].get_legend_handles_labels() + fig.legend(handles, labels, frameon=False, ncol=len(labels), fontsize=6.8, loc="lower center", bbox_to_anchor=(0.5, 0.03)) + fig.subplots_adjust(left=0.16, right=0.98, top=0.90, bottom=0.30, wspace=0.36) + save_fig(fig, "fig11_phase_allocation_original_vs_cluster") + + +def plot_ssl_embeddings(ssl_df: pd.DataFrame) -> None: + if ssl_df.empty or "ssl_pc1" not in ssl_df.columns: + return + fig, axes = plt.subplots(1, 3, figsize=(7.4, 3.25)) + ax = axes[0] + for ds in DATASET_ORDER: + g = ssl_df[ssl_df["dataset"] == ds] + ax.scatter(g["ssl_pc1"], g["ssl_pc2"], s=18, alpha=0.8, color=DATASET_COLORS[ds], label=dataset_label(ds)) + ax.set_xlabel("Embedding PC1") + ax.set_ylabel("Embedding PC2") + ax.legend(frameon=False, fontsize=6) + panel_label(ax, "a") + + ax = axes[1] + for skill in SKILL_ORDER: + g = ssl_df[ssl_df["skill_category"] == skill] + ax.scatter(g["ssl_pc1"], g["ssl_pc2"], s=18, alpha=0.75, color=SKILL_COLORS[skill], label=nice(skill)) + ax.set_xlabel("Embedding PC1") + ax.set_ylabel("Embedding PC2") + panel_label(ax, "b") + + ax = axes[2] + sc = ax.scatter( + ssl_df["ssl_pc1"], + ssl_df["ssl_pc2"], + c=np.log10(ssl_df["total_procedures"] + 1), + cmap="viridis", + s=18, + alpha=0.8, + ) + ax.set_xlabel("Embedding PC1") + ax.set_ylabel("Embedding PC2") + panel_label(ax, "c") + fig.subplots_adjust(left=0.08, right=0.98, top=0.93, bottom=0.34, wspace=0.34) + cax = fig.add_axes([0.32, 0.11, 0.36, 0.045]) + cbar = fig.colorbar(sc, cax=cax, orientation="horizontal") + cbar.set_label("log10(total procedures + 1)") + save_fig(fig, "fig12_unsupervised_motion_embeddings") + + +def plot_tool_specific_motion_boxplots(df: pd.DataFrame) -> None: + panels = [ + ("Path length", "path_length", "mm", False), + ("Average speed", "avg_speed", "mm/s", True), + ("Speed variability", "speed_cv", "CV", False), + ("Normalised jerk", "normalized_jerk", "a.u.", False), + ("Rotation per path", "rotation_per_path", "degrees/mm", False), + ("Depth range", "range_z", "mm", False), + ("Working area", "working_area_xy", "mm$^2$", False), + ("Movement count", "num_movements", "count", False), + ] + offsets = {"novice": -0.22, "intermediate": 0.0, "expert": 0.22} + for ds in DATASET_ORDER: + cohort = df[df["dataset"] == ds] + if cohort.empty: + continue + fig, axes = plt.subplots(3, 3, figsize=(9.1, 7.4), sharex=False) + axes = axes.ravel() + for ax, (title, suffix, ylabel, higher_is_more), label in zip(axes, panels, list("abcdefgh")): + positions = [] + data = [] + colors = [] + for base, tool in enumerate(["tool1", "tool2"]): + col = f"{tool}_{suffix}" + if col not in cohort.columns: + continue + for skill in SKILL_ORDER: + values = cohort.loc[cohort["skill_category"] == skill, col].dropna() + data.append(values) + positions.append(base + offsets[skill]) + colors.append(SKILL_COLORS[skill]) + if not data: + ax.axis("off") + continue + bp = ax.boxplot(data, positions=positions, widths=0.16, patch_artist=True, showfliers=False) + for patch, color in zip(bp["boxes"], colors): + patch.set_facecolor(color) + patch.set_alpha(0.65) + direction = "$\\uparrow$" if higher_is_more else "$\\downarrow$" + ax.set_title(f"{title} ({direction})", fontsize=8) + ax.set_xticks([0, 1], ["Tool 1", "Tool 2"]) + ax.set_ylabel(ylabel) + panel_label(ax, label) + for ax in axes[len(panels) :]: + ax.axis("off") + handles = [plt.Rectangle((0, 0), 1, 1, color=SKILL_COLORS[s], alpha=0.65) for s in SKILL_ORDER] + fig.legend(handles, [nice(s) for s in SKILL_ORDER], frameon=False, ncol=3, loc="lower center", bbox_to_anchor=(0.5, 0.02)) + fig.subplots_adjust(left=0.07, right=0.99, top=0.93, bottom=0.12, wspace=0.34, hspace=0.44) + save_fig(fig, f"fig13_tool_specific_motion_boxplots_{dataset_slug(ds)}") + + +def plot_feature_importance() -> None: + path = OUT / "skill_feature_importance.csv" + if not path.exists(): + return + imp = pd.read_csv(path).head(20).sort_values("importance") + fig, ax = plt.subplots(figsize=(7.2, 5.8)) + y = np.arange(len(imp)) + ax.barh(y, imp["importance"], color="#4C78A8") + ax.set_yticks(y, imp["label"]) + ax.set_xlabel("Random forest feature importance") + ax.set_ylabel("Kinematic feature") + fig.tight_layout() + save_fig(fig, "fig14_skill_feature_importance") + + +def plot_sorted_participant_bars(df: pd.DataFrame, proxy: pd.DataFrame) -> None: + merged = df.copy() + merged["Coordination-control composite"] = ( + proxy["Coordination-control composite"].to_numpy() if len(proxy) == len(df) else np.nan + ) + specs = [ + ("Coordination-control composite", "Coordination and\ncontrol score ($\\uparrow$)", True), + ("bimanual_correlation", "Bimanual\ncorrelation ($\\uparrow$)", True), + ("combined_idle_ratio", "Idle fraction ($\\downarrow$)", False), + ("tool1_normalized_jerk", "Tool 1 normalised\njerk ($\\downarrow$)", False), + ("tool2_normalized_jerk", "Tool 2 normalised\njerk ($\\downarrow$)", False), + ("total_time", "Analysed\nduration ($\\downarrow$)", False), + ] + fig, axes = plt.subplots(len(specs), len(DATASET_ORDER), figsize=(9.4, 9.2), sharex=False) + for r, (col, ylabel, higher_better) in enumerate(specs): + for c, ds in enumerate(DATASET_ORDER): + ax = axes[r, c] + if col not in merged.columns: + ax.axis("off") + continue + d = ( + merged.loc[merged["dataset"] == ds, ["skill_category", "dataset", col]] + .dropna() + .sort_values(col, ascending=not higher_better) + ) + colors = [SKILL_COLORS.get(s, "#999999") for s in d["skill_category"]] + ax.bar(np.arange(len(d)), d[col], color=colors, width=0.85) + if c == 0: + ax.set_ylabel( + ylabel, + rotation=0, + ha="right", + va="center", + labelpad=24, + fontsize=7, + ) + if r == 0: + ax.set_title(dataset_label(ds), fontsize=8) + ax.set_xticks([]) + panel_label(ax, chr(ord("a") + r * len(DATASET_ORDER) + c)) + handles = [plt.Rectangle((0, 0), 1, 1, color=SKILL_COLORS[s]) for s in SKILL_ORDER] + fig.legend(handles, [nice(s) for s in SKILL_ORDER], frameon=False, ncol=3, loc="lower center", bbox_to_anchor=(0.5, 0.02)) + fig.subplots_adjust(left=0.20, right=0.99, top=0.94, bottom=0.08, wspace=0.24, hspace=0.46) + save_fig(fig, "fig15_sorted_trial_metric_bars") + + +def _normalised_cohort_means(df: pd.DataFrame, specs: list[tuple[str, str, int]]) -> pd.DataFrame: + rows = [] + for col, label, direction in specs: + if col not in df.columns: + continue + x = pd.to_numeric(df[col], errors="coerce") * direction + finite = x[np.isfinite(x)] + if finite.empty: + continue + lo, hi = finite.quantile([0.05, 0.95]) + denom = hi - lo if hi != lo else 1.0 + scaled = ((x - lo) / denom).clip(0, 1) + for ds in DATASET_ORDER: + rows.append({"dataset": ds, "metric": label, "value": float(scaled[df["dataset"] == ds].mean())}) + return pd.DataFrame(rows) + + +def plot_cohort_radar(df: pd.DataFrame) -> None: + specs = [ + ("bimanual_correlation", "Bimanual\nsynchrony", 1), + ("simultaneous_motion_ratio", "Simultaneous\nmotion", 1), + ("combined_idle_ratio", "Low idle\ntime", -1), + ("total_time", "Shorter\ntrial", -1), + ("tool1_normalized_jerk", "Tool 1\nnormalised jerk", -1), + ("tool2_normalized_jerk", "Tool 2\nnormalised jerk", -1), + ("tool1_working_volume", "Tool 1 compact\nworkspace", -1), + ("tool2_working_volume", "Tool 2 compact\nworkspace", -1), + ] + norm = _normalised_cohort_means(df, specs) + if norm.empty: + return + metrics = [label for _, label, _ in specs if label in set(norm["metric"])] + theta = np.linspace(0, 2 * np.pi, len(metrics), endpoint=False) + fig = plt.figure(figsize=(7.2, 5.9)) + ax = fig.add_subplot(111, polar=True) + for ds in DATASET_ORDER: + vals = [] + for metric in metrics: + value = norm[(norm["dataset"] == ds) & (norm["metric"] == metric)]["value"] + vals.append(float(value.iloc[0]) if len(value) else np.nan) + vals = np.asarray(vals, dtype=float) + vals = np.nan_to_num(vals, nan=np.nanmean(vals) if np.isfinite(vals).any() else 0.0) + ax.plot(np.r_[theta, theta[0]], np.r_[vals, vals[0]], color=DATASET_COLORS[ds], lw=1.4, label=dataset_label(ds)) + ax.fill(np.r_[theta, theta[0]], np.r_[vals, vals[0]], color=DATASET_COLORS[ds], alpha=0.08) + ax.set_xticks(theta, metrics) + ax.set_ylim(0, 1) + ax.set_yticks([0.25, 0.5, 0.75], ["0.25", "0.50", "0.75"]) + ax.tick_params(axis="x", pad=10, labelsize=14) + ax.tick_params(axis="y", labelsize=12) + ax.legend(frameon=False, loc="upper center", bbox_to_anchor=(0.5, -0.12), ncol=3, fontsize=13) + fig.tight_layout(rect=(0, 0.08, 1, 1)) + save_fig(fig, "fig16_cohort_radar_summary") + + +def plot_dataset_metric_matrix(df: pd.DataFrame, proxy: pd.DataFrame) -> None: + merged = df.copy() + merged["Coordination-control composite"] = ( + proxy["Coordination-control composite"].to_numpy() if len(proxy) == len(df) else np.nan + ) + specs = [ + ("Coordination-control composite", "Coordination and control score", 1), + ("total_time", "Analysed duration", -1), + ("bimanual_correlation", "Bimanual correlation", 1), + ("bimanual_lag_s", "Bimanual lag", -1), + ("bimanual_concurrent_efficiency", "Concurrent efficiency", 1), + ("simultaneous_motion_ratio", "Simultaneous motion", 1), + ("combined_idle_ratio", "Idle fraction", -1), + ("tool_distance_cv", "Tool distance variability", -1), + ("tool_close_proximity_ratio", "Close proximity", -1), + ("tool1_normalized_jerk", "Tool 1 normalised jerk", -1), + ("tool2_normalized_jerk", "Tool 2 normalised jerk", -1), + ("tool1_avg_speed", "Tool 1 speed", 1), + ("tool2_avg_speed", "Tool 2 speed", 1), + ("tool1_path_length", "Tool 1 path length", -1), + ("tool2_path_length", "Tool 2 path length", -1), + ("tool1_rotation_per_path", "Tool 1 rotation economy", -1), + ("tool2_rotation_per_path", "Tool 2 rotation economy", -1), + ("tool1_range_z", "Tool 1 depth range", -1), + ("tool2_range_z", "Tool 2 depth range", -1), + ("tool1_working_volume", "Tool 1 working volume", -1), + ("tool2_working_volume", "Tool 2 working volume", -1), + ("tool1_num_speed_peaks", "Tool 1 speed peaks", -1), + ("tool2_num_speed_peaks", "Tool 2 speed peaks", -1), + ] + rows = [] + for col, label, direction in specs: + if col not in merged.columns: + continue + vals = pd.to_numeric(merged[col], errors="coerce") * direction + sd = vals.std(ddof=0) + z = (vals - vals.mean()) / (sd if sd and np.isfinite(sd) else 1.0) + for ds in DATASET_ORDER: + rows.append({"metric": label, "dataset": ds, "z": float(z[merged["dataset"] == ds].mean())}) + mat_df = pd.DataFrame(rows) + if mat_df.empty: + return + metrics = mat_df["metric"].drop_duplicates().to_list() + mat = np.full((len(DATASET_ORDER), len(metrics)), np.nan) + for r, metric in enumerate(metrics): + for c, ds in enumerate(DATASET_ORDER): + v = mat_df[(mat_df["metric"] == metric) & (mat_df["dataset"] == ds)]["z"] + mat[c, r] = float(v.iloc[0]) if len(v) else np.nan + fig = plt.figure(figsize=(10.8, 3.8)) + ax = fig.add_axes([0.08, 0.30, 0.90, 0.38]) + im = ax.imshow(mat, cmap="RdBu_r", vmin=-1.2, vmax=1.2, aspect="auto") + ax.set_xticks(np.arange(len(metrics)), metrics, rotation=45, ha="left") + ax.xaxis.tick_top() + ax.tick_params(axis="x", top=True, bottom=False, labeltop=True, labelbottom=False, pad=1) + ax.set_yticks(np.arange(len(DATASET_ORDER)), [dataset_label(ds) for ds in DATASET_ORDER]) + for r in range(mat.shape[0]): + for c in range(mat.shape[1]): + if np.isfinite(mat[r, c]): + ax.text(c, r, f"{mat[r, c]:.1f}", ha="center", va="center", fontsize=6) + else: + ax.text(c, r, "NA", ha="center", va="center", fontsize=6, color="#555555") + cax = fig.add_axes([0.24, 0.10, 0.52, 0.055]) + cbar = fig.colorbar(im, cax=cax, orientation="horizontal") + cbar.set_label("Cohort mean, oriented z-score") + save_fig(fig, "fig17_dataset_metric_matrix") + + +def plot_procedure_motion_disagreement(clusters: pd.DataFrame) -> None: + """Show how self-reported experience and measured motion bands diverge.""" + if clusters.empty or "metric_cluster" not in clusters: + return + d = clusters.dropna(subset=["skill_category", "metric_cluster"]).copy() + if d.empty: + return + matrix = ( + pd.crosstab(d["skill_category"], d["metric_cluster"]) + .reindex(index=SKILL_ORDER, columns=BAND_ORDER) + .fillna(0) + .astype(int) + ) + mismatch_specs = [ + ("Novice in\nupper band", (d["skill_category"] == "novice") & (d["metric_cluster"] == "metric_high"), int((d["skill_category"] == "novice").sum())), + ("Expert in\nlower band", (d["skill_category"] == "expert") & (d["metric_cluster"] == "metric_low"), int((d["skill_category"] == "expert").sum())), + ("Novice outside\nlower band", (d["skill_category"] == "novice") & (d["metric_cluster"] != "metric_low"), int((d["skill_category"] == "novice").sum())), + ("Expert outside\nupper band", (d["skill_category"] == "expert") & (d["metric_cluster"] != "metric_high"), int((d["skill_category"] == "expert").sum())), + ] + + fig, axes = plt.subplots(1, 2, figsize=(7.4, 3.4), gridspec_kw={"width_ratios": [1.0, 1.25]}) + ax = axes[0] + im = ax.imshow(matrix.to_numpy(), cmap="YlGnBu", vmin=0) + ax.set_xticks(np.arange(len(BAND_ORDER)), [BAND_SHORT_LABELS[b].replace(" ", "\n") for b in BAND_ORDER]) + ax.set_yticks(np.arange(len(SKILL_ORDER)), [nice(s) for s in SKILL_ORDER]) + ax.set_xlabel("Motion-score band") + ax.set_ylabel("Procedure-count label") + for r in range(matrix.shape[0]): + for c in range(matrix.shape[1]): + value = int(matrix.iloc[r, c]) + ax.text(c, r, str(value), ha="center", va="center", fontsize=9, weight="bold", color="#111111") + cbar = fig.colorbar(im, ax=ax, fraction=0.046, pad=0.03) + cbar.set_label("Trials") + panel_label(ax, "a") + + ax = axes[1] + counts = np.array([int(mask.sum()) for _, mask, _ in mismatch_specs], dtype=float) + denoms = np.array([denom if denom else np.nan for _, _, denom in mismatch_specs], dtype=float) + perc = 100 * counts / denoms + colors = ["#4C78A8", "#E45756", "#72B7B2", "#F58518"] + x = np.arange(len(mismatch_specs)) + ax.bar(x, counts, color=colors, width=0.72) + for i, (count, pct) in enumerate(zip(counts, perc)): + ax.text(i, count + 0.7, f"{int(count)}\n({pct:.0f}%)", ha="center", va="bottom", fontsize=7) + ax.set_xticks(x, [label for label, _, _ in mismatch_specs]) + ax.set_ylabel("Trials") + ax.set_ylim(0, max(counts) * 1.22 if len(counts) else 1) + panel_label(ax, "b") + fig.tight_layout() + save_fig(fig, "fig18_procedure_motion_disagreement") + + +def plot_feedback_targets(clusters: pd.DataFrame) -> None: + """Plot prevalence and magnitude of domain scores below the cohort midpoint.""" + domain_cols = [c for c in OSATS_DOMAINS if c in clusters.columns] + if clusters.empty or not domain_cols: + return + d = clusters.dropna(subset=["metric_cluster", "cohort_label"]).copy() + domain_order = [ + x + for x in [*CORE_PERFORMANCE_DOMAINS, "Workspace excursion"] + if x in domain_cols + ] + short = { + "Bimanual coordination": "Bimanual\ncoordination", + "Task efficiency": "Task\nefficiency", + "Instrument motion control": "Instrument\nmotion control", + "Workspace excursion": "Workspace\nexcursion", + } + midpoint = 3.0 + prevalence = pd.DataFrame(index=domain_order, columns=BAND_ORDER, dtype=float) + deficit = pd.DataFrame(index=domain_order, columns=BAND_ORDER, dtype=float) + for domain in domain_order: + for band in BAND_ORDER: + values = d.loc[d["metric_cluster"] == band, domain].dropna() + prevalence.loc[domain, band] = 100 * (values < midpoint).mean() if len(values) else np.nan + deficit.loc[domain, band] = np.maximum(0, midpoint - values).mean() if len(values) else np.nan + + fig, axes = plt.subplots(1, 2, figsize=(7.7, 3.85), gridspec_kw={"width_ratios": [0.95, 1.35]}) + ax = axes[0] + im = ax.imshow(prevalence.to_numpy(float), vmin=0, vmax=100, cmap="YlOrRd", aspect="auto") + ax.set_xticks(range(len(BAND_ORDER)), [BAND_SHORT_LABELS[b] for b in BAND_ORDER]) + ax.set_yticks(range(len(domain_order)), [short[d].replace("\n", " ") for d in domain_order]) + for r in range(len(domain_order)): + for c in range(len(BAND_ORDER)): + value = prevalence.iloc[r, c] + ax.text(c, r, "NA" if not np.isfinite(value) else f"{value:.0f}%", ha="center", va="center", fontsize=7) + cbar = fig.colorbar(im, ax=ax, orientation="horizontal", pad=0.20, fraction=0.08) + cbar.set_label("Trials below score 3.0") + panel_label(ax, "a") + + ax = axes[1] + x = np.arange(len(domain_order)) + width = 0.24 + for i, band in enumerate(BAND_ORDER): + ax.bar( + x + (i - 1) * width, + deficit[band].to_numpy(float), + width=width, + color=CLUSTER_COLORS[band], + label=BAND_SHORT_LABELS[band], + ) + ax.set_xticks(x, [short[domain] for domain in domain_order]) + ax.set_ylabel("Mean shortfall below score 3.0") + ax.legend(frameon=False, fontsize=6.6, loc="upper right") + panel_label(ax, "b") + fig.subplots_adjust(left=0.12, right=0.98, top=0.91, bottom=0.22, wspace=0.38) + save_fig(fig, "fig19_feedback_targets") + + +def _phase_trials_with_clusters(trials: pd.DataFrame, clusters: pd.DataFrame) -> pd.DataFrame: + phase = trials.copy() + phase["trial_key"] = [phase_trial_key(ds, ts) for ds, ts in zip(phase["dataset"], phase["trial_short"])] + band_col = "metric_cluster" + cluster_lookup = ( + clusters[["dataset", "trial_name", band_col, "cohort_label"]] + .rename(columns={band_col: "metric_cluster"}) + .drop_duplicates(["dataset", "trial_name"]) + ) + phase = phase.merge( + cluster_lookup, + left_on=["dataset", "trial_short"], + right_on=["dataset", "trial_name"], + how="left", + validate="one_to_one", + ) + identifier_count = phase.groupby("trial_key")["trial_key"].transform("size") + return phase.loc[identifier_count.eq(1)].copy() + + +def _phase_cycles_with_clusters(cycles: pd.DataFrame, clusters: pd.DataFrame) -> pd.DataFrame: + cyc = cycles.copy() + cyc["trial_key"] = [ + phase_trial_key(ds, trial_short) + for ds, trial_short in zip(cyc["dataset"], cyc["trial_short"]) + ] + band_col = "metric_cluster" + cluster_lookup = ( + clusters[["dataset", "trial_name", band_col, "cohort_label"]] + .rename(columns={band_col: "metric_cluster"}) + .drop_duplicates(["dataset", "trial_name"]) + ) + cyc = cyc.merge( + cluster_lookup, + left_on=["dataset", "trial_short"], + right_on=["dataset", "trial_name"], + how="left", + validate="many_to_one", + ) + identifier_recordings = ( + cyc[["trial_key", "trial_id"]].drop_duplicates().groupby("trial_key")["trial_id"].transform("size") + ) + unique_keys = set( + cyc[["trial_key", "trial_id"]] + .drop_duplicates() + .loc[identifier_recordings.eq(1), "trial_key"] + ) + return cyc.loc[cyc["trial_key"].isin(unique_keys)].copy() + + +def plot_phase_bottlenecks(trials: pd.DataFrame, cycles: pd.DataFrame, clusters: pd.DataFrame) -> None: + """Compare phase timing across descriptive motion-score bands.""" + if clusters.empty: + return + phase = _phase_trials_with_clusters(trials, clusters).dropna(subset=["metric_cluster"]) + cyc = _phase_cycles_with_clusters(cycles, clusters).dropna(subset=["metric_cluster"]) + if phase.empty or cyc.empty: + return + trial_cycle = ( + cyc.groupby(["metric_cluster", "trial_key"], as_index=False) + .agg( + mean_cycle_duration_s=("duration_s", "mean"), + mean_transitions=("n_transitions", "mean"), + ) + ) + fig, axes = plt.subplots( + 1, + 4, + figsize=(9.4, 3.35), + gridspec_kw={"width_ratios": [0.9, 1.2, 0.9, 0.9]}, + ) + + ax = axes[0] + means = [] + los = [] + his = [] + for band in BAND_ORDER: + vals = trial_cycle.loc[ + trial_cycle["metric_cluster"] == band, + "mean_cycle_duration_s", + ].dropna() + mean = vals.mean() + lo, hi = bootstrap_ci(vals) + means.append(mean) + los.append(mean - lo) + his.append(hi - mean) + x = np.arange(len(BAND_ORDER)) + ax.bar(x, means, color=[CLUSTER_COLORS[b] for b in BAND_ORDER], width=0.72) + ax.errorbar(x, means, yerr=[los, his], color="#222222", fmt="none", capsize=3, lw=0.8) + ax.set_xticks(x, [BAND_SHORT_LABELS[b].replace(" ", "\n") for b in BAND_ORDER]) + ax.set_ylabel("Cycle duration (s)") + panel_label(ax, "a") + + ax = axes[1] + phase_cols = ["frac_reach", "frac_grasp", "frac_transfer", "frac_place", "frac_nudge", "frac_idle", "frac_dropped"] + means_df = phase.groupby("metric_cluster")[phase_cols].mean().reindex(BAND_ORDER).fillna(0) + left = np.zeros(len(BAND_ORDER)) + y = np.arange(len(BAND_ORDER)) + for col in phase_cols: + phase_name = col.replace("frac_", "") + vals = means_df[col].to_numpy() + ax.barh(y, vals, left=left, color=PHASE_COLORS[phase_name], label=nice(phase_name), height=0.68) + left += vals + ax.set_yticks(y, [BAND_SHORT_LABELS[b] for b in BAND_ORDER]) + ax.set_xlabel("Mean fraction of corrected trial time") + ax.set_xlim(0, 1) + ax.invert_yaxis() + panel_label(ax, "b") + + ax = axes[2] + trans_means = ( + trial_cycle.groupby("metric_cluster")["mean_transitions"] + .mean() + .reindex(BAND_ORDER) + ) + trans_los = [] + trans_his = [] + for band in BAND_ORDER: + vals = trial_cycle.loc[ + trial_cycle["metric_cluster"] == band, + "mean_transitions", + ].dropna() + mean = vals.mean() + lo, hi = bootstrap_ci(vals) + trans_los.append(mean - lo) + trans_his.append(hi - mean) + ax.bar(x, trans_means, color=[CLUSTER_COLORS[b] for b in BAND_ORDER], width=0.72) + ax.errorbar(x, trans_means, yerr=[trans_los, trans_his], color="#222222", fmt="none", capsize=3, lw=0.8) + ax.set_xticks(x, [BAND_SHORT_LABELS[b].replace(" ", "\n") for b in BAND_ORDER]) + ax.set_ylabel("Phase transitions per cycle") + panel_label(ax, "c") + + ax = axes[3] + trial_drops = ( + cyc.groupby(["metric_cluster", "trial_key"], as_index=False) + .agg( + n_cycles=("cycle_index", "count"), + n_drop_events=("n_drop_events", "sum"), + ) + ) + trial_drops["drop_events_per_100_cycles"] = ( + 100 * trial_drops["n_drop_events"] / trial_drops["n_cycles"] + ) + drop_means = [] + drop_los = [] + drop_his = [] + for band in BAND_ORDER: + values = trial_drops.loc[ + trial_drops["metric_cluster"] == band, + "drop_events_per_100_cycles", + ].dropna() + mean = values.mean() + lo, hi = bootstrap_ci(values) + drop_means.append(mean) + drop_los.append(mean - lo) + drop_his.append(hi - mean) + ax.bar(x, drop_means, color=[CLUSTER_COLORS[b] for b in BAND_ORDER], width=0.72) + ax.errorbar( + x, + drop_means, + yerr=[drop_los, drop_his], + color="#222222", + fmt="none", + capsize=3, + lw=0.8, + ) + ax.set_xticks(x, [BAND_SHORT_LABELS[b].replace(" ", "\n") for b in BAND_ORDER]) + ax.set_ylabel("Drop episodes per 100 cycles") + panel_label(ax, "d") + + handles, labels = axes[1].get_legend_handles_labels() + fig.legend(handles, labels, frameon=False, ncol=7, fontsize=6.4, loc="lower center", bbox_to_anchor=(0.5, 0.02)) + fig.subplots_adjust(left=0.065, right=0.99, top=0.90, bottom=0.24, wspace=0.43) + save_fig(fig, "fig20_phase_bottlenecks") + + +def plot_cohort_effect_forest(df: pd.DataFrame) -> None: + """Cohort-specific expert-versus-novice effects for clinically important metrics.""" + identifier_counts = df.groupby("trial_key")["trial_key"].transform("size") + df = df.loc[identifier_counts.eq(1)].copy() + specs = [ + ("tool2_avg_speed", "Tool 2 speed", 1), + ("tool1_avg_speed", "Tool 1 speed", 1), + ("bimanual_correlation", "Bimanual correlation", 1), + ("tool2_num_speed_peaks", "Tool 2 stop-start peaks", -1), + ("tool1_num_speed_peaks", "Tool 1 stop-start peaks", -1), + ("total_time", "Analysed duration", -1), + ("tool2_range_z", "Tool 2 depth range", -1), + ] + rows = [] + rng = np.random.default_rng(20260618) + for col, label, direction in specs: + if col not in df.columns: + continue + for ds in DATASET_ORDER: + g = df[df["dataset"] == ds] + nov = (pd.to_numeric(g.loc[g["skill_category"] == "novice", col], errors="coerce") * direction).dropna().to_numpy() + exp = (pd.to_numeric(g.loc[g["skill_category"] == "expert", col], errors="coerce") * direction).dropna().to_numpy() + if len(nov) == 0 or len(exp) == 0: + continue + delta = cliffs_delta(exp, nov) + boot = [] + if len(nov) > 1 and len(exp) > 1: + for _ in range(1000): + boot.append(cliffs_delta(rng.choice(exp, len(exp), replace=True), rng.choice(nov, len(nov), replace=True))) + lo, hi = np.percentile(boot, [2.5, 97.5]) + else: + lo, hi = np.nan, np.nan + rows.append({"feature": label, "dataset": ds, "delta": delta, "lo": lo, "hi": hi}) + res = pd.DataFrame(rows) + if res.empty: + return + res.to_csv(OUT / "cohort_specific_effects.csv", index=False) + + features = [label for _, label, _ in specs if label in set(res["feature"])] + y_base = np.arange(len(features)) + offsets = {"BAPES2024": -0.22, "6DOF2023": 0.0, "7DOF2024": 0.22} + fig, ax = plt.subplots(figsize=(7.4, 4.8)) + for ds in DATASET_ORDER: + sub = res[res["dataset"] == ds] + ys = np.array([features.index(f) for f in sub["feature"]], dtype=float) + offsets[ds] + xerr = np.vstack([(sub["delta"] - sub["lo"]).clip(lower=0), (sub["hi"] - sub["delta"]).clip(lower=0)]) + ax.errorbar( + sub["delta"], + ys, + xerr=xerr, + fmt="o", + ms=4, + lw=0.8, + capsize=2.5, + color=DATASET_COLORS[ds], + label=dataset_label(ds), + alpha=0.92, + ) + ax.axvline(0, color="#333333", lw=0.8) + ax.set_yticks(y_base, features) + ax.invert_yaxis() + ax.set_xlabel("Oriented Cliff's delta for expert minus novice") + ax.set_xlim(-1.05, 1.05) + ax.text(0.01, 1.015, "Novice trials more favourable", transform=ax.transAxes, ha="left", va="bottom", fontsize=7) + ax.text(0.99, 1.015, "Expert trials more favourable", transform=ax.transAxes, ha="right", va="bottom", fontsize=7) + ax.legend(frameon=False, ncol=3, loc="lower center", bbox_to_anchor=(0.5, -0.22)) + fig.tight_layout(rect=(0, 0.07, 1, 1)) + save_fig(fig, "fig21_cohort_effect_forest") + + +def plot_time_synchrony_quadrants(df: pd.DataFrame, clusters: pd.DataFrame) -> None: + """Contrast completion time with coordination to show why time alone is incomplete.""" + if "total_time" not in df.columns or "bimanual_correlation" not in df.columns: + return + identifier_counts = df.groupby("trial_key")["trial_key"].transform("size") + d = df.loc[identifier_counts.eq(1)].copy() + correlation_rows = [] + fig, axes = plt.subplots(1, 3, figsize=(8.7, 3.1), sharex=False, sharey=True) + for ax, ds, label in zip(axes, DATASET_ORDER, "abc"): + g = d[d["dataset"] == ds].dropna(subset=["total_time", "bimanual_correlation"]) + if g.empty: + ax.axis("off") + continue + for skill in SKILL_ORDER: + sub = g[g["skill_category"] == skill] + ax.scatter( + sub["total_time"], + sub["bimanual_correlation"], + s=22, + alpha=0.82, + color=SKILL_COLORS[skill], + label=nice(skill), + ) + xmed = g["total_time"].median() + ymed = g["bimanual_correlation"].median() + ax.axvline(xmed, color="#555555", lw=0.8, ls=":") + ax.axhline(ymed, color="#555555", lw=0.8, ls=":") + rho, p = stats.spearmanr( + g["total_time"], + g["bimanual_correlation"], + nan_policy="omit", + ) + ci_low, ci_high = bootstrap_spearman_ci( + g["total_time"], + g["bimanual_correlation"], + ) + correlation_rows.append( + { + "dataset": ds, + "n": int(len(g)), + "spearman_rho": float(rho), + "ci_low": float(ci_low), + "ci_high": float(ci_high), + "p": float(p), + } + ) + ax.text( + 0.04, + 0.96, + f"$n={len(g)}$\n$\\rho={rho:.2f}$ [{ci_low:.2f}, {ci_high:.2f}]", + transform=ax.transAxes, + ha="left", + va="top", + fontsize=6.8, + bbox=dict(facecolor="white", edgecolor="none", alpha=0.78, pad=1.4), + ) + ax.text( + 0.04, + 0.06, + "Shorter", + transform=ax.transAxes, + ha="left", + va="bottom", + fontsize=6.2, + color="#555555", + ) + ax.text( + 0.96, + 0.06, + "Longer", + transform=ax.transAxes, + ha="right", + va="bottom", + fontsize=6.2, + color="#555555", + ) + ax.set_title(dataset_label(ds), fontsize=8) + ax.set_xlabel("Analysed duration (s)") + if ax is axes[0]: + ax.set_ylabel("Bimanual correlation") + panel_label(ax, label) + handles, labels = axes[-1].get_legend_handles_labels() + fig.legend(handles, labels, frameon=False, ncol=3, loc="lower center", bbox_to_anchor=(0.5, 0.02)) + fig.subplots_adjust(left=0.07, right=0.98, top=0.88, bottom=0.25, wspace=0.24) + save_fig(fig, "fig22_time_synchrony_quadrants") + pd.DataFrame(correlation_rows).to_csv( + OUT / "coordination_control_time_correlation.csv", + index=False, + ) + + +def plot_continuous_performance_validation( + proxy: pd.DataFrame, + clusters: pd.DataFrame, +) -> None: + """Show construction-excluded relationships without depending on cut points.""" + d = proxy.merge( + clusters[["dataset", "trial_name", "metric_cluster"]].drop_duplicates( + ["dataset", "trial_name"] + ), + on=["dataset", "trial_name"], + how="left", + validate="one_to_one", + ) + identifier_counts = d.groupby("trial_key")["trial_key"].transform("size") + d = d.loc[identifier_counts.eq(1)].copy() + outcomes = [ + ("Task efficiency", "Task efficiency score (1-5)"), + ("Analysed duration (s)", "Analysed duration (s)"), + ] + rows = [] + fig, axes = plt.subplots(2, 3, figsize=(8.9, 5.8), sharex="col") + for r, (outcome, ylabel) in enumerate(outcomes): + for c, (ax, ds, label) in enumerate(zip(axes[r], DATASET_ORDER, "abcdef"[r * 3 : r * 3 + 3])): + g = d[d["dataset"] == ds].dropna( + subset=["Coordination-control composite", outcome, "metric_cluster"] + ) + for band in BAND_ORDER: + sub = g[g["metric_cluster"] == band] + ax.scatter( + sub["Coordination-control composite"], + sub[outcome], + s=24, + alpha=0.78, + color=CLUSTER_COLORS[band], + edgecolor="white", + linewidth=0.25, + label=BAND_SHORT_LABELS[band], + ) + if len(g) >= 3: + x = g["Coordination-control composite"].to_numpy(float) + y = g[outcome].to_numpy(float) + slope, intercept = np.polyfit(x, y, 1) + xline = np.linspace(x.min(), x.max(), 100) + ax.plot(xline, intercept + slope * xline, color="#333333", lw=1.0) + rho, p_value = stats.spearmanr(x, y) + ci_low, ci_high = bootstrap_spearman_ci(x, y) + rows.append( + { + "dataset": ds, + "outcome": outcome, + "n": len(g), + "spearman_rho": rho, + "ci_low": ci_low, + "ci_high": ci_high, + "p": p_value, + } + ) + ax.text( + 0.04, + 0.96, + f"$n={len(g)}$\n$\\rho={rho:.2f}$ [{ci_low:.2f}, {ci_high:.2f}]", + transform=ax.transAxes, + ha="left", + va="top", + fontsize=6.8, + bbox=dict(facecolor="white", edgecolor="none", alpha=0.95, pad=1.4), + ) + if r == 0: + ax.set_title(dataset_label(ds), fontsize=8) + if r == len(outcomes) - 1: + ax.set_xlabel("Coordination and control score (1-5)") + if c == 0: + ax.set_ylabel(ylabel) + if outcome == "Task efficiency": + ax.set_ylim(1, 5) + panel_label(ax, label) + + result = pd.DataFrame(rows) + if not result.empty: + result["q"] = benjamini_hochberg(result["p"]) + result.to_csv(OUT / "continuous_performance_validation.csv", index=False) + result["dataset"] = pd.Categorical( + result["dataset"], categories=DATASET_ORDER, ordered=True + ) + result = result.sort_values(["dataset", "outcome"]) + table_rows = [] + for row in result.itertuples(index=False): + q_text = "$<0.001$" if row.q < 0.001 else f"{row.q:.3f}" + table_rows.append( + f"{dataset_label(row.dataset)} & {row.outcome.replace(' (s)', '')} & " + f"{int(row.n)} & {row.spearman_rho:.2f} " + f"[{row.ci_low:.2f}, {row.ci_high:.2f}] & {q_text} \\\\" + ) + (TAB / "continuous_performance_validation.tex").write_text( + "\n".join( + [ + "\\begin{tabular}{llrrr}", + "\\toprule", + "Cohort & Contextual outcome & $n$ & Spearman $\\rho$ [95\\% CI] & FDR $q$ \\\\", + "\\midrule", + *table_rows, + "\\bottomrule", + "\\end{tabular}", + ] + ) + ) + handles, labels = axes[0, -1].get_legend_handles_labels() + fig.legend( + handles, + labels, + frameon=False, + ncol=3, + loc="lower center", + bbox_to_anchor=(0.5, 0.015), + ) + fig.subplots_adjust(left=0.08, right=0.985, top=0.91, bottom=0.16, hspace=0.28, wspace=0.22) + save_fig(fig, "fig23_continuous_performance_validation") + + +def write_tables( + df: pd.DataFrame, + phase_trials: pd.DataFrame, + phase_cycles: pd.DataFrame, + feature_stats: pd.DataFrame, + adjusted_models: pd.DataFrame, + phase_kinematics: pd.DataFrame, + classification: dict[str, object], + proxy: pd.DataFrame, + clusters: pd.DataFrame, + cluster_summary: dict[str, object], + metric_band_classification: dict[str, object], +) -> None: + dataset_rows = [] + for ds in DATASET_ORDER: + g = df[df["dataset"] == ds] + skills = g["skill_category"].value_counts() + hands = g["handedness"].value_counts() + context = "BAPES training event" if ds == "BAPES2024" else "Urology" + dataset_rows.append( + f"{dataset_label(ds)} & {context} & {g['dof'].iloc[0]} & {len(g)} & " + f"{g['trial_key'].nunique()} & " + f"{int(skills.get('novice', 0))}/{int(skills.get('intermediate', 0))}/{int(skills.get('expert', 0))} & " + f"{int(hands.get('right', 0))}/{int(hands.get('left', 0))}/{int(hands.get('mixed', 0))} \\\\" + ) + (TAB / "dataset_characteristics.tex").write_text( + "\n".join( + [ + "\\begin{tabular}{lllrrrr}", + "\\toprule", + "Cohort & Collection context & Sensing & Recordings & Study IDs & Experience N/I/E & Hand R/L/M \\\\", + "\\midrule", + *dataset_rows, + "\\bottomrule", + "\\end{tabular}", + ] + ) + ) + + identifier_counts = df.groupby("trial_key")["trial_key"].transform("size") + independent_keys = set(df.loc[identifier_counts.eq(1), "trial_key"]) + phase_inferential = phase_trials[phase_trials["trial_key"].isin(independent_keys)] + analysis_unit_rows = [ + f"Dataset inventory & {len(df)} & {df['trial_key'].nunique()} & All available recordings \\\\", + f"Primary inferential set & {int(identifier_counts.eq(1).sum())} & " + f"{df.loc[identifier_counts.eq(1), 'trial_key'].nunique()} & " + "Identifiers represented by one recording \\\\", + f"Annotated phase inventory & {len(phase_trials)} & {phase_trials['trial_key'].nunique()} & " + "All densely annotated recordings \\\\", + f"Phase inferential set & {len(phase_inferential)} & " + f"{phase_inferential['trial_key'].nunique()} & " + "Identifiers represented once in full inventory \\\\", + ] + (TAB / "analysis_units.tex").write_text( + "\n".join( + [ + "\\begin{tabular}{lrrL{5.0cm}}", + "\\toprule", + "Analysis set & Recordings & Study IDs & Use \\\\", + "\\midrule", + *analysis_unit_rows, + "\\bottomrule", + "\\end{tabular}", + ] + ) + ) + + phase_rows = [] + for ds in DATASET_ORDER: + g = phase_trials[phase_trials["dataset"] == ds] + skills = g["skill_category"].value_counts() + phase_rows.append( + f"{dataset_label(ds)} & {len(g)} & {g['trial_key'].nunique()} & " + f"{int(skills.get('novice', 0))}/{int(skills.get('intermediate', 0))}/{int(skills.get('expert', 0))} & " + f"{int(g['n_frames'].sum()):,} & {int(g['n_cycles'].sum())} & {g['mean_cycle_s'].mean():.1f} \\\\" + ) + (TAB / "phase_subset.tex").write_text( + "\n".join( + [ + "\\begin{tabular}{lrrrrrr}", + "\\toprule", + "Dataset & Recordings & IDs & Procedure N/I/E & Frames & Cycles & Mean cycle (s) \\\\", + "\\midrule", + *phase_rows, + "\\bottomrule", + "\\end{tabular}", + ] + ) + ) + + annotated_names = set( + zip( + phase_trials["dataset"].astype(str), + phase_trials["trial_short"].astype(str), + ) + ) + selection = df[["dataset", "trial_name", "trial_key", "total_time"]].copy() + selection["phase_annotated"] = [ + (str(dataset), str(name)) in annotated_names + for dataset, name in zip(selection["dataset"], selection["trial_name"]) + ] + selection = selection.merge( + clusters[ + ["dataset", "trial_name", "Coordination-control composite"] + ].drop_duplicates(["dataset", "trial_name"]), + on=["dataset", "trial_name"], + how="left", + validate="one_to_one", + ) + selection_counts = selection.groupby("trial_key")["trial_key"].transform("size") + selection = selection.loc[selection_counts.eq(1)].copy() + selection_rows = [] + for ds in DATASET_ORDER: + cohort = selection[selection["dataset"] == ds] + annotated = cohort[cohort["phase_annotated"]] + other = cohort[~cohort["phase_annotated"]] + selection_rows.append( + f"{dataset_label(ds)} & {len(annotated)}/{len(cohort)} & " + f"{annotated['total_time'].median():.1f}/{other['total_time'].median():.1f} & " + f"{annotated['Coordination-control composite'].median():.2f}/" + f"{other['Coordination-control composite'].median():.2f} \\\\" + ) + (TAB / "phase_subset_selection.tex").write_text( + "\n".join( + [ + "\\begin{tabular}{lrrr}", + "\\toprule", + "Cohort & Annotated/available & Median time annotated/not (s) & " + "Median coordination and control score annotated/not \\\\", + "\\midrule", + *selection_rows, + "\\bottomrule", + "\\end{tabular}", + ] + ) + ) + + rows = [] + feature_rows = feature_stats[feature_stats["n_nonmissing"] >= 20].copy() + feature_rows = feature_rows.loc[ + feature_rows["cliffs_delta_novice_expert"] + .abs() + .sort_values(ascending=False) + .index + ].head(12) + for _, r in feature_rows.iterrows(): + arrow, interpretation = FEATURE_DIRECTIONS.get(r["feature"], ("", "")) + higher_is_better = "\\uparrow" in arrow + effect_orientation = -1 if higher_is_better else 1 + rho_orientation = 1 if higher_is_better else -1 + oriented_delta = effect_orientation * r["cliffs_delta_novice_expert"] + if effect_orientation > 0: + oriented_lo, oriented_hi = r["delta_ci_low"], r["delta_ci_high"] + else: + oriented_lo, oriented_hi = -r["delta_ci_high"], -r["delta_ci_low"] + kruskal_q = "$<0.001$" if r["kruskal_q"] < 0.001 else f"{r['kruskal_q']:.3f}" + spearman_q = "$<0.001$" if r["spearman_q"] < 0.001 else f"{r['spearman_q']:.3f}" + if rho_orientation > 0: + rho_lo, rho_hi = r["spearman_ci_low"], r["spearman_ci_high"] + else: + rho_lo, rho_hi = -r["spearman_ci_high"], -r["spearman_ci_low"] + rows.append( + f"{r['label']} ({arrow}) & {interpretation} & " + f"{oriented_delta:.2f} [{oriented_lo:.2f}, {oriented_hi:.2f}] & " + f"{kruskal_q} & {rho_orientation * r['spearman_rho_procedures']:.2f} " + f"[{rho_lo:.2f}, {rho_hi:.2f}; {spearman_q}] \\\\" + ) + (TAB / "feature_effects.tex").write_text( + "\n".join( + [ + "\\begin{tabular}{L{0.17\\linewidth}L{0.27\\linewidth}L{0.18\\linewidth}L{0.11\\linewidth}L{0.16\\linewidth}}", + "\\toprule", + "Feature & Prespecified interpretation & Oriented Cliff $\\delta$ [95\\% CI] & " + "Three-group $q$ & Oriented meta-$\\rho$ [95\\% CI; $q$] \\\\", + "\\midrule", + *rows, + "\\bottomrule", + "\\end{tabular}", + ] + ) + ) + + if not adjusted_models.empty: + rows = [] + for _, r in adjusted_models.iterrows(): + q_text = ( + "$<0.001$" + if r["procedure_q"] < 0.001 + else f"{r['procedure_q']:.3f}" + ) + rows.append( + f"{r['label']} & {r['beta_log10_procedures']:.2f} & " + f"[{r['ci_low']:.2f}, {r['ci_high']:.2f}] & " + f"{q_text} & " + f"{r['r2_procedure_only']:.2f} & {r['r2_cohort_only']:.2f} & " + f"{r['r2_full']:.2f} & {int(r['n'])} \\\\" + ) + (TAB / "cohort_adjusted_models.tex").write_text( + "\n".join( + [ + "\\begin{tabular}{lrrrrrrr}", + "\\toprule", + "Outcome & $\\beta_{\\log_{10}\\mathrm{proc}}$ & 95\\% CI & FDR $q$ & $R^2$ proc. & $R^2$ cohort & Full $R^2$ & $n$ \\\\", + "\\midrule", + *rows, + "\\bottomrule", + "\\end{tabular}", + ] + ) + ) + + if classification.get("available"): + cv = classification["stratified_group_5fold"] + lodo = classification["leave_one_dataset_out"] + model_labels = { + "time_only_logistic_regression": "Duration-only logistic regression", + "time_only_random_forest": "Duration-only random forest", + "all_features_logistic_regression": "All-feature logistic regression", + "all_features_random_forest": "All-feature random forest", + } + lines = [ + "\\begin{tabular}{L{0.35\\linewidth}L{0.22\\linewidth}L{0.31\\linewidth}}", + "\\toprule", + "Model & 5-fold balanced accuracy, mean (SD) & Leave-one-cohort-out balanced accuracy, mean [range] \\\\", + "\\midrule", + ] + for model, vals in cv.items(): + lodo_scores = [r["balanced_accuracy"] for r in lodo[model]] + mean_lodo = np.mean(lodo_scores) + lines.append( + f"{model_labels.get(model, model.replace('_', ' ').title())} & " + f"{vals['mean_balanced_accuracy']:.2f} ({vals['sd']:.2f}) & " + f"{mean_lodo:.2f} [{np.min(lodo_scores):.2f}, {np.max(lodo_scores):.2f}] \\\\" + ) + lines += ["\\bottomrule", "\\end{tabular}"] + (TAB / "classification.tex").write_text("\n".join(lines)) + + if metric_band_classification.get("available"): + comparison_rows = metric_band_classification.get("comparison_rows", []) + target_order = { + "motion_defined": 0, + "procedure_fixed": 1, + "procedure_tertiles": 2, + "procedure_kmeans": 3, + } + model_order = {"logistic_regression": 0, "random_forest": 1} + comparison_rows = sorted( + comparison_rows, + key=lambda r: (target_order.get(r.get("target"), 99), model_order.get(r.get("model"), 99)), + ) + lines = [ + "\\begin{tabular}{L{0.18\\linewidth}L{0.18\\linewidth}L{0.27\\linewidth}L{0.28\\linewidth}}", + "\\toprule", + "Target rule & Model & Group cross-validation, mean (SD) & Cross-cohort rule recovery, mean [range] \\\\", + "\\midrule", + ] + for r in comparison_rows: + lines.append( + f"{r['target_label']} & {r['model_label']} & " + f"{r['stratified_cv_balanced_accuracy']:.2f} " + f"({r['stratified_cv_sd']:.2f}) & " + f"{r['leave_one_dataset_out_balanced_accuracy']:.2f} " + f"[{r['leave_one_dataset_out_min']:.2f}, " + f"{r['leave_one_dataset_out_max']:.2f}] \\\\" + ) + lines += ["\\bottomrule", "\\end{tabular}"] + (TAB / "metric_band_classification.tex").write_text("\n".join(lines)) + + if not clusters.empty and "metric_cluster" in clusters: + d = clusters.dropna(subset=["skill_category", "metric_cluster"]).copy() + denom_novice = int((d["skill_category"] == "novice").sum()) + denom_expert = int((d["skill_category"] == "expert").sum()) + disagreement_specs = [ + ("Novice procedure group in upper band", (d["skill_category"] == "novice") & (d["metric_cluster"] == "metric_high"), denom_novice), + ("Expert procedure group in lower band", (d["skill_category"] == "expert") & (d["metric_cluster"] == "metric_low"), denom_expert), + ("Novice procedure group outside lower band", (d["skill_category"] == "novice") & (d["metric_cluster"] != "metric_low"), denom_novice), + ("Expert procedure group outside upper band", (d["skill_category"] == "expert") & (d["metric_cluster"] != "metric_high"), denom_expert), + ] + rows = [] + for label, mask, denom in disagreement_specs: + count = int(mask.sum()) + pct = 100 * count / denom if denom else np.nan + rows.append(f"{label} & {count}/{denom} & {pct:.1f}\\% \\\\") + (TAB / "performance_disagreement.tex").write_text( + "\n".join( + [ + "\\begin{tabular}{lrr}", + "\\toprule", + "Comparison & Count & Percentage \\\\", + "\\midrule", + *rows, + "\\bottomrule", + "\\end{tabular}", + ] + ) + ) + + if not clusters.empty and cluster_summary.get("available"): + rows = [] + order = {cl: i for i, cl in enumerate(BAND_ORDER)} + thresholds = cluster_summary.get("metric_band_thresholds", [np.nan, np.nan]) + threshold_rules = { + "metric_low": f"$< {thresholds[0]:.2f}$", + "metric_middle": f"${thresholds[0]:.2f}$ to $< {thresholds[1]:.2f}$", + "metric_high": f"$\\geq {thresholds[1]:.2f}$", + } + for cl in BAND_ORDER: + g = clusters[clusters["metric_cluster"] == cl] + if g.empty: + continue + rows.append( + ( + order[cl], + f"{BAND_LABELS[cl]} & {threshold_rules[cl]} & {len(g)} & " + f"{g['Coordination-control composite'].mean():.2f} & {g['total_procedures'].median():.0f} " + f"[{g['total_procedures'].quantile(0.25):.0f}, {g['total_procedures'].quantile(0.75):.0f}] & " + f"{int((g['skill_category']=='novice').sum())}/{int((g['skill_category']=='intermediate').sum())}/{int((g['skill_category']=='expert').sum())} \\\\" + ) + ) + rows = [r for _, r in sorted(rows)] + (TAB / "kmeans_clusters.tex").write_text( + "\n".join( + [ + "\\begin{tabular}{llrrrr}", + "\\toprule", + "Motion-score band & Score rule & Trials & Mean score & Procedure median [IQR] & Procedure N/I/E \\\\", + "\\midrule", + *rows, + "\\bottomrule", + "\\end{tabular}", + ] + ) + ) + held_out_path = OUT / "held_out_band_statistics.csv" + if held_out_path.exists(): + held_out = pd.read_csv(held_out_path) + held_out_rows = [] + for _, row in held_out.iterrows(): + if row["outcome"] == "Workspace excursion": + outcome_label = "Workspace compactness (descriptive)" + arrow = "" + else: + outcome_label = str(row["outcome"]) + arrow = " ($\\uparrow$)" if row["direction"] == "higher" else " ($\\downarrow$)" + q_text = "$<0.001$" if row["upper_lower_q"] < 0.001 else f"{row['upper_lower_q']:.3f}" + held_out_rows.append( + f"{outcome_label}{arrow} & " + f"{row['median_lower']:.2f}/{row['median_middle']:.2f}/{row['median_upper']:.2f} & " + f"{row['oriented_cliffs_delta_upper_vs_lower']:.2f} " + f"[{row['delta_ci_low']:.2f}, {row['delta_ci_high']:.2f}] & " + f"{q_text} & " + f"{int(row['n_lower'])}/{int(row['n_middle'])}/{int(row['n_upper'])} \\\\" + ) + (TAB / "held_out_band_checks.tex").write_text( + "\n".join( + [ + "\\begin{tabular}{lrrrr}", + "\\toprule", + "Contextual outcome & Lower/middle/upper median & " + "Oriented Cliff $\\delta$ [95\\% CI] & FDR $q$ & $n$ L/M/U \\\\", + "\\midrule", + *held_out_rows, + "\\bottomrule", + "\\end{tabular}", + ] + ) + ) + validation = pd.read_csv(OUT / "kmeans_validation.csv") if (OUT / "kmeans_validation.csv").exists() else pd.DataFrame() + if not validation.empty: + val_rows = [ + f"{int(r['k'])} & {r['silhouette']:.2f} & {r['davies_bouldin']:.2f} & {r['calinski_harabasz']:.1f} & {r['inertia']:.1f} \\\\" + for _, r in validation.iterrows() + ] + (TAB / "kmeans_validation.tex").write_text( + "\n".join( + [ + "\\begin{tabular}{rrrrr}", + "\\toprule", + "$k$ & Silhouette $\\uparrow$ & Davies--Bouldin $\\downarrow$ & Calinski--Harabasz $\\uparrow$ & Inertia $\\downarrow$ \\\\", + "\\midrule", + *val_rows, + "\\bottomrule", + "\\end{tabular}", + ] + ) + ) + sensitivity = cluster_summary.get("sensitivity", {}) + sensitivity_rows = [] + complete = sensitivity.get("complete_case", {}) + complete_cuts = complete.get("thresholds", [np.nan, np.nan]) + sensitivity_rows.append( + "Complete raw-feature cases & " + f"$n={int(complete.get('n_trials', 0))}$ & " + f"{complete_cuts[0]:.2f}, {complete_cuts[1]:.2f} & " + f"{complete.get('assignment_adjusted_rand_index', np.nan):.2f} \\\\" + ) + perturb = sensitivity.get("domain_weight_perturbation", {}) + sensitivity_rows.append( + "Domain weights varied by 25\\% & " + f"{int(perturb.get('n_repetitions', 0))} repetitions & -- & " + f"{perturb.get('adjusted_rand_index_median', np.nan):.2f} " + f"[{perturb.get('adjusted_rand_index_ci_low', np.nan):.2f}, " + f"{perturb.get('adjusted_rand_index_ci_high', np.nan):.2f}] \\\\" + ) + for row in sensitivity.get("leave_one_domain_out", []): + cuts = row.get("thresholds", [np.nan, np.nan]) + sensitivity_rows.append( + f"Omit {row['omitted_domain']} & Remaining one-domain score & " + f"{cuts[0]:.2f}, {cuts[1]:.2f} & {row['adjusted_rand_index']:.2f} \\\\" + ) + for row in sensitivity.get("leave_one_cohort_out", []): + cuts = row.get("thresholds", [np.nan, np.nan]) + sensitivity_rows.append( + f"Fit without {dataset_label(row['held_out_dataset'])} & Excluded cohort & " + f"{cuts[0]:.2f}, {cuts[1]:.2f} & {row['held_out_adjusted_rand_index']:.2f} \\\\" + ) + (TAB / "band_sensitivity.tex").write_text( + "\n".join( + [ + "\\begin{tabular}{llll}", + "\\toprule", + "Sensitivity analysis & Data or perturbation & Cut points & Assignment ARI \\\\", + "\\midrule", + *sensitivity_rows, + "\\bottomrule", + "\\end{tabular}", + ] + ) + ) + + if not proxy.empty: + rows = [] + for skill in SKILL_ORDER: + g = proxy[proxy["skill_category"] == skill] + rows.append( + f"{nice(skill)} & {g['Bimanual coordination'].mean():.2f} & {g['Task efficiency'].mean():.2f} & " + f"{g['Instrument motion control'].mean():.2f} & {g['Workspace excursion'].mean():.2f} & " + f"{g['Coordination-control composite'].mean():.2f} \\\\" + ) + (TAB / "osats_proxy.tex").write_text( + "\n".join( + [ + "\\begin{tabular}{lrrrrr}", + "\\toprule", + "Procedure group & Bimanual $\\uparrow$ & Efficiency $\\uparrow$ & Motion control $\\uparrow$ & Workspace compactness & Coordination and control score $\\uparrow$ \\\\", + "\\midrule", + *rows, + "\\bottomrule", + "\\end{tabular}", + ] + ) + ) + + seg_summary_path = OUT / "phase_segment_summary.csv" + if seg_summary_path.exists(): + seg_summary = pd.read_csv(seg_summary_path) + rows = [] + for phase in PHASE_ORDER: + g = seg_summary[seg_summary["phase"] == phase] + if g.empty: + continue + r = g.iloc[0] + rows.append( + f"{nice(phase)} & {int(r['n_segments'])} & {r['mean_duration_s']:.2f} & {r['median_duration_s']:.2f} \\\\" + ) + (TAB / "phase_segment_summary.tex").write_text( + "\n".join( + [ + "\\begin{tabular}{lrrr}", + "\\toprule", + "Phase & Segments & Mean duration (s) & Median duration (s) \\\\", + "\\midrule", + *rows, + "\\bottomrule", + "\\end{tabular}", + ] + ) + ) + + if not phase_kinematics.empty: + rows = [] + for phase in ["reach", "grasp", "transfer", "place", "nudge"]: + g = phase_kinematics[phase_kinematics["phase"] == phase] + if g.empty: + continue + r = g.iloc[0] + rows.append( + f"{nice(phase)} & {int(r['n_trials'])} & {r['duration_s']:.0f} & " + f"{r['tool1_speed_mm_s']:.1f} & {r['tool2_speed_mm_s']:.1f} & " + f"{r['combined_path_rate_mm_s']:.1f} \\\\" + ) + (TAB / "phase_kinematics_summary.tex").write_text( + "\n".join( + [ + "\\begin{tabular}{lrrrrr}", + "\\toprule", + "Phase & Trials & Duration (s) & Tool 1 speed & Tool 2 speed & Combined path rate \\\\", + "\\midrule", + *rows, + "\\bottomrule", + "\\end{tabular}", + ] + ) + ) + + if not clusters.empty and not phase_trials.empty and not phase_cycles.empty: + phase = _phase_trials_with_clusters(phase_trials, clusters).dropna(subset=["metric_cluster"]) + cyc = _phase_cycles_with_clusters(phase_cycles, clusters).dropna(subset=["metric_cluster"]) + if not phase.empty and not cyc.empty: + trial_cycle = ( + cyc.groupby(["metric_cluster", "trial_key"], as_index=False) + .agg( + mean_cycle_duration_s=("duration_s", "mean"), + mean_transitions=("n_transitions", "mean"), + n_cycles=("cycle_index", "count"), + ) + ) + rows = [] + for band in BAND_ORDER: + pg = phase[phase["metric_cluster"] == band] + cg = cyc[cyc["metric_cluster"] == band] + tg = trial_cycle[trial_cycle["metric_cluster"] == band] + if pg.empty or cg.empty: + continue + rows.append( + f"{BAND_LABELS[band]} & {pg['trial_key'].nunique()} & {len(cg)} & " + f"{tg['mean_cycle_duration_s'].mean():.2f} & {tg['mean_transitions'].mean():.2f} & " + f"{100 * pg['frac_transfer'].mean():.1f}\\% & {100 * pg['frac_place'].mean():.1f}\\% \\\\" + ) + (TAB / "phase_bottlenecks.tex").write_text( + "\n".join( + [ + "\\begin{tabular}{lrrrrrr}", + "\\toprule", + "Motion-score band & Trials & Cycles & Cycle time (s) & Transitions & Transfer time & Place time \\\\", + "\\midrule", + *rows, + "\\bottomrule", + "\\end{tabular}", + ] + ) + ) + + drop_rows = [] + for band in BAND_ORDER: + cg = cyc[cyc["metric_cluster"] == band] + if cg.empty: + continue + per_trial = cg.groupby("trial_key").agg( + n_drop_events=("n_drop_events", "sum"), + n_cycles=("cycle_index", "count"), + ) + per_trial["rate"] = 100 * per_trial["n_drop_events"] / per_trial["n_cycles"] + drop_rows.append( + f"Motion-score band & {BAND_SHORT_LABELS[band]} & " + f"{len(per_trial)} & {len(cg)} & {int(cg['n_drop_events'].sum())} & " + f"{per_trial['rate'].mean():.2f} & " + f"{100 * (per_trial['n_drop_events'] > 0).mean():.1f}\\% \\\\" + ) + for skill in SKILL_ORDER: + cg = cyc[cyc["skill_category"] == skill] + if cg.empty: + continue + per_trial = cg.groupby("trial_key").agg( + n_drop_events=("n_drop_events", "sum"), + n_cycles=("cycle_index", "count"), + ) + per_trial["rate"] = 100 * per_trial["n_drop_events"] / per_trial["n_cycles"] + drop_rows.append( + f"Procedure count & {nice(skill)} & {len(per_trial)} & {len(cg)} & " + f"{int(cg['n_drop_events'].sum())} & " + f"{per_trial['rate'].mean():.2f} & " + f"{100 * (per_trial['n_drop_events'] > 0).mean():.1f}\\% \\\\" + ) + (TAB / "drop_episode_summary.tex").write_text( + "\n".join( + [ + "\\begin{tabular}{llrrrrr}", + "\\toprule", + "Grouping & Group & Trials & Cycles & Drop episodes & " + "Mean episodes per 100 cycles & Trials with drop \\\\", + "\\midrule", + *drop_rows, + "\\bottomrule", + "\\end{tabular}", + ] + ) + ) + + +def write_reproducibility_manifest() -> None: + """Write an auditable manifest for the caches used by this manuscript.""" + input_paths = analysis_input_paths() + packages = [ + "numpy", + "pandas", + "matplotlib", + "seaborn", + "scipy", + "scikit-learn", + "pyarrow", + ] + manifest = { + "analysis_script": { + "path": str((ROOT / "scripts" / "build_scirep_analysis.py").relative_to(ROOT)), + "sha256": file_sha256(ROOT / "scripts" / "build_scirep_analysis.py"), + }, + "python": { + "executable": Path(sys.executable).name, + "version": sys.version, + "platform": platform.platform(), + }, + "packages": {name: package_version(name) for name in packages}, + "inputs": { + key: { + "path": portable_input_path(path), + "exists": path.exists(), + "size_bytes": path.stat().st_size if path.exists() and path.is_file() else None, + "sha256": file_sha256(path), + } + for key, path in input_paths.items() + }, + "random_seeds": { + "bootstrap_and_sampling": 20260618, + "classification_cv": 13, + "pca": 20260618, + "kmeans": 20260618, + }, + } + (OUT / "analysis_manifest.json").write_text(json.dumps(manifest, indent=2)) + + +def main() -> None: + validate_analysis_inputs() + setup_style() + trials, cycles, frames = load_phase_data() + all_df = load_all_trial_data(frames) + feature_stats = feature_statistics(all_df) + classification = run_classification(all_df) + proxy = osats_proxy_scores(all_df) + adjusted_models = cohort_adjusted_models(all_df, proxy) + feature_dictionary = write_feature_dictionary(all_df) + clusters, cluster_summary = kmeans_skill_clusters(all_df, proxy) + trim_validation = trimming_rule_validation(frames, clusters) + processing_checks = processing_sensitivity( + all_df, + frames, + proxy, + clusters, + cluster_summary, + ) + held_out_band_statistics(clusters) + metric_band_classification = run_metric_band_classification(clusters) + motion_sample = load_origin_motion(all_df) + phase_statistics(trials, cycles, frames) + phase_kinematics = phase_specific_kinematics(frames, clusters) + plot_task_setup_overview() + plot_cohort(all_df, trials) + plot_feature_effects(feature_stats) + plot_phase_timing(cycles, trials) + plot_experience_links(all_df, trials) + plot_osats_proxy(proxy, clusters, cluster_summary) + plot_kmeans_clusters(clusters, cluster_summary) + plot_motion_domains(all_df) + plot_jaw_and_rotation(all_df) + plot_workspace_heatmaps(motion_sample) + plot_skill_workspace_xy(motion_sample) + plot_phase_by_cluster(trials, clusters) + plot_tool_specific_motion_boxplots(all_df) + plot_sorted_participant_bars(all_df, proxy) + plot_dataset_metric_matrix(all_df, proxy) + plot_procedure_motion_disagreement(clusters) + plot_phase_bottlenecks(trials, cycles, clusters) + plot_cohort_effect_forest(all_df) + plot_time_synchrony_quadrants(all_df, clusters) + plot_continuous_performance_validation(proxy, clusters) + write_tables( + all_df, + trials, + cycles, + feature_stats, + adjusted_models, + phase_kinematics, + classification, + proxy, + clusters, + cluster_summary, + metric_band_classification, + ) + write_reproducibility_manifest() + summary = { + "all_trials": int(len(all_df)), + "phase_trials": int(len(trials)), + "phase_cycles": int(len(cycles)), + "feature_dictionary_rows": int(len(feature_dictionary)), + "trimming_rule_validation": trim_validation.to_dict(orient="records"), + "processing_sensitivity": processing_checks.to_dict( + orient="records" + ), + "classification": classification, + "metric_band_classification": metric_band_classification, + "kmeans": cluster_summary, + "core_motion_score_mean_by_procedure_group": ( + proxy.groupby("skill_category")["Coordination-control composite"].mean().to_dict() + ), + "top_feature_effects": feature_stats.head(5).to_dict(orient="records"), + } + (OUT / "analysis_summary.json").write_text(json.dumps(summary, indent=2)) + print(json.dumps(summary, indent=2)[:4000]) + + +if __name__ == "__main__": + main() diff --git a/scripts/validate_release.py b/scripts/validate_release.py new file mode 100644 index 0000000..2c09d7a --- /dev/null +++ b/scripts/validate_release.py @@ -0,0 +1,126 @@ +#!/usr/bin/env python3 +"""Validate the aggregate LASK analysis release without private source data.""" +from __future__ import annotations + +import csv +import hashlib +import json +from pathlib import Path + + +ROOT = Path(__file__).resolve().parents[1] +DERIVED = ROOT / "data" / "derived" + + +def read_json(name: str) -> dict: + with (DERIVED / name).open(encoding="utf-8") as handle: + return json.load(handle) + + +def read_csv(name: str) -> list[dict[str, str]]: + with (DERIVED / name).open(newline="", encoding="utf-8") as handle: + return list(csv.DictReader(handle)) + + +def sha256(path: Path) -> str: + digest = hashlib.sha256() + with path.open("rb") as handle: + for chunk in iter(lambda: handle.read(1024 * 1024), b""): + digest.update(chunk) + return digest.hexdigest() + + +def require(condition: bool, message: str) -> None: + if not condition: + raise AssertionError(message) + + +def main() -> None: + required_files = [ + ROOT / "README.md", + ROOT / "CITATION.cff", + ROOT / "docs" / "DATASET.md", + ROOT / "docs" / "ANALYSIS.md", + ROOT / "docs" / "FEATURES.md", + ROOT / "docs" / "PHASES.md", + ROOT / "scripts" / "build_scirep_analysis.py", + DERIVED / "analysis_manifest.json", + DERIVED / "analysis_summary.json", + DERIVED / "feature_dictionary.csv", + DERIVED / "all_trial_feature_statistics.csv", + DERIVED / "cohort_adjusted_models.csv", + ] + for path in required_files: + require(path.is_file() and path.stat().st_size > 0, f"Missing required file: {path}") + + summary = read_json("analysis_summary.json") + require(summary["all_trials"] == 115, "Expected 115 recordings in the inventory") + require(summary["phase_trials"] == 38, "Expected 38 phase-labelled recordings") + require(summary["phase_cycles"] == 425, "Expected 425 placement-complete cycles") + require(summary["feature_dictionary_rows"] == 30, "Expected 30 defined motion features") + + feature_rows = read_csv("all_trial_feature_statistics.csv") + require(len(feature_rows) == 29, "Expected 29 inferential feature rows") + significant_correlations = [ + row for row in feature_rows if float(row["spearman_q"]) < 0.05 + ] + significant_group_tests = [ + row for row in feature_rows if float(row["kruskal_q"]) < 0.05 + ] + require( + len(significant_correlations) == 8, + "Expected eight false-discovery-rate-corrected procedure correlations", + ) + require( + not significant_group_tests, + "No three-group feature test should survive correction", + ) + + model_rows = read_csv("cohort_adjusted_models.csv") + significant_models = { + row["outcome"] for row in model_rows if float(row["procedure_q"]) < 0.05 + } + require( + significant_models + == { + "total_time", + "bimanual_correlation", + "Coordination-control composite", + }, + "Unexpected set of corrected cohort-adjusted procedure effects", + ) + + manifest = read_json("analysis_manifest.json") + analysis_script = ROOT / manifest["analysis_script"]["path"] + require( + sha256(analysis_script) == manifest["analysis_script"]["sha256"], + "Analysis script hash does not match the public manifest", + ) + for source in manifest["inputs"].values(): + require( + not source["path"].startswith("/"), + "Manifest contains a machine-specific absolute input path", + ) + require(len(source["sha256"]) == 64, "Input hash is missing or malformed") + + figures = [ + "fig1_cohort_structure.png", + "fig2_feature_effects.png", + "fig3_phase_timing.png", + "fig4_experience_links.png", + "fig21_cohort_effect_forest.png", + "fig23_continuous_performance_validation.png", + ] + for name in figures: + path = ROOT / "paper" / "figures" / name + require(path.is_file() and path.stat().st_size > 20_000, f"Missing figure: {name}") + + print( + "Validated LASK aggregate release: " + "115 recordings, 38 phase recordings, 425 cycles, " + "29 feature tests, 8 corrected continuous associations." + ) + + +if __name__ == "__main__": + main()