feat: version-agnostic analysis scripts with shared version helpers

- scripts/bicorder_common.py (new): single source of truth for historical
  gradient renames (COLUMN_RENAMES), version detection (bicorder_version col
  → version col → data/<type>_<version>/ dir convention), and training-CSV
  auto-selection (find_training_csv)
- classify_readings.py: auto-select training run by recorded bicorder version
  (excludes the input itself to avoid circular training), canonicalize old
  column names, version-mismatch warnings
- bicorder_classifier.py / export_model_for_js.py: share renames + dimension
  loader; instructive error when clustering results are missing;
  bicorder_version recorded in exported models
- compare_analyses.py: CLI (reference + comparison CSVs), rename
  canonicalization so runs of different versions align on shared gradients,
  Descriptor dedup (was silently skewing merges); legacy no-arg audit intact
- scripts/univariate_analysis.py (new): per-protocol/per-gradient averages,
  distributions, summary stats; --img publishes the three README summary
  charts to img/
- sync_readings.sh: defer classifier training to auto-matching; gitignore
  analysis/.venv and __pycache__
This commit is contained in:
Nathan Schneider committed 2026-10-02 08:32:59 -06:00
1 parent a4f16e8e4d
commit 55cbd6cd5d
8 files changed
+591 -109

No files matched your search

+43 -7
View File
@@ -2,14 +2,18 @@
"""
Apply the BicorderClassifier to all readings in a CSV and save results.
Uses the synthetic-trained LDA model by default. Missing dimensions are
Training data is selected automatically by version: the input CSV's recorded
`bicorder_version` (or `version`) is matched against the runs under data/ so
the classifier trains on a same-version synthetic run whenever one exists
(see bicorder_common.find_training_csv). Older-format columns are renamed to
the current bicorder terminology automatically. Missing dimensions are
filled with the neutral value (5), so shortform readings can still be
classified — though with lower confidence.
Usage:
python3 scripts/classify_readings.py data/manual_20260320/readings.csv
python3 scripts/classify_readings.py data/manual_20260320/readings.csv \\
--training data/synthetic_1.2.6/readings.csv \\
--training data/synthetic_1.4.0/readings.csv \\
--output data/manual_20260320/analysis/classifications.csv
"""
@@ -20,6 +24,7 @@ from pathlib import Path
import pandas as pd
from bicorder_classifier import BicorderClassifier
from bicorder_common import csv_version, find_training_csv, apply_renames
def main():
@@ -28,9 +33,10 @@ def main():
)
parser.add_argument('input_csv', help='Readings CSV to classify')
parser.add_argument(
'--training',
default='data/synthetic_1.2.6/readings.csv',
help='Training CSV for classifier (default: synthetic_1.2.6)'
'--training', default=None,
help='Training CSV for classifier (default: auto-selected to match '
"the input's recorded bicorder_version; falls back to the most "
'recent synthetic run)'
)
parser.add_argument(
'--output', default=None,
@@ -39,16 +45,46 @@ def main():
args = parser.parse_args()
input_path = Path(args.input_csv)
input_version = csv_version(input_path)
# Auto-select training data matching the input's recorded bicorder version
if args.training:
training_path = Path(args.training)
else:
training_path = find_training_csv(version=input_version, exclude=input_path)
if training_path is None:
raise SystemExit("No training CSV found under data/*/readings.csv — "
"pass --training explicitly.")
training_version = csv_version(training_path)
if input_version and training_version == input_version:
note = f"matching v{input_version}"
elif training_version:
note = (f"fell back to v{training_version} "
f"(no other v{input_version} run under data/ to avoid circular training)")
else:
note = "(input records no version; most recent run selected)"
print(f"Auto-selected training data: {training_path} ({note})")
# Warn when training data comes from a different bicorder version
# (older column names are auto-renamed via bicorder_common)
training_version = csv_version(training_path)
if input_version and training_version and input_version != training_version:
print(f"Warning: training data is bicorder v{training_version} "
f"but input is v{input_version}; renamed columns are aligned "
f"automatically, but compare versions (especially gradient "
f"orderings) when interpreting results.")
output_path = (
Path(args.output) if args.output
else input_path.parent / 'analysis' / 'classifications.csv'
)
output_path.parent.mkdir(parents=True, exist_ok=True)
print(f"Loading classifier (training: {args.training})...")
classifier = BicorderClassifier(diagnostic_csv=args.training)
print(f"Loading classifier (training: {training_path})...")
classifier = BicorderClassifier(diagnostic_csv=training_path)
df = pd.read_csv(input_path)
df = apply_renames(df) # canonicalize old gradient column names
print(f"Classifying {len(df)} readings from {input_path}...")
rows = []