feat: version-agnostic analysis scripts with shared version helpers
- scripts/bicorder_common.py (new): single source of truth for historical gradient renames (COLUMN_RENAMES), version detection (bicorder_version col → version col → data/<type>_<version>/ dir convention), and training-CSV auto-selection (find_training_csv) - classify_readings.py: auto-select training run by recorded bicorder version (excludes the input itself to avoid circular training), canonicalize old column names, version-mismatch warnings - bicorder_classifier.py / export_model_for_js.py: share renames + dimension loader; instructive error when clustering results are missing; bicorder_version recorded in exported models - compare_analyses.py: CLI (reference + comparison CSVs), rename canonicalization so runs of different versions align on shared gradients, Descriptor dedup (was silently skewing merges); legacy no-arg audit intact - scripts/univariate_analysis.py (new): per-protocol/per-gradient averages, distributions, summary stats; --img publishes the three README summary charts to img/ - sync_readings.sh: defer classifier training to auto-matching; gitignore analysis/.venv and __pycache__
This commit is contained in:
1 parent
a4f16e8e4d
commit
55cbd6cd5d
8 files changed
+591
-109
No files matched your search
@@ -2,14 +2,18 @@
|
||||
"""
|
||||
Apply the BicorderClassifier to all readings in a CSV and save results.
|
||||
|
||||
Uses the synthetic-trained LDA model by default. Missing dimensions are
|
||||
Training data is selected automatically by version: the input CSV's recorded
|
||||
`bicorder_version` (or `version`) is matched against the runs under data/ so
|
||||
the classifier trains on a same-version synthetic run whenever one exists
|
||||
(see bicorder_common.find_training_csv). Older-format columns are renamed to
|
||||
the current bicorder terminology automatically. Missing dimensions are
|
||||
filled with the neutral value (5), so shortform readings can still be
|
||||
classified — though with lower confidence.
|
||||
|
||||
Usage:
|
||||
python3 scripts/classify_readings.py data/manual_20260320/readings.csv
|
||||
python3 scripts/classify_readings.py data/manual_20260320/readings.csv \\
|
||||
--training data/synthetic_1.2.6/readings.csv \\
|
||||
--training data/synthetic_1.4.0/readings.csv \\
|
||||
--output data/manual_20260320/analysis/classifications.csv
|
||||
"""
|
||||
|
||||
@@ -20,6 +24,7 @@ from pathlib import Path
|
||||
import pandas as pd
|
||||
|
||||
from bicorder_classifier import BicorderClassifier
|
||||
from bicorder_common import csv_version, find_training_csv, apply_renames
|
||||
|
||||
|
||||
def main():
|
||||
@@ -28,9 +33,10 @@ def main():
|
||||
)
|
||||
parser.add_argument('input_csv', help='Readings CSV to classify')
|
||||
parser.add_argument(
|
||||
'--training',
|
||||
default='data/synthetic_1.2.6/readings.csv',
|
||||
help='Training CSV for classifier (default: synthetic_1.2.6)'
|
||||
'--training', default=None,
|
||||
help='Training CSV for classifier (default: auto-selected to match '
|
||||
"the input's recorded bicorder_version; falls back to the most "
|
||||
'recent synthetic run)'
|
||||
)
|
||||
parser.add_argument(
|
||||
'--output', default=None,
|
||||
@@ -39,16 +45,46 @@ def main():
|
||||
args = parser.parse_args()
|
||||
|
||||
input_path = Path(args.input_csv)
|
||||
input_version = csv_version(input_path)
|
||||
|
||||
# Auto-select training data matching the input's recorded bicorder version
|
||||
if args.training:
|
||||
training_path = Path(args.training)
|
||||
else:
|
||||
training_path = find_training_csv(version=input_version, exclude=input_path)
|
||||
if training_path is None:
|
||||
raise SystemExit("No training CSV found under data/*/readings.csv — "
|
||||
"pass --training explicitly.")
|
||||
training_version = csv_version(training_path)
|
||||
if input_version and training_version == input_version:
|
||||
note = f"matching v{input_version}"
|
||||
elif training_version:
|
||||
note = (f"fell back to v{training_version} "
|
||||
f"(no other v{input_version} run under data/ to avoid circular training)")
|
||||
else:
|
||||
note = "(input records no version; most recent run selected)"
|
||||
print(f"Auto-selected training data: {training_path} ({note})")
|
||||
|
||||
# Warn when training data comes from a different bicorder version
|
||||
# (older column names are auto-renamed via bicorder_common)
|
||||
training_version = csv_version(training_path)
|
||||
if input_version and training_version and input_version != training_version:
|
||||
print(f"Warning: training data is bicorder v{training_version} "
|
||||
f"but input is v{input_version}; renamed columns are aligned "
|
||||
f"automatically, but compare versions (especially gradient "
|
||||
f"orderings) when interpreting results.")
|
||||
|
||||
output_path = (
|
||||
Path(args.output) if args.output
|
||||
else input_path.parent / 'analysis' / 'classifications.csv'
|
||||
)
|
||||
output_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
print(f"Loading classifier (training: {args.training})...")
|
||||
classifier = BicorderClassifier(diagnostic_csv=args.training)
|
||||
print(f"Loading classifier (training: {training_path})...")
|
||||
classifier = BicorderClassifier(diagnostic_csv=training_path)
|
||||
|
||||
df = pd.read_csv(input_path)
|
||||
df = apply_renames(df) # canonicalize old gradient column names
|
||||
print(f"Classifying {len(df)} readings from {input_path}...")
|
||||
|
||||
rows = []
|
||||
|
||||
Reference in new issue
Block a user