#!/usr/bin/env python3 """ Apply the BicorderClassifier to all readings in a CSV and save results. Training data is selected automatically by version: the input CSV's recorded `bicorder_version` (or `version`) is matched against the runs under data/ so the classifier trains on a same-version synthetic run whenever one exists (see bicorder_common.find_training_csv). Older-format columns are renamed to the current bicorder terminology automatically. Missing dimensions are filled with the neutral value (5), so shortform readings can still be classified — though with lower confidence. Usage: python3 scripts/classify_readings.py data/manual_20260320/readings.csv python3 scripts/classify_readings.py data/manual_20260320/readings.csv \\ --training data/synthetic_1.4.0/readings.csv \\ --output data/manual_20260320/analysis/classifications.csv """ import argparse import csv from pathlib import Path import pandas as pd from bicorder_classifier import BicorderClassifier from bicorder_common import csv_version, find_training_csv, apply_renames def main(): parser = argparse.ArgumentParser( description='Classify all readings in a CSV using the BicorderClassifier' ) parser.add_argument('input_csv', help='Readings CSV to classify') parser.add_argument( '--training', default=None, help='Training CSV for classifier (default: auto-selected to match ' "the input's recorded bicorder_version; falls back to the most " 'recent synthetic run)' ) parser.add_argument( '--output', default=None, help='Output CSV path (default: /analysis/classifications.csv)' ) args = parser.parse_args() input_path = Path(args.input_csv) input_version = csv_version(input_path) # Auto-select training data matching the input's recorded bicorder version if args.training: training_path = Path(args.training) else: training_path = find_training_csv(version=input_version, exclude=input_path) if training_path is None: raise SystemExit("No training CSV found under data/*/readings.csv — " "pass --training explicitly.") training_version = csv_version(training_path) if input_version and training_version == input_version: note = f"matching v{input_version}" elif training_version: note = (f"fell back to v{training_version} " f"(no other v{input_version} run under data/ to avoid circular training)") else: note = "(input records no version; most recent run selected)" print(f"Auto-selected training data: {training_path} ({note})") # Warn when training data comes from a different bicorder version # (older column names are auto-renamed via bicorder_common) training_version = csv_version(training_path) if input_version and training_version and input_version != training_version: print(f"Warning: training data is bicorder v{training_version} " f"but input is v{input_version}; renamed columns are aligned " f"automatically, but compare versions (especially gradient " f"orderings) when interpreting results.") output_path = ( Path(args.output) if args.output else input_path.parent / 'analysis' / 'classifications.csv' ) output_path.parent.mkdir(parents=True, exist_ok=True) print(f"Loading classifier (training: {training_path})...") classifier = BicorderClassifier(diagnostic_csv=training_path) df = pd.read_csv(input_path) df = apply_renames(df) # canonicalize old gradient column names print(f"Classifying {len(df)} readings from {input_path}...") rows = [] for _, record in df.iterrows(): # Build ratings dict from dimension columns only ratings = { col: float(record[col]) for col in classifier.DIMENSIONS if col in record and pd.notna(record[col]) } result = classifier.predict(ratings, return_details=True) rows.append({ 'Descriptor': record.get('Descriptor', ''), 'analyst': record.get('analyst', ''), 'standpoint': record.get('standpoint', ''), 'shortform': record.get('shortform', ''), 'cluster': result['cluster'], 'cluster_name': result['cluster_name'], 'confidence': round(result['confidence'], 3), 'lda_score': round(result['lda_score'], 3), 'distance_to_boundary': round(result['distance_to_boundary'], 3), 'completeness': round(result['completeness'], 3), 'dimensions_provided': result['dimensions_provided'], 'key_dims_provided': result['key_dimensions_provided'], 'recommended_form': result['recommended_form'], }) out_df = pd.DataFrame(rows) out_df.to_csv(output_path, index=False) print(f"Classifications saved → {output_path}") # Summary counts = out_df['cluster_name'].value_counts() print(f"\nCluster summary:") for name, count in counts.items(): pct = count / len(out_df) * 100 print(f" {name}: {count} ({pct:.0f}%)") low_conf = (out_df['confidence'] < 0.4).sum() if low_conf: print(f"\n {low_conf} readings with low confidence (<0.4) — may be boundary cases") shortform_count = out_df[out_df['shortform'].astype(str) == 'True'].shape[0] if shortform_count: print(f"\n {shortform_count} shortform readings classified (missing dims filled with neutral 5)") if __name__ == '__main__': main()