- scripts/bicorder_common.py (new): single source of truth for historical gradient renames (COLUMN_RENAMES), version detection (bicorder_version col → version col → data/<type>_<version>/ dir convention), and training-CSV auto-selection (find_training_csv) - classify_readings.py: auto-select training run by recorded bicorder version (excludes the input itself to avoid circular training), canonicalize old column names, version-mismatch warnings - bicorder_classifier.py / export_model_for_js.py: share renames + dimension loader; instructive error when clustering results are missing; bicorder_version recorded in exported models - compare_analyses.py: CLI (reference + comparison CSVs), rename canonicalization so runs of different versions align on shared gradients, Descriptor dedup (was silently skewing merges); legacy no-arg audit intact - scripts/univariate_analysis.py (new): per-protocol/per-gradient averages, distributions, summary stats; --img publishes the three README summary charts to img/ - sync_readings.sh: defer classifier training to auto-matching; gitignore analysis/.venv and __pycache__
139 lines
5.5 KiB
Python
139 lines
5.5 KiB
Python
#!/usr/bin/env python3
|
|
"""
|
|
Apply the BicorderClassifier to all readings in a CSV and save results.
|
|
|
|
Training data is selected automatically by version: the input CSV's recorded
|
|
`bicorder_version` (or `version`) is matched against the runs under data/ so
|
|
the classifier trains on a same-version synthetic run whenever one exists
|
|
(see bicorder_common.find_training_csv). Older-format columns are renamed to
|
|
the current bicorder terminology automatically. Missing dimensions are
|
|
filled with the neutral value (5), so shortform readings can still be
|
|
classified — though with lower confidence.
|
|
|
|
Usage:
|
|
python3 scripts/classify_readings.py data/manual_20260320/readings.csv
|
|
python3 scripts/classify_readings.py data/manual_20260320/readings.csv \\
|
|
--training data/synthetic_1.4.0/readings.csv \\
|
|
--output data/manual_20260320/analysis/classifications.csv
|
|
"""
|
|
|
|
import argparse
|
|
import csv
|
|
from pathlib import Path
|
|
|
|
import pandas as pd
|
|
|
|
from bicorder_classifier import BicorderClassifier
|
|
from bicorder_common import csv_version, find_training_csv, apply_renames
|
|
|
|
|
|
def main():
|
|
parser = argparse.ArgumentParser(
|
|
description='Classify all readings in a CSV using the BicorderClassifier'
|
|
)
|
|
parser.add_argument('input_csv', help='Readings CSV to classify')
|
|
parser.add_argument(
|
|
'--training', default=None,
|
|
help='Training CSV for classifier (default: auto-selected to match '
|
|
"the input's recorded bicorder_version; falls back to the most "
|
|
'recent synthetic run)'
|
|
)
|
|
parser.add_argument(
|
|
'--output', default=None,
|
|
help='Output CSV path (default: <dataset>/analysis/classifications.csv)'
|
|
)
|
|
args = parser.parse_args()
|
|
|
|
input_path = Path(args.input_csv)
|
|
input_version = csv_version(input_path)
|
|
|
|
# Auto-select training data matching the input's recorded bicorder version
|
|
if args.training:
|
|
training_path = Path(args.training)
|
|
else:
|
|
training_path = find_training_csv(version=input_version, exclude=input_path)
|
|
if training_path is None:
|
|
raise SystemExit("No training CSV found under data/*/readings.csv — "
|
|
"pass --training explicitly.")
|
|
training_version = csv_version(training_path)
|
|
if input_version and training_version == input_version:
|
|
note = f"matching v{input_version}"
|
|
elif training_version:
|
|
note = (f"fell back to v{training_version} "
|
|
f"(no other v{input_version} run under data/ to avoid circular training)")
|
|
else:
|
|
note = "(input records no version; most recent run selected)"
|
|
print(f"Auto-selected training data: {training_path} ({note})")
|
|
|
|
# Warn when training data comes from a different bicorder version
|
|
# (older column names are auto-renamed via bicorder_common)
|
|
training_version = csv_version(training_path)
|
|
if input_version and training_version and input_version != training_version:
|
|
print(f"Warning: training data is bicorder v{training_version} "
|
|
f"but input is v{input_version}; renamed columns are aligned "
|
|
f"automatically, but compare versions (especially gradient "
|
|
f"orderings) when interpreting results.")
|
|
|
|
output_path = (
|
|
Path(args.output) if args.output
|
|
else input_path.parent / 'analysis' / 'classifications.csv'
|
|
)
|
|
output_path.parent.mkdir(parents=True, exist_ok=True)
|
|
|
|
print(f"Loading classifier (training: {training_path})...")
|
|
classifier = BicorderClassifier(diagnostic_csv=training_path)
|
|
|
|
df = pd.read_csv(input_path)
|
|
df = apply_renames(df) # canonicalize old gradient column names
|
|
print(f"Classifying {len(df)} readings from {input_path}...")
|
|
|
|
rows = []
|
|
for _, record in df.iterrows():
|
|
# Build ratings dict from dimension columns only
|
|
ratings = {
|
|
col: float(record[col])
|
|
for col in classifier.DIMENSIONS
|
|
if col in record and pd.notna(record[col])
|
|
}
|
|
|
|
result = classifier.predict(ratings, return_details=True)
|
|
|
|
rows.append({
|
|
'Descriptor': record.get('Descriptor', ''),
|
|
'analyst': record.get('analyst', ''),
|
|
'standpoint': record.get('standpoint', ''),
|
|
'shortform': record.get('shortform', ''),
|
|
'cluster': result['cluster'],
|
|
'cluster_name': result['cluster_name'],
|
|
'confidence': round(result['confidence'], 3),
|
|
'lda_score': round(result['lda_score'], 3),
|
|
'distance_to_boundary': round(result['distance_to_boundary'], 3),
|
|
'completeness': round(result['completeness'], 3),
|
|
'dimensions_provided': result['dimensions_provided'],
|
|
'key_dims_provided': result['key_dimensions_provided'],
|
|
'recommended_form': result['recommended_form'],
|
|
})
|
|
|
|
out_df = pd.DataFrame(rows)
|
|
out_df.to_csv(output_path, index=False)
|
|
print(f"Classifications saved → {output_path}")
|
|
|
|
# Summary
|
|
counts = out_df['cluster_name'].value_counts()
|
|
print(f"\nCluster summary:")
|
|
for name, count in counts.items():
|
|
pct = count / len(out_df) * 100
|
|
print(f" {name}: {count} ({pct:.0f}%)")
|
|
|
|
low_conf = (out_df['confidence'] < 0.4).sum()
|
|
if low_conf:
|
|
print(f"\n {low_conf} readings with low confidence (<0.4) — may be boundary cases")
|
|
|
|
shortform_count = out_df[out_df['shortform'].astype(str) == 'True'].shape[0]
|
|
if shortform_count:
|
|
print(f"\n {shortform_count} shortform readings classified (missing dims filled with neutral 5)")
|
|
|
|
|
|
if __name__ == '__main__':
|
|
main()
|