feat: version-agnostic analysis scripts with shared version helpers

- scripts/bicorder_common.py (new): single source of truth for historical
  gradient renames (COLUMN_RENAMES), version detection (bicorder_version col
  → version col → data/<type>_<version>/ dir convention), and training-CSV
  auto-selection (find_training_csv)
- classify_readings.py: auto-select training run by recorded bicorder version
  (excludes the input itself to avoid circular training), canonicalize old
  column names, version-mismatch warnings
- bicorder_classifier.py / export_model_for_js.py: share renames + dimension
  loader; instructive error when clustering results are missing;
  bicorder_version recorded in exported models
- compare_analyses.py: CLI (reference + comparison CSVs), rename
  canonicalization so runs of different versions align on shared gradients,
  Descriptor dedup (was silently skewing merges); legacy no-arg audit intact
- scripts/univariate_analysis.py (new): per-protocol/per-gradient averages,
  distributions, summary stats; --img publishes the three README summary
  charts to img/
- sync_readings.sh: defer classifier training to auto-matching; gitignore
  analysis/.venv and __pycache__
This commit is contained in:
Nathan Schneider committed 2026-10-02 08:32:59 -06:00
1 parent a4f16e8e4d
commit 55cbd6cd5d
8 files changed
+591 -109

No files matched your search

+28 -34
View File
@@ -30,34 +30,7 @@ from sklearn.discriminant_analysis import LinearDiscriminantAnalysis
import json
from pathlib import Path
# Path to bicorder.json (relative to this script)
_BICORDER_JSON = Path(__file__).parent.parent.parent / 'bicorder.json'
# Historical column renames: maps old CSV column names → current bicorder.json names.
# Add an entry here whenever gradient terms are renamed in bicorder.json.
_COLUMN_RENAMES = {
'Design_elite_vs_vernacular': 'Design_formal_vs_vernacular',
'Design_institutional_vs_vernacular': 'Design_formal_vs_vernacular',
'Entanglement_exclusive_vs_non-exclusive': 'Entanglement_monopolistic_vs_pluralistic',
'Experience_sufficient_vs_insufficient': 'Experience_sufficient_vs_limited',
'Experience_Kafka_vs_Whitehead': 'Experience_restraining_vs_liberating',
}
def _load_bicorder_dimensions(bicorder_path=_BICORDER_JSON):
"""Read DIMENSIONS and KEY_DIMENSIONS from bicorder.json."""
with open(bicorder_path) as f:
data = json.load(f)
dimensions = []
key_dimensions = []
for category in data['diagnostic']:
set_name = category['set_name']
for gradient in category['gradients']:
dim_name = f"{set_name}_{gradient['term_left']}_vs_{gradient['term_right']}"
dimensions.append(dim_name)
if gradient.get('shortform', False):
key_dimensions.append(dim_name)
return dimensions, key_dimensions
from bicorder_common import COLUMN_RENAMES as _COLUMN_RENAMES, find_training_csv, load_bicorder_dimensions
class BicorderClassifier:
@@ -71,19 +44,32 @@ class BicorderClassifier:
2: "Institutional/Bureaucratic"
}
def __init__(self, diagnostic_csv='data/synthetic_1.2.6/readings.csv',
def __init__(self, diagnostic_csv=None,
model_path=None):
"""Initialize classifier with pre-computed model data."""
"""Initialize classifier with pre-computed model data.
If diagnostic_csv is None, the most recent synthetic readings.csv is
selected automatically (see bicorder_common.find_training_csv).
"""
if diagnostic_csv is None:
diagnostic_csv = find_training_csv()
if diagnostic_csv is None:
raise FileNotFoundError(
"No training CSV found under data/*/readings.csv — "
"pass diagnostic_csv explicitly."
)
print(f"No training CSV specified; using {diagnostic_csv}")
if model_path is None:
model_path = str(Path(diagnostic_csv).parent / 'analysis' / 'data')
self._diagnostic_csv = diagnostic_csv
self._diagnostic_csv = str(diagnostic_csv)
self.model_path = Path(model_path)
self.scaler = StandardScaler()
self.lda = None
self.cluster_centroids = None
# Derive dimension lists from bicorder.json
self.DIMENSIONS, self.KEY_DIMENSIONS = _load_bicorder_dimensions()
# Derive dimension lists (and version) from bicorder.json
self.DIMENSIONS, self.KEY_DIMENSIONS, self.bicorder_version = (
load_bicorder_dimensions())
# Load training data to fit scaler and LDA
self._load_model()
@@ -92,7 +78,13 @@ class BicorderClassifier:
"""Load and fit the classification model from analysis results."""
# Load the original data and cluster assignments
df = pd.read_csv(self._diagnostic_csv)
clusters = pd.read_csv(self.model_path / 'kmeans_clusters.csv')
clusters_path = self.model_path / 'kmeans_clusters.csv'
if not clusters_path.exists():
raise FileNotFoundError(
f"No clustering results for training data: {clusters_path} not found. "
f"Run `python3 scripts/multivariate_analysis.py {self._diagnostic_csv}` "
f"first to generate cluster assignments.")
clusters = pd.read_csv(clusters_path)
# Rename old column names to match current bicorder.json
df = df.rename(columns=_COLUMN_RENAMES)
@@ -283,6 +275,8 @@ class BicorderClassifier:
def save_model(self, output_path='bicorder_classifier_model.json'):
"""Save model parameters for use without scikit-learn."""
model_data = {
'bicorder_version': self.bicorder_version,
'trained_on': self._diagnostic_csv,
'dimensions': self.DIMENSIONS,
'key_dimensions': self.KEY_DIMENSIONS,
'cluster_names': self.CLUSTER_NAMES,