From 55cbd6cd5da0caaaacd2b4350046b7405d0d3115 Mon Sep 17 00:00:00 2001 From: Nathan Schneider Date: Fri, 2 Oct 2026 08:32:59 -0600 Subject: [PATCH] feat: version-agnostic analysis scripts with shared version helpers MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - scripts/bicorder_common.py (new): single source of truth for historical gradient renames (COLUMN_RENAMES), version detection (bicorder_version col → version col → data/_/ dir convention), and training-CSV auto-selection (find_training_csv) - classify_readings.py: auto-select training run by recorded bicorder version (excludes the input itself to avoid circular training), canonicalize old column names, version-mismatch warnings - bicorder_classifier.py / export_model_for_js.py: share renames + dimension loader; instructive error when clustering results are missing; bicorder_version recorded in exported models - compare_analyses.py: CLI (reference + comparison CSVs), rename canonicalization so runs of different versions align on shared gradients, Descriptor dedup (was silently skewing merges); legacy no-arg audit intact - scripts/univariate_analysis.py (new): per-protocol/per-gradient averages, distributions, summary stats; --img publishes the three README summary charts to img/ - sync_readings.sh: defer classifier training to auto-matching; gitignore analysis/.venv and __pycache__ --- .gitignore | 2 + analysis/scripts/bicorder_classifier.py | 62 +++---- analysis/scripts/bicorder_common.py | 161 +++++++++++++++++ analysis/scripts/classify_readings.py | 50 +++++- analysis/scripts/compare_analyses.py | 119 ++++++++++--- analysis/scripts/export_model_for_js.py | 56 ++---- analysis/scripts/sync_readings.sh | 25 ++- analysis/scripts/univariate_analysis.py | 225 ++++++++++++++++++++++++ 8 files changed, 591 insertions(+), 109 deletions(-) create mode 100644 analysis/scripts/bicorder_common.py create mode 100644 analysis/scripts/univariate_analysis.py diff --git a/.gitignore b/.gitignore index 82cafc1..193e847 100644 --- a/.gitignore +++ b/.gitignore @@ -1,3 +1,5 @@ tmp analysis/venv +analysis/.venv +__pycache__/ .aider* diff --git a/analysis/scripts/bicorder_classifier.py b/analysis/scripts/bicorder_classifier.py index a9453f4..6512500 100644 --- a/analysis/scripts/bicorder_classifier.py +++ b/analysis/scripts/bicorder_classifier.py @@ -30,34 +30,7 @@ from sklearn.discriminant_analysis import LinearDiscriminantAnalysis import json from pathlib import Path -# Path to bicorder.json (relative to this script) -_BICORDER_JSON = Path(__file__).parent.parent.parent / 'bicorder.json' - -# Historical column renames: maps old CSV column names → current bicorder.json names. -# Add an entry here whenever gradient terms are renamed in bicorder.json. -_COLUMN_RENAMES = { - 'Design_elite_vs_vernacular': 'Design_formal_vs_vernacular', - 'Design_institutional_vs_vernacular': 'Design_formal_vs_vernacular', - 'Entanglement_exclusive_vs_non-exclusive': 'Entanglement_monopolistic_vs_pluralistic', - 'Experience_sufficient_vs_insufficient': 'Experience_sufficient_vs_limited', - 'Experience_Kafka_vs_Whitehead': 'Experience_restraining_vs_liberating', -} - - -def _load_bicorder_dimensions(bicorder_path=_BICORDER_JSON): - """Read DIMENSIONS and KEY_DIMENSIONS from bicorder.json.""" - with open(bicorder_path) as f: - data = json.load(f) - dimensions = [] - key_dimensions = [] - for category in data['diagnostic']: - set_name = category['set_name'] - for gradient in category['gradients']: - dim_name = f"{set_name}_{gradient['term_left']}_vs_{gradient['term_right']}" - dimensions.append(dim_name) - if gradient.get('shortform', False): - key_dimensions.append(dim_name) - return dimensions, key_dimensions +from bicorder_common import COLUMN_RENAMES as _COLUMN_RENAMES, find_training_csv, load_bicorder_dimensions class BicorderClassifier: @@ -71,19 +44,32 @@ class BicorderClassifier: 2: "Institutional/Bureaucratic" } - def __init__(self, diagnostic_csv='data/synthetic_1.2.6/readings.csv', + def __init__(self, diagnostic_csv=None, model_path=None): - """Initialize classifier with pre-computed model data.""" + """Initialize classifier with pre-computed model data. + + If diagnostic_csv is None, the most recent synthetic readings.csv is + selected automatically (see bicorder_common.find_training_csv). + """ + if diagnostic_csv is None: + diagnostic_csv = find_training_csv() + if diagnostic_csv is None: + raise FileNotFoundError( + "No training CSV found under data/*/readings.csv — " + "pass diagnostic_csv explicitly." + ) + print(f"No training CSV specified; using {diagnostic_csv}") if model_path is None: model_path = str(Path(diagnostic_csv).parent / 'analysis' / 'data') - self._diagnostic_csv = diagnostic_csv + self._diagnostic_csv = str(diagnostic_csv) self.model_path = Path(model_path) self.scaler = StandardScaler() self.lda = None self.cluster_centroids = None - # Derive dimension lists from bicorder.json - self.DIMENSIONS, self.KEY_DIMENSIONS = _load_bicorder_dimensions() + # Derive dimension lists (and version) from bicorder.json + self.DIMENSIONS, self.KEY_DIMENSIONS, self.bicorder_version = ( + load_bicorder_dimensions()) # Load training data to fit scaler and LDA self._load_model() @@ -92,7 +78,13 @@ class BicorderClassifier: """Load and fit the classification model from analysis results.""" # Load the original data and cluster assignments df = pd.read_csv(self._diagnostic_csv) - clusters = pd.read_csv(self.model_path / 'kmeans_clusters.csv') + clusters_path = self.model_path / 'kmeans_clusters.csv' + if not clusters_path.exists(): + raise FileNotFoundError( + f"No clustering results for training data: {clusters_path} not found. " + f"Run `python3 scripts/multivariate_analysis.py {self._diagnostic_csv}` " + f"first to generate cluster assignments.") + clusters = pd.read_csv(clusters_path) # Rename old column names to match current bicorder.json df = df.rename(columns=_COLUMN_RENAMES) @@ -283,6 +275,8 @@ class BicorderClassifier: def save_model(self, output_path='bicorder_classifier_model.json'): """Save model parameters for use without scikit-learn.""" model_data = { + 'bicorder_version': self.bicorder_version, + 'trained_on': self._diagnostic_csv, 'dimensions': self.DIMENSIONS, 'key_dimensions': self.KEY_DIMENSIONS, 'cluster_names': self.CLUSTER_NAMES, diff --git a/analysis/scripts/bicorder_common.py b/analysis/scripts/bicorder_common.py new file mode 100644 index 0000000..2ac99ac --- /dev/null +++ b/analysis/scripts/bicorder_common.py @@ -0,0 +1,161 @@ +#!/usr/bin/env python3 +""" +Shared bicorder-version helpers used by the analysis scripts. + +Single source of truth for: + - historical gradient column renames (old CSV/JSON names → current bicorder.json names) + - reading the bicorder version recorded in a readings.csv + - locating a training CSV matching a target bicorder version + - loading gradient definitions from bicorder.json + +Whenever gradients are renamed in ../bicorder.json, update COLUMN_RENAMES +here — scripts that read older readings data import from this module rather +than keeping their own copies of the map. +""" + +import csv +import json +import re +from pathlib import Path + +# analysis/ directory (this file lives in analysis/scripts/) +_ANALYSIS_DIR = Path(__file__).resolve().parent.parent + +# Shared protocol inputs / run directories +DATA_DIR = _ANALYSIS_DIR / 'data' + +# Repository-root bicorder.json (the current gradient definitions) +_BICORDER_JSON = _ANALYSIS_DIR.parent / 'bicorder.json' + +# Historical column renames: old CSV column names → current bicorder.json names. +# Add an entry here whenever gradient terms are renamed in bicorder.json. +COLUMN_RENAMES = { + 'Design_elite_vs_vernacular': 'Design_formal_vs_vernacular', + 'Design_institutional_vs_vernacular': 'Design_formal_vs_vernacular', + 'Entanglement_exclusive_vs_non-exclusive': 'Entanglement_monopolistic_vs_pluralistic', + 'Experience_sufficient_vs_insufficient': 'Experience_sufficient_vs_limited', + 'Experience_Kafka_vs_Whitehead': 'Experience_restraining_vs_liberating', +} + +# Dimension column prefixes, in bicorder.json set order +DIMENSION_PREFIXES = ('Design_', 'Entanglement_', 'Experience_') + + +def apply_renames(df_or_columns): + """Canonicalize old column names to the current bicorder.json terminology. + + Works on a pandas DataFrame (returns a renamed copy) or on any iterable of + column names (returns a mapped list). + """ + if hasattr(df_or_columns, 'rename'): # pandas DataFrame + return df_or_columns.rename(columns=COLUMN_RENAMES) + return [COLUMN_RENAMES.get(c, c) for c in df_or_columns] + + +def dimension_columns(columns): + """Filter an iterable of column names down to the diagnostic gradient columns.""" + return [c for c in columns if c.startswith(DIMENSION_PREFIXES)] + + +def load_bicorder_dimensions(bicorder_path=None): + """Read DIMENSIONS, KEY_DIMENSIONS, and version from bicorder.json. + + Returns (dimensions, key_dimensions, version) where dimension names follow + the CSV convention `Set_left_vs_right`. + """ + with open(bicorder_path or _BICORDER_JSON) as f: + data = json.load(f) + dimensions = [] + key_dimensions = [] + for category in data['diagnostic']: + set_name = category['set_name'] + for gradient in category['gradients']: + dim_name = f"{set_name}_{gradient['term_left']}_vs_{gradient['term_right']}" + dimensions.append(dim_name) + if gradient.get('shortform', False): + key_dimensions.append(dim_name) + return dimensions, key_dimensions, data.get('version', '') + + +def version_key(version): + """Sort key for dotted version strings like '1.4.0'. Unknown formats sort lowest.""" + try: + return tuple(int(part) for part in str(version).split('.')) + except ValueError: + return (0,) + + +# Run-directory naming convention: data/_/ with a dotted version +# (e.g. synthetic_1.4.0). Dates like manual_20260320 don't match (no dots). +_DIR_VERSION = re.compile(r'.*_(?P\d+(?:\.\d+)+)$') + + +def dir_version(dir_path): + """Version inferred from a run-directory name (data/_ convention).""" + name = Path(dir_path).name + match = _DIR_VERSION.match(name) + return match.group('version') if match else '' + + +def csv_version(csv_path): + """Bicorder version associated with a readings.csv, or ''. + + Prefers the version recorded in the file (bicorder_version / version column, + via read_csv_version); falls back to the run-directory naming convention + (data/_/) so legacy runs like synthetic_1.2.6 still resolve. + """ + path = Path(csv_path) + return read_csv_version(path) or dir_version(path.parent) + + +def read_csv_version(csv_path): + """Return the bicorder version recorded in a readings.csv, or ''. + + Prefers the `bicorder_version` provenance column (synthetic runs); falls + back to the per-reading `version` column (manual runs, via json_to_csv). + Uses the most common value so mixed-version datasets still resolve. + """ + with open(csv_path, newline='', encoding='utf-8-sig') as f: + reader = csv.DictReader(f) + if not reader.fieldnames: + return '' + for col in ('bicorder_version', 'version'): + if col in reader.fieldnames: + counts = {} + for row in reader: + value = (row.get(col) or '').strip() + if value: + counts[value] = counts.get(value, 0) + 1 + if counts: + return max(counts, key=counts.get) + return '' + + +def find_training_csv(version=None, data_dir=None, exclude=None): + """Locate a synthetic readings.csv suitable for classifier training. + + Scans data/*/readings.csv (whatever their directories are named — run + directories just need to be self-describing via a version column or a + data/_/ name) and prefers: + 1. a file whose version matches `version` (when given), excluding + `exclude` (the input file itself, to avoid circular training when + classifying a synthetic run against itself) + 2. failing that, the most recent version recorded in the data directory + + Returns a Path, or None if no candidate exists. + """ + data_dir = Path(data_dir) if data_dir else DATA_DIR + candidates = [] + for path in sorted(data_dir.glob('*/readings.csv')): + path = path.resolve() + if exclude and path == Path(exclude).resolve(): + continue + candidates.append((path, csv_version(path))) + if not candidates: + return None + if version: + matches = [path for path, ver in candidates if ver == version] + if matches: + return matches[0] + best = max(candidates, key=lambda item: version_key(item[1]) if item[1] else (0,)) + return best[0] \ No newline at end of file diff --git a/analysis/scripts/classify_readings.py b/analysis/scripts/classify_readings.py index 829cc1f..2255a38 100644 --- a/analysis/scripts/classify_readings.py +++ b/analysis/scripts/classify_readings.py @@ -2,14 +2,18 @@ """ Apply the BicorderClassifier to all readings in a CSV and save results. -Uses the synthetic-trained LDA model by default. Missing dimensions are +Training data is selected automatically by version: the input CSV's recorded +`bicorder_version` (or `version`) is matched against the runs under data/ so +the classifier trains on a same-version synthetic run whenever one exists +(see bicorder_common.find_training_csv). Older-format columns are renamed to +the current bicorder terminology automatically. Missing dimensions are filled with the neutral value (5), so shortform readings can still be classified — though with lower confidence. Usage: python3 scripts/classify_readings.py data/manual_20260320/readings.csv python3 scripts/classify_readings.py data/manual_20260320/readings.csv \\ - --training data/synthetic_1.2.6/readings.csv \\ + --training data/synthetic_1.4.0/readings.csv \\ --output data/manual_20260320/analysis/classifications.csv """ @@ -20,6 +24,7 @@ from pathlib import Path import pandas as pd from bicorder_classifier import BicorderClassifier +from bicorder_common import csv_version, find_training_csv, apply_renames def main(): @@ -28,9 +33,10 @@ def main(): ) parser.add_argument('input_csv', help='Readings CSV to classify') parser.add_argument( - '--training', - default='data/synthetic_1.2.6/readings.csv', - help='Training CSV for classifier (default: synthetic_1.2.6)' + '--training', default=None, + help='Training CSV for classifier (default: auto-selected to match ' + "the input's recorded bicorder_version; falls back to the most " + 'recent synthetic run)' ) parser.add_argument( '--output', default=None, @@ -39,16 +45,46 @@ def main(): args = parser.parse_args() input_path = Path(args.input_csv) + input_version = csv_version(input_path) + + # Auto-select training data matching the input's recorded bicorder version + if args.training: + training_path = Path(args.training) + else: + training_path = find_training_csv(version=input_version, exclude=input_path) + if training_path is None: + raise SystemExit("No training CSV found under data/*/readings.csv — " + "pass --training explicitly.") + training_version = csv_version(training_path) + if input_version and training_version == input_version: + note = f"matching v{input_version}" + elif training_version: + note = (f"fell back to v{training_version} " + f"(no other v{input_version} run under data/ to avoid circular training)") + else: + note = "(input records no version; most recent run selected)" + print(f"Auto-selected training data: {training_path} ({note})") + + # Warn when training data comes from a different bicorder version + # (older column names are auto-renamed via bicorder_common) + training_version = csv_version(training_path) + if input_version and training_version and input_version != training_version: + print(f"Warning: training data is bicorder v{training_version} " + f"but input is v{input_version}; renamed columns are aligned " + f"automatically, but compare versions (especially gradient " + f"orderings) when interpreting results.") + output_path = ( Path(args.output) if args.output else input_path.parent / 'analysis' / 'classifications.csv' ) output_path.parent.mkdir(parents=True, exist_ok=True) - print(f"Loading classifier (training: {args.training})...") - classifier = BicorderClassifier(diagnostic_csv=args.training) + print(f"Loading classifier (training: {training_path})...") + classifier = BicorderClassifier(diagnostic_csv=training_path) df = pd.read_csv(input_path) + df = apply_renames(df) # canonicalize old gradient column names print(f"Classifying {len(df)} readings from {input_path}...") rows = [] diff --git a/analysis/scripts/compare_analyses.py b/analysis/scripts/compare_analyses.py index 17ecffc..4d501e8 100644 --- a/analysis/scripts/compare_analyses.py +++ b/analysis/scripts/compare_analyses.py @@ -2,13 +2,42 @@ """ Compare multiple analysis CSV files to determine which most closely resembles a reference file. Uses Euclidean distance, correlation, and RMSE metrics. + +Readings files are canonicalized to the current bicorder terminology (history of +renames in bicorder_common.py), and comparison runs on the gradient columns +shared by all files — so any versions can be compared, and renamed gradients +remain comparable across version boundaries. + +Usage: + # Legacy audit: manual review vs. the three model test runs (as in README) + python3 scripts/compare_analyses.py + + # Explicit: reference file first, then any number of comparison files + python3 scripts/compare_analyses.py \ + data/synthetic_1.2.6/readings.csv data/synthetic_1.4.0/readings.csv """ +import argparse +import sys import pandas as pd import numpy as np from scipy.stats import pearsonr from pathlib import Path +from bicorder_common import apply_renames, csv_version + + +def load_canonical(path): + """Load a readings CSV, canonicalize columns, and coerce gradient values to numeric.""" + df = pd.read_csv(path, quotechar='"', escapechar='\\', engine='python') + df = apply_renames(df) + numeric_cols = [col for col in df.columns if + col.startswith(('Design_', 'Entanglement_', 'Experience_'))] + for col in numeric_cols: + df[col] = pd.to_numeric(df[col], errors='coerce') + return df, numeric_cols + + def calculate_euclidean_distance(df1, df2, numeric_cols): """Calculate Euclidean distance between two dataframes.""" distances = [] @@ -45,18 +74,23 @@ def calculate_correlation(df1, df2, numeric_cols): def compare_analyses(reference_file, comparison_files): """Compare multiple analysis files to a reference file.""" - # Read reference file + # Read and canonicalize reference file print(f"Reading reference file: {reference_file}") - ref_df = pd.read_csv(reference_file, quotechar='"', escapechar='\\', engine='python') - # Get numeric columns (all the rating dimensions) - numeric_cols = [col for col in ref_df.columns if - col.startswith(('Design_', 'Entanglement_', 'Experience_'))] + ref_version = csv_version(reference_file) + if ref_version: + print(f" bicorder version recorded: v{ref_version}") + ref_df, ref_numeric = load_canonical(reference_file) - # Convert numeric columns to numeric type, coercing errors to NaN - for col in numeric_cols: - ref_df[col] = pd.to_numeric(ref_df[col], errors='coerce') + # Repeated Descriptor entries (kept as control cases in the datasets) would + # multiply rows in the Descriptor-based merge; keep first occurrence like + # the classifier does. + if 'Descriptor' in ref_df.columns: + before = len(ref_df) + ref_df = ref_df.drop_duplicates(subset='Descriptor', keep='first') + if len(ref_df) < before: + print(f" Deduplicated reference: {before} → {len(ref_df)} rows (kept first of repeated Descriptor)") - print(f"\nFound {len(numeric_cols)} numeric dimensions to compare") + print(f"\nFound {len(ref_numeric)} numeric dimensions in reference file") print(f"Comparing {len(ref_df)} protocols\n") print("="*80) @@ -66,12 +100,23 @@ def compare_analyses(reference_file, comparison_files): print(f"\nComparing: {Path(comp_file).name}") print("-"*80) - # Read comparison file - comp_df = pd.read_csv(comp_file, quotechar='"', escapechar='\\', engine='python') + # Read and canonicalize comparison file + comp_version = csv_version(comp_file) + if comp_version: + print(f" bicorder version recorded: v{comp_version}") + comp_df, comp_numeric = load_canonical(comp_file) + if 'Descriptor' in comp_df.columns: + before = len(comp_df) + comp_df = comp_df.drop_duplicates(subset='Descriptor', keep='first') + if len(comp_df) < before: + print(f" Deduplicated comparison: {before} → {len(comp_df)} rows (kept first of repeated Descriptor)") - # Convert numeric columns to numeric type, coercing errors to NaN - for col in numeric_cols: - comp_df[col] = pd.to_numeric(comp_df[col], errors='coerce') + # Restrict to columns shared by both files (post-rename): enables + # comparing across bicorder versions when gradients were renamed + numeric_cols = [col for col in ref_numeric if col in comp_numeric] + missing = [col for col in ref_numeric if col not in comp_numeric] + if missing: + print(f" Note: {len(missing)} gradient(s) absent here are excluded: {', '.join(missing)}") # Ensure same protocols in same order (match by Descriptor) if 'Descriptor' in ref_df.columns and 'Descriptor' in comp_df.columns: @@ -155,32 +200,58 @@ def compare_analyses(reference_file, comparison_files): return results -if __name__ == "__main__": - # Define file paths - reference_file = "data/synthetic_1.2.6/readings_manual.csv" - comparison_files = [ +def main(argv=None): + """CLI entry point. + + With no arguments, falls back to the legacy audit: the 1.2.6 manual review + against the three model test runs (as described in README.md). + """ + legacy_reference = "data/synthetic_1.2.6/readings_manual.csv" + legacy_comparisons = [ "data/synthetic_1.2.6/readings_gemma3-12b.csv", "data/synthetic_1.2.6/readings_gpt-oss.csv", - "data/synthetic_1.2.6/readings_mistral.csv" + "data/synthetic_1.2.6/readings_mistral.csv", ] + if argv is None: + argv = sys.argv[1:] + if argv: + parser = argparse.ArgumentParser( + description='Compare readings CSVs to a reference (Euclidean distance, RMSE, correlation)', + epilog="""Example (cross-version): + python3 scripts/compare_analyses.py \\ + data/synthetic_1.4.0/readings.csv data/synthetic_1.2.6/readings.csv +""", + ) + parser.add_argument('reference', help='Reference readings CSV') + parser.add_argument('comparisons', nargs='+', help='Comparison readings CSVs') + args = parser.parse_args(argv) + reference_file, comparison_files = args.reference, args.comparisons + else: + reference_file, comparison_files = legacy_reference, legacy_comparisons + # Check if files exist if not Path(reference_file).exists(): print(f"Error: Reference file '{reference_file}' not found") - exit(1) + sys.exit(1) + existing = [file for file in comparison_files if Path(file).exists()] for file in comparison_files: if not Path(file).exists(): print(f"Warning: Comparison file '{file}' not found, skipping...") - comparison_files.remove(file) - if not comparison_files: + if not existing: print("Error: No comparison files found") - exit(1) + sys.exit(1) # Run comparison - results = compare_analyses(reference_file, comparison_files) + results = compare_analyses(reference_file, existing) print("\n" + "="*80) print("Analysis complete!") print("="*80) + return results + + +if __name__ == "__main__": + main() diff --git a/analysis/scripts/export_model_for_js.py b/analysis/scripts/export_model_for_js.py index c7aab92..1f97dcd 100644 --- a/analysis/scripts/export_model_for_js.py +++ b/analysis/scripts/export_model_for_js.py @@ -3,10 +3,8 @@ Export the cluster classification model to JSON for use in JavaScript. Reads dimension names directly from bicorder.json so the model always -stays in sync with the current bicorder structure. - -When gradients are renamed in bicorder.json, add the old→new mapping to -COLUMN_RENAMES so the training CSV columns are correctly aligned. +stays in sync with the current bicorder structure. Column renames and the +version record live in bicorder_common.py (shared by all analysis scripts). Usage: python3 scripts/export_model_for_js.py data/synthetic_1.2.6/readings.csv @@ -22,37 +20,7 @@ import numpy as np from sklearn.preprocessing import StandardScaler from sklearn.discriminant_analysis import LinearDiscriminantAnalysis -# Path to bicorder.json (relative to this script) -BICORDER_JSON = Path(__file__).parent.parent.parent / 'bicorder.json' - -# Historical column renames: maps old CSV column names → current bicorder.json names. -# Add an entry here whenever gradient terms are renamed in bicorder.json. -COLUMN_RENAMES = { - 'Design_elite_vs_vernacular': 'Design_formal_vs_vernacular', - 'Design_institutional_vs_vernacular': 'Design_formal_vs_vernacular', - 'Entanglement_exclusive_vs_non-exclusive': 'Entanglement_monopolistic_vs_pluralistic', - 'Experience_sufficient_vs_insufficient': 'Experience_sufficient_vs_limited', - 'Experience_Kafka_vs_Whitehead': 'Experience_restraining_vs_liberating', -} - - -def load_bicorder_dimensions(bicorder_path): - """Read DIMENSIONS and KEY_DIMENSIONS from bicorder.json.""" - with open(bicorder_path) as f: - data = json.load(f) - - dimensions = [] - key_dimensions = [] - - for category in data['diagnostic']: - set_name = category['set_name'] - for gradient in category['gradients']: - dim_name = f"{set_name}_{gradient['term_left']}_vs_{gradient['term_right']}" - dimensions.append(dim_name) - if gradient.get('shortform', False): - key_dimensions.append(dim_name) - - return dimensions, key_dimensions, data['version'] +from bicorder_common import COLUMN_RENAMES, csv_version, load_bicorder_dimensions def main(): @@ -74,14 +42,28 @@ Example usage: analysis_dir = dataset_dir / 'analysis' # Derive dimensions and version from bicorder.json - DIMENSIONS, KEY_DIMENSIONS, BICORDER_VERSION = load_bicorder_dimensions(BICORDER_JSON) + DIMENSIONS, KEY_DIMENSIONS, BICORDER_VERSION = load_bicorder_dimensions() print(f"Loaded bicorder.json v{BICORDER_VERSION}") print(f"Dimensions: {len(DIMENSIONS)}, key dimensions: {len(KEY_DIMENSIONS)}") # Load data df = pd.read_csv(args.input_csv) - clusters = pd.read_csv(analysis_dir / 'data' / 'kmeans_clusters.csv') + clusters_path = analysis_dir / 'data' / 'kmeans_clusters.csv' + if not clusters_path.exists(): + raise FileNotFoundError( + f"No clustering results for {args.input_csv}: {clusters_path} not found. " + f"Run `python3 scripts/multivariate_analysis.py {args.input_csv}` first " + f"to generate cluster assignments.") + clusters = pd.read_csv(clusters_path) + + # Flag a version mismatch between the training data and current bicorder.json + # (column renames keep the data aligned, but the clustering itself was done + # against the recorded version's gradient structure) + recorded_version = csv_version(args.input_csv) + if recorded_version and recorded_version != BICORDER_VERSION: + print(f"Note: training data records bicorder v{recorded_version}; " + f"current bicorder.json is v{BICORDER_VERSION} (columns auto-renamed)") # Rename old column names to match current bicorder.json df = df.rename(columns=COLUMN_RENAMES) diff --git a/analysis/scripts/sync_readings.sh b/analysis/scripts/sync_readings.sh index 90fb5d8..d76d248 100755 --- a/analysis/scripts/sync_readings.sh +++ b/analysis/scripts/sync_readings.sh @@ -7,7 +7,11 @@ # scripts/sync_readings.sh data/manual_20260320 # scripts/sync_readings.sh data/manual_20260320 --no-analysis # scripts/sync_readings.sh data/manual_20260320 --min-coverage 0.8 -# scripts/sync_readings.sh data/manual_20260320 --training data/synthetic_1.2.6/readings.csv +# scripts/sync_readings.sh data/manual_20260320 --training data/synthetic_1.4.0/readings.csv +# +# By default the classifier training CSV is auto-selected to match the synced +# dataset's recorded bicorder version (see scripts/bicorder_common.py); --training +# overrides that. # # .sync_source format: # REMOTE_URL=https://git.example.org/user/repo @@ -18,7 +22,7 @@ set -euo pipefail DATASET_DIR="${1:?Usage: $0 [--no-analysis] [--min-coverage N]}" RUN_ANALYSIS=true MIN_COVERAGE=0.8 -TRAINING_CSV="data/synthetic_1.2.6/readings.csv" +TRAINING_CSV="" # empty → let classify_readings.py auto-match the dataset's bicorder version shift || true while [[ $# -gt 0 ]]; do @@ -96,11 +100,18 @@ if [[ "$RUN_ANALYSIS" == true ]]; then echo "Generating LDA visualization..." "$PYTHON" scripts/lda_visualization.py "$DATASET_DIR/readings.csv" - echo "" - echo "Classifying readings (training: $TRAINING_CSV)..." - "$PYTHON" scripts/classify_readings.py \ - "$DATASET_DIR/readings.csv" \ - --training "$TRAINING_CSV" + if [[ -n "$TRAINING_CSV" ]]; then + echo "" + echo "Classifying readings (training: $TRAINING_CSV)..." + "$PYTHON" scripts/classify_readings.py \ + "$DATASET_DIR/readings.csv" \ + --training "$TRAINING_CSV" + else + echo "" + echo "Classifying readings (training auto-matched by bicorder version)..." + "$PYTHON" scripts/classify_readings.py \ + "$DATASET_DIR/readings.csv" + fi fi echo "" diff --git a/analysis/scripts/univariate_analysis.py b/analysis/scripts/univariate_analysis.py new file mode 100644 index 0000000..4a2d953 --- /dev/null +++ b/analysis/scripts/univariate_analysis.py @@ -0,0 +1,225 @@ +#!/usr/bin/env python3 +""" +Univariate analysis of bicorder readings: per-protocol and per-gradient averages. + +Reproduces the ad-hoc averages workflow described in README.md for any +readings CSV, version-agnostic: gradient columns are discovered from the file +itself (canonicalized via bicorder_common.COLUMN_RENAMES), so every run +directory can be analyzed with the same command. + +Outputs (default /analysis/): + plots/protocol_averages.png — protocol averages, ascending + plots/gradient_averages.png — gradient averages with gaps between the three sets + plots/averages_histogram.png — distribution of protocol averages + data/protocol_averages.csv — per-protocol mean over all gradients + data/gradient_averages.csv — per-gradient mean/median/coverage + reports/univariate_summary.txt — printed summary + +With --img, also publishes the three summary PNGs to an image directory +(analysis/img/ by default — the charts the README links from img/). + +Usage: + python3 scripts/univariate_analysis.py data/synthetic_1.4.0/readings.csv + python3 scripts/univariate_analysis.py data/synthetic_1.4.0/readings.csv --output data/synthetic_1.4.0/analysis + python3 scripts/univariate_analysis.py data/synthetic_1.4.0/readings.csv --img # also refresh img/ +""" + +import argparse +import shutil +import sys +from pathlib import Path +import warnings +warnings.filterwarnings('ignore') + +import pandas as pd +import numpy as np +import matplotlib.pyplot as plt + +from bicorder_common import apply_renames, csv_version, dimension_columns + +MIDPOINT = 5 # all gradients run 1 (hard) .. 9 (soft) + + +def main(): + parser = argparse.ArgumentParser( + description='Univariate analysis of Protocol Bicorder readings', + formatter_class=argparse.RawDescriptionHelpFormatter, + epilog=""" +Examples: + python3 scripts/univariate_analysis.py data/synthetic_1.4.0/readings.csv + python3 scripts/univariate_analysis.py data/synthetic_1.2.6/readings.csv --output data/synthetic_1.2.6/analysis + """, + ) + parser.add_argument('csv_file', help='Readings CSV (e.g. data/synthetic_1.4.0/readings.csv)') + parser.add_argument('--output', '-o', default=None, + help='Output directory (default: /analysis)') + parser.add_argument('--img', nargs='?', const='img', default=None, + help='Publish the three summary PNGs to an image directory in addition ' + 'to the run analysis outputs (default with --img: img/ at the ' + "analysis root — the charts the README links from img/)") + args = parser.parse_args() + + if not Path(args.csv_file).exists(): + print(f"Error: File not found: {args.csv_file}") + sys.exit(1) + + dataset_dir = Path(args.csv_file).parent + output_dir = Path(args.output) if args.output else dataset_dir / 'analysis' + (output_dir / 'plots').mkdir(parents=True, exist_ok=True) + (output_dir / 'data').mkdir(parents=True, exist_ok=True) + (output_dir / 'reports').mkdir(parents=True, exist_ok=True) + + version = csv_version(args.csv_file) + print("=" * 80) + print("PROTOCOL BICORDER - UNIVARIATE ANALYSIS" + + (f" (bicorder v{version})" if version else "")) + print(f"Source: {args.csv_file}") + print("=" * 80) + + df = pd.read_csv(args.csv_file) + df = apply_renames(df) + + # Identify gradient columns, grouped by set in bicorder.json order + all_dims = dimension_columns(df.columns.tolist()) + groups = [[c for c in all_dims if c.startswith(prefix)] + for prefix in ('Design_', 'Entanglement_', 'Experience_')] + dimension_cols = [c for g in groups for c in g] + + if not dimension_cols: + print("Error: no gradient columns found") + sys.exit(1) + + values = df[dimension_cols].apply(pd.to_numeric, errors='coerce') + + # --- Per-protocol averages --- + protocol = pd.DataFrame({ + 'Descriptor': df.get('Descriptor', pd.Series(range(len(df)))), + 'average': values.mean(axis=1), + 'n_gradients_scored': values.notna().sum(axis=1), + }).dropna(subset=['average']).sort_values('average').reset_index(drop=True) + protocol.to_csv(output_dir / 'data' / 'protocol_averages.csv', index=False) + print(f"\nSaved: {output_dir / 'data' / 'protocol_averages.csv'}") + + # --- Per-gradient averages --- + gradient = pd.DataFrame({ + 'gradient': dimension_cols, + 'set': [c.split('_', 1)[0] for c in dimension_cols], + 'mean': [values[c].mean() for c in dimension_cols], + 'median': [values[c].median() for c in dimension_cols], + 'coverage': [values[c].notna().mean() for c in dimension_cols], + }).sort_values('mean', ascending=False) + gradient.to_csv(output_dir / 'data' / 'gradient_averages.csv', index=False) + print(f"Saved: {output_dir / 'data' / 'gradient_averages.csv'}") + + # --- Summary statistics on protocol averages --- + avg = protocol['average'] + mean, median, std = avg.mean(), avg.median(), avg.std() + pearson_skew = 3 * (mean - median) / std if std else float('nan') + moment_skew = avg.skew() # bias-corrected Fisher moment skewness + + lines = [] + lines.append(f"Readings analyzed: {len(protocol)} protocols x {len(dimension_cols)} gradients") + if version: + lines.append(f"Bicorder version: {version}") + lines.append("") + lines.append("Protocol averages:") + for label, value in [('mean', mean), ('median', median), ('std', std), + ('min', avg.min()), ('max', avg.max())]: + lines.append(f" {label:8s} = {value:.3f}") + lines.append(f" midpoint = {MIDPOINT}") + lines.append(f" deviation from midpoint = {mean - MIDPOINT:+.3f} " + f"(normalized over half-range 4: {(mean - MIDPOINT) / 4:+.3f})") + lines.append(f" Pearson skew (3*(mean-median)/std) = {pearson_skew:+.3f}") + lines.append(f" Fisher moment skew = {moment_skew:+.3f}") + lines.append("") + lines.append("Gradient averages (all sets):") + for _, row in gradient.iterrows(): + lines.append(f" {row['gradient'][:60]:60s} mean={row['mean']:.2f} coverage={row['coverage']:.0%}") + lines.append("") + lines.append(f"Extremes: highest = {gradient.iloc[0]['gradient']} ({gradient.iloc[0]['mean']:.2f}); " + f"lowest = {gradient.iloc[-1]['gradient']} ({gradient.iloc[-1]['mean']:.2f})") + + summary_text = "\n".join(lines) + report_path = output_dir / 'reports' / 'univariate_summary.txt' + report_path.write_text(summary_text + "\n") + print(f"\nSaved: {report_path}\n") + print(summary_text) + + # --- Plots --- + # Protocol averages, ascending (cf. img/protocol_averages.png) + fig, ax = plt.subplots(figsize=(12, 8)) + ax.plot(range(len(protocol)), protocol['average'], linewidth=1.2) + ax.axhline(MIDPOINT, color='gray', linestyle='--', linewidth=1, label=f'midpoint ({MIDPOINT})') + ax.set_title('Protocol averages (ascending order)' + (f' — bicorder v{version}' if version else '')) + ax.set_xlabel('Protocol (ranked by average)') + ax.set_ylabel('Average gradient value') + ax.set_ylim(0, 10) + ax.legend() + plt.tight_layout() + path = output_dir / 'plots' / 'protocol_averages.png' + plt.savefig(path, dpi=300, bbox_inches='tight') + plt.close() + print(f"\nSaved: {path}") + + # Histogram (cf. img/averages_histogram.png) + fig, ax = plt.subplots(figsize=(10, 6)) + ax.hist(avg, bins=40, color='#4C72B0', edgecolor='white') + ax.axvline(MIDPOINT, color='gray', linestyle='--', linewidth=1, label=f'midpoint ({MIDPOINT})') + ax.axvline(mean, color='#C44E52', linewidth=1, label=f'mean ({mean:.2f})') + ax.set_title('Distribution of protocol averages') + ax.set_xlabel('Average gradient value') + ax.set_ylabel('Number of protocols') + ax.legend() + plt.tight_layout() + path = output_dir / 'plots' / 'averages_histogram.png' + plt.savefig(path, dpi=300, bbox_inches='tight') + plt.close() + print(f"Saved: {path}") + + # Gradient averages, sorted within each set, with visual gaps between sets + # (cf. img/gradient_averages.png) + palette = {'Design': '#4C72B0', 'Entanglement': '#55A868', 'Experience': '#C44E52'} + ordered_cols = [c for g in groups + for c in sorted(g, key=lambda cv: values[cv].mean(), reverse=True)] + means = [values[c].mean() for c in ordered_cols] + colors = [palette[c.split('_', 1)[0]] for c in ordered_cols] + x_pos, pos = [], 0.0 + for gi, group in enumerate(groups): + group_sorted = sorted(group, key=lambda cv: values[cv].mean(), reverse=True) + if gi > 0: + pos += 1.5 # visual gap between gradient sets + x_pos.extend(pos + i for i in range(len(group_sorted))) + pos += len(group_sorted) + fig, ax = plt.subplots(figsize=(14, 6)) + ax.bar(x_pos, means, color=colors) + ax.axhline(MIDPOINT, color='gray', linestyle='--', linewidth=1, label=f'midpoint ({MIDPOINT})') + ax.set_title('Gradient averages (gaps separate the three gradient sets)' + + (f' — bicorder v{version}' if version else '')) + ax.set_xticks(x_pos) + ax.set_xticklabels([c.replace('_vs_', '\nvs\n') for c in ordered_cols], fontsize=7) + ax.set_ylabel('Average value') + ax.set_ylim(0, 10) + ax.legend() + plt.tight_layout() + path = output_dir / 'plots' / 'gradient_averages.png' + plt.savefig(path, dpi=300, bbox_inches='tight') + plt.close() + print(f"Saved: {path}") + + # Optionally publish the summary charts for docs (the README links img/…) + if args.img: + img_dir = Path(args.img) + if not img_dir.is_absolute(): + img_dir = Path(__file__).resolve().parent.parent / img_dir + img_dir.mkdir(parents=True, exist_ok=True) + published = [] + for name in ('protocol_averages.png', 'averages_histogram.png', 'gradient_averages.png'): + shutil.copy2(output_dir / 'plots' / name, img_dir / name) + published.append(str(img_dir / name)) + print(f"Published charts → {', '.join(published)}") + + print("\nDone.") + + +if __name__ == '__main__': + main() \ No newline at end of file