feat: version-agnostic analysis scripts with shared version helpers

- scripts/bicorder_common.py (new): single source of truth for historical
  gradient renames (COLUMN_RENAMES), version detection (bicorder_version col
  → version col → data/<type>_<version>/ dir convention), and training-CSV
  auto-selection (find_training_csv)
- classify_readings.py: auto-select training run by recorded bicorder version
  (excludes the input itself to avoid circular training), canonicalize old
  column names, version-mismatch warnings
- bicorder_classifier.py / export_model_for_js.py: share renames + dimension
  loader; instructive error when clustering results are missing;
  bicorder_version recorded in exported models
- compare_analyses.py: CLI (reference + comparison CSVs), rename
  canonicalization so runs of different versions align on shared gradients,
  Descriptor dedup (was silently skewing merges); legacy no-arg audit intact
- scripts/univariate_analysis.py (new): per-protocol/per-gradient averages,
  distributions, summary stats; --img publishes the three README summary
  charts to img/
- sync_readings.sh: defer classifier training to auto-matching; gitignore
  analysis/.venv and __pycache__
This commit is contained in:
Nathan Schneider committed 2026-10-02 08:32:59 -06:00
1 parent a4f16e8e4d
commit 55cbd6cd5d
8 files changed
+591 -109

No files matched your search

+2
View File
@@ -1,3 +1,5 @@
tmp
analysis/venv
analysis/.venv
__pycache__/
.aider*
+28 -34
View File
@@ -30,34 +30,7 @@ from sklearn.discriminant_analysis import LinearDiscriminantAnalysis
import json
from pathlib import Path
# Path to bicorder.json (relative to this script)
_BICORDER_JSON = Path(__file__).parent.parent.parent / 'bicorder.json'
# Historical column renames: maps old CSV column names → current bicorder.json names.
# Add an entry here whenever gradient terms are renamed in bicorder.json.
_COLUMN_RENAMES = {
'Design_elite_vs_vernacular': 'Design_formal_vs_vernacular',
'Design_institutional_vs_vernacular': 'Design_formal_vs_vernacular',
'Entanglement_exclusive_vs_non-exclusive': 'Entanglement_monopolistic_vs_pluralistic',
'Experience_sufficient_vs_insufficient': 'Experience_sufficient_vs_limited',
'Experience_Kafka_vs_Whitehead': 'Experience_restraining_vs_liberating',
}
def _load_bicorder_dimensions(bicorder_path=_BICORDER_JSON):
"""Read DIMENSIONS and KEY_DIMENSIONS from bicorder.json."""
with open(bicorder_path) as f:
data = json.load(f)
dimensions = []
key_dimensions = []
for category in data['diagnostic']:
set_name = category['set_name']
for gradient in category['gradients']:
dim_name = f"{set_name}_{gradient['term_left']}_vs_{gradient['term_right']}"
dimensions.append(dim_name)
if gradient.get('shortform', False):
key_dimensions.append(dim_name)
return dimensions, key_dimensions
from bicorder_common import COLUMN_RENAMES as _COLUMN_RENAMES, find_training_csv, load_bicorder_dimensions
class BicorderClassifier:
@@ -71,19 +44,32 @@ class BicorderClassifier:
2: "Institutional/Bureaucratic"
}
def __init__(self, diagnostic_csv='data/synthetic_1.2.6/readings.csv',
def __init__(self, diagnostic_csv=None,
model_path=None):
"""Initialize classifier with pre-computed model data."""
"""Initialize classifier with pre-computed model data.
If diagnostic_csv is None, the most recent synthetic readings.csv is
selected automatically (see bicorder_common.find_training_csv).
"""
if diagnostic_csv is None:
diagnostic_csv = find_training_csv()
if diagnostic_csv is None:
raise FileNotFoundError(
"No training CSV found under data/*/readings.csv — "
"pass diagnostic_csv explicitly."
)
print(f"No training CSV specified; using {diagnostic_csv}")
if model_path is None:
model_path = str(Path(diagnostic_csv).parent / 'analysis' / 'data')
self._diagnostic_csv = diagnostic_csv
self._diagnostic_csv = str(diagnostic_csv)
self.model_path = Path(model_path)
self.scaler = StandardScaler()
self.lda = None
self.cluster_centroids = None
# Derive dimension lists from bicorder.json
self.DIMENSIONS, self.KEY_DIMENSIONS = _load_bicorder_dimensions()
# Derive dimension lists (and version) from bicorder.json
self.DIMENSIONS, self.KEY_DIMENSIONS, self.bicorder_version = (
load_bicorder_dimensions())
# Load training data to fit scaler and LDA
self._load_model()
@@ -92,7 +78,13 @@ class BicorderClassifier:
"""Load and fit the classification model from analysis results."""
# Load the original data and cluster assignments
df = pd.read_csv(self._diagnostic_csv)
clusters = pd.read_csv(self.model_path / 'kmeans_clusters.csv')
clusters_path = self.model_path / 'kmeans_clusters.csv'
if not clusters_path.exists():
raise FileNotFoundError(
f"No clustering results for training data: {clusters_path} not found. "
f"Run `python3 scripts/multivariate_analysis.py {self._diagnostic_csv}` "
f"first to generate cluster assignments.")
clusters = pd.read_csv(clusters_path)
# Rename old column names to match current bicorder.json
df = df.rename(columns=_COLUMN_RENAMES)
@@ -283,6 +275,8 @@ class BicorderClassifier:
def save_model(self, output_path='bicorder_classifier_model.json'):
"""Save model parameters for use without scikit-learn."""
model_data = {
'bicorder_version': self.bicorder_version,
'trained_on': self._diagnostic_csv,
'dimensions': self.DIMENSIONS,
'key_dimensions': self.KEY_DIMENSIONS,
'cluster_names': self.CLUSTER_NAMES,
+161
View File
@@ -0,0 +1,161 @@
#!/usr/bin/env python3
"""
Shared bicorder-version helpers used by the analysis scripts.
Single source of truth for:
- historical gradient column renames (old CSV/JSON names → current bicorder.json names)
- reading the bicorder version recorded in a readings.csv
- locating a training CSV matching a target bicorder version
- loading gradient definitions from bicorder.json
Whenever gradients are renamed in ../bicorder.json, update COLUMN_RENAMES
here — scripts that read older readings data import from this module rather
than keeping their own copies of the map.
"""
import csv
import json
import re
from pathlib import Path
# analysis/ directory (this file lives in analysis/scripts/)
_ANALYSIS_DIR = Path(__file__).resolve().parent.parent
# Shared protocol inputs / run directories
DATA_DIR = _ANALYSIS_DIR / 'data'
# Repository-root bicorder.json (the current gradient definitions)
_BICORDER_JSON = _ANALYSIS_DIR.parent / 'bicorder.json'
# Historical column renames: old CSV column names → current bicorder.json names.
# Add an entry here whenever gradient terms are renamed in bicorder.json.
COLUMN_RENAMES = {
'Design_elite_vs_vernacular': 'Design_formal_vs_vernacular',
'Design_institutional_vs_vernacular': 'Design_formal_vs_vernacular',
'Entanglement_exclusive_vs_non-exclusive': 'Entanglement_monopolistic_vs_pluralistic',
'Experience_sufficient_vs_insufficient': 'Experience_sufficient_vs_limited',
'Experience_Kafka_vs_Whitehead': 'Experience_restraining_vs_liberating',
}
# Dimension column prefixes, in bicorder.json set order
DIMENSION_PREFIXES = ('Design_', 'Entanglement_', 'Experience_')
def apply_renames(df_or_columns):
"""Canonicalize old column names to the current bicorder.json terminology.
Works on a pandas DataFrame (returns a renamed copy) or on any iterable of
column names (returns a mapped list).
"""
if hasattr(df_or_columns, 'rename'): # pandas DataFrame
return df_or_columns.rename(columns=COLUMN_RENAMES)
return [COLUMN_RENAMES.get(c, c) for c in df_or_columns]
def dimension_columns(columns):
"""Filter an iterable of column names down to the diagnostic gradient columns."""
return [c for c in columns if c.startswith(DIMENSION_PREFIXES)]
def load_bicorder_dimensions(bicorder_path=None):
"""Read DIMENSIONS, KEY_DIMENSIONS, and version from bicorder.json.
Returns (dimensions, key_dimensions, version) where dimension names follow
the CSV convention `Set_left_vs_right`.
"""
with open(bicorder_path or _BICORDER_JSON) as f:
data = json.load(f)
dimensions = []
key_dimensions = []
for category in data['diagnostic']:
set_name = category['set_name']
for gradient in category['gradients']:
dim_name = f"{set_name}_{gradient['term_left']}_vs_{gradient['term_right']}"
dimensions.append(dim_name)
if gradient.get('shortform', False):
key_dimensions.append(dim_name)
return dimensions, key_dimensions, data.get('version', '')
def version_key(version):
"""Sort key for dotted version strings like '1.4.0'. Unknown formats sort lowest."""
try:
return tuple(int(part) for part in str(version).split('.'))
except ValueError:
return (0,)
# Run-directory naming convention: data/<type>_<version>/ with a dotted version
# (e.g. synthetic_1.4.0). Dates like manual_20260320 don't match (no dots).
_DIR_VERSION = re.compile(r'.*_(?P<version>\d+(?:\.\d+)+)$')
def dir_version(dir_path):
"""Version inferred from a run-directory name (data/<type>_<version> convention)."""
name = Path(dir_path).name
match = _DIR_VERSION.match(name)
return match.group('version') if match else ''
def csv_version(csv_path):
"""Bicorder version associated with a readings.csv, or ''.
Prefers the version recorded in the file (bicorder_version / version column,
via read_csv_version); falls back to the run-directory naming convention
(data/<type>_<version>/) so legacy runs like synthetic_1.2.6 still resolve.
"""
path = Path(csv_path)
return read_csv_version(path) or dir_version(path.parent)
def read_csv_version(csv_path):
"""Return the bicorder version recorded in a readings.csv, or ''.
Prefers the `bicorder_version` provenance column (synthetic runs); falls
back to the per-reading `version` column (manual runs, via json_to_csv).
Uses the most common value so mixed-version datasets still resolve.
"""
with open(csv_path, newline='', encoding='utf-8-sig') as f:
reader = csv.DictReader(f)
if not reader.fieldnames:
return ''
for col in ('bicorder_version', 'version'):
if col in reader.fieldnames:
counts = {}
for row in reader:
value = (row.get(col) or '').strip()
if value:
counts[value] = counts.get(value, 0) + 1
if counts:
return max(counts, key=counts.get)
return ''
def find_training_csv(version=None, data_dir=None, exclude=None):
"""Locate a synthetic readings.csv suitable for classifier training.
Scans data/*/readings.csv (whatever their directories are named — run
directories just need to be self-describing via a version column or a
data/<type>_<version>/ name) and prefers:
1. a file whose version matches `version` (when given), excluding
`exclude` (the input file itself, to avoid circular training when
classifying a synthetic run against itself)
2. failing that, the most recent version recorded in the data directory
Returns a Path, or None if no candidate exists.
"""
data_dir = Path(data_dir) if data_dir else DATA_DIR
candidates = []
for path in sorted(data_dir.glob('*/readings.csv')):
path = path.resolve()
if exclude and path == Path(exclude).resolve():
continue
candidates.append((path, csv_version(path)))
if not candidates:
return None
if version:
matches = [path for path, ver in candidates if ver == version]
if matches:
return matches[0]
best = max(candidates, key=lambda item: version_key(item[1]) if item[1] else (0,))
return best[0]
+43 -7
View File
@@ -2,14 +2,18 @@
"""
Apply the BicorderClassifier to all readings in a CSV and save results.
Uses the synthetic-trained LDA model by default. Missing dimensions are
Training data is selected automatically by version: the input CSV's recorded
`bicorder_version` (or `version`) is matched against the runs under data/ so
the classifier trains on a same-version synthetic run whenever one exists
(see bicorder_common.find_training_csv). Older-format columns are renamed to
the current bicorder terminology automatically. Missing dimensions are
filled with the neutral value (5), so shortform readings can still be
classified — though with lower confidence.
Usage:
python3 scripts/classify_readings.py data/manual_20260320/readings.csv
python3 scripts/classify_readings.py data/manual_20260320/readings.csv \\
--training data/synthetic_1.2.6/readings.csv \\
--training data/synthetic_1.4.0/readings.csv \\
--output data/manual_20260320/analysis/classifications.csv
"""
@@ -20,6 +24,7 @@ from pathlib import Path
import pandas as pd
from bicorder_classifier import BicorderClassifier
from bicorder_common import csv_version, find_training_csv, apply_renames
def main():
@@ -28,9 +33,10 @@ def main():
)
parser.add_argument('input_csv', help='Readings CSV to classify')
parser.add_argument(
'--training',
default='data/synthetic_1.2.6/readings.csv',
help='Training CSV for classifier (default: synthetic_1.2.6)'
'--training', default=None,
help='Training CSV for classifier (default: auto-selected to match '
"the input's recorded bicorder_version; falls back to the most "
'recent synthetic run)'
)
parser.add_argument(
'--output', default=None,
@@ -39,16 +45,46 @@ def main():
args = parser.parse_args()
input_path = Path(args.input_csv)
input_version = csv_version(input_path)
# Auto-select training data matching the input's recorded bicorder version
if args.training:
training_path = Path(args.training)
else:
training_path = find_training_csv(version=input_version, exclude=input_path)
if training_path is None:
raise SystemExit("No training CSV found under data/*/readings.csv — "
"pass --training explicitly.")
training_version = csv_version(training_path)
if input_version and training_version == input_version:
note = f"matching v{input_version}"
elif training_version:
note = (f"fell back to v{training_version} "
f"(no other v{input_version} run under data/ to avoid circular training)")
else:
note = "(input records no version; most recent run selected)"
print(f"Auto-selected training data: {training_path} ({note})")
# Warn when training data comes from a different bicorder version
# (older column names are auto-renamed via bicorder_common)
training_version = csv_version(training_path)
if input_version and training_version and input_version != training_version:
print(f"Warning: training data is bicorder v{training_version} "
f"but input is v{input_version}; renamed columns are aligned "
f"automatically, but compare versions (especially gradient "
f"orderings) when interpreting results.")
output_path = (
Path(args.output) if args.output
else input_path.parent / 'analysis' / 'classifications.csv'
)
output_path.parent.mkdir(parents=True, exist_ok=True)
print(f"Loading classifier (training: {args.training})...")
classifier = BicorderClassifier(diagnostic_csv=args.training)
print(f"Loading classifier (training: {training_path})...")
classifier = BicorderClassifier(diagnostic_csv=training_path)
df = pd.read_csv(input_path)
df = apply_renames(df) # canonicalize old gradient column names
print(f"Classifying {len(df)} readings from {input_path}...")
rows = []
+95 -24
View File
@@ -2,13 +2,42 @@
"""
Compare multiple analysis CSV files to determine which most closely resembles a reference file.
Uses Euclidean distance, correlation, and RMSE metrics.
Readings files are canonicalized to the current bicorder terminology (history of
renames in bicorder_common.py), and comparison runs on the gradient columns
shared by all files — so any versions can be compared, and renamed gradients
remain comparable across version boundaries.
Usage:
# Legacy audit: manual review vs. the three model test runs (as in README)
python3 scripts/compare_analyses.py
# Explicit: reference file first, then any number of comparison files
python3 scripts/compare_analyses.py \
data/synthetic_1.2.6/readings.csv data/synthetic_1.4.0/readings.csv
"""
import argparse
import sys
import pandas as pd
import numpy as np
from scipy.stats import pearsonr
from pathlib import Path
from bicorder_common import apply_renames, csv_version
def load_canonical(path):
"""Load a readings CSV, canonicalize columns, and coerce gradient values to numeric."""
df = pd.read_csv(path, quotechar='"', escapechar='\\', engine='python')
df = apply_renames(df)
numeric_cols = [col for col in df.columns if
col.startswith(('Design_', 'Entanglement_', 'Experience_'))]
for col in numeric_cols:
df[col] = pd.to_numeric(df[col], errors='coerce')
return df, numeric_cols
def calculate_euclidean_distance(df1, df2, numeric_cols):
"""Calculate Euclidean distance between two dataframes."""
distances = []
@@ -45,18 +74,23 @@ def calculate_correlation(df1, df2, numeric_cols):
def compare_analyses(reference_file, comparison_files):
"""Compare multiple analysis files to a reference file."""
# Read reference file
# Read and canonicalize reference file
print(f"Reading reference file: {reference_file}")
ref_df = pd.read_csv(reference_file, quotechar='"', escapechar='\\', engine='python')
# Get numeric columns (all the rating dimensions)
numeric_cols = [col for col in ref_df.columns if
col.startswith(('Design_', 'Entanglement_', 'Experience_'))]
ref_version = csv_version(reference_file)
if ref_version:
print(f" bicorder version recorded: v{ref_version}")
ref_df, ref_numeric = load_canonical(reference_file)
# Convert numeric columns to numeric type, coercing errors to NaN
for col in numeric_cols:
ref_df[col] = pd.to_numeric(ref_df[col], errors='coerce')
# Repeated Descriptor entries (kept as control cases in the datasets) would
# multiply rows in the Descriptor-based merge; keep first occurrence like
# the classifier does.
if 'Descriptor' in ref_df.columns:
before = len(ref_df)
ref_df = ref_df.drop_duplicates(subset='Descriptor', keep='first')
if len(ref_df) < before:
print(f" Deduplicated reference: {before} → {len(ref_df)} rows (kept first of repeated Descriptor)")
print(f"\nFound {len(numeric_cols)} numeric dimensions to compare")
print(f"\nFound {len(ref_numeric)} numeric dimensions in reference file")
print(f"Comparing {len(ref_df)} protocols\n")
print("="*80)
@@ -66,12 +100,23 @@ def compare_analyses(reference_file, comparison_files):
print(f"\nComparing: {Path(comp_file).name}")
print("-"*80)
# Read comparison file
comp_df = pd.read_csv(comp_file, quotechar='"', escapechar='\\', engine='python')
# Read and canonicalize comparison file
comp_version = csv_version(comp_file)
if comp_version:
print(f" bicorder version recorded: v{comp_version}")
comp_df, comp_numeric = load_canonical(comp_file)
if 'Descriptor' in comp_df.columns:
before = len(comp_df)
comp_df = comp_df.drop_duplicates(subset='Descriptor', keep='first')
if len(comp_df) < before:
print(f" Deduplicated comparison: {before} → {len(comp_df)} rows (kept first of repeated Descriptor)")
# Convert numeric columns to numeric type, coercing errors to NaN
for col in numeric_cols:
comp_df[col] = pd.to_numeric(comp_df[col], errors='coerce')
# Restrict to columns shared by both files (post-rename): enables
# comparing across bicorder versions when gradients were renamed
numeric_cols = [col for col in ref_numeric if col in comp_numeric]
missing = [col for col in ref_numeric if col not in comp_numeric]
if missing:
print(f" Note: {len(missing)} gradient(s) absent here are excluded: {', '.join(missing)}")
# Ensure same protocols in same order (match by Descriptor)
if 'Descriptor' in ref_df.columns and 'Descriptor' in comp_df.columns:
@@ -155,32 +200,58 @@ def compare_analyses(reference_file, comparison_files):
return results
if __name__ == "__main__":
# Define file paths
reference_file = "data/synthetic_1.2.6/readings_manual.csv"
comparison_files = [
def main(argv=None):
"""CLI entry point.
With no arguments, falls back to the legacy audit: the 1.2.6 manual review
against the three model test runs (as described in README.md).
"""
legacy_reference = "data/synthetic_1.2.6/readings_manual.csv"
legacy_comparisons = [
"data/synthetic_1.2.6/readings_gemma3-12b.csv",
"data/synthetic_1.2.6/readings_gpt-oss.csv",
"data/synthetic_1.2.6/readings_mistral.csv"
"data/synthetic_1.2.6/readings_mistral.csv",
]
if argv is None:
argv = sys.argv[1:]
if argv:
parser = argparse.ArgumentParser(
description='Compare readings CSVs to a reference (Euclidean distance, RMSE, correlation)',
epilog="""Example (cross-version):
python3 scripts/compare_analyses.py \\
data/synthetic_1.4.0/readings.csv data/synthetic_1.2.6/readings.csv
""",
)
parser.add_argument('reference', help='Reference readings CSV')
parser.add_argument('comparisons', nargs='+', help='Comparison readings CSVs')
args = parser.parse_args(argv)
reference_file, comparison_files = args.reference, args.comparisons
else:
reference_file, comparison_files = legacy_reference, legacy_comparisons
# Check if files exist
if not Path(reference_file).exists():
print(f"Error: Reference file '{reference_file}' not found")
exit(1)
sys.exit(1)
existing = [file for file in comparison_files if Path(file).exists()]
for file in comparison_files:
if not Path(file).exists():
print(f"Warning: Comparison file '{file}' not found, skipping...")
comparison_files.remove(file)
if not comparison_files:
if not existing:
print("Error: No comparison files found")
exit(1)
sys.exit(1)
# Run comparison
results = compare_analyses(reference_file, comparison_files)
results = compare_analyses(reference_file, existing)
print("\n" + "="*80)
print("Analysis complete!")
print("="*80)
return results
if __name__ == "__main__":
main()
+19 -37
View File
@@ -3,10 +3,8 @@
Export the cluster classification model to JSON for use in JavaScript.
Reads dimension names directly from bicorder.json so the model always
stays in sync with the current bicorder structure.
When gradients are renamed in bicorder.json, add the old→new mapping to
COLUMN_RENAMES so the training CSV columns are correctly aligned.
stays in sync with the current bicorder structure. Column renames and the
version record live in bicorder_common.py (shared by all analysis scripts).
Usage:
python3 scripts/export_model_for_js.py data/synthetic_1.2.6/readings.csv
@@ -22,37 +20,7 @@ import numpy as np
from sklearn.preprocessing import StandardScaler
from sklearn.discriminant_analysis import LinearDiscriminantAnalysis
# Path to bicorder.json (relative to this script)
BICORDER_JSON = Path(__file__).parent.parent.parent / 'bicorder.json'
# Historical column renames: maps old CSV column names → current bicorder.json names.
# Add an entry here whenever gradient terms are renamed in bicorder.json.
COLUMN_RENAMES = {
'Design_elite_vs_vernacular': 'Design_formal_vs_vernacular',
'Design_institutional_vs_vernacular': 'Design_formal_vs_vernacular',
'Entanglement_exclusive_vs_non-exclusive': 'Entanglement_monopolistic_vs_pluralistic',
'Experience_sufficient_vs_insufficient': 'Experience_sufficient_vs_limited',
'Experience_Kafka_vs_Whitehead': 'Experience_restraining_vs_liberating',
}
def load_bicorder_dimensions(bicorder_path):
"""Read DIMENSIONS and KEY_DIMENSIONS from bicorder.json."""
with open(bicorder_path) as f:
data = json.load(f)
dimensions = []
key_dimensions = []
for category in data['diagnostic']:
set_name = category['set_name']
for gradient in category['gradients']:
dim_name = f"{set_name}_{gradient['term_left']}_vs_{gradient['term_right']}"
dimensions.append(dim_name)
if gradient.get('shortform', False):
key_dimensions.append(dim_name)
return dimensions, key_dimensions, data['version']
from bicorder_common import COLUMN_RENAMES, csv_version, load_bicorder_dimensions
def main():
@@ -74,14 +42,28 @@ Example usage:
analysis_dir = dataset_dir / 'analysis'
# Derive dimensions and version from bicorder.json
DIMENSIONS, KEY_DIMENSIONS, BICORDER_VERSION = load_bicorder_dimensions(BICORDER_JSON)
DIMENSIONS, KEY_DIMENSIONS, BICORDER_VERSION = load_bicorder_dimensions()
print(f"Loaded bicorder.json v{BICORDER_VERSION}")
print(f"Dimensions: {len(DIMENSIONS)}, key dimensions: {len(KEY_DIMENSIONS)}")
# Load data
df = pd.read_csv(args.input_csv)
clusters = pd.read_csv(analysis_dir / 'data' / 'kmeans_clusters.csv')
clusters_path = analysis_dir / 'data' / 'kmeans_clusters.csv'
if not clusters_path.exists():
raise FileNotFoundError(
f"No clustering results for {args.input_csv}: {clusters_path} not found. "
f"Run `python3 scripts/multivariate_analysis.py {args.input_csv}` first "
f"to generate cluster assignments.")
clusters = pd.read_csv(clusters_path)
# Flag a version mismatch between the training data and current bicorder.json
# (column renames keep the data aligned, but the clustering itself was done
# against the recorded version's gradient structure)
recorded_version = csv_version(args.input_csv)
if recorded_version and recorded_version != BICORDER_VERSION:
print(f"Note: training data records bicorder v{recorded_version}; "
f"current bicorder.json is v{BICORDER_VERSION} (columns auto-renamed)")
# Rename old column names to match current bicorder.json
df = df.rename(columns=COLUMN_RENAMES)
+18 -7
View File
@@ -7,7 +7,11 @@
# scripts/sync_readings.sh data/manual_20260320
# scripts/sync_readings.sh data/manual_20260320 --no-analysis
# scripts/sync_readings.sh data/manual_20260320 --min-coverage 0.8
# scripts/sync_readings.sh data/manual_20260320 --training data/synthetic_1.2.6/readings.csv
# scripts/sync_readings.sh data/manual_20260320 --training data/synthetic_1.4.0/readings.csv
#
# By default the classifier training CSV is auto-selected to match the synced
# dataset's recorded bicorder version (see scripts/bicorder_common.py); --training
# overrides that.
#
# .sync_source format:
# REMOTE_URL=https://git.example.org/user/repo
@@ -18,7 +22,7 @@ set -euo pipefail
DATASET_DIR="${1:?Usage: $0 <dataset_dir> [--no-analysis] [--min-coverage N]}"
RUN_ANALYSIS=true
MIN_COVERAGE=0.8
TRAINING_CSV="data/synthetic_1.2.6/readings.csv"
TRAINING_CSV="" # empty → let classify_readings.py auto-match the dataset's bicorder version
shift || true
while [[ $# -gt 0 ]]; do
@@ -96,11 +100,18 @@ if [[ "$RUN_ANALYSIS" == true ]]; then
echo "Generating LDA visualization..."
"$PYTHON" scripts/lda_visualization.py "$DATASET_DIR/readings.csv"
echo ""
echo "Classifying readings (training: $TRAINING_CSV)..."
"$PYTHON" scripts/classify_readings.py \
"$DATASET_DIR/readings.csv" \
--training "$TRAINING_CSV"
if [[ -n "$TRAINING_CSV" ]]; then
echo ""
echo "Classifying readings (training: $TRAINING_CSV)..."
"$PYTHON" scripts/classify_readings.py \
"$DATASET_DIR/readings.csv" \
--training "$TRAINING_CSV"
else
echo ""
echo "Classifying readings (training auto-matched by bicorder version)..."
"$PYTHON" scripts/classify_readings.py \
"$DATASET_DIR/readings.csv"
fi
fi
echo ""
+225
View File
@@ -0,0 +1,225 @@
#!/usr/bin/env python3
"""
Univariate analysis of bicorder readings: per-protocol and per-gradient averages.
Reproduces the ad-hoc averages workflow described in README.md for any
readings CSV, version-agnostic: gradient columns are discovered from the file
itself (canonicalized via bicorder_common.COLUMN_RENAMES), so every run
directory can be analyzed with the same command.
Outputs (default <dataset_dir>/analysis/):
plots/protocol_averages.png — protocol averages, ascending
plots/gradient_averages.png — gradient averages with gaps between the three sets
plots/averages_histogram.png — distribution of protocol averages
data/protocol_averages.csv — per-protocol mean over all gradients
data/gradient_averages.csv — per-gradient mean/median/coverage
reports/univariate_summary.txt — printed summary
With --img, also publishes the three summary PNGs to an image directory
(analysis/img/ by default — the charts the README links from img/).
Usage:
python3 scripts/univariate_analysis.py data/synthetic_1.4.0/readings.csv
python3 scripts/univariate_analysis.py data/synthetic_1.4.0/readings.csv --output data/synthetic_1.4.0/analysis
python3 scripts/univariate_analysis.py data/synthetic_1.4.0/readings.csv --img # also refresh img/
"""
import argparse
import shutil
import sys
from pathlib import Path
import warnings
warnings.filterwarnings('ignore')
import pandas as pd
import numpy as np
import matplotlib.pyplot as plt
from bicorder_common import apply_renames, csv_version, dimension_columns
MIDPOINT = 5 # all gradients run 1 (hard) .. 9 (soft)
def main():
parser = argparse.ArgumentParser(
description='Univariate analysis of Protocol Bicorder readings',
formatter_class=argparse.RawDescriptionHelpFormatter,
epilog="""
Examples:
python3 scripts/univariate_analysis.py data/synthetic_1.4.0/readings.csv
python3 scripts/univariate_analysis.py data/synthetic_1.2.6/readings.csv --output data/synthetic_1.2.6/analysis
""",
)
parser.add_argument('csv_file', help='Readings CSV (e.g. data/synthetic_1.4.0/readings.csv)')
parser.add_argument('--output', '-o', default=None,
help='Output directory (default: <dataset_dir>/analysis)')
parser.add_argument('--img', nargs='?', const='img', default=None,
help='Publish the three summary PNGs to an image directory in addition '
'to the run analysis outputs (default with --img: img/ at the '
"analysis root — the charts the README links from img/)")
args = parser.parse_args()
if not Path(args.csv_file).exists():
print(f"Error: File not found: {args.csv_file}")
sys.exit(1)
dataset_dir = Path(args.csv_file).parent
output_dir = Path(args.output) if args.output else dataset_dir / 'analysis'
(output_dir / 'plots').mkdir(parents=True, exist_ok=True)
(output_dir / 'data').mkdir(parents=True, exist_ok=True)
(output_dir / 'reports').mkdir(parents=True, exist_ok=True)
version = csv_version(args.csv_file)
print("=" * 80)
print("PROTOCOL BICORDER - UNIVARIATE ANALYSIS"
+ (f" (bicorder v{version})" if version else ""))
print(f"Source: {args.csv_file}")
print("=" * 80)
df = pd.read_csv(args.csv_file)
df = apply_renames(df)
# Identify gradient columns, grouped by set in bicorder.json order
all_dims = dimension_columns(df.columns.tolist())
groups = [[c for c in all_dims if c.startswith(prefix)]
for prefix in ('Design_', 'Entanglement_', 'Experience_')]
dimension_cols = [c for g in groups for c in g]
if not dimension_cols:
print("Error: no gradient columns found")
sys.exit(1)
values = df[dimension_cols].apply(pd.to_numeric, errors='coerce')
# --- Per-protocol averages ---
protocol = pd.DataFrame({
'Descriptor': df.get('Descriptor', pd.Series(range(len(df)))),
'average': values.mean(axis=1),
'n_gradients_scored': values.notna().sum(axis=1),
}).dropna(subset=['average']).sort_values('average').reset_index(drop=True)
protocol.to_csv(output_dir / 'data' / 'protocol_averages.csv', index=False)
print(f"\nSaved: {output_dir / 'data' / 'protocol_averages.csv'}")
# --- Per-gradient averages ---
gradient = pd.DataFrame({
'gradient': dimension_cols,
'set': [c.split('_', 1)[0] for c in dimension_cols],
'mean': [values[c].mean() for c in dimension_cols],
'median': [values[c].median() for c in dimension_cols],
'coverage': [values[c].notna().mean() for c in dimension_cols],
}).sort_values('mean', ascending=False)
gradient.to_csv(output_dir / 'data' / 'gradient_averages.csv', index=False)
print(f"Saved: {output_dir / 'data' / 'gradient_averages.csv'}")
# --- Summary statistics on protocol averages ---
avg = protocol['average']
mean, median, std = avg.mean(), avg.median(), avg.std()
pearson_skew = 3 * (mean - median) / std if std else float('nan')
moment_skew = avg.skew() # bias-corrected Fisher moment skewness
lines = []
lines.append(f"Readings analyzed: {len(protocol)} protocols x {len(dimension_cols)} gradients")
if version:
lines.append(f"Bicorder version: {version}")
lines.append("")
lines.append("Protocol averages:")
for label, value in [('mean', mean), ('median', median), ('std', std),
('min', avg.min()), ('max', avg.max())]:
lines.append(f" {label:8s} = {value:.3f}")
lines.append(f" midpoint = {MIDPOINT}")
lines.append(f" deviation from midpoint = {mean - MIDPOINT:+.3f} "
f"(normalized over half-range 4: {(mean - MIDPOINT) / 4:+.3f})")
lines.append(f" Pearson skew (3*(mean-median)/std) = {pearson_skew:+.3f}")
lines.append(f" Fisher moment skew = {moment_skew:+.3f}")
lines.append("")
lines.append("Gradient averages (all sets):")
for _, row in gradient.iterrows():
lines.append(f" {row['gradient'][:60]:60s} mean={row['mean']:.2f} coverage={row['coverage']:.0%}")
lines.append("")
lines.append(f"Extremes: highest = {gradient.iloc[0]['gradient']} ({gradient.iloc[0]['mean']:.2f}); "
f"lowest = {gradient.iloc[-1]['gradient']} ({gradient.iloc[-1]['mean']:.2f})")
summary_text = "\n".join(lines)
report_path = output_dir / 'reports' / 'univariate_summary.txt'
report_path.write_text(summary_text + "\n")
print(f"\nSaved: {report_path}\n")
print(summary_text)
# --- Plots ---
# Protocol averages, ascending (cf. img/protocol_averages.png)
fig, ax = plt.subplots(figsize=(12, 8))
ax.plot(range(len(protocol)), protocol['average'], linewidth=1.2)
ax.axhline(MIDPOINT, color='gray', linestyle='--', linewidth=1, label=f'midpoint ({MIDPOINT})')
ax.set_title('Protocol averages (ascending order)' + (f' — bicorder v{version}' if version else ''))
ax.set_xlabel('Protocol (ranked by average)')
ax.set_ylabel('Average gradient value')
ax.set_ylim(0, 10)
ax.legend()
plt.tight_layout()
path = output_dir / 'plots' / 'protocol_averages.png'
plt.savefig(path, dpi=300, bbox_inches='tight')
plt.close()
print(f"\nSaved: {path}")
# Histogram (cf. img/averages_histogram.png)
fig, ax = plt.subplots(figsize=(10, 6))
ax.hist(avg, bins=40, color='#4C72B0', edgecolor='white')
ax.axvline(MIDPOINT, color='gray', linestyle='--', linewidth=1, label=f'midpoint ({MIDPOINT})')
ax.axvline(mean, color='#C44E52', linewidth=1, label=f'mean ({mean:.2f})')
ax.set_title('Distribution of protocol averages')
ax.set_xlabel('Average gradient value')
ax.set_ylabel('Number of protocols')
ax.legend()
plt.tight_layout()
path = output_dir / 'plots' / 'averages_histogram.png'
plt.savefig(path, dpi=300, bbox_inches='tight')
plt.close()
print(f"Saved: {path}")
# Gradient averages, sorted within each set, with visual gaps between sets
# (cf. img/gradient_averages.png)
palette = {'Design': '#4C72B0', 'Entanglement': '#55A868', 'Experience': '#C44E52'}
ordered_cols = [c for g in groups
for c in sorted(g, key=lambda cv: values[cv].mean(), reverse=True)]
means = [values[c].mean() for c in ordered_cols]
colors = [palette[c.split('_', 1)[0]] for c in ordered_cols]
x_pos, pos = [], 0.0
for gi, group in enumerate(groups):
group_sorted = sorted(group, key=lambda cv: values[cv].mean(), reverse=True)
if gi > 0:
pos += 1.5 # visual gap between gradient sets
x_pos.extend(pos + i for i in range(len(group_sorted)))
pos += len(group_sorted)
fig, ax = plt.subplots(figsize=(14, 6))
ax.bar(x_pos, means, color=colors)
ax.axhline(MIDPOINT, color='gray', linestyle='--', linewidth=1, label=f'midpoint ({MIDPOINT})')
ax.set_title('Gradient averages (gaps separate the three gradient sets)'
+ (f' — bicorder v{version}' if version else ''))
ax.set_xticks(x_pos)
ax.set_xticklabels([c.replace('_vs_', '\nvs\n') for c in ordered_cols], fontsize=7)
ax.set_ylabel('Average value')
ax.set_ylim(0, 10)
ax.legend()
plt.tight_layout()
path = output_dir / 'plots' / 'gradient_averages.png'
plt.savefig(path, dpi=300, bbox_inches='tight')
plt.close()
print(f"Saved: {path}")
# Optionally publish the summary charts for docs (the README links img/…)
if args.img:
img_dir = Path(args.img)
if not img_dir.is_absolute():
img_dir = Path(__file__).resolve().parent.parent / img_dir
img_dir.mkdir(parents=True, exist_ok=True)
published = []
for name in ('protocol_averages.png', 'averages_histogram.png', 'gradient_averages.png'):
shutil.copy2(output_dir / 'plots' / name, img_dir / name)
published.append(str(img_dir / name))
print(f"Published charts → {', '.join(published)}")
print("\nDone.")
if __name__ == '__main__':
main()