feat: version-agnostic analysis scripts with shared version helpers
- scripts/bicorder_common.py (new): single source of truth for historical gradient renames (COLUMN_RENAMES), version detection (bicorder_version col → version col → data/<type>_<version>/ dir convention), and training-CSV auto-selection (find_training_csv) - classify_readings.py: auto-select training run by recorded bicorder version (excludes the input itself to avoid circular training), canonicalize old column names, version-mismatch warnings - bicorder_classifier.py / export_model_for_js.py: share renames + dimension loader; instructive error when clustering results are missing; bicorder_version recorded in exported models - compare_analyses.py: CLI (reference + comparison CSVs), rename canonicalization so runs of different versions align on shared gradients, Descriptor dedup (was silently skewing merges); legacy no-arg audit intact - scripts/univariate_analysis.py (new): per-protocol/per-gradient averages, distributions, summary stats; --img publishes the three README summary charts to img/ - sync_readings.sh: defer classifier training to auto-matching; gitignore analysis/.venv and __pycache__
This commit is contained in:
1 parent
a4f16e8e4d
commit
55cbd6cd5d
8 files changed
+586
-104
No files matched your search
@@ -1,3 +1,5 @@
|
||||
tmp
|
||||
analysis/venv
|
||||
analysis/.venv
|
||||
__pycache__/
|
||||
.aider*
|
||||
@@ -30,34 +30,7 @@ from sklearn.discriminant_analysis import LinearDiscriminantAnalysis
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
# Path to bicorder.json (relative to this script)
|
||||
_BICORDER_JSON = Path(__file__).parent.parent.parent / 'bicorder.json'
|
||||
|
||||
# Historical column renames: maps old CSV column names → current bicorder.json names.
|
||||
# Add an entry here whenever gradient terms are renamed in bicorder.json.
|
||||
_COLUMN_RENAMES = {
|
||||
'Design_elite_vs_vernacular': 'Design_formal_vs_vernacular',
|
||||
'Design_institutional_vs_vernacular': 'Design_formal_vs_vernacular',
|
||||
'Entanglement_exclusive_vs_non-exclusive': 'Entanglement_monopolistic_vs_pluralistic',
|
||||
'Experience_sufficient_vs_insufficient': 'Experience_sufficient_vs_limited',
|
||||
'Experience_Kafka_vs_Whitehead': 'Experience_restraining_vs_liberating',
|
||||
}
|
||||
|
||||
|
||||
def _load_bicorder_dimensions(bicorder_path=_BICORDER_JSON):
|
||||
"""Read DIMENSIONS and KEY_DIMENSIONS from bicorder.json."""
|
||||
with open(bicorder_path) as f:
|
||||
data = json.load(f)
|
||||
dimensions = []
|
||||
key_dimensions = []
|
||||
for category in data['diagnostic']:
|
||||
set_name = category['set_name']
|
||||
for gradient in category['gradients']:
|
||||
dim_name = f"{set_name}_{gradient['term_left']}_vs_{gradient['term_right']}"
|
||||
dimensions.append(dim_name)
|
||||
if gradient.get('shortform', False):
|
||||
key_dimensions.append(dim_name)
|
||||
return dimensions, key_dimensions
|
||||
from bicorder_common import COLUMN_RENAMES as _COLUMN_RENAMES, find_training_csv, load_bicorder_dimensions
|
||||
|
||||
|
||||
class BicorderClassifier:
|
||||
@@ -71,19 +44,32 @@ class BicorderClassifier:
|
||||
2: "Institutional/Bureaucratic"
|
||||
}
|
||||
|
||||
def __init__(self, diagnostic_csv='data/synthetic_1.2.6/readings.csv',
|
||||
def __init__(self, diagnostic_csv=None,
|
||||
model_path=None):
|
||||
"""Initialize classifier with pre-computed model data."""
|
||||
"""Initialize classifier with pre-computed model data.
|
||||
|
||||
If diagnostic_csv is None, the most recent synthetic readings.csv is
|
||||
selected automatically (see bicorder_common.find_training_csv).
|
||||
"""
|
||||
if diagnostic_csv is None:
|
||||
diagnostic_csv = find_training_csv()
|
||||
if diagnostic_csv is None:
|
||||
raise FileNotFoundError(
|
||||
"No training CSV found under data/*/readings.csv — "
|
||||
"pass diagnostic_csv explicitly."
|
||||
)
|
||||
print(f"No training CSV specified; using {diagnostic_csv}")
|
||||
if model_path is None:
|
||||
model_path = str(Path(diagnostic_csv).parent / 'analysis' / 'data')
|
||||
self._diagnostic_csv = diagnostic_csv
|
||||
self._diagnostic_csv = str(diagnostic_csv)
|
||||
self.model_path = Path(model_path)
|
||||
self.scaler = StandardScaler()
|
||||
self.lda = None
|
||||
self.cluster_centroids = None
|
||||
|
||||
# Derive dimension lists from bicorder.json
|
||||
self.DIMENSIONS, self.KEY_DIMENSIONS = _load_bicorder_dimensions()
|
||||
# Derive dimension lists (and version) from bicorder.json
|
||||
self.DIMENSIONS, self.KEY_DIMENSIONS, self.bicorder_version = (
|
||||
load_bicorder_dimensions())
|
||||
|
||||
# Load training data to fit scaler and LDA
|
||||
self._load_model()
|
||||
@@ -92,7 +78,13 @@ class BicorderClassifier:
|
||||
"""Load and fit the classification model from analysis results."""
|
||||
# Load the original data and cluster assignments
|
||||
df = pd.read_csv(self._diagnostic_csv)
|
||||
clusters = pd.read_csv(self.model_path / 'kmeans_clusters.csv')
|
||||
clusters_path = self.model_path / 'kmeans_clusters.csv'
|
||||
if not clusters_path.exists():
|
||||
raise FileNotFoundError(
|
||||
f"No clustering results for training data: {clusters_path} not found. "
|
||||
f"Run `python3 scripts/multivariate_analysis.py {self._diagnostic_csv}` "
|
||||
f"first to generate cluster assignments.")
|
||||
clusters = pd.read_csv(clusters_path)
|
||||
|
||||
# Rename old column names to match current bicorder.json
|
||||
df = df.rename(columns=_COLUMN_RENAMES)
|
||||
@@ -283,6 +275,8 @@ class BicorderClassifier:
|
||||
def save_model(self, output_path='bicorder_classifier_model.json'):
|
||||
"""Save model parameters for use without scikit-learn."""
|
||||
model_data = {
|
||||
'bicorder_version': self.bicorder_version,
|
||||
'trained_on': self._diagnostic_csv,
|
||||
'dimensions': self.DIMENSIONS,
|
||||
'key_dimensions': self.KEY_DIMENSIONS,
|
||||
'cluster_names': self.CLUSTER_NAMES,
|
||||
|
||||
@@ -0,0 +1,161 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Shared bicorder-version helpers used by the analysis scripts.
|
||||
|
||||
Single source of truth for:
|
||||
- historical gradient column renames (old CSV/JSON names → current bicorder.json names)
|
||||
- reading the bicorder version recorded in a readings.csv
|
||||
- locating a training CSV matching a target bicorder version
|
||||
- loading gradient definitions from bicorder.json
|
||||
|
||||
Whenever gradients are renamed in ../bicorder.json, update COLUMN_RENAMES
|
||||
here — scripts that read older readings data import from this module rather
|
||||
than keeping their own copies of the map.
|
||||
"""
|
||||
|
||||
import csv
|
||||
import json
|
||||
import re
|
||||
from pathlib import Path
|
||||
|
||||
# analysis/ directory (this file lives in analysis/scripts/)
|
||||
_ANALYSIS_DIR = Path(__file__).resolve().parent.parent
|
||||
|
||||
# Shared protocol inputs / run directories
|
||||
DATA_DIR = _ANALYSIS_DIR / 'data'
|
||||
|
||||
# Repository-root bicorder.json (the current gradient definitions)
|
||||
_BICORDER_JSON = _ANALYSIS_DIR.parent / 'bicorder.json'
|
||||
|
||||
# Historical column renames: old CSV column names → current bicorder.json names.
|
||||
# Add an entry here whenever gradient terms are renamed in bicorder.json.
|
||||
COLUMN_RENAMES = {
|
||||
'Design_elite_vs_vernacular': 'Design_formal_vs_vernacular',
|
||||
'Design_institutional_vs_vernacular': 'Design_formal_vs_vernacular',
|
||||
'Entanglement_exclusive_vs_non-exclusive': 'Entanglement_monopolistic_vs_pluralistic',
|
||||
'Experience_sufficient_vs_insufficient': 'Experience_sufficient_vs_limited',
|
||||
'Experience_Kafka_vs_Whitehead': 'Experience_restraining_vs_liberating',
|
||||
}
|
||||
|
||||
# Dimension column prefixes, in bicorder.json set order
|
||||
DIMENSION_PREFIXES = ('Design_', 'Entanglement_', 'Experience_')
|
||||
|
||||
|
||||
def apply_renames(df_or_columns):
|
||||
"""Canonicalize old column names to the current bicorder.json terminology.
|
||||
|
||||
Works on a pandas DataFrame (returns a renamed copy) or on any iterable of
|
||||
column names (returns a mapped list).
|
||||
"""
|
||||
if hasattr(df_or_columns, 'rename'): # pandas DataFrame
|
||||
return df_or_columns.rename(columns=COLUMN_RENAMES)
|
||||
return [COLUMN_RENAMES.get(c, c) for c in df_or_columns]
|
||||
|
||||
|
||||
def dimension_columns(columns):
|
||||
"""Filter an iterable of column names down to the diagnostic gradient columns."""
|
||||
return [c for c in columns if c.startswith(DIMENSION_PREFIXES)]
|
||||
|
||||
|
||||
def load_bicorder_dimensions(bicorder_path=None):
|
||||
"""Read DIMENSIONS, KEY_DIMENSIONS, and version from bicorder.json.
|
||||
|
||||
Returns (dimensions, key_dimensions, version) where dimension names follow
|
||||
the CSV convention `Set_left_vs_right`.
|
||||
"""
|
||||
with open(bicorder_path or _BICORDER_JSON) as f:
|
||||
data = json.load(f)
|
||||
dimensions = []
|
||||
key_dimensions = []
|
||||
for category in data['diagnostic']:
|
||||
set_name = category['set_name']
|
||||
for gradient in category['gradients']:
|
||||
dim_name = f"{set_name}_{gradient['term_left']}_vs_{gradient['term_right']}"
|
||||
dimensions.append(dim_name)
|
||||
if gradient.get('shortform', False):
|
||||
key_dimensions.append(dim_name)
|
||||
return dimensions, key_dimensions, data.get('version', '')
|
||||
|
||||
|
||||
def version_key(version):
|
||||
"""Sort key for dotted version strings like '1.4.0'. Unknown formats sort lowest."""
|
||||
try:
|
||||
return tuple(int(part) for part in str(version).split('.'))
|
||||
except ValueError:
|
||||
return (0,)
|
||||
|
||||
|
||||
# Run-directory naming convention: data/<type>_<version>/ with a dotted version
|
||||
# (e.g. synthetic_1.4.0). Dates like manual_20260320 don't match (no dots).
|
||||
_DIR_VERSION = re.compile(r'.*_(?P<version>\d+(?:\.\d+)+)$')
|
||||
|
||||
|
||||
def dir_version(dir_path):
|
||||
"""Version inferred from a run-directory name (data/<type>_<version> convention)."""
|
||||
name = Path(dir_path).name
|
||||
match = _DIR_VERSION.match(name)
|
||||
return match.group('version') if match else ''
|
||||
|
||||
|
||||
def csv_version(csv_path):
|
||||
"""Bicorder version associated with a readings.csv, or ''.
|
||||
|
||||
Prefers the version recorded in the file (bicorder_version / version column,
|
||||
via read_csv_version); falls back to the run-directory naming convention
|
||||
(data/<type>_<version>/) so legacy runs like synthetic_1.2.6 still resolve.
|
||||
"""
|
||||
path = Path(csv_path)
|
||||
return read_csv_version(path) or dir_version(path.parent)
|
||||
|
||||
|
||||
def read_csv_version(csv_path):
|
||||
"""Return the bicorder version recorded in a readings.csv, or ''.
|
||||
|
||||
Prefers the `bicorder_version` provenance column (synthetic runs); falls
|
||||
back to the per-reading `version` column (manual runs, via json_to_csv).
|
||||
Uses the most common value so mixed-version datasets still resolve.
|
||||
"""
|
||||
with open(csv_path, newline='', encoding='utf-8-sig') as f:
|
||||
reader = csv.DictReader(f)
|
||||
if not reader.fieldnames:
|
||||
return ''
|
||||
for col in ('bicorder_version', 'version'):
|
||||
if col in reader.fieldnames:
|
||||
counts = {}
|
||||
for row in reader:
|
||||
value = (row.get(col) or '').strip()
|
||||
if value:
|
||||
counts[value] = counts.get(value, 0) + 1
|
||||
if counts:
|
||||
return max(counts, key=counts.get)
|
||||
return ''
|
||||
|
||||
|
||||
def find_training_csv(version=None, data_dir=None, exclude=None):
|
||||
"""Locate a synthetic readings.csv suitable for classifier training.
|
||||
|
||||
Scans data/*/readings.csv (whatever their directories are named — run
|
||||
directories just need to be self-describing via a version column or a
|
||||
data/<type>_<version>/ name) and prefers:
|
||||
1. a file whose version matches `version` (when given), excluding
|
||||
`exclude` (the input file itself, to avoid circular training when
|
||||
classifying a synthetic run against itself)
|
||||
2. failing that, the most recent version recorded in the data directory
|
||||
|
||||
Returns a Path, or None if no candidate exists.
|
||||
"""
|
||||
data_dir = Path(data_dir) if data_dir else DATA_DIR
|
||||
candidates = []
|
||||
for path in sorted(data_dir.glob('*/readings.csv')):
|
||||
path = path.resolve()
|
||||
if exclude and path == Path(exclude).resolve():
|
||||
continue
|
||||
candidates.append((path, csv_version(path)))
|
||||
if not candidates:
|
||||
return None
|
||||
if version:
|
||||
matches = [path for path, ver in candidates if ver == version]
|
||||
if matches:
|
||||
return matches[0]
|
||||
best = max(candidates, key=lambda item: version_key(item[1]) if item[1] else (0,))
|
||||
return best[0]
|
||||
@@ -2,14 +2,18 @@
|
||||
"""
|
||||
Apply the BicorderClassifier to all readings in a CSV and save results.
|
||||
|
||||
Uses the synthetic-trained LDA model by default. Missing dimensions are
|
||||
Training data is selected automatically by version: the input CSV's recorded
|
||||
`bicorder_version` (or `version`) is matched against the runs under data/ so
|
||||
the classifier trains on a same-version synthetic run whenever one exists
|
||||
(see bicorder_common.find_training_csv). Older-format columns are renamed to
|
||||
the current bicorder terminology automatically. Missing dimensions are
|
||||
filled with the neutral value (5), so shortform readings can still be
|
||||
classified — though with lower confidence.
|
||||
|
||||
Usage:
|
||||
python3 scripts/classify_readings.py data/manual_20260320/readings.csv
|
||||
python3 scripts/classify_readings.py data/manual_20260320/readings.csv \\
|
||||
--training data/synthetic_1.2.6/readings.csv \\
|
||||
--training data/synthetic_1.4.0/readings.csv \\
|
||||
--output data/manual_20260320/analysis/classifications.csv
|
||||
"""
|
||||
|
||||
@@ -20,6 +24,7 @@ from pathlib import Path
|
||||
import pandas as pd
|
||||
|
||||
from bicorder_classifier import BicorderClassifier
|
||||
from bicorder_common import csv_version, find_training_csv, apply_renames
|
||||
|
||||
|
||||
def main():
|
||||
@@ -28,9 +33,10 @@ def main():
|
||||
)
|
||||
parser.add_argument('input_csv', help='Readings CSV to classify')
|
||||
parser.add_argument(
|
||||
'--training',
|
||||
default='data/synthetic_1.2.6/readings.csv',
|
||||
help='Training CSV for classifier (default: synthetic_1.2.6)'
|
||||
'--training', default=None,
|
||||
help='Training CSV for classifier (default: auto-selected to match '
|
||||
"the input's recorded bicorder_version; falls back to the most "
|
||||
'recent synthetic run)'
|
||||
)
|
||||
parser.add_argument(
|
||||
'--output', default=None,
|
||||
@@ -39,16 +45,46 @@ def main():
|
||||
args = parser.parse_args()
|
||||
|
||||
input_path = Path(args.input_csv)
|
||||
input_version = csv_version(input_path)
|
||||
|
||||
# Auto-select training data matching the input's recorded bicorder version
|
||||
if args.training:
|
||||
training_path = Path(args.training)
|
||||
else:
|
||||
training_path = find_training_csv(version=input_version, exclude=input_path)
|
||||
if training_path is None:
|
||||
raise SystemExit("No training CSV found under data/*/readings.csv — "
|
||||
"pass --training explicitly.")
|
||||
training_version = csv_version(training_path)
|
||||
if input_version and training_version == input_version:
|
||||
note = f"matching v{input_version}"
|
||||
elif training_version:
|
||||
note = (f"fell back to v{training_version} "
|
||||
f"(no other v{input_version} run under data/ to avoid circular training)")
|
||||
else:
|
||||
note = "(input records no version; most recent run selected)"
|
||||
print(f"Auto-selected training data: {training_path} ({note})")
|
||||
|
||||
# Warn when training data comes from a different bicorder version
|
||||
# (older column names are auto-renamed via bicorder_common)
|
||||
training_version = csv_version(training_path)
|
||||
if input_version and training_version and input_version != training_version:
|
||||
print(f"Warning: training data is bicorder v{training_version} "
|
||||
f"but input is v{input_version}; renamed columns are aligned "
|
||||
f"automatically, but compare versions (especially gradient "
|
||||
f"orderings) when interpreting results.")
|
||||
|
||||
output_path = (
|
||||
Path(args.output) if args.output
|
||||
else input_path.parent / 'analysis' / 'classifications.csv'
|
||||
)
|
||||
output_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
print(f"Loading classifier (training: {args.training})...")
|
||||
classifier = BicorderClassifier(diagnostic_csv=args.training)
|
||||
print(f"Loading classifier (training: {training_path})...")
|
||||
classifier = BicorderClassifier(diagnostic_csv=training_path)
|
||||
|
||||
df = pd.read_csv(input_path)
|
||||
df = apply_renames(df) # canonicalize old gradient column names
|
||||
print(f"Classifying {len(df)} readings from {input_path}...")
|
||||
|
||||
rows = []
|
||||
|
||||
@@ -2,13 +2,42 @@
|
||||
"""
|
||||
Compare multiple analysis CSV files to determine which most closely resembles a reference file.
|
||||
Uses Euclidean distance, correlation, and RMSE metrics.
|
||||
|
||||
Readings files are canonicalized to the current bicorder terminology (history of
|
||||
renames in bicorder_common.py), and comparison runs on the gradient columns
|
||||
shared by all files — so any versions can be compared, and renamed gradients
|
||||
remain comparable across version boundaries.
|
||||
|
||||
Usage:
|
||||
# Legacy audit: manual review vs. the three model test runs (as in README)
|
||||
python3 scripts/compare_analyses.py
|
||||
|
||||
# Explicit: reference file first, then any number of comparison files
|
||||
python3 scripts/compare_analyses.py \
|
||||
data/synthetic_1.2.6/readings.csv data/synthetic_1.4.0/readings.csv
|
||||
"""
|
||||
|
||||
import argparse
|
||||
import sys
|
||||
import pandas as pd
|
||||
import numpy as np
|
||||
from scipy.stats import pearsonr
|
||||
from pathlib import Path
|
||||
|
||||
from bicorder_common import apply_renames, csv_version
|
||||
|
||||
|
||||
def load_canonical(path):
|
||||
"""Load a readings CSV, canonicalize columns, and coerce gradient values to numeric."""
|
||||
df = pd.read_csv(path, quotechar='"', escapechar='\\', engine='python')
|
||||
df = apply_renames(df)
|
||||
numeric_cols = [col for col in df.columns if
|
||||
col.startswith(('Design_', 'Entanglement_', 'Experience_'))]
|
||||
for col in numeric_cols:
|
||||
df[col] = pd.to_numeric(df[col], errors='coerce')
|
||||
return df, numeric_cols
|
||||
|
||||
|
||||
def calculate_euclidean_distance(df1, df2, numeric_cols):
|
||||
"""Calculate Euclidean distance between two dataframes."""
|
||||
distances = []
|
||||
@@ -45,18 +74,23 @@ def calculate_correlation(df1, df2, numeric_cols):
|
||||
def compare_analyses(reference_file, comparison_files):
|
||||
"""Compare multiple analysis files to a reference file."""
|
||||
|
||||
# Read reference file
|
||||
# Read and canonicalize reference file
|
||||
print(f"Reading reference file: {reference_file}")
|
||||
ref_df = pd.read_csv(reference_file, quotechar='"', escapechar='\\', engine='python')
|
||||
# Get numeric columns (all the rating dimensions)
|
||||
numeric_cols = [col for col in ref_df.columns if
|
||||
col.startswith(('Design_', 'Entanglement_', 'Experience_'))]
|
||||
ref_version = csv_version(reference_file)
|
||||
if ref_version:
|
||||
print(f" bicorder version recorded: v{ref_version}")
|
||||
ref_df, ref_numeric = load_canonical(reference_file)
|
||||
|
||||
# Convert numeric columns to numeric type, coercing errors to NaN
|
||||
for col in numeric_cols:
|
||||
ref_df[col] = pd.to_numeric(ref_df[col], errors='coerce')
|
||||
# Repeated Descriptor entries (kept as control cases in the datasets) would
|
||||
# multiply rows in the Descriptor-based merge; keep first occurrence like
|
||||
# the classifier does.
|
||||
if 'Descriptor' in ref_df.columns:
|
||||
before = len(ref_df)
|
||||
ref_df = ref_df.drop_duplicates(subset='Descriptor', keep='first')
|
||||
if len(ref_df) < before:
|
||||
print(f" Deduplicated reference: {before} → {len(ref_df)} rows (kept first of repeated Descriptor)")
|
||||
|
||||
print(f"\nFound {len(numeric_cols)} numeric dimensions to compare")
|
||||
print(f"\nFound {len(ref_numeric)} numeric dimensions in reference file")
|
||||
print(f"Comparing {len(ref_df)} protocols\n")
|
||||
print("="*80)
|
||||
|
||||
@@ -66,12 +100,23 @@ def compare_analyses(reference_file, comparison_files):
|
||||
print(f"\nComparing: {Path(comp_file).name}")
|
||||
print("-"*80)
|
||||
|
||||
# Read comparison file
|
||||
comp_df = pd.read_csv(comp_file, quotechar='"', escapechar='\\', engine='python')
|
||||
# Read and canonicalize comparison file
|
||||
comp_version = csv_version(comp_file)
|
||||
if comp_version:
|
||||
print(f" bicorder version recorded: v{comp_version}")
|
||||
comp_df, comp_numeric = load_canonical(comp_file)
|
||||
if 'Descriptor' in comp_df.columns:
|
||||
before = len(comp_df)
|
||||
comp_df = comp_df.drop_duplicates(subset='Descriptor', keep='first')
|
||||
if len(comp_df) < before:
|
||||
print(f" Deduplicated comparison: {before} → {len(comp_df)} rows (kept first of repeated Descriptor)")
|
||||
|
||||
# Convert numeric columns to numeric type, coercing errors to NaN
|
||||
for col in numeric_cols:
|
||||
comp_df[col] = pd.to_numeric(comp_df[col], errors='coerce')
|
||||
# Restrict to columns shared by both files (post-rename): enables
|
||||
# comparing across bicorder versions when gradients were renamed
|
||||
numeric_cols = [col for col in ref_numeric if col in comp_numeric]
|
||||
missing = [col for col in ref_numeric if col not in comp_numeric]
|
||||
if missing:
|
||||
print(f" Note: {len(missing)} gradient(s) absent here are excluded: {', '.join(missing)}")
|
||||
|
||||
# Ensure same protocols in same order (match by Descriptor)
|
||||
if 'Descriptor' in ref_df.columns and 'Descriptor' in comp_df.columns:
|
||||
@@ -155,32 +200,58 @@ def compare_analyses(reference_file, comparison_files):
|
||||
|
||||
return results
|
||||
|
||||
if __name__ == "__main__":
|
||||
# Define file paths
|
||||
reference_file = "data/synthetic_1.2.6/readings_manual.csv"
|
||||
comparison_files = [
|
||||
def main(argv=None):
|
||||
"""CLI entry point.
|
||||
|
||||
With no arguments, falls back to the legacy audit: the 1.2.6 manual review
|
||||
against the three model test runs (as described in README.md).
|
||||
"""
|
||||
legacy_reference = "data/synthetic_1.2.6/readings_manual.csv"
|
||||
legacy_comparisons = [
|
||||
"data/synthetic_1.2.6/readings_gemma3-12b.csv",
|
||||
"data/synthetic_1.2.6/readings_gpt-oss.csv",
|
||||
"data/synthetic_1.2.6/readings_mistral.csv"
|
||||
"data/synthetic_1.2.6/readings_mistral.csv",
|
||||
]
|
||||
|
||||
if argv is None:
|
||||
argv = sys.argv[1:]
|
||||
if argv:
|
||||
parser = argparse.ArgumentParser(
|
||||
description='Compare readings CSVs to a reference (Euclidean distance, RMSE, correlation)',
|
||||
epilog="""Example (cross-version):
|
||||
python3 scripts/compare_analyses.py \\
|
||||
data/synthetic_1.4.0/readings.csv data/synthetic_1.2.6/readings.csv
|
||||
""",
|
||||
)
|
||||
parser.add_argument('reference', help='Reference readings CSV')
|
||||
parser.add_argument('comparisons', nargs='+', help='Comparison readings CSVs')
|
||||
args = parser.parse_args(argv)
|
||||
reference_file, comparison_files = args.reference, args.comparisons
|
||||
else:
|
||||
reference_file, comparison_files = legacy_reference, legacy_comparisons
|
||||
|
||||
# Check if files exist
|
||||
if not Path(reference_file).exists():
|
||||
print(f"Error: Reference file '{reference_file}' not found")
|
||||
exit(1)
|
||||
sys.exit(1)
|
||||
|
||||
existing = [file for file in comparison_files if Path(file).exists()]
|
||||
for file in comparison_files:
|
||||
if not Path(file).exists():
|
||||
print(f"Warning: Comparison file '{file}' not found, skipping...")
|
||||
comparison_files.remove(file)
|
||||
|
||||
if not comparison_files:
|
||||
if not existing:
|
||||
print("Error: No comparison files found")
|
||||
exit(1)
|
||||
sys.exit(1)
|
||||
|
||||
# Run comparison
|
||||
results = compare_analyses(reference_file, comparison_files)
|
||||
results = compare_analyses(reference_file, existing)
|
||||
|
||||
print("\n" + "="*80)
|
||||
print("Analysis complete!")
|
||||
print("="*80)
|
||||
return results
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -3,10 +3,8 @@
|
||||
Export the cluster classification model to JSON for use in JavaScript.
|
||||
|
||||
Reads dimension names directly from bicorder.json so the model always
|
||||
stays in sync with the current bicorder structure.
|
||||
|
||||
When gradients are renamed in bicorder.json, add the old→new mapping to
|
||||
COLUMN_RENAMES so the training CSV columns are correctly aligned.
|
||||
stays in sync with the current bicorder structure. Column renames and the
|
||||
version record live in bicorder_common.py (shared by all analysis scripts).
|
||||
|
||||
Usage:
|
||||
python3 scripts/export_model_for_js.py data/synthetic_1.2.6/readings.csv
|
||||
@@ -22,37 +20,7 @@ import numpy as np
|
||||
from sklearn.preprocessing import StandardScaler
|
||||
from sklearn.discriminant_analysis import LinearDiscriminantAnalysis
|
||||
|
||||
# Path to bicorder.json (relative to this script)
|
||||
BICORDER_JSON = Path(__file__).parent.parent.parent / 'bicorder.json'
|
||||
|
||||
# Historical column renames: maps old CSV column names → current bicorder.json names.
|
||||
# Add an entry here whenever gradient terms are renamed in bicorder.json.
|
||||
COLUMN_RENAMES = {
|
||||
'Design_elite_vs_vernacular': 'Design_formal_vs_vernacular',
|
||||
'Design_institutional_vs_vernacular': 'Design_formal_vs_vernacular',
|
||||
'Entanglement_exclusive_vs_non-exclusive': 'Entanglement_monopolistic_vs_pluralistic',
|
||||
'Experience_sufficient_vs_insufficient': 'Experience_sufficient_vs_limited',
|
||||
'Experience_Kafka_vs_Whitehead': 'Experience_restraining_vs_liberating',
|
||||
}
|
||||
|
||||
|
||||
def load_bicorder_dimensions(bicorder_path):
|
||||
"""Read DIMENSIONS and KEY_DIMENSIONS from bicorder.json."""
|
||||
with open(bicorder_path) as f:
|
||||
data = json.load(f)
|
||||
|
||||
dimensions = []
|
||||
key_dimensions = []
|
||||
|
||||
for category in data['diagnostic']:
|
||||
set_name = category['set_name']
|
||||
for gradient in category['gradients']:
|
||||
dim_name = f"{set_name}_{gradient['term_left']}_vs_{gradient['term_right']}"
|
||||
dimensions.append(dim_name)
|
||||
if gradient.get('shortform', False):
|
||||
key_dimensions.append(dim_name)
|
||||
|
||||
return dimensions, key_dimensions, data['version']
|
||||
from bicorder_common import COLUMN_RENAMES, csv_version, load_bicorder_dimensions
|
||||
|
||||
|
||||
def main():
|
||||
@@ -74,14 +42,28 @@ Example usage:
|
||||
analysis_dir = dataset_dir / 'analysis'
|
||||
|
||||
# Derive dimensions and version from bicorder.json
|
||||
DIMENSIONS, KEY_DIMENSIONS, BICORDER_VERSION = load_bicorder_dimensions(BICORDER_JSON)
|
||||
DIMENSIONS, KEY_DIMENSIONS, BICORDER_VERSION = load_bicorder_dimensions()
|
||||
|
||||
print(f"Loaded bicorder.json v{BICORDER_VERSION}")
|
||||
print(f"Dimensions: {len(DIMENSIONS)}, key dimensions: {len(KEY_DIMENSIONS)}")
|
||||
|
||||
# Load data
|
||||
df = pd.read_csv(args.input_csv)
|
||||
clusters = pd.read_csv(analysis_dir / 'data' / 'kmeans_clusters.csv')
|
||||
clusters_path = analysis_dir / 'data' / 'kmeans_clusters.csv'
|
||||
if not clusters_path.exists():
|
||||
raise FileNotFoundError(
|
||||
f"No clustering results for {args.input_csv}: {clusters_path} not found. "
|
||||
f"Run `python3 scripts/multivariate_analysis.py {args.input_csv}` first "
|
||||
f"to generate cluster assignments.")
|
||||
clusters = pd.read_csv(clusters_path)
|
||||
|
||||
# Flag a version mismatch between the training data and current bicorder.json
|
||||
# (column renames keep the data aligned, but the clustering itself was done
|
||||
# against the recorded version's gradient structure)
|
||||
recorded_version = csv_version(args.input_csv)
|
||||
if recorded_version and recorded_version != BICORDER_VERSION:
|
||||
print(f"Note: training data records bicorder v{recorded_version}; "
|
||||
f"current bicorder.json is v{BICORDER_VERSION} (columns auto-renamed)")
|
||||
|
||||
# Rename old column names to match current bicorder.json
|
||||
df = df.rename(columns=COLUMN_RENAMES)
|
||||
|
||||
@@ -7,7 +7,11 @@
|
||||
# scripts/sync_readings.sh data/manual_20260320
|
||||
# scripts/sync_readings.sh data/manual_20260320 --no-analysis
|
||||
# scripts/sync_readings.sh data/manual_20260320 --min-coverage 0.8
|
||||
# scripts/sync_readings.sh data/manual_20260320 --training data/synthetic_1.2.6/readings.csv
|
||||
# scripts/sync_readings.sh data/manual_20260320 --training data/synthetic_1.4.0/readings.csv
|
||||
#
|
||||
# By default the classifier training CSV is auto-selected to match the synced
|
||||
# dataset's recorded bicorder version (see scripts/bicorder_common.py); --training
|
||||
# overrides that.
|
||||
#
|
||||
# .sync_source format:
|
||||
# REMOTE_URL=https://git.example.org/user/repo
|
||||
@@ -18,7 +22,7 @@ set -euo pipefail
|
||||
DATASET_DIR="${1:?Usage: $0 <dataset_dir> [--no-analysis] [--min-coverage N]}"
|
||||
RUN_ANALYSIS=true
|
||||
MIN_COVERAGE=0.8
|
||||
TRAINING_CSV="data/synthetic_1.2.6/readings.csv"
|
||||
TRAINING_CSV="" # empty → let classify_readings.py auto-match the dataset's bicorder version
|
||||
|
||||
shift || true
|
||||
while [[ $# -gt 0 ]]; do
|
||||
@@ -96,11 +100,18 @@ if [[ "$RUN_ANALYSIS" == true ]]; then
|
||||
echo "Generating LDA visualization..."
|
||||
"$PYTHON" scripts/lda_visualization.py "$DATASET_DIR/readings.csv"
|
||||
|
||||
if [[ -n "$TRAINING_CSV" ]]; then
|
||||
echo ""
|
||||
echo "Classifying readings (training: $TRAINING_CSV)..."
|
||||
"$PYTHON" scripts/classify_readings.py \
|
||||
"$DATASET_DIR/readings.csv" \
|
||||
--training "$TRAINING_CSV"
|
||||
else
|
||||
echo ""
|
||||
echo "Classifying readings (training auto-matched by bicorder version)..."
|
||||
"$PYTHON" scripts/classify_readings.py \
|
||||
"$DATASET_DIR/readings.csv"
|
||||
fi
|
||||
fi
|
||||
|
||||
echo ""
|
||||
|
||||
@@ -0,0 +1,225 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Univariate analysis of bicorder readings: per-protocol and per-gradient averages.
|
||||
|
||||
Reproduces the ad-hoc averages workflow described in README.md for any
|
||||
readings CSV, version-agnostic: gradient columns are discovered from the file
|
||||
itself (canonicalized via bicorder_common.COLUMN_RENAMES), so every run
|
||||
directory can be analyzed with the same command.
|
||||
|
||||
Outputs (default <dataset_dir>/analysis/):
|
||||
plots/protocol_averages.png — protocol averages, ascending
|
||||
plots/gradient_averages.png — gradient averages with gaps between the three sets
|
||||
plots/averages_histogram.png — distribution of protocol averages
|
||||
data/protocol_averages.csv — per-protocol mean over all gradients
|
||||
data/gradient_averages.csv — per-gradient mean/median/coverage
|
||||
reports/univariate_summary.txt — printed summary
|
||||
|
||||
With --img, also publishes the three summary PNGs to an image directory
|
||||
(analysis/img/ by default — the charts the README links from img/).
|
||||
|
||||
Usage:
|
||||
python3 scripts/univariate_analysis.py data/synthetic_1.4.0/readings.csv
|
||||
python3 scripts/univariate_analysis.py data/synthetic_1.4.0/readings.csv --output data/synthetic_1.4.0/analysis
|
||||
python3 scripts/univariate_analysis.py data/synthetic_1.4.0/readings.csv --img # also refresh img/
|
||||
"""
|
||||
|
||||
import argparse
|
||||
import shutil
|
||||
import sys
|
||||
from pathlib import Path
|
||||
import warnings
|
||||
warnings.filterwarnings('ignore')
|
||||
|
||||
import pandas as pd
|
||||
import numpy as np
|
||||
import matplotlib.pyplot as plt
|
||||
|
||||
from bicorder_common import apply_renames, csv_version, dimension_columns
|
||||
|
||||
MIDPOINT = 5 # all gradients run 1 (hard) .. 9 (soft)
|
||||
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser(
|
||||
description='Univariate analysis of Protocol Bicorder readings',
|
||||
formatter_class=argparse.RawDescriptionHelpFormatter,
|
||||
epilog="""
|
||||
Examples:
|
||||
python3 scripts/univariate_analysis.py data/synthetic_1.4.0/readings.csv
|
||||
python3 scripts/univariate_analysis.py data/synthetic_1.2.6/readings.csv --output data/synthetic_1.2.6/analysis
|
||||
""",
|
||||
)
|
||||
parser.add_argument('csv_file', help='Readings CSV (e.g. data/synthetic_1.4.0/readings.csv)')
|
||||
parser.add_argument('--output', '-o', default=None,
|
||||
help='Output directory (default: <dataset_dir>/analysis)')
|
||||
parser.add_argument('--img', nargs='?', const='img', default=None,
|
||||
help='Publish the three summary PNGs to an image directory in addition '
|
||||
'to the run analysis outputs (default with --img: img/ at the '
|
||||
"analysis root — the charts the README links from img/)")
|
||||
args = parser.parse_args()
|
||||
|
||||
if not Path(args.csv_file).exists():
|
||||
print(f"Error: File not found: {args.csv_file}")
|
||||
sys.exit(1)
|
||||
|
||||
dataset_dir = Path(args.csv_file).parent
|
||||
output_dir = Path(args.output) if args.output else dataset_dir / 'analysis'
|
||||
(output_dir / 'plots').mkdir(parents=True, exist_ok=True)
|
||||
(output_dir / 'data').mkdir(parents=True, exist_ok=True)
|
||||
(output_dir / 'reports').mkdir(parents=True, exist_ok=True)
|
||||
|
||||
version = csv_version(args.csv_file)
|
||||
print("=" * 80)
|
||||
print("PROTOCOL BICORDER - UNIVARIATE ANALYSIS"
|
||||
+ (f" (bicorder v{version})" if version else ""))
|
||||
print(f"Source: {args.csv_file}")
|
||||
print("=" * 80)
|
||||
|
||||
df = pd.read_csv(args.csv_file)
|
||||
df = apply_renames(df)
|
||||
|
||||
# Identify gradient columns, grouped by set in bicorder.json order
|
||||
all_dims = dimension_columns(df.columns.tolist())
|
||||
groups = [[c for c in all_dims if c.startswith(prefix)]
|
||||
for prefix in ('Design_', 'Entanglement_', 'Experience_')]
|
||||
dimension_cols = [c for g in groups for c in g]
|
||||
|
||||
if not dimension_cols:
|
||||
print("Error: no gradient columns found")
|
||||
sys.exit(1)
|
||||
|
||||
values = df[dimension_cols].apply(pd.to_numeric, errors='coerce')
|
||||
|
||||
# --- Per-protocol averages ---
|
||||
protocol = pd.DataFrame({
|
||||
'Descriptor': df.get('Descriptor', pd.Series(range(len(df)))),
|
||||
'average': values.mean(axis=1),
|
||||
'n_gradients_scored': values.notna().sum(axis=1),
|
||||
}).dropna(subset=['average']).sort_values('average').reset_index(drop=True)
|
||||
protocol.to_csv(output_dir / 'data' / 'protocol_averages.csv', index=False)
|
||||
print(f"\nSaved: {output_dir / 'data' / 'protocol_averages.csv'}")
|
||||
|
||||
# --- Per-gradient averages ---
|
||||
gradient = pd.DataFrame({
|
||||
'gradient': dimension_cols,
|
||||
'set': [c.split('_', 1)[0] for c in dimension_cols],
|
||||
'mean': [values[c].mean() for c in dimension_cols],
|
||||
'median': [values[c].median() for c in dimension_cols],
|
||||
'coverage': [values[c].notna().mean() for c in dimension_cols],
|
||||
}).sort_values('mean', ascending=False)
|
||||
gradient.to_csv(output_dir / 'data' / 'gradient_averages.csv', index=False)
|
||||
print(f"Saved: {output_dir / 'data' / 'gradient_averages.csv'}")
|
||||
|
||||
# --- Summary statistics on protocol averages ---
|
||||
avg = protocol['average']
|
||||
mean, median, std = avg.mean(), avg.median(), avg.std()
|
||||
pearson_skew = 3 * (mean - median) / std if std else float('nan')
|
||||
moment_skew = avg.skew() # bias-corrected Fisher moment skewness
|
||||
|
||||
lines = []
|
||||
lines.append(f"Readings analyzed: {len(protocol)} protocols x {len(dimension_cols)} gradients")
|
||||
if version:
|
||||
lines.append(f"Bicorder version: {version}")
|
||||
lines.append("")
|
||||
lines.append("Protocol averages:")
|
||||
for label, value in [('mean', mean), ('median', median), ('std', std),
|
||||
('min', avg.min()), ('max', avg.max())]:
|
||||
lines.append(f" {label:8s} = {value:.3f}")
|
||||
lines.append(f" midpoint = {MIDPOINT}")
|
||||
lines.append(f" deviation from midpoint = {mean - MIDPOINT:+.3f} "
|
||||
f"(normalized over half-range 4: {(mean - MIDPOINT) / 4:+.3f})")
|
||||
lines.append(f" Pearson skew (3*(mean-median)/std) = {pearson_skew:+.3f}")
|
||||
lines.append(f" Fisher moment skew = {moment_skew:+.3f}")
|
||||
lines.append("")
|
||||
lines.append("Gradient averages (all sets):")
|
||||
for _, row in gradient.iterrows():
|
||||
lines.append(f" {row['gradient'][:60]:60s} mean={row['mean']:.2f} coverage={row['coverage']:.0%}")
|
||||
lines.append("")
|
||||
lines.append(f"Extremes: highest = {gradient.iloc[0]['gradient']} ({gradient.iloc[0]['mean']:.2f}); "
|
||||
f"lowest = {gradient.iloc[-1]['gradient']} ({gradient.iloc[-1]['mean']:.2f})")
|
||||
|
||||
summary_text = "\n".join(lines)
|
||||
report_path = output_dir / 'reports' / 'univariate_summary.txt'
|
||||
report_path.write_text(summary_text + "\n")
|
||||
print(f"\nSaved: {report_path}\n")
|
||||
print(summary_text)
|
||||
|
||||
# --- Plots ---
|
||||
# Protocol averages, ascending (cf. img/protocol_averages.png)
|
||||
fig, ax = plt.subplots(figsize=(12, 8))
|
||||
ax.plot(range(len(protocol)), protocol['average'], linewidth=1.2)
|
||||
ax.axhline(MIDPOINT, color='gray', linestyle='--', linewidth=1, label=f'midpoint ({MIDPOINT})')
|
||||
ax.set_title('Protocol averages (ascending order)' + (f' — bicorder v{version}' if version else ''))
|
||||
ax.set_xlabel('Protocol (ranked by average)')
|
||||
ax.set_ylabel('Average gradient value')
|
||||
ax.set_ylim(0, 10)
|
||||
ax.legend()
|
||||
plt.tight_layout()
|
||||
path = output_dir / 'plots' / 'protocol_averages.png'
|
||||
plt.savefig(path, dpi=300, bbox_inches='tight')
|
||||
plt.close()
|
||||
print(f"\nSaved: {path}")
|
||||
|
||||
# Histogram (cf. img/averages_histogram.png)
|
||||
fig, ax = plt.subplots(figsize=(10, 6))
|
||||
ax.hist(avg, bins=40, color='#4C72B0', edgecolor='white')
|
||||
ax.axvline(MIDPOINT, color='gray', linestyle='--', linewidth=1, label=f'midpoint ({MIDPOINT})')
|
||||
ax.axvline(mean, color='#C44E52', linewidth=1, label=f'mean ({mean:.2f})')
|
||||
ax.set_title('Distribution of protocol averages')
|
||||
ax.set_xlabel('Average gradient value')
|
||||
ax.set_ylabel('Number of protocols')
|
||||
ax.legend()
|
||||
plt.tight_layout()
|
||||
path = output_dir / 'plots' / 'averages_histogram.png'
|
||||
plt.savefig(path, dpi=300, bbox_inches='tight')
|
||||
plt.close()
|
||||
print(f"Saved: {path}")
|
||||
|
||||
# Gradient averages, sorted within each set, with visual gaps between sets
|
||||
# (cf. img/gradient_averages.png)
|
||||
palette = {'Design': '#4C72B0', 'Entanglement': '#55A868', 'Experience': '#C44E52'}
|
||||
ordered_cols = [c for g in groups
|
||||
for c in sorted(g, key=lambda cv: values[cv].mean(), reverse=True)]
|
||||
means = [values[c].mean() for c in ordered_cols]
|
||||
colors = [palette[c.split('_', 1)[0]] for c in ordered_cols]
|
||||
x_pos, pos = [], 0.0
|
||||
for gi, group in enumerate(groups):
|
||||
group_sorted = sorted(group, key=lambda cv: values[cv].mean(), reverse=True)
|
||||
if gi > 0:
|
||||
pos += 1.5 # visual gap between gradient sets
|
||||
x_pos.extend(pos + i for i in range(len(group_sorted)))
|
||||
pos += len(group_sorted)
|
||||
fig, ax = plt.subplots(figsize=(14, 6))
|
||||
ax.bar(x_pos, means, color=colors)
|
||||
ax.axhline(MIDPOINT, color='gray', linestyle='--', linewidth=1, label=f'midpoint ({MIDPOINT})')
|
||||
ax.set_title('Gradient averages (gaps separate the three gradient sets)'
|
||||
+ (f' — bicorder v{version}' if version else ''))
|
||||
ax.set_xticks(x_pos)
|
||||
ax.set_xticklabels([c.replace('_vs_', '\nvs\n') for c in ordered_cols], fontsize=7)
|
||||
ax.set_ylabel('Average value')
|
||||
ax.set_ylim(0, 10)
|
||||
ax.legend()
|
||||
plt.tight_layout()
|
||||
path = output_dir / 'plots' / 'gradient_averages.png'
|
||||
plt.savefig(path, dpi=300, bbox_inches='tight')
|
||||
plt.close()
|
||||
print(f"Saved: {path}")
|
||||
|
||||
# Optionally publish the summary charts for docs (the README links img/…)
|
||||
if args.img:
|
||||
img_dir = Path(args.img)
|
||||
if not img_dir.is_absolute():
|
||||
img_dir = Path(__file__).resolve().parent.parent / img_dir
|
||||
img_dir.mkdir(parents=True, exist_ok=True)
|
||||
published = []
|
||||
for name in ('protocol_averages.png', 'averages_histogram.png', 'gradient_averages.png'):
|
||||
shutil.copy2(output_dir / 'plots' / name, img_dir / name)
|
||||
published.append(str(img_dir / name))
|
||||
print(f"Published charts → {', '.join(published)}")
|
||||
|
||||
print("\nDone.")
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
Reference in new issue
Block a user