Files
protocol-bicorder/analysis/scripts/bicorder_common.py
T
Nathan Schneider 55cbd6cd5d feat: version-agnostic analysis scripts with shared version helpers
- scripts/bicorder_common.py (new): single source of truth for historical
  gradient renames (COLUMN_RENAMES), version detection (bicorder_version col
  → version col → data/<type>_<version>/ dir convention), and training-CSV
  auto-selection (find_training_csv)
- classify_readings.py: auto-select training run by recorded bicorder version
  (excludes the input itself to avoid circular training), canonicalize old
  column names, version-mismatch warnings
- bicorder_classifier.py / export_model_for_js.py: share renames + dimension
  loader; instructive error when clustering results are missing;
  bicorder_version recorded in exported models
- compare_analyses.py: CLI (reference + comparison CSVs), rename
  canonicalization so runs of different versions align on shared gradients,
  Descriptor dedup (was silently skewing merges); legacy no-arg audit intact
- scripts/univariate_analysis.py (new): per-protocol/per-gradient averages,
  distributions, summary stats; --img publishes the three README summary
  charts to img/
- sync_readings.sh: defer classifier training to auto-matching; gitignore
  analysis/.venv and __pycache__
2026-10-02 08:32:59 -06:00

161 lines
6.2 KiB
Python

#!/usr/bin/env python3
"""
Shared bicorder-version helpers used by the analysis scripts.
Single source of truth for:
- historical gradient column renames (old CSV/JSON names → current bicorder.json names)
- reading the bicorder version recorded in a readings.csv
- locating a training CSV matching a target bicorder version
- loading gradient definitions from bicorder.json
Whenever gradients are renamed in ../bicorder.json, update COLUMN_RENAMES
here — scripts that read older readings data import from this module rather
than keeping their own copies of the map.
"""
import csv
import json
import re
from pathlib import Path
# analysis/ directory (this file lives in analysis/scripts/)
_ANALYSIS_DIR = Path(__file__).resolve().parent.parent
# Shared protocol inputs / run directories
DATA_DIR = _ANALYSIS_DIR / 'data'
# Repository-root bicorder.json (the current gradient definitions)
_BICORDER_JSON = _ANALYSIS_DIR.parent / 'bicorder.json'
# Historical column renames: old CSV column names → current bicorder.json names.
# Add an entry here whenever gradient terms are renamed in bicorder.json.
COLUMN_RENAMES = {
'Design_elite_vs_vernacular': 'Design_formal_vs_vernacular',
'Design_institutional_vs_vernacular': 'Design_formal_vs_vernacular',
'Entanglement_exclusive_vs_non-exclusive': 'Entanglement_monopolistic_vs_pluralistic',
'Experience_sufficient_vs_insufficient': 'Experience_sufficient_vs_limited',
'Experience_Kafka_vs_Whitehead': 'Experience_restraining_vs_liberating',
}
# Dimension column prefixes, in bicorder.json set order
DIMENSION_PREFIXES = ('Design_', 'Entanglement_', 'Experience_')
def apply_renames(df_or_columns):
"""Canonicalize old column names to the current bicorder.json terminology.
Works on a pandas DataFrame (returns a renamed copy) or on any iterable of
column names (returns a mapped list).
"""
if hasattr(df_or_columns, 'rename'): # pandas DataFrame
return df_or_columns.rename(columns=COLUMN_RENAMES)
return [COLUMN_RENAMES.get(c, c) for c in df_or_columns]
def dimension_columns(columns):
"""Filter an iterable of column names down to the diagnostic gradient columns."""
return [c for c in columns if c.startswith(DIMENSION_PREFIXES)]
def load_bicorder_dimensions(bicorder_path=None):
"""Read DIMENSIONS, KEY_DIMENSIONS, and version from bicorder.json.
Returns (dimensions, key_dimensions, version) where dimension names follow
the CSV convention `Set_left_vs_right`.
"""
with open(bicorder_path or _BICORDER_JSON) as f:
data = json.load(f)
dimensions = []
key_dimensions = []
for category in data['diagnostic']:
set_name = category['set_name']
for gradient in category['gradients']:
dim_name = f"{set_name}_{gradient['term_left']}_vs_{gradient['term_right']}"
dimensions.append(dim_name)
if gradient.get('shortform', False):
key_dimensions.append(dim_name)
return dimensions, key_dimensions, data.get('version', '')
def version_key(version):
"""Sort key for dotted version strings like '1.4.0'. Unknown formats sort lowest."""
try:
return tuple(int(part) for part in str(version).split('.'))
except ValueError:
return (0,)
# Run-directory naming convention: data/<type>_<version>/ with a dotted version
# (e.g. synthetic_1.4.0). Dates like manual_20260320 don't match (no dots).
_DIR_VERSION = re.compile(r'.*_(?P<version>\d+(?:\.\d+)+)$')
def dir_version(dir_path):
"""Version inferred from a run-directory name (data/<type>_<version> convention)."""
name = Path(dir_path).name
match = _DIR_VERSION.match(name)
return match.group('version') if match else ''
def csv_version(csv_path):
"""Bicorder version associated with a readings.csv, or ''.
Prefers the version recorded in the file (bicorder_version / version column,
via read_csv_version); falls back to the run-directory naming convention
(data/<type>_<version>/) so legacy runs like synthetic_1.2.6 still resolve.
"""
path = Path(csv_path)
return read_csv_version(path) or dir_version(path.parent)
def read_csv_version(csv_path):
"""Return the bicorder version recorded in a readings.csv, or ''.
Prefers the `bicorder_version` provenance column (synthetic runs); falls
back to the per-reading `version` column (manual runs, via json_to_csv).
Uses the most common value so mixed-version datasets still resolve.
"""
with open(csv_path, newline='', encoding='utf-8-sig') as f:
reader = csv.DictReader(f)
if not reader.fieldnames:
return ''
for col in ('bicorder_version', 'version'):
if col in reader.fieldnames:
counts = {}
for row in reader:
value = (row.get(col) or '').strip()
if value:
counts[value] = counts.get(value, 0) + 1
if counts:
return max(counts, key=counts.get)
return ''
def find_training_csv(version=None, data_dir=None, exclude=None):
"""Locate a synthetic readings.csv suitable for classifier training.
Scans data/*/readings.csv (whatever their directories are named — run
directories just need to be self-describing via a version column or a
data/<type>_<version>/ name) and prefers:
1. a file whose version matches `version` (when given), excluding
`exclude` (the input file itself, to avoid circular training when
classifying a synthetic run against itself)
2. failing that, the most recent version recorded in the data directory
Returns a Path, or None if no candidate exists.
"""
data_dir = Path(data_dir) if data_dir else DATA_DIR
candidates = []
for path in sorted(data_dir.glob('*/readings.csv')):
path = path.resolve()
if exclude and path == Path(exclude).resolve():
continue
candidates.append((path, csv_version(path)))
if not candidates:
return None
if version:
matches = [path for path, ver in candidates if ver == version]
if matches:
return matches[0]
best = max(candidates, key=lambda item: version_key(item[1]) if item[1] else (0,))
return best[0]