feat: version-agnostic analysis scripts with shared version helpers
- scripts/bicorder_common.py (new): single source of truth for historical gradient renames (COLUMN_RENAMES), version detection (bicorder_version col → version col → data/<type>_<version>/ dir convention), and training-CSV auto-selection (find_training_csv) - classify_readings.py: auto-select training run by recorded bicorder version (excludes the input itself to avoid circular training), canonicalize old column names, version-mismatch warnings - bicorder_classifier.py / export_model_for_js.py: share renames + dimension loader; instructive error when clustering results are missing; bicorder_version recorded in exported models - compare_analyses.py: CLI (reference + comparison CSVs), rename canonicalization so runs of different versions align on shared gradients, Descriptor dedup (was silently skewing merges); legacy no-arg audit intact - scripts/univariate_analysis.py (new): per-protocol/per-gradient averages, distributions, summary stats; --img publishes the three README summary charts to img/ - sync_readings.sh: defer classifier training to auto-matching; gitignore analysis/.venv and __pycache__
This commit is contained in:
1 parent
a4f16e8e4d
commit
55cbd6cd5d
8 files changed
+591
-109
No files matched your search
@@ -0,0 +1,161 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Shared bicorder-version helpers used by the analysis scripts.
|
||||
|
||||
Single source of truth for:
|
||||
- historical gradient column renames (old CSV/JSON names → current bicorder.json names)
|
||||
- reading the bicorder version recorded in a readings.csv
|
||||
- locating a training CSV matching a target bicorder version
|
||||
- loading gradient definitions from bicorder.json
|
||||
|
||||
Whenever gradients are renamed in ../bicorder.json, update COLUMN_RENAMES
|
||||
here — scripts that read older readings data import from this module rather
|
||||
than keeping their own copies of the map.
|
||||
"""
|
||||
|
||||
import csv
|
||||
import json
|
||||
import re
|
||||
from pathlib import Path
|
||||
|
||||
# analysis/ directory (this file lives in analysis/scripts/)
|
||||
_ANALYSIS_DIR = Path(__file__).resolve().parent.parent
|
||||
|
||||
# Shared protocol inputs / run directories
|
||||
DATA_DIR = _ANALYSIS_DIR / 'data'
|
||||
|
||||
# Repository-root bicorder.json (the current gradient definitions)
|
||||
_BICORDER_JSON = _ANALYSIS_DIR.parent / 'bicorder.json'
|
||||
|
||||
# Historical column renames: old CSV column names → current bicorder.json names.
|
||||
# Add an entry here whenever gradient terms are renamed in bicorder.json.
|
||||
COLUMN_RENAMES = {
|
||||
'Design_elite_vs_vernacular': 'Design_formal_vs_vernacular',
|
||||
'Design_institutional_vs_vernacular': 'Design_formal_vs_vernacular',
|
||||
'Entanglement_exclusive_vs_non-exclusive': 'Entanglement_monopolistic_vs_pluralistic',
|
||||
'Experience_sufficient_vs_insufficient': 'Experience_sufficient_vs_limited',
|
||||
'Experience_Kafka_vs_Whitehead': 'Experience_restraining_vs_liberating',
|
||||
}
|
||||
|
||||
# Dimension column prefixes, in bicorder.json set order
|
||||
DIMENSION_PREFIXES = ('Design_', 'Entanglement_', 'Experience_')
|
||||
|
||||
|
||||
def apply_renames(df_or_columns):
|
||||
"""Canonicalize old column names to the current bicorder.json terminology.
|
||||
|
||||
Works on a pandas DataFrame (returns a renamed copy) or on any iterable of
|
||||
column names (returns a mapped list).
|
||||
"""
|
||||
if hasattr(df_or_columns, 'rename'): # pandas DataFrame
|
||||
return df_or_columns.rename(columns=COLUMN_RENAMES)
|
||||
return [COLUMN_RENAMES.get(c, c) for c in df_or_columns]
|
||||
|
||||
|
||||
def dimension_columns(columns):
|
||||
"""Filter an iterable of column names down to the diagnostic gradient columns."""
|
||||
return [c for c in columns if c.startswith(DIMENSION_PREFIXES)]
|
||||
|
||||
|
||||
def load_bicorder_dimensions(bicorder_path=None):
|
||||
"""Read DIMENSIONS, KEY_DIMENSIONS, and version from bicorder.json.
|
||||
|
||||
Returns (dimensions, key_dimensions, version) where dimension names follow
|
||||
the CSV convention `Set_left_vs_right`.
|
||||
"""
|
||||
with open(bicorder_path or _BICORDER_JSON) as f:
|
||||
data = json.load(f)
|
||||
dimensions = []
|
||||
key_dimensions = []
|
||||
for category in data['diagnostic']:
|
||||
set_name = category['set_name']
|
||||
for gradient in category['gradients']:
|
||||
dim_name = f"{set_name}_{gradient['term_left']}_vs_{gradient['term_right']}"
|
||||
dimensions.append(dim_name)
|
||||
if gradient.get('shortform', False):
|
||||
key_dimensions.append(dim_name)
|
||||
return dimensions, key_dimensions, data.get('version', '')
|
||||
|
||||
|
||||
def version_key(version):
|
||||
"""Sort key for dotted version strings like '1.4.0'. Unknown formats sort lowest."""
|
||||
try:
|
||||
return tuple(int(part) for part in str(version).split('.'))
|
||||
except ValueError:
|
||||
return (0,)
|
||||
|
||||
|
||||
# Run-directory naming convention: data/<type>_<version>/ with a dotted version
|
||||
# (e.g. synthetic_1.4.0). Dates like manual_20260320 don't match (no dots).
|
||||
_DIR_VERSION = re.compile(r'.*_(?P<version>\d+(?:\.\d+)+)$')
|
||||
|
||||
|
||||
def dir_version(dir_path):
|
||||
"""Version inferred from a run-directory name (data/<type>_<version> convention)."""
|
||||
name = Path(dir_path).name
|
||||
match = _DIR_VERSION.match(name)
|
||||
return match.group('version') if match else ''
|
||||
|
||||
|
||||
def csv_version(csv_path):
|
||||
"""Bicorder version associated with a readings.csv, or ''.
|
||||
|
||||
Prefers the version recorded in the file (bicorder_version / version column,
|
||||
via read_csv_version); falls back to the run-directory naming convention
|
||||
(data/<type>_<version>/) so legacy runs like synthetic_1.2.6 still resolve.
|
||||
"""
|
||||
path = Path(csv_path)
|
||||
return read_csv_version(path) or dir_version(path.parent)
|
||||
|
||||
|
||||
def read_csv_version(csv_path):
|
||||
"""Return the bicorder version recorded in a readings.csv, or ''.
|
||||
|
||||
Prefers the `bicorder_version` provenance column (synthetic runs); falls
|
||||
back to the per-reading `version` column (manual runs, via json_to_csv).
|
||||
Uses the most common value so mixed-version datasets still resolve.
|
||||
"""
|
||||
with open(csv_path, newline='', encoding='utf-8-sig') as f:
|
||||
reader = csv.DictReader(f)
|
||||
if not reader.fieldnames:
|
||||
return ''
|
||||
for col in ('bicorder_version', 'version'):
|
||||
if col in reader.fieldnames:
|
||||
counts = {}
|
||||
for row in reader:
|
||||
value = (row.get(col) or '').strip()
|
||||
if value:
|
||||
counts[value] = counts.get(value, 0) + 1
|
||||
if counts:
|
||||
return max(counts, key=counts.get)
|
||||
return ''
|
||||
|
||||
|
||||
def find_training_csv(version=None, data_dir=None, exclude=None):
|
||||
"""Locate a synthetic readings.csv suitable for classifier training.
|
||||
|
||||
Scans data/*/readings.csv (whatever their directories are named — run
|
||||
directories just need to be self-describing via a version column or a
|
||||
data/<type>_<version>/ name) and prefers:
|
||||
1. a file whose version matches `version` (when given), excluding
|
||||
`exclude` (the input file itself, to avoid circular training when
|
||||
classifying a synthetic run against itself)
|
||||
2. failing that, the most recent version recorded in the data directory
|
||||
|
||||
Returns a Path, or None if no candidate exists.
|
||||
"""
|
||||
data_dir = Path(data_dir) if data_dir else DATA_DIR
|
||||
candidates = []
|
||||
for path in sorted(data_dir.glob('*/readings.csv')):
|
||||
path = path.resolve()
|
||||
if exclude and path == Path(exclude).resolve():
|
||||
continue
|
||||
candidates.append((path, csv_version(path)))
|
||||
if not candidates:
|
||||
return None
|
||||
if version:
|
||||
matches = [path for path, ver in candidates if ver == version]
|
||||
if matches:
|
||||
return matches[0]
|
||||
best = max(candidates, key=lambda item: version_key(item[1]) if item[1] else (0,))
|
||||
return best[0]
|
||||
Reference in new issue
Block a user