#!/usr/bin/env python3 """ Shared bicorder-version helpers used by the analysis scripts. Single source of truth for: - historical gradient column renames (old CSV/JSON names → current bicorder.json names) - reading the bicorder version recorded in a readings.csv - locating a training CSV matching a target bicorder version - loading gradient definitions from bicorder.json Whenever gradients are renamed in ../bicorder.json, update COLUMN_RENAMES here — scripts that read older readings data import from this module rather than keeping their own copies of the map. """ import csv import json import re from pathlib import Path # analysis/ directory (this file lives in analysis/scripts/) _ANALYSIS_DIR = Path(__file__).resolve().parent.parent # Shared protocol inputs / run directories DATA_DIR = _ANALYSIS_DIR / 'data' # Repository-root bicorder.json (the current gradient definitions) _BICORDER_JSON = _ANALYSIS_DIR.parent / 'bicorder.json' # Historical column renames: old CSV column names → current bicorder.json names. # Add an entry here whenever gradient terms are renamed in bicorder.json. COLUMN_RENAMES = { 'Design_elite_vs_vernacular': 'Design_formal_vs_vernacular', 'Design_institutional_vs_vernacular': 'Design_formal_vs_vernacular', 'Entanglement_exclusive_vs_non-exclusive': 'Entanglement_monopolistic_vs_pluralistic', 'Experience_sufficient_vs_insufficient': 'Experience_sufficient_vs_limited', 'Experience_Kafka_vs_Whitehead': 'Experience_restraining_vs_liberating', } # Dimension column prefixes, in bicorder.json set order DIMENSION_PREFIXES = ('Design_', 'Entanglement_', 'Experience_') def apply_renames(df_or_columns): """Canonicalize old column names to the current bicorder.json terminology. Works on a pandas DataFrame (returns a renamed copy) or on any iterable of column names (returns a mapped list). """ if hasattr(df_or_columns, 'rename'): # pandas DataFrame return df_or_columns.rename(columns=COLUMN_RENAMES) return [COLUMN_RENAMES.get(c, c) for c in df_or_columns] def dimension_columns(columns): """Filter an iterable of column names down to the diagnostic gradient columns.""" return [c for c in columns if c.startswith(DIMENSION_PREFIXES)] def load_bicorder_dimensions(bicorder_path=None): """Read DIMENSIONS, KEY_DIMENSIONS, and version from bicorder.json. Returns (dimensions, key_dimensions, version) where dimension names follow the CSV convention `Set_left_vs_right`. """ with open(bicorder_path or _BICORDER_JSON) as f: data = json.load(f) dimensions = [] key_dimensions = [] for category in data['diagnostic']: set_name = category['set_name'] for gradient in category['gradients']: dim_name = f"{set_name}_{gradient['term_left']}_vs_{gradient['term_right']}" dimensions.append(dim_name) if gradient.get('shortform', False): key_dimensions.append(dim_name) return dimensions, key_dimensions, data.get('version', '') def version_key(version): """Sort key for dotted version strings like '1.4.0'. Unknown formats sort lowest.""" try: return tuple(int(part) for part in str(version).split('.')) except ValueError: return (0,) # Run-directory naming convention: data/_/ with a dotted version # (e.g. synthetic_1.4.0). Dates like manual_20260320 don't match (no dots). _DIR_VERSION = re.compile(r'.*_(?P\d+(?:\.\d+)+)$') def dir_version(dir_path): """Version inferred from a run-directory name (data/_ convention).""" name = Path(dir_path).name match = _DIR_VERSION.match(name) return match.group('version') if match else '' def csv_version(csv_path): """Bicorder version associated with a readings.csv, or ''. Prefers the version recorded in the file (bicorder_version / version column, via read_csv_version); falls back to the run-directory naming convention (data/_/) so legacy runs like synthetic_1.2.6 still resolve. """ path = Path(csv_path) return read_csv_version(path) or dir_version(path.parent) def read_csv_version(csv_path): """Return the bicorder version recorded in a readings.csv, or ''. Prefers the `bicorder_version` provenance column (synthetic runs); falls back to the per-reading `version` column (manual runs, via json_to_csv). Uses the most common value so mixed-version datasets still resolve. """ with open(csv_path, newline='', encoding='utf-8-sig') as f: reader = csv.DictReader(f) if not reader.fieldnames: return '' for col in ('bicorder_version', 'version'): if col in reader.fieldnames: counts = {} for row in reader: value = (row.get(col) or '').strip() if value: counts[value] = counts.get(value, 0) + 1 if counts: return max(counts, key=counts.get) return '' def find_training_csv(version=None, data_dir=None, exclude=None): """Locate a synthetic readings.csv suitable for classifier training. Scans data/*/readings.csv (whatever their directories are named — run directories just need to be self-describing via a version column or a data/_/ name) and prefers: 1. a file whose version matches `version` (when given), excluding `exclude` (the input file itself, to avoid circular training when classifying a synthetic run against itself) 2. failing that, the most recent version recorded in the data directory Returns a Path, or None if no candidate exists. """ data_dir = Path(data_dir) if data_dir else DATA_DIR candidates = [] for path in sorted(data_dir.glob('*/readings.csv')): path = path.resolve() if exclude and path == Path(exclude).resolve(): continue candidates.append((path, csv_version(path))) if not candidates: return None if version: matches = [path for path, ver in candidates if ver == version] if matches: return matches[0] best = max(candidates, key=lambda item: version_key(item[1]) if item[1] else (0,)) return best[0]