feat: version-agnostic analysis scripts with shared version helpers
- scripts/bicorder_common.py (new): single source of truth for historical gradient renames (COLUMN_RENAMES), version detection (bicorder_version col → version col → data/<type>_<version>/ dir convention), and training-CSV auto-selection (find_training_csv) - classify_readings.py: auto-select training run by recorded bicorder version (excludes the input itself to avoid circular training), canonicalize old column names, version-mismatch warnings - bicorder_classifier.py / export_model_for_js.py: share renames + dimension loader; instructive error when clustering results are missing; bicorder_version recorded in exported models - compare_analyses.py: CLI (reference + comparison CSVs), rename canonicalization so runs of different versions align on shared gradients, Descriptor dedup (was silently skewing merges); legacy no-arg audit intact - scripts/univariate_analysis.py (new): per-protocol/per-gradient averages, distributions, summary stats; --img publishes the three README summary charts to img/ - sync_readings.sh: defer classifier training to auto-matching; gitignore analysis/.venv and __pycache__
This commit is contained in:
1 parent
a4f16e8e4d
commit
55cbd6cd5d
8 files changed
+591
-109
No files matched your search
@@ -30,34 +30,7 @@ from sklearn.discriminant_analysis import LinearDiscriminantAnalysis
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
# Path to bicorder.json (relative to this script)
|
||||
_BICORDER_JSON = Path(__file__).parent.parent.parent / 'bicorder.json'
|
||||
|
||||
# Historical column renames: maps old CSV column names → current bicorder.json names.
|
||||
# Add an entry here whenever gradient terms are renamed in bicorder.json.
|
||||
_COLUMN_RENAMES = {
|
||||
'Design_elite_vs_vernacular': 'Design_formal_vs_vernacular',
|
||||
'Design_institutional_vs_vernacular': 'Design_formal_vs_vernacular',
|
||||
'Entanglement_exclusive_vs_non-exclusive': 'Entanglement_monopolistic_vs_pluralistic',
|
||||
'Experience_sufficient_vs_insufficient': 'Experience_sufficient_vs_limited',
|
||||
'Experience_Kafka_vs_Whitehead': 'Experience_restraining_vs_liberating',
|
||||
}
|
||||
|
||||
|
||||
def _load_bicorder_dimensions(bicorder_path=_BICORDER_JSON):
|
||||
"""Read DIMENSIONS and KEY_DIMENSIONS from bicorder.json."""
|
||||
with open(bicorder_path) as f:
|
||||
data = json.load(f)
|
||||
dimensions = []
|
||||
key_dimensions = []
|
||||
for category in data['diagnostic']:
|
||||
set_name = category['set_name']
|
||||
for gradient in category['gradients']:
|
||||
dim_name = f"{set_name}_{gradient['term_left']}_vs_{gradient['term_right']}"
|
||||
dimensions.append(dim_name)
|
||||
if gradient.get('shortform', False):
|
||||
key_dimensions.append(dim_name)
|
||||
return dimensions, key_dimensions
|
||||
from bicorder_common import COLUMN_RENAMES as _COLUMN_RENAMES, find_training_csv, load_bicorder_dimensions
|
||||
|
||||
|
||||
class BicorderClassifier:
|
||||
@@ -71,19 +44,32 @@ class BicorderClassifier:
|
||||
2: "Institutional/Bureaucratic"
|
||||
}
|
||||
|
||||
def __init__(self, diagnostic_csv='data/synthetic_1.2.6/readings.csv',
|
||||
def __init__(self, diagnostic_csv=None,
|
||||
model_path=None):
|
||||
"""Initialize classifier with pre-computed model data."""
|
||||
"""Initialize classifier with pre-computed model data.
|
||||
|
||||
If diagnostic_csv is None, the most recent synthetic readings.csv is
|
||||
selected automatically (see bicorder_common.find_training_csv).
|
||||
"""
|
||||
if diagnostic_csv is None:
|
||||
diagnostic_csv = find_training_csv()
|
||||
if diagnostic_csv is None:
|
||||
raise FileNotFoundError(
|
||||
"No training CSV found under data/*/readings.csv — "
|
||||
"pass diagnostic_csv explicitly."
|
||||
)
|
||||
print(f"No training CSV specified; using {diagnostic_csv}")
|
||||
if model_path is None:
|
||||
model_path = str(Path(diagnostic_csv).parent / 'analysis' / 'data')
|
||||
self._diagnostic_csv = diagnostic_csv
|
||||
self._diagnostic_csv = str(diagnostic_csv)
|
||||
self.model_path = Path(model_path)
|
||||
self.scaler = StandardScaler()
|
||||
self.lda = None
|
||||
self.cluster_centroids = None
|
||||
|
||||
# Derive dimension lists from bicorder.json
|
||||
self.DIMENSIONS, self.KEY_DIMENSIONS = _load_bicorder_dimensions()
|
||||
# Derive dimension lists (and version) from bicorder.json
|
||||
self.DIMENSIONS, self.KEY_DIMENSIONS, self.bicorder_version = (
|
||||
load_bicorder_dimensions())
|
||||
|
||||
# Load training data to fit scaler and LDA
|
||||
self._load_model()
|
||||
@@ -92,7 +78,13 @@ class BicorderClassifier:
|
||||
"""Load and fit the classification model from analysis results."""
|
||||
# Load the original data and cluster assignments
|
||||
df = pd.read_csv(self._diagnostic_csv)
|
||||
clusters = pd.read_csv(self.model_path / 'kmeans_clusters.csv')
|
||||
clusters_path = self.model_path / 'kmeans_clusters.csv'
|
||||
if not clusters_path.exists():
|
||||
raise FileNotFoundError(
|
||||
f"No clustering results for training data: {clusters_path} not found. "
|
||||
f"Run `python3 scripts/multivariate_analysis.py {self._diagnostic_csv}` "
|
||||
f"first to generate cluster assignments.")
|
||||
clusters = pd.read_csv(clusters_path)
|
||||
|
||||
# Rename old column names to match current bicorder.json
|
||||
df = df.rename(columns=_COLUMN_RENAMES)
|
||||
@@ -283,6 +275,8 @@ class BicorderClassifier:
|
||||
def save_model(self, output_path='bicorder_classifier_model.json'):
|
||||
"""Save model parameters for use without scikit-learn."""
|
||||
model_data = {
|
||||
'bicorder_version': self.bicorder_version,
|
||||
'trained_on': self._diagnostic_csv,
|
||||
'dimensions': self.DIMENSIONS,
|
||||
'key_dimensions': self.KEY_DIMENSIONS,
|
||||
'cluster_names': self.CLUSTER_NAMES,
|
||||
|
||||
Reference in new issue
Block a user