feat: version-agnostic analysis scripts with shared version helpers
- scripts/bicorder_common.py (new): single source of truth for historical gradient renames (COLUMN_RENAMES), version detection (bicorder_version col → version col → data/<type>_<version>/ dir convention), and training-CSV auto-selection (find_training_csv) - classify_readings.py: auto-select training run by recorded bicorder version (excludes the input itself to avoid circular training), canonicalize old column names, version-mismatch warnings - bicorder_classifier.py / export_model_for_js.py: share renames + dimension loader; instructive error when clustering results are missing; bicorder_version recorded in exported models - compare_analyses.py: CLI (reference + comparison CSVs), rename canonicalization so runs of different versions align on shared gradients, Descriptor dedup (was silently skewing merges); legacy no-arg audit intact - scripts/univariate_analysis.py (new): per-protocol/per-gradient averages, distributions, summary stats; --img publishes the three README summary charts to img/ - sync_readings.sh: defer classifier training to auto-matching; gitignore analysis/.venv and __pycache__
This commit is contained in:
1 parent
a4f16e8e4d
commit
55cbd6cd5d
8 files changed
+591
-109
No files matched your search
@@ -7,7 +7,11 @@
|
||||
# scripts/sync_readings.sh data/manual_20260320
|
||||
# scripts/sync_readings.sh data/manual_20260320 --no-analysis
|
||||
# scripts/sync_readings.sh data/manual_20260320 --min-coverage 0.8
|
||||
# scripts/sync_readings.sh data/manual_20260320 --training data/synthetic_1.2.6/readings.csv
|
||||
# scripts/sync_readings.sh data/manual_20260320 --training data/synthetic_1.4.0/readings.csv
|
||||
#
|
||||
# By default the classifier training CSV is auto-selected to match the synced
|
||||
# dataset's recorded bicorder version (see scripts/bicorder_common.py); --training
|
||||
# overrides that.
|
||||
#
|
||||
# .sync_source format:
|
||||
# REMOTE_URL=https://git.example.org/user/repo
|
||||
@@ -18,7 +22,7 @@ set -euo pipefail
|
||||
DATASET_DIR="${1:?Usage: $0 <dataset_dir> [--no-analysis] [--min-coverage N]}"
|
||||
RUN_ANALYSIS=true
|
||||
MIN_COVERAGE=0.8
|
||||
TRAINING_CSV="data/synthetic_1.2.6/readings.csv"
|
||||
TRAINING_CSV="" # empty → let classify_readings.py auto-match the dataset's bicorder version
|
||||
|
||||
shift || true
|
||||
while [[ $# -gt 0 ]]; do
|
||||
@@ -96,11 +100,18 @@ if [[ "$RUN_ANALYSIS" == true ]]; then
|
||||
echo "Generating LDA visualization..."
|
||||
"$PYTHON" scripts/lda_visualization.py "$DATASET_DIR/readings.csv"
|
||||
|
||||
echo ""
|
||||
echo "Classifying readings (training: $TRAINING_CSV)..."
|
||||
"$PYTHON" scripts/classify_readings.py \
|
||||
"$DATASET_DIR/readings.csv" \
|
||||
--training "$TRAINING_CSV"
|
||||
if [[ -n "$TRAINING_CSV" ]]; then
|
||||
echo ""
|
||||
echo "Classifying readings (training: $TRAINING_CSV)..."
|
||||
"$PYTHON" scripts/classify_readings.py \
|
||||
"$DATASET_DIR/readings.csv" \
|
||||
--training "$TRAINING_CSV"
|
||||
else
|
||||
echo ""
|
||||
echo "Classifying readings (training auto-matched by bicorder version)..."
|
||||
"$PYTHON" scripts/classify_readings.py \
|
||||
"$DATASET_DIR/readings.csv"
|
||||
fi
|
||||
fi
|
||||
|
||||
echo ""
|
||||
|
||||
Reference in new issue
Block a user