refactor: reorganize analysis data to support multiple runs
Strategy: key runs by bicorder version (not date), promote shared inputs
to analysis/data/, and stamp every output with its bicorder_version so
a re-run on edited gradients is self-describing.
Data layout:
- Promote the shared protocol inputs out of the run directory:
analysis/data/protocols_edited.csv (411 cleaned protocols)
analysis/data/protocols_raw.csv (774 un-cleaned entries)
- Rename the v1.2.6 synthetic run:
data/synthetic_20251116/ -> data/synthetic_1.2.6/
so the bicorder version it was scored against is explicit (gradient
structure changes between versions make date-based names ambiguous)
Provenance:
- bicorder_analyze.py now writes a 'bicorder_version' column into every
output readings.csv, recording which gradient structure produced it
Scripts:
- Update the real code defaults that pointed at the old run path
(bicorder_classifier.py, classify_readings.py, sync_readings.sh,
compare_analyses.py) and refresh docstring/help examples
- Remove a stray committed __pycache__/.pyc
Docs: analysis/README.md documents the new layout + how to add a run;
WORKFLOW.md, TEST_COMMANDS.md, INTEGRATION_GUIDE.md paths updated.
This commit is contained in:
1 parent
459015fe17
commit
6ae77a4f9b
474 files changed
+90
-66
No files matched your search
Binary file not shown.
@@ -55,6 +55,7 @@ def process_csv(input_csv, output_csv, bicorder_path, analyst=None, standpoint=N
|
||||
# Load bicorder configuration
|
||||
bicorder_data = load_bicorder_config(bicorder_path)
|
||||
gradients = extract_gradients(bicorder_data)
|
||||
bicorder_version = bicorder_data.get('version', '')
|
||||
|
||||
with open(input_csv, 'r', encoding='utf-8') as infile, \
|
||||
open(output_csv, 'w', newline='', encoding='utf-8') as outfile:
|
||||
@@ -68,6 +69,9 @@ def process_csv(input_csv, output_csv, bicorder_path, analyst=None, standpoint=N
|
||||
gradient_columns = [g['column_name'] for g in gradients]
|
||||
output_fields = list(original_fields) + gradient_columns
|
||||
|
||||
# Add the bicorder version as a provenance column
|
||||
output_fields.append('bicorder_version')
|
||||
|
||||
# Add metadata columns if provided
|
||||
if analyst is not None:
|
||||
output_fields.append('analyst')
|
||||
@@ -87,6 +91,9 @@ def process_csv(input_csv, output_csv, bicorder_path, analyst=None, standpoint=N
|
||||
for gradient in gradients:
|
||||
output_row[gradient['column_name']] = ''
|
||||
|
||||
# Record which bicorder version these gradients came from
|
||||
output_row['bicorder_version'] = bicorder_version
|
||||
|
||||
# Add metadata if provided
|
||||
if analyst is not None:
|
||||
output_row['analyst'] = analyst
|
||||
|
||||
@@ -105,16 +105,16 @@ def main():
|
||||
epilog="""
|
||||
Example usage:
|
||||
# Process all protocols
|
||||
python3 bicorder_batch.py data/synthetic_20251116/protocols_edited.csv -o data/synthetic_20251116/readings.csv
|
||||
python3 bicorder_batch.py data/protocols_edited.csv -o data/synthetic_1.2.6/readings.csv
|
||||
|
||||
# Process specific rows
|
||||
python3 bicorder_batch.py data/synthetic_20251116/protocols_edited.csv -o data/synthetic_20251116/readings.csv --start 1 --end 5
|
||||
python3 bicorder_batch.py data/protocols_edited.csv -o data/synthetic_1.2.6/readings.csv --start 1 --end 5
|
||||
|
||||
# With specific model
|
||||
python3 bicorder_batch.py data/synthetic_20251116/protocols_edited.csv -o data/synthetic_20251116/readings.csv -m mistral
|
||||
python3 bicorder_batch.py data/protocols_edited.csv -o data/synthetic_1.2.6/readings.csv -m mistral
|
||||
|
||||
# With metadata
|
||||
python3 bicorder_batch.py data/synthetic_20251116/protocols_edited.csv -o data/synthetic_20251116/readings.csv -a "Your Name" -s "Your standpoint"
|
||||
python3 bicorder_batch.py data/protocols_edited.csv -o data/synthetic_1.2.6/readings.csv -a "Your Name" -s "Your standpoint"
|
||||
"""
|
||||
)
|
||||
|
||||
|
||||
@@ -70,7 +70,7 @@ class BicorderClassifier:
|
||||
2: "Institutional/Bureaucratic"
|
||||
}
|
||||
|
||||
def __init__(self, diagnostic_csv='data/synthetic_20251116/readings.csv',
|
||||
def __init__(self, diagnostic_csv='data/synthetic_1.2.6/readings.csv',
|
||||
model_path=None):
|
||||
"""Initialize classifier with pre-computed model data."""
|
||||
if model_path is None:
|
||||
|
||||
@@ -9,7 +9,7 @@ classified — though with lower confidence.
|
||||
Usage:
|
||||
python3 scripts/classify_readings.py data/manual_20260320/readings.csv
|
||||
python3 scripts/classify_readings.py data/manual_20260320/readings.csv \\
|
||||
--training data/synthetic_20251116/readings.csv \\
|
||||
--training data/synthetic_1.2.6/readings.csv \\
|
||||
--output data/manual_20260320/analysis/classifications.csv
|
||||
"""
|
||||
|
||||
@@ -29,8 +29,8 @@ def main():
|
||||
parser.add_argument('input_csv', help='Readings CSV to classify')
|
||||
parser.add_argument(
|
||||
'--training',
|
||||
default='data/synthetic_20251116/readings.csv',
|
||||
help='Training CSV for classifier (default: synthetic_20251116)'
|
||||
default='data/synthetic_1.2.6/readings.csv',
|
||||
help='Training CSV for classifier (default: synthetic_1.2.6)'
|
||||
)
|
||||
parser.add_argument(
|
||||
'--output', default=None,
|
||||
|
||||
@@ -157,11 +157,11 @@ def compare_analyses(reference_file, comparison_files):
|
||||
|
||||
if __name__ == "__main__":
|
||||
# Define file paths
|
||||
reference_file = "data/synthetic_20251116/readings_manual.csv"
|
||||
reference_file = "data/synthetic_1.2.6/readings_manual.csv"
|
||||
comparison_files = [
|
||||
"data/synthetic_20251116/readings_gemma3-12b.csv",
|
||||
"data/synthetic_20251116/readings_gpt-oss.csv",
|
||||
"data/synthetic_20251116/readings_mistral.csv"
|
||||
"data/synthetic_1.2.6/readings_gemma3-12b.csv",
|
||||
"data/synthetic_1.2.6/readings_gpt-oss.csv",
|
||||
"data/synthetic_1.2.6/readings_mistral.csv"
|
||||
]
|
||||
|
||||
# Check if files exist
|
||||
|
||||
@@ -137,11 +137,11 @@ def main():
|
||||
formatter_class=argparse.RawDescriptionHelpFormatter,
|
||||
epilog="""
|
||||
Example usage:
|
||||
python3 scripts/convert_csv_to_json.py data/synthetic_20251116/readings.csv
|
||||
python3 scripts/convert_csv_to_json.py data/synthetic_1.2.6/readings.csv
|
||||
python3 scripts/convert_csv_to_json.py data/manual_20260101/readings.csv --output-dir data/manual_20260101/json
|
||||
"""
|
||||
)
|
||||
parser.add_argument('input_csv', help='Diagnostic readings CSV (e.g. data/synthetic_20251116/readings.csv)')
|
||||
parser.add_argument('input_csv', help='Diagnostic readings CSV (e.g. data/synthetic_1.2.6/readings.csv)')
|
||||
parser.add_argument('--output-dir', default=None,
|
||||
help='Output directory for JSON files (default: <dataset_dir>/json)')
|
||||
parser.add_argument('--bicorder', default='../bicorder.json',
|
||||
|
||||
@@ -9,7 +9,7 @@ When gradients are renamed in bicorder.json, add the old→new mapping to
|
||||
COLUMN_RENAMES so the training CSV columns are correctly aligned.
|
||||
|
||||
Usage:
|
||||
python3 scripts/export_model_for_js.py data/synthetic_20251116/readings.csv
|
||||
python3 scripts/export_model_for_js.py data/synthetic_1.2.6/readings.csv
|
||||
python3 scripts/export_model_for_js.py data/manual_20260101/readings.csv --output bicorder_model.json
|
||||
"""
|
||||
|
||||
@@ -60,11 +60,11 @@ def main():
|
||||
formatter_class=argparse.RawDescriptionHelpFormatter,
|
||||
epilog="""
|
||||
Example usage:
|
||||
python3 scripts/export_model_for_js.py data/synthetic_20251116/readings.csv
|
||||
python3 scripts/export_model_for_js.py data/synthetic_1.2.6/readings.csv
|
||||
python3 scripts/export_model_for_js.py data/manual_20260101/readings.csv --output bicorder_model.json
|
||||
"""
|
||||
)
|
||||
parser.add_argument('input_csv', help='Diagnostic readings CSV (e.g. data/synthetic_20251116/readings.csv)')
|
||||
parser.add_argument('input_csv', help='Diagnostic readings CSV (e.g. data/synthetic_1.2.6/readings.csv)')
|
||||
parser.add_argument('--output', default='bicorder_model.json',
|
||||
help='Output model JSON path (default: bicorder_model.json)')
|
||||
args = parser.parse_args()
|
||||
|
||||
@@ -3,8 +3,8 @@
|
||||
Create LDA visualization to maximize cluster separation.
|
||||
|
||||
Usage:
|
||||
python3 scripts/lda_visualization.py data/synthetic_20251116.csv
|
||||
python3 scripts/lda_visualization.py data/synthetic_20251116.csv --results-dir analysis_results/synthetic_20251116
|
||||
python3 scripts/lda_visualization.py data/synthetic_1.2.6/readings.csv
|
||||
python3 scripts/lda_visualization.py data/synthetic_1.2.6/readings.csv --results-dir analysis_results/synthetic_1.2.6
|
||||
"""
|
||||
|
||||
import argparse
|
||||
@@ -22,11 +22,11 @@ def main():
|
||||
formatter_class=argparse.RawDescriptionHelpFormatter,
|
||||
epilog="""
|
||||
Example usage:
|
||||
python3 scripts/lda_visualization.py data/synthetic_20251116/readings.csv
|
||||
python3 scripts/lda_visualization.py data/synthetic_1.2.6/readings.csv
|
||||
python3 scripts/lda_visualization.py data/manual_20260101/readings.csv --analysis-dir data/manual_20260101/analysis
|
||||
"""
|
||||
)
|
||||
parser.add_argument('input_csv', help='Diagnostic readings CSV (e.g. data/synthetic_20251116/readings.csv)')
|
||||
parser.add_argument('input_csv', help='Diagnostic readings CSV (e.g. data/synthetic_1.2.6/readings.csv)')
|
||||
parser.add_argument('--analysis-dir', default=None,
|
||||
help='Analysis directory (default: <dataset_dir>/analysis)')
|
||||
args = parser.parse_args()
|
||||
|
||||
@@ -764,13 +764,13 @@ def main():
|
||||
formatter_class=argparse.RawDescriptionHelpFormatter,
|
||||
epilog="""
|
||||
Examples:
|
||||
python3 scripts/multivariate_analysis.py data/synthetic_20251116/readings.csv
|
||||
python3 scripts/multivariate_analysis.py data/synthetic_20251116/readings.csv --output data/synthetic_20251116/analysis
|
||||
python3 scripts/multivariate_analysis.py data/synthetic_20251116/readings.csv --analyses clustering pca
|
||||
python3 scripts/multivariate_analysis.py data/synthetic_1.2.6/readings.csv
|
||||
python3 scripts/multivariate_analysis.py data/synthetic_1.2.6/readings.csv --output data/synthetic_1.2.6/analysis
|
||||
python3 scripts/multivariate_analysis.py data/synthetic_1.2.6/readings.csv --analyses clustering pca
|
||||
"""
|
||||
)
|
||||
|
||||
parser.add_argument('csv_file', help='Diagnostic readings CSV (e.g. data/synthetic_20251116/readings.csv)')
|
||||
parser.add_argument('csv_file', help='Diagnostic readings CSV (e.g. data/synthetic_1.2.6/readings.csv)')
|
||||
parser.add_argument('--output', '-o', default=None,
|
||||
help='Output directory (default: <dataset_dir>/analysis)')
|
||||
parser.add_argument('--min-coverage', type=float, default=0.0,
|
||||
|
||||
@@ -3,7 +3,7 @@
|
||||
Comprehensive review of the analysis for errors and inconsistencies.
|
||||
|
||||
Usage:
|
||||
python3 scripts/review_analysis.py data/synthetic_20251116.csv
|
||||
python3 scripts/review_analysis.py data/synthetic_1.2.6/readings.csv
|
||||
python3 scripts/review_analysis.py data/manual_20260101.csv --results-dir analysis_results/manual_20260101
|
||||
"""
|
||||
|
||||
@@ -19,11 +19,11 @@ def main():
|
||||
formatter_class=argparse.RawDescriptionHelpFormatter,
|
||||
epilog="""
|
||||
Example usage:
|
||||
python3 scripts/review_analysis.py data/synthetic_20251116/readings.csv
|
||||
python3 scripts/review_analysis.py data/synthetic_1.2.6/readings.csv
|
||||
python3 scripts/review_analysis.py data/manual_20260101/readings.csv --analysis-dir data/manual_20260101/analysis
|
||||
"""
|
||||
)
|
||||
parser.add_argument('input_csv', help='Diagnostic readings CSV (e.g. data/synthetic_20251116/readings.csv)')
|
||||
parser.add_argument('input_csv', help='Diagnostic readings CSV (e.g. data/synthetic_1.2.6/readings.csv)')
|
||||
parser.add_argument('--analysis-dir', default=None,
|
||||
help='Analysis directory (default: <dataset_dir>/analysis)')
|
||||
args = parser.parse_args()
|
||||
|
||||
@@ -7,7 +7,7 @@
|
||||
# scripts/sync_readings.sh data/manual_20260320
|
||||
# scripts/sync_readings.sh data/manual_20260320 --no-analysis
|
||||
# scripts/sync_readings.sh data/manual_20260320 --min-coverage 0.8
|
||||
# scripts/sync_readings.sh data/manual_20260320 --training data/synthetic_20251116/readings.csv
|
||||
# scripts/sync_readings.sh data/manual_20260320 --training data/synthetic_1.2.6/readings.csv
|
||||
#
|
||||
# .sync_source format:
|
||||
# REMOTE_URL=https://git.example.org/user/repo
|
||||
@@ -18,7 +18,7 @@ set -euo pipefail
|
||||
DATASET_DIR="${1:?Usage: $0 <dataset_dir> [--no-analysis] [--min-coverage N]}"
|
||||
RUN_ANALYSIS=true
|
||||
MIN_COVERAGE=0.8
|
||||
TRAINING_CSV="data/synthetic_20251116/readings.csv"
|
||||
TRAINING_CSV="data/synthetic_1.2.6/readings.csv"
|
||||
|
||||
shift || true
|
||||
while [[ $# -gt 0 ]]; do
|
||||
|
||||
@@ -3,7 +3,7 @@
|
||||
Create visualizations of k-means clusters overlaid on dimensionality reduction plots.
|
||||
|
||||
Usage:
|
||||
python3 scripts/visualize_clusters.py data/synthetic_20251116.csv
|
||||
python3 scripts/visualize_clusters.py data/synthetic_1.2.6/readings.csv
|
||||
python3 scripts/visualize_clusters.py data/manual_20260101.csv --results-dir analysis_results/manual_20260101
|
||||
"""
|
||||
|
||||
@@ -20,11 +20,11 @@ def main():
|
||||
formatter_class=argparse.RawDescriptionHelpFormatter,
|
||||
epilog="""
|
||||
Example usage:
|
||||
python3 scripts/visualize_clusters.py data/synthetic_20251116/readings.csv
|
||||
python3 scripts/visualize_clusters.py data/synthetic_1.2.6/readings.csv
|
||||
python3 scripts/visualize_clusters.py data/manual_20260101/readings.csv --analysis-dir data/manual_20260101/analysis
|
||||
"""
|
||||
)
|
||||
parser.add_argument('input_csv', help='Diagnostic readings CSV (e.g. data/synthetic_20251116/readings.csv)')
|
||||
parser.add_argument('input_csv', help='Diagnostic readings CSV (e.g. data/synthetic_1.2.6/readings.csv)')
|
||||
parser.add_argument('--analysis-dir', default=None,
|
||||
help='Analysis directory (default: <dataset_dir>/analysis)')
|
||||
args = parser.parse_args()
|
||||
|
||||
Reference in new issue
Block a user