{ "experiment": "TSFA-SPR: Temporal-Shared-Private Representation", "seed": 42, "folds": 5, "grouped_by": "video_id/group_id", "grid_size": 50, "feature_dimensions": { "text": 768, "audio": 25, "vision": 192 }, "common_dim": 128, "shared_dim": 64, "private_dim": 64, "batch_size": 8, "epochs": 40, "learning_rate": 0.001, "temperature": 0.1, "orthogonality_weight": 0.1, "reconstruction_weight": 1.0, "contrastive_pairs": [ [ "text", "audio" ], [ "text", "vision" ], [ "audio", "vision" ] ], "hard_negative_offsets": [ -5, -3, -2, 2, 3, 5 ], "loss_variants": { "SP-noOrth-noRec": [ true, false, false ], "SP+Orth": [ true, true, false ], "SPR": [ true, true, true ], "SPR-noSharedContrastive": [ false, true, true ] }, "emotion_probe": { "classification": "StandardScaler + LogisticRegression(C=0.05, max_iter=5000)", "regression": "StandardScaler + Ridge(alpha=25), clipped to [-3, 3]", "temporal_pooling": "50 slots to five consecutive 10-slot mean bins", "all_scalers_fit_on_training_fold_only": true }, "dimension_controls": { "SPR_main": "unfused shared 3x64 plus private 3x64 = 384 per slot; 1920 after five-bin pooling", "SPR_dim_matched": "mean fused shared 64 plus all private 3x64 = 256 per slot; 1280 after pooling", "RawPrivate_PCA": "train-fold PCA from raw private 985-d slot vectors to 256 per slot; 1280 after pooling if 256 components" }, "emotion_labels_used_in_factorizer_training": false, "frozen_m4_temporal_branch": true, "explicit_time_code_or_slot_index_in_shared_encoder": false, "bootstrap_repeats": 2000 }