80 lines
1.8 KiB
JSON
80 lines
1.8 KiB
JSON
{
|
|
"experiment": "TSFA-SPR: Temporal-Shared-Private Representation",
|
|
"seed": 42,
|
|
"folds": 5,
|
|
"grouped_by": "video_id/group_id",
|
|
"grid_size": 50,
|
|
"feature_dimensions": {
|
|
"text": 768,
|
|
"audio": 25,
|
|
"vision": 192
|
|
},
|
|
"common_dim": 128,
|
|
"shared_dim": 64,
|
|
"private_dim": 64,
|
|
"batch_size": 8,
|
|
"epochs": 40,
|
|
"learning_rate": 0.001,
|
|
"temperature": 0.1,
|
|
"orthogonality_weight": 0.1,
|
|
"reconstruction_weight": 1.0,
|
|
"contrastive_pairs": [
|
|
[
|
|
"text",
|
|
"audio"
|
|
],
|
|
[
|
|
"text",
|
|
"vision"
|
|
],
|
|
[
|
|
"audio",
|
|
"vision"
|
|
]
|
|
],
|
|
"hard_negative_offsets": [
|
|
-5,
|
|
-3,
|
|
-2,
|
|
2,
|
|
3,
|
|
5
|
|
],
|
|
"loss_variants": {
|
|
"SP-noOrth-noRec": [
|
|
true,
|
|
false,
|
|
false
|
|
],
|
|
"SP+Orth": [
|
|
true,
|
|
true,
|
|
false
|
|
],
|
|
"SPR": [
|
|
true,
|
|
true,
|
|
true
|
|
],
|
|
"SPR-noSharedContrastive": [
|
|
false,
|
|
true,
|
|
true
|
|
]
|
|
},
|
|
"emotion_probe": {
|
|
"classification": "StandardScaler + LogisticRegression(C=0.05, max_iter=5000)",
|
|
"regression": "StandardScaler + Ridge(alpha=25), clipped to [-3, 3]",
|
|
"temporal_pooling": "50 slots to five consecutive 10-slot mean bins",
|
|
"all_scalers_fit_on_training_fold_only": true
|
|
},
|
|
"dimension_controls": {
|
|
"SPR_main": "unfused shared 3x64 plus private 3x64 = 384 per slot; 1920 after five-bin pooling",
|
|
"SPR_dim_matched": "mean fused shared 64 plus all private 3x64 = 256 per slot; 1280 after pooling",
|
|
"RawPrivate_PCA": "train-fold PCA from raw private 985-d slot vectors to 256 per slot; 1280 after pooling if 256 components"
|
|
},
|
|
"emotion_labels_used_in_factorizer_training": false,
|
|
"frozen_m4_temporal_branch": true,
|
|
"explicit_time_code_or_slot_index_in_shared_encoder": false,
|
|
"bootstrap_repeats": 2000
|
|
} |