加入实验输出(批次 14/14)
This commit is contained in:
@@ -0,0 +1,43 @@
|
||||
{
|
||||
"seed": 42,
|
||||
"grid_size": 50,
|
||||
"hidden_size": 128,
|
||||
"heads_in_frozen_alignment": 4,
|
||||
"delta_normalized_time": 0.1,
|
||||
"hard_negative_offsets_slots": [
|
||||
-5,
|
||||
-3,
|
||||
-2,
|
||||
2,
|
||||
3,
|
||||
5
|
||||
],
|
||||
"semantic_epochs": 40,
|
||||
"semantic_learning_rate": 0.001,
|
||||
"local_contrastive_temperature": 0.1,
|
||||
"content_probe_epochs": 40,
|
||||
"content_probe_temperature": 0.1,
|
||||
"reconstruction_probe_epochs": 40,
|
||||
"shuffle_repeats": 20,
|
||||
"random_window_repeats": 20,
|
||||
"window_ablation_status": "delta sweep and 80-percent attention-mass window deferred until the seven-method MVP is reviewed",
|
||||
"semantic_query_key_inputs": "Frozen M4 attention-pooled text value output is the semantic query input; M4 pre-attention projected Audio/Vision content is the semantic key/value input. No explicit source-time code, absolute PE, slot ID, or latent positional vector is passed to semantic Q/K. Temporal selection can still encode time indirectly.",
|
||||
"baseline_content_values": "M4 and M3 Audio/Vision use each checkpoint's actual MHA attention output. M3 Text uses attention-pooled pre-attention text features to remove its source-time query residual.",
|
||||
"temporal_role": "Frozen M4 attention supplies candidate masks; TSFA-multiply additionally multiplies semantic probabilities by M4 temporal attention",
|
||||
"local_contrastive": "same-slot positives with within-clip slot-offset negatives at +/-2, +/-3, +/-5; symmetric Text-Audio and Text-Vision loss",
|
||||
"random_candidate": "same delta-width windows with independently sampled centers at inference; 20 held-out random draws, one train draw for the train-only probe",
|
||||
"global_candidate": "all valid source positions available to semantic attention",
|
||||
"content_probe": "same train-only symmetric within-clip InfoNCE probe applied to each method; held-out negatives are positions more than two slots away",
|
||||
"content_shuffle": "independently permute each modality's 50 projected content rows within each held-out clip; M4 and TSFA-main",
|
||||
"reconstruction": "fold-train decoder predicts one aligned content representation from the other two; shift one partner by signed offsets +/-1,2,5,10 on identical target support",
|
||||
"confidence_intervals": "95-percent video_id-cluster bootstrap; 2,000 repetitions for CSV summaries",
|
||||
"methods": [
|
||||
"M3_noSourceTime",
|
||||
"M3_sourceTime",
|
||||
"M4_sourceTime",
|
||||
"TSFA-main",
|
||||
"TSFA-multiply",
|
||||
"TSFA-random",
|
||||
"TSFA-global"
|
||||
]
|
||||
}
|
||||
Reference in New Issue
Block a user