43 lines
2.4 KiB
JSON
43 lines
2.4 KiB
JSON
{
|
|
"seed": 42,
|
|
"grid_size": 50,
|
|
"hidden_size": 128,
|
|
"heads_in_frozen_alignment": 4,
|
|
"delta_normalized_time": 0.1,
|
|
"hard_negative_offsets_slots": [
|
|
-5,
|
|
-3,
|
|
-2,
|
|
2,
|
|
3,
|
|
5
|
|
],
|
|
"semantic_epochs": 40,
|
|
"semantic_learning_rate": 0.001,
|
|
"local_contrastive_temperature": 0.1,
|
|
"content_probe_epochs": 40,
|
|
"content_probe_temperature": 0.1,
|
|
"reconstruction_probe_epochs": 40,
|
|
"shuffle_repeats": 20,
|
|
"random_window_repeats": 20,
|
|
"window_ablation_status": "delta sweep and 80-percent attention-mass window deferred until the seven-method MVP is reviewed",
|
|
"semantic_query_key_inputs": "Frozen M4 attention-pooled text value output is the semantic query input; M4 pre-attention projected Audio/Vision content is the semantic key/value input. No explicit source-time code, absolute PE, slot ID, or latent positional vector is passed to semantic Q/K. Temporal selection can still encode time indirectly.",
|
|
"baseline_content_values": "M4 and M3 Audio/Vision use each checkpoint's actual MHA attention output. M3 Text uses attention-pooled pre-attention text features to remove its source-time query residual.",
|
|
"temporal_role": "Frozen M4 attention supplies candidate masks; TSFA-multiply additionally multiplies semantic probabilities by M4 temporal attention",
|
|
"local_contrastive": "same-slot positives with within-clip slot-offset negatives at +/-2, +/-3, +/-5; symmetric Text-Audio and Text-Vision loss",
|
|
"random_candidate": "same delta-width windows with independently sampled centers at inference; 20 held-out random draws, one train draw for the train-only probe",
|
|
"global_candidate": "all valid source positions available to semantic attention",
|
|
"content_probe": "same train-only symmetric within-clip InfoNCE probe applied to each method; held-out negatives are positions more than two slots away",
|
|
"content_shuffle": "independently permute each modality's 50 projected content rows within each held-out clip; M4 and TSFA-main",
|
|
"reconstruction": "fold-train decoder predicts one aligned content representation from the other two; shift one partner by signed offsets +/-1,2,5,10 on identical target support",
|
|
"confidence_intervals": "95-percent video_id-cluster bootstrap; 2,000 repetitions for CSV summaries",
|
|
"methods": [
|
|
"M3_noSourceTime",
|
|
"M3_sourceTime",
|
|
"M4_sourceTime",
|
|
"TSFA-main",
|
|
"TSFA-multiply",
|
|
"TSFA-random",
|
|
"TSFA-global"
|
|
]
|
|
} |