{ "seed": 42, "grid_size": 50, "hidden_size": 128, "heads_in_frozen_alignment": 4, "delta_normalized_time": 0.1, "hard_negative_offsets_slots": [ -5, -3, -2, 2, 3, 5 ], "semantic_epochs": 40, "semantic_learning_rate": 0.001, "local_contrastive_temperature": 0.1, "content_probe_epochs": 40, "content_probe_temperature": 0.1, "reconstruction_probe_epochs": 40, "shuffle_repeats": 20, "random_window_repeats": 20, "window_ablation_status": "delta sweep and 80-percent attention-mass window deferred until the seven-method MVP is reviewed", "semantic_query_key_inputs": "Frozen M4 attention-pooled text value output is the semantic query input; M4 pre-attention projected Audio/Vision content is the semantic key/value input. No explicit source-time code, absolute PE, slot ID, or latent positional vector is passed to semantic Q/K. Temporal selection can still encode time indirectly.", "baseline_content_values": "M4 and M3 Audio/Vision use each checkpoint's actual MHA attention output. M3 Text uses attention-pooled pre-attention text features to remove its source-time query residual.", "temporal_role": "Frozen M4 attention supplies candidate masks; TSFA-multiply additionally multiplies semantic probabilities by M4 temporal attention", "local_contrastive": "same-slot positives with within-clip slot-offset negatives at +/-2, +/-3, +/-5; symmetric Text-Audio and Text-Vision loss", "random_candidate": "same delta-width windows with independently sampled centers at inference; 20 held-out random draws, one train draw for the train-only probe", "global_candidate": "all valid source positions available to semantic attention", "content_probe": "same train-only symmetric within-clip InfoNCE probe applied to each method; held-out negatives are positions more than two slots away", "content_shuffle": "independently permute each modality's 50 projected content rows within each held-out clip; M4 and TSFA-main", "reconstruction": "fold-train decoder predicts one aligned content representation from the other two; shift one partner by signed offsets +/-1,2,5,10 on identical target support", "confidence_intervals": "95-percent video_id-cluster bootstrap; 2,000 repetitions for CSV summaries", "methods": [ "M3_noSourceTime", "M3_sourceTime", "M4_sourceTime", "TSFA-main", "TSFA-multiply", "TSFA-random", "TSFA-global" ] }