Files
modeling_zhaocui/deep_learning/Q1/outputs/tsfa/tsfa_config.json
T

43 lines
2.4 KiB
JSON

{
"seed": 42,
"grid_size": 50,
"hidden_size": 128,
"heads_in_frozen_alignment": 4,
"delta_normalized_time": 0.1,
"hard_negative_offsets_slots": [
-5,
-3,
-2,
2,
3,
5
],
"semantic_epochs": 40,
"semantic_learning_rate": 0.001,
"local_contrastive_temperature": 0.1,
"content_probe_epochs": 40,
"content_probe_temperature": 0.1,
"reconstruction_probe_epochs": 40,
"shuffle_repeats": 20,
"random_window_repeats": 20,
"window_ablation_status": "delta sweep and 80-percent attention-mass window deferred until the seven-method MVP is reviewed",
"semantic_query_key_inputs": "Frozen M4 attention-pooled text value output is the semantic query input; M4 pre-attention projected Audio/Vision content is the semantic key/value input. No explicit source-time code, absolute PE, slot ID, or latent positional vector is passed to semantic Q/K. Temporal selection can still encode time indirectly.",
"baseline_content_values": "M4 and M3 Audio/Vision use each checkpoint's actual MHA attention output. M3 Text uses attention-pooled pre-attention text features to remove its source-time query residual.",
"temporal_role": "Frozen M4 attention supplies candidate masks; TSFA-multiply additionally multiplies semantic probabilities by M4 temporal attention",
"local_contrastive": "same-slot positives with within-clip slot-offset negatives at +/-2, +/-3, +/-5; symmetric Text-Audio and Text-Vision loss",
"random_candidate": "same delta-width windows with independently sampled centers at inference; 20 held-out random draws, one train draw for the train-only probe",
"global_candidate": "all valid source positions available to semantic attention",
"content_probe": "same train-only symmetric within-clip InfoNCE probe applied to each method; held-out negatives are positions more than two slots away",
"content_shuffle": "independently permute each modality's 50 projected content rows within each held-out clip; M4 and TSFA-main",
"reconstruction": "fold-train decoder predicts one aligned content representation from the other two; shift one partner by signed offsets +/-1,2,5,10 on identical target support",
"confidence_intervals": "95-percent video_id-cluster bootstrap; 2,000 repetitions for CSV summaries",
"methods": [
"M3_noSourceTime",
"M3_sourceTime",
"M4_sourceTime",
"TSFA-main",
"TSFA-multiply",
"TSFA-random",
"TSFA-global"
]
}