{ "created_utc": "2026-09-23T10:26:41.743208+00:00", "sample_count": 100, "diagnostic_sample": "-tPCytz4rww/12", "seed": 42, "device": "cuda", "gpu_name": "NVIDIA GeForce RTX 5070 Ti", "python": "3.14.7", "torch": "2.14.0+cu130", "experiments": [ { "name": "D0", "scope": "single sample; span + barycenter band only", "steps_per_model": 1000, "loss": "5 * L_span + 10 * L_band" }, { "name": "D1", "scope": "single sample; Gaussian target KL only", "steps_per_model": 1000, "sigma_normalized_time": 0.1, "M4_control": "no PE versus fixed sinusoidal PE" }, { "name": "D2", "scope": "all 100 clips; Gaussian target KL only; one seed; in-sample diagnostic", "steps_per_model": 500, "sigma_normalized_time": 0.1, "M4_position_encoding": "fixed sinusoidal" }, { "name": "D3", "scope": "all 100 clips; Gaussian KL + masked reconstruction + contrastive; one seed; in-sample diagnostic", "steps_per_model": 500, "sigma_normalized_time": 0.1, "M4_position_encoding": "fixed sinusoidal" } ], "optimizer": "AdamW", "learning_rate": 0.001, "dropout": 0.0, "batch_size": 8, "gradient_logging_interval_steps": 20, "gradient_metrics": [ "W_Q", "W_K", "M4 latent slots Z" ], "features_changed": false, "M1_M2_changed": false, "elapsed_seconds": 103.73381090164185, "interpretation_limits": [ "D0 and D1 overfit one selected sample and diagnose optimization/representability only.", "D2 and D3 train and evaluate on the same 100 clips; they diagnose whether the target can be optimized, not generalization.", "Gaussian targets are weak temporal priors constructed from timestamps; they are not human alignment ground truth.", "M4 absolute sinusoidal encoding is enabled only for D1's PE control and D2/D3; prior M1-M4 and v2 results are unchanged." ] }