{ "created_utc": "2026-09-23T09:07:49.626404+00:00", "sample_count": 100, "group_count": 37, "folds": 5, "seeds": [ 42, 3407, 2026 ], "device": "cuda", "gpu_name": "NVIDIA GeForce RTX 5070 Ti", "python": "3.14.7", "torch": "2.14.0+cu130", "parameters": { "grid_size": 50, "hidden_size": 128, "heads": 4, "dropout": 0.1, "batch_size": 8, "epochs_max": 50, "early_stopping_patience": 8, "learning_rate": 0.0001, "mask_ratio": 0.2, "retrieval_probe_epochs": 20, "reconstruction_probe_epochs": 25, "mvr_epsilon": 0.02 }, "objective": "masked reconstruction + cross-modal contrastive + temporal monotonicity; emotion labels unused", "split_rule": "GroupKFold by group_id/video_id", "example_sample_id": "-tPCytz4rww/12", "elapsed_seconds": 516.2944533824921, "interpretation_limits": [ "Grid-index retrieval is a representation-consistency probe, not independent temporal ground truth.", "Masked reconstruction uses a decoder trained on the training fold and reports standardized-feature errors.", "The emotion probe is a small-sample downstream utility check, not a claim of generalization to MOSEI.", "No human event timestamps are available, so human IoU/MATE is not reported." ] }