{ "seed": 20260924, "text_encoder": "google-bert/bert-base-uncased", "training_configuration": { "student_epoch_limit": 12, "imputer_epochs": 8, "batch_size": 64, "early_stopping_patience": 3, "device": "cuda", "device_name": "NVIDIA GeForce RTX 5070 Ti", "optimizer": "AdamW", "student_learning_rate": 0.0003, "student_weight_decay": 0.001, "imputer_learning_rate": 0.0003, "imputer_weight_decay": 0.0001, "early_stopping_metric": "mean untempered selection_nll over fixed group-disjoint internal training scenarios", "inner_selection_scenarios": [ "0.0/natural", "0.3/single", "0.3/sync", "0.5/async" ], "inner_selection_source_video_groups": 76 }, "training_input": "E题数据/附件2-数据集特征文件/aligned_50.pkl", "training_sha256": "66e867aa74bc70a844e806e5571e371c9abb4a35f9e2887ce9b4d97ff2cb8fcd", "official_group_overlap": { "train_valid": 0, "train_test": 0, "valid_test": 0 }, "official_splits": { "train": { "n": 3395, "source_video_groups": 1528 }, "valid": { "n": 728, "source_video_groups": 239 }, "test": { "n": 727, "source_video_groups": 381 } }, "internal_train_holdouts": { "fit": { "n": 3030, "video_groups": 1375 }, "reliability_selection": { "n": 190, "video_groups": 76 }, "temperature_calibration": { "n": 175, "video_groups": 77 }, "all_group_disjoint": true }, "feature_standardization": "fit-only observed rows, per-dimension; fixed for valid/test/attachment3", "missing_mask": "row-level all-zero convention; q*=1 only where currently visible, J_Q=0; hidden metadata is zeroed", "observation_quality": { "quality_score_fields_present": false, "quality_available_flag_present": false, "fallback": "q*=1 and J_Q=0 for visible rows; R_eff=R", "quality_noise_mapping_ablation": "not identifiable on aligned_50 because no row quality score varies" }, "imputer": { "type": "structured linear Gaussian shared-private state space", "state_dims": { "shared": 8, "private_each": 4 }, "posterior": "block-tridiagonal equivalent Kalman information filter + RTS smoother", "sampling": "joint latent trajectories and missing emissions; observed features copied exactly", "fit_objective": "train-only observed Gaussian marginal likelihood including log determinants", "epochs": 8, "frozen_before_teacher_student": true }, "architecture": { "projection": 32, "bigru_hidden_each_direction": 16, "cross_source_layers": 1, "cross_time_read": true, "rank": 4, "reliability_gru": "directional hidden decay; reset applied before candidate map; update gate multiplied by rho", "final_gate": "rho times bounded content score plus positive null prior", "output": "neutral point mass plus sign-specific Beta magnitudes; K-path probabilities mixed before decoding" }, "reliability_hyperparameters": { "selected_per_model_on": "group-disjoint internal training reliability-validation slice", "candidate_values": [ [ 0.5, 0.0, 0.0, 0.0 ], [ 0.5, 0.05, 0.05, 0.05 ], [ 0.5, 0.1, 0.0, 0.0 ], [ 0.3, 0.05, 0.05, 0.05 ], [ 0.7, 0.05, 0.05, 0.05 ] ], "validation_scenarios": [ "0.0/natural", "0.3/single", "0.3/sync", "0.5/async" ], "selected_by_model": { "teacher": [ 0.3, 0.05, 0.05, 0.05 ], "C1": [ 0.5, 0.05, 0.05, 0.05 ], "C2": [ 0.5, 0.05, 0.05, 0.05 ], "C3": [ 0.3, 0.05, 0.05, 0.05 ], "C4": [ 0.7, 0.05, 0.05, 0.05 ], "C5": [ 0.3, 0.05, 0.05, 0.05 ], "C6": [ 0.3, 0.05, 0.05, 0.05 ], "C6_no_distance": [ 0.5, 0.0, 0.0, 0.0 ], "C6_no_reconstruction": [ 0.3, 0.05, 0.05, 0.05 ], "C6_pointmask": [ 0.3, 0.05, 0.05, 0.05 ], "C7_distill": [ 0.3, 0.05, 0.05, 0.05 ], "C7_group": [ 0.3, 0.05, 0.05, 0.05 ] } }, "group_risk_hyperparameters": { "selection": "lambda_group and group_temperature jointly selected with reliability hyperparameters on fixed group-disjoint internal training scenarios", "candidate_values": [ [ 0.05, 0.1 ], [ 0.1, 0.05 ], [ 0.1, 0.1 ], [ 0.1, 0.2 ], [ 0.2, 0.1 ] ], "selected": [ 0.1, 0.05 ], "selection_split": "reliability_validation" }, "loss": { "supervision": "negative log mixture of Beta interval masses plus scaled Huber mean term", "delta_u": 0.027777499999999997, "delta_u_source": "half the minimum positive spacing of nonzero absolute labels in fit only", "lambda_y": 1.0, "lambda_distill": 0.1, "lambda_reconstruction": 0.05, "lambda_group_default": 0.1, "group_temperature_default": 0.1, "selected_group_risk": [ 0.1, 0.05 ], "distill_temperature": 2.0, "distill_retention_exponent": 1.0, "imputer_regularization": { "emission_l2": 0.0001, "transition_l2": 0.0001 }, "group_and_distill_separate": true }, "calibration": { "method": "temperature scaling on a group-disjoint internal official-train holdout, separated from reliability selection", "temperature": 1.212728800581531, "valid_used_for_selection": true, "test_used_for_selection_or_calibration": false }, "selected_model": "C5", "attachment3_low_information_priors": { "class_probability_method": "fit counts + one pseudocount per class", "class_probability_values": [ 0.2786020441806792, 0.2205736894164194, 0.5008242664029015 ], "negative_beta": [ 1.1441766023635864, 1.8200817108154297 ], "positive_beta": [ 1.271026611328125, 2.4681642055511475 ] }, "ablation_definitions": { "C1": "masked BiGRU; no posterior imputation, explicit reliability or source gate", "C2": "exact Gaussian posterior mean; no joint trajectory integral", "C3": "joint trajectory integral plus final reliability/content fusion gate", "C4": "C3 plus bounded cross-time source attention and null source", "C5": "C4 plus reliability-modulated BiGRU update", "C6": "C5 plus optional rank-4 CP residual", "C6_no_distance": "C6 with uncertainty retained but both distance/span reliability penalties fixed to zero", "C6_no_reconstruction": "C6 trained without the auxiliary hidden-feature reconstruction loss", "C6_pointmask": "C6 trained with independent point masking instead of contiguous spans", "C7_distill": "C6 plus entropy/retention-weighted teacher distillation only", "C7_group": "C6 plus smooth worst-group risk only" }, "masking": { "rates": [ 0.0, 0.1, 0.3, 0.5, 0.7 ], "patterns": [ "single", "sync", "partial", "async" ], "preserve_at_least_fraction_per_selected_modality": 0.2, "controlled_sweep_split": "official validation", "identical_masks_across_models": true, "scenario_count": 42, "scenario_seed": 20261833, "training_mask_rng_seed": 20261227, "reliability_scenario_seed": 20261830, "controlled_torch_sampling_seed": 20261476, "paired_control_bootstrap_seed": 20261477, "mask_audit_file": "controlled_mask_audit.csv", "additional_one_factor_controls": [ "modality T/A/V and combinations", "start/middle/end", "one-long/multiple-short", "sync/partial/async" ], "semantic_position_control": "not run: aligned_50 does not provide audited semantic boundary indices; raw text is prohibited in student inputs" }, "final_test_metrics": { "n": 727, "accuracy": 0.6740027510316369, "macro_f1": 0.5861296508417881, "negative_support": 207, "neutral_support": 158, "positive_support": 362, "negative_recall": 0.7439613526570048, "middle_recall": 0.2088607594936709, "positive_recall": 0.8370165745856354, "regression_mae": 0.6979971528053284, "regression_rmse": 0.9674004106251411, "pearson": 0.6300334334373474, "brier": 0.44093362507172046, "classification_nll": 0.7613825798034668, "ece_15": 0.04994105640926912, "selection_nll": 2.868818521499634, "interval_90_coverage": 0.8968363136176066, "interval_90_mean_width": 2.4823575019836426, "predictive_variance_mean_uncalibrated": 0.601151168346405, "within_trajectory_variance_mean": 0.601142168045044, "between_trajectory_variance_mean": 8.998179509944748e-06, "predictive_mean_mean_calibrated": 0.12094182521104813, "predictive_variance_mean_calibrated": 0.6566913723945618, "selected_model": "C5", "temperature": 1.212728800581531 }, "attachment3_cases": 30, "attachment3_labeled_metrics": null, "completed_utc": "2026-09-24T20:01:49Z", "attachment3_prediction_file": "attachment3_predictions.csv", "attachment3_audit_file": "attachment3_audit.csv", "quality_flags": { "text": "unavailable; q*=1 fallback for visible rows, unknown flag retained", "audio": "unavailable; q*=1 fallback for visible rows, unknown flag retained", "vision": "unavailable; q*=1 fallback for visible rows, unknown flag retained" }, "neutral_output": "exact zero when neutral is the predicted class; no near-zero threshold", "diagnostic_figures": [ "q2_diagnostics.png", "q2_gate_positions.png" ] }