Prepare minimum submission bundle
This commit is contained in:
Binary file not shown.
Binary file not shown.
Binary file not shown.
@@ -0,0 +1,422 @@
|
||||
{
|
||||
"seed": 20260924,
|
||||
"text_encoder": "official precomputed text field; encoder revision not supplied",
|
||||
"training_configuration": {
|
||||
"student_epoch_limit": 12,
|
||||
"imputer_epochs": 8,
|
||||
"batch_size": 128,
|
||||
"early_stopping_patience": 3,
|
||||
"device": "cuda",
|
||||
"device_name": "NVIDIA GeForce RTX 5070 Ti",
|
||||
"optimizer": "AdamW",
|
||||
"student_learning_rate": 0.0003,
|
||||
"student_weight_decay": 0.001,
|
||||
"imputer_learning_rate": 0.0003,
|
||||
"imputer_weight_decay": 0.0001,
|
||||
"early_stopping_metric": "mean untempered selection_nll over fixed group-disjoint internal training scenarios",
|
||||
"inner_selection_scenarios": [
|
||||
"0.0/natural",
|
||||
"0.3/single",
|
||||
"0.3/sync",
|
||||
"0.5/async"
|
||||
],
|
||||
"inner_selection_source_video_groups": 76
|
||||
},
|
||||
"training_input": "E题数据/附件2-数据集特征文件/unaligned_50.pkl",
|
||||
"input_version": "unaligned_50",
|
||||
"q1_alignment_adapter": {
|
||||
"train": {
|
||||
"method": "shared_interval_overlap_on_normalized_progress",
|
||||
"coordinate_mode": "relative",
|
||||
"physical_time_alignment": false,
|
||||
"samples": 3395,
|
||||
"vision_length_conflict_samples": 618,
|
||||
"vision_tail_ambiguous_samples": 618,
|
||||
"nonzero_text_rows_outside_attention": 86078,
|
||||
"observed_target_rows": {
|
||||
"text": 169750,
|
||||
"audio": 169750,
|
||||
"vision": 163302
|
||||
},
|
||||
"mean_target_coverage": {
|
||||
"text": 1.0,
|
||||
"audio": 1.0,
|
||||
"vision": 0.9540337701314328
|
||||
},
|
||||
"quality_fields_available": false,
|
||||
"word_or_frame_timestamps_available": false
|
||||
},
|
||||
"valid": {
|
||||
"method": "shared_interval_overlap_on_normalized_progress",
|
||||
"coordinate_mode": "relative",
|
||||
"physical_time_alignment": false,
|
||||
"samples": 728,
|
||||
"vision_length_conflict_samples": 141,
|
||||
"vision_tail_ambiguous_samples": 141,
|
||||
"nonzero_text_rows_outside_attention": 17772,
|
||||
"observed_target_rows": {
|
||||
"text": 36400,
|
||||
"audio": 36400,
|
||||
"vision": 35315
|
||||
},
|
||||
"mean_target_coverage": {
|
||||
"text": 1.0,
|
||||
"audio": 1.0,
|
||||
"vision": 0.96090314748523
|
||||
},
|
||||
"quality_fields_available": false,
|
||||
"word_or_frame_timestamps_available": false
|
||||
},
|
||||
"test": {
|
||||
"method": "shared_interval_overlap_on_normalized_progress",
|
||||
"coordinate_mode": "relative",
|
||||
"physical_time_alignment": false,
|
||||
"samples": 727,
|
||||
"vision_length_conflict_samples": 131,
|
||||
"vision_tail_ambiguous_samples": 131,
|
||||
"nonzero_text_rows_outside_attention": 18041,
|
||||
"observed_target_rows": {
|
||||
"text": 36350,
|
||||
"audio": 36350,
|
||||
"vision": 35034
|
||||
},
|
||||
"mean_target_coverage": {
|
||||
"text": 1.0,
|
||||
"audio": 1.0,
|
||||
"vision": 0.9549938172651288
|
||||
},
|
||||
"quality_fields_available": false,
|
||||
"word_or_frame_timestamps_available": false
|
||||
}
|
||||
},
|
||||
"training_sha256": "77eda14a06be9749a96c52ae45470c7cffcfa7219011eae391d231a0664c3762",
|
||||
"official_group_overlap": {
|
||||
"train_valid": 0,
|
||||
"train_test": 0,
|
||||
"valid_test": 0
|
||||
},
|
||||
"official_splits": {
|
||||
"train": {
|
||||
"n": 3395,
|
||||
"source_video_groups": 1528
|
||||
},
|
||||
"valid": {
|
||||
"n": 728,
|
||||
"source_video_groups": 239
|
||||
},
|
||||
"test": {
|
||||
"n": 727,
|
||||
"source_video_groups": 381
|
||||
}
|
||||
},
|
||||
"internal_train_holdouts": {
|
||||
"fit": {
|
||||
"n": 3030,
|
||||
"video_groups": 1375
|
||||
},
|
||||
"reliability_selection": {
|
||||
"n": 190,
|
||||
"video_groups": 76
|
||||
},
|
||||
"temperature_calibration": {
|
||||
"n": 175,
|
||||
"video_groups": 77
|
||||
},
|
||||
"all_group_disjoint": true
|
||||
},
|
||||
"feature_standardization": "fit-only observed rows, per-dimension; fixed for valid/test/attachment3",
|
||||
"missing_mask": "official text attention and source lengths plus row observation; normalized-progress overlap preserves empty bins; q*=1 only where visible, J_Q=0",
|
||||
"observation_quality": {
|
||||
"quality_score_fields_present": false,
|
||||
"quality_available_flag_present": false,
|
||||
"fallback": "q*=1 and J_Q=0 for visible rows; R_eff=R",
|
||||
"quality_noise_mapping_ablation": "not identifiable on unaligned_50 because no row quality score varies"
|
||||
},
|
||||
"imputer": {
|
||||
"type": "structured linear Gaussian shared-private state space",
|
||||
"state_dims": {
|
||||
"shared": 8,
|
||||
"private_each": 4
|
||||
},
|
||||
"posterior": "block-tridiagonal equivalent Kalman information filter + RTS smoother",
|
||||
"sampling": "joint latent trajectories and missing emissions; observed features copied exactly",
|
||||
"fit_objective": "train-only observed Gaussian marginal likelihood including log determinants",
|
||||
"epochs": 8,
|
||||
"frozen_before_teacher_student": true
|
||||
},
|
||||
"architecture": {
|
||||
"projection": 32,
|
||||
"bigru_hidden_each_direction": 16,
|
||||
"cross_source_layers": 1,
|
||||
"cross_time_read": true,
|
||||
"rank": 4,
|
||||
"reliability_gru": "directional hidden decay; reset applied before candidate map; update gate multiplied by rho",
|
||||
"final_gate": "rho times bounded content score plus positive null prior",
|
||||
"output": "neutral point mass plus sign-specific Beta magnitudes; K-path probabilities mixed before decoding"
|
||||
},
|
||||
"reliability_hyperparameters": {
|
||||
"selected_per_model_on": "group-disjoint internal training reliability-validation slice",
|
||||
"candidate_values": [
|
||||
[
|
||||
0.5,
|
||||
0.0,
|
||||
0.0,
|
||||
0.0
|
||||
],
|
||||
[
|
||||
0.5,
|
||||
0.05,
|
||||
0.05,
|
||||
0.05
|
||||
],
|
||||
[
|
||||
0.5,
|
||||
0.1,
|
||||
0.0,
|
||||
0.0
|
||||
],
|
||||
[
|
||||
0.3,
|
||||
0.05,
|
||||
0.05,
|
||||
0.05
|
||||
],
|
||||
[
|
||||
0.7,
|
||||
0.05,
|
||||
0.05,
|
||||
0.05
|
||||
]
|
||||
],
|
||||
"validation_scenarios": [
|
||||
"0.0/natural",
|
||||
"0.3/single",
|
||||
"0.3/sync",
|
||||
"0.5/async"
|
||||
],
|
||||
"selected_by_model": {
|
||||
"teacher": [
|
||||
0.3,
|
||||
0.05,
|
||||
0.05,
|
||||
0.05
|
||||
],
|
||||
"C1": [
|
||||
0.5,
|
||||
0.05,
|
||||
0.05,
|
||||
0.05
|
||||
],
|
||||
"C2": [
|
||||
0.5,
|
||||
0.05,
|
||||
0.05,
|
||||
0.05
|
||||
],
|
||||
"C3": [
|
||||
0.5,
|
||||
0.0,
|
||||
0.0,
|
||||
0.0
|
||||
],
|
||||
"C4": [
|
||||
0.7,
|
||||
0.05,
|
||||
0.05,
|
||||
0.05
|
||||
],
|
||||
"C5": [
|
||||
0.3,
|
||||
0.05,
|
||||
0.05,
|
||||
0.05
|
||||
],
|
||||
"C6": [
|
||||
0.3,
|
||||
0.05,
|
||||
0.05,
|
||||
0.05
|
||||
],
|
||||
"C6_no_distance": [
|
||||
0.5,
|
||||
0.0,
|
||||
0.0,
|
||||
0.0
|
||||
],
|
||||
"C6_no_reconstruction": [
|
||||
0.3,
|
||||
0.05,
|
||||
0.05,
|
||||
0.05
|
||||
],
|
||||
"C6_pointmask": [
|
||||
0.3,
|
||||
0.05,
|
||||
0.05,
|
||||
0.05
|
||||
],
|
||||
"C7_distill": [
|
||||
0.5,
|
||||
0.05,
|
||||
0.05,
|
||||
0.05
|
||||
],
|
||||
"C7_group": [
|
||||
0.5,
|
||||
0.0,
|
||||
0.0,
|
||||
0.0
|
||||
]
|
||||
}
|
||||
},
|
||||
"group_risk_hyperparameters": {
|
||||
"selection": "lambda_group and group_temperature jointly selected with reliability hyperparameters on fixed group-disjoint internal training scenarios",
|
||||
"candidate_values": [
|
||||
[
|
||||
0.05,
|
||||
0.1
|
||||
],
|
||||
[
|
||||
0.1,
|
||||
0.05
|
||||
],
|
||||
[
|
||||
0.1,
|
||||
0.1
|
||||
],
|
||||
[
|
||||
0.1,
|
||||
0.2
|
||||
],
|
||||
[
|
||||
0.2,
|
||||
0.1
|
||||
]
|
||||
],
|
||||
"selected": [
|
||||
0.1,
|
||||
0.1
|
||||
],
|
||||
"selection_split": "reliability_validation"
|
||||
},
|
||||
"loss": {
|
||||
"supervision": "negative log mixture of Beta interval masses plus scaled Huber mean term",
|
||||
"delta_u": 0.027777499999999997,
|
||||
"delta_u_source": "half the minimum positive spacing of nonzero absolute labels in fit only",
|
||||
"lambda_y": 1.0,
|
||||
"lambda_distill": 0.1,
|
||||
"lambda_reconstruction": 0.05,
|
||||
"lambda_group_default": 0.1,
|
||||
"group_temperature_default": 0.1,
|
||||
"selected_group_risk": [
|
||||
0.1,
|
||||
0.1
|
||||
],
|
||||
"distill_temperature": 2.0,
|
||||
"distill_retention_exponent": 1.0,
|
||||
"imputer_regularization": {
|
||||
"emission_l2": 0.0001,
|
||||
"transition_l2": 0.0001
|
||||
},
|
||||
"group_and_distill_separate": true
|
||||
},
|
||||
"calibration": {
|
||||
"method": "temperature scaling on a group-disjoint internal official-train holdout, separated from reliability selection",
|
||||
"temperature": 1.122980387832455,
|
||||
"valid_used_for_selection": true,
|
||||
"test_used_for_selection_or_calibration": false
|
||||
},
|
||||
"selected_model": "C6",
|
||||
"attachment3_low_information_priors": {
|
||||
"class_probability_method": "fit counts + one pseudocount per class",
|
||||
"class_probability_values": [
|
||||
0.2786020441806792,
|
||||
0.2205736894164194,
|
||||
0.5008242664029015
|
||||
],
|
||||
"negative_beta": [
|
||||
1.1441766023635864,
|
||||
1.8200817108154297
|
||||
],
|
||||
"positive_beta": [
|
||||
1.271026611328125,
|
||||
2.4681642055511475
|
||||
]
|
||||
},
|
||||
"ablation_definitions": {
|
||||
"C1": "masked BiGRU; no posterior imputation, explicit reliability or source gate",
|
||||
"C2": "exact Gaussian posterior mean; no joint trajectory integral",
|
||||
"C3": "joint trajectory integral plus final reliability/content fusion gate",
|
||||
"C4": "C3 plus bounded cross-time source attention and null source",
|
||||
"C5": "C4 plus reliability-modulated BiGRU update",
|
||||
"C6": "C5 plus optional rank-4 CP residual",
|
||||
"C6_no_distance": "C6 with uncertainty retained but both distance/span reliability penalties fixed to zero",
|
||||
"C6_no_reconstruction": "C6 trained without the auxiliary hidden-feature reconstruction loss",
|
||||
"C6_pointmask": "C6 trained with independent point masking instead of contiguous spans",
|
||||
"C7_distill": "C6 plus entropy/retention-weighted teacher distillation only",
|
||||
"C7_group": "C6 plus smooth worst-group risk only"
|
||||
},
|
||||
"masking": {
|
||||
"rates": [
|
||||
0.0,
|
||||
0.1,
|
||||
0.3,
|
||||
0.5,
|
||||
0.7
|
||||
],
|
||||
"patterns": [
|
||||
"single",
|
||||
"sync",
|
||||
"partial",
|
||||
"async"
|
||||
],
|
||||
"preserve_at_least_fraction_per_selected_modality": 0.2,
|
||||
"controlled_sweep_split": "official validation",
|
||||
"identical_masks_across_models": true,
|
||||
"scenario_count": 42,
|
||||
"scenario_seed": 20261833,
|
||||
"training_mask_rng_seed": 20261227,
|
||||
"reliability_scenario_seed": 20261830,
|
||||
"controlled_torch_sampling_seed": 20261476,
|
||||
"paired_control_bootstrap_seed": 20261477,
|
||||
"mask_audit_file": null,
|
||||
"mask_audit_omitted_reason": "Row-level audit omitted from the size-limited deliverable; regenerated by rerunning final.q2.math.train.",
|
||||
"additional_one_factor_controls": [
|
||||
"modality T/A/V and combinations",
|
||||
"start/middle/end",
|
||||
"one-long/multiple-short",
|
||||
"sync/partial/async"
|
||||
],
|
||||
"semantic_position_control": "not run: unaligned_50 does not provide audited semantic boundary indices; raw text is prohibited in student inputs"
|
||||
},
|
||||
"final_test_metrics": {
|
||||
"n": 727,
|
||||
"accuracy": 0.672627235213205,
|
||||
"macro_f1": 0.5484581573154621,
|
||||
"negative_support": 207,
|
||||
"neutral_support": 158,
|
||||
"positive_support": 362,
|
||||
"negative_recall": 0.7439613526570048,
|
||||
"middle_recall": 0.10126582278481013,
|
||||
"positive_recall": 0.8812154696132597,
|
||||
"regression_mae": 0.7100059986114502,
|
||||
"regression_rmse": 0.969292458045841,
|
||||
"pearson": 0.6424147486686707,
|
||||
"brier": 0.4367243729993746,
|
||||
"classification_nll": 0.7538501024246216,
|
||||
"ece_15": 0.047142177615237854,
|
||||
"selection_nll": 2.8449835777282715,
|
||||
"interval_90_coverage": 0.8954607977991746,
|
||||
"interval_90_mean_width": 2.48697829246521,
|
||||
"predictive_variance_mean_uncalibrated": 0.6001424193382263,
|
||||
"within_trajectory_variance_mean": 0.6001414060592651,
|
||||
"between_trajectory_variance_mean": 1.021712705551181e-06,
|
||||
"predictive_mean_mean_calibrated": 0.18709982931613922,
|
||||
"predictive_variance_mean_calibrated": 0.6344733238220215
|
||||
},
|
||||
"test_gate_diagnostics_file": null,
|
||||
"test_gate_diagnostics_omitted_reason": "Row-level audit omitted from the size-limited deliverable; regenerated by rerunning final.q2.math.train.",
|
||||
"attachment3_cases": 0,
|
||||
"attachment3_labeled_metrics": null,
|
||||
"completed_utc": "2026-09-25T12:10:16Z"
|
||||
}
|
||||
Binary file not shown.
@@ -0,0 +1,27 @@
|
||||
{
|
||||
"n": 728,
|
||||
"accuracy": 0.6085164835164835,
|
||||
"macro_f1": 0.518781225964892,
|
||||
"negative_support": 206,
|
||||
"neutral_support": 184,
|
||||
"positive_support": 338,
|
||||
"negative_recall": 0.6796116504854369,
|
||||
"middle_recall": 0.125,
|
||||
"positive_recall": 0.8284023668639053,
|
||||
"regression_mae": 0.6846789717674255,
|
||||
"regression_rmse": 0.9208215740080159,
|
||||
"pearson": 0.6101368069648743,
|
||||
"brier": 0.4977383080922297,
|
||||
"classification_nll": 0.8427478075027466,
|
||||
"ece_15": 0.03950894476620705,
|
||||
"selection_nll": 2.783693552017212,
|
||||
"interval_90_coverage": 0.8873626373626373,
|
||||
"interval_90_mean_width": 2.3910491466522217,
|
||||
"predictive_variance_mean_uncalibrated": 0.5537729859352112,
|
||||
"within_trajectory_variance_mean": 0.5537727475166321,
|
||||
"between_trajectory_variance_mean": 2.9140662149984564e-07,
|
||||
"predictive_mean_mean_calibrated": 0.2094990462064743,
|
||||
"predictive_variance_mean_calibrated": 0.5816943049430847,
|
||||
"selected_model": "C6",
|
||||
"temperature": 1.122980387832455
|
||||
}
|
||||
@@ -0,0 +1,37 @@
|
||||
{
|
||||
"selected_method": "A0",
|
||||
"provisional_seed42_method": "A2",
|
||||
"candidate_methods_with_three_seeds": [
|
||||
"A2",
|
||||
"A0",
|
||||
"A1"
|
||||
],
|
||||
"selection_rule": "lowest mean fixed four-scenario validation task loss across seeds 42, 3407, 2026",
|
||||
"candidate_summary": [
|
||||
{
|
||||
"method": "A0",
|
||||
"seed_losses": "[0.8643234267339601, 0.8756466648735842, 0.8574190991265433]",
|
||||
"mean_validation_selection_loss": 0.8657963969113626,
|
||||
"std_validation_selection_loss": 0.009202622948013908,
|
||||
"seeds": 3,
|
||||
"validation_only_selection": true
|
||||
},
|
||||
{
|
||||
"method": "A1",
|
||||
"seed_losses": "[0.8649765662439577, 0.8810990981675766, 0.8557776766163963]",
|
||||
"mean_validation_selection_loss": 0.8672844470093102,
|
||||
"std_validation_selection_loss": 0.012817501026466038,
|
||||
"seeds": 3,
|
||||
"validation_only_selection": true
|
||||
},
|
||||
{
|
||||
"method": "A2",
|
||||
"seed_losses": "[0.8634063961741689, 0.880271397449158, 0.862087192279952]",
|
||||
"mean_validation_selection_loss": 0.8685883286344263,
|
||||
"std_validation_selection_loss": 0.010139311979910028,
|
||||
"seeds": 3,
|
||||
"validation_only_selection": true
|
||||
}
|
||||
],
|
||||
"attachment4_labels_used": false
|
||||
}
|
||||
Binary file not shown.
@@ -0,0 +1,42 @@
|
||||
{
|
||||
"method": "A0",
|
||||
"seed": 2026,
|
||||
"best_epoch": 4,
|
||||
"best_selection_loss": 0.8574190991265433,
|
||||
"batch_size": 64,
|
||||
"epoch_limit": 12,
|
||||
"patience": 3,
|
||||
"optimizer": "AdamW",
|
||||
"learning_rate": 0.0003,
|
||||
"weight_decay": 0.001,
|
||||
"gradient_clip_norm": 1.0,
|
||||
"training_mask_rates": [
|
||||
0.0,
|
||||
0.1,
|
||||
0.3,
|
||||
0.5,
|
||||
0.7
|
||||
],
|
||||
"training_mask_patterns": [
|
||||
"single",
|
||||
"sync",
|
||||
"partial",
|
||||
"async"
|
||||
],
|
||||
"training_mask_seed_base": 20261227,
|
||||
"same_orders_and_masks_across_methods_for_same_seed": true,
|
||||
"config": {
|
||||
"name": "A0_main_effects",
|
||||
"low_rank": false,
|
||||
"cross_attention": false,
|
||||
"anchored": true,
|
||||
"lambda_interaction": 0.001,
|
||||
"lambda_mask": 0.0,
|
||||
"rank": 4,
|
||||
"hidden": 64,
|
||||
"gru_hidden_per_direction": 32,
|
||||
"attention_heads": 4,
|
||||
"attention_ffn": 128,
|
||||
"eta_init": 0.1
|
||||
}
|
||||
}
|
||||
Binary file not shown.
@@ -0,0 +1,42 @@
|
||||
{
|
||||
"method": "A0",
|
||||
"seed": 3407,
|
||||
"best_epoch": 3,
|
||||
"best_selection_loss": 0.8756466648735842,
|
||||
"batch_size": 64,
|
||||
"epoch_limit": 12,
|
||||
"patience": 3,
|
||||
"optimizer": "AdamW",
|
||||
"learning_rate": 0.0003,
|
||||
"weight_decay": 0.001,
|
||||
"gradient_clip_norm": 1.0,
|
||||
"training_mask_rates": [
|
||||
0.0,
|
||||
0.1,
|
||||
0.3,
|
||||
0.5,
|
||||
0.7
|
||||
],
|
||||
"training_mask_patterns": [
|
||||
"single",
|
||||
"sync",
|
||||
"partial",
|
||||
"async"
|
||||
],
|
||||
"training_mask_seed_base": 20261227,
|
||||
"same_orders_and_masks_across_methods_for_same_seed": true,
|
||||
"config": {
|
||||
"name": "A0_main_effects",
|
||||
"low_rank": false,
|
||||
"cross_attention": false,
|
||||
"anchored": true,
|
||||
"lambda_interaction": 0.001,
|
||||
"lambda_mask": 0.0,
|
||||
"rank": 4,
|
||||
"hidden": 64,
|
||||
"gru_hidden_per_direction": 32,
|
||||
"attention_heads": 4,
|
||||
"attention_ffn": 128,
|
||||
"eta_init": 0.1
|
||||
}
|
||||
}
|
||||
Binary file not shown.
@@ -0,0 +1,42 @@
|
||||
{
|
||||
"method": "A0",
|
||||
"seed": 42,
|
||||
"best_epoch": 4,
|
||||
"best_selection_loss": 0.8643234267339601,
|
||||
"batch_size": 64,
|
||||
"epoch_limit": 12,
|
||||
"patience": 3,
|
||||
"optimizer": "AdamW",
|
||||
"learning_rate": 0.0003,
|
||||
"weight_decay": 0.001,
|
||||
"gradient_clip_norm": 1.0,
|
||||
"training_mask_rates": [
|
||||
0.0,
|
||||
0.1,
|
||||
0.3,
|
||||
0.5,
|
||||
0.7
|
||||
],
|
||||
"training_mask_patterns": [
|
||||
"single",
|
||||
"sync",
|
||||
"partial",
|
||||
"async"
|
||||
],
|
||||
"training_mask_seed_base": 20261227,
|
||||
"same_orders_and_masks_across_methods_for_same_seed": true,
|
||||
"config": {
|
||||
"name": "A0_main_effects",
|
||||
"low_rank": false,
|
||||
"cross_attention": false,
|
||||
"anchored": true,
|
||||
"lambda_interaction": 0.001,
|
||||
"lambda_mask": 0.0,
|
||||
"rank": 4,
|
||||
"hidden": 64,
|
||||
"gru_hidden_per_direction": 32,
|
||||
"attention_heads": 4,
|
||||
"attention_ffn": 128,
|
||||
"eta_init": 0.1
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,119 @@
|
||||
{
|
||||
"experiment": "ATI\u2013HO Q3 staged training and structural attribution audit",
|
||||
"created_utc": "2026-09-26T07:33:41Z",
|
||||
"device": "cuda",
|
||||
"torch_version": "2.14.0+cu130",
|
||||
"cuda_available": true,
|
||||
"cuda_version": "13.0",
|
||||
"gpu": "NVIDIA GeForce RTX 5070 Ti",
|
||||
"python": "3.14.7",
|
||||
"seeds": [
|
||||
42,
|
||||
3407,
|
||||
2026
|
||||
],
|
||||
"epochs_max": 12,
|
||||
"training_protocol": {
|
||||
"batch_size": 64,
|
||||
"early_stopping_patience": 3,
|
||||
"optimizer": "AdamW",
|
||||
"learning_rate": 0.0003,
|
||||
"weight_decay": 0.001,
|
||||
"gradient_clip_norm": 1.0,
|
||||
"training_mask_rates": [
|
||||
0.0,
|
||||
0.1,
|
||||
0.3,
|
||||
0.5,
|
||||
0.7
|
||||
],
|
||||
"training_mask_patterns": [
|
||||
"single",
|
||||
"sync",
|
||||
"partial",
|
||||
"async"
|
||||
],
|
||||
"validation_selection_scenarios": [
|
||||
"0.0/none",
|
||||
"0.3/single",
|
||||
"0.3/sync",
|
||||
"0.5/async"
|
||||
],
|
||||
"held_out_attachment4_touched_during_training": false
|
||||
},
|
||||
"ati_output": {
|
||||
"parameter_vector": "3 centered class logits + r_negative + r_positive",
|
||||
"intensity": "negative/positive magnitudes are 3*sigmoid(r); neutral class is exactly zero",
|
||||
"loss": "cross entropy + conditional magnitude SmoothL1 + 0.2*Huber(delta=0.25) + configured regularizers",
|
||||
"baseline_checkpoint_reuse": "No: retrain B0 and B1 on the fixed ATI split/mask schedule because existing Q2 checkpoints differ in seeds, batch size, and schedule.",
|
||||
"calibration_temperature": 1.0
|
||||
},
|
||||
"data": {
|
||||
"feature_file": "${FINAL_DATA_DIR}/attachment2/unaligned_50.pkl",
|
||||
"feature_sha256": "77eda14a06be9749a96c52ae45470c7cffcfa7219011eae391d231a0664c3762",
|
||||
"scaler_file": "final/experiments/q2/unaligned_deep_two_b128/unaligned_50_robust_stats.npz",
|
||||
"scaler_max_abs_difference_from_train_only_recompute": 0.0,
|
||||
"representation": "Q1 adapter Relative-Progress projection; 50 slots; not physical-time alignment",
|
||||
"adapter": "final.adapter.adapt_official_split; shared train-only robust scaler retained from Q2 V2",
|
||||
"dimensions": [
|
||||
768,
|
||||
74,
|
||||
35
|
||||
],
|
||||
"train_samples": 3395,
|
||||
"valid_samples": 728,
|
||||
"train_source_video_groups": 1528,
|
||||
"valid_source_video_groups": 239,
|
||||
"test_samples": 727,
|
||||
"test_source_video_groups": 381,
|
||||
"source_video_overlap_counts": {
|
||||
"train/valid": 0,
|
||||
"train/test": 0,
|
||||
"valid/test": 0
|
||||
},
|
||||
"adapter_audit": {
|
||||
"train": {
|
||||
"method": "shared_interval_overlap_on_normalized_progress",
|
||||
"coordinate_mode": "relative",
|
||||
"physical_time_alignment": false,
|
||||
"samples": 3395,
|
||||
"vision_length_conflict_samples": 618,
|
||||
"vision_tail_ambiguous_samples": 618,
|
||||
"nonzero_text_rows_outside_attention": 86078,
|
||||
"observed_target_rows": {
|
||||
"text": 169750,
|
||||
"audio": 169750,
|
||||
"vision": 163302
|
||||
},
|
||||
"mean_target_coverage": {
|
||||
"text": 1.0,
|
||||
"audio": 1.0,
|
||||
"vision": 0.9540337701314328
|
||||
},
|
||||
"quality_fields_available": false,
|
||||
"word_or_frame_timestamps_available": false
|
||||
},
|
||||
"valid": {
|
||||
"method": "shared_interval_overlap_on_normalized_progress",
|
||||
"coordinate_mode": "relative",
|
||||
"physical_time_alignment": false,
|
||||
"samples": 728,
|
||||
"vision_length_conflict_samples": 141,
|
||||
"vision_tail_ambiguous_samples": 141,
|
||||
"nonzero_text_rows_outside_attention": 17772,
|
||||
"observed_target_rows": {
|
||||
"text": 36400,
|
||||
"audio": 36400,
|
||||
"vision": 35315
|
||||
},
|
||||
"mean_target_coverage": {
|
||||
"text": 1.0,
|
||||
"audio": 1.0,
|
||||
"vision": 0.96090314748523
|
||||
},
|
||||
"quality_fields_available": false,
|
||||
"word_or_frame_timestamps_available": false
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user