119 lines
3.8 KiB
JSON
119 lines
3.8 KiB
JSON
{
|
||
"experiment": "ATI–HO Q3 staged training and structural attribution audit",
|
||
"created_utc": "2026-09-26T07:33:41Z",
|
||
"device": "cuda",
|
||
"torch_version": "2.14.0+cu130",
|
||
"cuda_available": true,
|
||
"cuda_version": "13.0",
|
||
"gpu": "NVIDIA GeForce RTX 5070 Ti",
|
||
"python": "3.14.7",
|
||
"seeds": [
|
||
42,
|
||
3407,
|
||
2026
|
||
],
|
||
"epochs_max": 12,
|
||
"training_protocol": {
|
||
"batch_size": 64,
|
||
"early_stopping_patience": 3,
|
||
"optimizer": "AdamW",
|
||
"learning_rate": 0.0003,
|
||
"weight_decay": 0.001,
|
||
"gradient_clip_norm": 1.0,
|
||
"training_mask_rates": [
|
||
0.0,
|
||
0.1,
|
||
0.3,
|
||
0.5,
|
||
0.7
|
||
],
|
||
"training_mask_patterns": [
|
||
"single",
|
||
"sync",
|
||
"partial",
|
||
"async"
|
||
],
|
||
"validation_selection_scenarios": [
|
||
"0.0/none",
|
||
"0.3/single",
|
||
"0.3/sync",
|
||
"0.5/async"
|
||
],
|
||
"held_out_attachment4_touched_during_training": false
|
||
},
|
||
"ati_output": {
|
||
"parameter_vector": "3 centered class logits + r_negative + r_positive",
|
||
"intensity": "negative/positive magnitudes are 3*sigmoid(r); neutral class is exactly zero",
|
||
"loss": "cross entropy + conditional magnitude SmoothL1 + 0.2*Huber(delta=0.25) + configured regularizers",
|
||
"baseline_checkpoint_reuse": "No: retrain B0 and B1 on the fixed ATI split/mask schedule because existing Q2 checkpoints differ in seeds, batch size, and schedule.",
|
||
"calibration_temperature": 1.0
|
||
},
|
||
"data": {
|
||
"feature_file": "/home/gloamxun/modeling_zhaocui/E题数据/附件2-数据集特征文件/unaligned_50.pkl",
|
||
"feature_sha256": "77eda14a06be9749a96c52ae45470c7cffcfa7219011eae391d231a0664c3762",
|
||
"scaler_file": "/home/gloamxun/modeling_zhaocui/final/experiments/q2/unaligned_deep_two_b128/unaligned_50_robust_stats.npz",
|
||
"scaler_max_abs_difference_from_train_only_recompute": 0.0,
|
||
"representation": "Q1 adapter Relative-Progress projection; 50 slots; not physical-time alignment",
|
||
"adapter": "final.adapter.adapt_official_split; shared train-only robust scaler retained from Q2 V2",
|
||
"dimensions": [
|
||
768,
|
||
74,
|
||
35
|
||
],
|
||
"train_samples": 3395,
|
||
"valid_samples": 728,
|
||
"train_source_video_groups": 1528,
|
||
"valid_source_video_groups": 239,
|
||
"test_samples": 727,
|
||
"test_source_video_groups": 381,
|
||
"source_video_overlap_counts": {
|
||
"train/valid": 0,
|
||
"train/test": 0,
|
||
"valid/test": 0
|
||
},
|
||
"adapter_audit": {
|
||
"train": {
|
||
"method": "shared_interval_overlap_on_normalized_progress",
|
||
"coordinate_mode": "relative",
|
||
"physical_time_alignment": false,
|
||
"samples": 3395,
|
||
"vision_length_conflict_samples": 618,
|
||
"vision_tail_ambiguous_samples": 618,
|
||
"nonzero_text_rows_outside_attention": 86078,
|
||
"observed_target_rows": {
|
||
"text": 169750,
|
||
"audio": 169750,
|
||
"vision": 163302
|
||
},
|
||
"mean_target_coverage": {
|
||
"text": 1.0,
|
||
"audio": 1.0,
|
||
"vision": 0.9540337701314328
|
||
},
|
||
"quality_fields_available": false,
|
||
"word_or_frame_timestamps_available": false
|
||
},
|
||
"valid": {
|
||
"method": "shared_interval_overlap_on_normalized_progress",
|
||
"coordinate_mode": "relative",
|
||
"physical_time_alignment": false,
|
||
"samples": 728,
|
||
"vision_length_conflict_samples": 141,
|
||
"vision_tail_ambiguous_samples": 141,
|
||
"nonzero_text_rows_outside_attention": 17772,
|
||
"observed_target_rows": {
|
||
"text": 36400,
|
||
"audio": 36400,
|
||
"vision": 35315
|
||
},
|
||
"mean_target_coverage": {
|
||
"text": 1.0,
|
||
"audio": 1.0,
|
||
"vision": 0.96090314748523
|
||
},
|
||
"quality_fields_available": false,
|
||
"word_or_frame_timestamps_available": false
|
||
}
|
||
}
|
||
}
|
||
} |