Files
modeling_zhaocui/deep_learning/Q2/q2/evaluate_math_protocol.py
T

615 lines
29 KiB
Python

"""Score the frozen EarlyConcat and MoFE checkpoints using the math-Q2 protocol.
This script performs no training and selects no models. It evaluates the saved
three-seed checkpoints on the official labeled test split once, and reuses the
fixed 42-scenario validation-mask audit as the controlled-missingness protocol.
All generated files stay under deep_learning/Q2/outputs/followups/.
"""
from __future__ import annotations
import csv
import argparse
import hashlib
import json
import math
import statistics
import time
from collections import defaultdict
from pathlib import Path
from typing import Any
import numpy as np
import torch
from sklearn.metrics import accuracy_score, f1_score, mean_absolute_error, mean_squared_error
from torch import nn
from .data import (
ATTACHMENT2,
MODALITIES,
RobustStats,
Split,
_ids_and_targets,
_text_mask,
_unpickle,
apply_robust_stats,
fit_robust_stats,
load_aligned,
)
from .models import AlignedFusionModel
from .mofe import MixtureOfFusionExperts
from .train_mofe import MODEL_CONFIG, _predict, _device_for, EARLYCONCAT, MOFE7_MLP
Q2_ROOT = Path(__file__).resolve().parents[1]
REPO_ROOT = Q2_ROOT.parents[1]
REFERENCE_DIR = Q2_ROOT / "outputs" / "followups" / "R01_selected_model_reevaluation"
OUTPUT_DIR = Q2_ROOT / "outputs" / "followups" / "R02_math_protocol_evaluation"
SEEDS = (42, 3407, 2026)
BOOTSTRAP_REPS = 1000
TEST_BOOTSTRAP_SEED = 20260925
AURC_BOOTSTRAP_SEED = 20260926
SCENARIO_SEED = 20261833
METHODS = (EARLYCONCAT, MOFE7_MLP)
CURVE_MODES = ("single", "sync", "partial", "async")
CURVE_RATES = (0.0, 0.1, 0.3, 0.5, 0.7)
def write_csv(path: Path, rows: list[dict[str, Any]]) -> None:
if not rows:
return
path.parent.mkdir(parents=True, exist_ok=True)
fields = list(dict.fromkeys(key for row in rows for key in row))
with path.open("w", newline="", encoding="utf-8-sig") as stream:
writer = csv.DictWriter(stream, fieldnames=fields)
writer.writeheader()
writer.writerows(rows)
def sha256(path: Path) -> str:
digest = hashlib.sha256()
with path.open("rb") as stream:
for block in iter(lambda: stream.read(1024 * 1024), b""):
digest.update(block)
return digest.hexdigest()
def split_from_part(part: dict[str, Any]) -> Split:
xs = tuple(np.asarray(part[name], dtype=np.float32) for name in MODALITIES)
masks = [
_text_mask(part),
np.any(np.isfinite(xs[1]) & (xs[1] != 0), axis=-1),
np.any(np.isfinite(xs[2]) & (xs[2] != 0), axis=-1),
]
ids, y_cls, y_reg = _ids_and_targets(part)
return Split(xs, np.stack(masks, axis=-1), y_cls, y_reg, ids)
def load_splits(feature_path: Path) -> dict[str, Split]:
raw = _unpickle(feature_path)
usual = load_aligned(feature_path)
splits = {"train": usual["train"], "valid": usual["valid"], "test": split_from_part(raw["test"])}
groups = {
name: {sample_id.split("$_$", 1)[0] for sample_id in split.ids}
for name, split in splits.items()
}
for first, second in (("train", "valid"), ("train", "test"), ("valid", "test")):
overlap = groups[first] & groups[second]
if overlap:
raise ValueError(f"official {first}/{second} source-video groups overlap: {len(overlap)}")
for name, split in splits.items():
expected = np.where(split.y_reg < 0, 0, np.where(split.y_reg == 0, 1, 2))
if not np.array_equal(expected, split.y_cls):
raise ValueError(f"{name}: classification labels disagree with strict sign of regression labels")
return splits
def metrics(split: Split, logits: np.ndarray, intensity: np.ndarray, indices: np.ndarray | None = None) -> dict[str, float]:
if indices is None:
indices = np.arange(split.n)
y_cls = split.y_cls[indices]
y_reg = split.y_reg[indices]
pred_cls = np.asarray(logits)[indices].argmax(axis=-1)
pred_reg = np.clip(np.asarray(intensity).reshape(-1)[indices], -3.0, 3.0)
return {
"accuracy": float(accuracy_score(y_cls, pred_cls)),
"macro_f1": float(f1_score(y_cls, pred_cls, labels=[0, 1, 2], average="macro", zero_division=0)),
"mae": float(mean_absolute_error(y_reg, pred_reg)),
"rmse": float(math.sqrt(mean_squared_error(y_reg, pred_reg))),
"pearson": float(np.corrcoef(y_reg, pred_reg)[0, 1]) if np.std(y_reg) > 0 and np.std(pred_reg) > 0 else float("nan"),
}
def load_model(method: str, seed: int, dims: tuple[int, int, int], device: torch.device) -> nn.Module:
if method == EARLYCONCAT:
checkpoint = REFERENCE_DIR / "models" / "baselines" / "concat" / f"seed_{seed}" / "model_best.pt"
model: nn.Module = AlignedFusionModel("concat", dims=dims).to(device)
state = torch.load(checkpoint, map_location=device, weights_only=False)
if state.get("kind") != "concat" or int(state.get("seed", -1)) != seed:
raise ValueError(f"unexpected EarlyConcat checkpoint: {checkpoint}")
elif method == MOFE7_MLP:
checkpoint = REFERENCE_DIR / "models" / MOFE7_MLP / f"seed_{seed}" / "model_best.pt"
model = MixtureOfFusionExperts(dims=dims, **MODEL_CONFIG).to(device)
state = torch.load(checkpoint, map_location=device, weights_only=False)
if state.get("config") != MODEL_CONFIG or int(state.get("seed", -1)) != seed:
raise ValueError(f"unexpected MoFE checkpoint: {checkpoint}")
else:
raise ValueError(f"unknown model {method}")
if tuple(state.get("dims", ())) != dims:
raise ValueError(f"feature dimensions do not match checkpoint: {checkpoint}")
model.load_state_dict(state["state_dict"])
model.eval()
return model
def best_interval(visible: np.ndarray, wanted: int, cap: int, location: str, rng: np.random.Generator) -> tuple[int, int] | None:
steps = len(visible)
candidates: list[tuple[int, int, int, int]] = []
for left in range(steps):
hits = 0
for right in range(left, steps):
hits += int(visible[right])
count = min(hits, cap)
if count:
candidates.append((abs(count - wanted), right - left + 1, left, right))
if not candidates:
return None
best = min((error, span) for error, span, _, _ in candidates)
tied = [(left, right) for error, span, left, right in candidates if (error, span) == best]
if location == "start":
return min(tied, key=lambda pair: (pair[0], pair[1]))
if location == "end":
return max(tied, key=lambda pair: (pair[1], pair[0]))
if location == "middle":
center = (steps - 1) / 2
return min(tied, key=lambda pair: (abs((pair[0] + pair[1]) / 2 - center), pair[0]))
if location != "random":
raise ValueError(f"unknown interval location: {location}")
return tied[int(rng.integers(0, len(tied)))]
def spread_short_spans(visible: np.ndarray, wanted: int, cap: int) -> np.ndarray:
positions = np.flatnonzero(visible)
count = min(int(wanted), int(cap), len(positions))
chosen = np.zeros(len(visible), dtype=bool)
if count <= 0:
return chosen
n_spans = min(3, count)
chunks = np.array_split(positions, n_spans)
allocations = [count // n_spans + int(i < count % n_spans) for i in range(n_spans)]
for chunk, amount in zip(chunks, allocations):
if amount <= 0 or len(chunk) == 0:
continue
amount = min(amount, len(chunk))
start = max(0, (len(chunk) - amount) // 2)
chosen[chunk[start:start + amount]] = True
return chosen
def continuous_mask(
original: np.ndarray,
rate: float,
mode: str,
rng: np.random.Generator,
*,
modalities: tuple[int, ...] | None = None,
location: str = "random",
span_structure: str = "long",
) -> np.ndarray:
"""Reproduce math/Q2 continuous masking on this model's observed positions."""
observed = np.asarray(original, dtype=bool)
result = observed.copy()
if rate <= 0 or mode == "none":
return result
steps, modality_count = observed.shape
present = [m for m in range(modality_count) if observed[:, m].any()]
if not present:
return result
if modalities is not None:
selected = [int(m) for m in modalities if int(m) in present]
if not selected:
return result
elif mode == "single":
selected = [int(rng.choice(present))]
elif mode in {"sync", "partial", "async"}:
if len(present) == 1:
selected = present
else:
count = int(rng.integers(2, min(3, len(present)) + 1))
selected = sorted(int(v) for v in rng.choice(present, size=count, replace=False))
else:
raise ValueError(f"unknown mask mode: {mode}")
def max_hide(modality: int) -> int:
count = int(observed[:, modality].sum())
keep = max(1, int(math.ceil(0.2 * count)))
return max(0, count - keep)
target = {m: min(max_hide(m), int(round(rate * int(observed[:, m].sum())))) for m in selected}
if mode == "sync":
span = max(1, int(round(rate * steps)))
if location == "start":
left = 0
elif location == "end":
left = steps - span
elif location == "middle":
left = (steps - span) // 2
elif location == "random":
left = int(rng.integers(0, max(1, steps - span + 1)))
else:
raise ValueError(f"unknown interval location: {location}")
right = min(steps - 1, left + span - 1)
for m in selected:
candidates = np.flatnonzero(observed[left:right + 1, m]) + left
amount = min(len(candidates), max_hide(m), target[m])
if amount:
offset = 0 if location != "end" else len(candidates) - amount
result[candidates[max(0, offset):max(0, offset) + amount], m] = False
else:
common_span = max(1, int(round(rate * steps)))
for rank, m in enumerate(selected):
wanted = target[m]
if wanted <= 0:
continue
cap = max_hide(m)
if span_structure == "multi_short":
hide = spread_short_spans(observed[:, m], wanted, cap)
elif span_structure != "long":
raise ValueError(f"unknown span structure: {span_structure}")
elif mode == "single" and location != "random":
# Place a contiguous block at the requested relative location
# among observed positions, while keeping the selected-source
# missing amount fixed. This avoids treating padding as time.
interval = best_interval(observed[:, m], wanted, cap, location, rng)
hide = np.zeros(steps, dtype=bool)
if interval is not None:
left, right = interval
candidates = np.flatnonzero(observed[left:right + 1, m]) + left
amount = min(len(candidates), wanted, cap)
if amount:
offset = 0 if location != "end" else len(candidates) - amount
hide[candidates[max(0, offset):max(0, offset) + amount]] = True
elif mode in {"partial", "async"}:
if mode == "partial":
base_left = int(rng.integers(0, max(1, steps - common_span + 1))) if location == "random" else (
0 if location == "start" else steps - common_span if location == "end" else (steps - common_span) // 2
)
offset = int(round(rank * common_span * 0.5))
else:
base_left = 0 if location == "random" else (
0 if location == "start" else steps - common_span if location == "end" else (steps - common_span) // 2
)
available = max(1, steps - common_span + 1)
offsets = np.rint(np.linspace(0, max(0, available - 1), len(selected))).astype(int)
if location == "random":
rng.shuffle(offsets)
offset = int(offsets[rank])
left = min(max(0, base_left + offset), max(0, steps - common_span))
right = min(steps - 1, left + common_span - 1)
hide = np.zeros(steps, dtype=bool)
candidates = np.flatnonzero(observed[left:right + 1, m]) + left
amount = min(len(candidates), wanted, cap)
if amount:
hide[candidates[:amount]] = True
else:
interval = best_interval(observed[:, m], wanted, cap, location, rng)
hide = np.zeros(steps, dtype=bool)
if interval is not None:
left, right = interval
candidates = np.flatnonzero(observed[left:right + 1, m]) + left
amount = min(len(candidates), wanted, cap)
if amount:
offset = 0 if location != "end" else len(candidates) - amount
hide[candidates[max(0, offset):max(0, offset) + amount]] = True
result[hide, m] = False
return result
def scenario_seed(seed: int, sample_id: str, key: str) -> int:
return int.from_bytes(hashlib.sha256(f"{seed}:{sample_id}:{key}".encode()).digest()[:8], "little")
def make_scenarios(valid: Split, seed: int = SCENARIO_SEED) -> dict[str, np.ndarray]:
scenarios = {"0.0/none": valid.mask.copy()}
for rate in CURVE_RATES[1:]:
for mode in CURVE_MODES:
key = f"{rate:.1f}/{mode}"
scenarios[key] = np.stack([
continuous_mask(mask, rate, mode, np.random.default_rng(scenario_seed(seed, sample_id, key)))
for sample_id, mask in zip(valid.ids, valid.mask)
])
modality_sets = (((0,), "T"), ((1,), "A"), ((2,), "V"), ((0, 1), "TA"), ((0, 2), "TV"), ((1, 2), "AV"), ((0, 1, 2), "TAV"))
for selected, label in modality_sets:
key = f"0.3/modality_{label}"
scenarios[key] = np.stack([
continuous_mask(mask, 0.3, "sync", np.random.default_rng(scenario_seed(seed, sample_id, key)), modalities=selected)
for sample_id, mask in zip(valid.ids, valid.mask)
])
for modality_index, label in enumerate(("T", "A", "V")):
for location in ("start", "middle", "end"):
key = f"0.3/location_{location}_{label}"
scenarios[key] = np.stack([
continuous_mask(mask, 0.3, "single", np.random.default_rng(scenario_seed(seed, sample_id, key)), modalities=(modality_index,), location=location)
for sample_id, mask in zip(valid.ids, valid.mask)
])
for structure in ("long", "multi_short"):
key = f"0.3/span_{structure}_{label}"
scenarios[key] = np.stack([
continuous_mask(mask, 0.3, "single", np.random.default_rng(scenario_seed(seed, sample_id, key)), modalities=(modality_index,), span_structure=structure)
for sample_id, mask in zip(valid.ids, valid.mask)
])
for mode in ("sync", "partial", "async"):
key = f"0.3/synchrony_{mode}"
scenarios[key] = np.stack([
continuous_mask(mask, 0.3, mode, np.random.default_rng(scenario_seed(seed, sample_id, key)), modalities=(0, 1, 2))
for sample_id, mask in zip(valid.ids, valid.mask)
])
return scenarios
def actual_additional_rates(base: np.ndarray, scenarios: dict[str, np.ndarray]) -> dict[str, np.ndarray]:
result = {}
observed = base.sum(axis=1)
for scenario, current in scenarios.items():
newly_hidden = base & ~current
hidden_count = newly_hidden.sum(axis=1)
by_modality = np.divide(
hidden_count,
observed,
out=np.full(hidden_count.shape, np.nan, dtype=np.float64),
where=observed > 0,
)
result[scenario] = np.nanmean(by_modality, axis=1)
return result
def aurc_from_curve(rates: list[float], maes: list[float]) -> float:
order = np.argsort(np.asarray(rates), kind="stable")
x = np.asarray(rates, dtype=np.float64)[order]
y = np.asarray(maes, dtype=np.float64)[order]
unique_x, inverse = np.unique(x, return_inverse=True)
unique_y = np.asarray([y[inverse == i].mean() for i in range(len(unique_x))])
if len(unique_x) <= 1 or unique_x[-1] <= 0:
return float(maes[0])
return float(np.trapezoid(unique_y, unique_x) / unique_x[-1])
def curve_scenarios(mode: str) -> list[str]:
return ["0.0/none"] + [f"{rate:.1f}/{mode}" for rate in CURVE_RATES[1:]]
def group_indices(ids: list[str]) -> tuple[list[str], dict[str, np.ndarray]]:
groups = sorted({sample_id.split("$_$", 1)[0] for sample_id in ids})
mapping = {group: np.flatnonzero(np.asarray([x.split("$_$", 1)[0] == group for x in ids])) for group in groups}
return groups, mapping
def bootstrap_clean_test(
split: Split,
preds: dict[tuple[str, int], dict[str, np.ndarray]],
) -> list[dict[str, Any]]:
groups, mapping = group_indices(split.ids)
rng = np.random.default_rng(TEST_BOOTSTRAP_SEED)
draws: dict[str, list[float]] = defaultdict(list)
for _ in range(BOOTSTRAP_REPS):
chosen = rng.choice(groups, size=len(groups), replace=True)
indices = np.concatenate([mapping[group] for group in chosen])
per_method = {}
for method in METHODS:
per_seed = [metrics(split, preds[(method, seed)]["logits"], preds[(method, seed)]["intensity"], indices) for seed in SEEDS]
per_method[method] = {key: float(np.mean([row[key] for row in per_seed])) for key in per_seed[0]}
for metric in per_method[EARLYCONCAT]:
draws[metric].append(per_method[MOFE7_MLP][metric] - per_method[EARLYCONCAT][metric])
rows = []
for metric, values in draws.items():
rows.append({
"comparison": "MoFE-7 + MLP Router minus EarlyConcat + BiGRU",
"metric": metric,
"delta_mean_over_seeds": float(np.mean([r[metric] for r in [
metrics(split, preds[(MOFE7_MLP, seed)]["logits"], preds[(MOFE7_MLP, seed)]["intensity"])
for seed in SEEDS
]]) - np.mean([r[metric] for r in [
metrics(split, preds[(EARLYCONCAT, seed)]["logits"], preds[(EARLYCONCAT, seed)]["intensity"])
for seed in SEEDS
]])),
"bootstrap_ci_2p5": float(np.quantile(values, 0.025)),
"bootstrap_ci_97p5": float(np.quantile(values, 0.975)),
"bootstrap_probability_delta_gt_0": float(np.mean(np.asarray(values) > 0)),
"replicates": BOOTSTRAP_REPS,
"resampling_unit": "source video id",
"paired": True,
"seed": TEST_BOOTSTRAP_SEED,
})
return rows
def bootstrap_aurc(
valid: Split,
predictions: dict[tuple[str, int, str], dict[str, np.ndarray]],
scenarios: dict[str, np.ndarray],
rates_by_sample: dict[str, np.ndarray],
) -> list[dict[str, Any]]:
groups, mapping = group_indices(valid.ids)
rng = np.random.default_rng(AURC_BOOTSTRAP_SEED)
delta_by_mode: dict[str, list[float]] = {mode: [] for mode in CURVE_MODES}
for _ in range(BOOTSTRAP_REPS):
chosen = rng.choice(groups, size=len(groups), replace=True)
indices = np.concatenate([mapping[group] for group in chosen])
for mode in CURVE_MODES:
keys = curve_scenarios(mode)
model_aucs: dict[str, list[float]] = {method: [] for method in METHODS}
for method in METHODS:
for seed in SEEDS:
xs = [float(np.nanmean(rates_by_sample[key][indices])) for key in keys]
ys = [float(np.abs(valid.y_reg[indices] - predictions[(method, seed, key)]["intensity"][indices]).mean()) for key in keys]
model_aucs[method].append(aurc_from_curve(xs, ys))
delta_by_mode[mode].append(float(np.mean(model_aucs[MOFE7_MLP]) - np.mean(model_aucs[EARLYCONCAT])))
point = {}
for mode in CURVE_MODES:
model_aucs = {}
for method in METHODS:
model_aucs[method] = []
for seed in SEEDS:
keys = curve_scenarios(mode)
xs = [float(np.nanmean(rates_by_sample[key])) for key in keys]
ys = [float(np.abs(valid.y_reg - predictions[(method, seed, key)]["intensity"]).mean()) for key in keys]
model_aucs[method].append(aurc_from_curve(xs, ys))
point[mode] = float(np.mean(model_aucs[MOFE7_MLP]) - np.mean(model_aucs[EARLYCONCAT]))
rows = []
for mode, values in delta_by_mode.items():
rows.append({
"mode": mode,
"delta_aurc_mae_mofe_minus_earlyconcat": point[mode],
"bootstrap_ci_2p5": float(np.quantile(values, 0.025)),
"bootstrap_ci_97p5": float(np.quantile(values, 0.975)),
"bootstrap_probability_delta_lt_0": float(np.mean(np.asarray(values) < 0)),
"replicates": BOOTSTRAP_REPS,
"resampling_unit": "source video id",
"paired": True,
"seed": AURC_BOOTSTRAP_SEED,
})
return rows
def run(device_name: str = "auto", batch_size: int = 64, masks_only: bool = False) -> None:
OUTPUT_DIR.mkdir(parents=True, exist_ok=True)
device = _device_for(device_name)
torch.set_num_threads(4)
torch.backends.cudnn.deterministic = True
torch.backends.cudnn.benchmark = False
feature_path = ATTACHMENT2 / "aligned_50.pkl"
with (REFERENCE_DIR / "run_manifest.json").open("r", encoding="utf-8") as stream:
reference_manifest = json.load(stream)
if sha256(feature_path) != reference_manifest["feature_sha256"]:
raise ValueError("current official feature file hash differs from the checkpoint evaluation manifest")
raw_splits = load_splits(feature_path)
train = raw_splits["train"]
valid = raw_splits["valid"]
test = raw_splits["test"]
scaler_path = REFERENCE_DIR / "aligned_robust_stats.npz"
stats = RobustStats.load(scaler_path)
computed = fit_robust_stats(train)
scaler_diff = max(
max(float(np.max(np.abs(a - b))) for a, b in zip(computed.center, stats.center)),
max(float(np.max(np.abs(a - b))) for a, b in zip(computed.scale, stats.scale)),
)
if scaler_diff > 1e-6:
raise ValueError(f"checkpoint scaler is not the train-only scaler (max difference {scaler_diff})")
valid = apply_robust_stats(valid, stats)
test = apply_robust_stats(test, stats)
dims = tuple(x.shape[-1] for x in train.x)
if not masks_only:
# Final, clean official-test evaluation; no retraining or selection occurs here.
test_predictions: dict[tuple[str, int], dict[str, np.ndarray]] = {}
test_rows: list[dict[str, Any]] = []
for method in METHODS:
for seed in SEEDS:
model = load_model(method, seed, dims, device)
prediction = _predict(model, test, test.mask, device, batch_size)
test_predictions[(method, seed)] = prediction
test_rows.append({"method": method, "seed": seed, "n_test": test.n, **metrics(test, prediction["logits"], prediction["intensity"])})
del model
if torch.cuda.is_available():
torch.cuda.empty_cache()
summary_rows = []
for method in METHODS:
subset = [row for row in test_rows if row["method"] == method]
for metric in ("accuracy", "macro_f1", "mae", "rmse", "pearson"):
values = [float(row[metric]) for row in subset]
summary_rows.append({"method": method, "metric": metric, "mean": float(np.mean(values)), "sd_across_seeds": float(np.std(values, ddof=1))})
write_csv(OUTPUT_DIR / "official_test_metrics_by_seed.csv", test_rows)
write_csv(OUTPUT_DIR / "official_test_summary.csv", summary_rows)
write_csv(OUTPUT_DIR / "official_test_paired_bootstrap.csv", bootstrap_clean_test(test, test_predictions))
# Reproduce the math-Q2 42-scenario design with a per-sample stable seed,
# while applying it to the observation masks used to train these models.
scenario_masks = make_scenarios(valid)
rates_by_sample = actual_additional_rates(valid.mask, scenario_masks)
condition_predictions: dict[tuple[str, int, str], dict[str, np.ndarray]] = {}
condition_rows: list[dict[str, Any]] = []
for method in METHODS:
for seed in SEEDS:
model = load_model(method, seed, dims, device)
for scenario, masks in scenario_masks.items():
prediction = _predict(model, valid, masks, device, batch_size)
condition_predictions[(method, seed, scenario)] = prediction
values = metrics(valid, prediction["logits"], prediction["intensity"])
condition_rows.append({
"method": method,
"seed": seed,
"scenario": scenario,
"realized_additional_global_rate": float(np.nanmean(rates_by_sample[scenario])),
"n_valid": valid.n,
**values,
})
del model
if torch.cuda.is_available():
torch.cuda.empty_cache()
write_csv(OUTPUT_DIR / "controlled_metrics_by_scenario.csv", condition_rows)
auc_rows: list[dict[str, Any]] = []
for method in METHODS:
for seed in SEEDS:
for mode in CURVE_MODES:
keys = curve_scenarios(mode)
xs = [float(np.nanmean(rates_by_sample[key])) for key in keys]
ys = [float(np.abs(valid.y_reg - condition_predictions[(method, seed, key)]["intensity"]).mean()) for key in keys]
auc_rows.append({"method": method, "seed": seed, "mask_mode": mode, "aurc_mae": aurc_from_curve(xs, ys), "rates_realized": json.dumps(xs)})
write_csv(OUTPUT_DIR / "aurc_mae_by_mode_seed.csv", auc_rows)
auc_summary = []
for method in METHODS:
for mode in CURVE_MODES:
values = [row["aurc_mae"] for row in auc_rows if row["method"] == method and row["mask_mode"] == mode]
auc_summary.append({"method": method, "mask_mode": mode, "mean": float(np.mean(values)), "sd_across_seeds": float(np.std(values, ddof=1))})
write_csv(OUTPUT_DIR / "aurc_mae_summary.csv", auc_summary)
write_csv(OUTPUT_DIR / "aurc_mae_paired_bootstrap.csv", bootstrap_aurc(valid, condition_predictions, scenario_masks, rates_by_sample))
manifest = {
"experiment": "Frozen EarlyConcat vs MoFE-7 evaluation under math/Q2 test protocol",
"created_unix": time.time(),
"device": str(device),
"cuda_device": torch.cuda.get_device_name(0) if device.type == "cuda" else None,
"feature_file": str(feature_path),
"feature_sha256": sha256(feature_path),
"representation": "official aligned_50 ordered positions; not physical-time bins",
"train_valid_test_counts": {name: split.n for name, split in raw_splits.items()},
"source_video_groups": {name: len({sample_id.split("$_$", 1)[0] for sample_id in split.ids}) for name, split in raw_splits.items()},
"official_group_splits_disjoint": True,
"test_evaluation": (
"one final clean evaluation on official labeled test split; no training/model selection/calibration"
if not masks_only else "test outputs preserved from the earlier single evaluation; no test prediction was rerun"
),
"test_prediction_performed_this_invocation": not masks_only,
"seeds": list(SEEDS),
"checkpoint_source": str(REFERENCE_DIR / "models"),
"train_only_scaler": str(scaler_path),
"scaler_max_abs_difference_from_train_refit": scaler_diff,
"test_labels_used_for_training_or_selection": False,
"controlled_missingness": {
"scenario_seed": SCENARIO_SEED,
"scenario_design": "math/Q2 42-scenario design regenerated on the Q2 models' BERT attention-mask base",
"scenarios": len(scenario_masks),
"AURC": "normalized trapezoidal area of MAE over realized equal-modality-weighted added missing rate, at 0/.1/.3/.5/.7 for single/sync/partial/async",
},
"bootstrap": {
"replicates": BOOTSTRAP_REPS,
"test_seed": TEST_BOOTSTRAP_SEED,
"aurc_seed": AURC_BOOTSTRAP_SEED,
"unit": "source video id",
"paired": True,
},
}
(OUTPUT_DIR / "run_manifest.json").write_text(json.dumps(manifest, indent=2), encoding="utf-8")
print(f"wrote math-protocol comparison to {OUTPUT_DIR}")
print(f"n_test={test.n}; n_valid={valid.n}; device={device}; scenarios={len(scenario_masks)}")
if __name__ == "__main__":
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--masks-only", action="store_true", help="Recompute validation mask scenarios without rerunning official-test inference")
args = parser.parse_args()
run(masks_only=args.masks_only)