116 lines
5.2 KiB
Python
116 lines
5.2 KiB
Python
"""Consolidate Q2 validation and selected official-test scores on adapter data."""
|
|
from __future__ import annotations
|
|
|
|
import csv
|
|
import argparse
|
|
import gzip
|
|
import json
|
|
from pathlib import Path
|
|
|
|
|
|
ROOT = Path(__file__).resolve().parent
|
|
OUT = ROOT / "output" / "q2"
|
|
EXPERIMENTS = ROOT / "experiments" / "q2"
|
|
MATH = EXPERIMENTS / "unaligned_math_all_b128"
|
|
DEEP = EXPERIMENTS / "unaligned_deep_two_b128"
|
|
|
|
|
|
def rows(path: Path) -> list[dict[str, str]]:
|
|
opener = gzip.open if path.suffix == ".gz" else open
|
|
with opener(path, "rt", newline="", encoding="utf-8-sig") as stream:
|
|
return list(csv.DictReader(stream))
|
|
|
|
|
|
def write(path: Path, records: list[dict[str, str]]) -> None:
|
|
with path.open("w", newline="", encoding="utf-8") as stream:
|
|
writer = csv.DictWriter(stream, fieldnames=list(records[0]))
|
|
writer.writeheader()
|
|
writer.writerows(records)
|
|
|
|
|
|
def record(name: str, family: str, collection: str, scores: dict) -> dict[str, str]:
|
|
return {
|
|
"model": name,
|
|
"family": family,
|
|
"split": collection,
|
|
"n": str(scores.get("n", scores.get("n_valid", scores.get("n_test", "")))),
|
|
"accuracy": str(scores["accuracy"]),
|
|
"macro_f1": str(scores["macro_f1"]),
|
|
"mae": str(scores.get("regression_mae", scores.get("mae"))),
|
|
"rmse": str(scores.get("regression_rmse", scores.get("rmse"))),
|
|
"pearson": str(scores["pearson"]),
|
|
}
|
|
|
|
|
|
def normalized_auc(points: list[tuple[float, float]]) -> float:
|
|
points = sorted(points)
|
|
assert len(points) == 5, f"expected five missing-rate points, got {len(points)}"
|
|
width = points[-1][0] - points[0][0]
|
|
assert points[0][0] == 0.0 and width > 0
|
|
area = sum((x1 - x0) * (y0 + y1) / 2
|
|
for (x0, y0), (x1, y1) in zip(points, points[1:]))
|
|
return area / width
|
|
|
|
|
|
def main() -> None:
|
|
global OUT, MATH, DEEP
|
|
parser = argparse.ArgumentParser(description="Consolidate the unaligned Q2 comparison outputs.")
|
|
parser.add_argument("--math-dir", type=Path, default=MATH)
|
|
parser.add_argument("--deep-dir", type=Path, default=DEEP)
|
|
parser.add_argument("--output-dir", type=Path, default=OUT)
|
|
args = parser.parse_args()
|
|
MATH, DEEP, OUT = args.math_dir.resolve(), args.deep_dir.resolve(), args.output_dir.resolve()
|
|
OUT.mkdir(parents=True, exist_ok=True)
|
|
math_valid = rows(MATH / "ablation_validation.csv")
|
|
deep_valid = [r for r in rows(DEEP / "controlled_metrics_by_scenario.csv")
|
|
if r["scenario"] == "0.0/none"]
|
|
assert len(math_valid) == 12, f"expected 12 math variants, got {len(math_valid)}"
|
|
assert {r["method"] for r in deep_valid} == {"B0_early_concat", "B5_mofe_mlp"}
|
|
valid = [record(r["model"], "math", "official_valid", r) for r in math_valid]
|
|
valid += [record("EarlyConcat" if r["method"] == "B0_early_concat" else "MoFE-7",
|
|
"deep_learning", "official_valid", r) for r in deep_valid]
|
|
assert {r["n"] for r in valid} == {"728"}
|
|
write(OUT / "comparison_validation.csv", valid)
|
|
|
|
math_test = json.loads((MATH / "test_metrics.json").read_text(encoding="utf-8"))
|
|
deep_test = rows(DEEP / "official_test_metrics_by_seed.csv")
|
|
selected = math_test.get("model")
|
|
if not selected:
|
|
manifest = json.loads((MATH / "run_manifest.json").read_text(encoding="utf-8"))
|
|
selected = manifest["selected_model"]
|
|
test = [record(selected, "math", "official_test", {"n": 727, **math_test})]
|
|
test += [record("EarlyConcat" if r["method"] == "B0_early_concat" else "MoFE-7",
|
|
"deep_learning", "official_test", r) for r in deep_test]
|
|
assert {r["n"] for r in test} == {"727"}
|
|
write(OUT / "comparison_test.csv", test)
|
|
|
|
modes = ("single", "sync", "partial", "async")
|
|
math_sweep = rows(MATH / "controlled_missingness.csv.gz")
|
|
math_models = sorted({r["model"] for r in math_sweep})
|
|
aurc: list[dict[str, str]] = []
|
|
for model in math_models:
|
|
metric_row = {"model": model, "family": "math", "split": "official_valid", "n": "728"}
|
|
for mode in modes:
|
|
curve = [r for r in math_sweep
|
|
if r["model"] == model and r["mask_pattern"] in ("none", mode)]
|
|
assert len(curve) == 5, f"{model}/{mode}: expected five curve points, got {len(curve)}"
|
|
metric_row[mode] = f"{normalized_auc([(float(r['rate_realized_additional_global']), float(r['regression_mae'])) for r in curve]):.12g}"
|
|
aurc.append(metric_row)
|
|
|
|
deep_map = {"B0_early_concat": "EarlyConcat", "B5_mofe_mlp": "MoFE-7"}
|
|
deep_auc = rows(DEEP / "aurc_mae_by_mode_seed.csv")
|
|
for method, model in deep_map.items():
|
|
metric_row = {"model": model, "family": "deep_learning", "split": "official_valid", "n": "728"}
|
|
for mode in modes:
|
|
values = [r for r in deep_auc if r["method"] == method and r["mask_mode"] == mode]
|
|
assert len(values) == 1, f"{model}/{mode}: expected one AURC row, got {len(values)}"
|
|
metric_row[mode] = values[0]["aurc_mae"]
|
|
aurc.append(metric_row)
|
|
assert len(aurc) == 14, f"expected 14 AURC rows, got {len(aurc)}"
|
|
write(OUT / "comparison_aurc.csv", aurc)
|
|
print(f"wrote {len(valid)} validation, {len(test)} test and {len(aurc)} AURC rows; math test selection={selected}")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|