Complete standalone final deliverable and unaligned Q2 results
This commit is contained in:
@@ -0,0 +1,115 @@
|
||||
"""Consolidate Q2 validation and selected official-test scores on adapter data."""
|
||||
from __future__ import annotations
|
||||
|
||||
import csv
|
||||
import argparse
|
||||
import gzip
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
ROOT = Path(__file__).resolve().parent
|
||||
OUT = ROOT / "output" / "q2"
|
||||
EXPERIMENTS = ROOT / "experiments" / "q2"
|
||||
MATH = EXPERIMENTS / "unaligned_math_all_b128"
|
||||
DEEP = EXPERIMENTS / "unaligned_deep_two_b128"
|
||||
|
||||
|
||||
def rows(path: Path) -> list[dict[str, str]]:
|
||||
opener = gzip.open if path.suffix == ".gz" else open
|
||||
with opener(path, "rt", newline="", encoding="utf-8-sig") as stream:
|
||||
return list(csv.DictReader(stream))
|
||||
|
||||
|
||||
def write(path: Path, records: list[dict[str, str]]) -> None:
|
||||
with path.open("w", newline="", encoding="utf-8") as stream:
|
||||
writer = csv.DictWriter(stream, fieldnames=list(records[0]))
|
||||
writer.writeheader()
|
||||
writer.writerows(records)
|
||||
|
||||
|
||||
def record(name: str, family: str, collection: str, scores: dict) -> dict[str, str]:
|
||||
return {
|
||||
"model": name,
|
||||
"family": family,
|
||||
"split": collection,
|
||||
"n": str(scores.get("n", scores.get("n_valid", scores.get("n_test", "")))),
|
||||
"accuracy": str(scores["accuracy"]),
|
||||
"macro_f1": str(scores["macro_f1"]),
|
||||
"mae": str(scores.get("regression_mae", scores.get("mae"))),
|
||||
"rmse": str(scores.get("regression_rmse", scores.get("rmse"))),
|
||||
"pearson": str(scores["pearson"]),
|
||||
}
|
||||
|
||||
|
||||
def normalized_auc(points: list[tuple[float, float]]) -> float:
|
||||
points = sorted(points)
|
||||
assert len(points) == 5, f"expected five missing-rate points, got {len(points)}"
|
||||
width = points[-1][0] - points[0][0]
|
||||
assert points[0][0] == 0.0 and width > 0
|
||||
area = sum((x1 - x0) * (y0 + y1) / 2
|
||||
for (x0, y0), (x1, y1) in zip(points, points[1:]))
|
||||
return area / width
|
||||
|
||||
|
||||
def main() -> None:
|
||||
global OUT, MATH, DEEP
|
||||
parser = argparse.ArgumentParser(description="Consolidate the unaligned Q2 comparison outputs.")
|
||||
parser.add_argument("--math-dir", type=Path, default=MATH)
|
||||
parser.add_argument("--deep-dir", type=Path, default=DEEP)
|
||||
parser.add_argument("--output-dir", type=Path, default=OUT)
|
||||
args = parser.parse_args()
|
||||
MATH, DEEP, OUT = args.math_dir.resolve(), args.deep_dir.resolve(), args.output_dir.resolve()
|
||||
OUT.mkdir(parents=True, exist_ok=True)
|
||||
math_valid = rows(MATH / "ablation_validation.csv")
|
||||
deep_valid = [r for r in rows(DEEP / "controlled_metrics_by_scenario.csv")
|
||||
if r["scenario"] == "0.0/none"]
|
||||
assert len(math_valid) == 12, f"expected 12 math variants, got {len(math_valid)}"
|
||||
assert {r["method"] for r in deep_valid} == {"B0_early_concat", "B5_mofe_mlp"}
|
||||
valid = [record(r["model"], "math", "official_valid", r) for r in math_valid]
|
||||
valid += [record("EarlyConcat" if r["method"] == "B0_early_concat" else "MoFE-7",
|
||||
"deep_learning", "official_valid", r) for r in deep_valid]
|
||||
assert {r["n"] for r in valid} == {"728"}
|
||||
write(OUT / "comparison_validation.csv", valid)
|
||||
|
||||
math_test = json.loads((MATH / "test_metrics.json").read_text(encoding="utf-8"))
|
||||
deep_test = rows(DEEP / "official_test_metrics_by_seed.csv")
|
||||
selected = math_test.get("model")
|
||||
if not selected:
|
||||
manifest = json.loads((MATH / "run_manifest.json").read_text(encoding="utf-8"))
|
||||
selected = manifest["selected_model"]
|
||||
test = [record(selected, "math", "official_test", {"n": 727, **math_test})]
|
||||
test += [record("EarlyConcat" if r["method"] == "B0_early_concat" else "MoFE-7",
|
||||
"deep_learning", "official_test", r) for r in deep_test]
|
||||
assert {r["n"] for r in test} == {"727"}
|
||||
write(OUT / "comparison_test.csv", test)
|
||||
|
||||
modes = ("single", "sync", "partial", "async")
|
||||
math_sweep = rows(MATH / "controlled_missingness.csv.gz")
|
||||
math_models = sorted({r["model"] for r in math_sweep})
|
||||
aurc: list[dict[str, str]] = []
|
||||
for model in math_models:
|
||||
metric_row = {"model": model, "family": "math", "split": "official_valid", "n": "728"}
|
||||
for mode in modes:
|
||||
curve = [r for r in math_sweep
|
||||
if r["model"] == model and r["mask_pattern"] in ("none", mode)]
|
||||
assert len(curve) == 5, f"{model}/{mode}: expected five curve points, got {len(curve)}"
|
||||
metric_row[mode] = f"{normalized_auc([(float(r['rate_realized_additional_global']), float(r['regression_mae'])) for r in curve]):.12g}"
|
||||
aurc.append(metric_row)
|
||||
|
||||
deep_map = {"B0_early_concat": "EarlyConcat", "B5_mofe_mlp": "MoFE-7"}
|
||||
deep_auc = rows(DEEP / "aurc_mae_by_mode_seed.csv")
|
||||
for method, model in deep_map.items():
|
||||
metric_row = {"model": model, "family": "deep_learning", "split": "official_valid", "n": "728"}
|
||||
for mode in modes:
|
||||
values = [r for r in deep_auc if r["method"] == method and r["mask_mode"] == mode]
|
||||
assert len(values) == 1, f"{model}/{mode}: expected one AURC row, got {len(values)}"
|
||||
metric_row[mode] = values[0]["aurc_mae"]
|
||||
aurc.append(metric_row)
|
||||
assert len(aurc) == 14, f"expected 14 AURC rows, got {len(aurc)}"
|
||||
write(OUT / "comparison_aurc.csv", aurc)
|
||||
print(f"wrote {len(valid)} validation, {len(test)} test and {len(aurc)} AURC rows; math test selection={selected}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user