58 lines
2.5 KiB
Python
58 lines
2.5 KiB
Python
"""C0: observed statistics and mask baseline from E题V2, table 5.9."""
|
|
from __future__ import annotations
|
|
|
|
import numpy as np
|
|
from sklearn.linear_model import LogisticRegression, Ridge
|
|
|
|
MODALITIES = ("text", "audio", "vision")
|
|
|
|
|
|
def sample_statistics(arrays: dict[str, np.ndarray], mask: np.ndarray) -> np.ndarray:
|
|
"""Mean, standard deviation, missing fraction and longest gap per modality."""
|
|
mask = np.asarray(mask, dtype=bool)
|
|
if mask.ndim != 3 or mask.shape[-1] != 3:
|
|
raise ValueError("mask must have shape (N, T, 3)")
|
|
parts = []
|
|
for index, name in enumerate(MODALITIES):
|
|
x = np.asarray(arrays[name], dtype=np.float32)
|
|
if x.shape[:2] != mask.shape[:2]:
|
|
raise ValueError(f"{name}: feature and mask shapes disagree")
|
|
visible = mask[:, :, index]
|
|
count = visible.sum(axis=1, keepdims=True)
|
|
mean = (x * visible[:, :, None]).sum(axis=1) / np.maximum(count, 1)
|
|
variance = (((x - mean[:, None, :]) ** 2) * visible[:, :, None]).sum(axis=1) / np.maximum(count, 1)
|
|
missing = 1.0 - visible.mean(axis=1, keepdims=True)
|
|
max_gap = []
|
|
for row in visible:
|
|
longest = current = 0
|
|
for observed in row:
|
|
current = 0 if observed else current + 1
|
|
longest = max(longest, current)
|
|
max_gap.append(longest / max(len(row), 1))
|
|
parts.extend((mean, np.sqrt(variance), missing, np.asarray(max_gap, np.float32)[:, None]))
|
|
return np.concatenate(parts, axis=1).astype(np.float32)
|
|
|
|
|
|
class C0:
|
|
"""Logistic polarity classifier and Ridge intensity regressor."""
|
|
|
|
def __init__(self) -> None:
|
|
self.classifier = LogisticRegression(C=0.05, max_iter=2500, random_state=20260924)
|
|
self.regressor = Ridge(alpha=25.0)
|
|
|
|
def fit(self, arrays: dict[str, np.ndarray], mask: np.ndarray,
|
|
polarity: np.ndarray, intensity: np.ndarray) -> "C0":
|
|
features = sample_statistics(arrays, mask)
|
|
self.classifier.fit(features, polarity)
|
|
self.regressor.fit(features, intensity)
|
|
return self
|
|
|
|
def predict(self, arrays: dict[str, np.ndarray], mask: np.ndarray) -> dict[str, np.ndarray]:
|
|
features = sample_statistics(arrays, mask)
|
|
probabilities = np.zeros((len(features), 3), np.float64)
|
|
probabilities[:, self.classifier.classes_] = self.classifier.predict_proba(features)
|
|
return {
|
|
"probabilities": probabilities,
|
|
"intensity": np.clip(self.regressor.predict(features), -3.0, 3.0),
|
|
}
|