BigDong's picture
update UltraData-Code-L2 Classifier
d9099a0
Raw
History Blame Contribute Delete
13.3 kB
"""Loaders and inference for the C / A / Q code-selection scorers.
A scorer set is a flat directory, one per language, such as ``classifiers/cpp``::
<lang>/
C_file_role.bin C_file_role.json # classifier + its class names
A_relevance.bin # checkpoint carries its own config
Q_quality.bin # architecture read from its shapes
Q_scaler.npz # Q's StandardScaler (mean / scale)
All five files are required. Layer widths are recovered from the checkpoints'
own tensor shapes rather than declared in config, so the only metadata shipped
alongside the weights is ``C_file_role.json`` — the index-to-category map, the
one thing the weights cannot carry.
``LanguageScorers.load(dir)`` reads the artifacts and configures itself from
what they declare; nothing about the featurization is hard-coded per language.
All three models consume a precomputed code embedding. Embeddings must come
from the same encoder and dimension used at training time — the loader checks
the dimension, but cannot detect a wrong encoder at the right dimension.
"""
from __future__ import annotations
import json
from pathlib import Path
import numpy as np
import torch
import torch.nn as nn
from path_features import PATH_HASH_ALGORITHM, build_feature_matrix
# --------------------------------------------------------------------------- #
# architectures (must match the training scripts' saved state_dicts)
# --------------------------------------------------------------------------- #
class FileRoleClassifier(nn.Module):
"""C: Linear/LayerNorm/GELU/Dropout stack -> num_classes logits."""
def __init__(self, input_dim, hidden_dims, num_classes, dropout=0.3):
super().__init__()
layers = []
prev = input_dim
for i, h in enumerate(hidden_dims):
layers += [
nn.Linear(prev, h),
nn.LayerNorm(h),
nn.GELU(),
nn.Dropout(dropout if i == 0 else dropout * 0.7),
]
prev = h
layers.append(nn.Linear(prev, num_classes))
self.net = nn.Sequential(*layers)
def forward(self, x):
return self.net(x)
class RelevanceMLP(nn.Module):
"""A: two hidden ReLU layers -> single logit."""
def __init__(self, in_dim, hidden1, hidden2):
super().__init__()
self.fc1 = nn.Linear(in_dim, hidden1)
self.fc2 = nn.Linear(hidden1, hidden2)
self.fc3 = nn.Linear(hidden2, 1)
def forward(self, x):
x = torch.relu(self.fc1(x))
x = torch.relu(self.fc2(x))
return self.fc3(x).squeeze(-1)
class _QualityBackbone(nn.Module):
"""Q's shared trunk.
Kept as its own module holding a ``self.net`` Sequential because the saved
state_dicts are keyed ``backbone.net.0.*``. Inlining the Sequential into
QualityMLP would renumber the keys to ``backbone.0.*`` and fail to load.
"""
def __init__(self, in_dim, hidden_dim, dropout):
super().__init__()
mid = max(64, hidden_dim // 2)
self.net = nn.Sequential(
nn.Linear(in_dim, hidden_dim),
nn.GELU(),
nn.LayerNorm(hidden_dim),
nn.Dropout(dropout),
nn.Linear(hidden_dim, mid),
nn.GELU(),
nn.LayerNorm(mid),
nn.Dropout(dropout),
)
def forward(self, x):
return self.net(x)
class QualityMLP(nn.Module):
"""Q: shared trunk -> zero-score head + 1..10 ordinal head.
Final score is ``(1 - sigmoid(zero)) * (1 + sum(sigmoid(ordinal)))``: the
zero head gates out config/data/no-logic files, the ordinal head places
everything else on 1..10.
"""
def __init__(self, in_dim, hidden_dim=512, dropout=0.1, ordinal_num_classes=10):
super().__init__()
mid = max(64, hidden_dim // 2)
self.backbone = _QualityBackbone(in_dim, hidden_dim, dropout)
self.zero_head = nn.Linear(mid, 1)
self.ordinal_levels = ordinal_num_classes - 1
self.ordinal_head = nn.Linear(mid, self.ordinal_levels)
def forward(self, x):
features = self.backbone(x)
return self.zero_head(features).squeeze(-1), self.ordinal_head(features)
# --------------------------------------------------------------------------- #
# standard scaler without an sklearn version dependency
# --------------------------------------------------------------------------- #
class _Scaler:
"""(x - mean) / scale, loaded from ``Q_scaler.npz``."""
def __init__(self, mean: np.ndarray, scale: np.ndarray, source: str):
self.mean = mean.astype(np.float32)
self.scale = scale.astype(np.float32)
self.n_features_in = int(self.mean.size)
self.source = source
@classmethod
def load(cls, scorer_dir: Path) -> "_Scaler":
npz = scorer_dir / "Q_scaler.npz"
payload = np.load(npz)
return cls(payload["mean"], payload["scale"], npz.name)
def transform(self, x: np.ndarray) -> np.ndarray:
return ((x - self.mean) / self.scale).astype(np.float32)
# --------------------------------------------------------------------------- #
# scorer set
# --------------------------------------------------------------------------- #
class LanguageScorers:
"""Loads one language's scorer set and scores batches of embeddings."""
def __init__(self, scorer_dir, device="cpu"):
self.scorer_dir = Path(scorer_dir)
self.device = torch.device(device)
self.language = self.scorer_dir.name
# -- loading ---------------------------------------------------------- #
@classmethod
def load(cls, scorer_dir, device="cpu") -> "LanguageScorers":
self = cls(scorer_dir, device)
self._load_file_role()
self._load_relevance()
self._load_quality()
if self.quality_in_dim != self.embedding_dim:
raise ValueError(
f"Q expects {self.quality_in_dim}-d input but C implies "
f"{self.embedding_dim}-d embeddings"
)
return self
def _load_file_role(self) -> None:
cfg = json.loads((self.scorer_dir / "C_file_role.json").read_text())
state = torch.load(
self.scorer_dir / "C_file_role.bin", map_location=self.device
)
# Layer widths come from the checkpoint itself: every Linear weight is
# [out, in], so the stack of 2-D weights gives input_dim, the hidden
# widths and num_classes. Dropout is inactive under eval().
linears = [v.shape for k, v in state.items()
if k.endswith("weight") and v.ndim == 2]
input_dim = int(linears[0][1])
hidden_dims = [int(s[0]) for s in linears[:-1]]
num_classes = int(linears[-1][0])
model = FileRoleClassifier(input_dim, hidden_dims, num_classes)
model.load_state_dict(state)
self.file_role_model = model.eval().to(self.device)
self.embedding_dim = input_dim
# The one thing weights cannot carry: which class each index means.
self.idx2cat = {int(k): v for k, v in cfg["idx2cat"].items()}
if len(self.idx2cat) != num_classes:
raise ValueError(
f"C_file_role.json maps {len(self.idx2cat)} categories but the "
f"checkpoint has {num_classes} output units"
)
def _load_relevance(self) -> None:
# Named explicitly rather than globbed: in a flat scorer-set directory a
# glob for "*.bin" would also match the C and Q checkpoints.
ckpt = torch.load(
self.scorer_dir / "A_relevance.bin", map_location=self.device
)
arch = ckpt["arch"]
model = RelevanceMLP(arch["in_dim"], arch["hidden1"], arch["hidden2"])
model.load_state_dict(ckpt["state_dict"])
self.relevance_model = model.eval().to(self.device)
self.relevance_in_dim = int(arch["in_dim"])
cfg = ckpt["feature_config"]
if not cfg.get("use_path_feature", False):
raise ValueError(
"this A checkpoint declares no path features; every scorer set "
"in this release feeds A a 1024-d embedding plus a 256-d path hash"
)
self.path_hash_dim = int(cfg["path_hash_dim"])
self.path_feature_weight = float(cfg["path_feature_weight"])
self.path_column = "relative_path"
self.path_algorithm = PATH_HASH_ALGORITHM
expected = self.embedding_dim + self.path_hash_dim
if expected != self.relevance_in_dim:
raise ValueError(
f"A expects {self.relevance_in_dim}-d input but the configured "
f"featurization produces {expected}-d"
)
def _load_quality(self) -> None:
self.quality_scaler = _Scaler.load(self.scorer_dir)
state = torch.load(self.scorer_dir / "Q_quality.bin", map_location=self.device)
# Shapes carry the architecture: the trunk's first Linear is
# [hidden_dim, in_dim] and the ordinal head is [num_classes - 1, mid].
# No config file is needed, and load_state_dict below is the real guard
# — a differently-headed checkpoint fails on key names, not on a flag.
in_dim = self.quality_scaler.n_features_in or int(
state["backbone.net.0.weight"].shape[1]
)
hidden_dim = int(state["backbone.net.0.weight"].shape[0])
ordinal_num_classes = int(state["ordinal_head.weight"].shape[0]) + 1
model = QualityMLP(in_dim, hidden_dim, ordinal_num_classes=ordinal_num_classes)
model.load_state_dict(state)
self.quality_model = model.eval().to(self.device)
self.quality_in_dim = in_dim
# -- inference -------------------------------------------------------- #
def predict_file_role(self, embeddings: np.ndarray):
with torch.no_grad():
logits = self.file_role_model(torch.from_numpy(embeddings).to(self.device))
probs = torch.softmax(logits, dim=1)
confidence, index = probs.max(dim=1)
roles = [self.idx2cat.get(int(i), "EXCLUDE") for i in index.cpu().numpy()]
return roles, confidence.cpu().numpy()
def predict_relevance(self, embeddings: np.ndarray, paths) -> np.ndarray:
features = build_feature_matrix(
embeddings,
paths,
path_hash_dim=self.path_hash_dim,
path_feature_weight=self.path_feature_weight,
)
with torch.no_grad():
tensor = torch.from_numpy(np.ascontiguousarray(features, dtype=np.float32))
logits = self.relevance_model(tensor.to(self.device))
scores = torch.sigmoid(logits).cpu().numpy()
return np.asarray(scores).reshape(-1)
def predict_quality(self, embeddings: np.ndarray) -> np.ndarray:
scaled = self.quality_scaler.transform(embeddings)
with torch.no_grad():
zero_logit, ordinal_logits = self.quality_model(
torch.from_numpy(scaled).to(self.device)
)
p_zero = torch.sigmoid(zero_logit)
expected_positive = 1.0 + torch.sigmoid(ordinal_logits).sum(dim=1)
out = (1.0 - p_zero) * expected_positive
return np.asarray(out.cpu().numpy()).reshape(-1)
def score(self, embeddings, paths=None) -> dict:
"""Score a batch. Returns the four columns as numpy arrays / lists."""
embeddings = np.ascontiguousarray(embeddings, dtype=np.float32)
if embeddings.ndim != 2 or embeddings.shape[1] != self.embedding_dim:
raise ValueError(
f"expected embeddings of shape [N, {self.embedding_dim}], "
f"got {tuple(embeddings.shape)}"
)
if paths is None:
paths = [""] * len(embeddings)
roles, confidence = self.predict_file_role(embeddings)
return {
"category": roles,
"cls_confidence": confidence,
"algo_rel_score": self.predict_relevance(embeddings, paths),
"quality_score": self.predict_quality(embeddings),
}
def needs_paths(self) -> bool:
"""True when any model consumes the file path. Always true: A does."""
return self.path_hash_dim > 0
def describe(self) -> dict:
return {
"language": self.language,
"scorer_dir": str(self.scorer_dir),
"embedding_dim": self.embedding_dim,
"C": {
"input_dim": self.embedding_dim,
"categories": [self.idx2cat[i] for i in sorted(self.idx2cat)],
},
"A": {
"input_dim": self.relevance_in_dim,
"path_hash_dim": self.path_hash_dim,
"path_feature_weight": self.path_feature_weight,
"path_algorithm": self.path_algorithm,
"path_column": self.path_column,
},
"Q": {
"strategy": "two_stage",
"input_dim": self.quality_in_dim,
"ordinal_num_classes": self.quality_model.ordinal_levels + 1,
"scaler_source": self.quality_scaler.source,
},
}