Download scripts/scorers.py from openbmb/UltraData-Code-L2-Classifier: direct link, hf CLI and curl.
- Browser
- Download file 13.3 kB
-
https://huggingface.co/openbmb/UltraData-Code-L2-Classifier/resolve/main/scripts/scorers.py
- Command line
-
hf download hf://openbmb/UltraData-Code-L2-Classifier/scripts/scorers.py
-
curl -L -o scorers.py https://huggingface.co/openbmb/UltraData-Code-L2-Classifier/resolve/main/scripts/scorers.py
13.3 kB
| """Loaders and inference for the C / A / Q code-selection scorers. | |
| A scorer set is a flat directory, one per language, such as ``classifiers/cpp``:: | |
| <lang>/ | |
| C_file_role.bin C_file_role.json # classifier + its class names | |
| A_relevance.bin # checkpoint carries its own config | |
| Q_quality.bin # architecture read from its shapes | |
| Q_scaler.npz # Q's StandardScaler (mean / scale) | |
| All five files are required. Layer widths are recovered from the checkpoints' | |
| own tensor shapes rather than declared in config, so the only metadata shipped | |
| alongside the weights is ``C_file_role.json`` — the index-to-category map, the | |
| one thing the weights cannot carry. | |
| ``LanguageScorers.load(dir)`` reads the artifacts and configures itself from | |
| what they declare; nothing about the featurization is hard-coded per language. | |
| All three models consume a precomputed code embedding. Embeddings must come | |
| from the same encoder and dimension used at training time — the loader checks | |
| the dimension, but cannot detect a wrong encoder at the right dimension. | |
| """ | |
| from __future__ import annotations | |
| import json | |
| from pathlib import Path | |
| import numpy as np | |
| import torch | |
| import torch.nn as nn | |
| from path_features import PATH_HASH_ALGORITHM, build_feature_matrix | |
| # --------------------------------------------------------------------------- # | |
| # architectures (must match the training scripts' saved state_dicts) | |
| # --------------------------------------------------------------------------- # | |
| class FileRoleClassifier(nn.Module): | |
| """C: Linear/LayerNorm/GELU/Dropout stack -> num_classes logits.""" | |
| def __init__(self, input_dim, hidden_dims, num_classes, dropout=0.3): | |
| super().__init__() | |
| layers = [] | |
| prev = input_dim | |
| for i, h in enumerate(hidden_dims): | |
| layers += [ | |
| nn.Linear(prev, h), | |
| nn.LayerNorm(h), | |
| nn.GELU(), | |
| nn.Dropout(dropout if i == 0 else dropout * 0.7), | |
| ] | |
| prev = h | |
| layers.append(nn.Linear(prev, num_classes)) | |
| self.net = nn.Sequential(*layers) | |
| def forward(self, x): | |
| return self.net(x) | |
| class RelevanceMLP(nn.Module): | |
| """A: two hidden ReLU layers -> single logit.""" | |
| def __init__(self, in_dim, hidden1, hidden2): | |
| super().__init__() | |
| self.fc1 = nn.Linear(in_dim, hidden1) | |
| self.fc2 = nn.Linear(hidden1, hidden2) | |
| self.fc3 = nn.Linear(hidden2, 1) | |
| def forward(self, x): | |
| x = torch.relu(self.fc1(x)) | |
| x = torch.relu(self.fc2(x)) | |
| return self.fc3(x).squeeze(-1) | |
| class _QualityBackbone(nn.Module): | |
| """Q's shared trunk. | |
| Kept as its own module holding a ``self.net`` Sequential because the saved | |
| state_dicts are keyed ``backbone.net.0.*``. Inlining the Sequential into | |
| QualityMLP would renumber the keys to ``backbone.0.*`` and fail to load. | |
| """ | |
| def __init__(self, in_dim, hidden_dim, dropout): | |
| super().__init__() | |
| mid = max(64, hidden_dim // 2) | |
| self.net = nn.Sequential( | |
| nn.Linear(in_dim, hidden_dim), | |
| nn.GELU(), | |
| nn.LayerNorm(hidden_dim), | |
| nn.Dropout(dropout), | |
| nn.Linear(hidden_dim, mid), | |
| nn.GELU(), | |
| nn.LayerNorm(mid), | |
| nn.Dropout(dropout), | |
| ) | |
| def forward(self, x): | |
| return self.net(x) | |
| class QualityMLP(nn.Module): | |
| """Q: shared trunk -> zero-score head + 1..10 ordinal head. | |
| Final score is ``(1 - sigmoid(zero)) * (1 + sum(sigmoid(ordinal)))``: the | |
| zero head gates out config/data/no-logic files, the ordinal head places | |
| everything else on 1..10. | |
| """ | |
| def __init__(self, in_dim, hidden_dim=512, dropout=0.1, ordinal_num_classes=10): | |
| super().__init__() | |
| mid = max(64, hidden_dim // 2) | |
| self.backbone = _QualityBackbone(in_dim, hidden_dim, dropout) | |
| self.zero_head = nn.Linear(mid, 1) | |
| self.ordinal_levels = ordinal_num_classes - 1 | |
| self.ordinal_head = nn.Linear(mid, self.ordinal_levels) | |
| def forward(self, x): | |
| features = self.backbone(x) | |
| return self.zero_head(features).squeeze(-1), self.ordinal_head(features) | |
| # --------------------------------------------------------------------------- # | |
| # standard scaler without an sklearn version dependency | |
| # --------------------------------------------------------------------------- # | |
| class _Scaler: | |
| """(x - mean) / scale, loaded from ``Q_scaler.npz``.""" | |
| def __init__(self, mean: np.ndarray, scale: np.ndarray, source: str): | |
| self.mean = mean.astype(np.float32) | |
| self.scale = scale.astype(np.float32) | |
| self.n_features_in = int(self.mean.size) | |
| self.source = source | |
| def load(cls, scorer_dir: Path) -> "_Scaler": | |
| npz = scorer_dir / "Q_scaler.npz" | |
| payload = np.load(npz) | |
| return cls(payload["mean"], payload["scale"], npz.name) | |
| def transform(self, x: np.ndarray) -> np.ndarray: | |
| return ((x - self.mean) / self.scale).astype(np.float32) | |
| # --------------------------------------------------------------------------- # | |
| # scorer set | |
| # --------------------------------------------------------------------------- # | |
| class LanguageScorers: | |
| """Loads one language's scorer set and scores batches of embeddings.""" | |
| def __init__(self, scorer_dir, device="cpu"): | |
| self.scorer_dir = Path(scorer_dir) | |
| self.device = torch.device(device) | |
| self.language = self.scorer_dir.name | |
| # -- loading ---------------------------------------------------------- # | |
| def load(cls, scorer_dir, device="cpu") -> "LanguageScorers": | |
| self = cls(scorer_dir, device) | |
| self._load_file_role() | |
| self._load_relevance() | |
| self._load_quality() | |
| if self.quality_in_dim != self.embedding_dim: | |
| raise ValueError( | |
| f"Q expects {self.quality_in_dim}-d input but C implies " | |
| f"{self.embedding_dim}-d embeddings" | |
| ) | |
| return self | |
| def _load_file_role(self) -> None: | |
| cfg = json.loads((self.scorer_dir / "C_file_role.json").read_text()) | |
| state = torch.load( | |
| self.scorer_dir / "C_file_role.bin", map_location=self.device | |
| ) | |
| # Layer widths come from the checkpoint itself: every Linear weight is | |
| # [out, in], so the stack of 2-D weights gives input_dim, the hidden | |
| # widths and num_classes. Dropout is inactive under eval(). | |
| linears = [v.shape for k, v in state.items() | |
| if k.endswith("weight") and v.ndim == 2] | |
| input_dim = int(linears[0][1]) | |
| hidden_dims = [int(s[0]) for s in linears[:-1]] | |
| num_classes = int(linears[-1][0]) | |
| model = FileRoleClassifier(input_dim, hidden_dims, num_classes) | |
| model.load_state_dict(state) | |
| self.file_role_model = model.eval().to(self.device) | |
| self.embedding_dim = input_dim | |
| # The one thing weights cannot carry: which class each index means. | |
| self.idx2cat = {int(k): v for k, v in cfg["idx2cat"].items()} | |
| if len(self.idx2cat) != num_classes: | |
| raise ValueError( | |
| f"C_file_role.json maps {len(self.idx2cat)} categories but the " | |
| f"checkpoint has {num_classes} output units" | |
| ) | |
| def _load_relevance(self) -> None: | |
| # Named explicitly rather than globbed: in a flat scorer-set directory a | |
| # glob for "*.bin" would also match the C and Q checkpoints. | |
| ckpt = torch.load( | |
| self.scorer_dir / "A_relevance.bin", map_location=self.device | |
| ) | |
| arch = ckpt["arch"] | |
| model = RelevanceMLP(arch["in_dim"], arch["hidden1"], arch["hidden2"]) | |
| model.load_state_dict(ckpt["state_dict"]) | |
| self.relevance_model = model.eval().to(self.device) | |
| self.relevance_in_dim = int(arch["in_dim"]) | |
| cfg = ckpt["feature_config"] | |
| if not cfg.get("use_path_feature", False): | |
| raise ValueError( | |
| "this A checkpoint declares no path features; every scorer set " | |
| "in this release feeds A a 1024-d embedding plus a 256-d path hash" | |
| ) | |
| self.path_hash_dim = int(cfg["path_hash_dim"]) | |
| self.path_feature_weight = float(cfg["path_feature_weight"]) | |
| self.path_column = "relative_path" | |
| self.path_algorithm = PATH_HASH_ALGORITHM | |
| expected = self.embedding_dim + self.path_hash_dim | |
| if expected != self.relevance_in_dim: | |
| raise ValueError( | |
| f"A expects {self.relevance_in_dim}-d input but the configured " | |
| f"featurization produces {expected}-d" | |
| ) | |
| def _load_quality(self) -> None: | |
| self.quality_scaler = _Scaler.load(self.scorer_dir) | |
| state = torch.load(self.scorer_dir / "Q_quality.bin", map_location=self.device) | |
| # Shapes carry the architecture: the trunk's first Linear is | |
| # [hidden_dim, in_dim] and the ordinal head is [num_classes - 1, mid]. | |
| # No config file is needed, and load_state_dict below is the real guard | |
| # — a differently-headed checkpoint fails on key names, not on a flag. | |
| in_dim = self.quality_scaler.n_features_in or int( | |
| state["backbone.net.0.weight"].shape[1] | |
| ) | |
| hidden_dim = int(state["backbone.net.0.weight"].shape[0]) | |
| ordinal_num_classes = int(state["ordinal_head.weight"].shape[0]) + 1 | |
| model = QualityMLP(in_dim, hidden_dim, ordinal_num_classes=ordinal_num_classes) | |
| model.load_state_dict(state) | |
| self.quality_model = model.eval().to(self.device) | |
| self.quality_in_dim = in_dim | |
| # -- inference -------------------------------------------------------- # | |
| def predict_file_role(self, embeddings: np.ndarray): | |
| with torch.no_grad(): | |
| logits = self.file_role_model(torch.from_numpy(embeddings).to(self.device)) | |
| probs = torch.softmax(logits, dim=1) | |
| confidence, index = probs.max(dim=1) | |
| roles = [self.idx2cat.get(int(i), "EXCLUDE") for i in index.cpu().numpy()] | |
| return roles, confidence.cpu().numpy() | |
| def predict_relevance(self, embeddings: np.ndarray, paths) -> np.ndarray: | |
| features = build_feature_matrix( | |
| embeddings, | |
| paths, | |
| path_hash_dim=self.path_hash_dim, | |
| path_feature_weight=self.path_feature_weight, | |
| ) | |
| with torch.no_grad(): | |
| tensor = torch.from_numpy(np.ascontiguousarray(features, dtype=np.float32)) | |
| logits = self.relevance_model(tensor.to(self.device)) | |
| scores = torch.sigmoid(logits).cpu().numpy() | |
| return np.asarray(scores).reshape(-1) | |
| def predict_quality(self, embeddings: np.ndarray) -> np.ndarray: | |
| scaled = self.quality_scaler.transform(embeddings) | |
| with torch.no_grad(): | |
| zero_logit, ordinal_logits = self.quality_model( | |
| torch.from_numpy(scaled).to(self.device) | |
| ) | |
| p_zero = torch.sigmoid(zero_logit) | |
| expected_positive = 1.0 + torch.sigmoid(ordinal_logits).sum(dim=1) | |
| out = (1.0 - p_zero) * expected_positive | |
| return np.asarray(out.cpu().numpy()).reshape(-1) | |
| def score(self, embeddings, paths=None) -> dict: | |
| """Score a batch. Returns the four columns as numpy arrays / lists.""" | |
| embeddings = np.ascontiguousarray(embeddings, dtype=np.float32) | |
| if embeddings.ndim != 2 or embeddings.shape[1] != self.embedding_dim: | |
| raise ValueError( | |
| f"expected embeddings of shape [N, {self.embedding_dim}], " | |
| f"got {tuple(embeddings.shape)}" | |
| ) | |
| if paths is None: | |
| paths = [""] * len(embeddings) | |
| roles, confidence = self.predict_file_role(embeddings) | |
| return { | |
| "category": roles, | |
| "cls_confidence": confidence, | |
| "algo_rel_score": self.predict_relevance(embeddings, paths), | |
| "quality_score": self.predict_quality(embeddings), | |
| } | |
| def needs_paths(self) -> bool: | |
| """True when any model consumes the file path. Always true: A does.""" | |
| return self.path_hash_dim > 0 | |
| def describe(self) -> dict: | |
| return { | |
| "language": self.language, | |
| "scorer_dir": str(self.scorer_dir), | |
| "embedding_dim": self.embedding_dim, | |
| "C": { | |
| "input_dim": self.embedding_dim, | |
| "categories": [self.idx2cat[i] for i in sorted(self.idx2cat)], | |
| }, | |
| "A": { | |
| "input_dim": self.relevance_in_dim, | |
| "path_hash_dim": self.path_hash_dim, | |
| "path_feature_weight": self.path_feature_weight, | |
| "path_algorithm": self.path_algorithm, | |
| "path_column": self.path_column, | |
| }, | |
| "Q": { | |
| "strategy": "two_stage", | |
| "input_dim": self.quality_in_dim, | |
| "ordinal_num_classes": self.quality_model.ordinal_levels + 1, | |
| "scaler_source": self.quality_scaler.source, | |
| }, | |
| } | |