Download scripts/path_features.py from openbmb/UltraData-Code-L2-Classifier: direct link, hf CLI and curl.
- Browser
- Download file 5.03 kB
-
https://huggingface.co/openbmb/UltraData-Code-L2-Classifier/resolve/main/scripts/path_features.py
- Command line
-
hf download hf://openbmb/UltraData-Code-L2-Classifier/scripts/path_features.py
-
curl -L -o path_features.py https://huggingface.co/openbmb/UltraData-Code-L2-Classifier/resolve/main/scripts/path_features.py
5.03 kB
| """Path featurization for the C / A / Q scorers. | |
| Only the algorithmic-relevance model A consumes the file path. It is turned | |
| into a fixed-width vector by a hashing trick and concatenated to the code | |
| embedding: | |
| features = [ embedding (1024) | path_hash (256) * weight (0.4) ] | |
| The algorithm is named ``crc32_signed_tokens_legacy`` in the checkpoints. | |
| "legacy" is a name inherited from the training pipeline, not a deprecation -- | |
| this is the only path featurizer, and every shipped scorer set uses it. | |
| Two details matter for reproducing training-time behaviour exactly: | |
| * Benchmark names (``leetcode``, ``humaneval``, ``mbpp``, ``codeforces`` ...) | |
| are stripped from the path first, so the model cannot shortcut on them. | |
| * Tokens are a bag of path components plus 3-character prefixes of the longer | |
| ones. Position is discarded, which is what makes it tolerant of repository | |
| layout differences. | |
| This is reproduced verbatim from the training-time implementation so a scorer | |
| set is self-contained. | |
| """ | |
| from __future__ import annotations | |
| import json | |
| import re | |
| import zlib | |
| from typing import Sequence | |
| import numpy as np | |
| PATH_HASH_ALGORITHM = "crc32_signed_tokens_legacy" | |
| # --------------------------------------------------------------------------- # | |
| # read repository-relative paths out of a pyarrow table | |
| # --------------------------------------------------------------------------- # | |
| def _path_from_meta(value: object) -> str: | |
| if isinstance(value, str): | |
| try: | |
| value = json.loads(value) | |
| except (TypeError, ValueError): | |
| return "" | |
| if isinstance(value, dict): | |
| path = value.get("file_path") | |
| return path if isinstance(path, str) else "" | |
| return "" | |
| def extract_relative_paths(table, path_col: str = "relative_path") -> list[str]: | |
| """Read paths from ``path_col``, falling back to ``meta.file_path``.""" | |
| count = table.num_rows | |
| if path_col in table.column_names: | |
| paths = [ | |
| "" if value is None else str(value) | |
| for value in table.column(path_col).to_pylist() | |
| ] | |
| else: | |
| paths = [""] * count | |
| if "meta" in table.column_names and not all(paths): | |
| paths = [ | |
| path or _path_from_meta(meta) | |
| for path, meta in zip(paths, table.column("meta").to_pylist()) | |
| ] | |
| return paths | |
| # --------------------------------------------------------------------------- # | |
| # benchmark-name stripping | |
| # --------------------------------------------------------------------------- # | |
| _SEPARATOR = r"[-_./\\\s]*" | |
| _BENCHMARK_PATTERNS = ( | |
| rf"human{_SEPARATOR}eval(?:{_SEPARATOR}plus|{_SEPARATOR}x)?", | |
| rf"mbpp(?:{_SEPARATOR}plus)?", | |
| r"multipl[-_./\\\s]+e", | |
| rf"ds{_SEPARATOR}1000", | |
| rf"crux{_SEPARATOR}eval", | |
| rf"big{_SEPARATOR}code{_SEPARATOR}bench", | |
| rf"live{_SEPARATOR}code{_SEPARATOR}bench", | |
| rf"code{_SEPARATOR}contests?", | |
| r"leetcode", | |
| r"codeforces", | |
| r"atcoder", | |
| r"acm", | |
| ) | |
| _BENCHMARK_RE = re.compile( | |
| rf"(?<![a-z0-9])(?:{'|'.join(_BENCHMARK_PATTERNS)})(?=$|[^a-z0-9])", | |
| re.IGNORECASE, | |
| ) | |
| _TOKEN_SPLIT = re.compile(r"[/\\._\-]+") | |
| def sanitize_benchmark_path(relative_path: str) -> str: | |
| """Remove benchmark names so the model cannot shortcut on them.""" | |
| path = str(relative_path or "").replace("\\", "/").lower() | |
| return _BENCHMARK_RE.sub("/", path) | |
| # --------------------------------------------------------------------------- # | |
| # hashing trick | |
| # --------------------------------------------------------------------------- # | |
| def path_tokens(relative_path: str) -> list[str]: | |
| """Lowercased path components plus 3-char prefixes of longer tokens.""" | |
| if not relative_path: | |
| return [] | |
| raw = _TOKEN_SPLIT.split(sanitize_benchmark_path(relative_path)) | |
| tokens = [t for t in raw if t] | |
| extra = [t[:3] for t in tokens if len(t) > 3] | |
| return tokens + extra | |
| def path_hash_vector(relative_path: str, dim: int) -> np.ndarray: | |
| """crc32 hashing-trick bag-of-tokens, L2-normalized, shape [dim].""" | |
| vec = np.zeros(dim, dtype=np.float32) | |
| for tok in path_tokens(relative_path): | |
| encoded = tok.encode("utf-8") | |
| bucket = zlib.crc32(encoded) % dim | |
| sign = 1.0 if (zlib.crc32(b"s:" + encoded) & 1) == 0 else -1.0 | |
| vec[bucket] += sign | |
| norm = float(np.linalg.norm(vec)) | |
| if norm > 0.0: | |
| vec /= norm | |
| return vec | |
| def build_feature_matrix( | |
| embeddings: Sequence[Sequence[float]], | |
| paths: Sequence[str], | |
| *, | |
| path_hash_dim: int, | |
| path_feature_weight: float, | |
| ) -> np.ndarray: | |
| """Build ``[ embedding | path_hash * weight ]``, shape [N, D+H].""" | |
| rows = [] | |
| for embedding, path in zip(embeddings, paths): | |
| emb = np.asarray(embedding, dtype=np.float32).reshape(-1) | |
| hashed = path_hash_vector(path, path_hash_dim) * float(path_feature_weight) | |
| rows.append(np.concatenate([emb, hashed.astype(np.float32)])) | |
| if not rows: | |
| return np.zeros((0, 0), dtype=np.float32) | |
| return np.stack(rows).astype(np.float32) | |