File size: 10,583 Bytes
58258b8 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 253 254 255 256 257 258 259 260 261 262 263 264 265 266 267 268 269 270 271 272 273 274 275 | #!/usr/bin/env python3
# Copyright The Marin Authors
# SPDX-License-Identifier: Apache-2.0
"""Patch a cloned evalchemy checkout for single-GPU offline vLLM evaluation.
This is the GPU analogue of the runtime patching that
``lib/marin/src/marin/evaluation/evaluators/evalchemy_evaluator.py`` does on TPU.
The TPU-only patches (per-request seed stripping, fake ``vllm-tpu`` version
metadata, ``MODEL_IMPL_TYPE=vllm``, GCS ``AutoConfig`` redirection) are dropped;
two GPU-specific ones are added.
Patches applied (all idempotent, all fail loudly if their marker is missing so a
pin bump cannot silently no-op):
1. ``eval/eval_tracker.py`` -- lm-eval ``eval_logger`` import shim.
2. ``eval/eval.py`` -- ``utils.eval_logger`` attribute shim.
3. ``*/eval_instruct.py`` -- ``self.n_repeat`` becomes env-driven
(``EVALCHEMY_N_REPEAT``) so one vLLM engine can produce all repetitions.
4. ``*/livecodebench_utils.py`` -- fork -> spawn for the sandboxed test runner.
REQUIRED on GPU: CUDA context and vLLM/filelock state are not fork-safe and a
forked child either deadlocks or corrupts the parent's CUDA context.
5. ``LiveCodeBench*/eval_instruct.py`` -- dataset ``num_proc`` becomes env-driven
(``EVALCHEMY_NUM_PROC``) so LCB's dataset filter does not oversubscribe the
box while a training job is running.
6. Writes ``_marin_gpu_entry.py`` -- the launcher the driver invokes. It applies
the ``vllm.utils.get_open_port`` compatibility shim and lifts Python's
int->str digit cap before delegating to ``eval.eval``.
Usage:
python patch_evalchemy_gpu.py /opt/evalchemy
"""
from __future__ import annotations
import argparse
import logging
import re
import sys
from pathlib import Path
logger = logging.getLogger("patch_evalchemy_gpu")
# Every benchmark that carries a ``self.n_repeat`` assignment at the pinned
# commits. Mirrors N_REPEAT_BENCHMARK_PATHS in Marin's TPU evaluator.
N_REPEAT_BENCHMARKS = (
"AIME24",
"AIME25",
"AIME26",
"AMC23",
"HMMT",
"GPQADiamond",
"JEEBench",
"HLE",
"LiveCodeBench",
"LiveCodeBenchv5_official",
"LiveCodeBenchv6_official",
"CodeForces",
"CodeElo",
)
# Benchmarks with no ``n_repeat`` at all -- single pass by construction.
SINGLE_PASS_BENCHMARKS = ("MATH500", "OlympiadBench", "OlympiadBench_Physics")
LCB_BENCHMARKS = ("LiveCodeBench", "LiveCodeBenchv5_official", "LiveCodeBenchv6_official")
LCB_UTILITY_FILES = (
"eval/chat_benchmarks/LiveCodeBench/livecodebench_utils.py",
"eval/chat_benchmarks/LiveCodeBenchv5/livecodebench_utils.py",
"eval/chat_benchmarks/LiveCodeBenchv5_official/livecodebench_utils.py",
"eval/chat_benchmarks/LiveCodeBenchv6_official/livecodebench_utils.py",
)
OLD_TRACKER_IMPORT = (
"from lm_eval.utils import eval_logger, handle_non_serializable, hash_string, simple_parse_args_string"
)
NEW_TRACKER_IMPORT = '''try:
from lm_eval.utils import eval_logger, handle_non_serializable, hash_string, simple_parse_args_string
except ImportError:
try:
from lm_eval.logging_utils import eval_logger
from lm_eval.utils import handle_non_serializable, hash_string, simple_parse_args_string
except ImportError:
import logging
eval_logger = logging.getLogger("lm-eval")
from lm_eval.utils import handle_non_serializable, hash_string, simple_parse_args_string'''
EVAL_PY_MARKER = "from lm_eval.tasks import TaskManager as PretrainTaskManager"
EVAL_PY_LOGGER_PATCH = '''
# Patch utils.eval_logger for lm-eval compatibility (marin gpu port)
if not hasattr(utils, "eval_logger"):
try:
from lm_eval.logging_utils import eval_logger as _eval_logger
utils.eval_logger = _eval_logger
except ImportError:
import logging as _logging
utils.eval_logger = _logging.getLogger("lm-eval")
'''
OLD_LCB_FORK = """ manager = multiprocessing.Manager()
result = manager.list()
p = multiprocessing.Process(target=run_tests_for_one_example, args=(test_cases, completion, result, is_extracted))"""
NEW_LCB_SPAWN = """ # marin gpu port: CUDA/vLLM/filelock state is not safe to inherit through
# fork. A forked grader child either deadlocks on the CUDA driver mutex or
# corrupts the parent engine's context. Spawn is the only safe context here.
multiprocessing_context = multiprocessing.get_context("spawn")
manager = multiprocessing_context.Manager()
result = manager.list()
p = multiprocessing_context.Process(
target=run_tests_for_one_example,
args=(test_cases, completion, result, is_extracted),
)"""
ENTRY_MODULE = '''#!/usr/bin/env python3
"""Launcher for evalchemy under a local CUDA vLLM install (marin gpu port).
Generated by patch_evalchemy_gpu.py -- do not edit by hand.
Applies two process-wide fixes that must happen before ``eval.eval`` imports
lm-eval, then delegates. Invoked as::
python _marin_gpu_entry.py --model vllm --tasks AIME24 ...
"""
import runpy
import socket
import sys
# HMMT and some OlympiadBench answers are enormous integers (e.g. (26!)^3).
# json.dumps calls str() on them; without this the results file is never written
# and the eval exits 0 with no output.
sys.set_int_max_str_digits(0)
# lm-eval's vllm_causallms imports get_open_port from vllm.utils, which vLLM
# moved/removed in newer releases. Re-provide it before lm-eval is imported.
try:
import vllm.utils as _vllm_utils
except ImportError:
_vllm_utils = None
if _vllm_utils is not None and not hasattr(_vllm_utils, "get_open_port"):
def _get_open_port() -> int:
with socket.socket(socket.AF_INET, socket.SOCK_STREAM) as sock:
sock.bind(("", 0))
return sock.getsockname()[1]
_vllm_utils.get_open_port = _get_open_port
sys.argv = ["eval.eval", *sys.argv[1:]]
runpy.run_module("eval.eval", run_name="__main__", alter_sys=True)
'''
def _patch_eval_tracker(root: Path) -> None:
path = root / "eval" / "eval_tracker.py"
source = path.read_text()
if "from lm_eval.logging_utils import eval_logger" in source:
logger.info("eval_tracker.py already patched")
return
if OLD_TRACKER_IMPORT not in source:
raise RuntimeError(f"marker missing in {path}: lm-eval import shim cannot be applied")
path.write_text(source.replace(OLD_TRACKER_IMPORT, NEW_TRACKER_IMPORT, 1))
logger.info("patched %s", path)
def _patch_eval_py(root: Path) -> None:
path = root / "eval" / "eval.py"
source = path.read_text()
if "# Patch utils.eval_logger" in source:
logger.info("eval.py already patched")
return
if EVAL_PY_MARKER not in source:
raise RuntimeError(f"marker missing in {path}: eval_logger shim cannot be applied")
path.write_text(source.replace(EVAL_PY_MARKER, EVAL_PY_MARKER + EVAL_PY_LOGGER_PATCH, 1))
logger.info("patched %s", path)
def _patch_n_repeat(root: Path) -> list[str]:
patched: list[str] = []
for name in N_REPEAT_BENCHMARKS:
path = root / "eval" / "chat_benchmarks" / name / "eval_instruct.py"
if not path.exists():
logger.warning("benchmark %s not present at this pin; skipping n_repeat patch", name)
continue
source = path.read_text()
if "EVALCHEMY_N_REPEAT" in source:
patched.append(name)
continue
# __import__ rather than a bare ``os`` reference: several benchmark
# modules do not import os, and inserting a top-level import is fragile
# across pin bumps (it would shadow a module docstring).
new_source, count = re.subn(
r"self\.n_repeat\s*=\s*\d+",
'self.n_repeat = int(__import__("os").environ.get("EVALCHEMY_N_REPEAT", "1"))',
source,
count=1,
)
if count != 1:
raise RuntimeError(f"no 'self.n_repeat = <int>' assignment found in {path}")
path.write_text(new_source)
patched.append(name)
logger.info("patched n_repeat in %s", path)
return patched
def _patch_lcb_num_proc(root: Path) -> None:
for name in LCB_BENCHMARKS:
path = root / "eval" / "chat_benchmarks" / name / "eval_instruct.py"
if not path.exists():
continue
source = path.read_text()
if "EVALCHEMY_NUM_PROC" in source:
continue
if "num_proc=cpu_count" not in source:
raise RuntimeError(f"'num_proc=cpu_count' not found in {path}")
source = source.replace(
"num_proc=cpu_count",
'num_proc=int(__import__("os").environ.get("EVALCHEMY_NUM_PROC", str(cpu_count)))',
1,
)
path.write_text(source)
logger.info("patched num_proc in %s", path)
def _patch_lcb_spawn(root: Path) -> None:
for relative in LCB_UTILITY_FILES:
path = root / relative
if not path.exists():
logger.warning("%s not present at this pin; skipping spawn patch", relative)
continue
source = path.read_text()
if "multiprocessing_context = multiprocessing.get_context" in source:
continue
if OLD_LCB_FORK not in source:
raise RuntimeError(f"LiveCodeBench fork block not found in {path}")
path.write_text(source.replace(OLD_LCB_FORK, NEW_LCB_SPAWN, 1))
logger.info("patched fork->spawn in %s", path)
def _write_entry(root: Path) -> Path:
path = root / "_marin_gpu_entry.py"
path.write_text(ENTRY_MODULE)
path.chmod(0o755)
logger.info("wrote %s", path)
return path
def patch(root: Path) -> None:
if not (root / "eval" / "eval.py").exists():
raise RuntimeError(f"{root} does not look like an evalchemy checkout (no eval/eval.py)")
_patch_eval_tracker(root)
_patch_eval_py(root)
patched = _patch_n_repeat(root)
_patch_lcb_num_proc(root)
_patch_lcb_spawn(root)
_write_entry(root)
logger.info("n_repeat is env-driven for: %s", ", ".join(patched))
logger.info("single-pass benchmarks (no n_repeat): %s", ", ".join(SINGLE_PASS_BENCHMARKS))
logger.info("evalchemy at %s is ready for single-GPU vLLM evaluation", root)
def main() -> int:
logging.basicConfig(level=logging.INFO, format="%(levelname)s %(message)s")
parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
parser.add_argument("evalchemy_dir", type=Path, help="Path to the cloned evalchemy checkout")
args = parser.parse_args()
patch(args.evalchemy_dir.resolve())
return 0
if __name__ == "__main__":
sys.exit(main())
|