patdev commited on
Commit
d68b399
·
verified ·
1 Parent(s): b0a3b0b

banc adapte a l image publique vllm

Browse files
Files changed (1) hide show
  1. banc_ada_controle.py +3 -3
banc_ada_controle.py CHANGED
@@ -26,6 +26,7 @@ Reperes, meme depot `NVIDIA-Nemotron-3.5-Lightning-30B-A3B-NVFP4` :
26
  RTX 6000 Ada, notre pod, ctx 1M, all : 238,5 | 911 @16
27
  """
28
  import json
 
29
  import re
30
  import statistics
31
  import subprocess
@@ -34,7 +35,7 @@ import time
34
  import urllib.request
35
 
36
  MODEL = "nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-NVFP4"
37
- PY = "/opt/venv/bin/python"
38
  PORT = 8000
39
  URL = "http://127.0.0.1:%d" % PORT
40
 
@@ -54,8 +55,7 @@ dire("=" * 78)
54
 
55
  # Recette NVIDIA mot pour mot, contexte 131072 : la MEME que le job Blackwell.
56
  # Toute divergence ici invaliderait le controle.
57
- BASE = [PY, "-m", "vllm.entrypoints.openai.api_server",
58
- "--model", MODEL,
59
  "--served-model-name", "ornith",
60
  "--host", "127.0.0.1", "--port", str(PORT),
61
  "--trust-remote-code",
 
26
  RTX 6000 Ada, notre pod, ctx 1M, all : 238,5 | 911 @16
27
  """
28
  import json
29
+ import sys
30
  import re
31
  import statistics
32
  import subprocess
 
35
  import urllib.request
36
 
37
  MODEL = "nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-NVFP4"
38
+ PY = sys.executable # image publique : pas de /opt/venv
39
  PORT = 8000
40
  URL = "http://127.0.0.1:%d" % PORT
41
 
 
55
 
56
  # Recette NVIDIA mot pour mot, contexte 131072 : la MEME que le job Blackwell.
57
  # Toute divergence ici invaliderait le controle.
58
+ BASE = ["vllm", "serve", MODEL,
 
59
  "--served-model-name", "ornith",
60
  "--host", "127.0.0.1", "--port", str(PORT),
61
  "--trust-remote-code",