Spaces:
Running
Running
File size: 756 Bytes
0a2e8cf | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 | # One GPU: vLLM serves the weights, openjev-server exposes the decision API on :3000.
services:
vllm:
image: vllm/vllm-openai:v0.29.0
command: >
--model openjev/openjev-FP8 --served-model-name qwen --port 8000 --enable-prefix-caching --max-model-len 16384
--gpu-memory-utilization 0.90 --limit-mm-per-prompt '{"image":1}' --trust-remote-code --max-num-seqs 64
--max-logprobs 64 --gdn-prefill-backend triton --quantization fp8
deploy: { resources: { reservations: { devices: [{ driver: nvidia, count: 1, capabilities: [gpu] }] } } }
volumes: ["~/.cache/huggingface:/root/.cache/huggingface"]
openjev:
build: .
depends_on: [vllm]
ports: ["3000:3000"]
environment:
OPENJEV_TOKEN: ${OPENJEV_TOKEN:-}
|