CMVR-AI-ANALYSIS/deploy/compose.sglang.yaml

83 lines
1.9 KiB
YAML
Raw Normal View History

services:
qwen38-fast:
container_name: cmvr-qwen38-fast
image: quay.io/gpustack/runner:cuda13.0-sglang0.5.15.post1
restart: unless-stopped
ipc: host
shm_size: 32gb
ports:
- "192.168.28.10:14081:30000"
environment:
SGLANG_USE_MODELSCOPE: "true"
MODELSCOPE_CACHE: /root/.cache/modelscope
MODELSCOPE_DOWNLOAD_PARALLEL_WORKERS: "4"
volumes:
- ../models/modelscope:/root/.cache/modelscope
gpus:
- driver: nvidia
device_ids: ["2"]
capabilities: [gpu]
command:
- python
- -m
- sglang.launch_server
- --model-path
- Qwen/Qwen3.8-27B-FP8
- --served-model-name
- Qwen/Qwen3.8-27B-FP8
- --host
- 0.0.0.0
- --port
- "30000"
- --tp-size
- "1"
- --context-length
- "65536"
- --kv-cache-dtype
- fp8_e4m3
- --mem-fraction-static
- "0.80"
- --reasoning-parser
- qwen3
qwen38-accurate:
container_name: cmvr-qwen38-accurate
image: quay.io/gpustack/runner:cuda13.0-sglang0.5.15.post1
restart: unless-stopped
ipc: host
shm_size: 32gb
ports:
- "192.168.28.10:14082:30000"
environment:
SGLANG_USE_MODELSCOPE: "true"
MODELSCOPE_CACHE: /root/.cache/modelscope
MODELSCOPE_DOWNLOAD_PARALLEL_WORKERS: "4"
volumes:
- ../models/modelscope:/root/.cache/modelscope
gpus:
- driver: nvidia
device_ids: ["3"]
capabilities: [gpu]
command:
- python
- -m
- sglang.launch_server
- --model-path
- Qwen/Qwen3.8-27B
- --served-model-name
- Qwen/Qwen3.8-27B
- --host
- 0.0.0.0
- --port
- "30000"
- --tp-size
- "1"
- --context-length
- "65536"
- --kv-cache-dtype
- fp8_e4m3
- --mem-fraction-static
- "0.84"
- --reasoning-parser
- qwen3