Repository navigation
Expand file tree
/
Copy pathfleet.example.json
More file actions
87 lines (87 loc) · 1.94 KB
/
Copy pathfleet.example.json
File metadata and controls
87 lines (87 loc) · 1.94 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
{
"version": 1,
"api": {
"listen": "127.0.0.1:8090"
},
"runtime": {
"docker_binary": "docker",
"poll_interval": "2s",
"operation_timeout": "30s",
"readiness_timeout": "10m",
"model_port_range": {
"start": 8000,
"end": 8000
},
"remove_on_unload": true
},
"models": [
{
"id": "qwen3-8b",
"description": "Qwen3 8B served by vLLM",
"image": "vllm/vllm-openai:latest",
"command": [
"Qwen/Qwen3-8B",
"--host", "0.0.0.0",
"--port", "8000",
"--gpu-memory-utilization", "0.90"
],
"environment": {
"HF_HOME": "/root/.cache/huggingface"
},
"mounts": [
{
"source": "/var/lib/lil-fleet/huggingface",
"target": "/root/.cache/huggingface"
}
],
"ports": [
{
"host_ip": "127.0.0.1",
"host_port": 8000,
"container_port": 8000
}
],
"gpus": "all",
"shm_size": "16g",
"ipc": "host",
"readiness": {
"url": "http://127.0.0.1:8000/health",
"success_status": 200
}
},
{
"id": "mistral-7b",
"description": "Mistral 7B served by vLLM",
"image": "vllm/vllm-openai:latest",
"command": [
"mistralai/Mistral-7B-Instruct-v0.3",
"--host", "0.0.0.0",
"--port", "8000",
"--gpu-memory-utilization", "0.90"
],
"environment": {
"HF_HOME": "/root/.cache/huggingface"
},
"mounts": [
{
"source": "/var/lib/lil-fleet/huggingface",
"target": "/root/.cache/huggingface"
}
],
"ports": [
{
"host_ip": "127.0.0.1",
"host_port": 8000,
"container_port": 8000
}
],
"gpus": "all",
"shm_size": "16g",
"ipc": "host",
"readiness": {
"url": "http://127.0.0.1:8000/health",
"success_status": 200
}
}
]
}