-
Notifications
You must be signed in to change notification settings - Fork 83
Expand file tree
/
Copy pathgemma4-server.yaml
More file actions
114 lines (112 loc) · 3.13 KB
/
Copy pathgemma4-server.yaml
File metadata and controls
114 lines (112 loc) · 3.13 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
apiVersion: v1
kind: Namespace
metadata:
name: gemma4-test
---
# TODO: Update the storage class if required (or if using a standard GKE storage class)
apiVersion: storage.k8s.io/v1
kind: StorageClass
metadata:
name: hyperdisk-balanced-tpu
provisioner: pd.csi.storage.gke.io
parameters:
type: hyperdisk-balanced
reclaimPolicy: Delete
volumeBindingMode: WaitForFirstConsumer
allowVolumeExpansion: true
---
apiVersion: v1
kind: PersistentVolumeClaim
metadata:
name: hd-claim-v2
namespace: gemma4-test
spec:
storageClassName: hyperdisk-balanced-tpu
accessModes:
- ReadWriteOnce
resources:
requests:
storage: 100Gi
---
apiVersion: apps/v1
kind: Deployment
metadata:
name: vllm-tpu
namespace: gemma4-test
spec:
replicas: 1
selector:
matchLabels:
app: vllm-tpu
template:
metadata:
labels:
app: vllm-tpu
spec:
# TODO: Update the topology to match your allocated TPU node pool (e.g., 2x2x1 is 4 chips)
nodeSelector:
cloud.google.com/gke-tpu-accelerator: tpu7x
cloud.google.com/gke-tpu-topology: 2x2x1
containers:
- name: vllm-tpu
image: vllm/vllm-tpu:nightly-20260604-2aaa59e-4f423bd
command: ["python3", "-m", "vllm.entrypoints.openai.api_server"]
args:
- --host=0.0.0.0
- --port=8000
- --seed=42
# TODO: Update the model if you wish to switch to MoE or another variant
- --model=google/gemma-4-31B-it
- --max-model-len=1524
- --max-num-seqs=256
- --data-parallel-size=4
- --tensor-parallel-size=2
- --max-num-batched-tokens=16384
- --no-enable-prefix-caching
- --kv-cache-dtype=fp8
- --gpu-memory-utilization=0.9
- --async-scheduling
- "--limit-mm-per-prompt={\"image\": 0, \"video\": 0, \"audio\": 0}"
- --block-size=256
- "--additional-config={\"quantization\": { \"qwix\": { \"rules\": [{ \"module_path\": \".*\", \"weight_qtype\": \"float8_e4m3fn\", \"act_qtype\": \"float8_e4m3fn\"}]}}}"
env:
- name: VLLM_ENGINE_READY_TIMEOUT_S
value: "1800"
- name: TPU_MULTIPROCESS_DP
value: "1"
- name: USE_BATCHED_RPA_KERNEL
value: "1"
- name: MODEL_IMPL_TYPE
value: "flax_nnx"
- name: HF_HOME
value: /data
- name: HF_TOKEN
valueFrom:
secretKeyRef:
# TODO: Update the secret name if you use a custom name
name: hf-secret
key: hf_api_token
ports:
- containerPort: 8000
resources:
limits:
google.com/tpu: '4' # Number of chips required by topology
requests:
google.com/tpu: '4'
readinessProbe:
tcpSocket:
port: 8000
initialDelaySeconds: 15
periodSeconds: 10
volumeMounts:
- mountPath: "/data"
name: data-volume
- mountPath: /dev/shm
name: dshm
volumes:
- emptyDir:
medium: Memory
name: dshm
- name: data-volume
persistentVolumeClaim:
claimName: hd-claim-v2