diff --git a/charts/vllm/Chart.yaml b/charts/vllm/Chart.yaml new file mode 100644 index 0000000..bf00f14 --- /dev/null +++ b/charts/vllm/Chart.yaml @@ -0,0 +1,7 @@ +apiVersion: v2 +description: The stack deployment of vLLM +maintainers: +- name: apostac +name: vllm-stack +type: application +version: 0.0.11 diff --git a/charts/vllm/README.md b/charts/vllm/README.md new file mode 100644 index 0000000..c05eaf0 --- /dev/null +++ b/charts/vllm/README.md @@ -0,0 +1,27 @@ +# vLLM Production Stack helm chart + +This helm chart lets users deploy multiple serving engines and a router into the Kubernetes cluster. + +## Key features + +- Support running multiple serving engines with multiple different models +- Load the model weights directly from the existing PersistentVolumes + +## Prerequisites + +1. A running Kubernetes cluster with GPU. (You can set it up through `minikube`: ) +2. [Helm](https://helm.sh/docs/intro/install/) + +## Install the helm chart + +```bash +helm install llmstack . -f values-example.yaml +``` + +## Uninstall the deployment + +run `helm uninstall llmstack` + +## Configure the deployments + +See `helm/values.yaml` for mode details. diff --git a/charts/vllm/custom-values.yaml b/charts/vllm/custom-values.yaml new file mode 100644 index 0000000..cefca80 --- /dev/null +++ b/charts/vllm/custom-values.yaml @@ -0,0 +1,37 @@ +servingEngineSpec: + modelSpec: [] + resources: + requests: + cpu: "4" + memory: "16G" + limits: + cpu: "8" + memory: "32G" + +routerSpec: + # -- The docker image of the router. The following values are defaults: + repository: "lmcache/lmstack-router" + tag: "latest" + resources: + requests: + cpu: "2" + memory: "8G" + limits: + cpu: "4" + memory: "16G" + ingress: + enabled: true + className: "" + annotations: + cert-manager.io/cluster-issuer: "selfsigned-issuer" + cert-manager.io/duration: 8760h + cert-manager.io/renew-before: 720h + hosts: + - host: vllm.example.com + paths: + - path: / + pathType: Prefix + tls: + - secretName: vllm-tls-secret + hosts: + - vllm.example.com diff --git a/charts/vllm/templates/_helpers.tpl b/charts/vllm/templates/_helpers.tpl new file mode 100644 index 0000000..43ef2e5 --- /dev/null +++ b/charts/vllm/templates/_helpers.tpl @@ -0,0 +1,142 @@ +{{/* +Define ports for the pods +*/}} +{{- define "chart.container-port" -}} +{{- default "8000" .Values.servingEngineSpec.containerPort }} +{{- end }} + +{{/* +Define service port +*/}} +{{- define "chart.service-port" -}} +{{- if .Values.servingEngineSpec.servicePort }} +{{- .Values.servingEngineSpec.servicePort }} +{{- else }} +{{- include "chart.container-port" . }} +{{- end }} +{{- end }} + +{{/* +Define service port name +*/}} +{{- define "chart.service-port-name" -}} +"service-port" +{{- end }} + +{{/* +Define container port name +*/}} +{{- define "chart.container-port-name" -}} +"container-port" +{{- end }} + +{{/* +Define engine deployment strategy. +If .Values.engineStrategy is defined, use it. +Otherwise, fall back to the default rolling update strategy. +*/}} +{{- define "chart.engineStrategy" -}} +strategy: +{{- if .Values.servingEngineSpec.strategy }} +{{- toYaml .Values.servingEngineSpec.strategy | nindent 2 }} +{{- else }} + rollingUpdate: + maxSurge: 100% + maxUnavailable: 0 +{{- end }} +{{- end }} + +{{/* +Define router deployment strategy. +If .Values.routerStrategy is defined, use it. +Otherwise, fall back to the default rolling update strategy. +*/}} +{{- define "chart.routerStrategy" -}} +strategy: +{{- if .Values.routerSpec.strategy }} +{{- toYaml .Values.routerSpec.strategy | nindent 2 }} +{{- else }} + rollingUpdate: + maxSurge: 100% + maxUnavailable: 0 +{{- end }} +{{- end }} + +{{/* +Define additional ports +*/}} +{{- define "chart.extraPorts" }} +{{- with .Values.servingEngineSpec.extraPorts }} +{{ toYaml . }} +{{- end }} +{{- end }} + + +{{/* +Define liveness et readiness probes +*/}} +{{- define "chart.probes" -}} +{{- if .Values.servingEngineSpec.startupProbe }} +startupProbe: +{{- with .Values.servingEngineSpec.startupProbe }} +{{- toYaml . | nindent 2 }} +{{- end }} +{{- end }} +{{- if .Values.servingEngineSpec.livenessProbe }} +livenessProbe: +{{- with .Values.servingEngineSpec.livenessProbe }} +{{- toYaml . | nindent 2 }} +{{- end }} +{{- end }} +{{- end }} + +{{/* +Define resources with a variable model spec +*/}} +{{- define "chart.resources" -}} +{{- $modelSpec := . -}} +requests: + memory: {{ required "Value 'modelSpec.requestMemory' must be defined !" ($modelSpec.requestMemory | quote) }} + cpu: {{ required "Value 'modelSpec.requestCPU' must be defined !" ($modelSpec.requestCPU | quote) }} + {{- if (gt (int $modelSpec.requestGPU) 0) }} + nvidia.com/gpu: {{ required "Value 'modelSpec.requestGPU' must be defined !" (index $modelSpec.requestGPU | quote) }} + {{- end }} +limits: + memory: {{ required "Value 'modelSpec.requestMemory' must be defined !" ($modelSpec.requestMemory | quote) }} + cpu: {{ required "Value 'modelSpec.requestCPU' must be defined !" ($modelSpec.requestCPU | quote) }} + {{- if (gt (int $modelSpec.requestGPU) 0) }} + nvidia.com/gpu: {{ required "Value 'modelSpec.requestGPU' must be defined !" (index $modelSpec.requestGPU | quote) }} + {{- end }} +{{- end }} + + +{{/* + Define labels for serving engine and its service +*/}} +{{- define "chart.engineLabels" -}} +{{- with .Values.servingEngineSpec.labels -}} +{{ toYaml . }} +{{- end }} +{{- end }} + +{{/* + Define labels for router and its service +*/}} +{{- define "chart.routerLabels" -}} +{{- with .Values.routerSpec.labels -}} +{{ toYaml . }} +{{- end }} +{{- end }} + +{{/* + Define helper function to convert labels to a comma separated list +*/}} +{{- define "labels.toCommaSeparatedList" -}} +{{- $labels := . -}} +{{- $result := "" -}} +{{- range $key, $value := $labels -}} + {{- if $result }},{{ end -}} + {{ $key }}={{ $value }} + {{- $result = "," -}} +{{- end -}} +{{- end -}} diff --git a/charts/vllm/templates/configmap.yaml b/charts/vllm/templates/configmap.yaml new file mode 100644 index 0000000..2314086 --- /dev/null +++ b/charts/vllm/templates/configmap.yaml @@ -0,0 +1,11 @@ +{{- if .Values.servingEngineSpec.configs -}} +apiVersion: v1 +kind: ConfigMap +metadata: + name: "{{ .Release.Name }}-configs" + namespace: {{ .Release.Namespace }} +data: + {{- with .Values.servingEngineSpec.configs }} + {{- toYaml . | nindent 2 }} + {{- end }} +{{- end -}} diff --git a/charts/vllm/templates/deployment-router.yaml b/charts/vllm/templates/deployment-router.yaml new file mode 100644 index 0000000..5dacd67 --- /dev/null +++ b/charts/vllm/templates/deployment-router.yaml @@ -0,0 +1,85 @@ +{{- if .Values.routerSpec.enableRouter -}} +apiVersion: apps/v1 +kind: Deployment +metadata: + name: "{{ .Release.Name }}-deployment-router" + namespace: {{ .Release.Namespace }} + labels: + {{- include "chart.routerLabels" . | nindent 4 }} +spec: + replicas: {{ .Values.routerSpec.replicaCount }} + {{- include "chart.routerStrategy" . | nindent 2 }} + selector: + matchLabels: + {{- include "chart.routerLabels" . | nindent 6 }} + template: + metadata: + labels: + {{- include "chart.routerLabels" . | nindent 8 }} + spec: + serviceAccountName: {{ .Release.Name }}-router-service-account + containers: + - name: router-container + image: "{{ .Values.routerSpec.repository | default "lmcache/lmstack-router" }}:{{ .Values.routerSpec.tag | default "latest" }}" + imagePullPolicy: "{{ .Values.routerSpec.imagePullPolicy | default "Always" }}" + args: + - "--host" + - "0.0.0.0" + - "--port" + - "{{ .Values.routerSpec.containerPort }}" + - "--service-discovery" + - "{{ default "k8s" .Values.routerSpec.serviceDiscovery }}" + {{- if eq .Values.routerSpec.serviceDiscovery "k8s" }} + - "--k8s-namespace" + - "{{ .Release.Namespace }}" + - "--k8s-label-selector" + - {{ include "labels.toCommaSeparatedList" .Values.servingEngineSpec.labels }} + {{- end }} + {{- if eq .Values.routerSpec.serviceDiscovery "static" }} + - "--static-backends" + - "{{ required "When using static service discovery, .Values.routerSpec.staticBackends is a required value" .Values.routerSpec.staticBackends }}" + - "--static-models" + - "{{ required "When using static service discovery, .Values.routerSpec.staticModels is a required value" .Values.routerSpec.staticModels }}" + {{- end }} + - "--routing-logic" + - "{{ .Values.routerSpec.routingLogic }}" + {{- if .Values.routerSpec.sessionKey }} + - "--session-key" + - "{{ .Values.routerSpec.sessionKey }}" + {{- end }} + {{- if .Values.routerSpec.engineScrapeInterval }} + - "--engine-stats-interval" + - "{{ .Values.routerSpec.engineScrapeInterval }}" + {{- end }} + {{- if .Values.routerSpec.requestStatsWindow }} + - "--request-stats-window" + - "{{ .Values.routerSpec.requestStatsWindow }}" + {{- end }} + {{- if .Values.routerSpec.extraArgs }} + {{- toYaml .Values.routerSpec.extraArgs | nindent 10 }} + {{- end }} + {{- if .Values.routerSpec.resources }} + resources: + {{- if .Values.routerSpec.resources.requests }} + requests: + cpu: "{{ .Values.routerSpec.resources.requests.cpu }}" + memory: "{{ .Values.routerSpec.resources.requests.memory }}" + {{- end }} + {{- if .Values.routerSpec.resources.limits }} + limits: + cpu: "{{ .Values.routerSpec.resources.limits.cpu }}" + memory: "{{ .Values.routerSpec.resources.limits.memory }}" + {{- end }} + {{- end }} + ports: + - name: "router-cport" + containerPort: {{ .Values.routerSpec.containerPort }} + + livenessProbe: + initialDelaySeconds: 30 + periodSeconds: 5 + failureThreshold: 3 + httpGet: + path: /health + port: {{ .Values.routerSpec.containerPort }} +{{- end }} diff --git a/charts/vllm/templates/deployment-vllm-multi.yaml b/charts/vllm/templates/deployment-vllm-multi.yaml new file mode 100644 index 0000000..905084c --- /dev/null +++ b/charts/vllm/templates/deployment-vllm-multi.yaml @@ -0,0 +1,168 @@ +{{- range $modelSpec := .Values.servingEngineSpec.modelSpec }} +{{- with $ -}} +apiVersion: apps/v1 +kind: Deployment +metadata: + name: "{{ .Release.Name }}-{{$modelSpec.name}}-deployment-vllm" + namespace: {{ .Release.Namespace }} + labels: + {{- include "chart.engineLabels" . | nindent 4 }} +spec: + replicas: {{ $modelSpec.replicaCount }} + {{- include "chart.engineStrategy" . | nindent 2 }} + selector: + matchLabels: + {{- include "chart.engineLabels" . | nindent 6 }} + progressDeadlineSeconds: 1200 + template: + metadata: + labels: + {{- include "chart.engineLabels" . | nindent 8 }} + spec: + containers: + - name: "vllm" + image: "{{ required "Required value 'modelSpec.repository' must be defined !" $modelSpec.repository }}:{{ required "Required value 'modelSpec.tag' must be defined !" $modelSpec.tag }}" + + command: + - "vllm" + - "serve" + - {{ $modelSpec.modelURL | quote }} + - "--host" + - "0.0.0.0" + - "--port" + - {{ include "chart.container-port" . | quote }} + {{- with $modelSpec.vllmConfig }} + {{- if hasKey . "enableChunkedPrefill" }} + - "--enable-chunked-prefill" + - {{ .enableChunkedPrefill | quote }} + {{- end }} + {{- if .enablePrefixCaching }} + - "--enable-prefix-caching" + {{- end }} + {{- if hasKey . "maxModelLen" }} + - "--max-model-len" + - {{ .maxModelLen | quote }} + {{- end }} + {{- if hasKey . "dtype" }} + - "--dtype" + - {{ .dtype | quote }} + {{- end }} + {{- if hasKey . "tensorParallelSize" }} + - "--tensor-parallel-size" + - {{ .tensorParallelSize | quote }} + {{- end }} + {{- if .extraArgs }} + {{- range .extraArgs }} + - {{ . | quote }} + {{- end }} + {{- end }} + {{- end }} + {{- if $modelSpec.lmcacheConfig }} + {{- if $modelSpec.lmcacheConfig.enabled }} + - "--kv-transfer-config" + - '{"kv_connector":"LMCacheConnector","kv_role":"kv_both"}' + {{- end }} + {{- end }} + securityContext: + runAsNonRoot: false + imagePullPolicy: IfNotPresent + env: + - name: HF_HOME + {{- if hasKey $modelSpec "pvcStorage" }} + value: /data + {{- else }} + value: /tmp + {{- end }} + {{- with $modelSpec.vllmConfig}} + {{- if hasKey . "v1" }} + - name: VLLM_USE_V1 + value: {{ default 0 $modelSpec.vllmConfig.v1 | quote }} + {{- end}} + {{- end}} + {{- if $modelSpec.hf_token }} + - name: HF_TOKEN + valueFrom: + secretKeyRef: + name: {{ .Release.Name }}-secrets + key: hf_token_{{ $modelSpec.name }} + {{- end }} + {{- with $modelSpec.env }} + {{- toYaml . | nindent 10 }} + {{- end }} + {{- if $modelSpec.lmcacheConfig }} + {{- if $modelSpec.lmcacheConfig.enabled }} + - name: LMCACHE_USE_EXPERIMENTAL + value: "True" + - name: VLLM_RPC_TIMEOUT + value: "1000000" + {{- end }} + {{- if $modelSpec.lmcacheConfig.cpuOffloadingBufferSize }} + - name: LMCACHE_LOCAL_CPU + value: "True" + - name: LMCACHE_MAX_LOCAL_CPU_SIZE + value: "{{ $modelSpec.lmcacheConfig.cpuOffloadingBufferSize }}" + {{- end }} + {{- if $modelSpec.lmcacheConfig.diskOffloadingBufferSize }} + - name: LMCACHE_LOCAL_DISK + value: "True" + - name: LMCACHE_MAX_LOCAL_DISK_SIZE + value: "{{ $modelSpec.lmcacheConfig.diskOffloadingBufferSize }}" + {{- end }} + {{- end }} + {{- if .Values.servingEngineSpec.configs }} + envFrom: + - configMapRef: + name: "{{ .Release.Name }}-configs" + {{- end }} + ports: + - name: {{ include "chart.container-port-name" . }} + containerPort: {{ include "chart.container-port" . }} + {{- include "chart.probes" . | indent 10 }} + resources: {{- include "chart.resources" $modelSpec | nindent 12 }} + {{- if hasKey $modelSpec "pvcStorage" }} + volumeMounts: + - name: {{ .Release.Name }}-storage + mountPath: /data + {{- end }} + {{- with $modelSpec.vllmConfig }} + {{- if hasKey $modelSpec.vllmConfig "tensorParallelSize"}} + - name: shm + mountPath: /dev/shm + {{- end}} + {{- end}} + volumes: + {{- if hasKey $modelSpec "pvcStorage" }} + - name: {{ .Release.Name }}-storage + persistentVolumeClaim: + claimName: "{{ .Release.Name }}-{{$modelSpec.name}}-storage-claim" + {{- end }} + {{- with $modelSpec.vllmConfig }} + {{- if hasKey $modelSpec.vllmConfig "tensorParallelSize"}} + - name: shm + emptyDir: + medium: Memory + sizeLimit: {{ default "20Gi" $modelSpec.shmSize }} + {{- end}} + {{- end}} + {{- if .Values.servingEngineSpec.tolerations }} + {{- with .Values.servingEngineSpec.tolerations }} + tolerations: + {{- toYaml . | nindent 8 }} + {{- end }} + {{- end }} + + {{- if .Values.servingEngineSpec.runtimeClassName }} + runtimeClassName: {{ .Values.servingEngineSpec.runtimeClassName }} + {{- end }} + {{- if $modelSpec.nodeSelectorTerms}} + affinity: + nodeAffinity: + requiredDuringSchedulingIgnoredDuringExecution: + nodeSelectorTerms: + {{- with $modelSpec.nodeSelectorTerms }} + {{- toYaml . | nindent 12 }} + {{- end }} + {{- end }} +{{- end }} +--- +{{- end }} diff --git a/charts/vllm/templates/ingress.yaml b/charts/vllm/templates/ingress.yaml new file mode 100644 index 0000000..5008260 --- /dev/null +++ b/charts/vllm/templates/ingress.yaml @@ -0,0 +1,42 @@ +{{- if and (.Values.routerSpec.enableRouter) (.Values.routerSpec.ingress.enabled) -}} +apiVersion: networking.k8s.io/v1 +kind: Ingress +metadata: + name: "{{ .Release.Name }}-ingress-router" + namespace: {{ .Release.Namespace }} + labels: + {{- include "chart.routerLabels" . | nindent 4 }} + {{- with .Values.routerSpec.ingress.annotations }} + annotations: + {{- toYaml . | nindent 4 }} + {{- end }} +spec: + {{- if .Values.routerSpec.ingress.className }} + ingressClassName: {{ .Values.routerSpec.ingress.className }} + {{- end }} + {{- if .Values.routerSpec.ingress.tls }} + tls: + {{- range .Values.routerSpec.ingress.tls }} + - hosts: + {{- range .hosts }} + - {{ . | quote }} + {{- end }} + secretName: {{ .secretName }} + {{- end }} + {{- end }} + rules: + {{- range .Values.routerSpec.ingress.hosts }} + - host: {{ .host | quote }} + http: + paths: + {{- range .paths }} + - path: {{ .path }} + pathType: {{ .pathType }} + backend: + service: + name: "{{ $.Release.Name }}-router-service" + port: + number: {{ $.Values.routerSpec.servicePort }} + {{- end }} + {{- end }} +{{- end }} diff --git a/charts/vllm/templates/poddisruptionbudget.yaml b/charts/vllm/templates/poddisruptionbudget.yaml new file mode 100644 index 0000000..4b62f74 --- /dev/null +++ b/charts/vllm/templates/poddisruptionbudget.yaml @@ -0,0 +1,7 @@ +apiVersion: policy/v1 +kind: PodDisruptionBudget +metadata: + name: "{{ .Release.Name }}-pdb" + namespace: {{ .Release.Namespace }} +spec: + maxUnavailable: {{ default 1 .Values.servingEngineSpec.maxUnavailablePodDisruptionBudget }} diff --git a/charts/vllm/templates/pvc.yaml b/charts/vllm/templates/pvc.yaml new file mode 100644 index 0000000..625f296 --- /dev/null +++ b/charts/vllm/templates/pvc.yaml @@ -0,0 +1,30 @@ +{{- range $modelSpec := .Values.servingEngineSpec.modelSpec }} +{{- with $ -}} +{{- if and (hasKey $modelSpec "pvcStorage") (not (empty $modelSpec.pvcStorage)) }} +apiVersion: v1 +kind: PersistentVolumeClaim +metadata: + name: "{{ .Release.Name }}-{{$modelSpec.name}}-storage-claim" + namespace: {{ .Release.Namespace }} +spec: + accessModes: + {{- if $modelSpec.pvcAccessMode }} + {{- toYaml $modelSpec.pvcAccessMode | nindent 4 }} + {{- else }} + - ReadWriteOnce + {{- end }} + resources: + requests: + storage: {{ $modelSpec.pvcStorage | default "20Gi" }} # Default to 40Gi if not set + {{- if hasKey $modelSpec "storageClass" }} + storageClassName: "{{ $modelSpec.storageClass }}" + {{- end }} + {{- if not (empty $modelSpec.pvcMatchLabels) }} + selector: + matchLabels: + {{- toYaml $modelSpec.pvcMatchLabels | nindent 8 }} + {{- end }} +{{- end }} +{{- end }} +--- +{{- end }} diff --git a/charts/vllm/templates/role.yaml b/charts/vllm/templates/role.yaml new file mode 100644 index 0000000..3e45c9e --- /dev/null +++ b/charts/vllm/templates/role.yaml @@ -0,0 +1,11 @@ +{{- if .Values.routerSpec.enableRouter -}} +apiVersion: rbac.authorization.k8s.io/v1 +kind: Role +metadata: + name: "{{ .Release.Name }}-pod-reader" + namespace: {{ .Release.Namespace }} +rules: +- apiGroups: [""] # "" indicates the core API group + resources: ["pods"] + verbs: ["get", "watch", "list"] +{{- end }} diff --git a/charts/vllm/templates/rolebinding.yaml b/charts/vllm/templates/rolebinding.yaml new file mode 100644 index 0000000..a04218e --- /dev/null +++ b/charts/vllm/templates/rolebinding.yaml @@ -0,0 +1,15 @@ +{{- if .Values.routerSpec.enableRouter -}} +apiVersion: rbac.authorization.k8s.io/v1 +kind: RoleBinding +metadata: + name: {{ .Release.Name }}-deployment-access-binding + namespace: {{ .Release.Namespace }} +subjects: + - kind: ServiceAccount + name: {{ .Release.Name }}-router-service-account + namespace: {{ .Release.Namespace }} +roleRef: + kind: Role + name: {{ .Release.Name }}-pod-reader + apiGroup: rbac.authorization.k8s.io +{{- end }} diff --git a/charts/vllm/templates/secrets.yaml b/charts/vllm/templates/secrets.yaml new file mode 100644 index 0000000..b8d1719 --- /dev/null +++ b/charts/vllm/templates/secrets.yaml @@ -0,0 +1,14 @@ +apiVersion: v1 +kind: Secret +metadata: + name: "{{ .Release.Name }}-secrets" + namespace: {{ .Release.Namespace }} +type: Opaque +data: + {{- range $modelSpec := .Values.servingEngineSpec.modelSpec }} + {{- with $ -}} + {{- if $modelSpec.hf_token }} + hf_token_{{ $modelSpec.name }}: {{ $modelSpec.hf_token | b64enc | quote }} + {{- end }} + {{- end }} + {{- end }} diff --git a/charts/vllm/templates/service-router.yaml b/charts/vllm/templates/service-router.yaml new file mode 100644 index 0000000..e90529a --- /dev/null +++ b/charts/vllm/templates/service-router.yaml @@ -0,0 +1,18 @@ +{{- if .Values.routerSpec.enableRouter -}} +apiVersion: v1 +kind: Service +metadata: + name: "{{ .Release.Name }}-router-service" + namespace: {{ .Release.Namespace }} + labels: + {{- include "chart.routerLabels" . | nindent 4 }} +spec: + type: ClusterIP + ports: + - name: "router-sport" + port: {{ .Values.routerSpec.servicePort }} + targetPort: {{ .Values.routerSpec.containerPort }} + protocol: TCP + selector: + {{- include "chart.routerLabels" . | nindent 4 }} +{{- end }} diff --git a/charts/vllm/templates/service-vllm.yaml b/charts/vllm/templates/service-vllm.yaml new file mode 100644 index 0000000..efe6cc0 --- /dev/null +++ b/charts/vllm/templates/service-vllm.yaml @@ -0,0 +1,16 @@ +apiVersion: v1 +kind: Service +metadata: + name: "{{ .Release.Name }}-engine-service" + namespace: {{ .Release.Namespace }} + labels: + {{- include "chart.engineLabels" . | nindent 4 }} +spec: + type: ClusterIP + ports: + - name: {{ include "chart.service-port-name" . }} + port: {{ include "chart.service-port" . }} + targetPort: {{ include "chart.container-port-name" . }} + protocol: TCP + selector: + {{- include "chart.engineLabels" . | nindent 4 }} diff --git a/charts/vllm/templates/serviceaccount.yaml b/charts/vllm/templates/serviceaccount.yaml new file mode 100644 index 0000000..ad99564 --- /dev/null +++ b/charts/vllm/templates/serviceaccount.yaml @@ -0,0 +1,7 @@ +{{- if .Values.routerSpec.enableRouter -}} +apiVersion: v1 +kind: ServiceAccount +metadata: + name: "{{ .Release.Name }}-router-service-account" + namespace: {{ .Release.Namespace }} +{{- end }} diff --git a/charts/vllm/values.yaml b/charts/vllm/values.yaml new file mode 100644 index 0000000..e3ba0c1 --- /dev/null +++ b/charts/vllm/values.yaml @@ -0,0 +1,237 @@ +# -- Default values for llmstack helm chart +# -- Declare variables to be passed into your templates. + + +# -- Serving engine configuratoon +servingEngineSpec: + # -- Customized labels for the serving engine deployment + labels: + environment: "test" + release: "test" + + # modelSpec - configuring multiple serving engines deployments that runs different models + # Each entry in the modelSpec array should contain the following fields: + # - name: (string) The name of the model, e.g., "example-model" + # - repository: (string) The repository of the model, e.g., "vllm/vllm-openai" + # - tag: (string) The tag of the model, e.g., "latest" + # - modelURL: (string) The URL of the model, e.g., "facebook/opt-125m" + # + # - replicaCount: (int) The number of replicas for the model, e.g. 1 + # - requestCPU: (int) The number of CPUs requested for the model, e.g. 6 + # - requestMemory: (string) The amount of memory requested for the model, e.g., "16Gi" + # - requestGPU: (int) The number of GPUs requested for the model, e.g., 1 + # + # - pvcStorage: (Optional, string) The amount of storage requested for the model, e.g., "50Gi". + # - pvcAccessMode: (Optional, list) The access mode policy for the mounted volume, e.g., ["ReadWriteOnce"] + # - storageClass: (Optional, String) The storage class of the PVC e.g., "", default is "" + # - pvcMatchLabels: (Optional, map) The labels to match the PVC, e.g., {model: "opt125m"} + # + # - vllmConfig: (optional, map) The configuration for the VLLM model, supported options are: + # - enablePrefixCaching: (optional, bool) Enable prefix caching, e.g., false + # - enableChunkedPrefill: (optional, bool) Enable chunked prefill, e.g., false + # - maxModelLen: (optional, int) The maximum model length, e.g., 16384 + # - dtype: (optional, string) The data type, e.g., "bfloat16" + # - tensorParallelSize: (optional, int) The degree of tensor parallelism, e.g., 2 + # - extraArgs: (optional, list) Extra command line arguments to pass to vLLM, e.g., ["--disable-log-requests"] + # + # - lmcacheConfig: (optional, map) The configuration of the LMCache for KV offloading, supported options are: + # - enabled: (optional, bool) Enable LMCache, e.g., true + # - cpuOffloadingBufferSize: (optional, string) The CPU offloading buffer size, e.g., "30" + # + # - hf_token: (optional, string) the Huggingface tokens for this model + # + # - env: (optional, list) The environment variables to set in the container, e.g., your HF_TOKEN + # + # - nodeSelectorTerms: (optional, list) The node selector terms to match the nodes + # + # - shmSize: (optional, string) The size of the shared memory, e.g., "20Gi" + # + # Example: + # modelSpec: + # - name: "mistral" + # repository: "lmcache/vllm-openai" + # tag: "latest" + # modelURL: "mistralai/Mistral-7B-Instruct-v0.2" + # replicaCount: 1 + # + # requestCPU: 10 + # requestMemory: "64Gi" + # requestGPU: 1 + # + # pvcStorage: "50Gi" + # pvcAccessMode: + # - ReadWriteOnce + # pvcMatchLabels: + # model: "mistral" + # + # vllmConfig: + # enableChunkedPrefill: false + # enablePrefixCaching: false + # maxModelLen: 16384 + # dtype: "bfloat16" + # extraArgs: ["--disable-log-requests", "--gpu-memory-utilization", "0.8"] + # + # lmcacheConfig: + # enabled: true + # cpuOffloadingBufferSize: "30" + # + # hf_token: + # + # nodeSelectorTerms: + # - matchExpressions: + # - key: nvidia.com/gpu.product + # operator: "In" + # values: + # - "NVIDIA-RTX-A6000" + modelSpec: [] + + # -- Container port + containerPort: 8000 + # -- Service port + servicePort: 80 + + # -- Set other environment variables from config map + configs: {} + + # -- deployment strategy + strategy: {} + + # -- Readiness probe configuration + startupProbe: + # -- Number of seconds after the container has started before startup probe is initiated + initialDelaySeconds: 15 + # -- How often (in seconds) to perform the startup probe + periodSeconds: 10 + # -- Number of times after which if a probe fails in a row, Kubernetes considers that the overall check has failed: the container is not ready + failureThreshold: 60 + # -- Configuration of the Kubelet http request on the server + httpGet: + # -- Path to access on the HTTP server + path: /health + # -- Name or number of the port to access on the container, on which the server is listening + port: 8000 + + # -- Liveness probe configuration + livenessProbe: + # -- Number of seconds after the container has started before liveness probe is initiated + initialDelaySeconds: 15 + # -- Number of times after which if a probe fails in a row, Kubernetes considers that the overall check has failed: the container is not alive + failureThreshold: 3 + # -- How often (in seconds) to perform the liveness probe + periodSeconds: 10 + # -- Configuration of the Kubelet http request on the server + httpGet: + # -- Path to access on the HTTP server + path: /health + # -- Name or number of the port to access on the container, on which the server is listening + port: 8000 + + # -- Disruption Budget Configuration + maxUnavailablePodDisruptionBudget: "" + + # -- Tolerations configuration (when there are taints on nodes) + # Example: + # tolerations: + # - key: "node-role.kubernetes.io/control-plane" + # operator: "Exists" + # effect: "NoSchedule" + tolerations: [] + + # -- RuntimeClassName configuration, set to "nvidia" if the model requires GPU + runtimeClassName: "nvidia" + +routerSpec: + # -- The docker image of the router. The following values are defaults: + repository: "lmcache/lmstack-router" + tag: "latest" + imagePullPolicy: "Always" + + # -- Whether to enable the router service + enableRouter: true + + # -- Number of replicas + replicaCount: 1 + + # -- Container port + containerPort: 8000 + + # -- Service port + servicePort: 80 + + # -- Service discovery mode, supports "k8s" or "static". Defaults to "k8s" if not set. + serviceDiscovery: "k8s" + + # -- If serviceDiscovery is set to "static", the comma-separated values below are required. There needs to be the same number of backends and models + staticBackends: "" + staticModels: "" + + # -- routing logic, could be "roundrobin" or "session" + routingLogic: "roundrobin" + + # -- session key if using "session" routing logic + sessionKey: "" + + # -- extra router commandline arguments + extraArgs: [] + + # -- Interval in seconds to scrape the serving engine metrics + engineScrapeInterval: 15 + + # -- Window size in seconds to calculate the request statistics + requestStatsWindow: 60 + + # -- deployment strategy + strategy: {} + + # -- router resource requests and limits + resources: + requests: + cpu: "4" + memory: "16G" + limits: + cpu: "8" + memory: "32G" + + # -- Customized labels for the router deployment + labels: + environment: "router" + release: "router" + + ingress: + # -- Enable ingress controller resource + enabled: false + + # -- IngressClass that will be used to implement the Ingress + className: "" + + # -- Additional annotations for the Ingress resource + annotations: {} + # kubernetes.io/ingress.class: alb + # kubernetes.io/ingress.class: nginx + # kubernetes.io/tls-acme: "true" + + # The list of hostnames to be covered with this ingress record. + hosts: + - host: vllm-router.local + paths: + - path: / + pathType: Prefix + + # -- The tls configuration for hostnames to be covered with this ingress record. + tls: [] + # - secretName: vllm-router-tls + # hosts: + # - vllm-router.local + + # -- TODO: Readiness probe configuration + #startupProbe: + # # -- Number of seconds after the container has started before startup probe is initiated + # initialDelaySeconds: 5 + # # -- How often (in seconds) to perform the startup probe + # periodSeconds: 5 + # # -- Number of times after which if a probe fails in a row, Kubernetes considers that the overall check has failed: the container is not ready + # failureThreshold: 100 + # # -- Configuration of the Kubelet http request on the server + # httpGet: + # # -- Path to access on the HTTP server + #