version: 0.9.2-1 defaultConfig: namespace: llm model: Qwen/Qwen3-8B maxModelLen: 32768 gpuMemoryUtilization: 0.9 cpuRequest: '4' cpuLimit: '8' memoryRequest: 16Gi memoryLimit: 24Gi gpuCount: 1 domain: vllm.{{ .cloud.domain }} defaultSecrets: - key: apiKey