Helm Overview #
# Add Helm repository
helm repo add otwld https://helm.otwld.com/ &&
helm repo update
# List available Helm chart versions
helm search repo otwld/ollama --versions
# Shell output:
NAME CHART VERSION APP VERSION DESCRIPTION
otwld/ollama 1.84.0 0.34.4 Get up and running with large language models l...
otwld/ollama 1.83.0 0.34.2 Get up and running with large language models l...
# Save Helm chart values
helm show values otwld/ollama --version 1.84.0 > ollama-values-1.84.0.yml
Ansible Setup #
Playbook #
- name: Ollama Helm installation
hosts: localhost
connection: local
gather_facts: false
become: false
vars:
# Helm Configuration
helm_repo_url: "https://helm.otwld.com/"
helm_repo_name: "otwld"
helm_chart: "otwld/ollama"
helm_chart_version: "1.84.0"
helm_release_name: "ollama"
# Kubernetes Configuration
kubernetes_namespace: "ollama"
storage_class: "local-path"
pvc_size: "20Gi"
# Routing
ollama_domain: "ollama.jklug.work"
gateway_name: "kgateway-metallb"
gateway_namespace: "kgateway-infra"
gateway_listener: "https"
# GPU Configuration
runtime_class: "nvidia"
roles:
- k8s_ollama
# Run playbook:
ansible-playbook playbooks/k8s_ollama.yml
Tasks #
k8s_ollama/tasks/main.yml
- name: Create namespace
kubernetes.core.k8s:
api_version: v1
kind: Namespace
name: "{{ kubernetes_namespace }}"
state: present
- name: Add Helm repository
kubernetes.core.helm_repository:
name: "{{ helm_repo_name }}"
repo_url: "{{ helm_repo_url }}"
force_update: true
- name: Install Helm Chart
kubernetes.core.helm:
name: "{{ helm_release_name }}"
chart_ref: "{{ helm_chart }}"
chart_version: "{{ helm_chart_version }}"
release_namespace: "{{ kubernetes_namespace }}"
create_namespace: false
wait: true # Ansible waits till all resources are ready
wait_timeout: 5m0s
atomic: false # Auto-rollback on failure
values: "{{ lookup('template', 'helm-values.yml.j2') | from_yaml }}"
- name: Configure Ollama HTTPRoute TrafficPolicy for timeout
kubernetes.core.k8s:
state: present
definition:
apiVersion: gateway.kgateway.dev/v1alpha1
kind: TrafficPolicy
metadata:
name: ollama-timeout
namespace: ollama
spec:
targetRefs:
- group: gateway.networking.k8s.io
kind: HTTPRoute
name: ollama
timeouts:
request: 30m
Templates #
k8s_ollama/templates/helm-values.yml.j2
# Default values for ollama-helm.
# This is a YAML-formatted file.
# Declare variables to be passed into your templates.
# -- Number of replicas
replicaCount: 1
# Knative configuration
knative:
# -- Enable Knative integration
enabled: false
# -- Knative service container concurrency
containerConcurrency: 0
# -- Knative service timeout seconds
timeoutSeconds: 300
# -- Knative service response start timeout seconds
responseStartTimeoutSeconds: 300
# -- Knative service idle timeout seconds
idleTimeoutSeconds: 300
# -- Knative service annotations
annotations: {}
modelBootstrap:
# -- Time to keep completed Knative model bootstrap Jobs before cleanup. Set to null to disable TTL-based cleanup.
ttlSecondsAfterFinished: 300
# Docker image
image:
# -- Docker image registry
repository: ollama/ollama
# -- Docker pull policy
pullPolicy: IfNotPresent
# -- Docker image tag, overrides the image tag whose default is the chart appVersion.
tag: ""
# -- Docker registry secret names as an array
imagePullSecrets: []
# -- String to partially override template (will maintain the release name)
nameOverride: ""
# -- String to fully override template
fullnameOverride: ""
# -- String to fully override namespace
namespaceOverride: ""
# Ollama parameters
ollama:
# Port Ollama is listening on
port: 11434
gpu:
# -- Enable GPU integration
enabled: true
# -- Enable DRA GPU integration
# If enabled, it will use DRA instead of Device Driver Plugin and create a ResourceClaim and GpuClaimParameters
draEnabled: false
# -- DRA GPU DriverClass
draDriverClass: "gpu.nvidia.com"
# -- Existing DRA GPU ResourceClaim Template
draExistingClaimTemplate: ""
# -- GPU type: 'nvidia' or 'amd'
# If 'ollama.gpu.enabled', default value is nvidia
# If set to 'amd', this will add 'rocm' suffix to image tag if 'image.tag' is not override
# This is due cause AMD and CPU/CUDA are different images
type: 'nvidia'
# -- Specify the number of GPU
# If you use MIG section below then this parameter is ignored
number: 1
# -- only for nvidia cards; change to (example) 'nvidia.com/mig-1g.10gb' to use MIG slice
nvidiaResource: "nvidia.com/gpu"
# nvidiaResource: "nvidia.com/mig-1g.10gb" # example
# If you want to use more than one NVIDIA MIG you can use the following syntax (then nvidiaResource is ignored and only the configuration in the following MIG section is used)
mig:
# -- Enable multiple mig devices
# If enabled you will have to specify the mig devices
# If enabled is set to false this section is ignored
enabled: false
# -- Specify the mig devices and the corresponding number
devices: {}
# 1g.10gb: 1
# 3g.40gb: 1
models:
# -- List of models to pull at container startup
# The more you add, the longer the container will take to start if models are not present
# pull:
# - llama2
# - mistral
pull: []
# -- List of models to load in memory at container startup
# run:
# - llama2
# - mistral
run: []
# -- List of models to create at container startup, there are two options
# 1. Create a raw model
# 2. Load a model from configMaps, configMaps must be created before and are loaded as volume in "/models" directory.
# create:
# - name: llama3.1-ctx32768
# configMapRef: my-configmap
# configMapKeyRef: configmap-key
# - name: llama3.1-ctx32768
# template: |
# FROM llama3.1
# PARAMETER num_ctx 32768
create: []
# -- Automatically remove models present on the disk but not specified in the values file
clean: false
# -- Add insecure flag for pulling at container startup
insecure: false
# -- Override ollama-data volume mount path, default: "/root/.ollama"
mountPath: ""
# Service account
# ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-service-account/
serviceAccount:
# -- Specifies whether a service account should be created
create: true
# -- Automatically mount a ServiceAccount's API credentials?
automount: true
# -- Annotations to add to the service account
annotations: {}
# -- The name of the service account to use.
# If not set and create is true, a name is generated using the fullname template
name: ""
# -- Map of annotations to add to the pods
podAnnotations: {}
# -- Map of labels to add to the pods
podLabels: {}
# -- Pod Security Context
podSecurityContext: {}
# fsGroup: 2000
# -- Priority Class Name
priorityClassName: ""
# -- Container Security Context
securityContext: {}
# capabilities:
# drop:
# - ALL
# readOnlyRootFilesystem: true
# runAsNonRoot: true
# runAsUser: 1000
# -- Specify runtime class
runtimeClassName: "{{ runtime_class }}"
# Configure Service
service:
# -- Service type
type: ClusterIP
# -- Service port
port: 11434
# -- Service node port when service type is 'NodePort'
nodePort: 31434
# -- Load Balancer IP address
loadBalancerIP:
# -- Annotations to add to the service
annotations: {}
# -- Labels to add to the service
labels: {}
# -- IP Families for the service
ipFamilies: []
# - IPv4
# - IPv6
# -- IP Family Policy for the service
ipFamilyPolicy: ""
# SingleStack
# PreferDualStack
# RequireDualStack
# Configure Deployment
deployment:
# -- Labels to add to the deployment
labels: {}
# Configure Gateway
gateway:
# -- Create Gateway if gateway.enabled = true. Otherwise, httpRoute will need parentRefs to attach to an existing Gateway.
enabled: false
# -- Name of the Gateway. Defaults to the chart's full name if empty
name: ""
# -- Name of an existing Gateway Class. Mandatory non-empty field
className: ""
# -- Labels to add to the Gateway
labels: {}
# -- Annotations to add to the Gateway
annotations: {}
# -- Configure the listener
listener:
# -- Listener network port. May depend on implementation (eg. Traefik)
port: 80
# -- Define which Routes may be attached to this Listener
allowedRoutes: {}
# namespaces:
# from: Same
# -- TLS configuration
tls:
# -- Enable TLS
enabled: false
# -- Reference to valid certificates(s)
# -- See https://gateway-api.sigs.k8s.io/reference/spec/#secretobjectreference
certificateRefs: []
# - name: ollama-certificate
# kind: Secret
# group: ""
# Configure HTTPRoute
httpRoute:
# -- Enable HttpRoute
enabled: true
# -- Labels to add to the HTTPRoute
labels: {}
# -- Hostnames to match for this HTTPRoute
hostnames:
- {{ ollama_domain }}
# -- References to the existing Gateway(s) this route should attach to.
# -- Ignored if gateway.enabled is true. It will automatically attach to the created gateway.
# -- See https://gateway-api.sigs.k8s.io/reference/spec/#parentreference
parentRefs:
- name: "{{ gateway_name }}"
namespace: "{{ gateway_namespace }}"
sectionName: "{{ gateway_listener }}"
# - name: ollama
# namespace: default
# -- Routing rules. If empty, a default rule routing '/' to the Ollama service is created.
# -- See https://gateway-api.sigs.k8s.io/reference/spec/#httprouterule
rules: []
# - matches:
# - path:
# type: PathPrefix
# value: /api
# Configure the ingress resource that allows you to access the
ingress:
# -- Enable ingress controller resource
enabled: false
# -- IngressClass that will be used to implement the Ingress (Kubernetes 1.18+)
className: ""
# -- Additional annotations for the Ingress resource.
annotations: {}
# kubernetes.io/ingress.class: traefik
# kubernetes.io/ingress.class: nginx
# kubernetes.io/tls-acme: "true"
# The list of hostnames to be covered with this ingress record.
hosts:
- host: ollama.local
paths:
- path: /
pathType: Prefix
# -- The tls configuration for hostnames to be covered with this ingress record.
tls: []
# - secretName: chart-example-tls
# hosts:
# - chart-example.local
# Configure resource requests and limits
# ref: http://kubernetes.io/docs/user-guide/compute-resources/
resources:
# -- Pod requests
requests: {}
# Memory request
# memory: 4096Mi
# CPU request
# cpu: 2000m
# -- Pod limit
limits: {}
# Memory limit
# memory: 8192Mi
# CPU limit
# cpu: 4000m
# Configure extra options for liveness probe
# ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/#configure-probes
livenessProbe:
# -- Enable livenessProbe
enabled: true
# -- Request path for livenessProbe
path: /
# -- Initial delay seconds for livenessProbe
initialDelaySeconds: 60
# -- Period seconds for livenessProbe
periodSeconds: 10
# -- Timeout seconds for livenessProbe
timeoutSeconds: 5
# -- Failure threshold for livenessProbe
failureThreshold: 6
# -- Success threshold for livenessProbe
successThreshold: 1
# Configure extra options for readiness probe
# ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/#configure-probes
readinessProbe:
# -- Enable readinessProbe
enabled: true
# -- Request path for readinessProbe
path: /
# -- Initial delay seconds for readinessProbe
initialDelaySeconds: 30
# -- Period seconds for readinessProbe
periodSeconds: 5
# -- Timeout seconds for readinessProbe
timeoutSeconds: 3
# -- Failure threshold for readinessProbe
failureThreshold: 6
# -- Success threshold for readinessProbe
successThreshold: 1
# Configure autoscaling
autoscaling:
# -- Enable autoscaling
enabled: false
# -- Number of minimum replicas
minReplicas: 1
# -- Number of maximum replicas
maxReplicas: 100
# -- CPU usage to target replica
targetCPUUtilizationPercentage: 80
# -- targetMemoryUtilizationPercentage: 80
# -- Additional volumes on the output Deployment definition.
volumes: []
# -- - name: foo
# secret:
# secretName: mysecret
# optional: false
# -- Additional volumeMounts on the output Deployment definition.
volumeMounts: []
# -- - name: foo
# mountPath: "/etc/foo"
# readOnly: true
# -- Additional arguments on the output Deployment definition.
extraArgs: []
# -- Additional environments variables on the output Deployment definition.
# For extra OLLAMA env, please refer to https://github.com/ollama/ollama/blob/main/envconfig/config.go
extraEnv: []
# - name: OLLAMA_DEBUG
# value: "1"
# -- Additionl environment variables from external sources (like ConfigMap)
extraEnvFrom: []
# - configMapRef:
# name: my-env-configmap
# Enable persistence using Persistent Volume Claims
# ref: https://kubernetes.io/docs/concepts/storage/persistent-volumes/
persistentVolume:
# -- Enable persistence using PVC
enabled: true
# -- Ollama server data Persistent Volume access modes
# Must match those of existing PV or dynamic provisioner
# Ref: http://kubernetes.io/docs/user-guide/persistent-volumes/
accessModes:
- ReadWriteOnce
# -- Ollama server data Persistent Volume annotations
annotations: {}
# -- If you'd like to bring your own PVC for persisting Ollama state, pass the name of the
# created + ready PVC here. If set, this Chart will not create the default PVC.
# Requires server.persistentVolume.enabled: true
existingClaim: ""
# -- Ollama server data Persistent Volume size
size: "{{ pvc_size }}"
# -- Ollama server data Persistent Volume Storage Class
# If defined, storageClassName: <storageClass>
# If set to "-", storageClassName: "", which disables dynamic provisioning
# If undefined (the default) or set to null, no storageClassName spec is
# set, choosing the default provisioner. (gp2 on AWS, standard on
# GKE, AWS & OpenStack)
storageClass: "{{ storage_class }}"
# -- Ollama server data Persistent Volume Binding Mode
# If defined, volumeMode: <volumeMode>
# If empty (the default) or set to null, no volumeBindingMode spec is
# set, choosing the default mode.
volumeMode: ""
# -- Subdirectory of Ollama server data Persistent Volume to mount
# Useful if the volume's root directory is not empty
subPath: ""
# -- Pre-existing PV to attach this claim to
# Useful if a CSI auto-provisions a PV for you and you want to always
# reference the PV moving forward
volumeName: ""
# -- Node labels for pod assignment.
nodeSelector: {}
# -- Tolerations for pod assignment
tolerations: []
# -- Affinity for pod assignment
affinity: {}
# -- Lifecycle for pod assignment (override ollama.models startup pull/run)
lifecycle: {}
# How to replace existing pods
updateStrategy:
# -- Deployment strategy can be "Recreate" or "RollingUpdate". Default is Recreate
type: "Recreate"
# -- Number of old ReplicaSets to retain for rollback. Default is 10.
revisionHistoryLimit: 10
# -- Topology Spread Constraints for pod assignment
topologySpreadConstraints: {}
# -- Wait for a grace period
terminationGracePeriodSeconds: 120
# -- Init containers to add to the pod
initContainers: []
# - name: startup-tool
# image: alpine:3
# command: [sh, -c]
# args:
# - echo init
# -- Use the host’s ipc namespace.
hostIPC: false
# -- Use the host’s pid namespace
hostPID: false
# -- Use the host's network namespace.
hostNetwork: false
# -- Extra K8s manifests to deploy
extraObjects: []
# - apiVersion: v1
# kind: PersistentVolume
# metadata:
# name: aws-efs
# data:
# key: "value"
# - apiVersion: scheduling.k8s.io/v1
# kind: PriorityClass
# metadata:
# name: high-priority
# value: 1000000
# globalDefault: false
# description: "This priority class should be used for XYZ service pods only."
# Test connection pods
tests:
enabled: true
# -- Labels to add to the tests
labels: {}
# -- Annotations to add to the tests
annotations: {}
podSchedulerName: ""
Kubernetes Resources #
# List default resources
kubectl -n ollama get all
# Shell output:
NAME READY STATUS RESTARTS AGE
pod/ollama-7dbcc65f9d-7mhlz 1/1 Running 0 2m34s
NAME TYPE CLUSTER-IP EXTERNAL-IP PORT(S) AGE
service/ollama ClusterIP 10.201.53.237 <none> 11434/TCP 2m34s
NAME READY UP-TO-DATE AVAILABLE AGE
deployment.apps/ollama 1/1 1 1 2m34s
NAME DESIRED CURRENT READY AGE
replicaset.apps/ollama-7dbcc65f9d 1 1 1 2m34s
# List PVC
kubectl -n ollama get pvc
# Shell output:
NAME STATUS VOLUME CAPACITY ACCESS MODES STORAGECLASS VOLUMEATTRIBUTESCLASS AGE
ollama Bound pvc-e493a115-0d2d-44bc-bde9-9d323467f356 20Gi RWO local-path <unset> 2m48s
# List httproute
kubectl -n ollama get httproute
# Shell output:
NAME HOSTNAMES AGE
ollama ["ollama.jklug.work"] 3m
Ollama API #
# Curl Ollama status
curl https://ollama.jklug.work
# Shell output:
Ollama is running
# Pull qwen3.5:9b
curl --fail-with-body -N https://ollama.jklug.work/api/pull \
-H 'Content-Type: application/json' \
-d '{"model":"qwen3.5:9b"}'
# Shell output:
...
{"status":"verifying sha256 digest"}
{"status":"writing manifest"}
{"status":"success"}
# List available LLMs
curl https://ollama.jklug.work/api/tags | jq
# Shell output:
{
"models": [
{
"name": "qwen3.5:9b",
"model": "qwen3.5:9b",
"modified_at": "2026-09-27T20:26:50.215763652Z",
"size": 6594474711,
"digest": "6488c96fa5faab64bb65cbd30d4289e20e6130ef535a93ef9a49f42eda893ea7",
"details": {
"parent_model": "",
"format": "gguf",
"family": "qwen35",
"families": [
"qwen35"
],
"parameter_size": "9.7B",
"quantization_level": "Q4_K_M",
"context_length": 262144,
"embedding_length": 4096
},
"capabilities": [
"completion",
"vision",
"tools",
"thinking"
]
}
]
}
# List test query
curl https://ollama.jklug.work/api/chat \
-H 'Content-Type: application/json' \
-d '{
"model": "qwen3.5:9b",
"messages": [{"role": "user", "content": "Who is Linus Torvalds?"}],
"stream": false
}'
# Shell output:
{"model":"qwen3.5:9b","created_at":"2026-09-27T20:30:21.974203572Z","message":{"role":"assistant","content":"**Linus Torvalds** is a Finnish-American software engineer and computer scientist best known as the creator of the **Linux kernel**. ...
Verify GPU Usage #
# List Ollama loaded models
kubectl -n ollama exec deployment/ollama -- ollama ps
# Shell output:
NAME ID SIZE PROCESSOR CONTEXT UNTIL
qwen3.5:9b 6488c96fa5fa 5.5 GB 100% GPU 4096 4 minutes from now
Verify the GPU usage during the query:
# Nvidia GPU monitoring
nvidia-smi
# Shell output:
Sun Sep 27 22:30:53 2026
+-----------------------------------------------------------------------------------------+
| NVIDIA-SMI 615.71.09 KMD Version: 615.71.09 CUDA UMD Version: 13.4 |
+-----------------------------------------+------------------------+----------------------+
| GPU Name Persistence-M | Bus-Id Disp.A | Volatile Uncorr. ECC |
| Fan Temp Perf Pwr:Usage/Cap | Memory-Usage | GPU-Util Compute M. |
| | | MIG M. |
|=========================================+========================+======================|
| 0 NVIDIA GeForce RTX 4070 ... On | 00000000:01:00.0 Off | N/A |
| 0% 58C P2 193W / 220W | 6479MiB / 12282MiB | 92% Default |
| | | N/A |
+-----------------------------------------+------------------------+----------------------+
+-----------------------------------------------------------------------------------------+
| Processes: |
| GPU GI CI PID Type Process name GPU Memory |
| ID ID Usage |
|=========================================================================================|
| 0 N/A N/A 9041 C /usr/lib/ollama/llama-server 6470MiB |
+-----------------------------------------------------------------------------------------+
# Follow GPU usage
watch -n 1 nvidia-smi
Scale down and up #
# Scale deployment down
kubectl -n ollama scale deployment/ollama --replicas=0
# Scale the deployment up
kubectl -n ollama scale deployment/ollama --replicas=1