Decision. Run batch jobs in-cluster via Workflows
Alternative. Keeping a second CI runner fleet for jobs
Why. Jobs run where the data lives, and the runner count stays at one (GitLab CI only triggers them).
Runs batch jobs and CI-style pipelines as containerised workflow steps
Argo Workflows runs batch and CI-style jobs directly on the cluster: containerised steps in DAGs, with cluster RBAC and storage available to every step.
Excerpt from argocd-apps/values.yaml in the argocd-apps chart,
with annotations added for this site:
argo-workflows:
application: true
project: cluster-services
# false
autoSync: false
The full argo-workflows/values.yaml from the service's
own chart:
argo-workflows:
## Custom resource configuration
crds:
install: true
keep: true
singleNamespace: false
workflow:
serviceAccount:
create: true
name: "argo-workflow-sa"
rbac:
create: true
agentPermissions: true
artifactGC: true
controller:
#parallelism: 4
rbac:
create: true
configMap:
create: true
metricsConfig:
# Monitoring: expose workflow-controller Prometheus metrics on
# argo-workflows-workflow-controller:8080 (scraped by the `monitoring`
# Alloy DaemonSet → New Relic).
enabled: true
# -- the controller container's securityContext
securityContext:
readOnlyRootFilesystem: true
runAsNonRoot: true
allowPrivilegeEscalation: false
capabilities:
drop:
- ALL
persistence:
archive: true
postgresql:
host: postgres-pooler-rw.postgres.svc.cluster.local
port: 5432
database: argo_workflows
tableName: argo_workflows
userNameSecret:
name: postgres-user-secret
key: username
passwordSecret:
name: postgres-user-secret
key: password
ssl: true
sslMode: require
telemetryConfig:
enabled: false
serviceMonitor:
enabled: false
serviceAccount:
create: true
name: "argo-workflows-controller-sa"
name: controller
# -- Specify all namespaces where this workflow controller instance will manage
# workflows. This controls where the service account and RBAC resources will
# be created. Only valid when singleNamespace is false.
workflowNamespaces:
- argo-workflows
- openvino
instanceID:
enabled: false
useReleaseName: false
logging:
# -- Set the logging level (one of: `debug`, `info`, `warn`, `error`)
level: info
# -- Set the glog logging level
globallevel: "0"
# -- Set the logging format (one of: `text`, `json`)
format: "text"
# -- Service type of the controller Service
serviceType: ClusterIP
# -- Configure liveness [probe] for the controller
# @default -- See [values.yaml]
livenessProbe:
httpGet:
port: 6060
path: /healthz
failureThreshold: 3
initialDelaySeconds: 90
periodSeconds: 60
timeoutSeconds: 30
# -- Extra environment variables to provide to the controller container
extraEnv: []
# - name: FOO
# value: "bar"
# -- Extra arguments to be added to the controller
extraArgs: []
# -- Additional volume mounts to the controller main container
#volumeMounts:
#- name: database-ca-cert
# mountPath: /certs/
# subPath: ca.pem
#volumes:
#- name: database-ca-cert
# secret:
# secretName: postgres-user-secret
# items:
# - key: ca_cert
# path: ca.pem
replicas: 1
# -- The number of revisions to keep.
revisionHistoryLimit: 10
pdb:
# -- Configure [Pod Disruption Budget] for the controller pods
enabled: false
# minAvailable: 1
# maxUnavailable: 1
# -- [Node selector]
nodeSelector:
kubernetes.io/os: linux
clusterWorkflowTemplates:
enabled: true
nodeEvents:
enabled: true
workflowEvents:
enabled: true
server:
enabled: true
name: server
serviceType: ClusterIP
servicePort: 2746
authModes:
- server
serviceAccount:
create: true
name: "argo-workflows-server-sa"
# -- Volume to be mounted in Pods for temporary files.
tmpVolume:
emptyDir: {}
# -- Additional volume mounts to the server main container.
volumeMounts: []
# -- Additional volumes to the server pod.
volumes: []
ingress:
enabled: false
clusterWorkflowTemplates:
enabled: true
enableEditing: true
# -- Extra containers to be added to the server deployment
extraContainers: []
# -- Enables init containers to be added to the server deployment
extraInitContainers: []
# -- Specify postStart and preStop lifecycle hooks for server container
lifecycle: {}
# -- terminationGracePeriodSeconds for container lifecycle hook
terminationGracePeriodSeconds: 30
livenessProbe:
enabled: false
# -- Array of extra K8s manifests to deploy
extraObjects:
- apiVersion: external-secrets.io/v1
kind: ExternalSecret
metadata:
name: postgres-user-external-secret
namespace: argo-workflows
spec:
refreshInterval: 24h
secretStoreRef:
kind: ClusterSecretStore
name: vault-cluster-secret-store
target:
name: postgres-user-secret
creationPolicy: Owner
data:
- secretKey: password
remoteRef:
key: postgres
property: user_password
- secretKey: username
remoteRef:
key: postgres
property: user_name
- apiVersion: external-secrets.io/v1
kind: ExternalSecret
metadata:
name: object-storage-external-secret
namespace: argo-workflows
spec:
refreshInterval: 1h
secretStoreRef:
kind: ClusterSecretStore
name: vault-cluster-secret-store
target:
name: object-storage-secret
creationPolicy: Owner
template:
type: Opaque
metadata:
annotations:
reflector.v1.k8s.emberstack.com/reflection-allowed: "true"
reflector.v1.k8s.emberstack.com/reflection-allowed-namespaces: "openvino, automq"
reflector.v1.k8s.emberstack.com/reflection-auto-enabled: "true"
reflector.v1.k8s.emberstack.com/reflection-auto-namespaces: "openvino, automq"
data:
- secretKey: API_ENDPOINT_URL
remoteRef:
key: rustfs
property: internal-api-endpoint
- secretKey: WEB_ENDPOINT_URL
remoteRef:
key: rustfs
property: internal-console-endpoint
- secretKey: secret-key
remoteRef:
key: rustfs
property: admin-secret-key
- secretKey: access-key
remoteRef:
key: rustfs
property: admin-access-key
useStaticCredentials: true
artifactRepository:
archiveLogs: true
s3:
bucket: argo-workflows
endpoint: rustfs-service.rustfs.svc.cluster.local:9000
insecure: true
pathStyleForceEnabled: true
accessKeySecret:
name: object-storage-secret
key: access-key
secretKeySecret:
name: object-storage-secret
key: secret-key templates/http-route.yaml Gateway API HTTPRoute exposing the service through the Istio gateway under opi5cluster.co.uk.
apiVersion: gateway.networking.k8s.io/v1
kind: HTTPRoute
metadata:
name: {{ .Release.Namespace }}-httproute
namespace: {{ .Release.Namespace }}
annotations:
link.argocd.argoproj.io/external-link: "https://argo-workflows.opi5cluster.co.uk"
spec:
parentRefs:
- name: istio-gateway
namespace: istio
sectionName: websecure
hostnames:
- argo-workflows.opi5cluster.co.uk
rules:
- backendRefs:
- name: argo-workflows-server
port: 2746 templates/roles.yaml RBAC roles for the Argo Workflows service accounts.
apiVersion: rbac.authorization.k8s.io/v1
kind: Role
metadata:
name: argo-workflow-role
namespace: argo-workflows
rules:
- apiGroups: ["argoproj.io"]
resources: ["workflowtaskresults"]
verbs: ["create", "patch"]
---
apiVersion: rbac.authorization.k8s.io/v1
kind: RoleBinding
metadata:
name: argo-workflow-role-binding
namespace: argo-workflows
subjects:
- kind: ServiceAccount
name: default
namespace: argo-workflows
roleRef:
kind: Role
name: argo-workflow-role
apiGroup: rbac.authorization.k8s.io Decision. Run batch jobs in-cluster via Workflows
Alternative. Keeping a second CI runner fleet for jobs
Why. Jobs run where the data lives, and the runner count stays at one (GitLab CI only triggers them).