From c9a825ca519f3c345234bb285dbaa5018f326805 Mon Sep 17 00:00:00 2001 From: Pascal Matthiesen <434505+pmdroid@users.noreply.github.com> Date: Fri, 9 Oct 2026 14:54:17 -0700 Subject: [PATCH 1/7] docs(deploy): add self-hosted Arcade Deploy Helm guide Add 'Arcade Deploy on your own cluster (Helm)' under Operate > Deploy, covering prerequisites, deployments.* chart values, the engine-level resource request keys from monorepo #5306 (not yet exposed by the chart), deploying, status and logs, upgrades, and troubleshooting. Unverified items are marked with TODO comments. Fix build/arcade-deploy and the deploy overview so they say Arcade Deploy also runs in a self-hosted installation, and link the new guide from the Helm page. Refs DEP-371, DEP-365 --- app/en/build/arcade-deploy/page.mdx | 8 +- app/en/operate/deploy/_meta.tsx | 3 + .../deploy/helm-arcade-deploy/page.mdx | 287 ++++++++++++++++++ app/en/operate/deploy/helm/page.mdx | 1 + app/en/operate/deploy/page.mdx | 2 +- 5 files changed, 299 insertions(+), 2 deletions(-) create mode 100644 app/en/operate/deploy/helm-arcade-deploy/page.mdx diff --git a/app/en/build/arcade-deploy/page.mdx b/app/en/build/arcade-deploy/page.mdx index f413f91f7..7918727a6 100644 --- a/app/en/build/arcade-deploy/page.mdx +++ b/app/en/build/arcade-deploy/page.mdx @@ -11,7 +11,7 @@ import { SignupLink } from "@/app/_components/analytics"; Running your MCP servers locally is very convenient during development and testing. Once your MCP server is mature, however, you may want to access it from any MCP client, or to facilitate multi-user support. Doing all that from your computer comes with the complexity of running and maintaining a server, handling auth and high availability for all your users and all the integrations you want to support. Arcade Deploy takes care of all that for you. Your MCP server will be registered to Arcade, adding all the tools you created to the larger tool catalog. From there, you can create MCP Gateways to pick and choose which tools you want to use in your MCP clients, which can be from any connected MCP server. -Arcade Deploy hosts *your* MCP server on Arcade Cloud. It's a feature for serving tools — not a way to deploy the Arcade platform. For a full platform deployment, see the [marketplace guides](/operate/deploy) or [self-host with Helm](/operate/deploy/helm). +Arcade Deploy hosts *your* MCP server, either on Arcade Cloud or in a self-hosted Arcade installation on your own Kubernetes cluster. It's a feature for serving tools, not a way to deploy the Arcade platform. For a full platform deployment, see the [marketplace guides](/operate/deploy) or [self-host with Helm](/operate/deploy/helm). To turn on Arcade Deploy in a self-hosted installation, see [Arcade Deploy on your own cluster](/operate/deploy/helm-arcade-deploy). @@ -68,6 +68,12 @@ If you have not created an MCP server yet, then follow the steps outlined in [th ## Deploy your MCP Server + +Deploying to a self-hosted Arcade installation? Sign in to it first with `arcade login --url `. `arcade deploy` targets the installation you're signed in to. Your platform operator turns deployments on as described in [Arcade Deploy on your own cluster](/operate/deploy/helm-arcade-deploy). + + +{/* TODO(DEP-371): Verify `arcade login --url` + `arcade deploy` against a self-hosted installation end to end (spec SVD1.R2/SVD1.R6 scenarios are still @wip). */} + Run the deploy command in the directory where you started your MCP server (containing your `pyproject.toml` file) and specify the relative path to your entrypoint file via the `--entrypoint/-e` option. ```bash diff --git a/app/en/operate/deploy/_meta.tsx b/app/en/operate/deploy/_meta.tsx index 9ff53c164..5a96c101e 100644 --- a/app/en/operate/deploy/_meta.tsx +++ b/app/en/operate/deploy/_meta.tsx @@ -22,6 +22,9 @@ const meta: MetaRecord = { helm: { title: "Self-host with Helm", }, + "helm-arcade-deploy": { + title: "Arcade Deploy on your own cluster (Helm)", + }, worker: { title: "Arcade Worker", }, diff --git a/app/en/operate/deploy/helm-arcade-deploy/page.mdx b/app/en/operate/deploy/helm-arcade-deploy/page.mdx new file mode 100644 index 000000000..891cf6096 --- /dev/null +++ b/app/en/operate/deploy/helm-arcade-deploy/page.mdx @@ -0,0 +1,287 @@ +--- +title: "Arcade Deploy on your own cluster (Helm)" +description: "Turn on Arcade Deploy in a self-hosted Arcade installation so developers can run arcade deploy against your own Kubernetes cluster" +--- + +import { Callout, Steps } from "nextra/components"; + +# Arcade Deploy on your own cluster (Helm) + +This page is for platform operators who run Arcade with the [Helm chart](/operate/deploy/helm) and want developers to ship custom MCP servers into that installation with `arcade deploy`. Once you turn deployments on, the Arcade Engine runs each deployed server as a workload in your own cluster. Bundles go to an object store you own, and runner images come from a registry you choose. + +{/* TODO(DEP-371): Add the "nothing leaves your environment" guarantee (spec SEC1.R1) once its scenarios are promoted (DEP-238). */} + +{/* TODO(DEP-371): The chart's deployments.* block shipped in chart 1.11.1 (appVersion f0ceab0). Confirm the first chart version that shipped it and state it here as a minimum version. */} + +## How it works + +- A developer runs `arcade deploy` against your installation. The Arcade Engine publishes the server's bundle to your **artifact store** (S3-compatible, GCS, or Azure Blob). +- The Arcade Engine creates the server in the **workers namespace** (`arcade-workers` by default). Each deployed server runs as a Kubernetes Deployment of the **runner image** (`arcadedev/runner`), one pod by default, and the pod fetches its bundle from the artifact store. +- Every deployed server gets its own NetworkPolicy. It accepts traffic only from the Arcade Engine. Outbound, it reaches the public internet and cluster DNS, but not private address ranges, link-local addresses (including cloud metadata), or the cluster ranges you list, unless you allowlist a range. +- The **deployment budget** sizes the workers namespace's quota, which limits how many deployed servers can run at once. + +The Arcade Engine's service account gets a namespaced Role in the workers namespace. That Role covers Deployments, Services, Secrets, ServiceAccounts, NetworkPolicies, ResourceQuotas, Leases, read access to pods and pod logs, and read access to pod metrics. Developers never need cluster access. + +## Prerequisites + +- A working Arcade installation from the [Helm chart](/operate/deploy/helm), with `gateway.hostname`, `ingress.hostname`, or `deployments.domain` set +- A bucket in an S3-compatible store, Google Cloud Storage, or Azure Blob Storage for release bundles +- Credentials for that bucket that the Arcade Engine can pick up from its environment. The Arcade Engine reads them from the ambient cloud credential chain, never from the bucket URL and never from a deploy request. +- Access from your cluster to the runner image, either `arcadedev/runner` on Docker Hub or a copy in your own registry +- A CNI plugin that enforces Kubernetes NetworkPolicies, so the per-server isolation takes effect + +{/* TODO(DEP-371): Document how to hand bucket credentials to the Arcade Engine through the chart. There's no engine service account annotation value for workload identity (IRSA, GKE Workload Identity, Azure Workload Identity) yet; engine.extraEnv / engine.extraEnvFrom are the likely path for static keys. Confirm with the chart owners before publishing. */} +{/* TODO(DEP-371): Confirm whether metrics-server is required. The engine Role reads metrics.k8s.io pods to report CPU and memory usage per server. */} + +## Enable deployments + + + +### Add the deployments values + +Deployments are off by default. Turn them on and name your artifact store: + +```yaml filename="values.yaml" +deployments: + enabled: true + artifacts: + # s3://, gs:// or azblob:// + url: s3://arcade-bundles?region=us-east-1 +``` + +Use an `s3://`, `gs://`, or `azblob://` URL. The chart refuses to render when `deployments.enabled` is `true` and `deployments.artifacts.url` is empty. The Arcade Engine refuses to start when the URL scheme isn't one it supports, and names the supported schemes. + +{/* TODO(DEP-371): The engine's SupportedSchemes also lists mem and file (apparently for tests). Confirm they're not meant for production before saying anything about them. */} + +### Upgrade the release + +Apply the new values to your existing release: + +```bash +helm upgrade arcade \ + oci://public.ecr.aws/s5i6x9d1/charts/arcade \ + --namespace arcade \ + -f values.yaml +``` + +With the defaults, this creates the `arcade-workers` namespace, a default-deny NetworkPolicy for it, a ResourceQuota and LimitRange, and the Arcade Engine's Role in that namespace. + +### Deploy a test server + +Ask a developer to [deploy a server](#deploy-a-server) and check that it reaches **Running**. + + + +## Configuration + +All settings live under `deployments.*` in the chart values. Leave a value at its default unless you need to change it. + +### Core settings + +| Value | Default | Description | +| --- | --- | --- | +| `deployments.enabled` | `false` | Turns deployments on. | +| `deployments.provider` | `kubernetes` | The deployment provider. `kubernetes` runs deployed servers in your cluster. | +| `deployments.mutationsAllowed` | `true` | When `false`, the Arcade Engine blocks creating and updating deployments. | +| `deployments.domain` | `""` | Root domain for deployments, passed to the Arcade Engine. Falls back to `gateway.hostname`, then `ingress.hostname`. The chart refuses to render if none of them is set. In the platform's own cluster, the Arcade Engine reaches deployed servers on their in-cluster Service address. | +| `deployments.timeout` | `5m` | How long a deploy may take to become healthy before it's marked failed. | +| `deployments.autoRoll` | `true` | Rendered to the Arcade Engine as `auto_roll`. {/* TODO(DEP-371): Describe what auto_roll does (moving deployed servers onto an upgraded runner image) once DEP-243/DEP-248 land and the behavior is verified. */} | +| `deployments.reconciler.enabled` | `true` | Runs the deployment reconciler. It keeps the budget quota in step with `deployments.budget.default` and removes NetworkPolicies left behind by removed servers. | +| `deployments.budget.default` | `5` | Deployment budget. The reconciler sizes the `arcade-deployments-budget` quota from it: `budget + 1` pods, so one rollout can surge. {/* TODO(DEP-237): Document the refusal of a deploy at the budget once its scenario is promoted. */} | +| `deployments.readinessProbe` | `rollout` | How the Arcade Engine decides a deployed server is ready. {/* TODO(DEP-371): Document the values other than `rollout` and what each one checks. */} | + +{/* TODO(DEP-371): The PLE1 spec says the budget counts servers that are Running, Deploying, Degraded, or Disabled, that an unsigned default can't exceed 5, and that a license Secret raises it. The license and the budget display on the Servers page are still pending (DEP-237, DEP-246). Document them once they ship. */} + +### Artifact store + +| Value | Default | Description | +| --- | --- | --- | +| `deployments.artifacts.url` | `""` | Bucket URL for release bundles, for example `s3://bundles?region=us-east-1`, `gs://bundles`, or `azblob://bundles`. Required when deployments are on. | + +The deployed server's pod doesn't hold store credentials. The Arcade Engine gives the pod a presigned download URL for its bundle when the store can sign one: + +| Store | Signs when | Otherwise | +| --- | --- | --- | +| S3 | Any credential. With STS session credentials, the URL expires with the session. | The pod reads the store URL directly and needs store credentials. | +| GCS | The Arcade Engine's service account holds `iam.serviceAccounts.signBlob`. | Federated credentials can't sign, so the pod falls back to the store URL. | +| Azure Blob | A shared key is configured (`AZURE_STORAGE_KEY`). | Managed identity can't mint SAS URLs, so the pod falls back to the store URL. | + +A signing failure doesn't fail the deploy. + +### Runner image and registry + +| Value | Default | Description | +| --- | --- | --- | +| `deployments.image.repository` | `arcadedev/runner` | Runner image that deployed servers run in. | +| `deployments.image.tag` | `""` | Runner image tag. Empty uses the chart's `appVersion`, so the runner moves with chart upgrades. | +| `deployments.registry.server` | `""` | Registry host to pull the runner image from. When set, it's prefixed to `deployments.image.repository`. | +| `deployments.registry.imagePullSecrets` | `[]` | Names of image pull Secrets for deployed servers' pods. When `dockerRegistry.enabled` is `true` and the chart creates that Secret, the chart adds it to this list. | + +For a private mirror, push `arcadedev/runner` to your registry and point the chart at it: + +```yaml filename="values.yaml" +deployments: + registry: + server: registry.example.internal + imagePullSecrets: + - my-registry-pull-secret +``` + +### Sizing + +Every deployed server gets the same size. + +| Value | Default | Description | +| --- | --- | --- | +| `deployments.sizing.instances` | `1` | Instances per deployed server. {/* TODO(DEP-371): The PLE1 spec lists HA and autoscaling as non-goals (one serving instance per release). Confirm whether values above 1 are supported before documenting them. */} | +| `deployments.sizing.cpuLimit` | `1.0` | CPU limit per deployed server, in cores. | +| `deployments.sizing.memoryLimit` | `512` | Memory limit per deployed server, in mebibytes (an integer, not a quantity string). | +| `deployments.runtime.className` | `""` | RuntimeClass for deployed servers, for example a sandboxed runtime. Empty uses the cluster default. | +| `deployments.runtime.nodeSelector` | `{}` | Node selector for deployed servers' pods. | +| `deployments.runtime.tolerations` | `[]` | Tolerations for deployed servers' pods. | + +With the chart, each pod's requests equal its limits, so pods run in the Guaranteed QoS class. Each pod also has a fixed ephemeral-storage request and limit of `6Gi`, which covers a `5Gi` volume for the bundle and its virtual environment. + +#### Resource requests and ephemeral storage + +{/* TODO(DEP-454): These keys landed in the Arcade Engine with monorepo PR #5306 (merged 2026-10-09). They aren't in a released chart yet (chart 1.11.1 ships appVersion f0ceab0, which predates #5306), and the chart has no values for them: it renders only instances, cpu_limit and memory_limit, and engine.extraConfig can't set `deployments:`. Replace this subsection with deployments.sizing.* chart values once the chart exposes them and a chart release carries them. */} + + +Not yet available in a released chart. Newer Arcade Engine builds can reserve less than the limit for each deployed server. The chart doesn't expose these settings yet. + + +The Arcade Engine reads these settings from `deployments.kubernetes.instance` in its own configuration: + +| Resource | Request | Limit | Default | +| --- | --- | --- | --- | +| CPU | `cpu_request` (cores) | `cpu_limit` (cores) | Request equals the limit | +| Memory | `memory_request` (mebibytes) | `memory_limit` (mebibytes) | Request equals the limit | +| Ephemeral storage | `ephemeral_storage_request` (quantity) | `ephemeral_storage_limit` (quantity) | Request equals the limit; limit `6Gi` | + +```yaml filename="engine.yaml" +deployments: + kubernetes: + instance: + cpu_request: 0.1 + memory_request: 256 + ephemeral_storage_request: 512Mi + cpu_limit: 1 + memory_limit: 512 + ephemeral_storage_limit: 6Gi +``` + +- A request below its limit puts the pod in the Burstable QoS class. Under node pressure, the kubelet evicts Burstable pods before Guaranteed pods. +- The Arcade Engine refuses to start when a request exceeds its limit, a request is negative, or an ephemeral-storage value isn't a Kubernetes quantity. +- Keep `ephemeral_storage_limit` at `6Gi` or higher. The bundle, temporary, and home volumes add up to `6Gi`. +- The Arcade Engine sizes the namespace quota from limits, not requests. + +### Workers namespace + +| Value | Default | Description | +| --- | --- | --- | +| `deployments.namespace` | `arcade-workers` | Namespace deployed servers run in. | +| `deployments.createNamespace` | `true` | Creates the namespace with the `restricted` Pod Security Standard (enforce, audit, and warn) and a default-deny NetworkPolicy. Helm keeps both on uninstall. | +| `deployments.resourceQuota.enabled` | `true` | Renders a ResourceQuota named `-workers-quota` from `deployments.resourceQuota.hard`. | +| `deployments.resourceQuota.hard` | `requests.cpu: 10`, `requests.memory: 20Gi`, `limits.cpu: 20`, `limits.memory: 40Gi`, `pods: 50` | Hard limits for that quota. | +| `deployments.limitRange.enabled` | `true` | Renders a LimitRange for containers in the namespace. | +| `deployments.limitRange.default` | `cpu: 1`, `memory: 512Mi` | Default container limits. | +| `deployments.limitRange.defaultRequest` | `cpu: 250m`, `memory: 256Mi` | Default container requests. | + +To use a namespace you manage yourself, set `deployments.createNamespace: false`. The chart then adds no default-deny NetworkPolicy, unless you also set `deployments.network.defaultDenyExistingNamespace: true`. Only do that when the namespace holds nothing but deployed servers. + +The reconciler also maintains its own quota, `arcade-deployments-budget`. It allows `budget + 1` pods (one spare for a rollout), and sizes CPU and memory as that pod count times the per-server limits. + +{/* TODO(DEP-371): The runbook says arcade-deployments-budget should be the only quota in the workers namespace, but the chart also renders -workers-quota by default. Confirm the recommended deployments.resourceQuota.enabled setting when the reconciler is on. */} + +### Network + +| Value | Default | Description | +| --- | --- | --- | +| `deployments.egressAllowlist` | `[]` | CIDRs deployed servers may reach in addition to the public internet, for example an internal API. | +| `deployments.network.clusterCIDRs` | `[]` | Your cluster's pod and service CIDRs. Deployed servers can't reach them. `10.0.0.0/8`, `172.16.0.0/12`, `192.168.0.0/16`, `100.64.0.0/10`, and `169.254.0.0/16` stay unreachable as well. Ranges in `deployments.egressAllowlist` are reachable, and so is DNS (TCP and UDP port 53) to the resolvers below. | +| `deployments.network.dns.mode` | `cluster` | `cluster` uses cluster DNS. `external` replaces pod DNS with `deployments.network.dns.nameservers` and removes every cluster-DNS exception. | +| `deployments.network.dns.nameservers` | `[]` | Public IPv4 nameservers. Required in `external` mode. | +| `deployments.network.dns.namespace` | `kube-system` | Namespace of the cluster DNS pods (`cluster` mode only). | +| `deployments.network.dns.podSelector` | `{}` | Labels of the cluster DNS pods, for example `k8s-app: kube-dns` (`cluster` mode only). Empty allows every pod in that namespace. | +| `deployments.network.dns.cidrs` | `[]` | Extra resolver IP ranges, such as a NodeLocal DNS `/32`. These allow TCP and UDP port 53 only (`cluster` mode only). | + +`external` DNS mode also needs a bundle endpoint the pod can reach without cluster DNS. Existing deployed servers pick up a DNS mode change only after a redeploy. + +{/* TODO(DEP-371): Sterling's split into use-case guides (internal networking, IPv6, a different registry) fits here. The egress rules today are written against IPv4 ranges; confirm IPv6 behavior before writing an IPv6 guide. */} + +## Deploy a server + +Developers use the same workflow as on Arcade Cloud. They sign in to your installation once, then deploy from the project directory: + +```bash +arcade login --url https://arcade.example.com +arcade deploy -e src/my_server/server.py +``` + +`arcade login --url` saves a context for your installation, and `arcade deploy` targets the active context. See [Arcade Deploy](/build/arcade-deploy) for the full developer guide and the [CLI cheat sheet](/references/cli-cheat-sheet) for every flag. + +- Deploy MCP servers built with the Arcade MCP framework for Python. The Arcade Engine refuses a bundle without a Python project definition or with a missing entrypoint. +- The Arcade Engine refuses a changed redeploy while the previous one is still rolling out. +- The Arcade Engine also refuses a bundle over the installation's size limit, and discards a virtual environment that the bundle ships. +- A redeploy that changes the server's code or secrets replaces the release under the same name and gateway URL. A redeploy of a running release that changes nothing is a no-op. Deploying again after a failed release rolls it out again, even if nothing changed. + +{/* TODO(DEP-371): Saved-context sign-in and the framework gate in the CLI (spec SVD1.R2, SVD1.R4, SVD1.R6) are still @wip in the monorepo behaviors. Verify `arcade login --url` + `arcade deploy` end to end against a self-hosted install before publishing. */} +{/* TODO(DEP-371): Document non-interactive CI deploys (installation URL plus a project API key in the environment, per spec SVD1.R2): the exact environment variable names and the minimum CLI version. */} +{/* TODO(DEP-371): State the bundle size limit and where it's configured. */} + +## Check status and logs + +### Dashboard and CLI + +The **Servers** page in the dashboard lists deployed servers and their status. A running server whose instance keeps crashing shows as **Degraded**. A removed server disappears from the list. + +From the CLI: + +```bash +arcade server list +arcade server get +arcade server logs -f +``` + +Logs cover the current release. After a crash, they also include the immediately previous instance. Arcade doesn't keep logs beyond that, so ship pod logs to your own logging stack if you need them longer. + +{/* TODO(DEP-371): The spec's status words are Deploying, Running, Degraded, Failed, Disabled, Removed. The listing scenarios are still pending (DEP-239). Confirm which appear in the self-hosted dashboard today. Degraded for an unreachable artifact store is still pending (DEP-243/DEP-248); per-server CPU and memory usage is pending (DEP-235). */} + +### kubectl + +Deployed servers are ordinary Kubernetes objects in the workers namespace, so your existing tooling works: + +```bash +kubectl -n arcade-workers get deployments,pods,services +kubectl -n arcade-workers get networkpolicies -l app.kubernetes.io/managed-by=arcade-engine +kubectl -n arcade-workers get resourcequota +``` + +## Upgrades + +Upgrade the chart with `helm upgrade` as described in [Self-host with Helm](/operate/deploy/helm#upgrade-and-roll-back). Keep these points in mind: + +- **Runner image.** With `deployments.image.tag` empty, the runner tag follows the chart's `appVersion`. Pin `deployments.image.tag` to control when the runner changes. +- **Private registry.** Mirror the new runner tag into your registry before you upgrade. +- **Namespace.** With `deployments.createNamespace: true`, Helm keeps the workers namespace and its default-deny NetworkPolicy on uninstall, so deployed servers' objects aren't deleted with the release. + +{/* TODO(DEP-371): The PLE1 spec says an upgrade preserves each release and moves deployed servers onto the upgraded runtime one at a time, and that an installation can defer that move. These scenarios are still @wip (DEP-243, DEP-248, DEP-251). Document the upgrade flow once they're promoted. */} + +## Troubleshooting + +| Symptom | Cause and fix | +| --- | --- | +| `helm upgrade` fails with `deployments.artifacts.url must name an S3-compatible artifact store when deployments.enabled is true` | Set `deployments.artifacts.url`. | +| `helm upgrade` fails with `deployments.domain (or gateway.hostname / ingress.hostname) must be set when deployments.enabled is true` | Set `deployments.domain`, `gateway.hostname`, or `ingress.hostname`. | +| `helm upgrade` fails with `engine.extraConfig sets deployments:` | The chart owns the `deployments:` block in the Arcade Engine's configuration. Use `deployments.*` chart values instead of `engine.extraConfig`. | +| The Arcade Engine doesn't start and names an unsupported scheme | Use an `s3://`, `gs://`, or `azblob://` URL in `deployments.artifacts.url`. | +| A deploy fails because the server never became healthy | The pod started but didn't answer its health check within `deployments.timeout`. Read the server's logs, and check that it binds the port the platform probes. If many servers fail at once, suspect the runner image. If start-up is just slow, raise `deployments.timeout`. | +| Deploys stop creating pods | A ResourceQuota in the workers namespace is full. Run `kubectl -n arcade-workers get resourcequota`. The tighter of `arcade-deployments-budget` and `-workers-quota` is the one that refuses pods. Raise `deployments.budget.default` or `deployments.resourceQuota.hard`, and make sure your nodes can hold the extra servers. | +| A deployed server can't reach an internal service | Add the service's CIDR to `deployments.egressAllowlist`. | +| NetworkPolicies stay behind after you remove servers | With `deployments.reconciler.enabled: false`, nothing removes them. Delete NetworkPolicies labeled `app.kubernetes.io/managed-by=arcade-engine` by hand, or turn the reconciler back on. | + +## Next steps + +- [Deploy an MCP server with Arcade Deploy](/build/arcade-deploy) +- [Create an MCP Gateway](/operate/governance/mcp-gateways) to give clients access to deployed servers' tools +- [Platform architecture](/operate/deploy/architecture) diff --git a/app/en/operate/deploy/helm/page.mdx b/app/en/operate/deploy/helm/page.mdx index 41ddf4677..8a81b5a29 100644 --- a/app/en/operate/deploy/helm/page.mdx +++ b/app/en/operate/deploy/helm/page.mdx @@ -77,5 +77,6 @@ Upgrade to a new chart version with `helm upgrade`, and roll back with `helm rol ## Next steps +- [Turn on Arcade Deploy](/operate/deploy/helm-arcade-deploy) so developers can deploy custom MCP servers into your cluster - [Create an MCP Gateway](/operate/governance/mcp-gateways) to scope tools and auth for each client - [Set up a User Source](/operate/identity/user-sources) to authenticate end users with your own identity provider diff --git a/app/en/operate/deploy/page.mdx b/app/en/operate/deploy/page.mdx index 1d5652374..3acb1b8db 100644 --- a/app/en/operate/deploy/page.mdx +++ b/app/en/operate/deploy/page.mdx @@ -44,7 +44,7 @@ The marketplace and Helm options are **full platform deployments**. The features These connect your tools and clients to Arcade. They are not platform deployments: -- [**Arcade Deploy**](/build/arcade-deploy): host *your* MCP server on Arcade Cloud with the `arcade deploy` command. +- [**Arcade Deploy**](/build/arcade-deploy): host *your* MCP server with the `arcade deploy` command, on Arcade Cloud or [on your own cluster](/operate/deploy/helm-arcade-deploy) in a self-hosted installation. - [**Hybrid MCP servers**](/operate/deploy/on-prem): run MCP servers in your own environment and connect them to Arcade Cloud, so tools reach private resources. ## Customizing auth From 8d55deccf0539bfe44c3c4ad7db58000a18d7c21 Mon Sep 17 00:00:00 2001 From: Pascal Matthiesen <434505+pmdroid@users.noreply.github.com> Date: Fri, 9 Oct 2026 15:02:25 -0700 Subject: [PATCH 2/7] docs(deploy): don't claim the runner rolls with chart upgrades --- app/en/operate/deploy/helm-arcade-deploy/page.mdx | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/app/en/operate/deploy/helm-arcade-deploy/page.mdx b/app/en/operate/deploy/helm-arcade-deploy/page.mdx index 891cf6096..994dbf3d0 100644 --- a/app/en/operate/deploy/helm-arcade-deploy/page.mdx +++ b/app/en/operate/deploy/helm-arcade-deploy/page.mdx @@ -113,7 +113,7 @@ A signing failure doesn't fail the deploy. | Value | Default | Description | | --- | --- | --- | | `deployments.image.repository` | `arcadedev/runner` | Runner image that deployed servers run in. | -| `deployments.image.tag` | `""` | Runner image tag. Empty uses the chart's `appVersion`, so the runner moves with chart upgrades. | +| `deployments.image.tag` | `""` | Runner image tag. Empty uses the chart's `appVersion`. | | `deployments.registry.server` | `""` | Registry host to pull the runner image from. When set, it's prefixed to `deployments.image.repository`. | | `deployments.registry.imagePullSecrets` | `[]` | Names of image pull Secrets for deployed servers' pods. When `dockerRegistry.enabled` is `true` and the chart creates that Secret, the chart adds it to this list. | @@ -261,7 +261,7 @@ kubectl -n arcade-workers get resourcequota Upgrade the chart with `helm upgrade` as described in [Self-host with Helm](/operate/deploy/helm#upgrade-and-roll-back). Keep these points in mind: -- **Runner image.** With `deployments.image.tag` empty, the runner tag follows the chart's `appVersion`. Pin `deployments.image.tag` to control when the runner changes. +- **Runner image.** With `deployments.image.tag` empty, the runner tag follows the chart's `appVersion`. Pin `deployments.image.tag` to control which runner tag the Arcade Engine configures. - **Private registry.** Mirror the new runner tag into your registry before you upgrade. - **Namespace.** With `deployments.createNamespace: true`, Helm keeps the workers namespace and its default-deny NetworkPolicy on uninstall, so deployed servers' objects aren't deleted with the release. From 9738c5f99952adc7741e413192e124f7880a0730 Mon Sep 17 00:00:00 2001 From: "github-actions[bot]" <41898282+github-actions[bot]@users.noreply.github.com> Date: Fri, 9 Oct 2026 22:22:03 +0000 Subject: [PATCH 3/7] =?UTF-8?q?=F0=9F=A4=96=20Regenerate=20LLMs.txt?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- public/llms.txt | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/public/llms.txt b/public/llms.txt index 9dbeba352..9f3c2d8f0 100644 --- a/public/llms.txt +++ b/public/llms.txt @@ -1,4 +1,4 @@ - + # Arcade @@ -79,6 +79,7 @@ Arcade docs serve two audiences. Start with the path that matches your goal: - [Adding Resource Server Auth to Your MCP Server](https://docs.arcade.dev/en/build/create-tools/secure-your-server/secure-your-mcp-server): This documentation page teaches you how to secure an HTTP MCP server using OAuth 2.1 Resource Server authentication, enabling tool-level authorization and secrets for self-hosted deployments. It explains what Resource Server auth is, why it's needed to protect HTTP MCP servers, and provides guidance on configuring your server to validate Bearer tokens and support multiple authorization servers. - [Agentic development](https://docs.arcade.dev/en/get-started/setup/connect-arcade-docs): This documentation page provides guidance on utilizing agents in Integrated Development Environments (IDEs) to enhance development efficiency by accessing well-formatted markdown documentation directly from the Arcade site. It explains how AI agents can retrieve content without manual copying and introduces the LLM - [Arcade Cloud infrastructure](https://docs.arcade.dev/en/operate/deploy/arcade-cloud): Arcade Cloud infrastructure documentation explains the fully-managed SaaS platform's hosting, data storage, and security features, including information about data sovereignty (US-based), egress IP addresses, VPC peering options, and encryption standards. It details how Arcade stores and manages different categories of data—including user-controlled application data, platform-controlled monitoring data, and training data for ML/AI models—along with user controls for data retention, deletion, and opt-out options for training consent. +- [Arcade Deploy on your own cluster (Helm)](https://docs.arcade.dev/en/operate/deploy/helm-arcade-deploy): This documentation enables platform operators to configure Arcade Deploy on their Kubernetes cluster using Helm, allowing developers to deploy custom MCP servers into the cluster via the `arcade deploy` command. The page explains the deployment architecture, prerequisites (including artifact storage and CNI networking), and step-by-step instructions for enabling deployments through Helm chart configuration. - [Arcade Gateway Assistant](https://docs.arcade.dev/en/operate/governance/mcp-gateways/create-via-ai): The Arcade Gateway Assistant allows users to create and manage MCP gateways through natural language commands in their chat interface by connecting an MCP-compatible client to the Gateway Assistant server. After authentication with an Arcade account, users can describe what integrations they need (like Gmail and Google Calendar) and the AI will automatically select appropriate tools and create a gateway, which can then be added back to their chat client for use. - [Arcade Glossary](https://docs.arcade.dev/en/resources/glossary): The Arcade Glossary provides definitions and explanations of key terms and concepts related to the Arcade platform, including agents, harnesses, MCP servers, and tools. This resource helps users understand the components and functionalities necessary for building, deploying, and managing applications that - [Arcade with Agent Frameworks and MCP Clients](https://docs.arcade.dev/en/get-started/agent-frameworks): This documentation page provides developers with guidance on integrating Arcade with agent frameworks and MCP clients to enhance AI applications with tool-calling capabilities. It offers detailed instructions on authentication, tool loading, and execution, along with code examples and configuration steps for popular frameworks and From 8984d7bd92e7bf56b4d8abb04525064622ff7e02 Mon Sep 17 00:00:00 2001 From: Pascal Matthiesen <434505+pmdroid@users.noreply.github.com> Date: Sat, 10 Oct 2026 14:05:52 -0700 Subject: [PATCH 4/7] docs(deploy): resolve self-hosted Deploy lookup TODOs Answer seven DEP-371 lookup TODOs from monorepo and arcade-mcp source: minimum chart version, metrics-server, mem/file schemes, readinessProbe values, sizing.instances, CI deploy env vars, and bundle size limits. The resourceQuota recommendation stays a TODO: no source states one. --- .../operate/deploy/helm-arcade-deploy/page.mdx | 16 ++++++++-------- 1 file changed, 8 insertions(+), 8 deletions(-) diff --git a/app/en/operate/deploy/helm-arcade-deploy/page.mdx b/app/en/operate/deploy/helm-arcade-deploy/page.mdx index 994dbf3d0..9f29641b4 100644 --- a/app/en/operate/deploy/helm-arcade-deploy/page.mdx +++ b/app/en/operate/deploy/helm-arcade-deploy/page.mdx @@ -11,7 +11,7 @@ This page is for platform operators who run Arcade with the [Helm chart](/operat {/* TODO(DEP-371): Add the "nothing leaves your environment" guarantee (spec SEC1.R1) once its scenarios are promoted (DEP-238). */} -{/* TODO(DEP-371): The chart's deployments.* block shipped in chart 1.11.1 (appVersion f0ceab0). Confirm the first chart version that shipped it and state it here as a minimum version. */} +The `deployments.*` values first shipped in chart 1.11.0. Use chart 1.11.1 or later: in 1.11.0 you must set `deployments.image.repository` yourself, or the chart fails to render, and there are no `deployments.runtime.*` values. ## How it works @@ -31,7 +31,7 @@ The Arcade Engine's service account gets a namespaced Role in the workers namesp - A CNI plugin that enforces Kubernetes NetworkPolicies, so the per-server isolation takes effect {/* TODO(DEP-371): Document how to hand bucket credentials to the Arcade Engine through the chart. There's no engine service account annotation value for workload identity (IRSA, GKE Workload Identity, Azure Workload Identity) yet; engine.extraEnv / engine.extraEnvFrom are the likely path for static keys. Confirm with the chart owners before publishing. */} -{/* TODO(DEP-371): Confirm whether metrics-server is required. The engine Role reads metrics.k8s.io pods to report CPU and memory usage per server. */} +metrics-server is optional. The Arcade Engine's Role may read pod metrics, but the Arcade Engine doesn't query them yet, so nothing breaks without it. ## Enable deployments @@ -51,7 +51,7 @@ deployments: Use an `s3://`, `gs://`, or `azblob://` URL. The chart refuses to render when `deployments.enabled` is `true` and `deployments.artifacts.url` is empty. The Arcade Engine refuses to start when the URL scheme isn't one it supports, and names the supported schemes. -{/* TODO(DEP-371): The engine's SupportedSchemes also lists mem and file (apparently for tests). Confirm they're not meant for production before saying anything about them. */} +The Arcade Engine also accepts `mem://` and `file://`, but only its tests use them. They don't work in a cluster, because a deployed server's pod can't read the Arcade Engine's memory or local disk. ### Upgrade the release @@ -88,7 +88,7 @@ All settings live under `deployments.*` in the chart values. Leave a value at it | `deployments.autoRoll` | `true` | Rendered to the Arcade Engine as `auto_roll`. {/* TODO(DEP-371): Describe what auto_roll does (moving deployed servers onto an upgraded runner image) once DEP-243/DEP-248 land and the behavior is verified. */} | | `deployments.reconciler.enabled` | `true` | Runs the deployment reconciler. It keeps the budget quota in step with `deployments.budget.default` and removes NetworkPolicies left behind by removed servers. | | `deployments.budget.default` | `5` | Deployment budget. The reconciler sizes the `arcade-deployments-budget` quota from it: `budget + 1` pods, so one rollout can surge. {/* TODO(DEP-237): Document the refusal of a deploy at the budget once its scenario is promoted. */} | -| `deployments.readinessProbe` | `rollout` | How the Arcade Engine decides a deployed server is ready. {/* TODO(DEP-371): Document the values other than `rollout` and what each one checks. */} | +| `deployments.readinessProbe` | `rollout` | How the Arcade Engine decides a deployed server is ready. `rollout` waits for the Kubernetes rollout to finish, which needs the pod to accept TCP connections on the server's port. Any other value, such as `health`, also waits until the server's `/worker/health` endpoint answers HTTP 200 to the Arcade Engine. | {/* TODO(DEP-371): The PLE1 spec says the budget counts servers that are Running, Deploying, Degraded, or Disabled, that an unsigned default can't exceed 5, and that a license Secret raises it. The license and the budget display on the Servers page are still pending (DEP-237, DEP-246). Document them once they ship. */} @@ -133,7 +133,7 @@ Every deployed server gets the same size. | Value | Default | Description | | --- | --- | --- | -| `deployments.sizing.instances` | `1` | Instances per deployed server. {/* TODO(DEP-371): The PLE1 spec lists HA and autoscaling as non-goals (one serving instance per release). Confirm whether values above 1 are supported before documenting them. */} | +| `deployments.sizing.instances` | `1` | Instances per deployed server. Keep it at `1`: Arcade Deploy runs one serving instance per server, without high availability or autoscaling. A higher value makes the Arcade Engine run that many pods per server and size the budget quota at `budget × instances + 1` pods. | | `deployments.sizing.cpuLimit` | `1.0` | CPU limit per deployed server, in cores. | | `deployments.sizing.memoryLimit` | `512` | Memory limit per deployed server, in mebibytes (an integer, not a quantity string). | | `deployments.runtime.className` | `""` | RuntimeClass for deployed servers, for example a sandboxed runtime. Empty uses the cluster default. | @@ -222,12 +222,12 @@ arcade deploy -e src/my_server/server.py - Deploy MCP servers built with the Arcade MCP framework for Python. The Arcade Engine refuses a bundle without a Python project definition or with a missing entrypoint. - The Arcade Engine refuses a changed redeploy while the previous one is still rolling out. -- The Arcade Engine also refuses a bundle over the installation's size limit, and discards a virtual environment that the bundle ships. +- The Arcade Engine also refuses a bundle over its 16 MiB limit or one that unpacks past 256 MiB, and discards a virtual environment that the bundle ships. The 16 MiB limit caps the upload request, with a little headroom for encoding. Both limits are built into the Arcade Engine; no chart or configuration value changes them. - A redeploy that changes the server's code or secrets replaces the release under the same name and gateway URL. A redeploy of a running release that changes nothing is a no-op. Deploying again after a failed release rolls it out again, even if nothing changed. {/* TODO(DEP-371): Saved-context sign-in and the framework gate in the CLI (spec SVD1.R2, SVD1.R4, SVD1.R6) are still @wip in the monorepo behaviors. Verify `arcade login --url` + `arcade deploy` end to end against a self-hosted install before publishing. */} -{/* TODO(DEP-371): Document non-interactive CI deploys (installation URL plus a project API key in the environment, per spec SVD1.R2): the exact environment variable names and the minimum CLI version. */} -{/* TODO(DEP-371): State the bundle size limit and where it's configured. */} + +For non-interactive CI deploys, set `ARCADE_URL` and `ARCADE_API_KEY` (a project API key) in the CI environment, then run `arcade deploy`. With both set, the CLI skips the saved context and browser sign-in and uses `ARCADE_URL` directly as the Arcade Engine's URL, without discovery. This needs `arcade-mcp` 1.16.0 or later. ## Check status and logs From 1de2c2aa03f97c33750568463cc92703ba309248 Mon Sep 17 00:00:00 2001 From: Pascal Matthiesen <434505+pmdroid@users.noreply.github.com> Date: Sat, 10 Oct 2026 15:23:54 -0700 Subject: [PATCH 5/7] docs(deploy): document runtime moves, autoRoll and states; drop budget Close four DEP-371 TODOs from monorepo source (runtime move and halt, autoRoll, deployment states, artifact store and usage tracking) and remove every deployment budget and license mention (DEP-237, DEP-246). --- .../deploy/helm-arcade-deploy/page.mdx | 36 +++++++++++-------- 1 file changed, 21 insertions(+), 15 deletions(-) diff --git a/app/en/operate/deploy/helm-arcade-deploy/page.mdx b/app/en/operate/deploy/helm-arcade-deploy/page.mdx index 9f29641b4..6f55647f7 100644 --- a/app/en/operate/deploy/helm-arcade-deploy/page.mdx +++ b/app/en/operate/deploy/helm-arcade-deploy/page.mdx @@ -9,7 +9,7 @@ import { Callout, Steps } from "nextra/components"; This page is for platform operators who run Arcade with the [Helm chart](/operate/deploy/helm) and want developers to ship custom MCP servers into that installation with `arcade deploy`. Once you turn deployments on, the Arcade Engine runs each deployed server as a workload in your own cluster. Bundles go to an object store you own, and runner images come from a registry you choose. -{/* TODO(DEP-371): Add the "nothing leaves your environment" guarantee (spec SEC1.R1) once its scenarios are promoted (DEP-238). */} +Bundles stay in your environment: the chart requires an artifact store (`deployments.artifacts.url`), and the Arcade Engine publishes bundles only there. Usage-event tracking (`usageTracking.enabled`) is off by default in the chart. Leave it off so your installation sends no usage events to Arcade. The `deployments.*` values first shipped in chart 1.11.0. Use chart 1.11.1 or later: in 1.11.0 you must set `deployments.image.repository` yourself, or the chart fails to render, and there are no `deployments.runtime.*` values. @@ -18,7 +18,6 @@ The `deployments.*` values first shipped in chart 1.11.0. Use chart 1.11.1 or la - A developer runs `arcade deploy` against your installation. The Arcade Engine publishes the server's bundle to your **artifact store** (S3-compatible, GCS, or Azure Blob). - The Arcade Engine creates the server in the **workers namespace** (`arcade-workers` by default). Each deployed server runs as a Kubernetes Deployment of the **runner image** (`arcadedev/runner`), one pod by default, and the pod fetches its bundle from the artifact store. - Every deployed server gets its own NetworkPolicy. It accepts traffic only from the Arcade Engine. Outbound, it reaches the public internet and cluster DNS, but not private address ranges, link-local addresses (including cloud metadata), or the cluster ranges you list, unless you allowlist a range. -- The **deployment budget** sizes the workers namespace's quota, which limits how many deployed servers can run at once. The Arcade Engine's service account gets a namespaced Role in the workers namespace. That Role covers Deployments, Services, Secrets, ServiceAccounts, NetworkPolicies, ResourceQuotas, Leases, read access to pods and pod logs, and read access to pod metrics. Developers never need cluster access. @@ -85,13 +84,10 @@ All settings live under `deployments.*` in the chart values. Leave a value at it | `deployments.mutationsAllowed` | `true` | When `false`, the Arcade Engine blocks creating and updating deployments. | | `deployments.domain` | `""` | Root domain for deployments, passed to the Arcade Engine. Falls back to `gateway.hostname`, then `ingress.hostname`. The chart refuses to render if none of them is set. In the platform's own cluster, the Arcade Engine reaches deployed servers on their in-cluster Service address. | | `deployments.timeout` | `5m` | How long a deploy may take to become healthy before it's marked failed. | -| `deployments.autoRoll` | `true` | Rendered to the Arcade Engine as `auto_roll`. {/* TODO(DEP-371): Describe what auto_roll does (moving deployed servers onto an upgraded runner image) once DEP-243/DEP-248 land and the behavior is verified. */} | -| `deployments.reconciler.enabled` | `true` | Runs the deployment reconciler. It keeps the budget quota in step with `deployments.budget.default` and removes NetworkPolicies left behind by removed servers. | -| `deployments.budget.default` | `5` | Deployment budget. The reconciler sizes the `arcade-deployments-budget` quota from it: `budget + 1` pods, so one rollout can surge. {/* TODO(DEP-237): Document the refusal of a deploy at the budget once its scenario is promoted. */} | +| `deployments.autoRoll` | `true` | When `true`, the reconciler moves deployed servers onto a new runner image one at a time (see [Upgrades](#upgrades)). When `false`, the move is deferred and deployed servers stay on the runner image they already run. | +| `deployments.reconciler.enabled` | `true` | Runs the deployment reconciler. It moves deployed servers onto a new runner image and removes NetworkPolicies left behind by removed servers. | | `deployments.readinessProbe` | `rollout` | How the Arcade Engine decides a deployed server is ready. `rollout` waits for the Kubernetes rollout to finish, which needs the pod to accept TCP connections on the server's port. Any other value, such as `health`, also waits until the server's `/worker/health` endpoint answers HTTP 200 to the Arcade Engine. | -{/* TODO(DEP-371): The PLE1 spec says the budget counts servers that are Running, Deploying, Degraded, or Disabled, that an unsigned default can't exceed 5, and that a license Secret raises it. The license and the budget display on the Servers page are still pending (DEP-237, DEP-246). Document them once they ship. */} - ### Artifact store | Value | Default | Description | @@ -133,7 +129,7 @@ Every deployed server gets the same size. | Value | Default | Description | | --- | --- | --- | -| `deployments.sizing.instances` | `1` | Instances per deployed server. Keep it at `1`: Arcade Deploy runs one serving instance per server, without high availability or autoscaling. A higher value makes the Arcade Engine run that many pods per server and size the budget quota at `budget × instances + 1` pods. | +| `deployments.sizing.instances` | `1` | Instances per deployed server. Keep it at `1`: Arcade Deploy runs one serving instance per server, without high availability or autoscaling. A higher value makes the Arcade Engine run that many pods per server. | | `deployments.sizing.cpuLimit` | `1.0` | CPU limit per deployed server, in cores. | | `deployments.sizing.memoryLimit` | `512` | Memory limit per deployed server, in mebibytes (an integer, not a quantity string). | | `deployments.runtime.className` | `""` | RuntimeClass for deployed servers, for example a sandboxed runtime. Empty uses the cluster default. | @@ -189,8 +185,6 @@ deployments: To use a namespace you manage yourself, set `deployments.createNamespace: false`. The chart then adds no default-deny NetworkPolicy, unless you also set `deployments.network.defaultDenyExistingNamespace: true`. Only do that when the namespace holds nothing but deployed servers. -The reconciler also maintains its own quota, `arcade-deployments-budget`. It allows `budget + 1` pods (one spare for a rollout), and sizes CPU and memory as that pod count times the per-server limits. - {/* TODO(DEP-371): The runbook says arcade-deployments-budget should be the only quota in the workers namespace, but the chart also renders -workers-quota by default. Confirm the recommended deployments.resourceQuota.enabled setting when the reconciler is on. */} ### Network @@ -233,7 +227,15 @@ For non-interactive CI deploys, set `ARCADE_URL` and `ARCADE_API_KEY` (a project ### Dashboard and CLI -The **Servers** page in the dashboard lists deployed servers and their status. A running server whose instance keeps crashing shows as **Degraded**. A removed server disappears from the list. +The **Servers** page in the dashboard lists deployed servers. A removed server disappears from the list. The Arcade Engine tracks each deployed server in one of these states, and `arcade deploy` waits for a final one: + +| State | Meaning | +| --- | --- | +| `pending` | A new server is rolling out. The dashboard shows **Deploying**. | +| `updating` | A redeploy is rolling out. The Arcade Engine refuses another redeploy of the server until it finishes. The dashboard shows **Updating**. | +| `running` | The current release is ready and serving. | +| `degraded` | No instance of the current release is serving, for example because it keeps crashing, or a rollout failed while the previous instance still serves. The status message gives the reason. | +| `failed` | The release didn't become ready, because it failed to start or `deployments.timeout` ran out, and nothing is serving it. | From the CLI: @@ -245,8 +247,6 @@ arcade server logs -f Logs cover the current release. After a crash, they also include the immediately previous instance. Arcade doesn't keep logs beyond that, so ship pod logs to your own logging stack if you need them longer. -{/* TODO(DEP-371): The spec's status words are Deploying, Running, Degraded, Failed, Disabled, Removed. The listing scenarios are still pending (DEP-239). Confirm which appear in the self-hosted dashboard today. Degraded for an unreachable artifact store is still pending (DEP-243/DEP-248); per-server CPU and memory usage is pending (DEP-235). */} - ### kubectl Deployed servers are ordinary Kubernetes objects in the workers namespace, so your existing tooling works: @@ -265,7 +265,13 @@ Upgrade the chart with `helm upgrade` as described in [Self-host with Helm](/ope - **Private registry.** Mirror the new runner tag into your registry before you upgrade. - **Namespace.** With `deployments.createNamespace: true`, Helm keeps the workers namespace and its default-deny NetworkPolicy on uninstall, so deployed servers' objects aren't deleted with the release. -{/* TODO(DEP-371): The PLE1 spec says an upgrade preserves each release and moves deployed servers onto the upgraded runtime one at a time, and that an installation can defer that move. These scenarios are still @wip (DEP-243, DEP-248, DEP-251). Document the upgrade flow once they're promoted. */} +### Moving deployed servers onto a new runner image + +When an upgrade changes the runner image tag, the reconciler moves running deployed servers onto it one at a time. This needs `deployments.reconciler.enabled: true`. + +- On each pass (every 30 seconds), the reconciler starts at most one move, and only when no other move is still rolling out. +- If a server can't start on the new runner image, the Arcade Engine rolls it back to the previous one and marks it `degraded` with the message "could not start on the upgraded runtime; the previous runtime is still serving". The move then halts: no other server moves onto that image. +- Set `deployments.autoRoll: false` to defer the move. Deployed servers then stay on the runner image they already run. ## Troubleshooting @@ -276,7 +282,7 @@ Upgrade the chart with `helm upgrade` as described in [Self-host with Helm](/ope | `helm upgrade` fails with `engine.extraConfig sets deployments:` | The chart owns the `deployments:` block in the Arcade Engine's configuration. Use `deployments.*` chart values instead of `engine.extraConfig`. | | The Arcade Engine doesn't start and names an unsupported scheme | Use an `s3://`, `gs://`, or `azblob://` URL in `deployments.artifacts.url`. | | A deploy fails because the server never became healthy | The pod started but didn't answer its health check within `deployments.timeout`. Read the server's logs, and check that it binds the port the platform probes. If many servers fail at once, suspect the runner image. If start-up is just slow, raise `deployments.timeout`. | -| Deploys stop creating pods | A ResourceQuota in the workers namespace is full. Run `kubectl -n arcade-workers get resourcequota`. The tighter of `arcade-deployments-budget` and `-workers-quota` is the one that refuses pods. Raise `deployments.budget.default` or `deployments.resourceQuota.hard`, and make sure your nodes can hold the extra servers. | +| Deploys stop creating pods | A ResourceQuota in the workers namespace is full. Run `kubectl -n arcade-workers get resourcequota` to see which one. For `-workers-quota`, raise `deployments.resourceQuota.hard`, and make sure your nodes can hold the extra servers. | | A deployed server can't reach an internal service | Add the service's CIDR to `deployments.egressAllowlist`. | | NetworkPolicies stay behind after you remove servers | With `deployments.reconciler.enabled: false`, nothing removes them. Delete NetworkPolicies labeled `app.kubernetes.io/managed-by=arcade-engine` by hand, or turn the reconciler back on. | From b3f861ffa69fa1b32f75638444c98982a7d5629a Mon Sep 17 00:00:00 2001 From: Pascal Matthiesen <434505+pmdroid@users.noreply.github.com> Date: Sat, 10 Oct 2026 16:16:09 -0700 Subject: [PATCH 6/7] docs(deploy): document bucket credentials for self-hosted Deploy Replace the bucket-credentials TODO: EKS Pod Identity and static keys work with chart 1.11.1; IRSA, Azure and GKE Workload Identity and a GCP key file need the next chart release (engine workload-identity hooks). Note that deployed servers' pods hold no bucket credentials and rely on presigned URLs. --- .../deploy/helm-arcade-deploy/page.mdx | 126 +++++++++++++++++- 1 file changed, 119 insertions(+), 7 deletions(-) diff --git a/app/en/operate/deploy/helm-arcade-deploy/page.mdx b/app/en/operate/deploy/helm-arcade-deploy/page.mdx index 6f55647f7..7b5c861be 100644 --- a/app/en/operate/deploy/helm-arcade-deploy/page.mdx +++ b/app/en/operate/deploy/helm-arcade-deploy/page.mdx @@ -25,13 +25,125 @@ The Arcade Engine's service account gets a namespaced Role in the workers namesp - A working Arcade installation from the [Helm chart](/operate/deploy/helm), with `gateway.hostname`, `ingress.hostname`, or `deployments.domain` set - A bucket in an S3-compatible store, Google Cloud Storage, or Azure Blob Storage for release bundles -- Credentials for that bucket that the Arcade Engine can pick up from its environment. The Arcade Engine reads them from the ambient cloud credential chain, never from the bucket URL and never from a deploy request. +- Credentials for that bucket that the Arcade Engine can pick up from its environment. The Arcade Engine reads them from the ambient cloud credential chain, never from the bucket URL and never from a deploy request. See [Bucket credentials](#bucket-credentials). - Access from your cluster to the runner image, either `arcadedev/runner` on Docker Hub or a copy in your own registry - A CNI plugin that enforces Kubernetes NetworkPolicies, so the per-server isolation takes effect -{/* TODO(DEP-371): Document how to hand bucket credentials to the Arcade Engine through the chart. There's no engine service account annotation value for workload identity (IRSA, GKE Workload Identity, Azure Workload Identity) yet; engine.extraEnv / engine.extraEnvFrom are the likely path for static keys. Confirm with the chart owners before publishing. */} metrics-server is optional. The Arcade Engine's Role may read pod metrics, but the Arcade Engine doesn't query them yet, so nothing breaks without it. +## Bucket credentials + +The Arcade Engine opens the artifact store with each cloud SDK's default credential lookup. Give credentials to the Arcade Engine's ServiceAccount, `-engine` in the release namespace, or put them in its environment. Pick one method: + +| Method | Store | Chart | +| --- | --- | --- | +| [EKS Pod Identity](#eks-pod-identity) | S3 | 1.11.1 or later | +| [Static keys in a Secret](#static-keys) | S3, Azure Blob | 1.11.1 or later | +| [EKS IRSA](#workload-identity-and-key-files) | S3 | The first chart release after 1.11.1 | +| [Azure Workload Identity](#workload-identity-and-key-files) | Azure Blob | The first chart release after 1.11.1 | +| [GKE Workload Identity](#workload-identity-and-key-files) | GCS | The first chart release after 1.11.1 | +| [GCP service account key file](#workload-identity-and-key-files) | GCS | The first chart release after 1.11.1 | + +The identity you give the Arcade Engine needs to read and write objects in the bucket. Deployed servers don't need bucket credentials as long as the Arcade Engine can sign download URLs. See [Artifact store](#artifact-store). + +### EKS Pod Identity + +EKS Pod Identity needs no chart values. With the [EKS Pod Identity Agent](https://docs.aws.amazon.com/eks/latest/userguide/pod-id-agent-setup.html) installed, create an IAM role that EKS Pod Identity can assume and associate it with the Arcade Engine's ServiceAccount. For a release named `arcade` in the `arcade` namespace: + +```bash +aws eks create-pod-identity-association \ + --cluster-name my-cluster \ + --namespace arcade \ + --service-account arcade-engine \ + --role-arn arn:aws:iam::111122223333:role/arcade-engine +``` + +See [EKS Pod Identity](https://docs.aws.amazon.com/eks/latest/userguide/pod-identities.html) for the role's trust policy. + +### Static keys + +Put the keys in a Secret in the release namespace and pass it to the Arcade Engine with `engine.extraEnvFrom`: + +- S3: `AWS_ACCESS_KEY_ID` and `AWS_SECRET_ACCESS_KEY` +- Azure Blob: `AZURE_STORAGE_ACCOUNT` and `AZURE_STORAGE_KEY`. A shared key also lets the Arcade Engine sign download URLs. + +```bash +kubectl -n arcade create secret generic arcade-bundle-store \ + --from-literal=AWS_ACCESS_KEY_ID=... \ + --from-literal=AWS_SECRET_ACCESS_KEY=... +``` + +```yaml filename="values.yaml" +engine: + extraEnvFrom: + - secretRef: + name: arcade-bundle-store +``` + +### Workload identity and key files + + +These methods need the first chart release after 1.11.1. Chart 1.11.1 has no values for ServiceAccount annotations, pod labels, or extra volumes. + + +Set up the identity on the cloud side first, following [EKS IRSA](https://docs.aws.amazon.com/eks/latest/userguide/iam-roles-for-service-accounts.html), [Azure Workload Identity](https://learn.microsoft.com/en-us/azure/aks/workload-identity-overview), or [GKE Workload Identity](https://cloud.google.com/kubernetes-engine/docs/how-to/workload-identity) for the Arcade Engine's ServiceAccount. Then set the matching values: + +| Value | Default | Description | +| --- | --- | --- | +| `engine.serviceAccount.annotations` | `{}` | Annotations on the Arcade Engine's ServiceAccount. | +| `engine.podLabels` | `{}` | Labels on the Arcade Engine's pods. Keys the chart already sets are rejected. | +| `engine.podAnnotations` | `{}` | Annotations on the Arcade Engine's pods. `checksum/*` keys are rejected. | +| `engine.extraVolumes` | `[]` | Extra volumes for the Arcade Engine's pods. | +| `engine.extraVolumeMounts` | `[]` | Extra volume mounts for the Arcade Engine's container. | + +The identity applies to the Arcade Engine's Deployment only. The Arcade Engine's Jobs (migration, gateway bootstrap, worker reconcile) use their own ServiceAccounts. Extra volumes and mounts also go into those Jobs. + +EKS IRSA: + +```yaml filename="values.yaml" +engine: + serviceAccount: + annotations: + eks.amazonaws.com/role-arn: arn:aws:iam::123456789012:role/arcade-engine +``` + +Azure Workload Identity. Also name the storage account, with `AZURE_STORAGE_ACCOUNT` in `engine.extraEnv` or `storage_account=` in `deployments.artifacts.url`: + +```yaml filename="values.yaml" +engine: + serviceAccount: + annotations: + azure.workload.identity/client-id: 00000000-0000-0000-0000-000000000000 + podLabels: + azure.workload.identity/use: "true" +``` + +GKE Workload Identity: + +```yaml filename="values.yaml" +engine: + serviceAccount: + annotations: + iam.gke.io/gcp-service-account: arcade-engine@my-project.iam.gserviceaccount.com +``` + +GCP service account key file, mounted from a Secret named `gcp-key` with a `key.json` entry: + +```yaml filename="values.yaml" +engine: + extraVolumes: + - name: gcp-key + secret: + secretName: gcp-key + extraVolumeMounts: + - name: gcp-key + mountPath: /var/secrets/gcp + readOnly: true + extraEnv: + - name: GOOGLE_APPLICATION_CREDENTIALS + value: /var/secrets/gcp/key.json +``` + ## Enable deployments @@ -94,15 +206,15 @@ All settings live under `deployments.*` in the chart values. Leave a value at it | --- | --- | --- | | `deployments.artifacts.url` | `""` | Bucket URL for release bundles, for example `s3://bundles?region=us-east-1`, `gs://bundles`, or `azblob://bundles`. Required when deployments are on. | -The deployed server's pod doesn't hold store credentials. The Arcade Engine gives the pod a presigned download URL for its bundle when the store can sign one: +Deployed servers' pods get no bucket credentials: their ServiceAccount mounts no token, and their NetworkPolicy blocks the cloud metadata endpoint. Instead, the Arcade Engine gives each pod a presigned download URL for its bundle when it can sign one: | Store | Signs when | Otherwise | | --- | --- | --- | -| S3 | Any credential. With STS session credentials, the URL expires with the session. | The pod reads the store URL directly and needs store credentials. | -| GCS | The Arcade Engine's service account holds `iam.serviceAccounts.signBlob`. | Federated credentials can't sign, so the pod falls back to the store URL. | -| Azure Blob | A shared key is configured (`AZURE_STORAGE_KEY`). | Managed identity can't mint SAS URLs, so the pod falls back to the store URL. | +| S3 | Any credential. With STS session credentials, the URL expires with the session. | The pod falls back to the store URL, which it has no credentials for. | +| GCS | A key file is configured, or the Arcade Engine's service account holds `iam.serviceAccounts.signBlob`. | The pod falls back to the store URL, which it has no credentials for. | +| Azure Blob | A shared key is configured (`AZURE_STORAGE_KEY`). | Workload and managed identities can't sign, so the pod falls back to the store URL, which it has no credentials for. | -A signing failure doesn't fail the deploy. +A signing failure doesn't fail the deploy, but the pod then can't fetch its bundle. With Azure Workload Identity or GKE Workload Identity, make sure signing works as described above. ### Runner image and registry From 5fbff1601c3c86cdd8621b6fa02c319960bce628 Mon Sep 17 00:00:00 2001 From: Pascal Matthiesen <434505+pmdroid@users.noreply.github.com> Date: Sat, 10 Oct 2026 18:56:47 -0700 Subject: [PATCH 7/7] docs(deploy): qualify autoRoll, rollback reason and signing failure; describe engine quota --- app/en/operate/deploy/helm-arcade-deploy/page.mdx | 12 +++++++----- 1 file changed, 7 insertions(+), 5 deletions(-) diff --git a/app/en/operate/deploy/helm-arcade-deploy/page.mdx b/app/en/operate/deploy/helm-arcade-deploy/page.mdx index 7b5c861be..0d17bf7f7 100644 --- a/app/en/operate/deploy/helm-arcade-deploy/page.mdx +++ b/app/en/operate/deploy/helm-arcade-deploy/page.mdx @@ -196,7 +196,7 @@ All settings live under `deployments.*` in the chart values. Leave a value at it | `deployments.mutationsAllowed` | `true` | When `false`, the Arcade Engine blocks creating and updating deployments. | | `deployments.domain` | `""` | Root domain for deployments, passed to the Arcade Engine. Falls back to `gateway.hostname`, then `ingress.hostname`. The chart refuses to render if none of them is set. In the platform's own cluster, the Arcade Engine reaches deployed servers on their in-cluster Service address. | | `deployments.timeout` | `5m` | How long a deploy may take to become healthy before it's marked failed. | -| `deployments.autoRoll` | `true` | When `true`, the reconciler moves deployed servers onto a new runner image one at a time (see [Upgrades](#upgrades)). When `false`, the move is deferred and deployed servers stay on the runner image they already run. | +| `deployments.autoRoll` | `true` | When `true`, the reconciler moves deployed servers onto a new runner image one at a time (see [Upgrades](#upgrades)). When `false`, the move is deferred and deployed servers stay on the runner image they already run until they're redeployed. A redeploy picks up the configured runner image. | | `deployments.reconciler.enabled` | `true` | Runs the deployment reconciler. It moves deployed servers onto a new runner image and removes NetworkPolicies left behind by removed servers. | | `deployments.readinessProbe` | `rollout` | How the Arcade Engine decides a deployed server is ready. `rollout` waits for the Kubernetes rollout to finish, which needs the pod to accept TCP connections on the server's port. Any other value, such as `health`, also waits until the server's `/worker/health` endpoint answers HTTP 200 to the Arcade Engine. | @@ -214,7 +214,7 @@ Deployed servers' pods get no bucket credentials: their ServiceAccount mounts no | GCS | A key file is configured, or the Arcade Engine's service account holds `iam.serviceAccounts.signBlob`. | The pod falls back to the store URL, which it has no credentials for. | | Azure Blob | A shared key is configured (`AZURE_STORAGE_KEY`). | Workload and managed identities can't sign, so the pod falls back to the store URL, which it has no credentials for. | -A signing failure doesn't fail the deploy, but the pod then can't fetch its bundle. With Azure Workload Identity or GKE Workload Identity, make sure signing works as described above. +A signing failure doesn't fail the deploy, but when no valid signed URL remains, the pod can't fetch its bundle. With Azure Workload Identity or GKE Workload Identity, make sure signing works as described above. ### Runner image and registry @@ -297,7 +297,9 @@ deployments: To use a namespace you manage yourself, set `deployments.createNamespace: false`. The chart then adds no default-deny NetworkPolicy, unless you also set `deployments.network.defaultDenyExistingNamespace: true`. Only do that when the namespace holds nothing but deployed servers. -{/* TODO(DEP-371): The runbook says arcade-deployments-budget should be the only quota in the workers namespace, but the chart also renders -workers-quota by default. Confirm the recommended deployments.resourceQuota.enabled setting when the reconciler is on. */} +When the reconciler is on, the Arcade Engine also applies its own ResourceQuota, `arcade-deployments-budget`, in the workers namespace. It limits `pods`, `limits.cpu`, and `limits.memory`, and the reconciler re-applies it on every pass, so manual edits to it don't last. A pod has to fit every quota in the namespace. + +{/* TODO(DEP-371): Confirm the recommended deployments.resourceQuota.enabled setting when the reconciler is on. */} ### Network @@ -382,8 +384,8 @@ Upgrade the chart with `helm upgrade` as described in [Self-host with Helm](/ope When an upgrade changes the runner image tag, the reconciler moves running deployed servers onto it one at a time. This needs `deployments.reconciler.enabled: true`. - On each pass (every 30 seconds), the reconciler starts at most one move, and only when no other move is still rolling out. -- If a server can't start on the new runner image, the Arcade Engine rolls it back to the previous one and marks it `degraded` with the message "could not start on the upgraded runtime; the previous runtime is still serving". The move then halts: no other server moves onto that image. -- Set `deployments.autoRoll: false` to defer the move. Deployed servers then stay on the runner image they already run. +- If a server can't start on the new runner image, the Arcade Engine rolls it back to the previous one and marks it `degraded` with the message "could not start on the upgraded runtime; the previous runtime is still serving". If no instance of the server is serving when it's marked `degraded`, the reason is "release artifact unavailable" instead. The move then halts: no other server moves onto that image. +- Set `deployments.autoRoll: false` to defer the move. Deployed servers then stay on the runner image they already run until they're redeployed. ## Troubleshooting