Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
12 changes: 10 additions & 2 deletions .env.example
Original file line number Diff line number Diff line change
Expand Up @@ -45,6 +45,14 @@ MODAL_ENVIRONMENT=
AWS_ACCESS_KEY_ID=
AWS_SECRET_ACCESS_KEY=

# --- RunPod provider -------------------------------------------------------
# One API key, from https://console.runpod.io/user/settings. It needs READ+WRITE on
# Pods: Nebula creates and deletes them, and also creates container-registry-auth
# objects when a workload uses an imagePullSecret. A read-only key registers fine
# and then fails every provision with an auth error, which blocklists the whole
# provider until it is replaced.
RUNPOD_API_KEY=

# --- Additional providers (add as adapters land) ---------------------------
# Each provider gets its OWN secret (see hack/deploy.sh PROVIDER_SECRETS), e.g.:
# RUNPOD_API_KEY=
# Each provider gets its OWN secret; see hack/deploy.sh PROVIDER_SECRETS and the
# RunPod block above for the shape.
3 changes: 3 additions & 0 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -62,6 +62,9 @@ metadata:
spec:
providers:
- name: modal # NeoCloud; regions omitted = place anywhere (cheapest)
- name: runpod # NeoCloud, OnDemand only; a region is a geography
regions: # ("us") or one data center ("EU-RO-1")
- us
- name: aws # hyperscaler; "us" expands to every US region
regions:
- us
Expand Down
2 changes: 1 addition & 1 deletion api/v1alpha1/nodepool_types.go
Original file line number Diff line number Diff line change
Expand Up @@ -181,7 +181,7 @@ const (
)

// CapacityType is the purchase model (the outer axis). Each provider maps it to
// its own concept — e.g. RunPod Spot -> interruptible/podRentInterruptable.
// its own concept — e.g. AWS Spot -> a spot-market CreateFleet request.
// +kubebuilder:validation:Enum=Spot;OnDemand
type CapacityType string

Expand Down
20 changes: 7 additions & 13 deletions cmd/main.go
Original file line number Diff line number Diff line change
Expand Up @@ -58,6 +58,7 @@ import (
awsprovider "github.com/InftyAI/Nebula/pkg/provider/aws"
"github.com/InftyAI/Nebula/pkg/provider/fake"
"github.com/InftyAI/Nebula/pkg/provider/modal"
"github.com/InftyAI/Nebula/pkg/provider/runpod"
"github.com/InftyAI/Nebula/pkg/vnode"
// +kubebuilder:scaffold:imports
)
Expand Down Expand Up @@ -544,7 +545,7 @@ func setupKubeletServer(mgr ctrl.Manager, addr, clientCA string, servingTLSBoots
// +kubebuilder:rbac:groups=certificates.k8s.io,resources=certificatesigningrequests,resourceNames=nebula-kubelet-serving,verbs=delete;get
// +kubebuilder:rbac:groups=certificates.k8s.io,resources=certificatesigningrequests/approval,resourceNames=nebula-kubelet-serving,verbs=update
// +kubebuilder:rbac:groups=certificates.k8s.io,resources=signers,resourceNames=kubernetes.io/kubelet-serving,verbs=approve
// +kubebuilder:rbac:groups="",resources=users,resourceNames={"system:node:nebula-aws","system:node:nebula-modal","system:node:nebula-fake"},verbs=impersonate
// +kubebuilder:rbac:groups="",resources=users,resourceNames={"system:node:nebula-aws","system:node:nebula-modal","system:node:nebula-runpod","system:node:nebula-fake"},verbs=impersonate
// +kubebuilder:rbac:groups="",resources=groups,resourceNames="system:nodes",verbs=impersonate

// addServingCertificateBootstrap requests a trusted serving certificate for the kubelet
Expand Down Expand Up @@ -647,21 +648,14 @@ func registerProviders(ctx context.Context, c client.Client, enabled map[string]
return modal.NewSDKClient(ctx, appName, os.Getenv("MODAL_ENVIRONMENT"))
})

// AWS. There is NO region env/flag: the regions this provider may use are declared
// per-pool in the NodePool (ProviderSpec.Regions) and read at call time via the
// region source below, so a pool added at runtime widens the fan-out without a
// restart. One AWS provider spans every such region (per-region clients are built
// lazily). The adapter is otherwise self-configuring: it resolves each region's
// GPU AMI and default-VPC subnets itself, so no launch template or pre-created
// infra is needed. Credentials are secrets and are NEVER read here: the SDK client
// uses the default credential chain (IRSA / instance-role / AWS_ACCESS_KEY_ID
// delivered via a Secret), and one account-global credential authorizes every
// region. Registration only fails (and is a non-fatal skip) if the price catalog
// cannot load — region config can no longer make it fail.
register(provider.ProviderAWS, func() (provider.Provider, error) {
return awsprovider.NewSDKClient(ctx, awsRegionSource(c))
})

register(provider.ProviderRunPod, func() (provider.Provider, error) {
return runpod.NewSDKClient(ctx)
})

// The fake provider is an in-memory backend used only by the e2e suite to
// exercise the full control-plane loop without cloud credentials. It ships in
// the binary but registers ONLY when explicitly enabled, so it can never place
Expand All @@ -675,7 +669,7 @@ func registerProviders(ctx context.Context, c client.Client, enabled map[string]

// knownProviders are the names --providers accepts, one per register call above. The fake
// provider is not among them: it stays gated on its env var alone.
var knownProviders = []string{provider.ProviderModal, provider.ProviderAWS}
var knownProviders = []string{provider.ProviderModal, provider.ProviderAWS, provider.ProviderRunPod}
Comment thread
kerthcet marked this conversation as resolved.

// parseProviders turns --providers into the enabled set. An unknown name is an error rather
// than ignored, so a typo cannot silently leave a provider off.
Expand Down
33 changes: 33 additions & 0 deletions cmd/main_test.go
Original file line number Diff line number Diff line change
Expand Up @@ -18,9 +18,15 @@ package main

import (
"maps"
"os"
"slices"
"strings"
"testing"

rbacv1 "k8s.io/api/rbac/v1"
"sigs.k8s.io/yaml"

"github.com/InftyAI/Nebula/pkg/vnode"
)

func TestParseProviders(t *testing.T) {
Expand Down Expand Up @@ -58,3 +64,30 @@ func TestParseProviders(t *testing.T) {
})
}
}

// TestImpersonateGrantCoversKnownProviders guards the serving-certificate bootstrap: it
// impersonates whichever provider registers first, and a missing grant only surfaces at
// runtime as a Forbidden retry loop. Reads the generated role, so a stale `make manifests`
// fails too.
func TestImpersonateGrantCoversKnownProviders(t *testing.T) {
raw, err := os.ReadFile("../config/rbac/role.yaml")
if err != nil {
t.Fatal(err)
}
var role rbacv1.ClusterRole
if err := yaml.Unmarshal(raw, &role); err != nil {
t.Fatal(err)
}
var users []string
for _, r := range role.Rules {
if slices.Contains(r.Resources, "users") && slices.Contains(r.Verbs, "impersonate") {
users = append(users, r.ResourceNames...)
}
}
for _, name := range knownProviders {
if id := vnode.NodeIdentity(vnode.NodeName(name)); !slices.Contains(users, id) {
t.Errorf("role.yaml grants no impersonate on %s; add it to the marker in main.go "+
"and run `make manifests`", id)
}
}
}
1 change: 1 addition & 0 deletions config/catalog/kustomization.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -17,6 +17,7 @@ configMapGenerator:
files:
- modal.csv=../../pkg/provider/catalog/data/modal.csv
- aws.csv=../../pkg/provider/catalog/data/aws.csv
- runpod.csv=../../pkg/provider/catalog/data/runpod.csv

generatorOptions:
# Stable name (no content-hash suffix) so `kubectl edit` and the volume
Expand Down
2 changes: 1 addition & 1 deletion config/crd/bases/nebula.inftyai.com_nodepools.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -90,7 +90,7 @@ spec:
items:
description: |-
CapacityType is the purchase model (the outer axis). Each provider maps it to
its own concept — e.g. RunPod Spot -> interruptible/podRentInterruptable.
its own concept — e.g. AWS Spot -> a spot-market CreateFleet request.
enum:
- Spot
- OnDemand
Expand Down
14 changes: 9 additions & 5 deletions config/manager/manager.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -67,7 +67,7 @@ spec:
# Every provider is registered by default. Restrict with --providers; it is the
# only way to turn AWS off, which registers even without credentials. Drain a
# provider first: once dropped, nothing terminates its running instances.
# - --providers=modal
# - --providers=runpod
image: controller:latest
name: manager
imagePullPolicy: IfNotPresent
Expand Down Expand Up @@ -131,10 +131,14 @@ spec:
- secretRef:
name: nebula-aws-credentials
optional: true
# Add one secretRef per provider as adapters land, e.g.:
# - secretRef:
# name: nebula-runpod-credentials
# optional: true
# RunPod: a single API key (RUNPOD_API_KEY). Unlike AWS there is no ambient
# identity to fall back on, so an absent Secret means the provider is simply
# skipped at registration. Regions come from the NodePool, so this is the only
# RunPod config here.
- secretRef:
name: nebula-runpod-credentials
optional: true
# Add one secretRef per provider as adapters land, following the pattern above.
ports:
# The kubelet API the API server dials for `kubectl logs` (10250, like a real
# kubelet). Declaring it is documentation and NetworkPolicy surface; the
Expand Down
1 change: 1 addition & 0 deletions config/rbac/role.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -66,6 +66,7 @@ rules:
- system:node:nebula-aws
- system:node:nebula-fake
- system:node:nebula-modal
- system:node:nebula-runpod
resources:
- users
verbs:
Expand Down
2 changes: 1 addition & 1 deletion config/samples/deployment.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -39,7 +39,7 @@ spec:
app: gpu-workload-sample
nebula.inftyai.com/enabled: "true"
nebula.inftyai.com/nodepool: sample
nebula.inftyai.com/accelerator-type: t4
nebula.inftyai.com/accelerator-type: l4
spec:
# Do NOT set nodeName or a provider nodeSelector yourself — the placement
# controller fills the nodeSelector in when it ungates the Pod. Setting
Expand Down
10 changes: 9 additions & 1 deletion config/samples/nodepool.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -19,7 +19,10 @@ spec:
- us
- eu
- ap-melbourne
# - name: runpod
- name: runpod
regions:
- us
- eu
# Outer axis: try OnDemand on every provider first, fall back to Spot.
capacityTypes:
- OnDemand
Expand All @@ -32,6 +35,11 @@ spec:
# sandbox stays reachable through its connect URL and token under every mode — it just
# cannot call out.
#
# Setting ANY mode other than Open narrows this pool to the providers that can enforce
# it: placement skips a provider whose Capabilities report SupportsEgressPolicy=false
# (RunPod, which exposes no outbound knob at all), rather than provisioning something
# with open internet access under a policy that says otherwise.
#
# Blocked permits nothing:
# egress:
# mode: Blocked
Expand Down
1 change: 1 addition & 0 deletions docs/deploy.md
Original file line number Diff line number Diff line change
Expand Up @@ -118,6 +118,7 @@ itself at startup (see [Webhook TLS](#webhook-tls-no-cert-manager)).
| `MODAL_ENVIRONMENT` | Modal | no | Modal Environment to create sandboxes in. Blank omits the key and the SDK uses the token profile's default. See [Modal Environments](#modal-environments). |
| `AWS_ACCESS_KEY_ID` | AWS | dev only | Prefer IRSA / instance role in production and leave blank — the SDK's default credential chain finds the role. Set only for local/dev. |
| `AWS_SECRET_ACCESS_KEY` | AWS | dev only | Pairs with `AWS_ACCESS_KEY_ID`; both required together or both blank. |
| `RUNPOD_API_KEY` | RunPod | yes | From [console.runpod.io/user/settings](https://console.runpod.io/user/settings). Needs **read+write on Pods**: a read-only key registers fine and then fails every provision with an auth error, which blocklists the whole provider. Unlike AWS there is no ambient identity, so blank skips RunPod entirely. |

Non-secret config, passed as `make` variables:

Expand Down
44 changes: 42 additions & 2 deletions docs/status.md
Original file line number Diff line number Diff line change
Expand Up @@ -25,6 +25,7 @@ enters the system.
- [Provider mappings](#provider-mappings)
- [AWS](#aws)
- [Modal](#modal)
- [RunPod](#runpod)
- [fake](#fake)
- [Logs and exec](#logs-and-exec)
- [What is not observable](#what-is-not-observable)
Expand Down Expand Up @@ -81,8 +82,9 @@ therefore emits without storing, and stores only once `Provision` returns.

`Provision` returns `(id, reserved, error)`. `reserved` means the provider committed
capacity, not merely accepted the request: AWS always does (`CreateFleet` with
`FleetTypeInstant` is synchronous), a fresh Modal sandbox never does (the GPU may
still be queued). Only a reserved instance advances to `Initializing`; an unreserved
`FleetTypeInstant` is synchronous), RunPod always does for the same reason (`POST
/pods` allocates a host before it answers, and a shortage comes back as an error), a
fresh Modal sandbox never does (the GPU may still be queued). Only a reserved instance advances to `Initializing`; an unreserved
id holds at `Provisioning`, which is still exactly true — the id is real and must be
reclaimed, but nothing is allocated. Either way the Pod is now tracked: `reserved`
constrains what the status may claim, not what is owed, and says nothing about
Expand Down Expand Up @@ -234,6 +236,44 @@ only two signals and has to record a third fact itself.
first poll tick. An *adopted* sandbox has been observed, so a `running` one is
known to be reserved. See below for why queued is not reported distinctly.

### RunPod

One RunPod Pod per NodeClaim, read through REST v2's `status`, which (unlike v1's
`desiredStatus`) is the state the Pod has *reached*.

| `status` | `InstanceState` | Pod |
|---|---|---|
| `RUNNING` | `Running` | `Running` / `Ready=True` |
| `PROVISIONING`, `STARTING` | `Pending` | `Pending` / `Initializing` |
| `EXITED`, `TERMINATED` | `Terminated` | `Failed` / `Terminated` |
| `ERROR` | `Failed` | `Failed` |
| anything else | `Pending` | `Pending` / `Initializing` |
| absent from `List` | `Terminated` | `Failed` / `Terminated` |

- **There is no readiness concept beyond `RUNNING`.** RunPod has no probe, so "started"
is the strongest signal available; a container that is up but not yet serving reads
`Running`. Contrast Modal, which has a real probe and latches it.
- **There is no queueing**, as with AWS: `POST /v2/pods` allocates a host before it
answers, and a shortfall is a synchronous error (`ErrNoCapacity`) that drives region
failover. So `Provision` always returns `reserved`.
- **`EXITED` hides crashes.** It covers a clean exit and a crash alike, with no exit
code, so a workload that died reads as `Terminated`, indistinguishable from teardown.
- **OnDemand only.** v2 has no interruptible tier, so nothing is reclaimed and the
default poll cadence applies.
- **Identity rides the Pod name**, not tags: RunPod Pods have none, so a Pod is named
after its claim (`<namespace>-<pod>`) and `List` reads the name back as the claim.
There is no ownership marker, so **the account must be dedicated to Nebula**: a Pod
someone else names like a claim is adopted and later terminated. A Pod whose name
would exceed RunPod's 191-character cap is refused at `Provision` rather than
truncated — two truncated claims would collide onto one Pod.
- The endpoint is **derived, not read back**: `https://<podID>-<port>.proxy.runpod.net`
is known at create time, so it is published from `CreatePod` like Modal's, but with
no token — that proxy is unauthenticated. Every TCP port is exposed as `/http`, so a
raw-TCP service is not reachable through it; UDP and SCTP ports are not exposed.
- **Neither `kubectl logs` nor `kubectl exec` works yet.** v2 streams logs over SSE
(`/v2/pods/{id}/logs`), which a `LogStreamer` could wrap; the only way into a
container is SSH, so exec answers NotFound.

### fake

The in-memory e2e provider reports `InstanceRunning` as soon as an instance is
Expand Down
7 changes: 6 additions & 1 deletion hack/deploy.sh
Original file line number Diff line number Diff line change
Expand Up @@ -102,7 +102,12 @@ PROVIDER_SECRETS=(
# instance role (the preferred path). Region is NON-SECRET (on the manager
# Deployment); the adapter self-configures the rest (GPU AMI + subnets).
"nebula-aws-credentials|AWS_ACCESS_KEY_ID AWS_SECRET_ACCESS_KEY|"
# "nebula-runpod-credentials|RUNPOD_API_KEY|"
# RunPod: one API key, and the only credential it has — there is no ambient identity to
# fall back on as AWS has, so a blank key skips the Secret AND the provider. Mint it at
# https://console.runpod.io/user/settings with read+write on Pods; a read-only key
# registers fine and then fails every create with an auth error, which blocklists the
# whole provider.
"nebula-runpod-credentials|RUNPOD_API_KEY|"
)

# create_provider_secret <secret-name> <required-keys> <optional-keys>
Expand Down
3 changes: 2 additions & 1 deletion internal/controller/nodeclaim_controller_test.go
Original file line number Diff line number Diff line change
Expand Up @@ -52,13 +52,14 @@ type fakeProvider struct {
gpus []string // accelerators MapAccelerator offers; nil = offer any
spot bool // Capabilities().SupportsSpot (placement skips Spot without it)
egress bool // Capabilities().SupportsEgressPolicy (placement skips restricted pools without it)
gpuOnly bool // !Capabilities().SupportsCPUOnly (placement skips CPU-only Pods)
// expandRegions overrides ResolveRegions; nil = pass the declaration through.
expandRegions func([]string) []string
}

func (f *fakeProvider) Name() string { return f.name }
func (f *fakeProvider) Capabilities() provider.Capabilities {
return provider.Capabilities{SupportsSpot: f.spot, SupportsEgressPolicy: f.egress}
return provider.Capabilities{SupportsSpot: f.spot, SupportsEgressPolicy: f.egress, SupportsCPUOnly: !f.gpuOnly}
}
func (f *fakeProvider) Provision(context.Context, *corev1.Pod, provider.ProvisionRequest) (provider.ProvisionResult, error) {
return provider.ProvisionResult{}, nil
Expand Down
5 changes: 2 additions & 3 deletions internal/controller/placement_metrics_test.go
Original file line number Diff line number Diff line change
Expand Up @@ -32,7 +32,6 @@ import (
"github.com/InftyAI/Nebula/pkg/failover"
"github.com/InftyAI/Nebula/pkg/metrics"
"github.com/InftyAI/Nebula/pkg/provider"
awsprovider "github.com/InftyAI/Nebula/pkg/provider/aws"
"github.com/InftyAI/Nebula/pkg/util"
)

Expand Down Expand Up @@ -247,10 +246,10 @@ func TestPlacement_NarrowingToNoRegionFilesNoAvailableRegions(t *testing.T) {
noRegions := skipLabels(provider.ProviderAWS, nebulav1alpha1.CapacityOnDemand, "", metrics.SkipNoAvailableRegions)
before := counterVal(t, metrics.CandidateSkips, noRegions)

pod := gatedPod("r1", "default", "uid-r1", "pool", "")
pod := gatedPod("r1", "default", "uid-r1", "pool", "T4")
pod.Annotations = map[string]string{nebulav1alpha1.RegionsAnnotation: "af"}
pool := poolWith("pool", []nebulav1alpha1.CapacityType{nebulav1alpha1.CapacityOnDemand}, provider.ProviderAWS)
r, _ := newPlacementReconciler(t, []client.Object{pod, pool}, awsprovider.New(nil, nil, nil))
r, _ := newPlacementReconciler(t, []client.Object{pod, pool}, catalogAWS(t))
reconcilePod(t, r, "default", "r1")

if got := counterVal(t, metrics.CandidateSkips, noRegions) - before; got != 1 {
Expand Down
2 changes: 0 additions & 2 deletions internal/controller/pod_placement_controller.go
Original file line number Diff line number Diff line change
Expand Up @@ -152,8 +152,6 @@ func (r *PodPlacementReconciler) Reconcile(ctx context.Context, req ctrl.Request
"pod", pod.Name, "pool", pool.Name, "retryAfter", retryAfter.String())
return ctrl.Result{RequeueAfter: retryAfter}, nil
}
log.Info("no provider in pool can serve the Pod; leaving it gated",
"pod", pod.Name, "pool", pool.Name)
return ctrl.Result{}, nil
}

Expand Down
Loading
Loading