Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
15 changes: 13 additions & 2 deletions .env.example
Original file line number Diff line number Diff line change
Expand Up @@ -45,6 +45,17 @@ MODAL_ENVIRONMENT=
AWS_ACCESS_KEY_ID=
AWS_SECRET_ACCESS_KEY=

# --- RunPod provider -------------------------------------------------------
# One API key, from https://console.runpod.io/user/settings. It needs READ+WRITE on
# Pods: Nebula creates and deletes them, and also creates container-registry-auth
# objects when a workload uses an imagePullSecret. A read-only key registers fine
# and then fails every provision with an auth error, which blocklists the whole
# provider until it is replaced.
#
# Unlike AWS there is no ambient identity to fall back on, so leaving this blank
# skips both the Secret and the provider registration — not fatal.
RUNPOD_API_KEY=

# --- Additional providers (add as adapters land) ---------------------------
# Each provider gets its OWN secret (see hack/deploy.sh PROVIDER_SECRETS), e.g.:
# RUNPOD_API_KEY=
# Each provider gets its OWN secret; see hack/deploy.sh PROVIDER_SECRETS and the
# RunPod block above for the shape.
3 changes: 3 additions & 0 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -62,6 +62,9 @@ metadata:
spec:
providers:
- name: modal # NeoCloud; regions omitted = place anywhere (cheapest)
- name: runpod # NeoCloud, OnDemand only; a region is a geography
regions: # ("us") or one data center ("EU-RO-1")
- us
- name: aws # hyperscaler; "us" expands to every US region
regions:
- us
Expand Down
2 changes: 1 addition & 1 deletion api/v1alpha1/nodepool_types.go
Original file line number Diff line number Diff line change
Expand Up @@ -181,7 +181,7 @@ const (
)

// CapacityType is the purchase model (the outer axis). Each provider maps it to
// its own concept — e.g. RunPod Spot -> interruptible/podRentInterruptable.
// its own concept — e.g. AWS Spot -> a spot-market CreateFleet request.
// +kubebuilder:validation:Enum=Spot;OnDemand
type CapacityType string

Expand Down
19 changes: 8 additions & 11 deletions cmd/main.go
Original file line number Diff line number Diff line change
Expand Up @@ -56,6 +56,7 @@ import (
awsprovider "github.com/InftyAI/Nebula/pkg/provider/aws"
"github.com/InftyAI/Nebula/pkg/provider/fake"
"github.com/InftyAI/Nebula/pkg/provider/modal"
"github.com/InftyAI/Nebula/pkg/provider/runpod"
"github.com/InftyAI/Nebula/pkg/vnode"
// +kubebuilder:scaffold:imports
)
Expand Down Expand Up @@ -619,24 +620,20 @@ func registerProviders(ctx context.Context, c client.Client) {
setupLog.Info("registered provider", "provider", p.Name())
}

// AWS. There is NO region env/flag: the regions this provider may use are declared
// per-pool in the NodePool (ProviderSpec.Regions) and read at call time via the
// region source below, so a pool added at runtime widens the fan-out without a
// restart. One AWS provider spans every such region (per-region clients are built
// lazily). The adapter is otherwise self-configuring: it resolves each region's
// GPU AMI and default-VPC subnets itself, so no launch template or pre-created
// infra is needed. Credentials are secrets and are NEVER read here: the SDK client
// uses the default credential chain (IRSA / instance-role / AWS_ACCESS_KEY_ID
// delivered via a Secret), and one account-global credential authorizes every
// region. Registration only fails (and is a non-fatal skip) if the price catalog
// cannot load — region config can no longer make it fail.
if p, err := awsprovider.NewSDKClient(ctx, awsRegionSource(c)); err != nil {
setupLog.Info("skipping AWS provider registration", "reason", err.Error())
} else {
provider.Register(p)
setupLog.Info("registered provider", "provider", p.Name())
}

if p, err := runpod.NewSDKClient(ctx); err != nil {
setupLog.Info("skipping RunPod provider registration", "reason", err.Error())
} else {
provider.Register(p)
setupLog.Info("registered provider", "provider", p.Name())
}

// The fake provider is an in-memory backend used only by the e2e suite to
// exercise the full control-plane loop without cloud credentials. It ships in
// the binary but registers ONLY when explicitly enabled, so it can never place
Expand Down
1 change: 1 addition & 0 deletions config/catalog/kustomization.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -17,6 +17,7 @@ configMapGenerator:
files:
- modal.csv=../../pkg/provider/catalog/data/modal.csv
- aws.csv=../../pkg/provider/catalog/data/aws.csv
- runpod.csv=../../pkg/provider/catalog/data/runpod.csv

generatorOptions:
# Stable name (no content-hash suffix) so `kubectl edit` and the volume
Expand Down
2 changes: 1 addition & 1 deletion config/crd/bases/nebula.inftyai.com_nodepools.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -90,7 +90,7 @@ spec:
items:
description: |-
CapacityType is the purchase model (the outer axis). Each provider maps it to
its own concept — e.g. RunPod Spot -> interruptible/podRentInterruptable.
its own concept — e.g. AWS Spot -> a spot-market CreateFleet request.
enum:
- Spot
- OnDemand
Expand Down
12 changes: 8 additions & 4 deletions config/manager/manager.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -127,10 +127,14 @@ spec:
- secretRef:
name: nebula-aws-credentials
optional: true
# Add one secretRef per provider as adapters land, e.g.:
# - secretRef:
# name: nebula-runpod-credentials
# optional: true
# RunPod: a single API key (RUNPOD_API_KEY). Unlike AWS there is no ambient
# identity to fall back on, so an absent Secret means the provider is simply
# skipped at registration. Regions come from the NodePool, so this is the only
# RunPod config here.
- secretRef:
name: nebula-runpod-credentials
optional: true
# Add one secretRef per provider as adapters land, following the pattern above.
ports:
# The kubelet API the API server dials for `kubectl logs` (10250, like a real
# kubelet). Declaring it is documentation and NetworkPolicy surface; the
Expand Down
12 changes: 11 additions & 1 deletion config/samples/nodepool.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -19,7 +19,12 @@ spec:
- us
- eu
- ap-melbourne
# - name: runpod
- name: runpod
# A region is a geography (us, eu, ...) or one exact data center (EU-RO-1). Each
# entry is one failover candidate, so a shortage in one blocklists only that one.
regions:
- us
- EU-RO-1
# Outer axis: try OnDemand on every provider first, fall back to Spot.
capacityTypes:
- OnDemand
Expand All @@ -32,6 +37,11 @@ spec:
# sandbox stays reachable through its connect URL and token under every mode — it just
# cannot call out.
#
# Setting ANY mode other than Open narrows this pool to the providers that can enforce
# it: placement skips a provider whose Capabilities report SupportsEgressPolicy=false
# (RunPod, which exposes no outbound knob at all), rather than provisioning something
# with open internet access under a policy that says otherwise.
#
# Blocked permits nothing:
# egress:
# mode: Blocked
Expand Down
1 change: 1 addition & 0 deletions docs/deploy.md
Original file line number Diff line number Diff line change
Expand Up @@ -118,6 +118,7 @@ itself at startup (see [Webhook TLS](#webhook-tls-no-cert-manager)).
| `MODAL_ENVIRONMENT` | Modal | no | Modal Environment to create sandboxes in. Blank omits the key and the SDK uses the token profile's default. See [Modal Environments](#modal-environments). |
| `AWS_ACCESS_KEY_ID` | AWS | dev only | Prefer IRSA / instance role in production and leave blank — the SDK's default credential chain finds the role. Set only for local/dev. |
| `AWS_SECRET_ACCESS_KEY` | AWS | dev only | Pairs with `AWS_ACCESS_KEY_ID`; both required together or both blank. |
| `RUNPOD_API_KEY` | RunPod | yes | From [console.runpod.io/user/settings](https://console.runpod.io/user/settings). Needs **read+write on Pods**: a read-only key registers fine and then fails every provision with an auth error, which blocklists the whole provider. Unlike AWS there is no ambient identity, so blank skips RunPod entirely. |

Non-secret config, passed as `make` variables:

Expand Down
42 changes: 40 additions & 2 deletions docs/status.md
Original file line number Diff line number Diff line change
Expand Up @@ -25,6 +25,7 @@ enters the system.
- [Provider mappings](#provider-mappings)
- [AWS](#aws)
- [Modal](#modal)
- [RunPod](#runpod)
- [fake](#fake)
- [Logs and exec](#logs-and-exec)
- [What is not observable](#what-is-not-observable)
Expand Down Expand Up @@ -81,8 +82,9 @@ therefore emits without storing, and stores only once `Provision` returns.

`Provision` returns `(id, reserved, error)`. `reserved` means the provider committed
capacity, not merely accepted the request: AWS always does (`CreateFleet` with
`FleetTypeInstant` is synchronous), a fresh Modal sandbox never does (the GPU may
still be queued). Only a reserved instance advances to `Initializing`; an unreserved
`FleetTypeInstant` is synchronous), RunPod always does for the same reason (`POST
/pods` allocates a host before it answers, and a shortage comes back as an error), a
fresh Modal sandbox never does (the GPU may still be queued). Only a reserved instance advances to `Initializing`; an unreserved
id holds at `Provisioning`, which is still exactly true — the id is real and must be
reclaimed, but nothing is allocated. Either way the Pod is now tracked: `reserved`
constrains what the status may claim, not what is owed, and says nothing about
Expand Down Expand Up @@ -234,6 +236,42 @@ only two signals and has to record a third fact itself.
first poll tick. An *adopted* sandbox has been observed, so a `running` one is
known to be reserved. See below for why queued is not reported distinctly.

### RunPod

One RunPod Pod per NodeClaim, read through REST v2's `status`, which (unlike v1's
`desiredStatus`) is the state the Pod has *reached*.

| `status` | `InstanceState` | Pod |
|---|---|---|
| `RUNNING` | `Running` | `Running` / `Ready=True` |
| `PROVISIONING`, `STARTING` | `Pending` | `Pending` / `Initializing` |
| `EXITED`, `TERMINATED` | `Terminated` | `Failed` / `Terminated` |
| `ERROR` | `Failed` | `Failed` |
| anything else | `Pending` | `Pending` / `Initializing` |
| absent from `List` | `Terminated` | `Failed` / `Terminated` |

- **There is no readiness concept beyond `RUNNING`.** RunPod has no probe, so "started"
is the strongest signal available; a container that is up but not yet serving reads
`Running`. Contrast Modal, which has a real probe and latches it.
- **There is no queueing**, as with AWS: `POST /v2/pods` allocates a host before it
answers, and a shortfall is a synchronous error (`ErrNoCapacity`) that drives region
failover. So `Provision` always returns `reserved`.
- **`EXITED` hides crashes.** It covers a clean exit and a crash alike, with no exit
code, so a workload that died reads as `Terminated`, indistinguishable from teardown.
- **OnDemand only.** v2 has no interruptible tier, so nothing is reclaimed and the
default poll cadence applies.
- **Identity rides the Pod name**, not tags: RunPod Pods have none, so `List` filters
on the `nebula-` prefix and the claim name is recovered by stripping it. A Pod whose
name would exceed RunPod's 191-character cap is refused at `Provision` rather than
truncated — two truncated claims would collide onto one Pod.
- The endpoint is **derived, not read back**: `https://<podID>-<port>.proxy.runpod.net`
is known at create time, so it is published from `CreatePod` like Modal's, but with
no token — that proxy is unauthenticated. A Pod with a public IP and an assigned
`/tcp` port mapping reports that direct address instead, once the poll loop sees it.
- **Neither `kubectl logs` nor `kubectl exec` works yet.** v2 streams logs over SSE
(`/v2/pods/{id}/logs`), which a `LogStreamer` could wrap; the only way into a
container is SSH, so exec answers NotFound.

### fake

The in-memory e2e provider reports `InstanceRunning` as soon as an instance is
Expand Down
7 changes: 6 additions & 1 deletion hack/deploy.sh
Original file line number Diff line number Diff line change
Expand Up @@ -102,7 +102,12 @@ PROVIDER_SECRETS=(
# instance role (the preferred path). Region is NON-SECRET (on the manager
# Deployment); the adapter self-configures the rest (GPU AMI + subnets).
"nebula-aws-credentials|AWS_ACCESS_KEY_ID AWS_SECRET_ACCESS_KEY|"
# "nebula-runpod-credentials|RUNPOD_API_KEY|"
# RunPod: one API key, and the only credential it has — there is no ambient identity to
# fall back on as AWS has, so a blank key skips the Secret AND the provider. Mint it at
# https://console.runpod.io/user/settings with read+write on Pods; a read-only key
# registers fine and then fails every create with an auth error, which blocklists the
# whole provider.
"nebula-runpod-credentials|RUNPOD_API_KEY|"
)

# create_provider_secret <secret-name> <required-keys> <optional-keys>
Expand Down
57 changes: 57 additions & 0 deletions pkg/provider/catalog/data/runpod.csv
Original file line number Diff line number Diff line change
@@ -0,0 +1,57 @@
# RunPod price/availability catalog — community-maintained.
#
# RunPod exposes GPU prices through its API, but only per GPU TYPE and without the
# canonical-name mapping Nebula needs, so these are SECURE-cloud list prices
# transcribed from https://runpod.io/pricing by hand. Treat them as a starting
# point, not a billing source, and refresh via `make update-catalog`
# (see hack/refresh.go).
#
# Prices are per GPU-HOUR, as Modal's are — not per instance-hour like aws.csv.
# RunPod bills per GPU, and the adapter passes the count as a runtime parameter.
#
# Only SECURE cloud is priced here. RunPod's COMMUNITY cloud rents the same GPUs
# from peer hosts for roughly 30-50% less, but this file has no cloud-type column
# to say which tier a row prices, so listing both would make the number the
# optimizer reads a coin flip. The adapter pins cloudType=SECURE to match; adding
# COMMUNITY means adding that column first (see pkg/provider/runpod's package doc).
#
# Columns (shared header across all provider CSVs; unused cells left blank):
# accelerator_type canonical Nebula accelerator type, matched case-insensitively
# against the nebula.inftyai.com/accelerator-type label
# accelerator_id ALWAYS SET here, unlike modal.csv: RunPod's ids are marketing
# strings ("NVIDIA H100 80GB HBM3") that share nothing with the
# canonical names, so every row carries its own translation.
#
# SEVERAL rows may share one accelerator_type, and the order
# matters: MapAccelerator returns them in file order, so the
# FIRST is the primary and the rest are interchangeable
# alternates. A REST v2 create takes ONE gpu id, so only the
# primary is launched; an alternate matters only once the rows
# above it are flipped to available=false. Put the variant with
# the best interconnect first (SXM before NVL before PCIe).
# gpu_count BLANK. RunPod takes the GPU count as a request parameter, so it
# is not a lookup dimension (contrast aws.csv, where the count is
# baked into the instance type). A blank row matches any count.
# capacity_type OnDemand only. REST v2 has no interruptible tier, so a Spot row
# would place Pods the adapter cannot launch.
# price_per_hour approximate USD per GPU-hour on SECURE cloud
# available whether Nebula may schedule onto it. Flipping a row to false
# removes it from placement everywhere without touching Go — the
# escape hatch for a GPU type RunPod has stopped offering.
# region BLANK. RunPod's prices are not partitioned by data center, so
# one row prices every region. Region is still a real placement
# axis for this provider (a pool's regions become dataCenterIds);
# it is just not a pricing one.
# updated documentation only, ignored by the parser
accelerator_type,accelerator_id,gpu_count,capacity_type,price_per_hour,available,region,updated
L4,NVIDIA L4,,OnDemand,0.43,true,,2026-08-29
A40,NVIDIA A40,,OnDemand,0.40,true,,2026-08-29
RTX4090,NVIDIA GeForce RTX 4090,,OnDemand,0.69,true,,2026-08-29
L40S,NVIDIA L40S,,OnDemand,0.86,true,,2026-08-29
A100-40GB,NVIDIA A100-PCIE-40GB,,OnDemand,1.19,true,,2026-08-29
A100-80GB,NVIDIA A100-SXM4-80GB,,OnDemand,1.74,true,,2026-08-29
A100-80GB,NVIDIA A100 80GB PCIe,,OnDemand,1.64,true,,2026-08-29
H100,NVIDIA H100 80GB HBM3,,OnDemand,2.99,true,,2026-08-29
H100,NVIDIA H100 NVL,,OnDemand,2.79,true,,2026-08-29
H100,NVIDIA H100 PCIe,,OnDemand,2.39,true,,2026-08-29
H200,NVIDIA H200,,OnDemand,3.99,true,,2026-08-29
25 changes: 8 additions & 17 deletions pkg/provider/errors.go
Original file line number Diff line number Diff line change
Expand Up @@ -22,6 +22,7 @@ import (
"strings"

nebulav1alpha1 "github.com/InftyAI/Nebula/api/v1alpha1"
"github.com/InftyAI/Nebula/pkg/util"
)

// Provision failure categories, shared by every adapter. The CATEGORIES are
Expand Down Expand Up @@ -169,7 +170,7 @@ func categorize(err error) failureCategory {
// — a manager shutdown, a leader handoff — and the provider may well have accepted the
// request. Nothing about the candidate was learned, so blocklisting would punish it for
// our own exit. Only the block scope; the caller still fails the Pod.
if errors.Is(err, context.Canceled) || containsAny(msg, "context canceled") {
if errors.Is(err, context.Canceled) || util.ContainsAny(msg, "context canceled") {
return catUnattributable
}

Expand All @@ -189,7 +190,7 @@ func categorize(err error) failureCategory {
// - "image build for": the Modal SDK's remote build verdict, the only thing it says when
// the build itself reached a decision. An API error from the same call does NOT carry
// it, which is what keeps an expired workspace token classifiable as auth below.
if containsAny(msg, "image pull credential", "image build for") {
if util.ContainsAny(msg, "image pull credential", "image build for") {
return catRequest
}

Expand All @@ -198,39 +199,29 @@ func categorize(err error) failureCategory {
// while every replacement Pod retried the same broken provider.
if strings.Contains(msg, "rpc error") {
switch {
case containsAny(msg, "code = unauthenticated", "code = permissiondenied"):
case util.ContainsAny(msg, "code = unauthenticated", "code = permissiondenied"):
return catAuth
case strings.Contains(msg, "code = resourceexhausted"):
return catCapacity
}
}

if containsAny(msg,
if util.ContainsAny(msg,
"rpc error", "connection refused", "connection reset", "broken pipe",
"no such host", "i/o timeout", "eof", "tls handshake",
"service unavailable", "bad gateway", "gateway timeout", "internal server error") {
return catUnattributable
}

switch {
case containsAny(msg, "unauthorized", "forbidden", "authentication",
case util.ContainsAny(msg, "unauthorized", "forbidden", "authentication",
"unauthenticated", "invalid token", "api key"):
return catAuth
case containsAny(msg, "quota", "limit exceeded", "rate limit"):
case util.ContainsAny(msg, "quota", "limit exceeded", "rate limit"):
return catCapacity
case containsAny(msg, "no capacity", "capacity", "unavailable", "out of", "no gpu"):
case util.ContainsAny(msg, "no capacity", "capacity", "unavailable", "out of", "no gpu"):
return catCapacity
default:
return catUnattributable
}
}

// containsAny reports whether s contains any of subs.
func containsAny(s string, subs ...string) bool {
for _, sub := range subs {
if strings.Contains(s, sub) {
return true
}
}
return false
}
Loading
Loading