Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions .gitignore
Original file line number Diff line number Diff line change
Expand Up @@ -50,3 +50,4 @@ __pycache__
.devcontainer
bin
.claude
*.html
4 changes: 2 additions & 2 deletions pkg/provider/catalog/base.go
Original file line number Diff line number Diff line change
Expand Up @@ -132,8 +132,8 @@ func (b Base) MapAccelerator(canonical string, count int32) (providerAccelerator
// adds them.
//
// Rows match as MapAccelerator matches them, plus capacity type: the one dimension
// MapAccelerator can ignore and pricing cannot (AWS p5.48xlarge is $34.412 Spot against
// $98.320 OnDemand). Among interchangeable alternates the FIRST row wins, so the price
// MapAccelerator can ignore and pricing cannot (AWS p5.48xlarge is $20.839 Spot against
// $55.040 OnDemand). Among interchangeable alternates the FIRST row wins, so the price
// describes the id a launch actually tries first.
func (b Base) PricePerHour(req provider.PriceRequest) (float64, error) {
// A CPU-only request. No row can match an empty type, so this is the same ErrNoPrice the
Expand Down
4 changes: 2 additions & 2 deletions pkg/provider/catalog/catalog_test.go
Original file line number Diff line number Diff line change
Expand Up @@ -403,8 +403,8 @@ func TestBasePricePerHour_EmbeddedCatalog(t *testing.T) {
if err != nil {
t.Fatalf("aws H100 x8: %v", err)
}
if got != 98.320 {
t.Fatalf("aws H100 x8 = %v, want 98.320 (whole-instance rate, unscaled)", got)
if got != 55.040 {
t.Fatalf("aws H100 x8 = %v, want 55.040 (whole-instance rate, unscaled)", got)
}
}

Expand Down
48 changes: 24 additions & 24 deletions pkg/provider/catalog/data/aws.csv
Original file line number Diff line number Diff line change
Expand Up @@ -32,27 +32,27 @@
# region BLANK — stamped per configured region by the live probe
# updated YYYY-MM-DD the row was last verified
accelerator_type,accelerator_id,gpu_count,capacity_type,price_per_hour,available,region,updated
T4,g4dn.xlarge,1,OnDemand,0.526,true,,2026-07-27
T4,g4dn.xlarge,1,Spot,0.158,true,,2026-07-27
T4,g4dn.12xlarge,4,OnDemand,3.912,true,,2026-07-27
T4,g4dn.12xlarge,4,Spot,1.174,true,,2026-07-27
T4,g4dn.metal,8,OnDemand,7.824,true,,2026-07-27
T4,g4dn.metal,8,Spot,2.347,true,,2026-07-27
A10G,g5.xlarge,1,OnDemand,1.006,true,,2026-07-27
A10G,g5.xlarge,1,Spot,0.352,true,,2026-07-27
A10G,g5.12xlarge,4,OnDemand,5.672,true,,2026-07-27
A10G,g5.12xlarge,4,Spot,1.985,true,,2026-07-27
A10G,g5.48xlarge,8,OnDemand,16.288,true,,2026-07-27
A10G,g5.48xlarge,8,Spot,5.700,true,,2026-07-27
L4,g6.xlarge,1,OnDemand,0.805,true,,2026-07-27
L4,g6.xlarge,1,Spot,0.282,true,,2026-07-27
L4,g6.12xlarge,4,OnDemand,4.602,true,,2026-07-27
L4,g6.12xlarge,4,Spot,1.610,true,,2026-07-27
L4,g6.48xlarge,8,OnDemand,13.350,true,,2026-07-27
L4,g6.48xlarge,8,Spot,4.672,true,,2026-07-27
A100-40GB,p4d.24xlarge,8,OnDemand,32.773,true,,2026-07-29
A100-40GB,p4d.24xlarge,8,Spot,11.470,true,,2026-07-29
A100-80GB,p4de.24xlarge,8,OnDemand,40.966,true,,2026-07-27
A100-80GB,p4de.24xlarge,8,Spot,14.338,true,,2026-07-27
H100,p5.48xlarge,8,OnDemand,98.320,true,,2026-07-27
H100,p5.48xlarge,8,Spot,34.412,true,,2026-07-27
T4,g4dn.xlarge,1,OnDemand,0.526,true,,2026-09-30
T4,g4dn.xlarge,1,Spot,0.274,true,,2026-09-30
T4,g4dn.12xlarge,4,OnDemand,3.912,true,,2026-09-30
T4,g4dn.12xlarge,4,Spot,1.561,true,,2026-09-30
T4,g4dn.metal,8,OnDemand,7.824,true,,2026-09-30
T4,g4dn.metal,8,Spot,5.683,true,,2026-09-30
A10G,g5.xlarge,1,OnDemand,1.006,true,,2026-09-30
A10G,g5.xlarge,1,Spot,0.466,true,,2026-09-30
A10G,g5.12xlarge,4,OnDemand,5.672,true,,2026-09-30
A10G,g5.12xlarge,4,Spot,4.576,true,,2026-09-30
A10G,g5.48xlarge,8,OnDemand,16.288,true,,2026-09-30
A10G,g5.48xlarge,8,Spot,8.402,true,,2026-09-30
L4,g6.xlarge,1,OnDemand,0.805,true,,2026-09-30
L4,g6.xlarge,1,Spot,0.601,true,,2026-09-30
L4,g6.12xlarge,4,OnDemand,4.602,true,,2026-09-30
L4,g6.12xlarge,4,Spot,3.300,true,,2026-09-30
L4,g6.48xlarge,8,OnDemand,13.350,true,,2026-09-30
L4,g6.48xlarge,8,Spot,8.292,true,,2026-09-30
A100-40GB,p4d.24xlarge,8,OnDemand,21.958,true,,2026-09-30
A100-40GB,p4d.24xlarge,8,Spot,17.861,true,,2026-09-30
A100-80GB,p4de.24xlarge,8,OnDemand,27.447,true,,2026-09-30
A100-80GB,p4de.24xlarge,8,Spot,21.514,true,,2026-09-30
H100,p5.48xlarge,8,OnDemand,55.040,true,,2026-09-30
H100,p5.48xlarge,8,Spot,20.839,true,,2026-09-30
5 changes: 2 additions & 3 deletions pkg/provider/catalog/data/pricing.go
Original file line number Diff line number Diff line change
Expand Up @@ -21,7 +21,7 @@ package data

// Modal meters CPU and memory SEPARATELY from the accelerator, so a sandbox's hourly
// cost is the GPU price PLUS these. Not universal: AWS bundles both into the instance
// price (p5.48xlarge's $98.320/hr already covers its vCPU and RAM), so a provider with
// price (p5.48xlarge's $55.040/hr already covers its vCPU and RAM), so a provider with
// no rates here is one whose CSV price is already all-in.
//
// Modal publishes these PER SECOND, so the literal stays exactly as printed on the price
Expand Down Expand Up @@ -53,8 +53,7 @@ const mibPerGiB = 1024
// (see modal.SandboxSpec) — so no conversion happens at the call site, which is where a
// factor-of-1024 slip would hide.
//
// Reservation, not usage: a sandbox bursting above its request toward CPULimit may bill
// above these.
// A floor, since Modal bills usage above the reservation (see util.PodReservation).
func ModalCPUCostPerHour(cpuCores float64) float64 {
return cpuCores * ModalCPUPricePerCoreHour
}
Expand Down
107 changes: 24 additions & 83 deletions pkg/provider/modal/modal.go
Original file line number Diff line number Diff line change
Expand Up @@ -55,7 +55,6 @@ import (
"time"

corev1 "k8s.io/api/core/v1"
"k8s.io/apimachinery/pkg/api/resource"

nebulav1alpha1 "github.com/InftyAI/Nebula/api/v1alpha1"
"github.com/InftyAI/Nebula/pkg/provider"
Expand Down Expand Up @@ -156,21 +155,11 @@ type SandboxSpec struct {
GPU string
// GPUCount is how many accelerators to attach (0 for CPU-only).
GPUCount int32
// CPU is the requested cores (fractional, physical), from the Pod's first
// container resource request. Zero lets Modal apply its own default.
CPU float64
// MemoryMiB is the requested memory in MiB, from the Pod's request. Zero lets
// Modal apply its own default.
MemoryMiB int
// CPULimit and MemoryLimitMiB are the HARD caps, from the Pod's limits only —
// never from its requests, unlike CPU/MemoryMiB above, which fall back to limits
// when no request is given.
//
// Zero means no cap, which is also what a Pod that declares no limit means, so the
// two vocabularies line up on everything but one case: a positive limit smaller than
// Modal's unit must not truncate into that sentinel (see limitMiB). Without these a Pod's limits
// reached Modal as nothing at all: a limits-only Pod became a RESERVATION of that
// size with an unbounded ceiling — the inverse of what it asked for, and billable.
// CPU (physical cores) and MemoryMiB are the reservation, CPULimit and MemoryLimitMiB the
// hard cap, all from util.PodResources. Zero is Modal's default on a request and no cap on
// a limit. The claim is priced at the request (see util.PodReservation).
CPU float64
MemoryMiB int
CPULimit float64
MemoryLimitMiB int
// Ports are the container ports to expose, from the Pod's containerPorts. They
Expand Down Expand Up @@ -412,7 +401,8 @@ func (p *Provider) ResolveRegions(declared, narrowTo []string) []string {
// would be read as free. A GPU sandbox in that state still prices, understating by those
// same defaults, which is immaterial beside the accelerator.
func (p *Provider) PricePerHour(req provider.PriceRequest) (float64, error) {
metered := data.ModalCPUCostPerHour(req.CPUCores) + data.ModalMemoryCostPerHour(req.MemoryMiB)
metered := data.ModalCPUCostPerHour(physicalCores(req.CPUCores)) +
data.ModalMemoryCostPerHour(req.MemoryMiB)

if req.AcceleratorType == "" {
if req.CPUCores <= 0 || req.MemoryMiB <= 0 {
Expand Down Expand Up @@ -624,6 +614,7 @@ func (p *Provider) sandboxSpecFromPod(pod *corev1.Pod, req provider.ProvisionReq
tags[ProbeTagKey] = probeTagValue
}

requests, limits := util.PodResources(pod)
spec := SandboxSpec{
Image: c.Image,
Command: slices.Clone(c.Command),
Expand All @@ -634,10 +625,10 @@ func (p *Provider) sandboxSpecFromPod(pod *corev1.Pod, req provider.ProvisionReq
// here: it holds references this adapter has no cluster access to follow. See
// provider.ProvisionRequest.Env.
Env: req.Env,
CPU: cpuCores(&c),
MemoryMiB: memoryMiB(&c),
CPULimit: cpuLimitCores(&c),
MemoryLimitMiB: memoryLimitMiB(&c),
CPU: physicalCores(requests.CPU),
MemoryMiB: requests.MemoryMiB,
CPULimit: physicalCores(limits.CPU),
MemoryLimitMiB: limits.MemoryMiB,
Ports: containerPorts(&c),
// An empty request region stays an empty slice, not a one-element [""]: that
// is the unconstrained case (no region declared on the pool), and it must
Expand Down Expand Up @@ -724,71 +715,21 @@ func checkRegistryAuth(a *provider.RegistryAuth) error {
}
}

// cpuCores reads the container's CPU request as fractional physical cores (Modal's
// unit). It prefers requests, falling back to limits, and returns 0 (→ Modal
// default) when neither is set.
func cpuCores(c *corev1.Container) float64 { return cores(resourceQty(c, corev1.ResourceCPU)) }

// memoryMiB reads the container's memory request in MiB (Modal's unit), preferring
// requests over limits. Returns 0 (→ Modal default) when neither is set.
func memoryMiB(c *corev1.Container) int { return mib(resourceQty(c, corev1.ResourceMemory)) }

// cpuLimitCores and memoryLimitMiB read the LIMITS, with no fallback to the request: a
// request is a floor, and reusing it as a ceiling would cap a burstable Pod that never
// asked to be capped. Zero (no limit declared) reaches Modal as "no limit", matching
// Kubernetes. The request/limit asymmetry is entirely in which lookup they use.
func cpuLimitCores(c *corev1.Container) float64 { return cores(limitQty(c, corev1.ResourceCPU)) }
func memoryLimitMiB(c *corev1.Container) int { return limitMiB(limitQty(c, corev1.ResourceMemory)) }

// cores converts a CPU quantity to Modal's unit, fractional physical cores. MilliValue
// is cores*1000. A nil quantity (unset) is 0, which lets Modal apply its own default.
func cores(q *resource.Quantity) float64 {
if q == nil {
return 0
}
return float64(q.MilliValue()) / 1000.0
}
const (
// vCPUsPerModalCore: a Kubernetes CPU is a vCPU, but Modal requests and bills physical
// cores of 2 vCPU each (modal.com/pricing).
vCPUsPerModalCore = 2
// minModalCores is Modal's per-container minimum.
minModalCores = 0.125
)

// mib converts a memory quantity to Modal's unit, MiB. Nil is 0, as in cores.
func mib(q *resource.Quantity) int {
if q == nil {
// physicalCores converts vCPUs to Modal physical cores, floored at minModalCores; zero
// stays zero (Modal's default). Shared with PricePerHour so price matches provisioning.
func physicalCores(vCPUs float64) float64 {
if vCPUs <= 0 {
return 0
}
const miB = 1024 * 1024
return int(q.Value() / miB)
}

// limitMiB is mib for a LIMIT, where 0 does not mean "unset" but "no cap". A positive
// quantity below 1 MiB truncates to 0 there, so the plain conversion would hand an
// UNBOUNDED sandbox to the one Pod that asked for the tightest ceiling — the inverse
// of its declaration. Any positive limit therefore floors at 1 MiB, the smallest cap
// Modal's unit can express. Modal may then refuse it as below its own minimum, which is
// the honest answer for a limit it cannot honour, and is not silently unlimited.
func limitMiB(q *resource.Quantity) int {
if m := mib(q); m != 0 || q == nil || q.Sign() <= 0 {
return m
}
return 1
}

// resourceQty returns the container's request for name, falling back to its limit,
// or nil when neither is present.
func resourceQty(c *corev1.Container, name corev1.ResourceName) *resource.Quantity {
if q, ok := c.Resources.Requests[name]; ok {
return &q
}
if q, ok := c.Resources.Limits[name]; ok {
return &q
}
return nil
}

// limitQty returns the container's limit for name, or nil when it has none.
func limitQty(c *corev1.Container, name corev1.ResourceName) *resource.Quantity {
if q, ok := c.Resources.Limits[name]; ok {
return &q
}
return nil
return max(vCPUs/vCPUsPerModalCore, minModalCores)
}

// containerPorts collects the container's declared ports, which is what tells Modal
Expand Down
60 changes: 38 additions & 22 deletions pkg/provider/modal/modal_test.go
Original file line number Diff line number Diff line change
Expand Up @@ -320,8 +320,8 @@ func TestProvision_MapsResourcesPortsAndTimeout(t *testing.T) {
if _, err := p.Provision(context.Background(), pod, provider.ProvisionRequest{ClaimName: "claim-res"}); err != nil {
t.Fatalf("Provision: %v", err)
}
if f.lastSpec.CPU != 2.5 {
t.Fatalf("CPU = %v, want 2.5", f.lastSpec.CPU)
if f.lastSpec.CPU != 1.25 {
t.Fatalf("CPU = %v, want 1.25 physical cores for 2500m (2.5 vCPU)", f.lastSpec.CPU)
}
if f.lastSpec.MemoryMiB != 4096 {
t.Fatalf("MemoryMiB = %d, want 4096", f.lastSpec.MemoryMiB)
Expand All @@ -334,6 +334,22 @@ func TestProvision_MapsResourcesPortsAndTimeout(t *testing.T) {
}
}

// Getting this wrong doubles both what the Pod gets and what it is billed.
func TestPhysicalCores(t *testing.T) {
for vCPUs, want := range map[float64]float64{
20: 10,
4: 2,
0.5: 0.25,
0.1: minModalCores,
0: 0, // Modal's default, never floored
-1: 0,
} {
if got := physicalCores(vCPUs); got != want {
t.Errorf("physicalCores(%v) = %v, want %v", vCPUs, got, want)
}
}
}

func TestProvision_MapsResourceLimits(t *testing.T) {
cases := []struct {
name string
Expand All @@ -342,16 +358,16 @@ func TestProvision_MapsResourceLimits(t *testing.T) {
wantMemMiB, wantMemLimMiB int
}{
{
name: "limits only: limit is the ceiling AND the request falls back to it",
name: "limits only: the request falls back to the limit",
limits: corev1.ResourceList{
corev1.ResourceCPU: resource.MustParse("2"),
corev1.ResourceMemory: resource.MustParse("8Gi"),
},
wantCPU: 2, wantCPULimit: 2,
wantCPU: 1, wantCPULimit: 1,
wantMemMiB: 8192, wantMemLimMiB: 8192,
},
{
name: "both: burstable, request below the ceiling",
name: "burstable: request below the ceiling",
requests: corev1.ResourceList{
corev1.ResourceCPU: resource.MustParse("500m"),
corev1.ResourceMemory: resource.MustParse("1Gi"),
Expand All @@ -360,27 +376,28 @@ func TestProvision_MapsResourceLimits(t *testing.T) {
corev1.ResourceCPU: resource.MustParse("4"),
corev1.ResourceMemory: resource.MustParse("16Gi"),
},
wantCPU: 0.5, wantCPULimit: 4,
wantCPU: 0.25, wantCPULimit: 2,
wantMemMiB: 1024, wantMemLimMiB: 16384,
},
{
name: "requests only: uncapped, as in Kubernetes",
requests: corev1.ResourceList{
corev1.ResourceCPU: resource.MustParse("500m"),
corev1.ResourceMemory: resource.MustParse("1Gi"),
},
wantCPU: 0.25, wantCPULimit: 0,
wantMemMiB: 1024, wantMemLimMiB: 0,
},
{
name: "neither: Modal applies its own defaults, uncapped",
wantCPU: 0, wantCPULimit: 0,
wantMemMiB: 0, wantMemLimMiB: 0,
},
{
// A ceiling below Modal's unit must not truncate into the zero that means
// "no cap" on the limit fields: it would leave the Pod asking for the
// TIGHTEST ceiling running unbounded. The matching request is a different
// question — zero there means "Modal's default", so falling back to 0 is
// correct and the asymmetry is deliberate.
name: "sub-MiB ceiling floors at 1 MiB instead of becoming uncapped",
limits: corev1.ResourceList{
corev1.ResourceCPU: resource.MustParse("1m"),
corev1.ResourceMemory: resource.MustParse("500Ki"),
},
wantCPU: 0.001, wantCPULimit: 0.001,
wantMemMiB: 0, wantMemLimMiB: 1,
// The limit floors with the request, or the SDK rejects limit < request.
name: "cpu below Modal's minimum floors on both sides",
limits: corev1.ResourceList{corev1.ResourceCPU: resource.MustParse("1m")},
wantCPU: minModalCores, wantCPULimit: minModalCores,
},
}

Expand Down Expand Up @@ -1958,10 +1975,9 @@ func TestPricePerHour_AddsCPUAndMemory(t *testing.T) {
p := newTestProvider(&fakeClient{})
od := nebulav1alpha1.CapacityOnDemand

// Modal's published per-second sandbox rates: 4 cores = $0.567648/hr, 8 GiB = $0.192096/hr.
// Transcribed again here rather than imported from data, so a slip in those constants fails
// this test instead of being multiplied through it.
const cpuAndMem = 4*0.00003942*60*60 + 8*0.00000667*60*60
// 4 vCPUs (2 physical cores) + 8 GiB at Modal's sandbox rates. Transcribed, not imported
// from data, so a slip in those constants fails here.
const cpuAndMem = 2*0.00003942*60*60 + 8*0.00000667*60*60

cases := map[string]struct {
req provider.PriceRequest
Expand Down
7 changes: 3 additions & 4 deletions pkg/provider/pricing.go
Original file line number Diff line number Diff line change
Expand Up @@ -49,11 +49,10 @@ type PriceRequest struct {
// matched row's GPUCount rather than hardcoded per provider — see Offering.GPUCount.
Count int32
// CapacityType selects between a row's Spot and OnDemand prices, which differ
// sharply (AWS p5.48xlarge: $34.412 Spot vs $98.320 OnDemand).
// sharply (AWS p5.48xlarge: $20.839 Spot vs $55.040 OnDemand).
CapacityType nebulav1alpha1.CapacityType
// CPUCores and MemoryMiB are the workload's RESERVATION, priced only by providers
// that meter them separately from the accelerator. Ignored by a provider whose
// instance price is all-in.
// CPUCores (vCPUs) and MemoryMiB are the workload's priced size (see util.PodReservation),
// priced only by providers that meter them apart from the accelerator.
CPUCores float64
MemoryMiB int
}
Expand Down
Loading
Loading