Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
17 changes: 3 additions & 14 deletions CubeMaster/conf.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -82,20 +82,9 @@ scheduler:
# during scheduling. Defaults to false: account for the Redis allocation
# records when computing schedulable capacity.
ignore_redis_allocation: false
# Global CPU/Mem overcommit ratio applied to the node-reported quota capacity.
# Defaults to CPU=3, Mem=2.
overcommit_ratio:
cpu_ratio: 3.0
mem_ratio: 2.0
# Per-instance-type overcommit ratio (takes precedence over the global
# overcommit_ratio).
# overcommit_ratio_conf:
# cubebox:
# cpu_ratio: 6.0
# mem_ratio: 4.0
# cubebox_gpu:
# cpu_ratio: 1.0
# mem_ratio: 1.0
# Deprecated: overcommit_ratio / overcommit_ratio_conf are no longer used.

Copy link
Copy Markdown

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Worth calling out that this isn't a no-op deprecation: schedulable capacity now equals the raw Cubelet-reported quota, whereas the default before was quota × 3 (CPU) / × 2 (mem). Cubelet still returns an explicitly configured host.quota.mcpu_limit/mem_limit verbatim (its overcommit factors only apply to auto-derived defaults), so a node with mcpu_limit: 24000 that used to offer 72000 milli-CPU now offers 24000 after upgrade. Suggest adding a short migration note in the docs telling operators to fold their previous ratio into the explicit host quota, or expect a one-time capacity drop (abrupt during rolling master/Cubelet upgrades).

# Leftover keys here are ignored; CubeMaster logs a warning at startup
# if they are present.
filter:
enable_filters:
- "cpu"
Expand Down
110 changes: 11 additions & 99 deletions CubeMaster/pkg/base/config/config.go
Original file line number Diff line number Diff line change
Expand Up @@ -251,12 +251,11 @@ type SchedulerConf struct {
// 0). A pointer is used so an unset value can default to false while still
// allowing operators to explicitly enable it. Defaults to false.
IgnoreRedisAllocation *bool `yaml:"ignore_redis_allocation"`
// OvercommitRatio is the global CPU/Mem overcommit ratio applied to the
// node-reported quota during scheduling. Defaults to CPU=3, Mem=2.
OvercommitRatio *OvercommitRatioConf `yaml:"overcommit_ratio"`
// OvercommitRatioByType overrides OvercommitRatio for specific instance
// types and takes precedence over the global ratio.
OvercommitRatioByType map[string]OvercommitRatioConf `yaml:"overcommit_ratio_conf"`
// Deprecated: overcommit_ratio is no longer used by CubeMaster.
// These fields are kept only so leftover YAML is parsed without error;
// the values are ignored at runtime.
DeprecatedOvercommitRatio *deprecatedOvercommitRatioConf `yaml:"overcommit_ratio"`

Copy link
Copy Markdown

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

After this change nothing reads CPURatio/MemRatio anymore — the deprecated struct is populated only so leftover YAML decodes without error, and hasDeprecatedOvercommitConfig() keys off pointer/map presence, never these values. The fields are now write-only. Consider a placeholder struct (or a comment noting the values are intentionally unused) so future readers don't expect them to feed the startup warning or any fallback logic.

DeprecatedOvercommitRatioByType map[string]deprecatedOvercommitRatioConf `yaml:"overcommit_ratio_conf"`
}

var defaultNodeAffinitySelectorAllowedKeys = []string{
Expand Down Expand Up @@ -295,55 +294,16 @@ func IsReservedLabelKey(k string) bool {
return false
}

// OvercommitRatioConf describes the CPU/Mem overcommit multipliers applied to
// a node's reported quota when computing schedulable capacity.
type OvercommitRatioConf struct {
type deprecatedOvercommitRatioConf struct {
CPURatio float64 `yaml:"cpu_ratio"`
MemRatio float64 `yaml:"mem_ratio"`
}

const (
defaultCPUOvercommitRatio = 3.0
defaultMemOvercommitRatio = 2.0
)

// GetEffectiveOvercommitRatio returns the overcommit ratio for the given
// instance type, falling back to the global ratio and then to the built-in
// defaults (CPU=3, Mem=2).
func (s *SchedulerConf) GetEffectiveOvercommitRatio(instanceType string) OvercommitRatioConf {
if s.OvercommitRatioByType != nil {
if v, ok := s.OvercommitRatioByType[instanceType]; ok {
return v.sanitized()
}
}
if s.OvercommitRatio != nil {
return s.OvercommitRatio.sanitized()
}
return OvercommitRatioConf{CPURatio: defaultCPUOvercommitRatio, MemRatio: defaultMemOvercommitRatio}
}

// sanitized guarantees non-positive, NaN, or infinite ratios fall back to the
// defaults so a malformed config never shrinks a node's schedulable capacity to
// zero or produces a garbage (NaN/Inf) capacity when multiplied with the quota.
func (c OvercommitRatioConf) sanitized() OvercommitRatioConf {
out := c
if !isValidRatio(out.CPURatio) {
out.CPURatio = defaultCPUOvercommitRatio
}
if !isValidRatio(out.MemRatio) {
out.MemRatio = defaultMemOvercommitRatio
}
return out
}

// isValidRatio reports whether r is a usable overcommit multiplier: it must be
// a finite, positive number. NaN and ±Inf (e.g. ".nan"/".inf" in YAML) are
// rejected so they never propagate into capacity arithmetic.
func isValidRatio(r float64) bool {
if math.IsNaN(r) || math.IsInf(r, 0) {
func (s *SchedulerConf) hasDeprecatedOvercommitConfig() bool {
if s == nil {
return false
}
return r > 0
return s.DeprecatedOvercommitRatio != nil || len(s.DeprecatedOvercommitRatioByType) > 0
}

// ShouldIgnoreRedisAllocation reports whether the scheduler must ignore the
Expand All @@ -355,40 +315,6 @@ func (s *SchedulerConf) ShouldIgnoreRedisAllocation() bool {
return *s.IgnoreRedisAllocation
}

// EffectiveQuotaCpu returns the schedulable CPU capacity (milli-cores) for a
// node after applying the configured overcommit ratio to its reported quota.
func (s *SchedulerConf) EffectiveQuotaCpu(instanceType string, quotaCpu int64) int64 {
ratio := s.GetEffectiveOvercommitRatio(instanceType)
return floatToInt64Clamped(float64(quotaCpu) * ratio.CPURatio)
}

// EffectiveQuotaMem returns the schedulable memory capacity (MB) for a node
// after applying the configured overcommit ratio to its reported quota.
func (s *SchedulerConf) EffectiveQuotaMem(instanceType string, quotaMem int64) int64 {
ratio := s.GetEffectiveOvercommitRatio(instanceType)
return floatToInt64Clamped(float64(quotaMem) * ratio.MemRatio)
}

// floatToInt64Clamped safely converts a float64 to int64. Converting an
// out-of-range or non-finite float64 to int64 is implementation-defined in Go
// and yields a garbage value, so NaN maps to 0 and values beyond the int64
// range (including ±Inf) are clamped to math.MaxInt64 / math.MinInt64. This
// guards capacity computation against quota * ratio overflowing int64.
func floatToInt64Clamped(f float64) int64 {
if math.IsNaN(f) {
return 0
}
// float64(math.MaxInt64) rounds up to 2^63, so use >= to treat the
// boundary and any larger value (incl. +Inf) as overflow.
if f >= float64(math.MaxInt64) {
return math.MaxInt64
}
if f <= float64(math.MinInt64) {
return math.MinInt64
}
return int64(f)
}

// EffectiveAllocated returns the allocated usage the scheduler should account
// for, which is 0 when Redis allocation records are ignored.
func (s *SchedulerConf) EffectiveAllocated(usage int64) int64 {
Expand Down Expand Up @@ -1043,22 +969,8 @@ func preHandleScheduler(config *Config) error {
ignore := false
config.Scheduler.IgnoreRedisAllocation = &ignore
}
// Default overcommit ratio: CPU=3, Mem=2. sanitized() guards against
// non-positive, NaN, or infinite values supplied by operators.
if config.Scheduler.OvercommitRatio == nil {
config.Scheduler.OvercommitRatio = &OvercommitRatioConf{
CPURatio: defaultCPUOvercommitRatio,
MemRatio: defaultMemOvercommitRatio,
}
} else {
sanitized := config.Scheduler.OvercommitRatio.sanitized()
config.Scheduler.OvercommitRatio = &sanitized
}
// Sanitize per-instance-type overrides at init time as well so malformed
// (non-positive/NaN/Inf) ratios are normalized once up front rather than
// relying solely on the lazy sanitize in GetEffectiveOvercommitRatio.
for k, v := range config.Scheduler.OvercommitRatioByType {
config.Scheduler.OvercommitRatioByType[k] = v.sanitized()
if config.Scheduler.hasDeprecatedOvercommitConfig() {
CubeLog.Warnf("scheduler.overcommit_ratio / overcommit_ratio_conf are deprecated and ignored; CubeMaster no longer applies overcommit to node-reported quota")

Copy link
Copy Markdown

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Blocking/coordination concern: this is only correct if Cubelet already reports effective (overcommitted) quota. Before this PR, CubeMaster applied CPU×3 / Mem×2 implicitly — preHandleScheduler populated those defaults even when overcommit_ratio was absent from YAML — so every current deployment (including ones that never set the key) is being scheduled against quota×3 / quota×2. The replacement mechanism referenced here and in conf.yaml (host.quota.cpu_overcommit_ratio / host.quota.mem_overcommit_ratio) does not exist in this repo's Cubelet today: Cubelet/pkg/config/config.go HostConfigQuota exposes only mcpu_limit/mem_limit/mvm_limit/creation_concurrent_num, and node_status.go applies only fixed derived factors (CPU×2, Mem×5/4) when quota is not explicitly configured.

Consequences if the Cubelet counterpart isn't shipped and deployed first:

  1. Nodes with an explicit host.quota silently lose ~2/3 of CPU and ~1/2 of Mem schedulable capacity → placements that previously fit get filtered/rejected.
  2. Because the old default applied with no YAML key present, hasDeprecatedOvercommitConfig() returns false for the majority of affected deployments — this warning never fires, so the capacity drop is silent.
  3. Per-instance-type overcommit (overcommit_ratio_conf, e.g. GPU nodes at 1.0/1.0) is dropped to a single host-level setting.

Please confirm the Cubelet-side PR is part of the same release (ideally landed first), and consider emitting the warning/upgrade note even for configs that relied on the implicit default.

}

if config.Scheduler.NodeMaxMvmNum == 0 {
Expand Down
132 changes: 17 additions & 115 deletions CubeMaster/pkg/base/config/config_test.go
Original file line number Diff line number Diff line change
Expand Up @@ -7,7 +7,7 @@ package config

import (
"fmt"
"math"

"os"
"path/filepath"
"testing"
Expand Down Expand Up @@ -93,143 +93,45 @@ func TestGetEffectiveNodeMaxMemReservedInMBKeepsConfiguredValue(t *testing.T) {
assert.Equal(t, int64(512), got)
}

func TestPreHandleSchedulerOvercommitAndIgnoreDefaults(t *testing.T) {
func TestPreHandleSchedulerIgnoreRedisAllocationDefault(t *testing.T) {
cfg := &Config{Scheduler: &WrapperSchedulerConf{}}
err := preHandleScheduler(cfg)
assert.NoError(t, err)

assert.NotNil(t, cfg.Scheduler.IgnoreRedisAllocation)
assert.False(t, cfg.Scheduler.ShouldIgnoreRedisAllocation())

ratio := cfg.Scheduler.GetEffectiveOvercommitRatio("cubebox")
assert.Equal(t, 3.0, ratio.CPURatio)
assert.Equal(t, 2.0, ratio.MemRatio)
}

func TestGetEffectiveOvercommitRatioPrecedence(t *testing.T) {
sconf := &SchedulerConf{
OvercommitRatio: &OvercommitRatioConf{CPURatio: 6.0, MemRatio: 4.0},
OvercommitRatioByType: map[string]OvercommitRatioConf{
"cubebox_gpu": {CPURatio: 1.0, MemRatio: 1.0},
func TestHasDeprecatedOvercommitConfig(t *testing.T) {
assert.False(t, (&SchedulerConf{}).hasDeprecatedOvercommitConfig())

assert.True(t, (&SchedulerConf{
DeprecatedOvercommitRatio: &deprecatedOvercommitRatioConf{CPURatio: 3, MemRatio: 2},
}).hasDeprecatedOvercommitConfig())

assert.True(t, (&SchedulerConf{
DeprecatedOvercommitRatioByType: map[string]deprecatedOvercommitRatioConf{
"cubebox_gpu": {CPURatio: 1, MemRatio: 1},
},
}
}).hasDeprecatedOvercommitConfig())

// per-type override wins
gpu := sconf.GetEffectiveOvercommitRatio("cubebox_gpu")
assert.Equal(t, 1.0, gpu.CPURatio)
assert.Equal(t, 1.0, gpu.MemRatio)

// fall back to global ratio
other := sconf.GetEffectiveOvercommitRatio("cubebox")
assert.Equal(t, 6.0, other.CPURatio)
assert.Equal(t, 4.0, other.MemRatio)

// fall back to built-in default when nothing configured
empty := &SchedulerConf{}
def := empty.GetEffectiveOvercommitRatio("cubebox")
assert.Equal(t, 3.0, def.CPURatio)
assert.Equal(t, 2.0, def.MemRatio)
var nilConf *SchedulerConf
assert.False(t, nilConf.hasDeprecatedOvercommitConfig())
}

func TestEffectiveQuotaAndAllocated(t *testing.T) {
func TestEffectiveAllocated(t *testing.T) {
ignore := false
sconf := &SchedulerConf{
IgnoreRedisAllocation: &ignore,
OvercommitRatio: &OvercommitRatioConf{CPURatio: 6.0, MemRatio: 4.0},
}

assert.Equal(t, int64(48000), sconf.EffectiveQuotaCpu("cubebox", 8000))
assert.Equal(t, int64(64000), sconf.EffectiveQuotaMem("cubebox", 16000))
// allocation kept when not ignoring
sconf := &SchedulerConf{IgnoreRedisAllocation: &ignore}
assert.Equal(t, int64(1234), sconf.EffectiveAllocated(1234))

// allocation kept by default (not ignoring)
defaultConf := &SchedulerConf{}
assert.Equal(t, int64(1234), defaultConf.EffectiveAllocated(1234))

// allocation zeroed when explicitly ignoring
ignoreTrue := true
ignoring := &SchedulerConf{IgnoreRedisAllocation: &ignoreTrue}
assert.Equal(t, int64(0), ignoring.EffectiveAllocated(1234))
}

func TestOvercommitRatioSanitizesNonPositive(t *testing.T) {
sconf := &SchedulerConf{
OvercommitRatio: &OvercommitRatioConf{CPURatio: 0, MemRatio: -1},
}
ratio := sconf.GetEffectiveOvercommitRatio("cubebox")
assert.Equal(t, 3.0, ratio.CPURatio)
assert.Equal(t, 2.0, ratio.MemRatio)
}

func TestOvercommitRatioSanitizesNaNAndInf(t *testing.T) {
cases := []OvercommitRatioConf{
{CPURatio: math.NaN(), MemRatio: math.NaN()},
{CPURatio: math.Inf(1), MemRatio: math.Inf(1)},
{CPURatio: math.Inf(-1), MemRatio: math.Inf(-1)},
}
for _, c := range cases {
sconf := &SchedulerConf{OvercommitRatio: &c}
ratio := sconf.GetEffectiveOvercommitRatio("cubebox")
assert.Equal(t, 3.0, ratio.CPURatio)
assert.Equal(t, 2.0, ratio.MemRatio)

// capacity arithmetic must stay finite after sanitizing
assert.Equal(t, int64(24000), sconf.EffectiveQuotaCpu("cubebox", 8000))
assert.Equal(t, int64(32000), sconf.EffectiveQuotaMem("cubebox", 16000))
}
}

func TestPreHandleSchedulerSanitizesPerTypeRatios(t *testing.T) {
cfg := &Config{Scheduler: &WrapperSchedulerConf{
SchedulerConf: SchedulerConf{
OvercommitRatioByType: map[string]OvercommitRatioConf{
"bad_zero": {CPURatio: 0, MemRatio: -1},
"bad_nan": {CPURatio: math.NaN(), MemRatio: math.Inf(1)},
"good": {CPURatio: 8, MemRatio: 5},
},
},
}}
err := preHandleScheduler(cfg)
assert.NoError(t, err)

// malformed per-type ratios are normalized to defaults at init time
bz := cfg.Scheduler.OvercommitRatioByType["bad_zero"]
assert.Equal(t, 3.0, bz.CPURatio)
assert.Equal(t, 2.0, bz.MemRatio)

bn := cfg.Scheduler.OvercommitRatioByType["bad_nan"]
assert.Equal(t, 3.0, bn.CPURatio)
assert.Equal(t, 2.0, bn.MemRatio)

// valid per-type ratios are preserved
g := cfg.Scheduler.OvercommitRatioByType["good"]
assert.Equal(t, 8.0, g.CPURatio)
assert.Equal(t, 5.0, g.MemRatio)
}

func TestFloatToInt64Clamped(t *testing.T) {
assert.Equal(t, int64(0), floatToInt64Clamped(math.NaN()))
assert.Equal(t, int64(math.MaxInt64), floatToInt64Clamped(math.Inf(1)))
assert.Equal(t, int64(math.MinInt64), floatToInt64Clamped(math.Inf(-1)))
// overflow beyond int64 range clamps instead of wrapping to garbage
assert.Equal(t, int64(math.MaxInt64), floatToInt64Clamped(1e30))
assert.Equal(t, int64(math.MinInt64), floatToInt64Clamped(-1e30))
// normal values convert (truncate toward zero) as usual
assert.Equal(t, int64(42), floatToInt64Clamped(42.9))
assert.Equal(t, int64(0), floatToInt64Clamped(0))
}

func TestEffectiveQuotaClampsOverflow(t *testing.T) {
// A huge quota combined with a large overcommit ratio must not wrap to a
// garbage int64; it clamps to MaxInt64 instead.
sconf := &SchedulerConf{
OvercommitRatio: &OvercommitRatioConf{CPURatio: 1e6, MemRatio: 1e6},
}
assert.Equal(t, int64(math.MaxInt64), sconf.EffectiveQuotaCpu("cubebox", math.MaxInt64))
assert.Equal(t, int64(math.MaxInt64), sconf.EffectiveQuotaMem("cubebox", math.MaxInt64))
}

func TestNodeAffinitySelectorAllowedKeySet(t *testing.T) {
sconf := &SchedulerConf{
NodeAffinitySelectorAllowedKeys: []string{"gpu"},
Expand Down
2 changes: 1 addition & 1 deletion CubeMaster/pkg/selector/filter/cpufilter.go
Original file line number Diff line number Diff line change
Expand Up @@ -40,7 +40,7 @@ func (l *cpuFilter) Select(selCtx *selctx.SelectorCtx) (node.NodeList, error) {
nodes := make(node.NodeList, 0, inList.Len())
for i := range inList {

quotaCpuFree := sconf.EffectiveQuotaCpu(inList[i].InstanceType, inList[i].QuotaCpu) -
quotaCpuFree := inList[i].QuotaCpu -
Comment thread
kinwin-ustc marked this conversation as resolved.

Copy link
Copy Markdown

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

This line now treats the Cubelet-reported QuotaCpu as the final schedulable capacity, dropping the CubeMaster-side overcommit multiplier. That is only behavior-preserving if Cubelet already reports quota scaled by its overcommit ratio — but at this PR's merge base it does not: resolveHostQuotaCPUMilli (Cubelet/pkg/cubelet/node_status.go) returns the raw configured host.quota.mcpu_limit, or a default of cpuCount*1000*2, and HostConfigQuota (Cubelet/pkg/config/config.go) has no cpu_overcommit_ratio/mem_overcommit_ratio keys. So for a node with an explicit quota, the schedulable CPU drops from quota*3 (previous default) to quota*1 as soon as this PR is deployed — roughly a 3×/2× capacity cut for CPU/mem until the Cubelet-side change lands and is configured with matching defaults. The same applies to memfilter.go:41, realtimescore.go:122-125, and score/utils.go:110,124. Please confirm the companion Cubelet change ships before/with this PR and that its defaults reproduce the old effective capacity (or that the capacity reduction is intentional and documented); otherwise this is a silent scheduling regression during any mixed-version upgrade.

sconf.EffectiveAllocated(inList[i].QuotaCpuUsage)
if quotaCpuFree <= cpuq.MilliValue() {
log.G(selCtx.Ctx).Warnf("%v select:%v, quotaCpuFree:%v, cpuq:%v",
Expand Down
2 changes: 1 addition & 1 deletion CubeMaster/pkg/selector/filter/memfilter.go
Original file line number Diff line number Diff line change
Expand Up @@ -38,7 +38,7 @@ func (l *memFilter) Select(selCtx *selctx.SelectorCtx) (node.NodeList, error) {
inList := selCtx.Nodes()
nodes := make(node.NodeList, 0, inList.Len())
for i := range inList {
quotaMemFree := sconf.EffectiveQuotaMem(inList[i].InstanceType, inList[i].QuotaMem) -
quotaMemFree := inList[i].QuotaMem -

Copy link
Copy Markdown

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

With this change n.QuotaMem is the overcommitted value, so GetEffectiveNodeMaxMemReservedInMB(instanceType, inList[i].QuotaMem) a few lines below scales the reserved amount by mem_overcommit_ratio (e.g., 10% reserve on a 2x-overcommitted node holds back 20% of physical memory). That tightens the physical check (loadMemFree <= request + reserved) beyond what the docs' "uses the reported (overcommitted) quota as the base" implies. The quota check here is equivalent to before, but the physical-reservation check gets stricter for default nodes.

sconf.EffectiveAllocated(inList[i].QuotaMemUsage)

if quotaMemFree <= memq.Value()/1024/1024 {
Expand Down
6 changes: 2 additions & 4 deletions CubeMaster/pkg/selector/score/realtimescore.go
Original file line number Diff line number Diff line change
Expand Up @@ -119,16 +119,14 @@ func getRealtimeWeightedAverageScore(n *node.Node, cpuq, memq *resource.Quantity
if cpuq != nil {

cpuReqValue := cpuq.MilliValue()
effCpu := schedConf.EffectiveQuotaCpu(n.InstanceType, n.QuotaCpu)
cpuLeft := getReciprocal(effCpu-schedConf.EffectiveAllocated(n.QuotaCpuUsage)-cpuReqValue, effCpu)
cpuLeft := getReciprocal(n.QuotaCpu-schedConf.EffectiveAllocated(n.QuotaCpuUsage)-cpuReqValue, n.QuotaCpu)
scores += cpuLeft * getFactorWeight(constants.WeightFactorReqCpu)
}

if memq != nil {

memReqValue := memq.Value() / 1024 / 1024
effMem := schedConf.EffectiveQuotaMem(n.InstanceType, n.QuotaMem)
memLeft := getReciprocal(effMem-schedConf.EffectiveAllocated(n.QuotaMemUsage)-memReqValue, effMem)
memLeft := getReciprocal(n.QuotaMem-schedConf.EffectiveAllocated(n.QuotaMemUsage)-memReqValue, n.QuotaMem)
scores += memLeft * getFactorWeight(constants.WeightFactorReqMem)
}
return scores
Expand Down
Loading
Loading