-
Notifications
You must be signed in to change notification settings - Fork 1.2k
refactor(scheduler): remove CubeMaster overcommit ratio #1575
New issue
Have a question about this project? Sign up for a free GitHub account to open an issue and contact its maintainers and the community.
By clicking “Sign up for GitHub”, you agree to our terms of service and privacy statement. We’ll occasionally send you account related emails.
Already on GitHub? Sign in to your account
Changes from all commits
File filter
Filter by extension
Conversations
Jump to
Diff view
Diff view
There are no files selected for viewing
| Original file line number | Diff line number | Diff line change |
|---|---|---|
|
|
@@ -251,12 +251,11 @@ type SchedulerConf struct { | |
| // 0). A pointer is used so an unset value can default to false while still | ||
| // allowing operators to explicitly enable it. Defaults to false. | ||
| IgnoreRedisAllocation *bool `yaml:"ignore_redis_allocation"` | ||
| // OvercommitRatio is the global CPU/Mem overcommit ratio applied to the | ||
| // node-reported quota during scheduling. Defaults to CPU=3, Mem=2. | ||
| OvercommitRatio *OvercommitRatioConf `yaml:"overcommit_ratio"` | ||
| // OvercommitRatioByType overrides OvercommitRatio for specific instance | ||
| // types and takes precedence over the global ratio. | ||
| OvercommitRatioByType map[string]OvercommitRatioConf `yaml:"overcommit_ratio_conf"` | ||
| // Deprecated: overcommit_ratio is no longer used by CubeMaster. | ||
| // These fields are kept only so leftover YAML is parsed without error; | ||
| // the values are ignored at runtime. | ||
| DeprecatedOvercommitRatio *deprecatedOvercommitRatioConf `yaml:"overcommit_ratio"` | ||
|
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. After this change nothing reads |
||
| DeprecatedOvercommitRatioByType map[string]deprecatedOvercommitRatioConf `yaml:"overcommit_ratio_conf"` | ||
| } | ||
|
|
||
| var defaultNodeAffinitySelectorAllowedKeys = []string{ | ||
|
|
@@ -295,55 +294,16 @@ func IsReservedLabelKey(k string) bool { | |
| return false | ||
| } | ||
|
|
||
| // OvercommitRatioConf describes the CPU/Mem overcommit multipliers applied to | ||
| // a node's reported quota when computing schedulable capacity. | ||
| type OvercommitRatioConf struct { | ||
| type deprecatedOvercommitRatioConf struct { | ||
| CPURatio float64 `yaml:"cpu_ratio"` | ||
| MemRatio float64 `yaml:"mem_ratio"` | ||
| } | ||
|
|
||
| const ( | ||
| defaultCPUOvercommitRatio = 3.0 | ||
| defaultMemOvercommitRatio = 2.0 | ||
| ) | ||
|
|
||
| // GetEffectiveOvercommitRatio returns the overcommit ratio for the given | ||
| // instance type, falling back to the global ratio and then to the built-in | ||
| // defaults (CPU=3, Mem=2). | ||
| func (s *SchedulerConf) GetEffectiveOvercommitRatio(instanceType string) OvercommitRatioConf { | ||
| if s.OvercommitRatioByType != nil { | ||
| if v, ok := s.OvercommitRatioByType[instanceType]; ok { | ||
| return v.sanitized() | ||
| } | ||
| } | ||
| if s.OvercommitRatio != nil { | ||
| return s.OvercommitRatio.sanitized() | ||
| } | ||
| return OvercommitRatioConf{CPURatio: defaultCPUOvercommitRatio, MemRatio: defaultMemOvercommitRatio} | ||
| } | ||
|
|
||
| // sanitized guarantees non-positive, NaN, or infinite ratios fall back to the | ||
| // defaults so a malformed config never shrinks a node's schedulable capacity to | ||
| // zero or produces a garbage (NaN/Inf) capacity when multiplied with the quota. | ||
| func (c OvercommitRatioConf) sanitized() OvercommitRatioConf { | ||
| out := c | ||
| if !isValidRatio(out.CPURatio) { | ||
| out.CPURatio = defaultCPUOvercommitRatio | ||
| } | ||
| if !isValidRatio(out.MemRatio) { | ||
| out.MemRatio = defaultMemOvercommitRatio | ||
| } | ||
| return out | ||
| } | ||
|
|
||
| // isValidRatio reports whether r is a usable overcommit multiplier: it must be | ||
| // a finite, positive number. NaN and ±Inf (e.g. ".nan"/".inf" in YAML) are | ||
| // rejected so they never propagate into capacity arithmetic. | ||
| func isValidRatio(r float64) bool { | ||
| if math.IsNaN(r) || math.IsInf(r, 0) { | ||
| func (s *SchedulerConf) hasDeprecatedOvercommitConfig() bool { | ||
| if s == nil { | ||
| return false | ||
| } | ||
| return r > 0 | ||
| return s.DeprecatedOvercommitRatio != nil || len(s.DeprecatedOvercommitRatioByType) > 0 | ||
| } | ||
|
|
||
| // ShouldIgnoreRedisAllocation reports whether the scheduler must ignore the | ||
|
|
@@ -355,40 +315,6 @@ func (s *SchedulerConf) ShouldIgnoreRedisAllocation() bool { | |
| return *s.IgnoreRedisAllocation | ||
| } | ||
|
|
||
| // EffectiveQuotaCpu returns the schedulable CPU capacity (milli-cores) for a | ||
| // node after applying the configured overcommit ratio to its reported quota. | ||
| func (s *SchedulerConf) EffectiveQuotaCpu(instanceType string, quotaCpu int64) int64 { | ||
| ratio := s.GetEffectiveOvercommitRatio(instanceType) | ||
| return floatToInt64Clamped(float64(quotaCpu) * ratio.CPURatio) | ||
| } | ||
|
|
||
| // EffectiveQuotaMem returns the schedulable memory capacity (MB) for a node | ||
| // after applying the configured overcommit ratio to its reported quota. | ||
| func (s *SchedulerConf) EffectiveQuotaMem(instanceType string, quotaMem int64) int64 { | ||
| ratio := s.GetEffectiveOvercommitRatio(instanceType) | ||
| return floatToInt64Clamped(float64(quotaMem) * ratio.MemRatio) | ||
| } | ||
|
|
||
| // floatToInt64Clamped safely converts a float64 to int64. Converting an | ||
| // out-of-range or non-finite float64 to int64 is implementation-defined in Go | ||
| // and yields a garbage value, so NaN maps to 0 and values beyond the int64 | ||
| // range (including ±Inf) are clamped to math.MaxInt64 / math.MinInt64. This | ||
| // guards capacity computation against quota * ratio overflowing int64. | ||
| func floatToInt64Clamped(f float64) int64 { | ||
| if math.IsNaN(f) { | ||
| return 0 | ||
| } | ||
| // float64(math.MaxInt64) rounds up to 2^63, so use >= to treat the | ||
| // boundary and any larger value (incl. +Inf) as overflow. | ||
| if f >= float64(math.MaxInt64) { | ||
| return math.MaxInt64 | ||
| } | ||
| if f <= float64(math.MinInt64) { | ||
| return math.MinInt64 | ||
| } | ||
| return int64(f) | ||
| } | ||
|
|
||
| // EffectiveAllocated returns the allocated usage the scheduler should account | ||
| // for, which is 0 when Redis allocation records are ignored. | ||
| func (s *SchedulerConf) EffectiveAllocated(usage int64) int64 { | ||
|
|
@@ -1043,22 +969,8 @@ func preHandleScheduler(config *Config) error { | |
| ignore := false | ||
| config.Scheduler.IgnoreRedisAllocation = &ignore | ||
| } | ||
| // Default overcommit ratio: CPU=3, Mem=2. sanitized() guards against | ||
| // non-positive, NaN, or infinite values supplied by operators. | ||
| if config.Scheduler.OvercommitRatio == nil { | ||
| config.Scheduler.OvercommitRatio = &OvercommitRatioConf{ | ||
| CPURatio: defaultCPUOvercommitRatio, | ||
| MemRatio: defaultMemOvercommitRatio, | ||
| } | ||
| } else { | ||
| sanitized := config.Scheduler.OvercommitRatio.sanitized() | ||
| config.Scheduler.OvercommitRatio = &sanitized | ||
| } | ||
| // Sanitize per-instance-type overrides at init time as well so malformed | ||
| // (non-positive/NaN/Inf) ratios are normalized once up front rather than | ||
| // relying solely on the lazy sanitize in GetEffectiveOvercommitRatio. | ||
| for k, v := range config.Scheduler.OvercommitRatioByType { | ||
| config.Scheduler.OvercommitRatioByType[k] = v.sanitized() | ||
| if config.Scheduler.hasDeprecatedOvercommitConfig() { | ||
| CubeLog.Warnf("scheduler.overcommit_ratio / overcommit_ratio_conf are deprecated and ignored; CubeMaster no longer applies overcommit to node-reported quota") | ||
|
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. Blocking/coordination concern: this is only correct if Cubelet already reports effective (overcommitted) quota. Before this PR, CubeMaster applied CPU×3 / Mem×2 implicitly — Consequences if the Cubelet counterpart isn't shipped and deployed first:
Please confirm the Cubelet-side PR is part of the same release (ideally landed first), and consider emitting the warning/upgrade note even for configs that relied on the implicit default. |
||
| } | ||
|
|
||
| if config.Scheduler.NodeMaxMvmNum == 0 { | ||
|
|
||
| Original file line number | Diff line number | Diff line change |
|---|---|---|
|
|
@@ -40,7 +40,7 @@ func (l *cpuFilter) Select(selCtx *selctx.SelectorCtx) (node.NodeList, error) { | |
| nodes := make(node.NodeList, 0, inList.Len()) | ||
| for i := range inList { | ||
|
|
||
| quotaCpuFree := sconf.EffectiveQuotaCpu(inList[i].InstanceType, inList[i].QuotaCpu) - | ||
| quotaCpuFree := inList[i].QuotaCpu - | ||
|
kinwin-ustc marked this conversation as resolved.
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. This line now treats the Cubelet-reported |
||
| sconf.EffectiveAllocated(inList[i].QuotaCpuUsage) | ||
| if quotaCpuFree <= cpuq.MilliValue() { | ||
| log.G(selCtx.Ctx).Warnf("%v select:%v, quotaCpuFree:%v, cpuq:%v", | ||
|
|
||
| Original file line number | Diff line number | Diff line change |
|---|---|---|
|
|
@@ -38,7 +38,7 @@ func (l *memFilter) Select(selCtx *selctx.SelectorCtx) (node.NodeList, error) { | |
| inList := selCtx.Nodes() | ||
| nodes := make(node.NodeList, 0, inList.Len()) | ||
| for i := range inList { | ||
| quotaMemFree := sconf.EffectiveQuotaMem(inList[i].InstanceType, inList[i].QuotaMem) - | ||
| quotaMemFree := inList[i].QuotaMem - | ||
|
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. With this change |
||
| sconf.EffectiveAllocated(inList[i].QuotaMemUsage) | ||
|
|
||
| if quotaMemFree <= memq.Value()/1024/1024 { | ||
|
|
||
There was a problem hiding this comment.
Choose a reason for hiding this comment
The reason will be displayed to describe this comment to others. Learn more.
Worth calling out that this isn't a no-op deprecation: schedulable capacity now equals the raw Cubelet-reported quota, whereas the default before was quota × 3 (CPU) / × 2 (mem). Cubelet still returns an explicitly configured
host.quota.mcpu_limit/mem_limitverbatim (its overcommit factors only apply to auto-derived defaults), so a node withmcpu_limit: 24000that used to offer 72000 milli-CPU now offers 24000 after upgrade. Suggest adding a short migration note in the docs telling operators to fold their previous ratio into the explicit host quota, or expect a one-time capacity drop (abrupt during rolling master/Cubelet upgrades).