Skip to content

Commit b88c57a

Browse files
authored
Merge branch 'main' into migrate/hami-core-on-pr1480
2 parents bd03290 + 86e02f2 commit b88c57a

10 files changed

Lines changed: 477 additions & 24 deletions

File tree

Lines changed: 71 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,71 @@
1+
# Copyright 2026 NVIDIA CORPORATION
2+
# SPDX-License-Identifier: Apache-2.0
3+
4+
# Runs benchmarks on every push to main and publishes a per-commit history
5+
# chart to GitHub Pages at /dev/bench/.
6+
#
7+
# Use workflow_dispatch with `gh-pages-branch: gh-pages-test` to dry-run
8+
# against a throwaway branch before merging changes to this workflow.
9+
10+
name: Benchmark History
11+
12+
on:
13+
push:
14+
branches:
15+
- main
16+
paths:
17+
- 'pkg/scheduler/**'
18+
- 'go.mod'
19+
- 'go.sum'
20+
workflow_dispatch:
21+
inputs:
22+
gh-pages-branch:
23+
description: 'Branch to publish history to (use a throwaway like gh-pages-test for dry runs)'
24+
required: false
25+
default: 'gh-pages'
26+
auto-push:
27+
description: 'Push results to gh-pages-branch'
28+
required: false
29+
type: boolean
30+
default: true
31+
32+
concurrency:
33+
group: benchmark-main
34+
cancel-in-progress: false
35+
36+
jobs:
37+
benchmark:
38+
name: Run and publish benchmarks
39+
runs-on: ubuntu-latest
40+
permissions:
41+
contents: write
42+
deployments: write
43+
steps:
44+
- uses: actions/checkout@v6
45+
46+
- uses: actions/setup-go@v6
47+
with:
48+
go-version: '1.26.1'
49+
cache: true
50+
51+
- name: Download envtest binaries
52+
run: |
53+
make envtest
54+
echo "KUBEBUILDER_ASSETS=$(bin/setup-envtest use 1.34.0 -p path --bin-dir bin)" >> $GITHUB_ENV
55+
56+
- name: Run benchmarks
57+
run: make benchmark BENCH_OUTPUT=bench.txt
58+
59+
- name: Publish benchmark history
60+
uses: benchmark-action/github-action-benchmark@v1
61+
with:
62+
name: KAI Scheduler
63+
tool: go
64+
output-file-path: bench.txt
65+
gh-pages-branch: ${{ github.event.inputs.gh-pages-branch || 'gh-pages' }}
66+
benchmark-data-dir-path: dev/bench
67+
github-token: ${{ secrets.GITHUB_TOKEN }}
68+
auto-push: ${{ github.event_name == 'push' || github.event.inputs.auto-push == 'true' }}
69+
alert-threshold: '150%'
70+
fail-on-alert: false
71+
comment-on-alert: true

‎README.md‎

Lines changed: 10 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -134,6 +134,16 @@ Please open a [GitHub issue](https://github.com/kai-scheduler/KAI-scheduler/issu
134134

135135
---
136136

137+
## Performance Dashboards
138+
139+
KAI Scheduler provides public dashboards for monitoring performance and scale testing:
140+
141+
- **[Scale Tests Dashboard](https://kai-scheduler.github.io/KAI-Scheduler/)**: View historical results from scale tests that validate scheduler performance at large cluster sizes (hundreds to thousands of nodes). Tests run every 24 hours on dedicated infrastructure and measure scheduling performance, topology-aware scheduling, resource allocation, and system stability under load. The dashboard displays execution times, pass/fail status, detailed failure logs, and 30-day historical trends. See [scale tests documentation](docs/developer/scale-tests.md) for technical details.
142+
143+
- **[Benchmarks Dashboard](https://kai-scheduler.github.io/KAI-Scheduler/dev/bench/)**: Track scheduler performance benchmarks across commits to the main branch. The dashboard shows per-commit benchmark history for core scheduler operations, with automatic alerts when performance regresses beyond thresholds.
144+
145+
---
146+
137147
<div align="center">
138148
<picture>
139149
<source media="(prefers-color-scheme: dark)" srcset="https://raw.githubusercontent.com/cncf/artwork/refs/heads/main/other/cncf/horizontal/color-whitetext/cncf-color-whitetext.svg">

‎docs/plugins/reflectjoborder.md‎

Lines changed: 83 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,83 @@
1+
# Reflect Job Order Plugin
2+
3+
## Overview
4+
5+
The `reflectjoborder` plugin exposes the order in which the scheduler intends to process pending PodGroups (jobs) during the next scheduling cycle. It is the supported way to answer the question *"where does my workload sit in its queue right now?"*.
6+
7+
The plugin is read-only: it observes the same ordering used by the `Allocate` action and serves it over an HTTP endpoint on the scheduler pod. Each scheduling cycle refreshes the data.
8+
9+
## Enabling the Plugin
10+
11+
The plugin is registered in the scheduler binary but **not enabled by default**. Enable it on the relevant `SchedulingShard`:
12+
13+
```yaml
14+
apiVersion: kai.scheduler/v1
15+
kind: SchedulingShard
16+
metadata:
17+
name: default
18+
spec:
19+
plugins:
20+
reflectjoborder:
21+
enabled: true
22+
```
23+
24+
When installing via Helm, set `scheduler.plugins` in your values file:
25+
26+
```yaml
27+
scheduler:
28+
plugins:
29+
reflectjoborder:
30+
enabled: true
31+
```
32+
33+
See `deployments/kai-scheduler/examples/custom-plugins-actions-values.yaml` for the full plugin-configuration syntax.
34+
35+
## Querying Job Order
36+
37+
The plugin registers an HTTP endpoint `/get-job-order` on the scheduler pod. Port-forward to the scheduler and call it:
38+
39+
```bash
40+
kubectl port-forward -n kai-scheduler deployment/kai-scheduler-default 8081 &
41+
sleep 2
42+
curl -s "localhost:8081/get-job-order" | jq
43+
```
44+
45+
## Response Format
46+
47+
```json
48+
{
49+
"global_order": [
50+
{ "id": "team-a/training-7d2", "priority": 100 },
51+
{ "id": "team-b/inference-9c1", "priority": 75 }
52+
],
53+
"queue_order": {
54+
"team-a": [
55+
{ "id": "team-a/training-7d2", "priority": 100 }
56+
],
57+
"team-b": [
58+
{ "id": "team-b/inference-9c1", "priority": 75 }
59+
]
60+
}
61+
}
62+
```
63+
64+
| Field | Description |
65+
|---|---|
66+
| `global_order` | All eligible jobs across all queues, in scheduling order. Index 0 is next to be considered. |
67+
| `queue_order` | Same jobs grouped by queue. Index within a queue is that workload's position in its queue. |
68+
| `id` | PodGroup UID (`<namespace>/<name>`). |
69+
| `priority` | Effective priority used for ordering. |
70+
71+
A workload's queue position is `index in queue_order[<queue>] + 1`.
72+
73+
## Caveats
74+
75+
- **Per-cycle snapshot.** The data is computed at the start of each scheduling cycle; it is not updated continuously. A workload's reported position may be a few seconds stale.
76+
- **Pending and ready jobs only.** The plugin filters with `FilterNonPending: true` and `FilterUnready: true`, so running jobs and jobs not yet ready for scheduling do not appear (`pkg/scheduler/actions/utils/input_jobs.go`).
77+
- **Bounded by `QueueDepthPerAction[Allocate]`.** Only the first *N* jobs per queue are reported, where *N* is the configured allocate queue depth. Jobs beyond that depth are excluded, even if pending.
78+
- **No history.** Only the most recent cycle is exposed; there is no time-series store. To track position over time, scrape the endpoint on an interval.
79+
- **Not authenticated.** The endpoint is served on the scheduler pod's HTTP port. Reach it via `kubectl port-forward` or an in-cluster client; do not expose it publicly.
80+
81+
## Implementation
82+
83+
Source: `pkg/scheduler/plugins/reflectjoborder/reflect_job_order.go`. The plugin populates `ReflectJobOrder` in `OnSessionOpen` by draining `utils.NewJobsOrderByQueues(...)` and registers `serveJobs` on `/get-job-order`.

‎pkg/apis/scheduling/v2alpha2/podgroup_webhook.go‎

Lines changed: 1 addition & 4 deletions
Original file line numberDiff line numberDiff line change
@@ -200,10 +200,7 @@ func validateSubGroupMinFields(subGroup *SubGroup, childrenMap map[string][]stri
200200
"subgroup %q: minMember cannot be set on a mid-level SubGroup (has child SubGroups); use minSubGroup instead", subGroup.Name)})
201201
}
202202
if subGroup.MinSubGroup != nil {
203-
if *subGroup.MinSubGroup <= 0 {
204-
minFieldsErrors = append(minFieldsErrors, &invalidMinSubGroupError{msg: fmt.Sprintf(
205-
"subgroup %q: minSubGroup must be greater than 0", subGroup.Name)})
206-
} else if int(*subGroup.MinSubGroup) > len(children) {
203+
if int(*subGroup.MinSubGroup) > len(children) {
207204
minFieldsErrors = append(minFieldsErrors, &minSubGroupExceedsChildCountError{msg: fmt.Sprintf(
208205
"subgroup %q: minSubGroup (%d) exceeds the number of direct child SubGroups (%d)",
209206
subGroup.Name, *subGroup.MinSubGroup, len(children))})

‎pkg/apis/scheduling/v2alpha2/podgroup_webhook_test.go‎

Lines changed: 0 additions & 20 deletions
Original file line numberDiff line numberDiff line change
@@ -171,15 +171,6 @@ func TestValidateSubGroups(t *testing.T) {
171171
},
172172
wantErr: nil,
173173
},
174-
{
175-
name: "Invalid: minSubGroup = 0 on SubGroup with children",
176-
subGroups: []SubGroup{
177-
{Name: "parent", MinSubGroup: ptr.To(int32(0))},
178-
{Name: "child-1", Parent: ptr.To("parent"), MinMember: ptr.To(int32(4))},
179-
{Name: "child-2", Parent: ptr.To("parent"), MinMember: ptr.To(int32(4))},
180-
},
181-
wantErr: &validationErrors{minDefinitionErrors: []error{&invalidMinSubGroupError{msg: `subgroup "parent": minSubGroup must be greater than 0`}}},
182-
},
183174
{
184175
name: "Invalid: minSubGroup = 0 on leaf SubGroup",
185176
subGroups: []SubGroup{
@@ -324,17 +315,6 @@ func TestValidatePodGroupSpec(t *testing.T) {
324315
},
325316
want: &validationErrors{minDefinitionErrors: []error{&minSubGroupExceedsChildCountError{msg: "minSubGroup (1) exceeds the number of direct child SubGroups (0)"}}},
326317
},
327-
{
328-
name: "Invalid: minSubGroup = 0 on PodGroup",
329-
spec: PodGroupSpec{
330-
MinSubGroup: ptr.To(int32(0)),
331-
SubGroups: []SubGroup{
332-
{Name: "a", MinMember: ptr.To(int32(4))},
333-
{Name: "b", MinMember: ptr.To(int32(4))},
334-
},
335-
},
336-
want: &validationErrors{minDefinitionErrors: []error{&invalidMinSubGroupError{msg: "minSubGroup at the podgroup level must be equal to or greater than 1"}}},
337-
},
338318
}
339319

340320
for _, tt := range tests {

‎pkg/scheduler/api/podgroup_info/allocation_info_test.go‎

Lines changed: 73 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -220,6 +220,79 @@ func Test_GetTasksToAllocate(t *testing.T) {
220220
}
221221
}
222222

223+
func Test_GetTasksToAllocate_MinSubGroupZero(t *testing.T) {
224+
tests := []struct {
225+
name string
226+
subGroupTasks map[string][]*pod_info.PodInfo
227+
minAvailMap map[string]int32
228+
wantNumTasks int
229+
}{
230+
{
231+
name: "root minSubGroup=0, all children unsatisfied: gang skipped, elastic returns one child's tasks",
232+
subGroupTasks: map[string][]*pod_info.PodInfo{
233+
"sgA": {
234+
simpleTask("taskA1", "sgA", pod_status.Pending),
235+
},
236+
"sgB": {
237+
simpleTask("taskB1", "sgB", pod_status.Pending),
238+
},
239+
},
240+
minAvailMap: map[string]int32{"sgA": 1, "sgB": 1},
241+
wantNumTasks: 1,
242+
},
243+
{
244+
name: "root minSubGroup=0, all children satisfied: elastic returns no tasks",
245+
subGroupTasks: map[string][]*pod_info.PodInfo{
246+
"sgA": {
247+
simpleTask("taskA1", "sgA", pod_status.Running),
248+
},
249+
"sgB": {
250+
simpleTask("taskB1", "sgB", pod_status.Running),
251+
},
252+
},
253+
minAvailMap: map[string]int32{"sgA": 1, "sgB": 1},
254+
wantNumTasks: 0,
255+
},
256+
{
257+
name: "root minSubGroup=0, one satisfied + one with elastic surplus: returns one elastic task",
258+
subGroupTasks: map[string][]*pod_info.PodInfo{
259+
"sgA": {
260+
simpleTask("taskA1", "sgA", pod_status.Running),
261+
},
262+
"sgB": {
263+
simpleTask("taskB1", "sgB", pod_status.Running),
264+
simpleTask("taskB2", "sgB", pod_status.Pending),
265+
},
266+
},
267+
minAvailMap: map[string]int32{"sgA": 1, "sgB": 1},
268+
wantNumTasks: 1,
269+
},
270+
}
271+
272+
for _, tt := range tests {
273+
t.Run(tt.name, func(t *testing.T) {
274+
pg := NewPodGroupInfo("pg")
275+
root := subgroup_info.NewSubGroupSet(subgroup_info.RootSubGroupSetName, nil)
276+
min := int32(0)
277+
root.SetMinSubGroup(&min)
278+
pg.RootSubGroupSet = root
279+
pg.PodSets = make(map[string]*subgroup_info.PodSet)
280+
for subGroupName, pods := range tt.subGroupTasks {
281+
ps := subgroup_info.NewPodSet(subGroupName, tt.minAvailMap[subGroupName], nil)
282+
root.AddPodSet(ps)
283+
pg.PodSets[subGroupName] = ps
284+
for _, pod := range pods {
285+
pg.AddTaskInfo(pod)
286+
}
287+
}
288+
gotTasks := GetTasksToAllocate(pg, subGroupOrderFn, tasksOrderFn, true)
289+
if len(gotTasks) != tt.wantNumTasks {
290+
t.Errorf("GetTasksToAllocate len = %d, want %d", len(gotTasks), tt.wantNumTasks)
291+
}
292+
})
293+
}
294+
}
295+
223296
func Test_GetTasksToAllocateRequestedGPUs(t *testing.T) {
224297
pg := NewPodGroupInfo("test-podgroup")
225298
pg.GetAllPodSets()[DefaultSubGroup].SetMinAvailable(1)

‎pkg/scheduler/api/podgroup_info/eviction_info_test.go‎

Lines changed: 72 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -324,6 +324,78 @@ func TestGetTasksToEvict_HierarchicalTree(t *testing.T) {
324324
expectedHasMoreTasks: true,
325325
numExpectTasks: 2,
326326
},
327+
{
328+
name: "MinSubGroupZero_RootEntireSubtreeElasticDroppable",
329+
job: func() *PodGroupInfo {
330+
// Root with minSubGroup=0 and one PodSet child at exactly min.
331+
// Phase 2 fires because 0 < numSatisfied(1) → drop the whole child.
332+
psA := subgroup_info.NewPodSet("ps-a", 1, nil)
333+
psA.AssignTask(simpleTask("pod-1", "ps-a", pod_status.Running))
334+
335+
root := subgroup_info.NewSubGroupSet(subgroup_info.RootSubGroupSetName, nil)
336+
root.SetMinSubGroup(ptr.To(int32(0)))
337+
root.AddPodSet(psA)
338+
339+
return &PodGroupInfo{
340+
RootSubGroupSet: root,
341+
PodSets: root.GetDescendantPodSets(),
342+
}
343+
}(),
344+
expectedHasMoreTasks: false,
345+
numExpectTasks: 1,
346+
},
347+
{
348+
name: "MinSubGroupZero_NestedInnerSubtreeElasticDroppable",
349+
job: func() *PodGroupInfo {
350+
// Inner SubGroupSet with minSubGroup=0 containing a PodSet at min.
351+
// Root requires inner (no minSubGroup). Inner is treated as fully elastic:
352+
// phase 1 finds elastic surplus inside inner → recurses → inner phase 2 drops ps-a.
353+
psA := subgroup_info.NewPodSet("ps-a", 1, nil)
354+
psA.AssignTask(simpleTask("pod-1", "ps-a", pod_status.Running))
355+
356+
inner := subgroup_info.NewSubGroupSet("inner", nil)
357+
inner.SetMinSubGroup(ptr.To(int32(0)))
358+
inner.AddPodSet(psA)
359+
360+
root := subgroup_info.NewSubGroupSet(subgroup_info.RootSubGroupSetName, nil)
361+
root.AddSubGroup(inner)
362+
363+
return &PodGroupInfo{
364+
RootSubGroupSet: root,
365+
PodSets: root.GetDescendantPodSets(),
366+
}
367+
}(),
368+
expectedHasMoreTasks: false,
369+
numExpectTasks: 1,
370+
},
371+
{
372+
name: "MinSubGroupZero_NestedInnerLeavesSiblingAlone",
373+
job: func() *PodGroupInfo {
374+
// Root requires both children; inner has minSubGroup=0 (fully elastic),
375+
// sibling required ps-keep is at min. Eviction must come from inner only.
376+
psElastic := subgroup_info.NewPodSet("ps-elastic", 1, nil)
377+
psElastic.AssignTask(simpleTask("pod-1", "ps-elastic", pod_status.Running))
378+
379+
inner := subgroup_info.NewSubGroupSet("inner", nil)
380+
inner.SetMinSubGroup(ptr.To(int32(0)))
381+
inner.AddPodSet(psElastic)
382+
383+
psKeep := subgroup_info.NewPodSet("ps-keep", 1, nil)
384+
psKeep.AssignTask(simpleTask("pod-2", "ps-keep", pod_status.Running))
385+
386+
root := subgroup_info.NewSubGroupSet(subgroup_info.RootSubGroupSetName, nil)
387+
root.AddSubGroup(inner)
388+
root.AddPodSet(psKeep)
389+
390+
return &PodGroupInfo{
391+
RootSubGroupSet: root,
392+
PodSets: root.GetDescendantPodSets(),
393+
}
394+
}(),
395+
expectedHasMoreTasks: true, // pod-2 in ps-keep stays
396+
numExpectTasks: 1,
397+
expectedTaskNames: []string{"pod-1"},
398+
},
327399
}
328400

329401
for _, tt := range tests {

0 commit comments

Comments
 (0)