Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
5 changes: 5 additions & 0 deletions sidebars-tutorials.js
Original file line number Diff line number Diff line change
Expand Up @@ -106,6 +106,11 @@ module.exports = {
id: "labs/hami-ascend-vnpu-slicing",
customProps: { level: "Intermediate", duration: "about 60 minutes" },
},
{
type: "doc",
id: "labs/init-sidecar-container-accounting",
customProps: { level: "Intermediate", duration: "about 45 minutes" },
},
],
},
],
Expand Down
Original file line number Diff line number Diff line change
@@ -0,0 +1,30 @@
apiVersion: v1

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Please read the contribution.md before contributing to the project.

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

please see this file also before contributing this is template for lab files .https://github.com/Project-HAMi/website/blob/master/tutorials/_template.md

kind: Pod
metadata:
name: regular-init-pod
labels:
app: regular-init-demo
spec:
restartPolicy: Never
initContainers:
- name: model-prep
image: busybox:1.36
command: ["sh", "-c", "echo 'Preparing model weights on GPU...' && sleep 15 && echo 'Done'"]
env:
- name: CUDA_DISABLE_CONTROL
value: "true" # Enables lightweight simulation in fake GPU / nvml-mock environments
resources:
limits:
nvidia.com/gpu: 1
nvidia.com/gpumem: 5000 # 5000 MiB allocated during init (peak allocation)
containers:
- name: inference-app
image: busybox:1.36
command: ["sh", "-c", "echo 'Inference app running...' && sleep infinity"]
env:
- name: CUDA_DISABLE_CONTROL
value: "true"
resources:
limits:
nvidia.com/gpu: 1
nvidia.com/gpumem: 2000 # 2000 MiB allocated to the main application (steady state)
Original file line number Diff line number Diff line change
@@ -0,0 +1,31 @@
apiVersion: v1
kind: Pod
metadata:
name: native-sidecar-pod
labels:
app: native-sidecar-demo
spec:
restartPolicy: Never
initContainers:
- name: proxy-sidecar
image: busybox:1.36
restartPolicy: Always # KEP-753 Native Sidecar Container: runs for the full Pod lifetime
command: ["sh", "-c", "echo 'Inference proxy sidecar running...' && sleep infinity"]
env:
- name: CUDA_DISABLE_CONTROL
value: "true"
resources:
limits:
nvidia.com/gpu: 1
nvidia.com/gpumem: 2000 # 2000 MiB permanently reserved for the sidecar
containers:
- name: inference-app
image: busybox:1.36
command: ["sh", "-c", "echo 'Main inference app running...' && sleep infinity"]
env:
- name: CUDA_DISABLE_CONTROL
value: "true"
resources:
limits:
nvidia.com/gpu: 1
nvidia.com/gpumem: 4000 # 4000 MiB reserved for the app (Total demand: 6000 MiB)
Original file line number Diff line number Diff line change
@@ -0,0 +1,43 @@
apiVersion: v1
kind: Pod
metadata:
name: mixed-workload-pod
labels:
app: mixed-workload-demo
spec:
restartPolicy: Never
initContainers:
# Sidecar declared before the regular init container
- name: cache-sidecar
image: busybox:1.36
restartPolicy: Always # Native sidecar: stays running
command: ["sh", "-c", "echo 'Cache sidecar active...' && sleep infinity"]
env:
- name: CUDA_DISABLE_CONTROL
value: "true"
resources:
limits:
nvidia.com/gpu: 1
nvidia.com/gpumem: 2000 # 2000 MiB
# Regular init container: runs to completion
- name: weights-loader
image: busybox:1.36
command: ["sh", "-c", "echo 'Loading model weights...' && sleep 15 && echo 'Done'"]
env:
- name: CUDA_DISABLE_CONTROL
value: "true"
resources:
limits:
nvidia.com/gpu: 1
nvidia.com/gpumem: 5000 # 5000 MiB (Peak with preceding sidecar: 7000 MiB)
containers:
- name: serving-engine
image: busybox:1.36
command: ["sh", "-c", "echo 'Serving engine active...' && sleep infinity"]
env:
- name: CUDA_DISABLE_CONTROL
value: "true"
resources:
limits:
nvidia.com/gpu: 1
nvidia.com/gpumem: 4000 # 4000 MiB (Steady-state with sidecar: 6000 MiB)
Original file line number Diff line number Diff line change
@@ -0,0 +1,23 @@
apiVersion: v1
kind: Pod
metadata:
name: competing-workload-pod
labels:
app: competing-workload
annotations:
# Pin this workload to the same GPU card hosting native-sidecar-pod
# Replace TARGET_GPU_UUID with the allocated GPU UUID from Step 3
nvidia.com/use-gpuuuid: "TARGET_GPU_UUID"
spec:
restartPolicy: Never
containers:
- name: worker
image: busybox:1.36
command: ["sh", "-c", "echo 'Competing workload running...' && sleep infinity"]
env:
- name: CUDA_DISABLE_CONTROL
value: "true"
resources:
limits:
nvidia.com/gpu: 1
nvidia.com/gpumem: 36000 # Requests 36000 MiB: exceeds the 34960 MiB remaining on a 40960 MiB A100 GPU
Loading
Loading