From 3ec46ce839309b9822420d01f61eee6a95296e3e Mon Sep 17 00:00:00 2001 From: praveen Date: Thu, 24 Sep 2026 11:32:41 -0700 Subject: [PATCH 1/4] docs(npd): add minimal dmesg-only NPD example Add docs/npd/npd-dmesg-example.md with a self-contained walkthrough for watching /dev/kmsg for AMD GPU kernel crashes using only NPD's built-in system-log monitor. No AMD Device Metrics Exporter or amdgpuhealth required. Covers RBAC, kernel-monitor.json ConfigMap with amdgpu.* regex rules (page fault, hang, reset, RAS error), a stripped DaemonSet with only /dev/kmsg mounted, and verify steps. Co-Authored-By: Claude --- docs/npd/npd-dmesg-example.md | 250 ++++++++++++++++++++++++++++++++++ 1 file changed, 250 insertions(+) create mode 100644 docs/npd/npd-dmesg-example.md diff --git a/docs/npd/npd-dmesg-example.md b/docs/npd/npd-dmesg-example.md new file mode 100644 index 000000000..ab2576057 --- /dev/null +++ b/docs/npd/npd-dmesg-example.md @@ -0,0 +1,250 @@ +# NPD dmesg-Only Example + +This page shows the minimal configuration to use Node Problem Detector to watch the +kernel ring buffer (`/dev/kmsg`) for AMD GPU crashes. No AMD Device Metrics Exporter +or `amdgpuhealth` binary is required — this example uses only NPD's built-in +**system-log monitor**. + +## How it works + +NPD reads `/dev/kmsg` continuously. When a log line matches a regex rule, NPD either +emits a one-shot `Event` (`type: temporary`) or flips a persistent `NodeCondition` +to `True` (`type: permanent`). The conditions are visible via `kubectl describe node`. + +## Step 1 — RBAC + +NPD needs permission to patch `nodes/status` (to write conditions) and create `events`. +No non-resource URL permissions are needed for the dmesg-only setup. + +```yaml +# npd-rbac.yaml +apiVersion: v1 +kind: ServiceAccount +metadata: + name: node-problem-detector + namespace: kube-system +--- +apiVersion: rbac.authorization.k8s.io/v1 +kind: ClusterRole +metadata: + name: node-problem-detector +rules: +- apiGroups: [""] + resources: ["nodes/status"] + verbs: ["patch"] +- apiGroups: [""] + resources: ["events"] + verbs: ["create", "patch"] +--- +apiVersion: rbac.authorization.k8s.io/v1 +kind: ClusterRoleBinding +metadata: + name: node-problem-detector +roleRef: + apiGroup: rbac.authorization.k8s.io + kind: ClusterRole + name: node-problem-detector +subjects: +- kind: ServiceAccount + name: node-problem-detector + namespace: kube-system +``` + +```bash +kubectl apply -f npd-rbac.yaml +``` + +## Step 2 — ConfigMap + +The `kernel-monitor.json` key defines which dmesg patterns to watch. + +```yaml +# npd-dmesg-config.yaml +apiVersion: v1 +kind: ConfigMap +metadata: + name: node-problem-detector-config + namespace: kube-system +data: + kernel-monitor.json: | + { + "plugin": "kmsg", + "logPath": "/dev/kmsg", + "lookback": "5m", + "bufferSize": 10, + "source": "kernel-monitor", + "conditions": [ + { + "type": "AMDGPUKernelCrash", + "reason": "NoAMDGPUKernelCrash", + "message": "no AMD GPU kernel crash detected" + } + ], + "rules": [ + { + "type": "temporary", + "reason": "AMDGPUHang", + "pattern": "amdgpu.*GPU hang detected.*" + }, + { + "type": "temporary", + "reason": "AMDGPUReset", + "pattern": "amdgpu.*GPU reset begin.*" + }, + { + "type": "permanent", + "condition": "AMDGPUKernelCrash", + "reason": "AMDGPUPageFault", + "pattern": "amdgpu.*GPU fault detected.*" + }, + { + "type": "permanent", + "condition": "AMDGPUKernelCrash", + "reason": "AMDGPUHangPermanent", + "pattern": "amdgpu.*GPU hang detected.*" + }, + { + "type": "permanent", + "condition": "AMDGPUKernelCrash", + "reason": "AMDGPURASError", + "pattern": "amdgpu.*RAS ERROR.*" + } + ] + } +``` + +```bash +kubectl apply -f npd-dmesg-config.yaml +``` + +### Rule reference + +| `type` | Effect | +|--------|--------| +| `temporary` | Emits a one-shot Kubernetes `Event`; does not change a node condition. | +| `permanent` | Sets the named `condition` to `True` and keeps it set until the node is restarted or the condition is explicitly cleared. Use this when downstream remediation needs to read the condition. | + +| Pattern | What it matches in dmesg | +|---------|--------------------------| +| `amdgpu.*GPU fault detected` | GPU page fault logged by the `amdgpu` kernel driver | +| `amdgpu.*GPU hang detected` | GPU hang / lockup | +| `amdgpu.*GPU reset begin` | Driver-initiated GPU reset | +| `amdgpu.*RAS ERROR` | Uncorrectable RAS error reported to the kernel | + +`lookback: "5m"` tells NPD to replay the last 5 minutes of the ring buffer on startup +so that crashes that happened just before NPD launched are not missed. + +## Step 3 — DaemonSet + +```yaml +# npd-dmesg.yaml +apiVersion: apps/v1 +kind: DaemonSet +metadata: + name: node-problem-detector + namespace: kube-system + labels: + app: node-problem-detector +spec: + selector: + matchLabels: + app: node-problem-detector + template: + metadata: + labels: + app: node-problem-detector + spec: + nodeSelector: + feature.node.kubernetes.io/amd-gpu: "true" + tolerations: + - effect: NoSchedule + operator: Exists + - effect: NoExecute + operator: Exists + serviceAccountName: node-problem-detector + containers: + - name: node-problem-detector + image: registry.k8s.io/node-problem-detector/node-problem-detector:v0.8.19 + command: + - /node-problem-detector + - --logtostderr + - --config.system-log-monitor=/config/kernel-monitor.json + securityContext: + privileged: true + env: + - name: NODE_NAME + valueFrom: + fieldRef: + fieldPath: spec.nodeName + resources: + limits: + cpu: 10m + memory: 80Mi + requests: + cpu: 10m + memory: 80Mi + volumeMounts: + - name: kmsg + mountPath: /dev/kmsg + readOnly: true + - name: localtime + mountPath: /etc/localtime + readOnly: true + - name: config + mountPath: /config + readOnly: true + volumes: + - name: kmsg + hostPath: + path: /dev/kmsg + - name: localtime + hostPath: + path: /etc/localtime + - name: config + configMap: + name: node-problem-detector-config + items: + - key: kernel-monitor.json + path: kernel-monitor.json +``` + +The key difference from the full integration example is: + +- Only `--config.system-log-monitor` is passed — no `--config.custom-plugin-monitor`. +- `/var/log` and the `amdexporter` host path are not mounted (not needed). +- The `/dev/kmsg` mount is read-only; `privileged: true` is still required for NPD to + open the device. + +```bash +kubectl apply -f npd-dmesg.yaml +``` + +## Step 4 — Verify + +```bash +# NPD pods running on GPU nodes +kubectl get pods -n kube-system -l app=node-problem-detector -o wide + +# Stream NPD logs to see pattern matches in real time +kubectl logs -n kube-system -l app=node-problem-detector -f + +# Check the node condition (False = healthy, True = crash detected) +kubectl describe node | sed -n '/Conditions:/,/Addresses:/p' +``` + +When a matching dmesg line appears, the `AMDGPUKernelCrash` condition flips to `True`: + +``` +Conditions: + Type Status ... Reason Message + ---- ------ --- ------ ------- + AMDGPUKernelCrash True ... AMDGPUHangPermanent amdgpu: GPU hang detected ... +``` + +## Next steps + +- To also check GPU ECC and other health metrics, add a `custom-plugin-monitor` config + and the `--config.custom-plugin-monitor` flag as described in + [node-problem-detector.md](node-problem-detector.md). +- To trigger automatic node remediation when a condition fires, see the + [Auto Node Remediation](../autoremediation/auto-remediation.md) documentation. From 90840a6a9ec9fbd0bb85f6e557723bb3261b0d86 Mon Sep 17 00:00:00 2001 From: praveen Date: Thu, 24 Sep 2026 11:39:52 -0700 Subject: [PATCH 2/4] docs(npd): rewrite dmesg example to integrate with GPU Operator auto-remediation Update npd-dmesg-example.md to use the operator-integrated NPD pattern: - dmesg system-log-monitor runs alongside amdgpuhealth custom plugin monitor - AMDGPUKernelCrash permanent condition wires into remediation ConfigMap - DaemonSet includes amd-gpu-unhealthy:NoSchedule toleration required by auto-remediation - Added remediation ConfigMap snippet showing nodeCondition alignment Add npd-dmesg-example to _toc.yml under Node Problem Detector section. Add new spellcheck wordlist entries for GPU condition/config terms used in the doc. Co-Authored-By: Claude --- .wordlist.txt | 849 +++++++++++++++++----------------- docs/npd/npd-dmesg-example.md | 240 ++++++---- docs/sphinx/_toc.yml | 2 + 3 files changed, 580 insertions(+), 511 deletions(-) diff --git a/.wordlist.txt b/.wordlist.txt index 504ac321b..e0de08f44 100644 --- a/.wordlist.txt +++ b/.wordlist.txt @@ -1,242 +1,31 @@ -amd +AAC +ABI +ACF +ACS AFID AFIDs -Affectioned -acf -ACF AGFHC -Allocatable -ACS +AINIC AKS -ARI -Autobuild -bb -burnin -CDI -CEL -CheckUnitStatus -CleanupPreState -CLI -ClusterRole -CN -CNI -computePartition -ConfigMap -ConfigMaps -ConditionalWorkflows -containerd -CoreOS -CPX -CrashLoopBackOff -CRD -CRDs -CRI -CRs -CronJob -Customizable -CustomResourceDefinition -daemonset -daemonsets -DaemonSet -Daemonsets -DaemonSets -DCM -dcm -Depricated -deivce -DeviceClass -DeviceConfig -DeviceIDs -DevicePlugin -DevicePluginArguments -DevicePluginImage -DevicePluginImagePullPolicy -DevicePluginSpec -DKMS -dma -DMC -DME -DNS -Dockerfile -DockerHub -DPX -DRA -DRADriver -DriverToolkit -ECC -enableDevicePlugin -EnableDevicePlugin -EnableNodeLabeller -ErrImagePull -flannel -GitOps -GPUs -gpup -Grafana -GracePeriodSeconds -gst -gpuagent -gpuClientSystemdServices -GKE -hbm -HealthThresholds -Helmify -hostname -hostnames -HSIO -HTTPS -iet -imageRegistrySecrets -IfNotPresent -IgnoreDaemonSets -IgnoreNamespaces -ImageStream -jq -json -kaniko -KMM -kmod -kubectl -Kubelet -KubeVirt -Kuberntes -Kubernetes -kubeconfig -labeller -Labeler -lifecycle -lvl -MachineConfig -MachineConfigOperator -MCO -Mericsclient -MaxParallelWorkflows -MaxUnavailable -MCO -memoryPartition -MetricsExporter -MetricsExporterSpec -MinIO -Minio -MOK -MTLS -namespace -NFD -NMC -NodeCondition -NodeDrainPolicy -NodeIP -NodeLabeller -NodeLabellerArguments -NodeLabellerImage -NodeLabellerImagePullPolicy -Nodelabeller -nodename -Nodeport -NodePort -NodeRemediationLabels -NodeRemediationTaints -NoExecute -NPD -NPD's -NotReady -numGPUsAssigned -Observability -oc -OCI -OLM -OOM -OpenShift -OperatorHub -Openshift -parition -paritioning -pbqt -pebb -PCI -pcie -perf -PFs -Perses -plugin -PodIP -PreFlight -PreStateDB -prometheus -Promethues -quay -QEMU -QPX -RAS -RBAC -Redhat -RedHat -ResourceSlices -RHCOS -RMA -rocminfo -rochpl -ROCm -runtime -runtime's -SAR -schedulable -SDK -selfcheck -ServiceAccounts -ServiceMonitor -ServiecMonitor -skippedGPUs -Slinkproject -SlinkProject -Slrum -Slurm -SPX -StopOnFailure -SubjectAccessReview -systemd -TestCategory -TesterImage -TimeoutSeconds -TokenReview -Tolerations -TODO -TLS -tolerations -tst -TTL -TtlForFailedWorkflows -ubuntu -UI -UID -UNCORRECT -Uncordoning -uninstallation -unschedulable -Upgrademgr -UpgradePolicy -validation -verison -VC -VCN -VFIO -VFs -VMs -webhook -xgmi -YAML -AAC -ABI ALU AMD AMDGPU +AMDGPUHang +AMDGPUHangPermanent +AMDGPUKernelCrash +AMDGPUPageFault +AMDGPURASError +AMDGPUReset +AMDGPUUnhealthy AMDGPUs AMDMIGraphX AMI +ANR AOCC AOMP APIC APIs +ARI ASIC ASICs ASan @@ -245,29 +34,44 @@ ATI AWQ AdaLoRA AddressSanitizer +Affectioned AlexNet +Allocatable Arb AutoAWQ AutoGPTQ +AutoStartWorkflow +Autobuild +BDFs BLAS BMC BitCode Blit Bluefield CCD +CDI CDNA +CEL CIFAR CLI CLion CMake CMakeLists CMakePackage +CN +CNI CP CPC +CPER CPF CPP CPU CPUs +CPX +CRD +CRDs +CRI +CRs CSC CSE CSV @@ -276,46 +80,84 @@ CTests CU CUDA CUs +CVE CXX Cavium CentOS ChatGPT +CheckUnitStatus +CleanupPreState +ClusterRole CoRR Codespaces Commitizen +CommonConfig CommonMark Concretized Conda +ConditionalWorkflows +ConfigMap +ConfigMaps ConnectX +CoreOS +CrashLoopBackOff +CronJob +CustomResourceDefinition +Customizable +DCM DDP DGEMM DKMS DL DLM DMA +DMC +DME DNN DNNL +DNS DPM +DPX +DRA +DRADriver DRI DW DWORD +DaemonSet +DaemonSets +Daemonsets Dask DataFrame DataLoader DataParallel DeepSpeed Dependabot +Depricated DevCap +DeviceClass +DeviceConfig +DeviceIDs +DevicePlugin +DevicePluginArguments +DevicePluginImage +DevicePluginImagePullPolicy +DevicePluginSpec Diffusers +DockerHub Dockerfile Dockerfiles Doxygen +DriverToolkit +ECC ELMo ENDPGM EPEL EPYC ESXi EU +EnableDevicePlugin +EnableNodeLabeller +ErrImagePull ExLlama FFT FFTs @@ -340,6 +182,7 @@ GEMM GEMMs GFortran GIM +GKE GL GLXT GMI @@ -352,10 +195,14 @@ GPU's GPUs GQA GRBM +GRE GenAI GenZ GitHub +GitOps Gitpod +GracePeriodSeconds +Grafana HBM HCA HIPCC @@ -366,8 +213,12 @@ HPCG HPE HPL HSA +HSIO +HTTPS HWE Haswell +HealthThresholds +Helmify Higgs Hyperparameters ICV @@ -378,11 +229,16 @@ IOMMU IOP IOPM IOV +IPC IRQ ISA ISV ISVs +IfNotPresent +IgnoreDaemonSets +IgnoreNamespaces ImageNet +ImageStream InfiniBand Inlines IntelliSense @@ -394,10 +250,15 @@ JIT JSON Jupyter KFD +KMM KVM Keras Khronos +Kube +KubeVirt +Kubelet Kubernetes +Kuberntes LAPACK LCLK LDS @@ -408,8 +269,10 @@ LM LSAN LSTM LTS +Labeler LinearReLU LoRA +MCO MEM MERCHANTABILITY MFMA @@ -425,18 +288,30 @@ MMA MMIO MMIOH MNIST +MOK MPI MQA MSVC +MTLS MVAPICH MVFFR +MachineConfig +MachineConfigOperator Makefile Makefiles Matplotlib +MaxParallelWorkflows +MaxUnavailable Megatron Mellanox Mellanox's +Mericsclient Meta's +MetricsExporter +MetricsExporterSpec +MicroK +MinIO +Minio MirroredStrategy MoE Multicore @@ -445,11 +320,15 @@ MyEnvironment MyST NBIO NBIOs +NFD NHWC NIC NICs NLI NLP +NMC +NPD +NPD's NPS NSP NUMA @@ -458,24 +337,44 @@ NVIDIA NVPTX Nano Navi -Noncoherently +NoAMDGPUKernelCrash +NoExecute NoSchedule +NodeCondition +NodeDrainPolicy +NodeIP +NodeLabeller +NodeLabellerArguments +NodeLabellerImage +NodeLabellerImagePullPolicy +NodePort +NodeRemediationLabels +NodeRemediationTaints +Nodelabeller +Nodeport +Noncoherently +NotReady NousResearch's NumPy OAM OAMs +OCI OCP OEM OFED +OLM OMP OMPI OMPT OMPX ONNX +OOM OSS OSU +Observability Omniperf Omnitrace +OnDelete OpenAI OpenCL OpenCV @@ -483,10 +382,15 @@ OpenFabrics OpenGL OpenMP OpenSSL +OpenShift +OpenShift's OpenVX +Openshift +OperatorHub PCI PCIe PEFT +PFs PIL PILImage PPO @@ -496,19 +400,30 @@ PaLM Pageable PeerDirect Perfetto +Perses PipelineParallel PnP +PodIP PowerShell +PreFlight +PreStateDB +Promethues PyPi PyTorch +QEMU QLoRA +QPX Qcycles RAII +RAS +RBAC RCCL RDC RDMA RDNA +RHCOS RHEL +RMA RNN ROC ROCProfiler @@ -527,13 +442,18 @@ RST RW Radeon ReLU +RedHat +Redhat RelWithDebInfo Req +ResourceSlices Rickle RoCE +RollingUpdate Roofline Ryzen SALU +SAR SBIOS SCA SDK @@ -554,6 +474,7 @@ SMEM SMI SMT SPI +SPX SQs SRAM SRAMECC @@ -561,12 +482,21 @@ SVD SWE SciPy SerDes +ServiceAccounts +ServiceMonitor +ServiecMonitor Shlens Skylake +SlinkProject +Slinkproject +Slrum +Slurm SmoothQuant Softmax Spack StarCoder +StopOnFailure +SubjectAccessReview Supermicro Szegedy TCA @@ -577,6 +507,8 @@ TCP TCR TFLOPS TGI +TLS +TODO TPOT TPU TPUs @@ -584,11 +516,18 @@ TRL TTFT TTGIR TTIR +TTL +Techsupport Templated TensorBoard TensorFlow TensorParallel +TestCategory +TesterImage +TimeoutSeconds ToC +TokenReview +Tolerations TorchAudio TorchInductor TorchMIGraphX @@ -597,27 +536,43 @@ TorchServe TorchVision TransferBench TrapStatus +TtlForFailedWorkflows Tunable TunableOp UAC UC UCC UCX +UI +UID UIF +UMC +UNCORRECT URI USM UTCL UTIL Uncached +Uncordoning +Uncorrectable Unhandled +UpgradePolicy +UpgradeStrategy +Upgrademgr +UtilsContainer VALU VBIOS +VC +VCN +VFIO +VFs VGPR VGPRs VGPU VM VMEM VMWare +VMs VRAM VSIX VSkipped @@ -646,40 +601,64 @@ YModel ZeRO ZenDNN accuracies +acf activations addr alloc +allocatable allocator allocators +allowPrivilegeEscalation +amd amdgpu +amdgpuhealth api +apiVersion +apiserver +args atmi atomics +attachMetadata +autobuild autogenerated autoregression autoregressive avx awk +aws backend backends backpropagation backtick +bb +bd +bearerTokenFile benchmarking +bh bilinear bitsandbytes blit +bool boson bosons +bufferSize buildable +burnin bursty bzip +caFile cacheable cd centos centric +certFile changelog +checkmark chiplet ckProfiler +clientCAConfigMap +clientName +clusterIP cmake cmd coalescable @@ -691,12 +670,26 @@ comgr completers composability composable +computePartition concretization config +configmap +configs +configurability conformant +containerSecurityContext +containerd +controllerConfigYaml +controllerManager +controllerMetricsService convolutional convolves cpp +cpu +cpx +crds +cron +cryptographic csn cuBLAS cuFFT @@ -705,6 +698,8 @@ cuRAND cuSOLVER cuSPARSE customizations +daemonset +daemonsets dataset dataset's datasets @@ -712,8 +707,11 @@ dataspace datatype datatypes dbgapi +dcd +dcm de deallocation +deivce denoise denoised denoises @@ -721,15 +719,29 @@ denormalize deserializers detections dev +devel +devicePlugin +devicePluginImage +devicePluginSpec +deviceconfig +deviceconfigs devicelibs devsel dimensionality +disableHttps disambiguates +discoverable distro +dkms +dma +dmesg doxysphinx dropdown el embeddings +emptyDir +enableDevicePlugin +enableNodeLabeller enablement endpgm env @@ -739,19 +751,38 @@ ethernet exascale executables ffmpeg +fieldPath +fieldRef filesystem finalizer +flannel fortran galb +gapped +gc gcc gdb +generationID gfortran gfx githooks github gnupg +gocheck +gpu +gpu's +gpuClientSystemdServices +gpuagent +gpup +gpus +grafana grayscale +grpc +gst +gz gzip +hbm +hcrxm heterogenous hipBLAS hipBLASLt @@ -771,21 +802,33 @@ hipfort hipify hipsolver hipsparse +honorLabels +honorTimestamps +hostname +hostnames hpp hsa hsakmt +hsio html hyperparameter ib_core +iet +imagePullPolicy +imagePullSecrets +imageRegistrySecrets inband incrementing inferencing inflight init initializer +initramfs inlining -installable +insecureSkipVerify installCRDs +installable +installdefaultNFDRule instantiation interprocedural intersphinx @@ -793,49 +836,123 @@ intra invariants invocating invoker +io ipo +isigned +jlzbs +jq +json +kaniko kdb +keyFile +keySecret +kfd +kmm +kmod +kmsg +kube +kubeconfig +kubectl +kubelet +kubernetes +kubernetesClusterDomain +labeller +labeller's libfabric libjpeg libs +lifecycle linearized linter linux llvm localscratch +localtime +logPath logits +lookback lossy +lvl +mTLS macOS +managerConfig +matchLabels matchers +maxParallel +md +mem +memoryPartition +metricsExporter +metricsclient microarchitecture migraphx miopen miopengemm +misconfigurations mivisionx mkdir +mkdocs mlirmiopen +modprobe +mortem mtypes mvffr myst +namesapace namespace +namespaced namespaces +nano natively +nm +nmc +nodeAffinity +nodeCondition +nodeLabellerImage +nodeName +nodePort +nodeSelector +nodeSelectorTerms +nodelabeller +nodename +notifyRemediationMessage +notifyTestFailureMessage +numGPUsAssigned +numa numref +observability +oc ocl +onwards opencl opencv openmp +openshift openssl optimizers os +osImage +oyaml pageable parallelization parallelize parameterization +parition +paritioning passthrough +pbqt +pci +pcie +pebb +peqt +perf perfcounter performant perl +pesm +physicalActionNeeded +plugin +podman pragma pre prebuilt @@ -849,22 +966,42 @@ preprocessing prequantized prerequisites profiler +programmatically +prometheus protobuf pseudorandom py +pytest quantized quantizing quasirandom +quay queueing +ras +rbac rccl +rcqt rdc reStructuredText +readded +rebootRequired +reconfiguring +recoveryPolicy reformats +relatedImageBuild +relatedImageBuildPullSecret +relatedImageSign +relatedImageSignPullSecret +relatedImageWorker +relatedImageWorkerPullSecret +remediations +repo repos representativeness req resampling rescaling +retorquing reusability roadmap roc @@ -872,6 +1009,7 @@ rocAL rocALUTION rocBLAS rocFFT +rocHPL rocLIB rocMLIR rocPRIM @@ -884,6 +1022,7 @@ rocalution rocblas rocclr rocfft +rochpl rocm rocminfo rocprim @@ -895,23 +1034,44 @@ rocsolver rocsparse rocthrust roctracer +rollout runtime +runtime's runtimes sL scalability scalable +schedulable +searchability +securityContext +selfcheck sendmsg serializers +serverName +serviceAccount +serviceAccountName +serviceAccountNamespaceSelector +serviceAccountSelector +serviceType shader sharded sharding sigmoid +signimage +simd +skipRebootStep +skippedGPUs +slurm sm smi softmax spack +spx src +staticAuthorization +stdout stochastically +stopOnFailure strided struct subdirectories @@ -921,13 +1081,26 @@ subfolder subfolders suboptimal supercomputing +svc +symlinks +sys +sysfs +sysmon +systemd +targetPort +teardown +techsupport templated +testRunner th +timeoutSeconds +tlsConfig tokenization tokenize tokenized tokenizer tokenizes +tolerations toolchain toolchains toolset @@ -936,24 +1109,37 @@ torchtune torchvision tqdm tracebacks +tst tunable tunings txt uarch +ubuntu +un unallocated uncached +uncordoned uncorrectable uninstallation +unpartitioned +unschedulable unsqueeze unstacking unswitching untrusted untuned +upgradeCRD +upgradePolicy +upgradeStrategy upstreamed upvote +url utils vL vLLM +validation +validationTestsProfile +valueFrom variational vdi vectorizable @@ -962,12 +1148,23 @@ vectorize vectorized vectorizer vectorizes +verison +vf +vfio +virtfn +virtualized vjxb +vram walkthrough walkthroughs wavefront wavefronts +webhook +webhook's +webhookServer +webhookService whitespaces +workflowTemplate workgroup workgroups writeback @@ -975,202 +1172,12 @@ writebacks wrreq wzo xFormers +xGMI xargs +xgmi +xtwbm xz yaml +yamls ysvmadyb zyppe -CommonConfig -Kube -OnDelete -OpenShift's -RollingUpdate -Techsupport -UpgradeStrategy -UtilsContainer -AutoStartWorkflow -allocatable -allowPrivilegeEscalation -amdgpuhealth -apiserver -apiVersion -args -attachMetadata -autobuild -aws -BDFs -bd -bearerTokenFile -bool -caFile -certFile -checkmark -clientCAConfigMap -clientName -clusterIP -configmap -configs -containerSecurityContext -controllerConfigYaml -controllerManager -controllerMetricsService -cpu -cpx -crds -cron -cryptographic -devel -devicePlugin -devicePluginImage -devicePluginSpec -deviceconfig -deviceconfigs -disableHttps -discoverable -dkms -dmesg -emptyDir -enableNodeLabeller -gapped -generationID -gpu -gpu's -gpus -grafana -grpc -honorLabels -honorTimestamps -hsio -imagePullPolicy -imagePullSecrets -initramfs -insecureSkipVerify -installdefaultNFDRule -io -isigned -kfd -keyFile -keySecret -kmm -kube -kubernetesClusterDomain -kubelet -kubernetes -labeller's -mTLS -managerConfig -maxParallel -metricsclient -metricsExporter -misconfigurations -mkdocs -modprobe -mortem -namesapace -namespaced -nano -nmc -nodeAffinity -nodeCondition -nodeLabellerImage -nodePort -nodeSelector -nodeSelectorTerms -nodelabeller -notifyRemediationMessage -notifyTestFailureMessage -numa -observability -onwards -openshift -osImage -oyaml -pci -physicalActionNeeded -podman -programmatically -pytest -ras -rbac -readded -rebootRequired -recoveryPolicy -relatedImageBuild -relatedImageBuildPullSecret -relatedImageSign -relatedImageSignPullSecret -relatedImageWorker -relatedImageWorkerPullSecret -repo -retorquing -rocHPL -rollout -searchability -serverName -serviceAccount -serviceAccountNamespaceSelector -serviceAccountSelector -serviceType -signimage -simd -skipRebootStep -slurm -spx -staticAuthorization -stdout -svc -symlinks -sys -sysfs -sysmon -targetPort -techsupport -testRunner -tlsConfig -un -uncordoned -unpartitioned -upgradeCRD -upgradePolicy -upgradeStrategy -url -validationTestsProfile -vf -vfio -virtfn -virtualized -vram -webhook's -webhookServer -webhookService -workflowTemplate -xGMI -yamls -sysfs -BDFs -bh -dcd -gc -gz -hcrxm -jlzbs -nm -xtwbm -OCI -gocheck -teardown -DME -reconfiguring -AINIC -ANR -CVE -IPC -MicroK -configurability -remediations -GRE -mem -peqt -pesm -rcqt -CPER diff --git a/docs/npd/npd-dmesg-example.md b/docs/npd/npd-dmesg-example.md index ab2576057..9a608f8a6 100644 --- a/docs/npd/npd-dmesg-example.md +++ b/docs/npd/npd-dmesg-example.md @@ -1,71 +1,90 @@ -# NPD dmesg-Only Example +# NPD dmesg Kernel-Crash Detection Example -This page shows the minimal configuration to use Node Problem Detector to watch the -kernel ring buffer (`/dev/kmsg`) for AMD GPU crashes. No AMD Device Metrics Exporter -or `amdgpuhealth` binary is required — this example uses only NPD's built-in -**system-log monitor**. +This page shows how to extend the Node Problem Detector (NPD) configuration to watch +the kernel ring buffer (`/dev/kmsg`) for AMD GPU crash patterns, alongside the standard +`amdgpuhealth` custom plugin monitor. The dmesg rules emit permanent node conditions +that the GPU Operator's auto-remediation controller can act on. -## How it works - -NPD reads `/dev/kmsg` continuously. When a log line matches a regex rule, NPD either -emits a one-shot `Event` (`type: temporary`) or flips a persistent `NodeCondition` -to `True` (`type: permanent`). The conditions are visible via `kubectl describe node`. +```{note} +This example extends the setup described in +[Node Problem Detector Integration](node-problem-detector.md). +Complete that setup first — RBAC, AMD Device Metrics Exporter, and the base DaemonSet +must all be in place before adding dmesg monitoring. +``` -## Step 1 — RBAC +## How dmesg monitoring works with the GPU Operator -NPD needs permission to patch `nodes/status` (to write conditions) and create `events`. -No non-resource URL permissions are needed for the dmesg-only setup. +NPD's `system-log-monitor` reads `/dev/kmsg` and matches log lines against regex rules. +When a line matches a `permanent` rule, NPD sets the named node condition to `True`. +The GPU Operator's remediation controller watches node conditions and triggers an Argo +workflow when it sees a condition that matches a `nodeCondition` entry in the +remediation ConfigMap. -```yaml -# npd-rbac.yaml -apiVersion: v1 -kind: ServiceAccount -metadata: - name: node-problem-detector - namespace: kube-system ---- -apiVersion: rbac.authorization.k8s.io/v1 -kind: ClusterRole -metadata: - name: node-problem-detector -rules: -- apiGroups: [""] - resources: ["nodes/status"] - verbs: ["patch"] -- apiGroups: [""] - resources: ["events"] - verbs: ["create", "patch"] ---- -apiVersion: rbac.authorization.k8s.io/v1 -kind: ClusterRoleBinding -metadata: - name: node-problem-detector -roleRef: - apiGroup: rbac.authorization.k8s.io - kind: ClusterRole - name: node-problem-detector -subjects: -- kind: ServiceAccount - name: node-problem-detector - namespace: kube-system ``` - -```bash -kubectl apply -f npd-rbac.yaml +/dev/kmsg (kernel ring buffer) + │ + ▼ +NPD system-log-monitor ──► NodeCondition = True (e.g. AMDGPUKernelCrash) + │ + ▼ +GPU Operator remediation controller ──► Argo Workflow ``` -## Step 2 — ConfigMap +## Step 1 — Add dmesg rules to the NPD ConfigMap + +Extend the existing `node-problem-detector-config` ConfigMap with a second key, +`kernel-monitor.json`. The `system-log-monitor` plugin reads this file. -The `kernel-monitor.json` key defines which dmesg patterns to watch. +The condition name (`AMDGPUKernelCrash` in the example below) must match the +`nodeCondition` field in the GPU Operator remediation ConfigMap so the operator +knows which workflow to trigger. ```yaml -# npd-dmesg-config.yaml +# node-problem-detector-config.yaml (extended) apiVersion: v1 kind: ConfigMap metadata: name: node-problem-detector-config namespace: kube-system data: + # Existing key — custom plugin monitor for amdgpuhealth metric checks + custom-plugin-monitor.json: | + { + "plugin": "custom", + "pluginConfig": { + "invoke_interval": "30s", + "timeout": "15s", + "max_output_length": 80, + "concurrency": 3, + "enable_message_change_based_condition_update": false + }, + "source": "amdgpu-custom-plugin-monitor", + "metricsReporting": true, + "conditions": [ + { + "type": "AMDGPUUnhealthy", + "reason": "AMDGPUIsUp", + "message": "AMDGPU is up" + } + ], + "rules": [ + { + "type": "permanent", + "condition": "AMDGPUUnhealthy", + "reason": "AMDGPUIsDown", + "path": "/var/lib/amd-metrics-exporter/amdgpuhealth", + "args": [ + "query", + "counter-metric", + "-m=GPU_ECC_UNCORRECT_UMC", + "-t=1" + ], + "timeout": "15s" + } + ] + } + + # New key — system-log monitor for dmesg / kmsg GPU crash patterns kernel-monitor.json: | { "plugin": "kmsg", @@ -86,11 +105,6 @@ data: "reason": "AMDGPUHang", "pattern": "amdgpu.*GPU hang detected.*" }, - { - "type": "temporary", - "reason": "AMDGPUReset", - "pattern": "amdgpu.*GPU reset begin.*" - }, { "type": "permanent", "condition": "AMDGPUKernelCrash", @@ -103,6 +117,12 @@ data: "reason": "AMDGPUHangPermanent", "pattern": "amdgpu.*GPU hang detected.*" }, + { + "type": "permanent", + "condition": "AMDGPUKernelCrash", + "reason": "AMDGPUReset", + "pattern": "amdgpu.*GPU reset begin.*" + }, { "type": "permanent", "condition": "AMDGPUKernelCrash", @@ -113,31 +133,30 @@ data: } ``` -```bash -kubectl apply -f npd-dmesg-config.yaml -``` - ### Rule reference | `type` | Effect | |--------|--------| -| `temporary` | Emits a one-shot Kubernetes `Event`; does not change a node condition. | -| `permanent` | Sets the named `condition` to `True` and keeps it set until the node is restarted or the condition is explicitly cleared. Use this when downstream remediation needs to read the condition. | +| `temporary` | Emits a one-shot Kubernetes `Event`. Does not flip a node condition. | +| `permanent` | Sets the named `condition` to `True` and keeps it set. Use this for conditions the remediation controller watches. | | Pattern | What it matches in dmesg | |---------|--------------------------| | `amdgpu.*GPU fault detected` | GPU page fault logged by the `amdgpu` kernel driver | -| `amdgpu.*GPU hang detected` | GPU hang / lockup | +| `amdgpu.*GPU hang detected` | GPU hang or lockup | | `amdgpu.*GPU reset begin` | Driver-initiated GPU reset | | `amdgpu.*RAS ERROR` | Uncorrectable RAS error reported to the kernel | -`lookback: "5m"` tells NPD to replay the last 5 minutes of the ring buffer on startup -so that crashes that happened just before NPD launched are not missed. +`lookback: "5m"` replays the last 5 minutes of the ring buffer on startup, so crashes +that occurred just before NPD launched are not missed. -## Step 3 — DaemonSet +## Step 2 — Add the system-log monitor to the DaemonSet + +Add `--config.system-log-monitor` and mount `/dev/kmsg`. The existing +`--config.custom-plugin-monitor` flag and all other mounts stay unchanged. ```yaml -# npd-dmesg.yaml +# node-problem-detector.yaml (extended) apiVersion: apps/v1 kind: DaemonSet metadata: @@ -157,6 +176,10 @@ spec: nodeSelector: feature.node.kubernetes.io/amd-gpu: "true" tolerations: + # Required: keeps NPD running on nodes tainted by auto-remediation + - key: amd-gpu-unhealthy + operator: Exists + effect: NoSchedule - effect: NoSchedule operator: Exists - effect: NoExecute @@ -168,7 +191,10 @@ spec: command: - /node-problem-detector - --logtostderr + # dmesg / kernel ring buffer monitoring - --config.system-log-monitor=/config/kernel-monitor.json + # amdgpuhealth metric monitoring + - --config.custom-plugin-monitor=/config/custom-plugin-monitor.json securityContext: privileged: true env: @@ -178,12 +204,14 @@ spec: fieldPath: spec.nodeName resources: limits: - cpu: 10m - memory: 80Mi + cpu: 20m + memory: 100Mi requests: cpu: 10m memory: 80Mi volumeMounts: + - name: log + mountPath: /var/log - name: kmsg mountPath: /dev/kmsg readOnly: true @@ -193,7 +221,12 @@ spec: - name: config mountPath: /config readOnly: true + - name: amdexporter + mountPath: /var/lib/amd-metrics-exporter volumes: + - name: log + hostPath: + path: /var/log/ - name: kmsg hostPath: path: /dev/kmsg @@ -204,47 +237,74 @@ spec: configMap: name: node-problem-detector-config items: + - key: custom-plugin-monitor.json + path: custom-plugin-monitor.json - key: kernel-monitor.json path: kernel-monitor.json + - name: amdexporter + hostPath: + path: /var/lib/amd-metrics-exporter +``` + +```{important} +The `amd-gpu-unhealthy:NoSchedule` toleration is required. When auto-remediation taints +a node to evict workloads, NPD must keep running so its final condition check can +confirm that the node has recovered. Without this toleration, NPD is evicted and the +remediation workflow gets stuck waiting for the condition to flip back to `False`. ``` -The key difference from the full integration example is: +## Step 3 — Wire the condition into the remediation ConfigMap -- Only `--config.system-log-monitor` is passed — no `--config.custom-plugin-monitor`. -- `/var/log` and the `amdexporter` host path are not mounted (not needed). -- The `/dev/kmsg` mount is read-only; `privileged: true` is still required for NPD to - open the device. +Add an entry for `AMDGPUKernelCrash` to the GPU Operator remediation ConfigMap so the +operator knows which Argo workflow to run when NPD sets that condition. -```bash -kubectl apply -f npd-dmesg.yaml +```yaml +remediation: + - nodeCondition: AMDGPUKernelCrash + workflowTemplate: default-template + validationTestsProfile: + framework: AGFHC + recipe: all_lvl4 + iterations: 1 + stopOnFailure: true + timeoutSeconds: 4800 + physicalActionNeeded: false + skipRebootStep: false ``` -## Step 4 — Verify +See the [Auto Node Remediation](../autoremediation/auto-remediation.md) documentation +for the full remediation ConfigMap schema and available fields. + +## Step 4 — Apply and verify ```bash -# NPD pods running on GPU nodes +kubectl apply -f node-problem-detector-config.yaml +kubectl rollout restart daemonset/node-problem-detector -n kube-system + +# Confirm NPD pods are running on GPU nodes kubectl get pods -n kube-system -l app=node-problem-detector -o wide -# Stream NPD logs to see pattern matches in real time +# Stream NPD logs to see both monitors active kubectl logs -n kube-system -l app=node-problem-detector -f -# Check the node condition (False = healthy, True = crash detected) +# Check node conditions — both AMDGPUUnhealthy and AMDGPUKernelCrash should appear kubectl describe node | sed -n '/Conditions:/,/Addresses:/p' ``` -When a matching dmesg line appears, the `AMDGPUKernelCrash` condition flips to `True`: +When healthy, both conditions are `False`: ``` Conditions: - Type Status ... Reason Message - ---- ------ --- ------ ------- - AMDGPUKernelCrash True ... AMDGPUHangPermanent amdgpu: GPU hang detected ... + Type Status Reason Message + ---- ------ ------ ------- + AMDGPUUnhealthy False AMDGPUIsUp AMDGPU is up + AMDGPUKernelCrash False NoAMDGPUKernelCrash no AMD GPU kernel crash detected ``` -## Next steps +When a matching dmesg line appears, `AMDGPUKernelCrash` flips to `True` and the GPU +Operator triggers an Argo workflow: -- To also check GPU ECC and other health metrics, add a `custom-plugin-monitor` config - and the `--config.custom-plugin-monitor` flag as described in - [node-problem-detector.md](node-problem-detector.md). -- To trigger automatic node remediation when a condition fires, see the - [Auto Node Remediation](../autoremediation/auto-remediation.md) documentation. +```bash +kubectl get workflows -A +kubectl get events -A --field-selector reason=amd-gpu-remediation-required +``` diff --git a/docs/sphinx/_toc.yml b/docs/sphinx/_toc.yml index 12b638378..67d4f9827 100644 --- a/docs/sphinx/_toc.yml +++ b/docs/sphinx/_toc.yml @@ -75,6 +75,8 @@ subtrees: - caption: Node Problem Detector entries: - file: npd/node-problem-detector + - file: npd/npd-dmesg-example + title: dmesg Kernel-Crash Detection Example - caption: Auto Remediation entries: - file: autoremediation/auto-remediation From 7e9a1c970744c502267d43afd2200e2c5ff79220 Mon Sep 17 00:00:00 2001 From: praveen Date: Thu, 24 Sep 2026 11:52:33 -0700 Subject: [PATCH 3/4] docs(npd): fix MD040 bare fenced code blocks Add 'text' language tag to the ASCII flow diagram and kubectl output blocks to satisfy markdownlint MD040 (fenced code blocks must have a language specified). Co-Authored-By: Claude --- docs/npd/npd-dmesg-example.md | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/docs/npd/npd-dmesg-example.md b/docs/npd/npd-dmesg-example.md index 9a608f8a6..9d452b168 100644 --- a/docs/npd/npd-dmesg-example.md +++ b/docs/npd/npd-dmesg-example.md @@ -20,7 +20,7 @@ The GPU Operator's remediation controller watches node conditions and triggers a workflow when it sees a condition that matches a `nodeCondition` entry in the remediation ConfigMap. -``` +```text /dev/kmsg (kernel ring buffer) │ ▼ @@ -293,7 +293,7 @@ kubectl describe node | sed -n '/Conditions:/,/Addresses:/p' When healthy, both conditions are `False`: -``` +```text Conditions: Type Status Reason Message ---- ------ ------ ------- From c9cf10e51cdc4e7d9b678d58f621091b3174ac2e Mon Sep 17 00:00:00 2001 From: praveen Date: Thu, 24 Sep 2026 12:02:31 -0700 Subject: [PATCH 4/4] docs(npd): fix MD060 table column alignment Pad table separator rows and cell content to align all pipes, satisfying markdownlint MD060 (table-column-style aligned). Co-Authored-By: Claude --- docs/npd/npd-dmesg-example.md | 20 ++++++++++---------- 1 file changed, 10 insertions(+), 10 deletions(-) diff --git a/docs/npd/npd-dmesg-example.md b/docs/npd/npd-dmesg-example.md index 9d452b168..bbd32493c 100644 --- a/docs/npd/npd-dmesg-example.md +++ b/docs/npd/npd-dmesg-example.md @@ -135,17 +135,17 @@ data: ### Rule reference -| `type` | Effect | -|--------|--------| -| `temporary` | Emits a one-shot Kubernetes `Event`. Does not flip a node condition. | -| `permanent` | Sets the named `condition` to `True` and keeps it set. Use this for conditions the remediation controller watches. | +| `type` | Effect | +|--------------|---------------------------------------------------------------------------------------------------------------------| +| `temporary` | Emits a one-shot Kubernetes `Event`. Does not flip a node condition. | +| `permanent` | Sets the named `condition` to `True` and keeps it set. Use this for conditions the remediation controller watches. | -| Pattern | What it matches in dmesg | -|---------|--------------------------| -| `amdgpu.*GPU fault detected` | GPU page fault logged by the `amdgpu` kernel driver | -| `amdgpu.*GPU hang detected` | GPU hang or lockup | -| `amdgpu.*GPU reset begin` | Driver-initiated GPU reset | -| `amdgpu.*RAS ERROR` | Uncorrectable RAS error reported to the kernel | +| Pattern | What it matches in dmesg | +|-------------------------------|-------------------------------------------------------| +| `amdgpu.*GPU fault detected` | GPU page fault logged by the `amdgpu` kernel driver | +| `amdgpu.*GPU hang detected` | GPU hang or lockup | +| `amdgpu.*GPU reset begin` | Driver-initiated GPU reset | +| `amdgpu.*RAS ERROR` | Uncorrectable RAS error reported to the kernel | `lookback: "5m"` replays the last 5 minutes of the ring buffer on startup, so crashes that occurred just before NPD launched are not missed.