diff --git a/.wordlist.txt b/.wordlist.txt index 504ac321b..e0de08f44 100644 --- a/.wordlist.txt +++ b/.wordlist.txt @@ -1,242 +1,31 @@ -amd +AAC +ABI +ACF +ACS AFID AFIDs -Affectioned -acf -ACF AGFHC -Allocatable -ACS +AINIC AKS -ARI -Autobuild -bb -burnin -CDI -CEL -CheckUnitStatus -CleanupPreState -CLI -ClusterRole -CN -CNI -computePartition -ConfigMap -ConfigMaps -ConditionalWorkflows -containerd -CoreOS -CPX -CrashLoopBackOff -CRD -CRDs -CRI -CRs -CronJob -Customizable -CustomResourceDefinition -daemonset -daemonsets -DaemonSet -Daemonsets -DaemonSets -DCM -dcm -Depricated -deivce -DeviceClass -DeviceConfig -DeviceIDs -DevicePlugin -DevicePluginArguments -DevicePluginImage -DevicePluginImagePullPolicy -DevicePluginSpec -DKMS -dma -DMC -DME -DNS -Dockerfile -DockerHub -DPX -DRA -DRADriver -DriverToolkit -ECC -enableDevicePlugin -EnableDevicePlugin -EnableNodeLabeller -ErrImagePull -flannel -GitOps -GPUs -gpup -Grafana -GracePeriodSeconds -gst -gpuagent -gpuClientSystemdServices -GKE -hbm -HealthThresholds -Helmify -hostname -hostnames -HSIO -HTTPS -iet -imageRegistrySecrets -IfNotPresent -IgnoreDaemonSets -IgnoreNamespaces -ImageStream -jq -json -kaniko -KMM -kmod -kubectl -Kubelet -KubeVirt -Kuberntes -Kubernetes -kubeconfig -labeller -Labeler -lifecycle -lvl -MachineConfig -MachineConfigOperator -MCO -Mericsclient -MaxParallelWorkflows -MaxUnavailable -MCO -memoryPartition -MetricsExporter -MetricsExporterSpec -MinIO -Minio -MOK -MTLS -namespace -NFD -NMC -NodeCondition -NodeDrainPolicy -NodeIP -NodeLabeller -NodeLabellerArguments -NodeLabellerImage -NodeLabellerImagePullPolicy -Nodelabeller -nodename -Nodeport -NodePort -NodeRemediationLabels -NodeRemediationTaints -NoExecute -NPD -NPD's -NotReady -numGPUsAssigned -Observability -oc -OCI -OLM -OOM -OpenShift -OperatorHub -Openshift -parition -paritioning -pbqt -pebb -PCI -pcie -perf -PFs -Perses -plugin -PodIP -PreFlight -PreStateDB -prometheus -Promethues -quay -QEMU -QPX -RAS -RBAC -Redhat -RedHat -ResourceSlices -RHCOS -RMA -rocminfo -rochpl -ROCm -runtime -runtime's -SAR -schedulable -SDK -selfcheck -ServiceAccounts -ServiceMonitor -ServiecMonitor -skippedGPUs -Slinkproject -SlinkProject -Slrum -Slurm -SPX -StopOnFailure -SubjectAccessReview -systemd -TestCategory -TesterImage -TimeoutSeconds -TokenReview -Tolerations -TODO -TLS -tolerations -tst -TTL -TtlForFailedWorkflows -ubuntu -UI -UID -UNCORRECT -Uncordoning -uninstallation -unschedulable -Upgrademgr -UpgradePolicy -validation -verison -VC -VCN -VFIO -VFs -VMs -webhook -xgmi -YAML -AAC -ABI ALU AMD AMDGPU +AMDGPUHang +AMDGPUHangPermanent +AMDGPUKernelCrash +AMDGPUPageFault +AMDGPURASError +AMDGPUReset +AMDGPUUnhealthy AMDGPUs AMDMIGraphX AMI +ANR AOCC AOMP APIC APIs +ARI ASIC ASICs ASan @@ -245,29 +34,44 @@ ATI AWQ AdaLoRA AddressSanitizer +Affectioned AlexNet +Allocatable Arb AutoAWQ AutoGPTQ +AutoStartWorkflow +Autobuild +BDFs BLAS BMC BitCode Blit Bluefield CCD +CDI CDNA +CEL CIFAR CLI CLion CMake CMakeLists CMakePackage +CN +CNI CP CPC +CPER CPF CPP CPU CPUs +CPX +CRD +CRDs +CRI +CRs CSC CSE CSV @@ -276,46 +80,84 @@ CTests CU CUDA CUs +CVE CXX Cavium CentOS ChatGPT +CheckUnitStatus +CleanupPreState +ClusterRole CoRR Codespaces Commitizen +CommonConfig CommonMark Concretized Conda +ConditionalWorkflows +ConfigMap +ConfigMaps ConnectX +CoreOS +CrashLoopBackOff +CronJob +CustomResourceDefinition +Customizable +DCM DDP DGEMM DKMS DL DLM DMA +DMC +DME DNN DNNL +DNS DPM +DPX +DRA +DRADriver DRI DW DWORD +DaemonSet +DaemonSets +Daemonsets Dask DataFrame DataLoader DataParallel DeepSpeed Dependabot +Depricated DevCap +DeviceClass +DeviceConfig +DeviceIDs +DevicePlugin +DevicePluginArguments +DevicePluginImage +DevicePluginImagePullPolicy +DevicePluginSpec Diffusers +DockerHub Dockerfile Dockerfiles Doxygen +DriverToolkit +ECC ELMo ENDPGM EPEL EPYC ESXi EU +EnableDevicePlugin +EnableNodeLabeller +ErrImagePull ExLlama FFT FFTs @@ -340,6 +182,7 @@ GEMM GEMMs GFortran GIM +GKE GL GLXT GMI @@ -352,10 +195,14 @@ GPU's GPUs GQA GRBM +GRE GenAI GenZ GitHub +GitOps Gitpod +GracePeriodSeconds +Grafana HBM HCA HIPCC @@ -366,8 +213,12 @@ HPCG HPE HPL HSA +HSIO +HTTPS HWE Haswell +HealthThresholds +Helmify Higgs Hyperparameters ICV @@ -378,11 +229,16 @@ IOMMU IOP IOPM IOV +IPC IRQ ISA ISV ISVs +IfNotPresent +IgnoreDaemonSets +IgnoreNamespaces ImageNet +ImageStream InfiniBand Inlines IntelliSense @@ -394,10 +250,15 @@ JIT JSON Jupyter KFD +KMM KVM Keras Khronos +Kube +KubeVirt +Kubelet Kubernetes +Kuberntes LAPACK LCLK LDS @@ -408,8 +269,10 @@ LM LSAN LSTM LTS +Labeler LinearReLU LoRA +MCO MEM MERCHANTABILITY MFMA @@ -425,18 +288,30 @@ MMA MMIO MMIOH MNIST +MOK MPI MQA MSVC +MTLS MVAPICH MVFFR +MachineConfig +MachineConfigOperator Makefile Makefiles Matplotlib +MaxParallelWorkflows +MaxUnavailable Megatron Mellanox Mellanox's +Mericsclient Meta's +MetricsExporter +MetricsExporterSpec +MicroK +MinIO +Minio MirroredStrategy MoE Multicore @@ -445,11 +320,15 @@ MyEnvironment MyST NBIO NBIOs +NFD NHWC NIC NICs NLI NLP +NMC +NPD +NPD's NPS NSP NUMA @@ -458,24 +337,44 @@ NVIDIA NVPTX Nano Navi -Noncoherently +NoAMDGPUKernelCrash +NoExecute NoSchedule +NodeCondition +NodeDrainPolicy +NodeIP +NodeLabeller +NodeLabellerArguments +NodeLabellerImage +NodeLabellerImagePullPolicy +NodePort +NodeRemediationLabels +NodeRemediationTaints +Nodelabeller +Nodeport +Noncoherently +NotReady NousResearch's NumPy OAM OAMs +OCI OCP OEM OFED +OLM OMP OMPI OMPT OMPX ONNX +OOM OSS OSU +Observability Omniperf Omnitrace +OnDelete OpenAI OpenCL OpenCV @@ -483,10 +382,15 @@ OpenFabrics OpenGL OpenMP OpenSSL +OpenShift +OpenShift's OpenVX +Openshift +OperatorHub PCI PCIe PEFT +PFs PIL PILImage PPO @@ -496,19 +400,30 @@ PaLM Pageable PeerDirect Perfetto +Perses PipelineParallel PnP +PodIP PowerShell +PreFlight +PreStateDB +Promethues PyPi PyTorch +QEMU QLoRA +QPX Qcycles RAII +RAS +RBAC RCCL RDC RDMA RDNA +RHCOS RHEL +RMA RNN ROC ROCProfiler @@ -527,13 +442,18 @@ RST RW Radeon ReLU +RedHat +Redhat RelWithDebInfo Req +ResourceSlices Rickle RoCE +RollingUpdate Roofline Ryzen SALU +SAR SBIOS SCA SDK @@ -554,6 +474,7 @@ SMEM SMI SMT SPI +SPX SQs SRAM SRAMECC @@ -561,12 +482,21 @@ SVD SWE SciPy SerDes +ServiceAccounts +ServiceMonitor +ServiecMonitor Shlens Skylake +SlinkProject +Slinkproject +Slrum +Slurm SmoothQuant Softmax Spack StarCoder +StopOnFailure +SubjectAccessReview Supermicro Szegedy TCA @@ -577,6 +507,8 @@ TCP TCR TFLOPS TGI +TLS +TODO TPOT TPU TPUs @@ -584,11 +516,18 @@ TRL TTFT TTGIR TTIR +TTL +Techsupport Templated TensorBoard TensorFlow TensorParallel +TestCategory +TesterImage +TimeoutSeconds ToC +TokenReview +Tolerations TorchAudio TorchInductor TorchMIGraphX @@ -597,27 +536,43 @@ TorchServe TorchVision TransferBench TrapStatus +TtlForFailedWorkflows Tunable TunableOp UAC UC UCC UCX +UI +UID UIF +UMC +UNCORRECT URI USM UTCL UTIL Uncached +Uncordoning +Uncorrectable Unhandled +UpgradePolicy +UpgradeStrategy +Upgrademgr +UtilsContainer VALU VBIOS +VC +VCN +VFIO +VFs VGPR VGPRs VGPU VM VMEM VMWare +VMs VRAM VSIX VSkipped @@ -646,40 +601,64 @@ YModel ZeRO ZenDNN accuracies +acf activations addr alloc +allocatable allocator allocators +allowPrivilegeEscalation +amd amdgpu +amdgpuhealth api +apiVersion +apiserver +args atmi atomics +attachMetadata +autobuild autogenerated autoregression autoregressive avx awk +aws backend backends backpropagation backtick +bb +bd +bearerTokenFile benchmarking +bh bilinear bitsandbytes blit +bool boson bosons +bufferSize buildable +burnin bursty bzip +caFile cacheable cd centos centric +certFile changelog +checkmark chiplet ckProfiler +clientCAConfigMap +clientName +clusterIP cmake cmd coalescable @@ -691,12 +670,26 @@ comgr completers composability composable +computePartition concretization config +configmap +configs +configurability conformant +containerSecurityContext +containerd +controllerConfigYaml +controllerManager +controllerMetricsService convolutional convolves cpp +cpu +cpx +crds +cron +cryptographic csn cuBLAS cuFFT @@ -705,6 +698,8 @@ cuRAND cuSOLVER cuSPARSE customizations +daemonset +daemonsets dataset dataset's datasets @@ -712,8 +707,11 @@ dataspace datatype datatypes dbgapi +dcd +dcm de deallocation +deivce denoise denoised denoises @@ -721,15 +719,29 @@ denormalize deserializers detections dev +devel +devicePlugin +devicePluginImage +devicePluginSpec +deviceconfig +deviceconfigs devicelibs devsel dimensionality +disableHttps disambiguates +discoverable distro +dkms +dma +dmesg doxysphinx dropdown el embeddings +emptyDir +enableDevicePlugin +enableNodeLabeller enablement endpgm env @@ -739,19 +751,38 @@ ethernet exascale executables ffmpeg +fieldPath +fieldRef filesystem finalizer +flannel fortran galb +gapped +gc gcc gdb +generationID gfortran gfx githooks github gnupg +gocheck +gpu +gpu's +gpuClientSystemdServices +gpuagent +gpup +gpus +grafana grayscale +grpc +gst +gz gzip +hbm +hcrxm heterogenous hipBLAS hipBLASLt @@ -771,21 +802,33 @@ hipfort hipify hipsolver hipsparse +honorLabels +honorTimestamps +hostname +hostnames hpp hsa hsakmt +hsio html hyperparameter ib_core +iet +imagePullPolicy +imagePullSecrets +imageRegistrySecrets inband incrementing inferencing inflight init initializer +initramfs inlining -installable +insecureSkipVerify installCRDs +installable +installdefaultNFDRule instantiation interprocedural intersphinx @@ -793,49 +836,123 @@ intra invariants invocating invoker +io ipo +isigned +jlzbs +jq +json +kaniko kdb +keyFile +keySecret +kfd +kmm +kmod +kmsg +kube +kubeconfig +kubectl +kubelet +kubernetes +kubernetesClusterDomain +labeller +labeller's libfabric libjpeg libs +lifecycle linearized linter linux llvm localscratch +localtime +logPath logits +lookback lossy +lvl +mTLS macOS +managerConfig +matchLabels matchers +maxParallel +md +mem +memoryPartition +metricsExporter +metricsclient microarchitecture migraphx miopen miopengemm +misconfigurations mivisionx mkdir +mkdocs mlirmiopen +modprobe +mortem mtypes mvffr myst +namesapace namespace +namespaced namespaces +nano natively +nm +nmc +nodeAffinity +nodeCondition +nodeLabellerImage +nodeName +nodePort +nodeSelector +nodeSelectorTerms +nodelabeller +nodename +notifyRemediationMessage +notifyTestFailureMessage +numGPUsAssigned +numa numref +observability +oc ocl +onwards opencl opencv openmp +openshift openssl optimizers os +osImage +oyaml pageable parallelization parallelize parameterization +parition +paritioning passthrough +pbqt +pci +pcie +pebb +peqt +perf perfcounter performant perl +pesm +physicalActionNeeded +plugin +podman pragma pre prebuilt @@ -849,22 +966,42 @@ preprocessing prequantized prerequisites profiler +programmatically +prometheus protobuf pseudorandom py +pytest quantized quantizing quasirandom +quay queueing +ras +rbac rccl +rcqt rdc reStructuredText +readded +rebootRequired +reconfiguring +recoveryPolicy reformats +relatedImageBuild +relatedImageBuildPullSecret +relatedImageSign +relatedImageSignPullSecret +relatedImageWorker +relatedImageWorkerPullSecret +remediations +repo repos representativeness req resampling rescaling +retorquing reusability roadmap roc @@ -872,6 +1009,7 @@ rocAL rocALUTION rocBLAS rocFFT +rocHPL rocLIB rocMLIR rocPRIM @@ -884,6 +1022,7 @@ rocalution rocblas rocclr rocfft +rochpl rocm rocminfo rocprim @@ -895,23 +1034,44 @@ rocsolver rocsparse rocthrust roctracer +rollout runtime +runtime's runtimes sL scalability scalable +schedulable +searchability +securityContext +selfcheck sendmsg serializers +serverName +serviceAccount +serviceAccountName +serviceAccountNamespaceSelector +serviceAccountSelector +serviceType shader sharded sharding sigmoid +signimage +simd +skipRebootStep +skippedGPUs +slurm sm smi softmax spack +spx src +staticAuthorization +stdout stochastically +stopOnFailure strided struct subdirectories @@ -921,13 +1081,26 @@ subfolder subfolders suboptimal supercomputing +svc +symlinks +sys +sysfs +sysmon +systemd +targetPort +teardown +techsupport templated +testRunner th +timeoutSeconds +tlsConfig tokenization tokenize tokenized tokenizer tokenizes +tolerations toolchain toolchains toolset @@ -936,24 +1109,37 @@ torchtune torchvision tqdm tracebacks +tst tunable tunings txt uarch +ubuntu +un unallocated uncached +uncordoned uncorrectable uninstallation +unpartitioned +unschedulable unsqueeze unstacking unswitching untrusted untuned +upgradeCRD +upgradePolicy +upgradeStrategy upstreamed upvote +url utils vL vLLM +validation +validationTestsProfile +valueFrom variational vdi vectorizable @@ -962,12 +1148,23 @@ vectorize vectorized vectorizer vectorizes +verison +vf +vfio +virtfn +virtualized vjxb +vram walkthrough walkthroughs wavefront wavefronts +webhook +webhook's +webhookServer +webhookService whitespaces +workflowTemplate workgroup workgroups writeback @@ -975,202 +1172,12 @@ writebacks wrreq wzo xFormers +xGMI xargs +xgmi +xtwbm xz yaml +yamls ysvmadyb zyppe -CommonConfig -Kube -OnDelete -OpenShift's -RollingUpdate -Techsupport -UpgradeStrategy -UtilsContainer -AutoStartWorkflow -allocatable -allowPrivilegeEscalation -amdgpuhealth -apiserver -apiVersion -args -attachMetadata -autobuild -aws -BDFs -bd -bearerTokenFile -bool -caFile -certFile -checkmark -clientCAConfigMap -clientName -clusterIP -configmap -configs -containerSecurityContext -controllerConfigYaml -controllerManager -controllerMetricsService -cpu -cpx -crds -cron -cryptographic -devel -devicePlugin -devicePluginImage -devicePluginSpec -deviceconfig -deviceconfigs -disableHttps -discoverable -dkms -dmesg -emptyDir -enableNodeLabeller -gapped -generationID -gpu -gpu's -gpus -grafana -grpc -honorLabels -honorTimestamps -hsio -imagePullPolicy -imagePullSecrets -initramfs -insecureSkipVerify -installdefaultNFDRule -io -isigned -kfd -keyFile -keySecret -kmm -kube -kubernetesClusterDomain -kubelet -kubernetes -labeller's -mTLS -managerConfig -maxParallel -metricsclient -metricsExporter -misconfigurations -mkdocs -modprobe -mortem -namesapace -namespaced -nano -nmc -nodeAffinity -nodeCondition -nodeLabellerImage -nodePort -nodeSelector -nodeSelectorTerms -nodelabeller -notifyRemediationMessage -notifyTestFailureMessage -numa -observability -onwards -openshift -osImage -oyaml -pci -physicalActionNeeded -podman -programmatically -pytest -ras -rbac -readded -rebootRequired -recoveryPolicy -relatedImageBuild -relatedImageBuildPullSecret -relatedImageSign -relatedImageSignPullSecret -relatedImageWorker -relatedImageWorkerPullSecret -repo -retorquing -rocHPL -rollout -searchability -serverName -serviceAccount -serviceAccountNamespaceSelector -serviceAccountSelector -serviceType -signimage -simd -skipRebootStep -slurm -spx -staticAuthorization -stdout -svc -symlinks -sys -sysfs -sysmon -targetPort -techsupport -testRunner -tlsConfig -un -uncordoned -unpartitioned -upgradeCRD -upgradePolicy -upgradeStrategy -url -validationTestsProfile -vf -vfio -virtfn -virtualized -vram -webhook's -webhookServer -webhookService -workflowTemplate -xGMI -yamls -sysfs -BDFs -bh -dcd -gc -gz -hcrxm -jlzbs -nm -xtwbm -OCI -gocheck -teardown -DME -reconfiguring -AINIC -ANR -CVE -IPC -MicroK -configurability -remediations -GRE -mem -peqt -pesm -rcqt -CPER diff --git a/docs/npd/npd-dmesg-example.md b/docs/npd/npd-dmesg-example.md new file mode 100644 index 000000000..bbd32493c --- /dev/null +++ b/docs/npd/npd-dmesg-example.md @@ -0,0 +1,310 @@ +# NPD dmesg Kernel-Crash Detection Example + +This page shows how to extend the Node Problem Detector (NPD) configuration to watch +the kernel ring buffer (`/dev/kmsg`) for AMD GPU crash patterns, alongside the standard +`amdgpuhealth` custom plugin monitor. The dmesg rules emit permanent node conditions +that the GPU Operator's auto-remediation controller can act on. + +```{note} +This example extends the setup described in +[Node Problem Detector Integration](node-problem-detector.md). +Complete that setup first — RBAC, AMD Device Metrics Exporter, and the base DaemonSet +must all be in place before adding dmesg monitoring. +``` + +## How dmesg monitoring works with the GPU Operator + +NPD's `system-log-monitor` reads `/dev/kmsg` and matches log lines against regex rules. +When a line matches a `permanent` rule, NPD sets the named node condition to `True`. +The GPU Operator's remediation controller watches node conditions and triggers an Argo +workflow when it sees a condition that matches a `nodeCondition` entry in the +remediation ConfigMap. + +```text +/dev/kmsg (kernel ring buffer) + │ + ▼ +NPD system-log-monitor ──► NodeCondition = True (e.g. AMDGPUKernelCrash) + │ + ▼ +GPU Operator remediation controller ──► Argo Workflow +``` + +## Step 1 — Add dmesg rules to the NPD ConfigMap + +Extend the existing `node-problem-detector-config` ConfigMap with a second key, +`kernel-monitor.json`. The `system-log-monitor` plugin reads this file. + +The condition name (`AMDGPUKernelCrash` in the example below) must match the +`nodeCondition` field in the GPU Operator remediation ConfigMap so the operator +knows which workflow to trigger. + +```yaml +# node-problem-detector-config.yaml (extended) +apiVersion: v1 +kind: ConfigMap +metadata: + name: node-problem-detector-config + namespace: kube-system +data: + # Existing key — custom plugin monitor for amdgpuhealth metric checks + custom-plugin-monitor.json: | + { + "plugin": "custom", + "pluginConfig": { + "invoke_interval": "30s", + "timeout": "15s", + "max_output_length": 80, + "concurrency": 3, + "enable_message_change_based_condition_update": false + }, + "source": "amdgpu-custom-plugin-monitor", + "metricsReporting": true, + "conditions": [ + { + "type": "AMDGPUUnhealthy", + "reason": "AMDGPUIsUp", + "message": "AMDGPU is up" + } + ], + "rules": [ + { + "type": "permanent", + "condition": "AMDGPUUnhealthy", + "reason": "AMDGPUIsDown", + "path": "/var/lib/amd-metrics-exporter/amdgpuhealth", + "args": [ + "query", + "counter-metric", + "-m=GPU_ECC_UNCORRECT_UMC", + "-t=1" + ], + "timeout": "15s" + } + ] + } + + # New key — system-log monitor for dmesg / kmsg GPU crash patterns + kernel-monitor.json: | + { + "plugin": "kmsg", + "logPath": "/dev/kmsg", + "lookback": "5m", + "bufferSize": 10, + "source": "kernel-monitor", + "conditions": [ + { + "type": "AMDGPUKernelCrash", + "reason": "NoAMDGPUKernelCrash", + "message": "no AMD GPU kernel crash detected" + } + ], + "rules": [ + { + "type": "temporary", + "reason": "AMDGPUHang", + "pattern": "amdgpu.*GPU hang detected.*" + }, + { + "type": "permanent", + "condition": "AMDGPUKernelCrash", + "reason": "AMDGPUPageFault", + "pattern": "amdgpu.*GPU fault detected.*" + }, + { + "type": "permanent", + "condition": "AMDGPUKernelCrash", + "reason": "AMDGPUHangPermanent", + "pattern": "amdgpu.*GPU hang detected.*" + }, + { + "type": "permanent", + "condition": "AMDGPUKernelCrash", + "reason": "AMDGPUReset", + "pattern": "amdgpu.*GPU reset begin.*" + }, + { + "type": "permanent", + "condition": "AMDGPUKernelCrash", + "reason": "AMDGPURASError", + "pattern": "amdgpu.*RAS ERROR.*" + } + ] + } +``` + +### Rule reference + +| `type` | Effect | +|--------------|---------------------------------------------------------------------------------------------------------------------| +| `temporary` | Emits a one-shot Kubernetes `Event`. Does not flip a node condition. | +| `permanent` | Sets the named `condition` to `True` and keeps it set. Use this for conditions the remediation controller watches. | + +| Pattern | What it matches in dmesg | +|-------------------------------|-------------------------------------------------------| +| `amdgpu.*GPU fault detected` | GPU page fault logged by the `amdgpu` kernel driver | +| `amdgpu.*GPU hang detected` | GPU hang or lockup | +| `amdgpu.*GPU reset begin` | Driver-initiated GPU reset | +| `amdgpu.*RAS ERROR` | Uncorrectable RAS error reported to the kernel | + +`lookback: "5m"` replays the last 5 minutes of the ring buffer on startup, so crashes +that occurred just before NPD launched are not missed. + +## Step 2 — Add the system-log monitor to the DaemonSet + +Add `--config.system-log-monitor` and mount `/dev/kmsg`. The existing +`--config.custom-plugin-monitor` flag and all other mounts stay unchanged. + +```yaml +# node-problem-detector.yaml (extended) +apiVersion: apps/v1 +kind: DaemonSet +metadata: + name: node-problem-detector + namespace: kube-system + labels: + app: node-problem-detector +spec: + selector: + matchLabels: + app: node-problem-detector + template: + metadata: + labels: + app: node-problem-detector + spec: + nodeSelector: + feature.node.kubernetes.io/amd-gpu: "true" + tolerations: + # Required: keeps NPD running on nodes tainted by auto-remediation + - key: amd-gpu-unhealthy + operator: Exists + effect: NoSchedule + - effect: NoSchedule + operator: Exists + - effect: NoExecute + operator: Exists + serviceAccountName: node-problem-detector + containers: + - name: node-problem-detector + image: registry.k8s.io/node-problem-detector/node-problem-detector:v0.8.19 + command: + - /node-problem-detector + - --logtostderr + # dmesg / kernel ring buffer monitoring + - --config.system-log-monitor=/config/kernel-monitor.json + # amdgpuhealth metric monitoring + - --config.custom-plugin-monitor=/config/custom-plugin-monitor.json + securityContext: + privileged: true + env: + - name: NODE_NAME + valueFrom: + fieldRef: + fieldPath: spec.nodeName + resources: + limits: + cpu: 20m + memory: 100Mi + requests: + cpu: 10m + memory: 80Mi + volumeMounts: + - name: log + mountPath: /var/log + - name: kmsg + mountPath: /dev/kmsg + readOnly: true + - name: localtime + mountPath: /etc/localtime + readOnly: true + - name: config + mountPath: /config + readOnly: true + - name: amdexporter + mountPath: /var/lib/amd-metrics-exporter + volumes: + - name: log + hostPath: + path: /var/log/ + - name: kmsg + hostPath: + path: /dev/kmsg + - name: localtime + hostPath: + path: /etc/localtime + - name: config + configMap: + name: node-problem-detector-config + items: + - key: custom-plugin-monitor.json + path: custom-plugin-monitor.json + - key: kernel-monitor.json + path: kernel-monitor.json + - name: amdexporter + hostPath: + path: /var/lib/amd-metrics-exporter +``` + +```{important} +The `amd-gpu-unhealthy:NoSchedule` toleration is required. When auto-remediation taints +a node to evict workloads, NPD must keep running so its final condition check can +confirm that the node has recovered. Without this toleration, NPD is evicted and the +remediation workflow gets stuck waiting for the condition to flip back to `False`. +``` + +## Step 3 — Wire the condition into the remediation ConfigMap + +Add an entry for `AMDGPUKernelCrash` to the GPU Operator remediation ConfigMap so the +operator knows which Argo workflow to run when NPD sets that condition. + +```yaml +remediation: + - nodeCondition: AMDGPUKernelCrash + workflowTemplate: default-template + validationTestsProfile: + framework: AGFHC + recipe: all_lvl4 + iterations: 1 + stopOnFailure: true + timeoutSeconds: 4800 + physicalActionNeeded: false + skipRebootStep: false +``` + +See the [Auto Node Remediation](../autoremediation/auto-remediation.md) documentation +for the full remediation ConfigMap schema and available fields. + +## Step 4 — Apply and verify + +```bash +kubectl apply -f node-problem-detector-config.yaml +kubectl rollout restart daemonset/node-problem-detector -n kube-system + +# Confirm NPD pods are running on GPU nodes +kubectl get pods -n kube-system -l app=node-problem-detector -o wide + +# Stream NPD logs to see both monitors active +kubectl logs -n kube-system -l app=node-problem-detector -f + +# Check node conditions — both AMDGPUUnhealthy and AMDGPUKernelCrash should appear +kubectl describe node | sed -n '/Conditions:/,/Addresses:/p' +``` + +When healthy, both conditions are `False`: + +```text +Conditions: + Type Status Reason Message + ---- ------ ------ ------- + AMDGPUUnhealthy False AMDGPUIsUp AMDGPU is up + AMDGPUKernelCrash False NoAMDGPUKernelCrash no AMD GPU kernel crash detected +``` + +When a matching dmesg line appears, `AMDGPUKernelCrash` flips to `True` and the GPU +Operator triggers an Argo workflow: + +```bash +kubectl get workflows -A +kubectl get events -A --field-selector reason=amd-gpu-remediation-required +``` diff --git a/docs/sphinx/_toc.yml b/docs/sphinx/_toc.yml index 12b638378..67d4f9827 100644 --- a/docs/sphinx/_toc.yml +++ b/docs/sphinx/_toc.yml @@ -75,6 +75,8 @@ subtrees: - caption: Node Problem Detector entries: - file: npd/node-problem-detector + - file: npd/npd-dmesg-example + title: dmesg Kernel-Crash Detection Example - caption: Auto Remediation entries: - file: autoremediation/auto-remediation