Repository navigation
Expand file tree
/
Copy pathgpu_operator_checks.py
More file actions
278 lines (259 loc) · 11.3 KB
/
Copy pathgpu_operator_checks.py
File metadata and controls
278 lines (259 loc) · 11.3 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
import subprocess
import sys
def check_oc_installation():
"""Check if the OpenShift CLI (oc) is installed on the system."""
try:
version_result = subprocess.run(
"oc version",
shell=True,
stdout=subprocess.PIPE,
stderr=subprocess.PIPE,
text=True
)
if version_result.returncode == 0:
print("OpenShift CLI (oc) is installed:")
print(version_result.stdout)
return True
else:
print("OpenShift CLI (oc) installation check failed:")
print(version_result.stderr)
return False
except FileNotFoundError:
print("Error: 'oc' command not found")
print("Please ensure the OpenShift CLI ('oc') is installed and added to your PATH")
return False
def check_nfd_create():
# Command to check NodeFeatureDiscovery
command = "oc get NodeFeatureDiscovery -n openshift-nfd"
try:
# Execute the command and capture output
result = subprocess.run(
command,
shell=True,
stdout=subprocess.PIPE,
stderr=subprocess.PIPE,
text=True
)
# Check the return code
if result.returncode == 0:
print("NodeFeatureDiscovery check executed successfully!")
print("Output:")
print(result.stdout)
else:
print("NodeFeatureDiscovery check failed with the following error:")
print(result.stderr)
print("\nPossible issues:")
print("- Not logged into an OpenShift cluster")
print("- No access to the 'openshift-nfd' namespace")
print("- NodeFeatureDiscovery resource doesn't exist")
except Exception as e:
print(f"An unexpected error occurred: {str(e)}")
def check_nvidia_gpu_nodes():
# Command to check nodes with NVIDIA GPUs
command = "oc get nodes -l feature.node.kubernetes.io/pci-10de.present"
try:
# Execute the command and capture output
result = subprocess.run(
command,
shell=True,
stdout=subprocess.PIPE,
stderr=subprocess.PIPE,
text=True
)
# Check the return code
if result.returncode == 0:
print("NVIDIA GPU nodes check executed successfully!")
if result.stdout.strip(): # Check if there's any output
print("Nodes with NVIDIA GPUs found:")
print(result.stdout)
else:
print("No nodes with NVIDIA GPUs found in the cluster")
else:
print("NVIDIA GPU nodes check failed with the following error:")
print(result.stderr)
print("\nPossible issues:")
print("- Not logged into an OpenShift cluster")
print("- Node Feature Discovery operator might not be installed")
print("- No nodes with NVIDIA GPUs exist in the cluster")
except Exception as e:
print(f"An unexpected error occurred: {str(e)}")
def check_clusterpolicy_crd():
"""Check if the ClusterPolicy CRD exists in the cluster."""
command = "oc get crd/clusterpolicies.nvidia.com"
try:
result = subprocess.run(
command,
shell=True,
stdout=subprocess.PIPE,
stderr=subprocess.PIPE,
text=True
)
if result.returncode == 0:
print("ClusterPolicy CRD check executed successfully!")
print("Output:")
print(result.stdout)
else:
print("ClusterPolicy CRD check failed with the following error:")
print(result.stderr)
print("\nPossible issues:")
print("- Not logged into an OpenShift cluster")
print("- NVIDIA GPU Operator might not be installed")
print("- ClusterPolicy CRD has not been deployed")
except Exception as e:
print(f"An unexpected error occurred: {str(e)}")
def check_clusterpolicy_status():
"""Check the status of ClusterPolicy instances and fetch details if not ready."""
# Step 1: Get the list of ClusterPolicies and their states
get_command = "oc get clusterpolicy -o custom-columns=NAME:.metadata.name,STATE:.status.state --no-headers"
try:
get_result = subprocess.run(
get_command,
shell=True,
stdout=subprocess.PIPE,
stderr=subprocess.PIPE,
text=True
)
# Check if the oc get command executed successfully
if get_result.returncode == 0:
output_lines = get_result.stdout.strip().splitlines()
if output_lines:
for line in output_lines:
# Parse name and state from each line
parts = line.split()
if len(parts) >= 2:
name, state = parts[0], parts[1]
print(f"ClusterPolicy {name}: {state}")
# Step 2: If NotReady, run oc describe to get the message
if state == "notReady":
describe_command = f"oc describe clusterpolicy {name}"
try:
describe_result = subprocess.run(
describe_command,
shell=True,
stdout=subprocess.PIPE,
stderr=subprocess.PIPE,
text=True
)
if describe_result.returncode == 0:
message = parse_ready_condition_message(describe_result.stdout)
if message:
print(f" Message: {message}")
else:
print(" No detailed message found for the Ready condition")
else:
print(" Failed to run oc describe:")
print(f" {describe_result.stderr}")
except Exception as e:
print(f" An error occurred while running oc describe: {str(e)}")
else:
print(f"Unexpected output format: {line}")
else:
print("No ClusterPolicy instances found in the cluster")
else:
print("Failed to retrieve ClusterPolicy status:")
print(get_result.stderr)
except Exception as e:
print(f"An unexpected error occurred: {str(e)}")
def parse_ready_condition_message(describe_output):
"""Parse the oc describe output to find the Message for the Ready condition with Status: False."""
lines = describe_output.splitlines()
in_conditions = False
current_type = None
current_status = None
for line in lines:
line = line.strip()
if line.startswith("Conditions:"):
in_conditions = True
elif in_conditions:
if line.startswith("Type:"):
current_type = line.split(":", 1)[1].strip()
elif line.startswith("Status:"):
current_status = line.split(":", 1)[1].strip()
elif line.startswith("Message:") and current_type == "Ready" and current_status == "False":
return line.split(":", 1)[1].strip()
return None
def check_gpu_operator_logs():
"""Fetch and display logs for GPU operator pods in the nvidia-gpu-operator namespace."""
command = "oc logs -n nvidia-gpu-operator -lapp=gpu-operator"
try:
result = subprocess.run(
command,
shell=True,
stdout=subprocess.PIPE,
stderr=subprocess.PIPE,
text=True
)
if result.returncode == 0:
print("GPU operator logs retrieved successfully!")
if result.stdout.strip():
print("Logs:")
print(result.stdout)
else:
print("No logs found for GPU operator pods.")
else:
print("Failed to retrieve GPU operator logs:")
print(result.stderr)
print("\nPossible issues:")
print("- Not logged into an OpenShift cluster")
print("- The 'nvidia-gpu-operator' namespace does not exist")
print("- No pods with label 'app=gpu-operator' found")
print("- No logs available (e.g., pods haven’t started yet)")
except Exception as e:
print(f"An unexpected error occurred: {str(e)}")
def check_gpu_operator_pods():
"""Check if the GPU operator pods are running in the nvidia-gpu-operator namespace."""
command = "oc get pods -n nvidia-gpu-operator -lapp=gpu-operator"
try:
result = subprocess.run(
command,
shell=True,
stdout=subprocess.PIPE,
stderr=subprocess.PIPE,
text=True
)
if result.returncode == 0:
lines = result.stdout.splitlines()
if len(lines) > 1: # Check if there are pods listed beyond the header
running_pods = 0
for line in lines[1:]: # Skip the first line as it is assumed to be a header
parts = line.split()
if len(parts) >= 3 and parts[2] == 'Running':
running_pods += 1
elif status == 'ImagePullBackOff':
error_messages.append("Pod in 'ImagePullBackOff' status: maybe the NVIDIA registry is down.")
elif status == 'CrashLoopBackOff':
error_messages.append("Pod in 'CrashLoopBackOff' status: review the operator logs.")
check_gpu_operator_logs()
if running_pods > 0:
print(f"GPU operator is running with {running_pods} pod(s) in 'Running' state.")
print("Pod details:")
print(result.stdout)
else:
print("GPU operator pods are present but none are in 'Running' state.")
print("Pod details:")
print(result.stdout)
else:
print("No GPU operator pods found in the nvidia-gpu-operator namespace.")
else:
print("Failed to check GPU operator pods:")
print(result.stderr)
print("\nPossible issues:")
print("- Not logged into an OpenShift cluster")
print("- The 'nvidia-gpu-operator' namespace does not exist")
print("- No pods with label 'app=gpu-operator' found")
except Exception as e:
print(f"An unexpected error occurred: {str(e)}")
# Run the script
if __name__ == "__main__":
print("Checking OpenShift CLI installation...")
check_oc_installation()
print("\nChecking NodeFeatureDiscovery command...")
check_nfd_create()
print("\nChecking for nodes with NVIDIA GPUs...")
check_nvidia_gpu_nodes()
print("\nChecking ClusterPolicy CRD...")
check_clusterpolicy_crd()
print("\nChecking ClusterPolicy status...")
check_clusterpolicy_status()
print("\nChecking GPU operator pods...")
check_gpu_operator_pods()