#!/usr/bin/env python3 """Run in your workload's Python environment. Prints a read-only Doctor report. No packages are installed, credentials collected, or data sent over the network. A tiny GPU operation verifies execution; localhost ports are checked for a TCP listener only. Run from your intended data directory to check that filesystem. """ import csv import importlib.metadata import json import platform import shutil import socket import subprocess def package(name): try: return importlib.metadata.version(name) except importlib.metadata.PackageNotFoundError: return None def collect(): report = {"schema_version": 1, "python_version": platform.python_version(), "torch_version": package("torch"), "vllm_version": package("vllm"), "gpus": [], "ports": {}, "probe_errors": []} try: report["disk_free_gb"] = round(shutil.disk_usage(".").free / (1024 ** 3), 2) except OSError: report["probe_errors"].append("Disk check unavailable") try: result = subprocess.run(["nvidia-smi", "--query-gpu=name,driver_version,memory.total", "--format=csv,noheader,nounits"], capture_output=True, text=True, timeout=10, check=True) for row in list(csv.reader(result.stdout.splitlines()))[:16]: name, driver, memory = (value.strip() for value in row) report["driver_version"] = driver report["gpus"].append({"name": name[:120], "memory_gb": round(float(memory) / 1024, 2)}) except (OSError, subprocess.SubprocessError, ValueError): report["probe_errors"].append("nvidia-smi unavailable") # Isolate CUDA imports/initialization so a stuck driver cannot hang Doctor. if report["torch_version"]: code = '''import json, torch r = {"cuda_available": torch.cuda.is_available(), "torch_cuda_version": torch.version.cuda, "cuda_archs": torch.cuda.get_arch_list() if torch.cuda.is_available() else [], "kernel_test_ok": False} gpus = [] try: for i in range(min(torch.cuda.device_count(), 16)): p = torch.cuda.get_device_properties(i) free, total = torch.cuda.mem_get_info(i) gpus.append({"name": p.name, "memory_gb": round(total / 1024**3, 2), "free_memory_gb": round(free / 1024**3, 2), "compute_capability": str(p.major)+"."+str(p.minor)}) if r["cuda_available"]: x = torch.ones(1, device="cuda") r["kernel_test_ok"] = bool((x + 1).item() == 2) except Exception: r["kernel_test_ok"] = False if gpus: r["gpus"] = gpus print(json.dumps(r)) ''' try: import sys result = subprocess.run([sys.executable, "-c", code], capture_output=True, text=True, timeout=30, check=True) report.update(json.loads(result.stdout.splitlines()[-1])) except (OSError, subprocess.SubprocessError, ValueError, IndexError): report["probe_errors"].append("PyTorch GPU check unavailable") for port in (8000, 8188, 8888): try: with socket.create_connection(("127.0.0.1", port), timeout=1): report["ports"][str(port)] = True except OSError: report["ports"][str(port)] = False return report if __name__ == "__main__": print(json.dumps(collect(), indent=2))