onwAMD / diag.py
ryugyosoft's picture
onw AMD 0.1.2: memlock on Linux, warn about another FastFlowLM
64f4a9a verified
Raw History Blame Contribute Delete
9.13 kB
"""onw AMD diagnostics: why FastFlowLM's NPU check fails on this PC. Standard library only; reads, changes nothing
(except --try-model, which pulls one small model into onw's model folder and runs it once).
curl -fsSL https://huggingface.co/ryugyosoft/onwAMD/resolve/main/diag.py | python3 - (Linux)
curl -fsSL https://huggingface.co/ryugyosoft/onwAMD/resolve/main/diag.py | python3 - --try-model (+ a real run)
irm https://huggingface.co/ryugyosoft/onwAMD/resolve/main/diag.py | python - (Windows)
The report is printed and saved as onw-amd-diag.txt in the home folder: send that file.
"""
import glob, json, os, platform, shutil, subprocess, sys, time, urllib.request
WIN = os.name == "nt"
OUT = []
def say(*a):
line = " ".join(str(x) for x in a)
print(line, flush=True)
OUT.append(line)
def section(t):
say("\n==== " + t)
def run(cmd, timeout=60, env=None):
try:
r = subprocess.run(cmd, capture_output=True, text=True, encoding="utf8", errors="replace", timeout=timeout,
env=env)
return r.returncode, r.stdout, r.stderr
except Exception as e:
return None, "", f"{type(e).__name__}: {e}"
def show(cmd, timeout=60, env=None, limit=3000):
rc, out, err = run(cmd, timeout, env)
say(f"$ {' '.join(cmd)} (exit {rc})")
for s in (out, err):
s = s.strip()
if s:
say(s[-limit:])
return rc, out, err
def config_dir():
base = os.environ.get("APPDATA") if WIN else os.path.join(os.path.expanduser("~"), ".config")
return os.path.join(base or os.path.expanduser("~"), "onw-amd")
def main():
try_model = "--try-model" in sys.argv
section("system")
say("python", sys.version.split()[0], "|", platform.platform())
if not WIN:
try:
say(next(l for l in open("/etc/os-release") if l.startswith("PRETTY_NAME")).strip())
except Exception:
pass
say("kernel", platform.release())
try:
say("cpu", next(l.split(":", 1)[1].strip() for l in open("/proc/cpuinfo") if l.startswith("model name")))
except Exception:
pass
if WIN:
section("NPU (Windows)")
show(["powershell", "-NoProfile", "-Command",
"Get-CimInstance Win32_PnPEntity | Where-Object { $_.PNPDeviceID -match 'VEN_1022&DEV_(17F0|1502)' } | "
"Select-Object Name,PNPDeviceID,Status | Format-List; "
"Get-CimInstance Win32_PnPSignedDriver | Where-Object { $_.DeviceName -match 'NPU Compute' } | "
"Select-Object DeviceName,DriverVersion,DriverDate | Format-List"])
else:
section("NPU (Linux): PCI")
for dev in sorted(glob.glob("/sys/bus/pci/devices/*")):
rd = lambda f: (open(os.path.join(dev, f)).read().strip() if os.path.exists(os.path.join(dev, f)) else "?")
if rd("vendor") == "0x1022" and rd("class").startswith("0x1180"):
drv = os.path.basename(os.path.realpath(os.path.join(dev, "driver"))) if os.path.exists(
os.path.join(dev, "driver")) else "(no driver bound)"
say(os.path.basename(dev), "device", rd("device"), "rev", rd("revision"), "driver", drv)
section("NPU (Linux): driver, firmware, device node")
for p in ("/sys/module/amdxdna/version", "/sys/module/amdxdna/srcversion"):
if os.path.exists(p):
say(p, open(p).read().strip())
say("amdxdna module loaded:", os.path.isdir("/sys/module/amdxdna"))
show(["modinfo", "amdxdna"], limit=1200)
fw = sorted(glob.glob("/lib/firmware/amdnpu/**/*", recursive=True))
say("firmware /lib/firmware/amdnpu:", len(fw), "files")
for f in fw[:40]:
say(" ", f)
nodes = glob.glob("/dev/accel/*")
say("device nodes:", nodes or "none")
for n in nodes:
st = os.stat(n)
say(" ", n, oct(st.st_mode & 0o777), "uid", st.st_uid, "gid", st.st_gid,
"| readable", os.access(n, os.R_OK), "writable", os.access(n, os.W_OK))
show(["id"])
import resource
soft, hard = resource.getrlimit(resource.RLIMIT_MEMLOCK)
fmt = lambda v: "unlimited" if v == resource.RLIM_INFINITY else f"{v // 1024} KiB"
say("memlock (ulimit -l): soft", fmt(soft), "hard", fmt(hard))
show(["sh", "-c", "grep -rh memlock /etc/security/limits.conf /etc/security/limits.d/ 2>/dev/null | grep -v '^#'"])
show(["sh", "-c", "journalctl -k --no-pager -g 'amdxdna|amdnpu' -n 40 2>/dev/null || dmesg 2>&1 | grep -i -E 'amdxdna|amdnpu' | tail -40"])
section("other FastFlowLM / XRT (Lemonade Server, system packages)")
if WIN:
show(["powershell", "-NoProfile", "-Command",
"Get-Process flm -ErrorAction SilentlyContinue | Select-Object Id,Path | Format-List"])
else:
show(["sh", "-c", "ps -eo pid,user,args | grep -E '[f]lm(-real)?( |$)|[l]emonade' || echo 'none running'"])
show(["sh", "-c", "dpkg -l 'fastflowlm*' 'lemonade*' 'libxrt*' 'xrt*' 'amdxdna*' 2>/dev/null | grep -E '^(ii|rc)' || echo 'no such packages'"])
show(["sh", "-c", "ls -d /opt/fastflowlm /opt/xilinx/xrt 2>&1"])
section("FastFlowLM")
cfg = config_dir()
exe_name = "flm.exe" if WIN else "flm"
found = [p for p in glob.glob(os.path.join(cfg, "bin", "flm", "npu", "**", exe_name), recursive=True)]
say("onw config folder:", cfg, "exists" if os.path.isdir(cfg) else "MISSING")
say("flm installed by onw:", found or "none")
say("flm on PATH:", shutil.which("flm") or "none")
try:
say("onw FastFlowLM version.txt:", open(os.path.join(cfg, "bin", "flm", "npu", "version.txt")).read().strip())
except OSError:
pass
flm = found[0] if found else shutil.which("flm")
try:
conf = json.load(open(os.path.join(cfg, "config.json"), encoding="utf8"))
except Exception:
conf = {}
models = conf.get("models_dir") or os.path.join(cfg, "flm")
env = {**os.environ, "FLM_MODEL_PATH": models}
say("FLM_MODEL_PATH onw uses:", models, "| in the environment:", os.environ.get("FLM_MODEL_PATH", "(unset)"))
if flm:
if not WIN:
say("executable:", os.access(flm, os.X_OK))
rc, out, err = run(["ldd", flm])
missing = [l.strip() for l in out.splitlines() if "not found" in l]
say("ldd: missing libraries:", missing or "none")
show([flm, "version"], env=env)
show([flm, "validate", "--json"], env=env)
show([flm, "validate"], env=env)
show([flm, "list", "--filter", "installed", "--quiet", "--json"], env=env, limit=1500)
section("onw logs")
for name in ("download.log", "server.log"):
p = os.path.join(cfg, name)
if os.path.exists(p):
say(f"--- {name} (last lines)")
say("".join(open(p, encoding="utf8", errors="replace").readlines()[-40:]).rstrip())
if try_model and flm:
section("try a model anyway (qwen3:0.6b, about 0.7 GB): is the NPU check wrong, or does the NPU really fail?")
show([flm, "pull", "qwen3:0.6b"], timeout=1800, env=env, limit=800)
log = os.path.join(os.path.expanduser("~"), "onw-amd-diag-serve.log")
with open(log, "w") as lf:
p = subprocess.Popen([flm, "serve", "qwen3:0.6b", "--port", "8099", "--host", "127.0.0.1"], env=env,
stdout=lf, stderr=subprocess.STDOUT)
try:
ok = False
for _ in range(180):
if p.poll() is not None:
break
try:
urllib.request.urlopen("http://127.0.0.1:8099/api/tags", timeout=2)
ok = True
break
except OSError:
time.sleep(1)
say("flm serve came up:", ok, "| exit code" if p.poll() is not None else "", p.poll() if p.poll() is not None else "")
if ok:
body = {"model": "qwen3:0.6b", "max_tokens": 40, "think": False,
"messages": [{"role": "user", "content": "Say hello in one short sentence."}]}
try:
j = json.loads(urllib.request.urlopen(urllib.request.Request(
"http://127.0.0.1:8099/v1/chat/completions", json.dumps(body).encode(),
{"Content-Type": "application/json"}), timeout=300).read())
say("answer:", j["choices"][0]["message"].get("content"))
say("usage:", json.dumps(j.get("usage")))
except Exception as e:
say("request failed:", e)
finally:
p.kill()
say("--- flm serve log (last lines)")
say("".join(open(log, errors="replace").readlines()[-40:]).rstrip())
path = os.path.join(os.path.expanduser("~"), "onw-amd-diag.txt")
with open(path, "w", encoding="utf8") as f:
f.write("\n".join(OUT) + "\n")
print(f"\nSaved: {path} (send this file)")
if __name__ == "__main__":
main()