IQ.Pilot Release Commit @ f2a861c

This commit is contained in:
IQ.Lvbs CI [bot]
2026-09-02 15:07:09 -05:00
parent b42569dbca
commit e8748fd704
5497 changed files with 316070 additions and 179848 deletions

View File

@@ -5,7 +5,7 @@ Copyright © IQ.Lvbs, apart of Project Teal Lvbs, All Rights Reserved, licensed
from __future__ import annotations
from openpilot.iqpilot.selfdrive.iqmodeld.tools.daemon_jit_compiler import main
from iqpilot.selfdrive.iqmodeld.tools.daemon_jit_compiler import main
if __name__ == "__main__":

View File

@@ -0,0 +1,471 @@
"""
Copyright © IQ.Lvbs, apart of Project Teal Lvbs, All Rights Reserved, licensed under https://konn3kt.com/tos/
"""
from __future__ import annotations
import argparse
import gc
import os
import pickle
import sys
import time
os.environ.setdefault("FLOAT16", "1")
os.environ.setdefault("JIT_BATCH_SIZE", "0")
os.environ.setdefault("GMMU", "0")
# TC_OPT=2 lets tinygrad pick tensor-core kernels; on some models a TC kernel miscompiles and biases
# the output (documented on Metal). A parity gate below catches it and re-compiles with TC off.
os.environ.setdefault("TC_OPT", "0" if ("--tc-off" in sys.argv or os.environ.get("IQ_EGPU_TC_OFF")) else "2")
HOST = "--host" in sys.argv
if HOST:
from iqpilot.selfdrive.iqmodeld.tools.egpu_host_mock import DEFAULT_ARCH, activate
activate(sys.argv[sys.argv.index("--arch") + 1] if "--arch" in sys.argv else DEFAULT_ARCH)
os.environ.setdefault("DEV", "USB+AMD:LLVM")
import numpy as np
from iqpilot.selfdrive.iqmodeld.egpu_helpers import egpu_pkl_path, local_onnx, patch_tinygrad_fetch_fw
from iqpilot.selfdrive.iqmodeld.egpu_model import EGPU_MODELS, get_egpu_model, resolve_egpu_model
from iqpilot.selfdrive.iqmodeld.temporal_state import MODEL_INPUT_SPEC, spec_from_meta
INPUT_SPEC = dict(MODEL_INPUT_SPEC)
patch_tinygrad_fetch_fw()
SEED = 42
class _ParityFail(RuntimeError):
pass
KERNEL_PROGRESS_SCALE = 260.0
def _progress_sampler(param: str, base: float, span: float, stop) -> None:
import math
from iqpilot.common.params import Params
from tinygrad.helpers import GlobalCounters
pm = Params()
last = -1.0
while not stop.wait(0.5):
kernels = float(getattr(GlobalCounters, "kernel_count", 0))
value = base + span * (1.0 - math.exp(-kernels / KERNEL_PROGRESS_SCALE))
if value - last >= 0.01:
last = value
pm.put(param, f"{min(base + span, value):.3f}")
def set_input_spec(meta: dict) -> None:
spec = spec_from_meta(meta)
if spec is not None:
INPUT_SPEC.clear()
INPUT_SPEC.update(spec)
def make_run_model(model_runner):
def run_model(**inputs):
out = next(iter(model_runner({k: inputs[k] for k in INPUT_SPEC}).values())).cast("float32")
return out.reshape(-1),
return run_model
def _random_inputs(seed: int):
from tinygrad.device import Device
from tinygrad.tensor import Tensor
rng = np.random.default_rng(seed)
out = {}
for name, (shape, dtype) in INPUT_SPEC.items():
if dtype == "uint8":
arr = rng.integers(0, 256, shape).astype(np.uint8)
else:
arr = rng.standard_normal(shape).astype(np.float32)
out[name] = Tensor(arr, device=Device.DEFAULT).realize()
return out
def _run(fn, seed: int) -> np.ndarray:
from tinygrad.device import Device
st = time.perf_counter()
outs = fn(**_random_inputs(seed))
Device.default.synchronize()
print(f" run(seed={seed}) {(time.perf_counter() - st) * 1e3:6.1f} ms")
return outs[0].numpy().reshape(-1)
def compile_model(meta: dict, onnx_path: str, out_path: str) -> str:
from tinygrad.device import Device
from tinygrad.engine.jit import TinyJit
from tinygrad.nn.onnx import OnnxRunner
if meta.get("split"):
raise RuntimeError(f"model {meta['key']} is a split model; eGPU v1 compiles fused models only")
jit = TinyJit(make_run_model(OnnxRunner(onnx_path)), prune=True)
print("capture + replay")
for _ in range(2):
baseline = _run(jit, SEED)
if baseline.shape[0] != meta["output_len"]:
raise RuntimeError(f"model output length {baseline.shape[0]} != registry {meta['output_len']}")
if not np.isfinite(baseline).all():
raise RuntimeError("compiled model produced non-finite outputs")
bundle = {
"run_model": jit,
"model_key": meta["key"],
"model_sha256": meta["sha256"],
"output_len": int(meta["output_len"]),
"frame_skip": int(meta["frame_skip"]),
"input_spec": {name: (tuple(shape), dtype) for name, (shape, dtype) in INPUT_SPEC.items()},
"input_device": Device.DEFAULT,
}
os.makedirs(os.path.dirname(out_path), exist_ok=True)
tmp = out_path + ".part"
print("serialize")
with open(tmp, "wb") as f:
pickle.dump(bundle, f, protocol=pickle.HIGHEST_PROTOCOL)
del bundle, jit
gc.collect()
print("reload + validate")
with open(tmp, "rb") as f:
jit = pickle.load(f)["run_model"]
if not np.array_equal(_run(jit, SEED), baseline):
raise RuntimeError("outputs differ from baseline after pickle round trip")
if np.array_equal(_run(jit, SEED + 1), baseline):
raise RuntimeError("outputs insensitive to inputs after pickle round trip")
from tinygrad.tensor import Tensor
zeros = {name: Tensor(np.zeros(shape, dtype=dtype), device=Device.DEFAULT).realize()
for name, (shape, dtype) in INPUT_SPEC.items()}
flat = jit(**zeros)[0].numpy().reshape(-1)
from iqpilot.selfdrive.iqmodeld.parser import PhaseParser
from iqpilot.selfdrive.iqmodeld.tools.compile_supercombo import _slice_outputs, _validate_pose_outputs
_validate_pose_outputs(PhaseParser().parse_vision_outputs(_slice_outputs(flat, meta["output_slices"])))
os.replace(tmp, out_path)
return out_path
def _policy_frame(seed: int, input_spec: dict):
from tinygrad.tensor import Tensor
rng = np.random.default_rng(seed)
img = input_spec["img"][0]
warped = Tensor(rng.integers(0, 256, (2, 6, img[2], img[3])).astype(np.uint8), device="NPY").realize()
return warped
def _tc_off_reference(onnx_path: str, meta: dict, fmt: int = 2, resolutions: tuple[tuple[int, int], ...] = ()):
"""Compile+run the model with tensor cores OFF in a child process and return the last of 3
policy frames. This is the trusted reference: TC-off kernels are the conservative path the
eMac gate also trusts. Used to catch a TC kernel miscompile that would bias steering."""
import subprocess
import tempfile
with tempfile.TemporaryDirectory() as td:
ref = os.path.join(td, "ref.npz" if fmt == 3 else "ref.npy")
env = {k: v for k, v in os.environ.items() if k not in ("TC_OPT", "BEAM")}
env["TC_OPT"] = "0"
env["IQ_EGPU_REFERENCE"] = ref
cmd = [sys.executable, "-m", "iqpilot.selfdrive.iqmodeld.tools.compile_egpu_model",
"--model", meta["key"], "--onnx", onnx_path, "--tc-off", "--format", str(fmt)]
if resolutions:
cmd += ["--camera-resolutions", *(f"{w}x{h}" for w, h in resolutions)]
r = subprocess.run(cmd, env=env, capture_output=True, text=True, timeout=14400)
if r.returncode != 0 or not os.path.isfile(ref):
raise RuntimeError(f"parity reference compile failed:\n{r.stderr[-2000:]}")
return np.load(ref)
def _parity_check(key: str, got: np.ndarray, ref: np.ndarray, label: str = "") -> None:
rel = float(np.abs(got - ref).mean() / max(1e-3, float(np.abs(ref).mean())))
if rel > 0.01:
raise _ParityFail(f"PARITY FAIL: TC kernels miscompiled {key} {label}(rel={rel:.4f} vs TC-off); recompiling with tensor cores disabled")
print(f" parity vs TC-off reference {label}: rel={rel:.6f} OK")
def compile_policy_model(meta: dict, onnx_path: str, out_path: str) -> str:
from tinygrad.device import Device
from tinygrad.engine.jit import TinyJit
from tinygrad.nn.onnx import OnnxRunner
from iqpilot.selfdrive.iqmodeld.egpu_policy import POLICY_FORMAT, PackedInputs, dump_oob, load_bundle, make_queues, make_run_policy
if meta.get("split"):
raise RuntimeError(f"model {meta['key']} is a split model; eGPU compiles fused models only")
input_spec = {name: (tuple(shape), dtype) for name, (shape, dtype) in INPUT_SPEC.items()}
frame_skip = int(meta["frame_skip"])
device = Device.DEFAULT
jit = TinyJit(make_run_policy(OnnxRunner(onnx_path), input_spec, frame_skip, device), prune=True)
queues = make_queues(input_spec, frame_skip, device)
packed = PackedInputs(input_spec)
def step(seed: int) -> np.ndarray:
packed.views["traffic_convention"][:] = [1, 0]
packed.views["action_t"][:] = [0.2, 0.3]
st = time.perf_counter()
out, = jit(warped=_policy_frame(seed, input_spec), packed_npy_inputs=packed.tensor, **queues)
flat = out.numpy().reshape(-1)
print(f" policy step(seed={seed}) {(time.perf_counter() - st) * 1e3:6.1f} ms")
packed.views["prev_feat"][:] = flat[meta["output_slices"]["hidden_state"]].reshape(packed.views["prev_feat"].shape)
return flat
print("capture + replay")
for i in range(3):
baseline = step(SEED + i)
if baseline.shape[0] != meta["output_len"]:
raise RuntimeError(f"model output length {baseline.shape[0]} != registry {meta['output_len']}")
if not HOST and not np.isfinite(baseline).all():
raise RuntimeError("compiled policy produced non-finite outputs")
bundle = {
"format": POLICY_FORMAT,
"run_policy": jit,
"model_key": meta["key"],
"model_sha256": meta["sha256"],
"output_len": int(meta["output_len"]),
"frame_skip": frame_skip,
"input_spec": input_spec,
"input_device": device,
}
os.makedirs(os.path.dirname(out_path), exist_ok=True)
tmp = out_path + ".part"
print("serialize (out-of-band buffers)")
with open(tmp, "wb") as f:
dump_oob(bundle, f)
del bundle, jit, queues, packed
gc.collect()
print("reload + validate")
jit = load_bundle(tmp)["run_policy"]
queues = make_queues(input_spec, frame_skip, device)
packed = PackedInputs(input_spec)
outs = []
for i in range(3):
packed.views["traffic_convention"][:] = [1, 0]
packed.views["action_t"][:] = [0.2, 0.3]
out, = jit(warped=_policy_frame(SEED + i, input_spec), packed_npy_inputs=packed.tensor, **queues)
flat = out.numpy().reshape(-1)
packed.views["prev_feat"][:] = flat[meta["output_slices"]["hidden_state"]].reshape(packed.views["prev_feat"].shape)
outs.append(flat)
ref_target = os.environ.get("IQ_EGPU_REFERENCE")
if ref_target:
np.save(ref_target, outs[-1])
return out_path
if HOST:
os.replace(tmp, out_path)
return out_path
if not np.array_equal(outs[-1], baseline):
raise RuntimeError("policy outputs differ from baseline after pickle round trip")
if np.array_equal(outs[0], outs[-1]):
raise RuntimeError("policy outputs insensitive to inputs after pickle round trip")
if not all(np.isfinite(o).all() for o in outs):
raise RuntimeError("reloaded policy produced non-finite outputs")
from iqpilot.selfdrive.iqmodeld.parser import PhaseParser
from iqpilot.selfdrive.iqmodeld.tools.compile_supercombo import _slice_outputs, _validate_pose_outputs
_validate_pose_outputs(PhaseParser().parse_vision_outputs(_slice_outputs(outs[-1], meta["output_slices"])))
if os.environ.get("TC_OPT") != "0" and not os.environ.get("IQ_EGPU_SKIP_PARITY"):
_parity_check(meta["key"], outs[-1], _tc_off_reference(onnx_path, meta))
os.replace(tmp, out_path)
return out_path
DEFAULT_CAMERA_RESOLUTIONS: tuple[tuple[int, int], ...] = ((1928, 1208), (1344, 760))
def camera_nv12(cam_w: int, cam_h: int) -> tuple[int, int, int, int, int]:
from iqpilot.system.camerad.cameras.nv12_info import get_nv12_info
stride, y_height, uv_height, _ = get_nv12_info(cam_w, cam_h)
return (cam_w, cam_h, stride, y_height, uv_height)
def _fill_model_frame(packed, seed: int, res: tuple[int, int], model_w: int, model_h: int) -> None:
rng = np.random.default_rng(seed)
cam_w, cam_h = res
scale = np.array([[cam_w / model_w, 0.0, 0.0], [0.0, cam_h / model_h, 0.0], [0.0, 0.0, 1.0]], dtype=np.float32)
for name in ("tfm", "big_tfm"):
packed.views[name][:, :] = scale * (1.0 + 0.02 * rng.standard_normal((3, 3))).astype(np.float32)
for v in packed.frames.values():
v[:] = rng.integers(0, 256, size=v.shape, dtype=np.uint8)
packed.views["traffic_convention"][:] = [1, 0]
packed.views["action_t"][:] = [0.2, 0.3]
def compile_model_v3(meta: dict, onnx_path: str, out_path: str,
resolutions: tuple[tuple[int, int], ...] = DEFAULT_CAMERA_RESOLUTIONS) -> str:
from tinygrad.device import Device
from tinygrad.engine.jit import TinyJit
from tinygrad.nn.onnx import OnnxRunner
from iqpilot.selfdrive.iqmodeld.egpu_policy import (
MODEL_FORMAT, dump_oob, load_bundle, make_model_queues, make_run_model, make_run_policy, make_warp, model_size, nv12_copy_size,
)
if meta.get("split"):
raise RuntimeError(f"model {meta['key']} is a split model; eGPU compiles fused models only")
input_spec = {name: (tuple(shape), dtype) for name, (shape, dtype) in INPUT_SPEC.items()}
frame_skip = int(meta["frame_skip"])
hidden = meta["output_slices"]["hidden_state"]
device = Device.DEFAULT
model_w, model_h = model_size(input_spec)
runner = OnnxRunner(onnx_path)
run_policy = make_run_policy(runner, input_spec, frame_skip, device)
def step(jit, queues, packed, seed: int, res: tuple[int, int]) -> np.ndarray:
_fill_model_frame(packed, seed, res, model_w, model_h)
st = time.perf_counter()
out, = jit(**queues)
flat = out.numpy().reshape(-1)
print(f" model step(seed={seed}, {res[0]}x{res[1]}) {(time.perf_counter() - st) * 1e3:6.1f} ms")
packed.views["prev_feat"][:] = flat[hidden].reshape(packed.views["prev_feat"].shape)
return flat
def run_three(jit, fcs: int, res: tuple[int, int]) -> list[np.ndarray]:
queues, packed = make_model_queues(input_spec, frame_skip, device, fcs)
return [step(jit, queues, packed, SEED + i, res) for i in range(3)]
jits: dict[tuple[int, int], object] = {}
sizes: dict[tuple[int, int], int] = {}
nv12s: dict[tuple[int, int], tuple[int, int, int, int, int]] = {}
baselines: dict[tuple[int, int], np.ndarray] = {}
for res in resolutions:
nv12 = camera_nv12(*res)
fcs = nv12_copy_size(nv12[2], nv12[3], nv12[4])
jit = TinyJit(make_run_model(make_warp(nv12, model_w, model_h, device), run_policy, input_spec, fcs, device), prune=True)
print(f"capture + replay {res[0]}x{res[1]} (frame copy {fcs} B)")
baseline = run_three(jit, fcs, res)[-1]
if baseline.shape[0] != meta["output_len"]:
raise RuntimeError(f"model output length {baseline.shape[0]} != registry {meta['output_len']}")
if not HOST and not np.isfinite(baseline).all():
raise RuntimeError("compiled model produced non-finite outputs")
jits[res], sizes[res], nv12s[res], baselines[res] = jit, fcs, nv12, baseline
bundle = {
"format": MODEL_FORMAT,
"run_model": jits,
"frame_copy_size": sizes,
"nv12": nv12s,
"model_key": meta["key"],
"model_sha256": meta["sha256"],
"output_len": int(meta["output_len"]),
"frame_skip": frame_skip,
"input_spec": input_spec,
"input_device": device,
}
os.makedirs(os.path.dirname(out_path), exist_ok=True)
tmp = out_path + ".part"
print("serialize (out-of-band buffers)")
with open(tmp, "wb") as f:
dump_oob(bundle, f)
del bundle, jits, run_policy, runner
gc.collect()
print("reload + validate")
loaded = load_bundle(tmp)
outs = {res: run_three(loaded["run_model"][res], loaded["frame_copy_size"][res], res) for res in resolutions}
ref_target = os.environ.get("IQ_EGPU_REFERENCE")
if ref_target:
np.savez(ref_target, **{f"{w}x{h}": outs[(w, h)][-1] for (w, h) in resolutions})
return out_path
if HOST:
os.replace(tmp, out_path)
return out_path
for res in resolutions:
if not np.array_equal(outs[res][-1], baselines[res]):
raise RuntimeError(f"model outputs differ from baseline after pickle round trip ({res[0]}x{res[1]})")
if np.array_equal(outs[res][0], outs[res][-1]):
raise RuntimeError(f"model outputs insensitive to inputs after pickle round trip ({res[0]}x{res[1]})")
if not all(np.isfinite(o).all() for o in outs[res]):
raise RuntimeError(f"reloaded model produced non-finite outputs ({res[0]}x{res[1]})")
from iqpilot.selfdrive.iqmodeld.parser import PhaseParser
from iqpilot.selfdrive.iqmodeld.tools.compile_supercombo import _slice_outputs, _validate_pose_outputs
_validate_pose_outputs(PhaseParser().parse_vision_outputs(_slice_outputs(outs[resolutions[0]][-1], meta["output_slices"])))
if os.environ.get("TC_OPT") != "0" and not os.environ.get("IQ_EGPU_SKIP_PARITY"):
ref = _tc_off_reference(onnx_path, meta, fmt=3, resolutions=resolutions)
for (w, h) in resolutions:
_parity_check(meta["key"], outs[(w, h)][-1], ref[f"{w}x{h}"], label=f"{w}x{h} ")
os.replace(tmp, out_path)
return out_path
def _parse_resolution(text: str) -> tuple[int, int]:
w, h = text.lower().split("x")
return int(w), int(h)
def main() -> None:
p = argparse.ArgumentParser()
p.add_argument("--model", default=None, help=f"registry key, one of {sorted(EGPU_MODELS)}")
p.add_argument("--onnx", default=None)
p.add_argument("--output", default=None)
p.add_argument("--progress-param", default=None)
p.add_argument("--progress-base", type=float, default=None)
p.add_argument("--progress-span", type=float, default=0.0)
p.add_argument("--format", type=int, default=3, choices=(1, 2, 3),
help="3 = warp on the dock from raw NV12 (comma master); 2 = device-warped policy bundle")
p.add_argument("--camera-resolutions", type=_parse_resolution, nargs="+", default=list(DEFAULT_CAMERA_RESOLUTIONS),
help="WxH camera sizes bundled into a format-3 artifact")
p.add_argument("--host", action="store_true", help="compile on a mock dock (no AMD hardware); outputs need a dock parity gate")
p.add_argument("--arch", default=None, help="target gfx arch for --host")
p.add_argument("--tc-off", action="store_true", help="disable tensor-core kernels (conservative; auto-set on parity failure)")
args = p.parse_args()
if args.host and args.format == 1:
raise SystemExit("--host supports formats 2 and 3 only")
if args.model is not None:
if args.model in EGPU_MODELS:
meta = get_egpu_model(args.model)
else:
from iqpilot.common.params import Params
meta = resolve_egpu_model(Params(), args.model)
if meta is None:
raise SystemExit(f"unknown model {args.model!r}: not a built-in ({sorted(EGPU_MODELS)}) and not in the synced catalog")
else:
meta = get_egpu_model()
set_input_spec(meta)
onnx_path = args.onnx or local_onnx(meta)
if onnx_path is None or not os.path.isfile(onnx_path):
raise SystemExit(f"onnx not found for {meta['key']}; pass --onnx or let iqegpumodeld download it first")
stop = None
sampler = None
if args.progress_param and args.progress_base is not None:
import threading
stop = threading.Event()
sampler = threading.Thread(target=_progress_sampler,
args=(args.progress_param, args.progress_base, args.progress_span, stop),
daemon=True)
sampler.start()
try:
if args.format == 3:
from functools import partial
build = partial(compile_model_v3, resolutions=tuple(args.camera_resolutions))
else:
build = compile_policy_model if args.format == 2 else compile_model
try:
out = build(meta, onnx_path, args.output or egpu_pkl_path(meta))
except _ParityFail as e:
if os.environ.get("TC_OPT") == "0" or args.format == 1:
raise
print(f"{e}\nretrying compile with tensor cores disabled", flush=True)
os.environ["TC_OPT"] = "0"
os.environ["IQ_EGPU_TC_OFF"] = "1"
out = build(meta, onnx_path, args.output or egpu_pkl_path(meta))
finally:
if stop is not None:
stop.set()
if sampler is not None:
sampler.join(timeout=2)
print(f"saved eGPU jit to {out} ({os.path.getsize(out) / 1e6:.2f} MB)")
if __name__ == "__main__":
main()

View File

@@ -0,0 +1,9 @@
"""
Copyright © IQ.Lvbs, apart of Project Teal Lvbs, All Rights Reserved, licensed under https://konn3kt.com/tos/
"""
from iqpilot.selfdrive.iqmodeld.tools.compile_warp import MODEL_SIZE, compile_warp, main
__all__ = ["MODEL_SIZE", "compile_warp", "main"]
if __name__ == "__main__":
main()

View File

@@ -0,0 +1,125 @@
"""
Copyright © IQ.Lvbs, apart of Project Teal Lvbs, All Rights Reserved, licensed under https://konn3kt.com/tos/
"""
import os
import pickle
import re
import shutil
import sys
import time
import numpy as np
if "JIT_BATCH_SIZE" not in os.environ:
os.environ["JIT_BATCH_SIZE"] = "0"
from tinygrad import Context, Device, GlobalCounters, Tensor, TinyJit, dtypes
from tinygrad.helpers import DEBUG, getenv
from tinygrad.nn.onnx import OnnxRunner
from tinygrad.uop.ops import Ops
def compile_model(onnx_file, output):
run_onnx = OnnxRunner(onnx_file)
print("loaded model")
input_shapes = {name: spec.shape for name, spec in run_onnx.graph_inputs.items()}
input_types = {name: spec.dtype for name, spec in run_onnx.graph_inputs.items()}
input_types = {key: dtypes.float32 if value is dtypes.float16 else value for key, value in input_types.items()}
input_shapes = {key: tuple(value if isinstance(value, int) else 1 for value in shape) for key, shape in input_shapes.items()}
Tensor.manual_seed(100)
inputs = {
key: Tensor(Tensor.randn(*shape, dtype=input_types[key]).mul(8).realize().numpy(), device="NPY")
for key, shape in sorted(input_shapes.items())
}
if not getenv("NPY_IMG"):
inputs = {key: Tensor(value.numpy(), device=Device.DEFAULT).realize() if "img" in key else value for key, value in inputs.items()}
print("created tensors")
run_onnx_jit = TinyJit(
lambda **kwargs: next(iter(run_onnx({key: value.to(Device.DEFAULT) for key, value in kwargs.items()}).values())).cast("float32"),
prune=True,
)
test_value = None
for iteration in range(3):
GlobalCounters.reset()
print(f"run {iteration}")
with Context(DEBUG=max(DEBUG.value, 2 if iteration == 2 else 1), OPENPILOT_HACKS=1):
result = run_onnx_jit(**inputs).numpy()
if iteration == 1:
test_value = np.copy(result)
kernel_asts = {Ops.PROGRAM}
kernel_calls = [
node for node in run_onnx_jit.captured.linear.toposort(gate=lambda value: value.op not in kernel_asts)
if node.op is Ops.CALL and node.src[0].op in kernel_asts
]
print(f"captured {len(kernel_calls)} kernels")
np.testing.assert_equal(test_value, result, "JIT run failed")
print("jit run validated")
kernel_count = 0
read_image_count = 0
gated_read_image_count = 0
for call in kernel_calls:
_, _, source, _ = call.src[0].src
rendered = source.arg
kernel_count += 1
read_image_count += rendered.count("read_image")
gated_read_image_count += rendered.count("?read_image")
for value in (match.group(1) for match in re.finditer(r"(val\d+)\s*=\s*read_imagef\(", rendered)):
if re.search(fr"[?:]{value}\.[xyzw]", rendered):
gated_read_image_count += 1
print(f"{kernel_count=}, {read_image_count=}, {gated_read_image_count=}")
expected = {
"kernel count": (kernel_count, getenv("ALLOWED_KERNEL_COUNT", -1)),
"read image count": (read_image_count, getenv("ALLOWED_READ_IMAGE", -1)),
"gated read image count": (gated_read_image_count, getenv("ALLOWED_GATED_READ_IMAGE", -1)),
}
for name, (actual, allowed) in expected.items():
if allowed != -1:
assert actual == allowed, f"different {name}: {actual}, expected {allowed}"
with open(output, "wb") as handle:
pickle.dump(run_onnx_jit, handle)
print(f"model size is {os.path.getsize(onnx_file) / 1e6:.2f}M")
print(f"pkl size is {os.path.getsize(output) / 1e6:.2f}M")
return run_onnx_jit, inputs, test_value
def test_compiled(run, inputs, test_value):
step_times = []
for _ in range(20):
start = time.perf_counter()
output = run(**inputs)
queued = time.perf_counter()
value = output.numpy()
end = time.perf_counter()
step_times.append((end - start) * 1e3)
print(f"enqueue {(queued - start) * 1e3:6.2f} ms -- total run {step_times[-1]:6.2f} ms")
minimum = getenv("ASSERT_MIN_STEP_TIME", 0.0)
if minimum:
assert min(step_times) < minimum, f"expected minimum step time below {minimum} ms, got {min(step_times)} ms"
np.testing.assert_equal(test_value, value)
changed_inputs = {key: Tensor(item.numpy() * 2, device=item.device) for key, item in inputs.items()}
changed_value = run(**changed_inputs).numpy()
np.testing.assert_raises(AssertionError, np.testing.assert_array_equal, value, changed_value)
if __name__ == "__main__":
model_path = sys.argv[1]
output_path = sys.argv[2]
if stash := os.environ.get("IQPILOT_MODEL_STASH"):
stashed_model = os.path.join(stash, os.path.basename(output_path))
if os.path.isfile(stashed_model) and os.path.getsize(stashed_model) > 0:
os.makedirs(os.path.dirname(output_path) or ".", exist_ok=True)
shutil.copyfile(stashed_model, output_path)
print(f"restored device-compiled model: {output_path}")
sys.exit(0)
_, input_values, expected_value = compile_model(model_path, output_path)
with open(output_path, "rb") as compiled_file:
compiled_model = pickle.load(compiled_file)
test_compiled(compiled_model, input_values, expected_value)

View File

@@ -62,8 +62,8 @@ WARP_DEVICE = os.getenv("WARP_DEV")
def _read_shared_copy(path: str) -> str:
from openpilot.common.file_chunker import read_file_chunked
from openpilot.system.hardware.hw import Paths
from iqpilot.common.file_chunker import read_file_chunked
from iqpilot.system.hardware.hw import Paths
shm_path = os.path.join(Paths.shm_path(), os.path.basename(path))
atexit.register(lambda: os.path.exists(shm_path) and os.remove(shm_path))
@@ -127,8 +127,8 @@ def _project_pixels(src_flat, inverse_matrix, dst_shape, src_shape, stride_pad,
dst_w, dst_h = dst_shape
src_h, src_w = src_shape
x_coords = Tensor.arange(dst_w, device=WARP_DEVICE).reshape(1, dst_w).expand(dst_h, dst_w).reshape(-1)
y_coords = Tensor.arange(dst_h, device=WARP_DEVICE).reshape(dst_h, 1).expand(dst_h, dst_w).reshape(-1)
x_coords = Tensor.arange(dst_w).to(WARP_DEVICE).reshape(1, dst_w).expand(dst_h, dst_w).reshape(-1)
y_coords = Tensor.arange(dst_h).to(WARP_DEVICE).reshape(dst_h, 1).expand(dst_h, dst_w).reshape(-1)
src_x = inverse_matrix[0, 0] * x_coords + inverse_matrix[0, 1] * y_coords + inverse_matrix[0, 2]
src_y = inverse_matrix[1, 0] * x_coords + inverse_matrix[1, 1] * y_coords + inverse_matrix[1, 2]
@@ -352,8 +352,8 @@ def _arg_parser() -> argparse.ArgumentParser:
def main(argv: list[str] | None = None) -> int:
from openpilot.iqpilot.selfdrive.iqmodeld.metadata import build_metadata_record
from openpilot.system.camerad.cameras.nv12_info import get_nv12_info
from iqpilot.selfdrive.iqmodeld.metadata import build_metadata_record
from iqpilot.system.camerad.cameras.nv12_info import get_nv12_info
args = _arg_parser().parse_args(argv)
model_w, model_h = args.model_size

View File

@@ -59,7 +59,6 @@ def warp_perspective_tinygrad(src_flat, M_inv, dst_shape, src_shape, stride_pad,
x = Tensor.arange(w_dst).reshape(1, w_dst).expand(h_dst, w_dst).reshape(-1)
y = Tensor.arange(h_dst).reshape(h_dst, 1).expand(h_dst, w_dst).reshape(-1)
# inline 3x3 matmul as elementwise to avoid reduce op (enables fusion with gather)
src_x = M_inv[0, 0] * x + M_inv[0, 1] * y + M_inv[0, 2]
src_y = M_inv[1, 0] * x + M_inv[1, 1] * y + M_inv[1, 2]
src_w = M_inv[2, 0] * x + M_inv[2, 1] * y + M_inv[2, 2]
@@ -100,9 +99,7 @@ def make_frame_prepare(nv12: NV12Frame, model_w, model_h):
stride_pad = stride - cam_w
def frame_prepare_tinygrad(input_frame, M_inv):
# UV_SCALE @ M_inv @ UV_SCALE_INV simplifies to elementwise scaling
M_inv_uv = M_inv * Tensor([[1.0, 1.0, 0.5], [1.0, 1.0, 0.5], [2.0, 2.0, 1.0]], device=WARP_DEV)
# deinterleave NV12 UV plane (UVUV... -> separate U, V)
uv = input_frame[uv_offset:uv_offset + uv_height * stride].reshape(uv_height, stride)
with Context(SPLIT_REDUCEOP=0):
y = warp_perspective_tinygrad(input_frame[:cam_h*stride],
@@ -142,7 +139,6 @@ def get_policy_npy_shapes(input_shapes):
tc = input_shapes['traffic_convention'] # (1, 2)
at = input_shapes['action_t'] # (1, 2)
fb = input_shapes['features_buffer'] # (1, 24, 512)
# TODO prev_feat shouldn't exist and be handled inside the JIT, but corrupt on QCOM for now
shapes = {'desire': (dp[2],), 'traffic_convention': tuple(tc), 'action_t': tuple(at), 'prev_feat': (fb[0], fb[2])}
return shapes, [math.prod(s) for s in shapes.values()]
@@ -155,7 +151,6 @@ def make_input_queues(input_shapes, frame_skip, device):
shapes, sizes = get_policy_npy_shapes(input_shapes)
packed_npy_inputs = np.zeros(sum(sizes), dtype=np.float32)
# views into the packed inputs, to be refilled at runtime
npy.update({k: v.reshape(s) for (k, s), v in zip(shapes.items(), np.split(packed_npy_inputs, np.cumsum(sizes[:-1])), strict=True)})
input_queues.update({
'feat_q': Tensor(np.zeros((frame_skip * fb[1], fb[0], fb[2]), dtype=np.float32), device=device).contiguous().realize(),
@@ -284,7 +279,7 @@ def _slice_outputs(model_outputs: np.ndarray, output_slices: dict[str, slice]) -
def _validate_pose_outputs(parsed_outputs: dict[str, np.ndarray]) -> None:
from openpilot.selfdrive.locationd.locationd import MIN_STD_SANITY_CHECK, ROTATION_SANITY_CHECK, TRANS_SANITY_CHECK
from iqpilot.selfdrive.locationd.locationd import MIN_STD_SANITY_CHECK, ROTATION_SANITY_CHECK, TRANS_SANITY_CHECK
required = (
'pose', 'pose_stds', 'wide_from_device_euler', 'wide_from_device_euler_stds',
@@ -326,7 +321,7 @@ def _validate_pose_outputs(parsed_outputs: dict[str, np.ndarray]) -> None:
def validate_supercombo_release(run_policy_jit, model_runner, model_metadata, frame_skip, expected_device: str) -> None:
from openpilot.iqpilot.selfdrive.iqmodeld.parser import PhaseParser
from iqpilot.selfdrive.iqmodeld.parser import PhaseParser
direct_fn = make_run_policy(model_runner, model_metadata, frame_skip)
parser = PhaseParser()
@@ -370,8 +365,8 @@ def _parse_size(s):
def read_file_chunked_to_shm(path):
from openpilot.common.file_chunker import read_file_chunked
from openpilot.system.hardware.hw import Paths
from iqpilot.common.file_chunker import read_file_chunked
from iqpilot.system.hardware.hw import Paths
with tempfile.NamedTemporaryFile(prefix='compile_modeld_', dir=Paths.shm_path(), delete=False) as f:
f.write(read_file_chunked(path))
tmp_path = f.name
@@ -381,8 +376,8 @@ def read_file_chunked_to_shm(path):
if __name__ == "__main__":
from tinygrad.nn.onnx import OnnxRunner
from openpilot.system.camerad.cameras.nv12_info import get_nv12_info
from openpilot.iqpilot.selfdrive.iqmodeld.metadata import build_metadata_record
from iqpilot.system.camerad.cameras.nv12_info import get_nv12_info
from iqpilot.selfdrive.iqmodeld.metadata import build_metadata_record
p = argparse.ArgumentParser()
p.add_argument('--model-size', type=_parse_size, required=True, help='model input WxH')
p.add_argument('--camera-resolutions', type=_parse_size, nargs='+', required=True,

View File

@@ -0,0 +1,119 @@
#!/usr/bin/env python3
"""
Copyright © IQ.Lvbs, apart of Project Teal Lvbs, All Rights Reserved, licensed under https://konn3kt.com/tos/
Compile the backend-neutral warp-only artifact: NV12 camera frames + 3x3
transforms -> (2, 6, model_h/2, model_w/2) uint8 warped tensor, on the device
GPU (QCOM). maciqmodeld runs this locally
and feed the output to their backend, so the big model's image pipeline is
bit-identical to comma's fused pkl warp stage.
Run ON the device (needs the QCOM backend):
cd /data/openpilot && DEV=QCOM WARP_DEV=QCOM IMAGE=1 FLOAT16=1 NOLOCALS=1 JIT_BATCH_SIZE=0 \
python3 iqpilot/selfdrive/iqmodeld/tools/compile_warp.py \
--camera-resolutions 1928x1208 --output /data/models/emac_warp.pkl
The artifact is then split per-resolution into Paths.model_root().
"""
from __future__ import annotations
import argparse
import hashlib
import os
import pickle
from functools import partial
import numpy as np
SELFTEST_SEED = 20260817
from iqpilot.selfdrive.iqmodeld.temporal_state import DEFAULT_FRAME_SKIP, MODEL_INPUT_SPEC
from iqpilot.selfdrive.iqmodeld.tools.compile_supercombo import (
NV12Frame, WARP_INPUTS, compile_jit, make_random_images, make_warp, make_warp_input_queues,
)
MODEL_SIZE = (MODEL_INPUT_SPEC["img"][0][3] * 2, MODEL_INPUT_SPEC["img"][0][2] * 2) # (512, 256)
def _parse_size(s: str) -> tuple[int, int]:
w, h = s.lower().split("x")
return int(w), int(h)
def compile_warp(cam_w: int, cam_h: int, out_path: str | None = None,
frame_skip: int = DEFAULT_FRAME_SKIP) -> str:
"""Compile the warp-only QCOM JIT for one camera resolution and write the pkl.
Returns the artifact path. Callable from the workers so a fresh device
self-provisions the warp instead of erroring — needs the QCOM backend."""
# the QCOM warp env must be set before tinygrad is imported here
os.environ.setdefault("DEV", "QCOM")
os.environ.setdefault("WARP_DEV", "QCOM")
os.environ.setdefault("IMAGE", "1")
os.environ.setdefault("FLOAT16", "1")
os.environ.setdefault("NOLOCALS", "1")
os.environ.setdefault("JIT_BATCH_SIZE", "0")
from tinygrad.engine.jit import TinyJit
from iqpilot.system.camerad.cameras.nv12_info import get_nv12_info
from iqpilot.system.hardware.hw import Paths
model_w, model_h = MODEL_SIZE
input_shapes = {name: shape for name, (shape, _) in MODEL_INPUT_SPEC.items()}
nv12 = NV12Frame(cam_w, cam_h, *get_nv12_info(cam_w, cam_h))
make_random_warp_inputs = partial(make_random_images, keys=["frame", "big_frame"],
shape=nv12.size, device=os.getenv("WARP_DEV"))
warp_jit = TinyJit(make_warp(nv12, model_w, model_h, frame_skip), prune=True)
make_warp_queues = partial(make_warp_input_queues, input_shapes, frame_skip)
compiled = compile_jit(warp_jit, make_random_warp_inputs, WARP_INPUTS, make_warp_queues)
# historical artifact name: already-provisioned devices keep their warp
out_path = out_path or os.path.join(Paths.model_root(), f"emac_warp_{cam_w}x{cam_h}_tinygrad.pkl")
os.makedirs(os.path.dirname(out_path), exist_ok=True)
tmp = out_path + ".part"
bundle = {(cam_w, cam_h): compiled, "frame_skip": frame_skip, "model_size": MODEL_SIZE}
bundle["selftest"] = selftest_digest(compiled, cam_w, cam_h, nv12.size)
with open(tmp, "wb") as f:
pickle.dump(bundle, f)
os.replace(tmp, out_path) # atomic: a reader never sees a half-written pkl
return out_path
def main() -> None:
p = argparse.ArgumentParser()
p.add_argument("--camera-resolutions", type=_parse_size, nargs="+", default=[(1928, 1208)])
p.add_argument("--output", default=None)
p.add_argument("--frame-skip", type=int, default=DEFAULT_FRAME_SKIP)
args = p.parse_args()
for cam_w, cam_h in args.camera_resolutions:
out = compile_warp(cam_w, cam_h, args.output, frame_skip=args.frame_skip)
print(f"saved warp JIT to {out} ({os.path.getsize(out) / 1e6:.2f} MB)")
if __name__ == "__main__":
main()
def selftest_inputs(cam_w: int, cam_h: int, nv12_size: int):
"""A fixed synthetic frame pair and pair of matrices. Deterministic so the
digest is reproducible on the device that compiled the artifact."""
rng = np.random.default_rng(SELFTEST_SEED)
frame = rng.integers(0, 256, nv12_size, dtype=np.uint8)
big_frame = rng.integers(0, 256, nv12_size, dtype=np.uint8)
tfm = np.array([[0.7, 0.02, 300.0], [0.01, 0.7, 240.0], [0.0, 0.0, 1.0]], dtype=np.float32)
big_tfm = np.array([[0.5, 0.01, 380.0], [0.02, 0.5, 300.0], [0.0, 0.0, 1.0]], dtype=np.float32)
return frame, big_frame, tfm, big_tfm
def selftest_digest(compiled, cam_w: int, cam_h: int, nv12_size: int) -> str:
"""Hash the warp's output for a fixed input.
A warp artifact pinned to one tinygrad can still unpickle under another and
then compute silently wrong, which reaches the model as a garbage image and
looks like a bad model rather than a stale artifact. A version string cannot
see that; running it can."""
from tinygrad.tensor import Tensor
frame, big_frame, tfm, big_tfm = selftest_inputs(cam_w, cam_h, nv12_size)
dev = os.getenv("WARP_DEV") or "QCOM"
out = compiled(tfm=Tensor(tfm, device="NPY").realize(),
big_tfm=Tensor(big_tfm, device="NPY").realize(),
frame=Tensor(frame, device=dev).realize(),
big_frame=Tensor(big_frame, device=dev).realize())
return hashlib.sha256(out.numpy().astype(np.uint8).tobytes()).hexdigest()

View File

@@ -0,0 +1,44 @@
"""
Copyright © IQ.Lvbs, apart of Project Teal Lvbs, All Rights Reserved, licensed under https://konn3kt.com/tos/
"""
from __future__ import annotations
import argparse
import gc
import os
os.environ.setdefault("DEV", "USB+AMD:LLVM")
os.environ.setdefault("GMMU", "0")
from iqpilot.selfdrive.iqmodeld.egpu_helpers import patch_tinygrad_fetch_fw
from iqpilot.selfdrive.iqmodeld.egpu_policy import dump_oob, is_oob, load_bundle
def convert(src: str, dst: str) -> str:
patch_tinygrad_fetch_fw()
if is_oob(src):
if src != dst:
os.replace(src, dst)
return dst
bundle = load_bundle(src)
tmp = dst + ".part"
with open(tmp, "wb") as f:
dump_oob(bundle, f)
del bundle
gc.collect()
load_bundle(tmp)
os.replace(tmp, dst)
return dst
def main() -> None:
p = argparse.ArgumentParser()
p.add_argument("src")
p.add_argument("--out", default=None)
args = p.parse_args()
out = convert(args.src, args.out or args.src)
print(f"converted -> {out} ({os.path.getsize(out) / 1e6:.1f} MB)")
if __name__ == "__main__":
main()

View File

@@ -272,8 +272,8 @@ def _parse_size(text: str) -> tuple[int, int]:
def _read_file_to_shared_memory(path: str) -> str:
from openpilot.common.file_chunker import read_file_chunked
from openpilot.system.hardware.hw import Paths
from iqpilot.common.file_chunker import read_file_chunked
from iqpilot.system.hardware.hw import Paths
shm_path = os.path.join(Paths.shm_path(), os.path.basename(path))
atexit.register(lambda: os.path.exists(shm_path) and os.remove(shm_path))
@@ -295,8 +295,8 @@ def _arg_parser() -> argparse.ArgumentParser:
def main(argv: list[str] | None = None) -> int:
from openpilot.iqpilot.selfdrive.iqmodeld.metadata import build_metadata_record
from openpilot.system.camerad.cameras.nv12_info import get_nv12_info
from iqpilot.selfdrive.iqmodeld.metadata import build_metadata_record
from iqpilot.system.camerad.cameras.nv12_info import get_nv12_info
args = _arg_parser().parse_args(argv)
model_w, model_h = args.model_size

View File

@@ -0,0 +1,64 @@
"""
Copyright © IQ.Lvbs, apart of Project Teal Lvbs, All Rights Reserved, licensed under https://konn3kt.com/tos/
"""
from __future__ import annotations
import os
import sys
import tempfile
DEFAULT_ARCH = "gfx1200"
MOCK_DEV = "MOCKUSB+AMD:LLVM"
def tinygrad_tree() -> str:
override = os.environ.get("IQ_TINYGRAD_TREE")
if override:
return override
here = os.path.dirname(os.path.abspath(__file__))
root = os.path.abspath(os.path.join(here, "..", "..", "..", ".."))
return os.path.join(root, "components", "tinygrad")
def activate(arch: str = DEFAULT_ARCH, execute: bool = False) -> None:
assert "tinygrad" not in sys.modules, "egpu_host_mock.activate must run before tinygrad is imported"
os.environ["DEV"] = f"{MOCK_DEV}:{arch}"
tree = tinygrad_tree()
if tree not in sys.path:
sys.path.insert(0, tree)
from test.mockgpu.am import amgpu
# The mock dock models 512MB VRAM; big-model weights alone exceed that. Must be set before amdriver binds it.
amgpu.VRAM_SIZE = int(os.environ.get("IQ_MOCK_VRAM_GB", "4")) << 30
from tinygrad.runtime.autogen import libc
if sys.platform == "darwin":
# A Homebrew-LLVM gfx1200 kernel (no s_code_end padding) hung a real dock; ship only container-built artifacts.
print("egpu_host_mock: native macOS LLVM output is for tests only; use scripts/iqpilot/host_egpu_compile_docker.sh for artifacts",
file=sys.stderr)
def memfd_create(name, flags):
fd, path = tempfile.mkstemp(prefix=b"iq_mock_" + bytes(name) + b"_")
os.unlink(path)
return fd
libc.memfd_create = memfd_create
if not hasattr(libc, "MFD_CLOEXEC"):
libc.MFD_CLOEXEC = 1
if not execute:
import ctypes
from test.mockgpu.amd import amdgpu
amdgpu.remu.run_asm = lambda *args, **kwargs: 0
pm4_wait = amdgpu.PM4Executor._exec_wait_reg_mem
sdma_poll = amdgpu.SDMAExecutor._execute_poll_regmem
# Without kernel execution no memory wait carries information; a blocked wait would need a host write to re-poll it.
def pm4_wait_passthrough(self, n):
if not pm4_wait(self, n):
self.rptr[0] += 7
return True
def sdma_poll_passthrough(self):
if not sdma_poll(self):
self.rptr[0] += ctypes.sizeof(amdgpu.sdma_pkts.poll_regmem)
return True
amdgpu.PM4Executor._exec_wait_reg_mem = pm4_wait_passthrough
amdgpu.SDMAExecutor._execute_poll_regmem = sdma_poll_passthrough

View File

@@ -13,7 +13,7 @@ from pathlib import Path
import onnx
from openpilot.system.hardware.hw import Paths
from iqpilot.system.hardware.hw import Paths
_MODEL_STEMS = ("driving_off_policy", "driving_on_policy", "driving_policy", "driving_vision")

View File

@@ -0,0 +1,64 @@
"""
Copyright © IQ.Lvbs, apart of Project Teal Lvbs, All Rights Reserved, licensed under https://konn3kt.com/tos/
"""
from __future__ import annotations
import argparse
import os
import pickletools
import struct
from iqpilot.selfdrive.iqmodeld.egpu_policy import OOB_MAGIC
MIN_OOB_BYTES = 1 << 16
NEXT_BUFFER = b"\x97"
READONLY_BUFFER = b"\x98"
def rewrite_oob(src: str, dst: str, min_bytes: int = MIN_OOB_BYTES) -> tuple[int, int]:
# tinygrad pickles device buffers as PickleBuffers, which land in-band as BYTEARRAY8/BINBYTES8
# without a buffer_callback; moving those opcodes out-of-band is byte-for-byte what a protocol-5
# dump with a buffer_callback produces, so nothing has to be unpickled (no dock needed).
with open(src, "rb") as f:
data = f.read()
ops = list(pickletools.genops(data))
proto = next((arg for op, arg, _ in ops if op.name == "PROTO"), 0)
if proto < 5:
raise ValueError(f"{src} is pickle protocol {proto}; out-of-band buffers need protocol 5")
moved = 0
tmp = dst + ".part"
with open(tmp, "wb") as out, open(tmp + ".buf", "wb") as bufs:
ops_stream = bytearray()
for i, (op, arg, pos) in enumerate(ops):
end = ops[i + 1][2] if i + 1 < len(ops) else len(data)
if op.name in ("BYTEARRAY8", "BINBYTES8", "BINBYTES") and len(arg) >= min_bytes:
ops_stream += NEXT_BUFFER
if op.name != "BYTEARRAY8":
ops_stream += READONLY_BUFFER
bufs.write(struct.pack("<q", len(arg)))
bufs.write(arg)
moved += 1
else:
ops_stream += data[pos:end]
out.write(OOB_MAGIC)
out.write(struct.pack("<q", len(ops_stream)))
out.write(ops_stream)
with open(tmp, "ab") as out, open(tmp + ".buf", "rb") as bufs:
while chunk := bufs.read(1 << 24):
out.write(chunk)
os.remove(tmp + ".buf")
os.replace(tmp, dst)
return moved, len(ops)
def main() -> None:
p = argparse.ArgumentParser()
p.add_argument("src")
p.add_argument("dst")
args = p.parse_args()
moved, total = rewrite_oob(args.src, args.dst)
print(f"{args.dst}: moved {moved} buffers out-of-band ({total} opcodes, {os.path.getsize(args.dst) / 1e6:.1f} MB)")
if __name__ == "__main__":
main()