From a9bf9dbe623ad934a528917ed4c8bab39db6445d Mon Sep 17 00:00:00 2001 From: Tristan Ross Date: Sun, 27 Sep 2026 18:25:29 -0700 Subject: [PATCH 1/2] feat: selectable gpu ordinal and cross-client peer imports --- src/nvidia/compute.zig | 337 +++++++++++++++++++++++++- src/nvidia/rm.zig | 119 +++++++++ src/nvidia/sdk.zig | 84 +++++++ src/nvidia/transport/freestanding.zig | 30 +++ src/nvidia/transport/linux.zig | 42 ++++ 5 files changed, 611 insertions(+), 1 deletion(-) diff --git a/src/nvidia/compute.zig b/src/nvidia/compute.zig index a751cb0..09137e3 100644 --- a/src/nvidia/compute.zig +++ b/src/nvidia/compute.zig @@ -324,6 +324,19 @@ pub const Buffer = struct { } }; +/// A peer allocation this runner can address. The bytes live on the other +/// runner's device, so there is no CPU mapping, except when the underlying +/// memory is host RAM, where `importPeer` maps it to the CPU anyway so +/// `copyFromPeer` can reach it with a memcpy (see the timings above that +/// function). `va` is a normal address in THIS runner's own GPU VA space: +/// importPeer already mapped the dup there, so a copy-engine transfer needs +/// no cross-address-space handling at all. +pub const Peer = struct { + va: u64, + size: u64, + memory: rm.Memory, +}; + /// The transfer size at which the copy engine overtakes the inline path on this /// hardware. Below it, put the payload in the pushbuffer with /// `Stream.uploadInline`; above it, hand the copy engine a source buffer with @@ -352,6 +365,31 @@ pub fn preferInlineUpload(bytes: u32) bool { return bytes < inline_upload_limit_bytes; } +/// Which path `Runner.copyFromPeer` should take, by where the two allocations +/// live. On an RTX 5070 on 2026-09-27, nanoseconds per transfer +/// (memcpy vs the copy engine, system<->system and VRAM<->VRAM, submission and +/// fence included for the copy engine column): +/// +/// bytes | system memcpy | system CE | VRAM memcpy* | VRAM CE +/// 4096 | 146 | 11076 | 438555 | 9214 +/// 65536 | 1746 | 20608 | 6988728 | 10880 +/// 1048576 | 77302 | 183415 | 137900445 | 41574 +/// +/// *the VRAM memcpy column reads a scratch BAR1 CPU mapping made only for +/// this measurement. copyFromPeer never maps VRAM to the CPU on the real +/// path, since the copy engine already reaches it directly. +/// +/// System memory needs no submission at all for a memcpy, so it wins across +/// the whole range, by two orders of magnitude at every size. VRAM +/// inverts it even harder than expected: the BAR1 CPU read is not merely +/// slower than the copy engine's own submission, it is 15x to 3300x slower, +/// climbing fast with size (a 1 MiB BAR1 read takes 138 ms, against 42 us on +/// the copy engine). Both halves of the rule hold decisively; neither needed +/// a threshold, only a location check. +pub fn preferMemcpyForPeerCopy(dst_location: rm.Memory.Location, src_location: rm.Memory.Location) bool { + return dst_location != .vram and src_location != .vram; +} + /// A ready-to-use compute context: a GPFIFO channel bound to the compute class, /// a VA space with a bump allocator behind `alloc`, and the code, descriptor, /// pushbuffer and semaphore a launch needs. Runner owns every CPU and GPU @@ -408,9 +446,28 @@ pub const Runner = struct { /// `error.SkipZigTest` when no GPU is reachable, so tests skip instead of /// failing on a machine without one. pub fn init() !Runner { + return initOn(0) catch |err| switch (err) { + // The faults `allocDevice` reports for GPU 0 all mean one thing to a caller + // that wants any GPU at all: there is not one here. Every test in this file + // depends on that, so keep it a skip. `initOn` reports them as real errors, + // because a caller that names an ordinal asks about one specific device. + error.NoDevice, + error.OpenFailed, + error.IoctlFailed, + error.RmAllocFailed, + => error.SkipZigTest, + else => err, + }; + } + + /// Like `init`, but on the `ordinal`-th GPU the driver knows about + /// (`rm.Client.deviceCount`). Only a completely unreachable driver skips + /// the test; a bad ordinal on a reachable driver is a real, reportable + /// error (`error.NoDevice`), not a skip. + pub fn initOn(ordinal: u32) !Runner { var client = rm.Client.open() catch return error.SkipZigTest; errdefer client.deinit(); - const dev = client.allocDevice(0) catch return error.SkipZigTest; + const dev = try client.allocDevice(ordinal); errdefer client.freeDevice(dev); const vaspace = try client.allocVaSpace(dev); @@ -534,6 +591,33 @@ pub const Runner = struct { return error.CopyTimeout; } + /// Copy `bytes` from `src_offset` in an imported peer allocation into + /// `dst_offset` in one of this runner's buffers, by the fastest path for + /// where the two allocations live (`preferMemcpyForPeerCopy` above, and + /// the numbers behind it). `dst_offset`/`bytes` and `src_offset`/`bytes` + /// are checked against `dst.bytes.len` and `src.size`: both come from a + /// runtime transfer request, not a value the caller chose in source, so + /// they are never trusted. + pub fn copyFromPeer(self: *Runner, dst: Buffer, dst_offset: u32, src: Peer, src_offset: u32, bytes: u32) !void { + const dst_end = std.math.add(u64, @as(u64, dst_offset), @as(u64, bytes)) catch + return error.OutOfBounds; + if (dst_end > @as(u64, dst.bytes.len)) return error.OutOfBounds; + const src_end = std.math.add(u64, @as(u64, src_offset), @as(u64, bytes)) catch + return error.OutOfBounds; + if (src_end > src.size) return error.OutOfBounds; + + if (preferMemcpyForPeerCopy(dst.memory.location, src.memory.location)) { + const index = self.findAllocation(src.memory) orelse unreachable; + // importPeer only leaves this null for a VRAM peer, and + // preferMemcpyForPeerCopy already ruled that out above. + const mapping = self.allocations[index].cpu_mapping orelse unreachable; + const source: []const volatile u8 = mapping.bytes[@as(usize, src_offset)..][0..bytes]; + @memcpy(dst.bytes[@as(usize, dst_offset)..][0..bytes], source); + return; + } + try self.copyLinear(dst.va + @as(u64, dst_offset), src.va + @as(u64, src_offset), bytes); + } + pub fn deinit(self: *Runner) void { self.releaseOwnedResources(); self.client.freeDevice(self.dev); @@ -594,6 +678,17 @@ pub const Runner = struct { return memory; } + /// Dup `memory` (owned by `src`, a different Runner and a different RM + /// client) into this Runner's own client, tracked through the same + /// bookkeeping `allocMemory` uses. + fn dupMemoryFrom(self: *Runner, src: *Runner, memory: rm.Memory) !rm.Memory { + if (self.allocation_count == MAX_ALLOCATIONS) return error.RunnerAllocationLimit; + const dup = try self.client.dupMemory(self.dev, src.dev, memory); + errdefer self.client.freeMemory(self.dev, dup); + try self.trackMemory(dup); + return dup; + } + fn allocUsermodeMemory(self: *Runner) !rm.Memory { if (self.allocation_count == MAX_ALLOCATIONS) return error.RunnerAllocationLimit; const handle = try self.client.allocUsermode(self.dev, sdk.BLACKWELL_USERMODE_A); @@ -652,6 +747,31 @@ pub const Runner = struct { self.releaseAllocation(index); } + /// Dup `buffer` (owned by `src`, a different Runner on a different RM + /// client, possibly a different GPU) into this Runner and map it into + /// this Runner's own GPU VA space. The result addresses the same physical + /// memory as `buffer`, and nothing is copied. Release with `releasePeer`. + /// + /// Host memory also gets a CPU mapping here (VRAM does not, see `Peer`), + /// tracked through the same bookkeeping `alloc` uses. That is why + /// `releasePeer` needs no separate cleanup path: `releaseAllocation` + /// already unmaps whatever is present and rmFrees the dup in this + /// client, and never touches `src`'s own allocation. + pub fn importPeer(self: *Runner, src: *Runner, buffer: Buffer) !Peer { + const dup = try self.dupMemoryFrom(src, buffer.memory); + errdefer self.releaseAllocation(self.findAllocation(dup) orelse unreachable); + const va = self.takeVa(dup.size); + const mapping = try self.mapToGpu(dup, va); + if (dup.location != .vram) _ = try self.mapToCpu(dup); + return .{ .va = mapping.gpu_va, .size = dup.size, .memory = dup }; + } + + /// Release a Peer returned by `importPeer`. + pub fn releasePeer(self: *Runner, imported: Peer) void { + const index = self.findAllocation(imported.memory) orelse unreachable; + self.releaseAllocation(index); + } + /// Send `data` into `dst_va` through the pushbuffer and wait for it to land. /// This is the small-transfer path: no source buffer and no copy engine, at /// the cost of carrying the payload in the pushbuffer. @@ -737,6 +857,221 @@ test "live: freeBuffer releases its tracked CPU and GPU mappings" { try std.testing.expect(runner.findAllocation(buffer.memory) == null); } +test "live: initOn a nonexistent ordinal gives a clean error, not a crash" { + // Only a completely unreachable driver is worth skipping; this box has a + // real GPU, so `count` names an ordinal one past the last valid one. + var probe = rm.Client.open() catch return error.SkipZigTest; + const count = probe.deviceCount() catch { + probe.deinit(); + return error.SkipZigTest; + }; + probe.deinit(); + try std.testing.expectError(error.NoDevice, Runner.initOn(count)); +} + +/// Fill `words` dwords of `buf` with a pattern that depends on the index, so +/// a wrong VA (garbage or all zero) or a half copy (a correct prefix and a +/// stale/zero tail) both show up as a mismatch somewhere in the range. +fn fillIndexPattern(buf: []u32, words: usize, seed: u32) void { + for (buf[0..words], 0..) |*w, i| w.* = seed +% @as(u32, @intCast(i)) *% 2654435761; +} + +fn expectIndexPattern(buf: Buffer, words: usize, seed: u32) !void { + for (0..words) |i| { + const want = seed +% @as(u32, @intCast(i)) *% 2654435761; + try std.testing.expectEqual(want, buf.read(u32, i)); + } +} + +test "live: importPeer + copyFromPeer takes the memcpy arm for host memory" { + var a = try Runner.init(); + defer a.deinit(); + var b = try Runner.init(); + defer b.deinit(); + + const words = 4096; // 16 KiB + const src_buf = try a.alloc(.system, words * 4); + defer a.freeBuffer(src_buf); + const dst_buf = try b.alloc(.system, words * 4); + defer b.freeBuffer(dst_buf); + fillIndexPattern(src_buf.slice(u32), words, 0xBEEF_0000); + + const peer = try b.importPeer(&a, src_buf); + defer b.releasePeer(peer); + try std.testing.expect(preferMemcpyForPeerCopy(dst_buf.memory.location, peer.memory.location)); + + try b.copyFromPeer(dst_buf, 0, peer, 0, words * 4); + try expectIndexPattern(dst_buf, words, 0xBEEF_0000); +} + +test "live: importPeer + copyFromPeer takes the copy-engine arm for VRAM" { + var a = try Runner.init(); + defer a.deinit(); + var b = try Runner.init(); + defer b.deinit(); + + const words = 4096; // 16 KiB + const src_buf = try a.alloc(.vram, words * 4); + defer a.freeBuffer(src_buf); + const dst_buf = try b.alloc(.vram, words * 4); + defer b.freeBuffer(dst_buf); + fillIndexPattern(src_buf.slice(u32), words, 0xFACE_0000); + + const peer = try b.importPeer(&a, src_buf); + defer b.releasePeer(peer); + try std.testing.expect(!preferMemcpyForPeerCopy(dst_buf.memory.location, peer.memory.location)); + + try b.copyFromPeer(dst_buf, 0, peer, 0, words * 4); + try expectIndexPattern(dst_buf, words, 0xFACE_0000); +} + +test "live: copyFromPeer applies dst_offset and src_offset to the correct side" { + var a = try Runner.init(); + defer a.deinit(); + var b = try Runner.init(); + defer b.deinit(); + + const words = 4096; // 16 KiB + const src_words_offset = 7; + const dst_words_offset = 13; + const copy_words = words - 32; + + const src_buf = try a.alloc(.system, words * 4); + defer a.freeBuffer(src_buf); + const dst_buf = try b.alloc(.system, words * 4); + defer b.freeBuffer(dst_buf); + // Two distinct seeds so a source-side offset and a destination-side offset + // applied to the wrong side would read or land on the wrong bytes instead + // of merely shifting the same pattern. + fillIndexPattern(src_buf.slice(u32), words, 0xA5A5_0000); + fillIndexPattern(dst_buf.slice(u32), words, 0x5A5A_0000); + + const peer = try b.importPeer(&a, src_buf); + defer b.releasePeer(peer); + try b.copyFromPeer( + dst_buf, + dst_words_offset * 4, + peer, + src_words_offset * 4, + copy_words * 4, + ); + + const dst_words = dst_buf.slice(u32); + for (0..dst_words_offset) |i| { + const want = 0x5A5A_0000 +% @as(u32, @intCast(i)) *% 2654435761; + try std.testing.expectEqual(want, dst_words[i]); + } + for (0..copy_words) |i| { + const want = 0xA5A5_0000 +% @as(u32, @intCast(src_words_offset + i)) *% 2654435761; + try std.testing.expectEqual(want, dst_words[dst_words_offset + i]); + } + for (dst_words_offset + copy_words..words) |i| { + const want = 0x5A5A_0000 +% @as(u32, @intCast(i)) *% 2654435761; + try std.testing.expectEqual(want, dst_words[i]); + } +} + +test "live: copyFromPeer rejects a destination range past the buffer" { + var a = try Runner.init(); + defer a.deinit(); + var b = try Runner.init(); + defer b.deinit(); + + const src_buf = try a.alloc(.system, 0x1000); + defer a.freeBuffer(src_buf); + const dst_buf = try b.alloc(.system, 0x1000); + defer b.freeBuffer(dst_buf); + const peer = try b.importPeer(&a, src_buf); + defer b.releasePeer(peer); + + try std.testing.expectError( + error.OutOfBounds, + b.copyFromPeer(dst_buf, @intCast(dst_buf.bytes.len), peer, 0, 4), + ); + try std.testing.expectError( + error.OutOfBounds, + b.copyFromPeer(dst_buf, 0, peer, @intCast(peer.size), 4), + ); +} + +/// Time `iters` memcpys of `bytes` bytes from `src` to `dst`. +fn timeMemcpy(dst: []u8, src: []const volatile u8, bytes: usize, iters: usize) u64 { + @memcpy(dst[0..bytes], src[0..bytes]); // warm the path + var start: std.Io.Timestamp = .now(std.testing.io, .awake); + for (0..iters) |_| @memcpy(dst[0..bytes], src[0..bytes]); + return @intCast(@divTrunc(start.durationTo(.now(std.testing.io, .awake)).nanoseconds, iters)); +} + +/// Time `iters` copy-engine transfers of `bytes` bytes. +fn timeCopyLinear(r: *Runner, dst_va: u64, src_va: u64, bytes: u32, iters: usize) !u64 { + try r.copyLinear(dst_va, src_va, bytes); // warm the path + var start: std.Io.Timestamp = .now(std.testing.io, .awake); + for (0..iters) |_| try r.copyLinear(dst_va, src_va, bytes); + return @intCast(@divTrunc(start.durationTo(.now(std.testing.io, .awake)).nanoseconds, iters)); +} + +test "live: a host memcpy beats the copy engine, and a BAR1 read loses to it" { + var a = try Runner.init(); + defer a.deinit(); + var b = try Runner.init(); + defer b.deinit(); + + const sizes = [_]u32{ 4 * 1024, 64 * 1024, 1024 * 1024 }; + // A BAR1 memcpy at 1 MiB alone takes hundreds of milliseconds, so this takes + // fewer samples than the 200 elsewhere in this file. The two paths differ by + // orders of magnitude, so few samples still separate them. + const iters = 20; + for (sizes) |size| { + const sys_src = try a.alloc(.system, size); + defer a.freeBuffer(sys_src); + const sys_dst = try b.alloc(.system, size); + defer b.freeBuffer(sys_dst); + const sys_peer = try b.importPeer(&a, sys_src); + defer b.releasePeer(sys_peer); + const sys_peer_index = b.findAllocation(sys_peer.memory) orelse unreachable; + const sys_peer_bytes = b.allocations[sys_peer_index].cpu_mapping.?.bytes; + const sys_memcpy_ns = timeMemcpy(sys_dst.bytes, sys_peer_bytes, size, iters); + const sys_ce_ns = try timeCopyLinear(&b, sys_dst.va, sys_peer.va, size, iters); + // Host memory pays no submission at all for a memcpy, and submission is + // most of a small transfer, so the memcpy has to win at every size here. + if (sys_memcpy_ns >= sys_ce_ns) { + std.debug.print( + "{d} B in host memory: memcpy {d} ns, copy engine {d} ns\n", + .{ size, sys_memcpy_ns, sys_ce_ns }, + ); + return error.HostMemcpyNoLongerWins; + } + + const vram_src = try a.alloc(.vram, size); + defer a.freeBuffer(vram_src); + const vram_dst = try b.alloc(.vram, size); + defer b.freeBuffer(vram_dst); + const vram_peer = try b.importPeer(&a, vram_src); + defer b.releasePeer(vram_peer); + // A scratch BAR1 CPU mapping, made only to time the path this rejects. + // The real copyFromPeer never maps VRAM to the CPU, see `Peer`. + const scratch = try b.client.mapMemory(b.dev, vram_peer.memory); + defer b.client.unmapMemory(scratch); + const vram_memcpy_ns = timeMemcpy(vram_dst.bytes, scratch.bytes, size, iters); + const vram_ce_ns = try timeCopyLinear(&b, vram_dst.va, vram_peer.va, size, iters); + // A CPU read of VRAM goes through the BAR1 window uncached, which costs + // far more than the copy engine's whole submission. + if (vram_memcpy_ns <= vram_ce_ns) { + std.debug.print( + "{d} B in VRAM: BAR1 memcpy {d} ns, copy engine {d} ns\n", + .{ size, vram_memcpy_ns, vram_ce_ns }, + ); + return error.Bar1ReadNoLongerLoses; + } + } + // And the path choice has to sit where the timings put it. + try std.testing.expect(preferMemcpyForPeerCopy(.system, .system)); + try std.testing.expect(preferMemcpyForPeerCopy(.system, .system_wc)); + try std.testing.expect(!preferMemcpyForPeerCopy(.vram, .vram)); + try std.testing.expect(!preferMemcpyForPeerCopy(.system, .vram)); + try std.testing.expect(!preferMemcpyForPeerCopy(.vram, .system)); +} + test "live: integer ALU (IADD3, IMAD, IMAD.WIDE, ISETP, SEL) computes on the SMs" { var r = try Runner.init(); defer r.deinit(); diff --git a/src/nvidia/rm.zig b/src/nvidia/rm.zig index 84a9ddc..a48a771 100644 --- a/src/nvidia/rm.zig +++ b/src/nvidia/rm.zig @@ -74,6 +74,12 @@ pub const Client = struct { self.t.rmFree(h_root, h_parent, h_object); } + /// How many GPUs the driver knows about (the valid entries of CARD_INFO). + /// A valid `index` for allocDevice is `0..deviceCount()`. + pub fn deviceCount(self: *Client) Error!u32 { + return self.t.deviceCount(); + } + /// Bring up the `index`-th GPU end to end: register its node, attach it, /// then allocate the root client, device, and subdevice. The full sequence /// required by the open kernel modules. @@ -183,6 +189,18 @@ pub const Client = struct { self.t.rmFree(dev.client, dev.device, mem.handle); } + /// Dup `memory` (owned by `src`'s client) into `dst`'s client, so `dst` + /// can map and use it the same as memory it allocated itself. The dup + /// shares the same physical pages as the original. It is a handle to + /// the same resource, not a copy, so a write through either side is + /// visible through the other. Free the returned Memory with freeMemory on + /// `dst`; that only drops this client's handle and never touches `src`'s + /// own allocation. + pub fn dupMemory(self: *Client, dst: Device, src: Device, memory: Memory) Error!Memory { + const handle = try self.t.dupObject(dst.client, dst.device, self.t.newHandle(), src.client, memory.handle); + return .{ .handle = handle, .size = memory.size, .location = memory.location }; + } + /// CPU-map a memory object: NV_ESC_RM_MAP_MEMORY sets up an mmap context on a /// dedicated fd, then mmap returns the pointer. System memory maps via the /// control node, VRAM via the device node. Supports many concurrent mappings. @@ -219,6 +237,69 @@ pub const Client = struct { return buf[0..n]; } + /// Decoded NV0000_CTRL_CMD_SYSTEM_GET_P2P_CAPS_V2 result for one GPU pair. + /// `status_ok` is true only when every entry in the kernel's status table + /// reads NV0000_P2P_CAPS_STATUS_OK. False means at least one of the caps + /// bits above was refused by chipset, topology, or a regkey rather than + /// genuinely supported. + pub const P2pCaps = struct { + writes: bool, + reads: bool, + prop: bool, + nvlink: bool, + atomics: bool, + loopback: bool, + pci: bool, + indirect_writes: bool, + indirect_reads: bool, + indirect_atomics: bool, + optimal_read_ces: sdk.NvU32, + optimal_write_ces: sdk.NvU32, + status_ok: bool, + /// The raw per-capability status table (sdk.p2p_caps_status_index + /// names the entries): sdk.P2P_CAPS_STATUS_OK, or a reason code for + /// why that one capability was refused. + status: [sdk.P2pCapsV2Params.CAPS_STATUS_TABLE_SIZE]sdk.NvU8, + }; + + /// Query the P2P capabilities between `a` and `b` (NV0000_CTRL_CMD_ + /// SYSTEM_GET_P2P_CAPS_V2, on `a`'s root client). `a` and `b` may be the + /// same Device, which needs only one GPU to reach, but do not read that + /// answer as "what this GPU can do with itself": a single RTX 5070 with no + /// peer fabric refuses every capability, loopback included. Use this to + /// learn about a real pair. For two runners on one GPU use + /// `Runner.importPeer`, which is a handle dup and needs no P2P capability. + pub fn p2pCaps(self: *Client, a: Device, b: Device) Error!P2pCaps { + var p = sdk.P2pCapsV2Params{}; + p.gpu_ids[0] = a.gpu_id; + p.gpu_ids[1] = b.gpu_id; + p.gpu_count = 2; + try self.control(a, a.client, sdk.NV0000_CTRL_CMD_SYSTEM_GET_P2P_CAPS_V2, &p, @sizeOf(sdk.P2pCapsV2Params)); + + const caps = p.p2p_caps; + const bit = sdk.p2p_caps; + var status_ok = true; + for (p.p2p_caps_status) |entry| { + if (entry != sdk.P2P_CAPS_STATUS_OK) status_ok = false; + } + return .{ + .writes = caps & bit.WRITES != 0, + .reads = caps & bit.READS != 0, + .prop = caps & bit.PROP != 0, + .nvlink = caps & bit.NVLINK != 0, + .atomics = caps & bit.ATOMICS != 0, + .loopback = caps & bit.LOOPBACK != 0, + .pci = caps & bit.PCI != 0, + .indirect_writes = caps & bit.INDIRECT_WRITES != 0, + .indirect_reads = caps & bit.INDIRECT_READS != 0, + .indirect_atomics = caps & bit.INDIRECT_ATOMICS != 0, + .optimal_read_ces = p.p2p_optimal_read_ces, + .optimal_write_ces = p.p2p_optimal_write_ces, + .status_ok = status_ok, + .status = p.p2p_caps_status, + }; + } + /// Allocate a GPU virtual address space (FERMI_VASPACE_A) under the device. /// First prerequisite for a GPFIFO channel. Free with rmFree(client, device, h). pub fn allocVaSpace(self: *Client, dev: Device) Error!sdk.NvHandle { @@ -493,6 +574,44 @@ test "live: RM_CONTROL queries GPU id and name" { try std.testing.expect(name.len > 0); } +test "live: deviceCount reports the GPUs the driver knows about" { + var c = try openOrSkip(); + defer c.deinit(); + const count = try c.deviceCount(); + try std.testing.expect(count >= 1); + // A valid index is always reachable up to the reported count. + const dev = c.allocDevice(count - 1) catch |e| switch (e) { + error.OpenFailed, error.NoDevice => return error.SkipZigTest, + else => return e, + }; + c.freeDevice(dev); +} + +test "live: p2pCaps against itself proves the struct layout, decoded" { + var c = try openOrSkip(); + defer c.deinit(); + const dev = c.allocDevice(0) catch |e| switch (e) { + error.OpenFailed, error.NoDevice => return error.SkipZigTest, + else => return e, + }; + defer c.freeDevice(dev); + + // The call succeeding at all is the acceptance criterion: NVOS54 only + // returns without error when the kernel's own status field came back 0, + // which only happens when it accepted this struct's exact layout. + // + // On a one-GPU box the answer is every capability false, status_ok + // false, and a status table of CHIPSET_NOT_SUPPORTED / NOT_SUPPORTED + // (the table reads {1, 1, 5, 5, 5, 5, 5, 5, 5}). A single GPU with no NVLink + // fabric has no peer to reach, not even itself, so the RM refuses every + // entry instead of granting a trivial loopback. Duping a memory object + // between two clients on this one GPU (see Runner.importPeer) is a + // separate RM primitive and does not depend on this capability at all. + const caps = try c.p2pCaps(dev, dev); + try std.testing.expect(!caps.status_ok); + for (caps.status) |entry| try std.testing.expect(entry != sdk.P2P_CAPS_STATUS_OK); +} + test "live: allocate a GPU VA space (channel prerequisite)" { var c = try openOrSkip(); defer c.deinit(); diff --git a/src/nvidia/sdk.zig b/src/nvidia/sdk.zig index d15c08e..fbe9f61 100644 --- a/src/nvidia/sdk.zig +++ b/src/nvidia/sdk.zig @@ -71,6 +71,24 @@ pub const Os00Params = extern struct { h_object_old: NvHandle, status: NvV32, }; +/// NVOS55_PARAMETERS (nvos.h): the NV_ESC_RM_DUP_OBJECT (NV04_DUP_OBJECT) +/// parameter block. Dups `hObjectSrc` (owned by `hClientSrc`) into a new +/// object under `hClient`/`hParent`, which shares the same underlying +/// resource. It hands a memory object from one RM client to another with no +/// copy. +/// used to hand a memory object from one RM client to another with no copy. +pub const Os55Params = extern struct { + h_client: NvHandle, // destination client + h_parent: NvHandle, // parent of the new object + h_object: NvHandle, // in: requested handle, out: the actual one + h_client_src: NvHandle, + h_object_src: NvHandle, + flags: NvU32, + status: NvV32, +}; + +/// NV04_DUP_HANDLE_FLAGS_NONE (nvos.h). +pub const NV04_DUP_HANDLE_FLAGS_NONE: NvU32 = 0; /// NVOS21_PARAMETERS (nvos.h): the NV_ESC_RM_ALLOC (NV04_ALLOC) parameter block. pub const Os21Params = extern struct { @@ -219,6 +237,59 @@ pub const GpuGetNameStringParams = extern struct { ascii: [GPU_NAME_MAX_LENGTH]u8 = [_]u8{0} ** GPU_NAME_MAX_LENGTH, }; +/// NV0000_CTRL_CMD_SYSTEM_GET_P2P_CAPS_V2 (ctrl0000system.h): the V2 form of +/// the P2P caps query. Use this one, not the deprecated V1 (0x127). V1 carries +/// an NvP64 pointer field that V2 replaced with plain NvU32s, which is why V2 +/// is the one that fits a fixed-layout extern struct cleanly. +pub const NV0000_CTRL_CMD_SYSTEM_GET_P2P_CAPS_V2: NvU32 = 0x12b; + +/// NV0000_CTRL_SYSTEM_GET_P2P_CAPS_V2_PARAMS (ctrl0000system.h). +pub const P2pCapsV2Params = extern struct { + gpu_ids: [MAX_ATTACHED_GPUS]NvU32 = [_]NvU32{0} ** MAX_ATTACHED_GPUS, + gpu_count: NvU32 = 0, + p2p_caps: NvU32 = 0, + p2p_optimal_read_ces: NvU32 = 0, + p2p_optimal_write_ces: NvU32 = 0, + p2p_caps_status: [CAPS_STATUS_TABLE_SIZE]NvU8 = [_]NvU8{0} ** CAPS_STATUS_TABLE_SIZE, + bus_peer_ids: [MAX_ATTACHED_GPUS_SQUARED]NvU32 = [_]NvU32{0} ** MAX_ATTACHED_GPUS_SQUARED, + bus_egm_peer_ids: [MAX_ATTACHED_GPUS_SQUARED]NvU32 = [_]NvU32{0} ** MAX_ATTACHED_GPUS_SQUARED, + + pub const MAX_ATTACHED_GPUS = 32; // NV0000_CTRL_SYSTEM_MAX_ATTACHED_GPUS + pub const MAX_ATTACHED_GPUS_SQUARED = 1024; // NV0000_CTRL_SYSTEM_MAX_ATTACHED_GPUS_SQUARED + pub const CAPS_STATUS_TABLE_SIZE = 9; // NV0000_CTRL_P2P_CAPS_INDEX_TABLE_SIZE +}; + +/// p2pCaps bit positions (NV0000_CTRL_SYSTEM_GET_P2P_CAPS_*_SUPPORTED). +pub const p2p_caps = struct { + pub const WRITES: NvU32 = 1 << 0; + pub const READS: NvU32 = 1 << 1; + pub const PROP: NvU32 = 1 << 2; + pub const NVLINK: NvU32 = 1 << 3; + pub const ATOMICS: NvU32 = 1 << 4; + pub const LOOPBACK: NvU32 = 1 << 5; + pub const PCI: NvU32 = 1 << 6; + pub const INDIRECT_WRITES: NvU32 = 1 << 7; + pub const INDIRECT_READS: NvU32 = 1 << 8; + pub const INDIRECT_ATOMICS: NvU32 = 1 << 9; +}; + +/// p2pCapsStatus table indices (NV0000_CTRL_P2P_CAPS_INDEX_*). +pub const p2p_caps_status_index = struct { + pub const READ = 0; + pub const WRITE = 1; + pub const NVLINK = 2; + pub const ATOMICS = 3; + pub const PROP = 4; + pub const LOOPBACK = 5; + pub const PCI = 6; + pub const C2C = 7; + pub const PCI_BAR1 = 8; +}; + +/// NV0000_P2P_CAPS_STATUS_OK: the capability at this table index is supported. +/// Any other value names a reason it is not (chipset, GPU, topology, regkey). +pub const P2P_CAPS_STATUS_OK: NvU8 = 0; + /// NV_MEMORY_VIRTUAL_ALLOCATION_PARAMS (cl0070.h): pAllocParms for /// NV01_MEMORY_VIRTUAL - reserves a GPU VA range [offset, limit] in a VA space. pub const VirtMemAllocParams = extern struct { @@ -480,6 +551,19 @@ test "RM struct layouts match the NVIDIA ABI" { try std.testing.expectEqual(@as(usize, 64), @offsetOf(ChannelAllocParams, "userd_offset")); try std.testing.expectEqual(@as(usize, 4), @sizeOf(ChannelBindParams)); try std.testing.expectEqual(@as(usize, 3), @sizeOf(GpfifoScheduleParams)); + try std.testing.expectEqual(@as(usize, 28), @sizeOf(Os55Params)); + try std.testing.expectEqual(@as(usize, 12), @offsetOf(Os55Params, "h_client_src")); + try std.testing.expectEqual(@as(usize, 24), @offsetOf(Os55Params, "status")); + // A wrong size or offset here is a silent garbage read past the kernel's + // own struct, since NV_ESC_RM_DUP_OBJECT/GET_P2P_CAPS_V2 trust params_size. + try std.testing.expectEqual(@as(usize, 8348), @sizeOf(P2pCapsV2Params)); + try std.testing.expectEqual(@as(usize, 128), @offsetOf(P2pCapsV2Params, "gpu_count")); + try std.testing.expectEqual(@as(usize, 140), @offsetOf(P2pCapsV2Params, "p2p_optimal_write_ces")); + // p2p_caps_status is a [9]u8 at 144; bus_peer_ids needs 4-byte alignment, + // so 3 bytes of padding sit between them (144+9=153, rounded up to 156). + try std.testing.expectEqual(@as(usize, 144), @offsetOf(P2pCapsV2Params, "p2p_caps_status")); + try std.testing.expectEqual(@as(usize, 156), @offsetOf(P2pCapsV2Params, "bus_peer_ids")); + try std.testing.expectEqual(@as(usize, 4252), @offsetOf(P2pCapsV2Params, "bus_egm_peer_ids")); } test "GPFIFO submission encodings" { diff --git a/src/nvidia/transport/freestanding.zig b/src/nvidia/transport/freestanding.zig index 94d6eec..2a2fa91 100644 --- a/src/nvidia/transport/freestanding.zig +++ b/src/nvidia/transport/freestanding.zig @@ -354,6 +354,34 @@ pub const Transport = struct { self.untrack(h_object); } + /// No CARD_INFO analog on baremetal, because there is no kernel node to enumerate + /// through. This HAL always addresses exactly the one GPU it was booted + /// against. + pub fn deviceCount(self: *Transport) Error!u32 { + if (!self.booted) return error.NotImplemented; + return 1; + } + + /// NV_ESC_RM_DUP_OBJECT has no GSP RPC form yet, so cross-client memory + /// duping (the peer-transfer path) is Linux-only for now. Kept + /// syntactically valid so rm.zig compiles against either transport. + pub fn dupObject( + self: *Transport, + h_client: sdk.NvHandle, + h_parent: sdk.NvHandle, + h_new: sdk.NvHandle, + h_client_src: sdk.NvHandle, + h_object_src: sdk.NvHandle, + ) Error!sdk.NvHandle { + _ = self; + _ = h_client; + _ = h_parent; + _ = h_new; + _ = h_client_src; + _ = h_object_src; + return error.NotImplemented; + } + // ======================================================================= // allocDevice -> the ROOT / DEVICE / SUBDEVICE alloc sequence over RPC. Mirrors // linux.zig allocDevice's order + the sdk params it passes (only the registration @@ -916,6 +944,8 @@ test "freestanding open() without metal inputs fails cleanly" { var t = Transport{}; try testing.expectError(error.NotImplemented, t.rmAlloc(0, 0, 1, sdk.NV01_ROOT, null, 0)); try testing.expectError(error.NotImplemented, t.allocDevice(0)); + try testing.expectError(error.NotImplemented, t.deviceCount()); + try testing.expectError(error.NotImplemented, t.dupObject(0, 0, 1, 0, 0)); } test { diff --git a/src/nvidia/transport/linux.zig b/src/nvidia/transport/linux.zig index 43d3a00..fd23639 100644 --- a/src/nvidia/transport/linux.zig +++ b/src/nvidia/transport/linux.zig @@ -115,6 +115,38 @@ pub const Transport = struct { } } + /// NV_ESC_RM_DUP_OBJECT (NVOS55): dup `h_object_src` (owned by client + /// `h_client_src`) into a new object `h_new` under `h_client`/`h_parent`. + /// The dup shares the same underlying resource as the source. For a + /// memory object that is the same physical pages, so writes through either + /// handle land in the same place. Returns the actual new handle. + pub fn dupObject( + self: *Transport, + h_client: sdk.NvHandle, + h_parent: sdk.NvHandle, + h_new: sdk.NvHandle, + h_client_src: sdk.NvHandle, + h_object_src: sdk.NvHandle, + ) Error!sdk.NvHandle { + var p = sdk.Os55Params{ + .h_client = h_client, + .h_parent = h_parent, + .h_object = h_new, + .h_client_src = h_client_src, + .h_object_src = h_object_src, + .flags = sdk.NV04_DUP_HANDLE_FLAGS_NONE, + .status = 0, + }; + const req = ioctl.iowr(ioctl.NV_ESC_RM_DUP_OBJECT, sdk.Os55Params); + const rc = std.os.linux.ioctl(self.fd, req, @intFromPtr(&p)); + switch (std.os.linux.errno(rc)) { + .SUCCESS => {}, + else => return error.IoctlFailed, + } + if (p.status != 0) return error.RmAllocFailed; + return p.h_object; + } + /// NV_ESC_RM_FREE: free an RM object. pub fn rmFree(self: *Transport, h_root: sdk.NvHandle, h_parent: sdk.NvHandle, h_object: sdk.NvHandle) void { var p = sdk.Os00Params{ .h_root = h_root, .h_object_parent = h_parent, .h_object_old = h_object, .status = 0 }; @@ -134,6 +166,16 @@ pub const Transport = struct { return cards; } + /// How many GPUs the driver knows about (the valid entries of CARD_INFO). + pub fn deviceCount(self: *Transport) Error!u32 { + const cards = try self.cardInfo(); + var count: u32 = 0; + for (cards) |entry| { + if (entry.valid != 0) count += 1; + } + return count; + } + /// NV_ESC_ATTACH_GPUS_TO_FD: attach a GPU (by gpu_id) to the control fd. The /// kernel reads size/4 ids; a trailing 0 terminates the list. fn attachGpu(self: *Transport, gpu_id: sdk.NvU32) Error!void { From b6dcae75412bf7cbe0eea391597c7cda90a8e7c7 Mon Sep 17 00:00:00 2001 From: Tristan Ross Date: Sun, 27 Sep 2026 18:27:20 -0700 Subject: [PATCH 2/2] chore: init ci --- .github/workflows/ci.yml | 20 ++++++++++++++++++++ flake.nix | 1 - 2 files changed, 20 insertions(+), 1 deletion(-) create mode 100644 .github/workflows/ci.yml diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml new file mode 100644 index 0000000..21f0d5f --- /dev/null +++ b/.github/workflows/ci.yml @@ -0,0 +1,20 @@ +on: + pull_request: + workflow_dispatch: + push: + branches: + - main + - master + tags: + - v?[0-9]+.[0-9]+.[0-9]+* + +concurrency: + group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.ref }} + cancel-in-progress: true + +jobs: + DeterminateCI: + uses: DeterminateSystems/ci/.github/workflows/workflow.yml@main + permissions: + id-token: write + contents: read diff --git a/flake.nix b/flake.nix index e2cfc30..49b3165 100644 --- a/flake.nix +++ b/flake.nix @@ -24,7 +24,6 @@ allSystems = [ "x86_64-linux" "aarch64-linux" - "aarch64-darwin" ]; flakeverConfig = flakever.lib.mkFlakever {