Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
20 changes: 20 additions & 0 deletions .github/workflows/ci.yml
Original file line number Diff line number Diff line change
@@ -0,0 +1,20 @@
on:
pull_request:
workflow_dispatch:
push:
branches:
- main
- master
tags:
- v?[0-9]+.[0-9]+.[0-9]+*

concurrency:
group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.ref }}
cancel-in-progress: true

jobs:
DeterminateCI:
uses: DeterminateSystems/ci/.github/workflows/workflow.yml@main
permissions:
id-token: write
contents: read
1 change: 0 additions & 1 deletion flake.nix
Original file line number Diff line number Diff line change
Expand Up @@ -24,7 +24,6 @@
allSystems = [
"x86_64-linux"
"aarch64-linux"
"aarch64-darwin"
];

flakeverConfig = flakever.lib.mkFlakever {
Expand Down
337 changes: 336 additions & 1 deletion src/nvidia/compute.zig

Large diffs are not rendered by default.

119 changes: 119 additions & 0 deletions src/nvidia/rm.zig
Original file line number Diff line number Diff line change
Expand Up @@ -74,6 +74,12 @@ pub const Client = struct {
self.t.rmFree(h_root, h_parent, h_object);
}

/// How many GPUs the driver knows about (the valid entries of CARD_INFO).
/// A valid `index` for allocDevice is `0..deviceCount()`.
pub fn deviceCount(self: *Client) Error!u32 {
return self.t.deviceCount();
}

/// Bring up the `index`-th GPU end to end: register its node, attach it,
/// then allocate the root client, device, and subdevice. The full sequence
/// required by the open kernel modules.
Expand Down Expand Up @@ -183,6 +189,18 @@ pub const Client = struct {
self.t.rmFree(dev.client, dev.device, mem.handle);
}

/// Dup `memory` (owned by `src`'s client) into `dst`'s client, so `dst`
/// can map and use it the same as memory it allocated itself. The dup
/// shares the same physical pages as the original. It is a handle to
/// the same resource, not a copy, so a write through either side is
/// visible through the other. Free the returned Memory with freeMemory on
/// `dst`; that only drops this client's handle and never touches `src`'s
/// own allocation.
pub fn dupMemory(self: *Client, dst: Device, src: Device, memory: Memory) Error!Memory {
const handle = try self.t.dupObject(dst.client, dst.device, self.t.newHandle(), src.client, memory.handle);
return .{ .handle = handle, .size = memory.size, .location = memory.location };
}

/// CPU-map a memory object: NV_ESC_RM_MAP_MEMORY sets up an mmap context on a
/// dedicated fd, then mmap returns the pointer. System memory maps via the
/// control node, VRAM via the device node. Supports many concurrent mappings.
Expand Down Expand Up @@ -219,6 +237,69 @@ pub const Client = struct {
return buf[0..n];
}

/// Decoded NV0000_CTRL_CMD_SYSTEM_GET_P2P_CAPS_V2 result for one GPU pair.
/// `status_ok` is true only when every entry in the kernel's status table
/// reads NV0000_P2P_CAPS_STATUS_OK. False means at least one of the caps
/// bits above was refused by chipset, topology, or a regkey rather than
/// genuinely supported.
pub const P2pCaps = struct {
writes: bool,
reads: bool,
prop: bool,
nvlink: bool,
atomics: bool,
loopback: bool,
pci: bool,
indirect_writes: bool,
indirect_reads: bool,
indirect_atomics: bool,
optimal_read_ces: sdk.NvU32,
optimal_write_ces: sdk.NvU32,
status_ok: bool,
/// The raw per-capability status table (sdk.p2p_caps_status_index
/// names the entries): sdk.P2P_CAPS_STATUS_OK, or a reason code for
/// why that one capability was refused.
status: [sdk.P2pCapsV2Params.CAPS_STATUS_TABLE_SIZE]sdk.NvU8,
};

/// Query the P2P capabilities between `a` and `b` (NV0000_CTRL_CMD_
/// SYSTEM_GET_P2P_CAPS_V2, on `a`'s root client). `a` and `b` may be the
/// same Device, which needs only one GPU to reach, but do not read that
/// answer as "what this GPU can do with itself": a single RTX 5070 with no
/// peer fabric refuses every capability, loopback included. Use this to
/// learn about a real pair. For two runners on one GPU use
/// `Runner.importPeer`, which is a handle dup and needs no P2P capability.
pub fn p2pCaps(self: *Client, a: Device, b: Device) Error!P2pCaps {
var p = sdk.P2pCapsV2Params{};
p.gpu_ids[0] = a.gpu_id;
p.gpu_ids[1] = b.gpu_id;
p.gpu_count = 2;
try self.control(a, a.client, sdk.NV0000_CTRL_CMD_SYSTEM_GET_P2P_CAPS_V2, &p, @sizeOf(sdk.P2pCapsV2Params));

const caps = p.p2p_caps;
const bit = sdk.p2p_caps;
var status_ok = true;
for (p.p2p_caps_status) |entry| {
if (entry != sdk.P2P_CAPS_STATUS_OK) status_ok = false;
}
return .{
.writes = caps & bit.WRITES != 0,
.reads = caps & bit.READS != 0,
.prop = caps & bit.PROP != 0,
.nvlink = caps & bit.NVLINK != 0,
.atomics = caps & bit.ATOMICS != 0,
.loopback = caps & bit.LOOPBACK != 0,
.pci = caps & bit.PCI != 0,
.indirect_writes = caps & bit.INDIRECT_WRITES != 0,
.indirect_reads = caps & bit.INDIRECT_READS != 0,
.indirect_atomics = caps & bit.INDIRECT_ATOMICS != 0,
.optimal_read_ces = p.p2p_optimal_read_ces,
.optimal_write_ces = p.p2p_optimal_write_ces,
.status_ok = status_ok,
.status = p.p2p_caps_status,
};
}

/// Allocate a GPU virtual address space (FERMI_VASPACE_A) under the device.
/// First prerequisite for a GPFIFO channel. Free with rmFree(client, device, h).
pub fn allocVaSpace(self: *Client, dev: Device) Error!sdk.NvHandle {
Expand Down Expand Up @@ -493,6 +574,44 @@ test "live: RM_CONTROL queries GPU id and name" {
try std.testing.expect(name.len > 0);
}

test "live: deviceCount reports the GPUs the driver knows about" {
var c = try openOrSkip();
defer c.deinit();
const count = try c.deviceCount();
try std.testing.expect(count >= 1);
// A valid index is always reachable up to the reported count.
const dev = c.allocDevice(count - 1) catch |e| switch (e) {
error.OpenFailed, error.NoDevice => return error.SkipZigTest,
else => return e,
};
c.freeDevice(dev);
}

test "live: p2pCaps against itself proves the struct layout, decoded" {
var c = try openOrSkip();
defer c.deinit();
const dev = c.allocDevice(0) catch |e| switch (e) {
error.OpenFailed, error.NoDevice => return error.SkipZigTest,
else => return e,
};
defer c.freeDevice(dev);

// The call succeeding at all is the acceptance criterion: NVOS54 only
// returns without error when the kernel's own status field came back 0,
// which only happens when it accepted this struct's exact layout.
//
// On a one-GPU box the answer is every capability false, status_ok
// false, and a status table of CHIPSET_NOT_SUPPORTED / NOT_SUPPORTED
// (the table reads {1, 1, 5, 5, 5, 5, 5, 5, 5}). A single GPU with no NVLink
// fabric has no peer to reach, not even itself, so the RM refuses every
// entry instead of granting a trivial loopback. Duping a memory object
// between two clients on this one GPU (see Runner.importPeer) is a
// separate RM primitive and does not depend on this capability at all.
const caps = try c.p2pCaps(dev, dev);
try std.testing.expect(!caps.status_ok);
for (caps.status) |entry| try std.testing.expect(entry != sdk.P2P_CAPS_STATUS_OK);
}

test "live: allocate a GPU VA space (channel prerequisite)" {
var c = try openOrSkip();
defer c.deinit();
Expand Down
84 changes: 84 additions & 0 deletions src/nvidia/sdk.zig
Original file line number Diff line number Diff line change
Expand Up @@ -71,6 +71,24 @@ pub const Os00Params = extern struct {
h_object_old: NvHandle,
status: NvV32,
};
/// NVOS55_PARAMETERS (nvos.h): the NV_ESC_RM_DUP_OBJECT (NV04_DUP_OBJECT)
/// parameter block. Dups `hObjectSrc` (owned by `hClientSrc`) into a new
/// object under `hClient`/`hParent`, which shares the same underlying
/// resource. It hands a memory object from one RM client to another with no
/// copy.
/// used to hand a memory object from one RM client to another with no copy.
pub const Os55Params = extern struct {
h_client: NvHandle, // destination client
h_parent: NvHandle, // parent of the new object
h_object: NvHandle, // in: requested handle, out: the actual one
h_client_src: NvHandle,
h_object_src: NvHandle,
flags: NvU32,
status: NvV32,
};

/// NV04_DUP_HANDLE_FLAGS_NONE (nvos.h).
pub const NV04_DUP_HANDLE_FLAGS_NONE: NvU32 = 0;

/// NVOS21_PARAMETERS (nvos.h): the NV_ESC_RM_ALLOC (NV04_ALLOC) parameter block.
pub const Os21Params = extern struct {
Expand Down Expand Up @@ -219,6 +237,59 @@ pub const GpuGetNameStringParams = extern struct {
ascii: [GPU_NAME_MAX_LENGTH]u8 = [_]u8{0} ** GPU_NAME_MAX_LENGTH,
};

/// NV0000_CTRL_CMD_SYSTEM_GET_P2P_CAPS_V2 (ctrl0000system.h): the V2 form of
/// the P2P caps query. Use this one, not the deprecated V1 (0x127). V1 carries
/// an NvP64 pointer field that V2 replaced with plain NvU32s, which is why V2
/// is the one that fits a fixed-layout extern struct cleanly.
pub const NV0000_CTRL_CMD_SYSTEM_GET_P2P_CAPS_V2: NvU32 = 0x12b;

/// NV0000_CTRL_SYSTEM_GET_P2P_CAPS_V2_PARAMS (ctrl0000system.h).
pub const P2pCapsV2Params = extern struct {
gpu_ids: [MAX_ATTACHED_GPUS]NvU32 = [_]NvU32{0} ** MAX_ATTACHED_GPUS,
gpu_count: NvU32 = 0,
p2p_caps: NvU32 = 0,
p2p_optimal_read_ces: NvU32 = 0,
p2p_optimal_write_ces: NvU32 = 0,
p2p_caps_status: [CAPS_STATUS_TABLE_SIZE]NvU8 = [_]NvU8{0} ** CAPS_STATUS_TABLE_SIZE,
bus_peer_ids: [MAX_ATTACHED_GPUS_SQUARED]NvU32 = [_]NvU32{0} ** MAX_ATTACHED_GPUS_SQUARED,
bus_egm_peer_ids: [MAX_ATTACHED_GPUS_SQUARED]NvU32 = [_]NvU32{0} ** MAX_ATTACHED_GPUS_SQUARED,

pub const MAX_ATTACHED_GPUS = 32; // NV0000_CTRL_SYSTEM_MAX_ATTACHED_GPUS
pub const MAX_ATTACHED_GPUS_SQUARED = 1024; // NV0000_CTRL_SYSTEM_MAX_ATTACHED_GPUS_SQUARED
pub const CAPS_STATUS_TABLE_SIZE = 9; // NV0000_CTRL_P2P_CAPS_INDEX_TABLE_SIZE
};

/// p2pCaps bit positions (NV0000_CTRL_SYSTEM_GET_P2P_CAPS_*_SUPPORTED).
pub const p2p_caps = struct {
pub const WRITES: NvU32 = 1 << 0;
pub const READS: NvU32 = 1 << 1;
pub const PROP: NvU32 = 1 << 2;
pub const NVLINK: NvU32 = 1 << 3;
pub const ATOMICS: NvU32 = 1 << 4;
pub const LOOPBACK: NvU32 = 1 << 5;
pub const PCI: NvU32 = 1 << 6;
pub const INDIRECT_WRITES: NvU32 = 1 << 7;
pub const INDIRECT_READS: NvU32 = 1 << 8;
pub const INDIRECT_ATOMICS: NvU32 = 1 << 9;
};

/// p2pCapsStatus table indices (NV0000_CTRL_P2P_CAPS_INDEX_*).
pub const p2p_caps_status_index = struct {
pub const READ = 0;
pub const WRITE = 1;
pub const NVLINK = 2;
pub const ATOMICS = 3;
pub const PROP = 4;
pub const LOOPBACK = 5;
pub const PCI = 6;
pub const C2C = 7;
pub const PCI_BAR1 = 8;
};

/// NV0000_P2P_CAPS_STATUS_OK: the capability at this table index is supported.
/// Any other value names a reason it is not (chipset, GPU, topology, regkey).
pub const P2P_CAPS_STATUS_OK: NvU8 = 0;

/// NV_MEMORY_VIRTUAL_ALLOCATION_PARAMS (cl0070.h): pAllocParms for
/// NV01_MEMORY_VIRTUAL - reserves a GPU VA range [offset, limit] in a VA space.
pub const VirtMemAllocParams = extern struct {
Expand Down Expand Up @@ -480,6 +551,19 @@ test "RM struct layouts match the NVIDIA ABI" {
try std.testing.expectEqual(@as(usize, 64), @offsetOf(ChannelAllocParams, "userd_offset"));
try std.testing.expectEqual(@as(usize, 4), @sizeOf(ChannelBindParams));
try std.testing.expectEqual(@as(usize, 3), @sizeOf(GpfifoScheduleParams));
try std.testing.expectEqual(@as(usize, 28), @sizeOf(Os55Params));
try std.testing.expectEqual(@as(usize, 12), @offsetOf(Os55Params, "h_client_src"));
try std.testing.expectEqual(@as(usize, 24), @offsetOf(Os55Params, "status"));
// A wrong size or offset here is a silent garbage read past the kernel's
// own struct, since NV_ESC_RM_DUP_OBJECT/GET_P2P_CAPS_V2 trust params_size.
try std.testing.expectEqual(@as(usize, 8348), @sizeOf(P2pCapsV2Params));
try std.testing.expectEqual(@as(usize, 128), @offsetOf(P2pCapsV2Params, "gpu_count"));
try std.testing.expectEqual(@as(usize, 140), @offsetOf(P2pCapsV2Params, "p2p_optimal_write_ces"));
// p2p_caps_status is a [9]u8 at 144; bus_peer_ids needs 4-byte alignment,
// so 3 bytes of padding sit between them (144+9=153, rounded up to 156).
try std.testing.expectEqual(@as(usize, 144), @offsetOf(P2pCapsV2Params, "p2p_caps_status"));
try std.testing.expectEqual(@as(usize, 156), @offsetOf(P2pCapsV2Params, "bus_peer_ids"));
try std.testing.expectEqual(@as(usize, 4252), @offsetOf(P2pCapsV2Params, "bus_egm_peer_ids"));
}

test "GPFIFO submission encodings" {
Expand Down
30 changes: 30 additions & 0 deletions src/nvidia/transport/freestanding.zig
Original file line number Diff line number Diff line change
Expand Up @@ -354,6 +354,34 @@ pub const Transport = struct {
self.untrack(h_object);
}

/// No CARD_INFO analog on baremetal, because there is no kernel node to enumerate
/// through. This HAL always addresses exactly the one GPU it was booted
/// against.
pub fn deviceCount(self: *Transport) Error!u32 {
if (!self.booted) return error.NotImplemented;
return 1;
}

/// NV_ESC_RM_DUP_OBJECT has no GSP RPC form yet, so cross-client memory
/// duping (the peer-transfer path) is Linux-only for now. Kept
/// syntactically valid so rm.zig compiles against either transport.
pub fn dupObject(
self: *Transport,
h_client: sdk.NvHandle,
h_parent: sdk.NvHandle,
h_new: sdk.NvHandle,
h_client_src: sdk.NvHandle,
h_object_src: sdk.NvHandle,
) Error!sdk.NvHandle {
_ = self;
_ = h_client;
_ = h_parent;
_ = h_new;
_ = h_client_src;
_ = h_object_src;
return error.NotImplemented;
}

// =======================================================================
// allocDevice -> the ROOT / DEVICE / SUBDEVICE alloc sequence over RPC. Mirrors
// linux.zig allocDevice's order + the sdk params it passes (only the registration
Expand Down Expand Up @@ -916,6 +944,8 @@ test "freestanding open() without metal inputs fails cleanly" {
var t = Transport{};
try testing.expectError(error.NotImplemented, t.rmAlloc(0, 0, 1, sdk.NV01_ROOT, null, 0));
try testing.expectError(error.NotImplemented, t.allocDevice(0));
try testing.expectError(error.NotImplemented, t.deviceCount());
try testing.expectError(error.NotImplemented, t.dupObject(0, 0, 1, 0, 0));
}

test {
Expand Down
42 changes: 42 additions & 0 deletions src/nvidia/transport/linux.zig
Original file line number Diff line number Diff line change
Expand Up @@ -115,6 +115,38 @@ pub const Transport = struct {
}
}

/// NV_ESC_RM_DUP_OBJECT (NVOS55): dup `h_object_src` (owned by client
/// `h_client_src`) into a new object `h_new` under `h_client`/`h_parent`.
/// The dup shares the same underlying resource as the source. For a
/// memory object that is the same physical pages, so writes through either
/// handle land in the same place. Returns the actual new handle.
pub fn dupObject(
self: *Transport,
h_client: sdk.NvHandle,
h_parent: sdk.NvHandle,
h_new: sdk.NvHandle,
h_client_src: sdk.NvHandle,
h_object_src: sdk.NvHandle,
) Error!sdk.NvHandle {
var p = sdk.Os55Params{
.h_client = h_client,
.h_parent = h_parent,
.h_object = h_new,
.h_client_src = h_client_src,
.h_object_src = h_object_src,
.flags = sdk.NV04_DUP_HANDLE_FLAGS_NONE,
.status = 0,
};
const req = ioctl.iowr(ioctl.NV_ESC_RM_DUP_OBJECT, sdk.Os55Params);
const rc = std.os.linux.ioctl(self.fd, req, @intFromPtr(&p));
switch (std.os.linux.errno(rc)) {
.SUCCESS => {},
else => return error.IoctlFailed,
}
if (p.status != 0) return error.RmAllocFailed;
return p.h_object;
}

/// NV_ESC_RM_FREE: free an RM object.
pub fn rmFree(self: *Transport, h_root: sdk.NvHandle, h_parent: sdk.NvHandle, h_object: sdk.NvHandle) void {
var p = sdk.Os00Params{ .h_root = h_root, .h_object_parent = h_parent, .h_object_old = h_object, .status = 0 };
Expand All @@ -134,6 +166,16 @@ pub const Transport = struct {
return cards;
}

/// How many GPUs the driver knows about (the valid entries of CARD_INFO).
pub fn deviceCount(self: *Transport) Error!u32 {
const cards = try self.cardInfo();
var count: u32 = 0;
for (cards) |entry| {
if (entry.valid != 0) count += 1;
}
return count;
}

/// NV_ESC_ATTACH_GPUS_TO_FD: attach a GPU (by gpu_id) to the control fd. The
/// kernel reads size/4 ids; a trailing 0 terminates the list.
fn attachGpu(self: *Transport, gpu_id: sdk.NvU32) Error!void {
Expand Down
Loading