From f155676d143fff39a452c4c4e57989112e1339cf Mon Sep 17 00:00:00 2001 From: Blaise Tine Date: Mon, 21 Sep 2026 21:23:11 -0700 Subject: [PATCH 01/31] simx: resume RTU traversal after any-hit and intersection verdicts Traversal was single-yield per lane: after the first candidate (a non-opaque triangle or a procedural AABB) was decided, the lane was resolved, so a rejected candidate hid everything behind it. Vulkan requires every candidate along the ray to be offered. A verdict that does not end the ray now resumes its walk above the decided candidate. Candidates are offered in ascending (t, key) order, key = (instance_id << 32) | record offset; the resumed walk is seeded with the committed hit and skips everything at or below the floor. An intersection shader's reported t commits only if it is nearer than the committed hit and inside the ray's interval. Also report a procedural leaf's gl_PrimitiveID as prim_base + index, as the RTL and the triangle path already do. Co-Authored-By: Claude Opus 5 (1M context) --- sim/simx/rtu/rtu_core.cpp | 40 ++++++++++++++++++------ sim/simx/rtu/rtu_types.h | 8 +++++ sim/simx/rtu/rtu_walker.cpp | 61 ++++++++++++++++++++++++++++++++----- 3 files changed, 92 insertions(+), 17 deletions(-) diff --git a/sim/simx/rtu/rtu_core.cpp b/sim/simx/rtu/rtu_core.cpp index 75492a860c..c3dce11781 100644 --- a/sim/simx/rtu/rtu_core.cpp +++ b/sim/simx/rtu/rtu_core.cpp @@ -340,9 +340,11 @@ class RtuCore::Impl { s.req = req; s.state = SlotState::READY; uint32_t first_active = uint32_t(-1); + s.lanes = {}; for (uint32_t t = 0; t < VX_CFG_NUM_THREADS; ++t) { if (s.req.tmask_bits & (1u << t)) { s.lanes[t].active = true; + s.lanes[t].walk_needed = true; if (first_active == uint32_t(-1)) first_active = t; } } @@ -375,7 +377,16 @@ class RtuCore::Impl { LaneState& l = s.lanes[t]; if (!l.cb_pending) continue; uint32_t action = req.cb_action[t]; - if (action == VX_RT_CB_ACCEPT || action == VX_RT_CB_TERMINATE) { + const bool decides = (l.cb_type == VX_RT_CB_TYPE_ANYHIT + || l.cb_type == VX_RT_CB_TYPE_PROC); + const bool accept = (action == VX_RT_CB_ACCEPT || action == VX_RT_CB_TERMINATE); + // An intersection shader reports its own t, which need not be nearer than + // what the walk already committed: only a nearer hit inside the ray's + // interval replaces it. + const float new_t = (l.cb_type == VX_RT_CB_TYPE_PROC) ? req.cb_hit_t[t] : l.cand_t; + const bool commits = accept && (!decides + || (new_t >= s.req.tmin[t] && new_t < (l.hit ? l.hit_t : s.req.tmax[t]))); + if (commits) { l.hit = true; // A procedural (IS) accept commits the shader's own hit_t; a triangle // AHS keeps the geometric candidate t. Either way the hitAttribute the @@ -398,15 +409,25 @@ class RtuCore::Impl { } } // IGNORE leaves the committed hit alone; DONE means the CHS dispatcher has - // finished shading an already-committed hit. Traversal is - // single-yield-per-lane, so either way the lane is resolved and the slot - // drops to its terminal record once every yielding lane has answered. + // finished shading an already-committed hit. An any-hit / intersection + // verdict that does not end the ray resumes its walk above the decided + // candidate, so every candidate along the ray is offered in (t, key) + // order; TERMINATE, or an accept under TERMINATE_ON_FIRST_HIT, ends it. l.cb_pending = false; - bool any_pending = false; + const bool ends = action == VX_RT_CB_TERMINATE + || (commits && (s.req.flags[t] & VX_RT_FLAG_TERMINATE_ON_FIRST_HIT)); + if (decides && !ends) { + l.has_floor = true; + l.floor_t = l.cand_t; + l.floor_key = l.cand_key; + l.walk_needed = true; + } + bool any_pending = false, any_walk = false; for (auto const& ll : s.lanes) { - if (ll.cb_pending) { any_pending = true; break; } + any_pending |= ll.cb_pending; + any_walk |= ll.walk_needed; } - if (!any_pending) s.state = SlotState::RESP; + if (!any_pending) s.state = any_walk ? SlotState::READY : SlotState::RESP; } // Clear this warp's callback-in-flight gate so the next queued CB_YIELD for // the same warp (e.g. the second SBT group) can be emitted. @@ -436,7 +457,7 @@ class RtuCore::Impl { Slot& s = pool_.at(best); uint32_t need = 0; for (uint32_t t = 0; t < VX_CFG_NUM_THREADS; ++t) { - if (s.lanes[t].active) ++need; + if (s.lanes[t].active && s.lanes[t].walk_needed) ++need; } uint32_t avail = 0; for (const auto& cx : contexts_) { @@ -452,7 +473,8 @@ class RtuCore::Impl { uint32_t next_free = 0; for (uint32_t t = 0; t < VX_CFG_NUM_THREADS; ++t) { - if (!s.lanes[t].active) continue; + if (!s.lanes[t].active || !s.lanes[t].walk_needed) continue; + s.lanes[t].walk_needed = false; while (contexts_[next_free].valid) ++next_free; bind_context(next_free, best, t, s.req.scene_root[t]); ++next_free; diff --git a/sim/simx/rtu/rtu_types.h b/sim/simx/rtu/rtu_types.h index 8d8858bce2..1a326407e7 100644 --- a/sim/simx/rtu/rtu_types.h +++ b/sim/simx/rtu/rtu_types.h @@ -436,6 +436,14 @@ struct LaneState { float cand_obj_d[3] = {0.f, 0.f, 0.f}; uint32_t hit_instance_id = 0; uint32_t hit_instance_custom = 0; + // Multi-candidate traversal. A verdict that does not end the ray resumes its + // walk above the decided candidate's (t, key); `walk_needed` marks the lanes + // the next promote binds contexts to. + uint64_t cand_key = 0; + bool has_floor = false; + float floor_t = 0.f; + uint64_t floor_key = 0; + bool walk_needed = false; }; struct Slot { diff --git a/sim/simx/rtu/rtu_walker.cpp b/sim/simx/rtu/rtu_walker.cpp index efb5724c21..1c76284fb6 100644 --- a/sim/simx/rtu/rtu_walker.cpp +++ b/sim/simx/rtu/rtu_walker.cpp @@ -105,8 +105,30 @@ struct WalkCtx { // at the top level (no instance). float best_obj_o[3], best_obj_d[3]; float yield_obj_o[3], yield_obj_d[3]; + // Candidates are offered one at a time in ascending (t, key) order, key = + // (instance_id << 32) | record offset -- unique per primitive per instance. A + // resumed walk skips everything at or below the last decided candidate's key. + uint64_t yield_key; + bool has_floor; + float floor_t; + uint64_t floor_key; }; +inline uint64_t cand_key(uint32_t instance_id, uint32_t record_off) { + return (uint64_t(instance_id) << 32) | record_off; +} + +// Whether a candidate at (t, key) replaces the pending one: nearer than the +// committed hit, above the resume floor, and first in (t, key) order. +inline bool cand_takes(const WalkCtx& ctx, float t, uint64_t key) { + if (!(t < ctx.best_t)) return false; + if (ctx.has_floor + && (t < ctx.floor_t || (t == ctx.floor_t && key <= ctx.floor_key))) + return false; + if (!ctx.yield_pending) return true; + return t < ctx.yield_t || (t == ctx.yield_t && key < ctx.yield_key); +} + // Depth-first walker for one BVH sub-tree under the supplied (object-space) // ray. Recurses on LeafInst so each instance's BLAS gets walked with its // transformed ray. ctx accumulates hits/yields across the whole call tree. @@ -175,8 +197,10 @@ void walk_bvh4_subtree(SceneView& sv, } } } else { // TriAction::Yield - if (t_hit < ctx.best_t && t_hit < ctx.yield_t) { + uint64_t key = cand_key(instance_id, tris_off + i * kVxBvhTriStride); + if (cand_takes(ctx, t_hit, key)) { ctx.yield_pending = true; + ctx.yield_key = key; ctx.yield_t = t_hit; ctx.yield_u = u; ctx.yield_v = v; ctx.yield_prim = leaf_prim_base + i; ctx.yield_instance = instance_id; @@ -220,10 +244,13 @@ void walk_bvh4_subtree(SceneView& sv, } // Procedural primitives are inherently non-opaque (the IS decides the // hit), so always stage an IS yield for the closest candidate. - if (t_near < ctx.best_t && t_near < ctx.yield_t) { + uint64_t key = cand_key(instance_id, + aabbs_off + i * uint32_t(sizeof(VxBvhProcAabb))); + if (cand_takes(ctx, t_near, key)) { ctx.yield_pending = true; + ctx.yield_key = key; ctx.yield_t = t_near; ctx.yield_u = 0.f; ctx.yield_v = 0.f; - ctx.yield_prim = i; + ctx.yield_prim = hdr->prim_base + i; // gl_PrimitiveID, as for LEAF_TRI ctx.yield_instance = instance_id; ctx.yield_custom = custom_id; ctx.yield_geom = hdr->geometry_index; @@ -418,6 +445,7 @@ bool emit_lane_result(const RtuReq& req, LaneState& l, uint32_t t, l.cand_prim = ctx.yield_prim; l.cand_instance = ctx.yield_instance; l.cand_custom = ctx.yield_custom; + l.cand_key = ctx.yield_key; return true; case LaneAction::YieldChs: l.cb_pending = true; @@ -448,9 +476,10 @@ bool emit_lane_result(const RtuReq& req, LaneState& l, uint32_t t, return false; // unreachable } -// Common init of the traversal accumulator from the ray. +// Common init of the traversal accumulator from the ray, and -- for a walk +// resumed after a callback verdict -- from the lane's committed hit and floor. WalkCtx init_ctx(const RtuReq& req, uint32_t t, - const float ro[3], const float rd[3]) { + const float ro[3], const float rd[3], const LaneState& l) { WalkCtx ctx; ctx.tmin = req.tmin[t]; ctx.tmax = req.tmax[t]; @@ -471,6 +500,20 @@ WalkCtx init_ctx(const RtuReq& req, uint32_t t, // under an instance). vcopy3(ctx.best_obj_o, ro); vcopy3(ctx.best_obj_d, rd); vcopy3(ctx.yield_obj_o, ro); vcopy3(ctx.yield_obj_d, rd); + ctx.yield_key = 0; + ctx.has_floor = l.has_floor; + ctx.floor_t = l.floor_t; + ctx.floor_key = l.floor_key; + if (l.has_floor && l.hit) { + ctx.any_hit = true; + ctx.best_t = l.hit_t; ctx.best_u = l.hit_u; ctx.best_v = l.hit_v; + ctx.best_prim = l.hit_prim; + ctx.best_instance = l.hit_instance_id; + ctx.best_custom = l.hit_instance_custom; + ctx.best_geom = l.hit_geometry; + vcopy3(ctx.best_obj_o, l.hit_obj_o); + vcopy3(ctx.best_obj_d, l.hit_obj_d); + } return ctx; } @@ -492,7 +535,7 @@ WalkResult FlatWalker::walk_lane(const RtuReq& req, uint32_t t, SceneView& sv, const float ro[3] = { req.origin_x[t], req.origin_y[t], req.origin_z[t] }; const float rd[3] = { req.dir_x[t], req.dir_y[t], req.dir_z[t] }; - WalkCtx ctx = init_ctx(req, t, ro, rd); + WalkCtx ctx = init_ctx(req, t, ro, rd, l); // TLAS scenes walk one or more instances; each instance points at a BLAS (a // triangle list) and (optionally) applies an object→world affine transform. @@ -626,8 +669,10 @@ WalkResult FlatWalker::walk_lane(const RtuReq& req, uint32_t t, SceneView& sv, } } } else { // TriAction::Yield - if (t_hit < ctx.best_t && t_hit < ctx.yield_t) { + uint64_t key = cand_key(inst_idx, blas_tri_off + i * kPhase2TriStride); + if (cand_takes(ctx, t_hit, key)) { ctx.yield_pending = true; + ctx.yield_key = key; ctx.yield_t = t_hit; ctx.yield_u = u; ctx.yield_v = v; ctx.yield_prim = i; ctx.yield_sbt = cls.yield_sbt_idx; @@ -661,7 +706,7 @@ WalkResult Bvh4Walker::walk_lane(const RtuReq& req, uint32_t t, SceneView& sv, const float ro[3] = { req.origin_x[t], req.origin_y[t], req.origin_z[t] }; const float rd[3] = { req.dir_x[t], req.dir_y[t], req.dir_z[t] }; - WalkCtx ctx = init_ctx(req, t, ro, rd); + WalkCtx ctx = init_ctx(req, t, ro, rd, l); // Top-level (non-instanced) triangles carry no instance flags. walk_bvh4_subtree(sv, ro, rd, root_off, 0, 0, 0, ctx, perf); From 991a8a5f39d3def13c3bd6974c6940f0431c441f Mon Sep 17 00:00:00 2001 From: Blaise Tine Date: Mon, 21 Sep 2026 21:23:11 -0700 Subject: [PATCH 02/31] kernel: pin WGATHER to the warp-gather registers; add diverge_loop test WGATHER writes every non-source lane of its destination, active or not, so a unit reading the packed operand across the warp sees all of it. In an allocatable register that clobbers inactive lanes' live values. The gather is now pinned to x31 (x30 for a second live operand), which the compiler reserves in any function naming them, and the RTU trace and DXA issue read the pinned registers directly. diverge_loop covers a divergent continue out of a loop body whose other path carries its own loop, both feeding one accumulator. Co-Authored-By: Claude Opus 5 (1M context) --- sw/kernel/include/vx_dxa.h | 95 +++++-------- sw/kernel/include/vx_intrinsics.h | 16 ++- sw/kernel/include/vx_raytrace.h | 5 +- tests/regression/diverge_loop/Makefile | 16 +++ tests/regression/diverge_loop/common.h | 20 +++ tests/regression/diverge_loop/kernel.cpp | 35 +++++ tests/regression/diverge_loop/main.cpp | 168 +++++++++++++++++++++++ 7 files changed, 294 insertions(+), 61 deletions(-) create mode 100644 tests/regression/diverge_loop/Makefile create mode 100644 tests/regression/diverge_loop/common.h create mode 100644 tests/regression/diverge_loop/kernel.cpp create mode 100644 tests/regression/diverge_loop/main.cpp diff --git a/sw/kernel/include/vx_dxa.h b/sw/kernel/include/vx_dxa.h index dbbc031e37..ea722b2752 100644 --- a/sw/kernel/include/vx_dxa.h +++ b/sw/kernel/include/vx_dxa.h @@ -45,6 +45,25 @@ extern "C" { // 1D and 2D: all rs2 lanes are zero, so rs2 = x0 (no second vx_wgather). // 3D–5D: rs2 carries coord2..coord4, requiring a second vx_wgather. +// The DXA reads its packed operands across the warp regardless of the thread +// mask, so it reads them from the warp-gather registers the gathers wrote +// (x31, and x30 for rs2; see __VX_WGATHER_IN). +#define __VX_DXA_ISSUE1(a0v) do { \ + register uint32_t __rs1 __asm__("x31") = (a0v); \ + __asm__ volatile (".insn r %0, 0, %1, x0, %2, x0\n\t" \ + : : "i"(VX_DXA_EXT_OPCODE), "i"(VX_DXA_FUNCT7), "r"(__rs1) \ + : "memory"); \ + } while (0) + +#define __VX_DXA_ISSUE2(a0v, a1v) do { \ + register uint32_t __rs1 __asm__("x31") = (a0v); \ + register uint32_t __rs2 __asm__("x30") = (a1v); \ + __asm__ volatile (".insn r %0, 0, %1, x0, %2, %3\n\t" \ + : : "i"(VX_DXA_EXT_OPCODE), "i"(VX_DXA_FUNCT7), "r"(__rs1), \ + "r"(__rs2) \ + : "memory"); \ + } while (0) + inline uint32_t vx_dxa_pack_meta(uint32_t desc_slot, uint32_t barrier_id) { return (barrier_id << 4) | desc_slot; } @@ -59,11 +78,7 @@ inline void vx_dxa_issue_1d_wg(uint32_t desc_slot, (size_t)meta, (size_t)coord0, (size_t)0u); - __asm__ volatile ( - ".insn r %0, 0, %1, x0, %2, x0\n\t" - : - : "i"(VX_DXA_EXT_OPCODE), "i"(VX_DXA_FUNCT7), "r"(a0) - : "memory"); + __VX_DXA_ISSUE1(a0); } // 2D: rs1 = wgather(smem_addr, meta, coord0, coord1), rs2 = x0 @@ -77,11 +92,7 @@ inline void vx_dxa_issue_2d_wg(uint32_t desc_slot, (size_t)meta, (size_t)coord0, (size_t)coord1); - __asm__ volatile ( - ".insn r %0, 0, %1, x0, %2, x0\n\t" - : - : "i"(VX_DXA_EXT_OPCODE), "i"(VX_DXA_FUNCT7), "r"(a0) - : "memory"); + __VX_DXA_ISSUE1(a0); } // 3D–5D: rs2 = wgather(coord2, coord3, coord4, 0) @@ -96,15 +107,11 @@ inline void vx_dxa_issue_3d_wg(uint32_t desc_slot, (size_t)meta, (size_t)coord0, (size_t)coord1); - const uint32_t a1 = (uint32_t)vx_wgather((size_t)coord2, + const uint32_t a1 = (uint32_t)__VX_WGATHER_IN("x30", 0, (size_t)coord2, (size_t)0u, (size_t)0u, (size_t)0u); - __asm__ volatile ( - ".insn r %0, 0, %1, x0, %2, %3\n\t" - : - : "i"(VX_DXA_EXT_OPCODE), "i"(VX_DXA_FUNCT7), "r"(a0), "r"(a1) - : "memory"); + __VX_DXA_ISSUE2(a0, a1); } inline void vx_dxa_issue_4d_wg(uint32_t desc_slot, @@ -119,15 +126,11 @@ inline void vx_dxa_issue_4d_wg(uint32_t desc_slot, (size_t)meta, (size_t)coord0, (size_t)coord1); - const uint32_t a1 = (uint32_t)vx_wgather((size_t)coord2, + const uint32_t a1 = (uint32_t)__VX_WGATHER_IN("x30", 0, (size_t)coord2, (size_t)coord3, (size_t)0u, (size_t)0u); - __asm__ volatile ( - ".insn r %0, 0, %1, x0, %2, %3\n\t" - : - : "i"(VX_DXA_EXT_OPCODE), "i"(VX_DXA_FUNCT7), "r"(a0), "r"(a1) - : "memory"); + __VX_DXA_ISSUE2(a0, a1); } inline void vx_dxa_issue_5d_wg(uint32_t desc_slot, @@ -143,15 +146,11 @@ inline void vx_dxa_issue_5d_wg(uint32_t desc_slot, (size_t)meta, (size_t)coord0, (size_t)coord1); - const uint32_t a1 = (uint32_t)vx_wgather((size_t)coord2, + const uint32_t a1 = (uint32_t)__VX_WGATHER_IN("x30", 0, (size_t)coord2, (size_t)coord3, (size_t)coord4, (size_t)0u); - __asm__ volatile ( - ".insn r %0, 0, %1, x0, %2, %3\n\t" - : - : "i"(VX_DXA_EXT_OPCODE), "i"(VX_DXA_FUNCT7), "r"(a0), "r"(a1) - : "memory"); + __VX_DXA_ISSUE2(a0, a1); } // Multicast DXA issues read GMEM once and replay SMEM writes to multiple @@ -168,15 +167,11 @@ inline void vx_dxa_issue_1d_multicast_wg(uint32_t desc_slot, (size_t)meta, (size_t)coord0, (size_t)0u); - const uint32_t a1 = (uint32_t)vx_wgather((size_t)0, + const uint32_t a1 = (uint32_t)__VX_WGATHER_IN("x30", 0, (size_t)0, (size_t)0, (size_t)0, (size_t)cta_mask); - __asm__ volatile ( - ".insn r %0, 0, %1, x0, %2, %3\n\t" - : - : "i"(VX_DXA_EXT_OPCODE), "i"(VX_DXA_FUNCT7), "r"(a0), "r"(a1) - : "memory"); + __VX_DXA_ISSUE2(a0, a1); } // 2D multicast: rs2 = wgather(0, 0, 0, cta_mask) @@ -191,15 +186,11 @@ inline void vx_dxa_issue_2d_multicast_wg(uint32_t desc_slot, (size_t)meta, (size_t)coord0, (size_t)coord1); - const uint32_t a1 = (uint32_t)vx_wgather((size_t)0, + const uint32_t a1 = (uint32_t)__VX_WGATHER_IN("x30", 0, (size_t)0, (size_t)0, (size_t)0, (size_t)cta_mask); - __asm__ volatile ( - ".insn r %0, 0, %1, x0, %2, %3\n\t" - : - : "i"(VX_DXA_EXT_OPCODE), "i"(VX_DXA_FUNCT7), "r"(a0), "r"(a1) - : "memory"); + __VX_DXA_ISSUE2(a0, a1); } // 3D multicast: rs2 = wgather(coord2, 0, 0, cta_mask) @@ -215,15 +206,11 @@ inline void vx_dxa_issue_3d_multicast_wg(uint32_t desc_slot, (size_t)meta, (size_t)coord0, (size_t)coord1); - const uint32_t a1 = (uint32_t)vx_wgather((size_t)coord2, + const uint32_t a1 = (uint32_t)__VX_WGATHER_IN("x30", 0, (size_t)coord2, (size_t)0, (size_t)0, (size_t)cta_mask); - __asm__ volatile ( - ".insn r %0, 0, %1, x0, %2, %3\n\t" - : - : "i"(VX_DXA_EXT_OPCODE), "i"(VX_DXA_FUNCT7), "r"(a0), "r"(a1) - : "memory"); + __VX_DXA_ISSUE2(a0, a1); } // 4D multicast: rs2 = wgather(coord2, coord3, 0, cta_mask) @@ -240,15 +227,11 @@ inline void vx_dxa_issue_4d_multicast_wg(uint32_t desc_slot, (size_t)meta, (size_t)coord0, (size_t)coord1); - const uint32_t a1 = (uint32_t)vx_wgather((size_t)coord2, + const uint32_t a1 = (uint32_t)__VX_WGATHER_IN("x30", 0, (size_t)coord2, (size_t)coord3, (size_t)0, (size_t)cta_mask); - __asm__ volatile ( - ".insn r %0, 0, %1, x0, %2, %3\n\t" - : - : "i"(VX_DXA_EXT_OPCODE), "i"(VX_DXA_FUNCT7), "r"(a0), "r"(a1) - : "memory"); + __VX_DXA_ISSUE2(a0, a1); } // 5D multicast: rs2 = wgather(coord2, coord3, coord4, cta_mask) @@ -266,15 +249,11 @@ inline void vx_dxa_issue_5d_multicast_wg(uint32_t desc_slot, (size_t)meta, (size_t)coord0, (size_t)coord1); - const uint32_t a1 = (uint32_t)vx_wgather((size_t)coord2, + const uint32_t a1 = (uint32_t)__VX_WGATHER_IN("x30", 0, (size_t)coord2, (size_t)coord3, (size_t)coord4, (size_t)cta_mask); - __asm__ volatile ( - ".insn r %0, 0, %1, x0, %2, %3\n\t" - : - : "i"(VX_DXA_EXT_OPCODE), "i"(VX_DXA_FUNCT7), "r"(a0), "r"(a1) - : "memory"); + __VX_DXA_ISSUE2(a0, a1); } #ifdef __cplusplus diff --git a/sw/kernel/include/vx_intrinsics.h b/sw/kernel/include/vx_intrinsics.h index 11ee4081a8..427169096e 100644 --- a/sw/kernel/include/vx_intrinsics.h +++ b/sw/kernel/include/vx_intrinsics.h @@ -507,8 +507,17 @@ inline __attribute__((const)) float vx_quad_ddy_f32(float value) { // Each lane gathers a value from the source lane's register file. // S = source lane (compile-time constant, 0-3). // The source lane retains its own rd value; lane (S+1) gets v1[S], (S+2) gets v2[S], (S+3) gets v3[S]. -#define __VX_WGATHER(src_lane, self_val, v1, v2, v3) ({ \ - size_t __ret = (self_val); \ +// +// WGATHER writes every non-source lane of rd, active or not, so a unit reading +// the packed operand across the warp (the RTU trace config, a DXA descriptor) +// sees all of it whatever the thread mask. In an allocatable register that +// would clobber an inactive lane's live value, so rd is pinned to a dedicated +// warp-gather register -- x31, or x30 for a second packed operand live at the +// same time -- which the compiler reserves in any function naming one. A +// consumer of the packed lanes must read that register directly (bind it with +// a register variable); a copy elsewhere is written on active lanes only. +#define __VX_WGATHER_IN(reg, src_lane, self_val, v1, v2, v3) ({ \ + register size_t __ret __asm__(reg) = (self_val); \ __asm__ volatile ( \ ".insn r4 %1, 0, %2, %0, %3, %4, %5" \ : "+r"(__ret) \ @@ -518,6 +527,9 @@ inline __attribute__((const)) float vx_quad_ddy_f32(float value) { __ret; \ }) +#define __VX_WGATHER(src_lane, self_val, v1, v2, v3) \ + __VX_WGATHER_IN("x31", src_lane, self_val, v1, v2, v3) + // Warp-level gather with source lane 0. inline __attribute__((const)) size_t vx_wgather(size_t self_val, size_t v1, size_t v2, size_t v3) { diff --git a/sw/kernel/include/vx_raytrace.h b/sw/kernel/include/vx_raytrace.h index 0ff732b88c..6b415edec4 100644 --- a/sw/kernel/include/vx_raytrace.h +++ b/sw/kernel/include/vx_raytrace.h @@ -170,9 +170,12 @@ uint32_t vx_rt_wtrace(uint32_t scene_ptr, uint32_t payload_ptr, // list (read by HW convention, like the tensor unit's fragment window); // the encoding itself only names rd/rs1. Named operands (not %0/%1) keep // the field references stable across the long register-binding list. + // The RTU reads the packed config lanes regardless of the thread mask, so it + // reads them from the warp-gather register itself (see __VX_WGATHER_IN). + register uint32_t cfg_reg __asm__("x31") = cfg; __asm__ volatile (".insn r %[op], 7, 0, %[hnd], %[cfg], x0" : [hnd]"=r"(handle) - : [op]"i"(RISCV_CUSTOM1), [cfg]"r"(cfg), + : [op]"i"(RISCV_CUSTOM1), [cfg]"r"(cfg_reg), "f"(r0), "f"(r1), "f"(r2), "f"(r3), "f"(r4), "f"(r5), "f"(r6), "f"(r7)); return handle; diff --git a/tests/regression/diverge_loop/Makefile b/tests/regression/diverge_loop/Makefile new file mode 100644 index 0000000000..4559039b6e --- /dev/null +++ b/tests/regression/diverge_loop/Makefile @@ -0,0 +1,16 @@ +ROOT_DIR := $(realpath ../../..) +include $(ROOT_DIR)/config.mk + +PROJECT := diverge_loop + +SRC_DIR := $(VORTEX_HOME)/tests/regression/$(PROJECT) + +SRCS := $(SRC_DIR)/main.cpp + +VX_SRCS := $(SRC_DIR)/kernel.cpp + +OPTS ?= -n256 + +KERNEL_LIB := vortex2 + +include ../common.mk \ No newline at end of file diff --git a/tests/regression/diverge_loop/common.h b/tests/regression/diverge_loop/common.h new file mode 100644 index 0000000000..cf485d40d1 --- /dev/null +++ b/tests/regression/diverge_loop/common.h @@ -0,0 +1,20 @@ +#ifndef _COMMON_H_ +#define _COMMON_H_ + +#include + +#ifndef TYPE +#define TYPE float +#endif + +typedef struct { + uint32_t num_points; + uint32_t num_samples; + uint32_t num_shadows; + uint64_t t_addr; + uint64_t color_addr; + uint64_t occl_addr; + uint64_t dst_addr; +} kernel_arg_t; + +#endif diff --git a/tests/regression/diverge_loop/kernel.cpp b/tests/regression/diverge_loop/kernel.cpp new file mode 100644 index 0000000000..558367613c --- /dev/null +++ b/tests/regression/diverge_loop/kernel.cpp @@ -0,0 +1,35 @@ +#include +#include "common.h" + +// A divergent early `continue` out of a loop body whose other path carries its +// own loop, both feeding one accumulator that is live around the outer loop: +// the reconvergence shape of a ray-tracing raygen shader (a missed ray adds the +// sky and continues, a hit one first attenuates by its shadow rays). +__kernel void kernel_main(kernel_arg_t* __UNIFORM__ arg) { + auto t_ptr = reinterpret_cast(arg->t_addr); + auto color_ptr = reinterpret_cast(arg->color_addr); + auto occl_ptr = reinterpret_cast(arg->occl_addr); + auto dst_ptr = reinterpret_cast(arg->dst_addr); + uint32_t idx = blockIdx.x * blockDim.x + threadIdx.x; + if (idx >= arg->num_points) + return; + + float acc = 0.0f; + for (uint32_t s = 0; s < arg->num_samples; ++s) { + float t = t_ptr[idx * arg->num_samples + s]; + float h = color_ptr[idx]; + if (t < 0.0f) { + acc += h; + continue; + } + for (uint32_t j = 0; j < arg->num_shadows; ++j) { + float occluded = 1.0f; + if (occl_ptr[idx] > 0.5f) + occluded = t - occl_ptr[idx]; + if (occluded > 0.0f) + h *= 0.3f; + } + acc += h; + } + dst_ptr[idx] = acc; +} diff --git a/tests/regression/diverge_loop/main.cpp b/tests/regression/diverge_loop/main.cpp new file mode 100644 index 0000000000..f6c05dd374 --- /dev/null +++ b/tests/regression/diverge_loop/main.cpp @@ -0,0 +1,168 @@ +// Copyright © 2019-2023 +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// http://www.apache.org/licenses/LICENSE-2.0 + +// diverge_loop — divergent continue + inner loop reconverging on one accumulator. +// +// Async pattern: uploads are fire-and-forget; the launch produces an event; +// the dst readback gates on that event; the host waits once at the end. +// The per-queue worker serializes ops in FIFO order, so no inter-step host sync is needed. + +#include +#include "common.h" + +#include +#include +#include +#include +#include +#include +#include + +#define CHECK(expr) do { \ + vx_result_t _r = (expr); \ + if (_r != VX_SUCCESS) { \ + std::fprintf(stderr, "FAIL %s:%d: '%s' returned %s\n", \ + __FILE__, __LINE__, #expr, vx_result_string(_r)); \ + std::exit(1); \ + } \ +} while (0) + +namespace { +const char* kernel_file = "kernel.vxbin"; +uint32_t size = 16; + +void parse_args(int argc, char** argv) { + int c; + while ((c = getopt(argc, argv, "n:k:h")) != -1) { + switch (c) { + case 'n': size = std::atoi(optarg); break; + case 'k': kernel_file = optarg; break; + default: + std::cout << "Usage: [-k kernel] [-n words] [-h]" << std::endl; + std::exit(c == 'h' ? 0 : -1); + } + } +} + +bool float_eq(float a, float b) { + union fi { float f; int32_t i; }; + fi fa{a}, fb{b}; + return std::abs(fa.i - fb.i) <= 6; +} +} // namespace + +int main(int argc, char** argv) { + parse_args(argc, argv); + std::srand(50); + + const uint32_t num_points = size; + const uint64_t buf_size = num_points * sizeof(TYPE); + const uint32_t num_samples = 2, num_shadows = 2; + std::cout << "diverge_loop vortex2: n=" << num_points << std::endl; + + vx_device_h dev = nullptr; + CHECK(vx_device_open(0, &dev)); + + vx_queue_info_t qi = { sizeof(qi), nullptr, VX_QUEUE_PRIORITY_NORMAL, 0 }; + vx_queue_h q = nullptr; + CHECK(vx_queue_create(dev, &qi, &q)); + + vx_buffer_h t_buf=nullptr, src0_buf=nullptr, src1_buf=nullptr, dst_buf=nullptr; + CHECK(vx_buffer_create(dev, buf_size * num_samples, VX_MEM_READ, &t_buf)); + CHECK(vx_buffer_create(dev, buf_size, VX_MEM_READ, &src0_buf)); + CHECK(vx_buffer_create(dev, buf_size, VX_MEM_READ, &src1_buf)); + CHECK(vx_buffer_create(dev, buf_size, VX_MEM_WRITE, &dst_buf)); + + vx_module_h mod = nullptr; + vx_kernel_h kern = nullptr; + CHECK(vx_module_load_file(dev, kernel_file, &mod)); + CHECK(vx_module_get_kernel(mod, "main", &kern)); + + kernel_arg_t kernel_arg{}; + kernel_arg.num_points = num_points; + kernel_arg.num_samples = num_samples; + kernel_arg.num_shadows = num_shadows; + CHECK(vx_buffer_address(t_buf, &kernel_arg.t_addr)); + CHECK(vx_buffer_address(src0_buf, &kernel_arg.color_addr)); + CHECK(vx_buffer_address(src1_buf, &kernel_arg.occl_addr)); + CHECK(vx_buffer_address(dst_buf, &kernel_arg.dst_addr)); + + // Every divergence pattern within a warp: a lane's samples miss or hit + // independently, and a hit lane is occluded or not. + std::vector h_t(num_points * num_samples), h_src0(num_points), + h_src1(num_points), h_dst(num_points); + for (uint32_t i = 0; i < num_points; ++i) { + for (uint32_t s = 0; s < num_samples; ++s) + h_t[i * num_samples + s] = ((i >> s) & 1) ? -1.0f : 10.0f + i; + h_src0[i] = 1.0f + static_cast(std::rand()) / RAND_MAX; + h_src1[i] = ((i >> 2) & 1) ? 1.0f + static_cast(std::rand()) / RAND_MAX : 0.0f; + } + + // ----- Async chain: 2 writes → launch → read → 1 wait ----- + // The kernel-args block is passed as a host blob — no args device buffer needed. + CHECK(vx_enqueue_write(q, t_buf, 0, h_t.data(), buf_size * num_samples, 0,nullptr,nullptr)); + CHECK(vx_enqueue_write(q, src0_buf, 0, h_src0.data(), buf_size, 0,nullptr,nullptr)); + CHECK(vx_enqueue_write(q, src1_buf, 0, h_src1.data(), buf_size, 0,nullptr,nullptr)); + + uint32_t grid[1], block[1]; + CHECK(vx_device_max_occupancy_grid(dev, 1, &num_points, grid, block)); + + vx_launch_info_t li{}; + li.struct_size = sizeof(li); + li.kernel = kern; + li.args_host = &kernel_arg; + li.args_size = sizeof(kernel_arg); + li.ndim = 1; + li.grid_dim[0] = grid[0]; + li.block_dim[0]= block[0]; + + vx_event_h launch_ev=nullptr, read_ev=nullptr; + CHECK(vx_enqueue_launch(q, &li, 0, nullptr, &launch_ev)); + CHECK(vx_enqueue_read(q, h_dst.data(), dst_buf, 0, buf_size, + 1, &launch_ev, &read_ev)); + CHECK(vx_event_wait_value(read_ev, 1, VX_TIMEOUT_INFINITE)); + + int errors = 0; + for (uint32_t i = 0; i < num_points; ++i) { + TYPE ref = 0.0f; + for (uint32_t s = 0; s < num_samples; ++s) { + float t = h_t[i * num_samples + s]; + float h = h_src0[i]; + if (t < 0.0f) { ref += h; continue; } + for (uint32_t j = 0; j < num_shadows; ++j) { + float occluded = 1.0f; + if (h_src1[i] > 0.5f) occluded = t - h_src1[i]; + if (occluded > 0.0f) h *= 0.3f; + } + ref += h; + } + if (!float_eq(h_dst[i], ref)) { + if (errors < 16) + std::printf("*** [%u] expected=%f actual=%f\n", i, ref, h_dst[i]); + ++errors; + } + } + + vx_event_release(read_ev); + vx_event_release(launch_ev); + vx_buffer_release(dst_buf); + vx_buffer_release(src1_buf); + vx_buffer_release(src0_buf); + vx_buffer_release(t_buf); + vx_kernel_release(kern); + vx_module_release(mod); + vx_queue_release(q); + vx_device_dump_perf(dev, stdout); + vx_device_release(dev); + + if (errors) { + std::cout << "Found " << errors << " errors!\nFAILED!" << std::endl; + return 1; + } + std::cout << "PASSED!" << std::endl; + return 0; +} From 47a98b720b0f4b13f68528b3653ddbbd41c78da4 Mon Sep 17 00:00:00 2001 From: Blaise Tine Date: Mon, 21 Sep 2026 22:04:47 -0700 Subject: [PATCH 03/31] rtu: resume RTL traversal after any-hit and intersection verdicts Mirror the SimX multi-candidate model in the RTL scheduler. A verdict that does not end the ray re-walks that lane from the root, seeded with the committed hit as its t_max and the decided candidate as a floor, so every candidate along the ray is offered in ascending (t, key) order, key = {instance id, record offset}. - context word: the staged candidate's key and the resume floor (t, key); the key compares are precomputed at ALIGN off the store row, so EXEC only adds the t compares to the staging condition - barrier walker: a RES job also reads the committed-t row and drops an intersection shader's accept that is not nearer than the committed hit, then relaunches the non-terminating deciding lanes (re-walk template keeps the world-ray reciprocals and skips the setup) and re-arms finalise for them only (live mask) - core: T_RWAIT takes the next candidate batch as well as the terminal record; the scheduler's yield is gated off the cycle a resume job retires, when the flags it clears still read old - flat TLAS walk scans every instance instead of stopping at the first one that staged a candidate New test rt_smoke_ahs_multi: six stacked triangles (a t tie, an opaque one behind) with per-lane accept targets; checks the offer order and the final hit per lane. Passes on simx and rtlsim. Co-Authored-By: Claude Opus 5 (1M context) --- hw/rtl/rtu/VX_rtu_core.sv | 9 +- hw/rtl/rtu/VX_rtu_scheduler.sv | 183 ++++++++++++++-- tests/raytracing/Makefile | 2 +- tests/raytracing/rt_smoke_ahs_multi/Makefile | 20 ++ tests/raytracing/rt_smoke_ahs_multi/common.h | 62 ++++++ .../raytracing/rt_smoke_ahs_multi/kernel.cpp | 58 +++++ tests/raytracing/rt_smoke_ahs_multi/main.cpp | 204 ++++++++++++++++++ 7 files changed, 512 insertions(+), 26 deletions(-) create mode 100644 tests/raytracing/rt_smoke_ahs_multi/Makefile create mode 100644 tests/raytracing/rt_smoke_ahs_multi/common.h create mode 100644 tests/raytracing/rt_smoke_ahs_multi/kernel.cpp create mode 100644 tests/raytracing/rt_smoke_ahs_multi/main.cpp diff --git a/hw/rtl/rtu/VX_rtu_core.sv b/hw/rtl/rtu/VX_rtu_core.sv index f7bd8af8ea..738f92085d 100644 --- a/hw/rtl/rtu/VX_rtu_core.sv +++ b/hw/rtl/rtu/VX_rtu_core.sv @@ -143,7 +143,7 @@ module VX_rtu_core import VX_gpu_pkg::*, VX_rtu_pkg::*; #( T_CBWAIT = 3'd4, // candidate returned; await the CONTINUE's t T_CBATTR = 3'd5, // ... and its hitAttribute (CONT beat 1) T_RESUME = 3'd6, // release this slot's yield barrier - T_RWAIT = 3'd7; // await the resume commit -> terminal record + T_RWAIT = 3'd7; // await the resume commit -> next batch | terminal record reg [NUM_SLOTS-1:0][2:0] tstate; reg [NUM_SLOTS-1:0][NUM_LANES-1:0] req_mask; @@ -692,7 +692,12 @@ module VX_rtu_core import VX_gpu_pkg::*, VX_rtu_pkg::*; #( tstate[s] <= T_RWAIT; end T_RWAIT: begin - if (sch_done[s]) begin + // a verdict that did not end a lane's ray re-walks it: + // the next candidate batch yields like the first + if (sch_yield[s]) begin + is_cand[s] <= 1'b1; + tstate[s] <= T_WRITE; + end else if (sch_done[s]) begin is_cand[s] <= 1'b0; tstate[s] <= T_WRITE; end diff --git a/hw/rtl/rtu/VX_rtu_scheduler.sv b/hw/rtl/rtu/VX_rtu_scheduler.sv index cec290c883..958251f322 100644 --- a/hw/rtl/rtu/VX_rtu_scheduler.sv +++ b/hw/rtl/rtu/VX_rtu_scheduler.sv @@ -184,6 +184,14 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( logic [2:0][31:0] inv_d; logic [31:0] best_t; logic [31:0] yld_t; // staged candidate's t (compare copy) + logic [31:0] yld_ki; // staged candidate's key: instance id + logic [31:0] yld_ko; // ... and record offset + // resume floor: a re-walk offers only candidates above the last + // decided one in (t, key) order + logic has_floor; + logic [31:0] floor_t; + logic [31:0] floor_ki; + logic [31:0] floor_ko; logic [31:0] cur_off; logic [LB-1:0] f_idx; logic [LB-1:0] f_total; @@ -228,6 +236,10 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( reg [NUM_CTX-1:0] rdy_set; // wake bits (event-driven) wire [NUM_CTX-1:0] rdy_next; // next-cycle wake vector (see SELECT) reg [NUM_CTX-1:0] fresh_set; // first pass runs the launch init + reg [NUM_CTX-1:0] rewalk_set;// first pass runs the re-walk init + reg [NUM_CTX-1:0] live_q; // lane walks in the current round + reg [NUM_CTX-1:0] seed_v_q; // a resume committed seed_t_q + reg [NUM_CTX-1:0][31:0] seed_t_q; reg [NUM_CTX-1:0] done_q; reg [NUM_CTX-1:0] mask_q; reg [NUM_CTX-1:0] hit_q; @@ -274,12 +286,19 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( reg s1_valid; reg [CTX_TAG_W-1:0] s1_sel; reg s1_fresh; + reg s1_rewalk; wire [SLOT_W-1:0] s1_slot = SLOT_W'(32'(s1_sel) / NUM_LANES); // ═══════════════════════ ALIGN snapshot ═══════════════════════════ reg x_valid; reg [CTX_TAG_W-1:0] sel_q; reg fresh_q; + reg rewalk_q; + // candidate key order, precomputed at ALIGN off the store row: the key of + // the record under test ({instance, record offset}) against the floor and + // the staged candidate, so EXEC only adds the t compares + reg key_gt_floor_q; + reg key_lt_yld_q; ctx_state_t word_q; lane_ray_t ray_q; reg [BUF_BITS-1:0] fbuf_q; @@ -457,6 +476,8 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( reg [COLL_SIZE-1:0][NODE_W-1:0][31:0] coll_ordt; reg [COLL_SIZE-1:0] coll_prochit; + wire [31:0] cand_ki_al = cs_word.in_blas ? cs_word.inst_id : 32'd0; + // ═══════════════════════ stage advance ════════════════════════════ always_ff @(posedge clk) begin if (reset) begin @@ -471,18 +492,22 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( s1_valid <= g1_valid; s1_sel <= g1_idx; s1_fresh <= g1_valid && fresh_set[g1_idx]; + s1_rewalk <= g1_valid && rewalk_set[g1_idx]; x_valid <= s1_valid; if (s1_valid) begin sel_q <= s1_sel; fresh_q <= s1_fresh; + rewalk_q <= s1_rewalk; + key_gt_floor_q <= {cand_ki_al, cs_word.cur_off} > {cs_word.floor_ki, cs_word.floor_ko}; + key_lt_yld_q <= {cand_ki_al, cs_word.cur_off} < {cs_word.yld_ki, cs_word.yld_ko}; word_q <= cs_word; ray_q <= lane_ray_t'(ray_rdata); fbuf_q <= fbuf; stacktop_q <= stk_rdata; - // a fresh context's store row is stale: its walk starts at the - // scene base (the init template's cur_off is 0) + // a fresh context's store row is stale and a re-walk restarts: + // either walk starts at the scene base (the template's cur_off is 0) structaddr_q <= slot_scene[s1_slot] - + (s1_fresh ? ADDRW'(0) : ADDRW'(cs_word.cur_off)); + + ((s1_fresh || s1_rewalk) ? ADDRW'(0) : ADDRW'(cs_word.cur_off)); sp_q <= sp_q_arr[s1_sel]; flags_q <= slot_flags[s1_slot]; cull_q <= slot_cull[s1_slot]; @@ -939,8 +964,11 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( // ═══════════════════════ barrier walker ═══════════════════════════ // Runs the two whole-slot record operations one row at a time: // FIN — stage CHS (hit->yld row copy) and MISS (zeroed attributes) - // RES — commit accepted candidates (yld->hit row copy, attr merge) + // RES — commit accepted candidates (yld->hit row copy, attr merge), then + // re-walk every lane whose verdict did not end its ray localparam [3:0] BW_IDLE = 4'd0, + BW_RD3 = 4'd10, // RES: request the committed-t row + BW_CAP3 = 4'd11, // ... and drop out-of-range IS accepts BW_RD = 4'd1, // request the source row BW_CAP = 4'd2, // capture it BW_RD2 = 4'd3, // request the CONT-t row (RES field 0) @@ -997,11 +1025,11 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( always @(*) begin for (integer j = 0; j < NUM_LANES; j = j + 1) begin fin_chs_mask[j] = fin_req && fin_chs_en - && mask_q[32'(fin_slot)*NUM_LANES + j] + && live_q[32'(fin_slot)*NUM_LANES + j] && !yld_q[32'(fin_slot)*NUM_LANES + j] && hit_q[32'(fin_slot)*NUM_LANES + j]; fin_miss_mask[j] = fin_req && fin_miss_en - && mask_q[32'(fin_slot)*NUM_LANES + j] + && live_q[32'(fin_slot)*NUM_LANES + j] && !yld_q[32'(fin_slot)*NUM_LANES + j] && !hit_q[32'(fin_slot)*NUM_LANES + j]; end @@ -1030,6 +1058,29 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( end end + // An intersection shader reports its own t: it commits only if nearer than + // the committed hit (the captured row at BW_CAP3). The shader already + // checked it against the ray interval it knows of, which cannot include + // the opaque hits the walk committed on its own. + reg [NUM_LANES-1:0] bw_keep_mask; + // the lanes whose verdict did not end the ray: they re-walk above it + reg [NUM_LANES-1:0] res_rewalk; + wire bw_term_first = ((32'(slot_flags[bw_slot]) & 32'(`VX_RT_FLAG_TERMINATE_ON_FIRST_HIT)) != 0); + always @(*) begin + for (integer j = 0; j < NUM_LANES; j = j + 1) begin + logic [CTX_TAG_W-1:0] c; + logic is_proc, decides, ends; + c = CTX_TAG_W'(32'(bw_slot) * NUM_LANES + j); + is_proc = (cbtype_q[c] == RTU_CB_TYPE_BITS'(`VX_RT_CB_TYPE_PROC)); + decides = is_proc || (cbtype_q[c] == RTU_CB_TYPE_BITS'(`VX_RT_CB_TYPE_ANYHIT)); + bw_keep_mask[j] = bw_copy_mask[j] + && !(is_proc && hit_q[c] && !(bw_data2[j*32 +: 32] < ws_rdata[j*32 +: 32])); + ends = (act_q[c] == RTU_CB_ACTION_BITS'(`VX_RT_CB_TERMINATE)) + || (bw_copy_mask[j] && bw_term_first); + res_rewalk[j] = bw_is_res && yld_q[c] && decides && !ends; + end + end + // walker window-port requests reg bw_rd_req, bw_wr_req; reg [WS_ADDRW-1:0] bw_rd_addr, bw_wr_addr; @@ -1057,6 +1108,10 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( bw_rd_req = 1'b1; bw_rd_addr = {bw_slot, RTU_WS_WORD_BITS'(RTU_WS_CONT_T)}; end + BW_RD3: begin + bw_rd_req = 1'b1; + bw_rd_addr = {bw_slot, RTU_WS_WORD_BITS'(RTU_WS_HIT_BASE)}; + end BW_WR: begin bw_wr_req = 1'b1; bw_wr_addr = {bw_slot, bw_dst_base + RTU_WS_WORD_BITS'(32'(bw_field))}; @@ -1126,7 +1181,10 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( // ═══════════════════════ EXEC: the context FSM ════════════════════ // Effective word: a fresh (just-launched) context ignores the stale store - // row and starts from the init template. + // row and starts from the init template. A re-walk (resumed after a + // callback verdict that did not end the ray) restarts from the root with + // the committed hit as its t_max and the decided candidate as its floor; + // the world-ray reciprocals are unchanged, so it skips the setup. ctx_state_t word_x; always @(*) begin word_x = word_q; @@ -1135,9 +1193,33 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( word_x.cstate = FLAT ? CS_HDR_REQ : CS_SETUP; word_x.best_t = ray_q.t_max; word_x.yld_t = ray_q.t_max; + end else if (rewalk_q) begin + word_x = '0; + word_x.cstate = CS_HDR_REQ; + word_x.inv_d = word_q.inv_d; + word_x.best_t = seed_v_q[sel_q] ? seed_t_q[sel_q] : word_q.best_t; + word_x.yld_t = ray_q.t_max; + word_x.has_floor = 1'b1; + word_x.floor_t = word_q.yld_t; + word_x.floor_ki = word_q.yld_ki; + word_x.floor_ko = word_q.yld_ko; end end + // Candidate order: ascending (t, key), key = {instance id, record offset}. + // A candidate is staged only above the floor and ahead of the staged one. + wire [31:0] cand_ki = word_q.in_blas ? word_q.inst_id : 32'd0; + function automatic logic above_floor(input logic [31:0] t); + above_floor = !word_q.has_floor + || (t > word_q.floor_t) + || ((t == word_q.floor_t) && key_gt_floor_q); + endfunction + function automatic logic before_yld(input logic [31:0] t); + before_yld = !yld_q[sel_q] + || (t < word_q.yld_t) + || ((t == word_q.yld_t) && key_lt_yld_q); + endfunction + // EXEC outcome (combinational) ctx_state_t word_n; reg wake_self; @@ -1394,7 +1476,8 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( // woken by the collector: the raw AABB result landed if (coll_hit_q && (coll_t0_q < word_x.best_t) - && (!yld_q[sel_q] || (coll_t0_q < word_x.yld_t))) begin + && above_floor(coll_t0_q) + && before_yld(coll_t0_q)) begin cf_din_r.kind = CK_YLDP; cf_din_r.t = coll_t0_q; cf_din_r.prim = word_x.prim_base; @@ -1405,6 +1488,8 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( exec_yld_set = 1'b1; exec_cbtype = RTU_CB_TYPE_BITS'(`VX_RT_CB_TYPE_PROC); word_n.yld_t = coll_t0_q; + word_n.yld_ki = cand_ki; + word_n.yld_ko = word_x.cur_off; coll_free_r = 1'b1; word_n.cstate = CS_POP; end @@ -1491,7 +1576,8 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( end end end else if (tri_committable - && (!yld_q[sel_q] || (trit_q < word_x.yld_t))) begin + && above_floor(trit_q) + && before_yld(trit_q)) begin cf_din_r.kind = CK_YLDA; cf_din_r.t = trit_q; cf_din_r.u = triu_q; @@ -1507,7 +1593,9 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( cf_push_r = 1'b1; exec_yld_set = 1'b1; exec_cbtype = cls_cbtype; - word_n.yld_t = trit_q; + word_n.yld_t = trit_q; + word_n.yld_ki = cand_ki; + word_n.yld_ko = word_x.cur_off; if ((word_x.tri_i + 32'd1) < word_x.tri_n) begin word_n.tri_i = word_x.tri_i + 32'd1; word_n.cur_off = word_x.cur_off + 32'(RTU_TRI_STRIDE); @@ -1647,12 +1735,10 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( wake_self = 1'b1; end CS_INST_NEXT: begin + // every instance is scanned: a candidate staged in one instance + // does not hide a nearer one in a later instance word_n.in_blas = 1'b0; - if ((FLAT_TLAS != 0) && yld_q[sel_q]) begin - // the flat instance loop stops on a staged candidate - word_n.cstate = CS_DONE; - exec_done = 1'b1; - end else if ((word_x.inst_idx + 32'd1) == word_x.inst_cnt) begin + if ((word_x.inst_idx + 32'd1) == word_x.inst_cnt) begin if (FLAT) begin word_n.cstate = CS_DONE; exec_done = 1'b1; @@ -1734,6 +1820,10 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( end end + // the contexts a resume re-walks (bw_job_done's RES commit) + wire [NUM_CTX-1:0] rewalk_wake = (bw_job_done && bw_is_res) + ? (NUM_CTX'(res_rewalk) << (32'(bw_slot) * NUM_LANES)) : NUM_CTX'(0); + wire [NUM_CTX-1:0] rdy_wake_mask = (ray_wr_valid ? NUM_CTX'(1) << ray_wr_ctx : NUM_CTX'(0)) | (mem_rsp_valid ? NUM_CTX'(1) << mem_rsp_tag : NUM_CTX'(0)) @@ -1741,7 +1831,8 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( | (xform_valid_out ? NUM_CTX'(1) << xform_tag_out : NUM_CTX'(0)) | ((recip_valid_out && recip_last_out) ? NUM_CTX'(1) << recip_tag_out : NUM_CTX'(0)) | (box_wake_r ? NUM_CTX'(1) << box_wake_ctx_r : NUM_CTX'(0)) - | (wake_self_r ? NUM_CTX'(1) << wake_self_ctx_r : NUM_CTX'(0)); + | (wake_self_r ? NUM_CTX'(1) << wake_self_ctx_r : NUM_CTX'(0)) + | rewalk_wake; assign rdy_next = (rdy_set & ~g1_onehot) | rdy_wake_mask; @@ -1750,6 +1841,9 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( if (reset) begin rdy_set <= '0; fresh_set <= '0; + rewalk_set <= '0; + live_q <= '0; + seed_v_q <= '0; done_q <= '0; mask_q <= '0; hit_q <= '0; @@ -1778,6 +1872,8 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( hit_q[32'(slot_start_slot)*NUM_LANES + k] <= 1'b0; yld_q[32'(slot_start_slot)*NUM_LANES + k] <= 1'b0; attr_q[32'(slot_start_slot)*NUM_LANES + k] <= 1'b0; + live_q[32'(slot_start_slot)*NUM_LANES + k] <= slot_start_mask[k]; + seed_v_q[32'(slot_start_slot)*NUM_LANES + k] <= 1'b0; end end @@ -1788,7 +1884,11 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( // EXEC outcomes if (x_valid) begin - fresh_set[sel_q] <= 1'b0; + fresh_set[sel_q] <= 1'b0; + rewalk_set[sel_q] <= 1'b0; + if (rewalk_q) begin + seed_v_q[sel_q] <= 1'b0; + end if (exec_done) begin done_q[sel_q] <= 1'b1; end @@ -1802,7 +1902,7 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( if (exec_yld_clr) begin yld_q[sel_q] <= 1'b0; end - if (fresh_q) begin + if (fresh_q || rewalk_q) begin sp_q_arr[sel_q] <= '0; end else if (sp_inc) begin sp_q_arr[sel_q] <= sp_q + RTU_STACK_BITS'(1); @@ -1834,10 +1934,20 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( if (bw_is_res) begin for (k = 0; k < NUM_LANES; k = k + 1) begin if (bw_copy_mask[k]) begin - hit_q[32'(bw_slot)*NUM_LANES + k] <= 1'b1; - attr_q[32'(bw_slot)*NUM_LANES + k] <= 1'b1; + hit_q[32'(bw_slot)*NUM_LANES + k] <= 1'b1; + attr_q[32'(bw_slot)*NUM_LANES + k] <= 1'b1; + seed_v_q[32'(bw_slot)*NUM_LANES + k] <= 1'b1; + end + yld_q[32'(bw_slot)*NUM_LANES + k] <= 1'b0; + live_q[32'(bw_slot)*NUM_LANES + k] <= res_rewalk[k]; + if (res_rewalk[k]) begin + done_q[32'(bw_slot)*NUM_LANES + k] <= 1'b0; + rewalk_set[32'(bw_slot)*NUM_LANES + k] <= 1'b1; end - yld_q[32'(bw_slot)*NUM_LANES + k] <= 1'b0; + end + // another round: its lanes finalise again when they finish + if (res_rewalk != '0) begin + finalised[bw_slot] <= 1'b0; end end else begin for (k = 0; k < NUM_LANES; k = k + 1) begin @@ -1865,6 +1975,17 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( end end + // a resume's committed t seeds the lane's re-walk + always_ff @(posedge clk) begin + if (bw_wr_gnt && (bw_state == BW_WR) && bw_is_res && (bw_field == 3'd0)) begin + for (integer j = 0; j < NUM_LANES; j = j + 1) begin + if (bw_copy_mask[j]) begin + seed_t_q[32'(bw_slot)*NUM_LANES + j] <= bw_wr_data[j*32 +: 32]; + end + end + end + end + // actions are captured at the resume pulse always_ff @(posedge clk) begin for (integer s2 = 0; s2 < NUM_SLOTS; s2 = s2 + 1) begin @@ -1960,7 +2081,21 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( end BW_CAP2: begin bw_data2 <= ws_rdata; - bw_state <= BW_WR; + bw_state <= BW_RD3; + end + BW_RD3: begin + if (bw_rd_gnt) begin + bw_state <= BW_CAP3; + end + end + BW_CAP3: begin + bw_copy_mask <= bw_keep_mask; + if (bw_keep_mask == '0) begin + bw_job_done <= 1'b1; // every accept was out of range + bw_state <= BW_IDLE; + end else begin + bw_state <= BW_WR; + end end BW_WR: begin if (bw_wr_gnt) begin @@ -2016,8 +2151,10 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( assign busy = running; assign done = done_r; for (genvar s = 0; s < NUM_SLOTS; ++s) begin : g_yield + // bw_job_done: the flags a resume clears (yld/done) update on this + // edge, so the old values must not read as a fresh yield assign yield[s] = running[s] && all_done[s] && finalised[s] && yld_any[s] - && ce_idle && bw_idle && !pend_resume[s]; + && ce_idle && bw_idle && !pend_resume[s] && !bw_job_done; end assign hit_bits = hit_q; assign yld_bits = yld_q; diff --git a/tests/raytracing/Makefile b/tests/raytracing/Makefile index 5745ed6c45..dac035b0bd 100644 --- a/tests/raytracing/Makefile +++ b/tests/raytracing/Makefile @@ -11,7 +11,7 @@ TESTS := \ rt_smoke_miss rt_smoke_is rt_smoke_sbt rt_smoke_tlas \ rt_smoke_ahs_mixed rt_smoke_recursive \ rt_smoke_bvh_basic rt_smoke_bvh_multilevel \ - rt_smoke_bvh_instanced rt_smoke_ahs_bvh \ + rt_smoke_bvh_instanced rt_smoke_ahs_bvh rt_smoke_ahs_multi \ rt_smoke_shadow rt_smoke_cull_back \ rt_smoke_async_batch rt_smoke_cull_mask \ rt_smoke_proc rt_smoke_bvh6 rt_bvh_multinode rt_smoke_numctx \ diff --git a/tests/raytracing/rt_smoke_ahs_multi/Makefile b/tests/raytracing/rt_smoke_ahs_multi/Makefile new file mode 100644 index 0000000000..8a0a1f8777 --- /dev/null +++ b/tests/raytracing/rt_smoke_ahs_multi/Makefile @@ -0,0 +1,20 @@ +ROOT_DIR := $(realpath ../../..) +include $(ROOT_DIR)/config.mk + +CONFIGS := $(if $(findstring -DVX_CFG_EXT_RTU_ENABLE,$(CONFIGS)),$(CONFIGS),$(CONFIGS) -DVX_CFG_EXT_RTU_ENABLE) +# CW-BVH4 scene -> build the RTU as a CW-BVH4 walker. +CONFIGS += -DVX_CFG_RTU_BVH_WIDTH=4 + +PROJECT := rt_smoke_ahs_multi + +SRC_DIR := $(VORTEX_HOME)/tests/raytracing/$(PROJECT) + +SRCS := $(SRC_DIR)/main.cpp + +VX_SRCS := $(SRC_DIR)/kernel.cpp + +OPTS ?= + +KERNEL_LIB := vortex2 + +include ../common.mk diff --git a/tests/raytracing/rt_smoke_ahs_multi/common.h b/tests/raytracing/rt_smoke_ahs_multi/common.h new file mode 100644 index 0000000000..cd2ea37c70 --- /dev/null +++ b/tests/raytracing/rt_smoke_ahs_multi/common.h @@ -0,0 +1,62 @@ +// Copyright © 2019-2023 +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. +// +// RTU multi-candidate any-hit smoke. +// +// Several non-opaque triangles lie along one ray -- two of them coplanar at the +// same t -- with an opaque triangle behind them. Every candidate the walk meets +// must be offered, one per callback round, in ascending (t, record) order; a +// verdict that does not end the ray resumes the walk above the decided +// candidate. Each lane accepts a different primitive and records the order it +// was offered candidates in. + +#ifndef _RTU_SMOKE_AHS_MULTI_COMMON_H_ +#define _RTU_SMOKE_AHS_MULTI_COMMON_H_ + +#include + +// CW-BVH4 scene layout (matches rt_smoke_bvh_basic). +#define VX_BVH_SCENE_KIND 2 +#define VX_BVH_SCENE_HDR_BYTES 16 +#define VX_BVH_LEAF_HDR_BYTES 16 +#define VX_BVH_TRI_STRIDE 40 +#define VX_BVH_TRI_FLAGS_OFFSET 36 +#define VX_BVH_KIND_LEAF_TRI 1 +#define VX_BVH_COUNT_SHIFT 8 +#define VX_BVH_TRI_FLAG_OPAQUE 0x1u + +#define RTU_MULTI_NUM_TRIS 6 +#define RTU_MULTI_MAX_OFFER 8 +#define RTU_MULTI_NONE 0xffu // accept nothing + +typedef struct { + uint32_t status; + float hit_t; + uint32_t primitive_id; + uint32_t num_offered; + uint8_t offered[RTU_MULTI_MAX_OFFER]; // primitive ids, in callback order +} rtu_result_t; + +typedef struct { + uint64_t scene_addr; + uint64_t results_addr; + uint64_t targets_addr; // uint8_t per lane: the primitive it accepts + uint32_t num_lanes; + uint32_t pad; + float ray_origin[3]; + float ray_direction[3]; + float tmin; + float tmax; +} kernel_arg_t; + +#endif // _RTU_SMOKE_AHS_MULTI_COMMON_H_ diff --git a/tests/raytracing/rt_smoke_ahs_multi/kernel.cpp b/tests/raytracing/rt_smoke_ahs_multi/kernel.cpp new file mode 100644 index 0000000000..fdbe122e8d --- /dev/null +++ b/tests/raytracing/rt_smoke_ahs_multi/kernel.cpp @@ -0,0 +1,58 @@ +// Copyright © 2019-2023 +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. +// +// RTU multi-candidate any-hit smoke kernel: every lane of the warp traces the +// same ray and ACCEPTs only its target primitive, IGNOREing every other +// candidate, so the lanes' walks diverge -- some end early, others keep being +// offered candidates (and read PENDING while they wait on the others). + +#include +#include +#include "common.h" + +__kernel void kernel_main(kernel_arg_t* arg) { + uint32_t tid = threadIdx.x; + if (tid >= arg->num_lanes) return; + uint32_t target = ((const uint8_t*)(uintptr_t)arg->targets_addr)[tid]; + + vx_ray_t ray = { + {arg->ray_origin[0], arg->ray_origin[1], arg->ray_origin[2]}, + {arg->ray_direction[0], arg->ray_direction[1], arg->ray_direction[2]}, + arg->tmin, + arg->tmax, + }; + + rtu_result_t* res = (rtu_result_t*)((uintptr_t)arg->results_addr) + tid; + uint32_t n = 0; + + uint32_t scene_lo = (uint32_t)(arg->scene_addr & 0xffffffffu); + uint32_t h = vx_rt_wtrace(scene_lo, 0u, 0u, 0xffu, &ray); + vx_hit_t hit; + uint32_t sts = vx_rt_wait(h, &hit); + while (vx_rt_sts_is_yield(sts)) { + uint32_t action = VX_RT_CB_IGNORE; + if (vx_rt_sts_has_candidate(sts)) { + if (n < RTU_MULTI_MAX_OFFER) + res->offered[n] = (uint8_t)hit.primitive_id; + ++n; + if (hit.primitive_id == target) + action = VX_RT_CB_ACCEPT; + } + sts = vx_rt_continue(h, action, hit.t, 0u, &hit); + } + + res->status = sts; + res->hit_t = hit.t; + res->primitive_id = hit.primitive_id; + res->num_offered = n; +} diff --git a/tests/raytracing/rt_smoke_ahs_multi/main.cpp b/tests/raytracing/rt_smoke_ahs_multi/main.cpp new file mode 100644 index 0000000000..ab7176e2c4 --- /dev/null +++ b/tests/raytracing/rt_smoke_ahs_multi/main.cpp @@ -0,0 +1,204 @@ +// Copyright © 2019-2023 +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. +// +// RTU multi-candidate any-hit smoke -- host driver. +// +// One CW-BVH4 leaf, six triangles stacked along the ray (+z from z=0): +// prim 0 t=7 non-opaque +// prim 1 t=3 non-opaque +// prim 2 t=5 non-opaque +// prim 3 t=5 non-opaque (coplanar with prim 2: a t tie) +// prim 4 t=8 OPAQUE +// prim 5 t=9 non-opaque (behind the opaque one: never offered) +// A lane that accepts nothing is offered 1, 2, 3, 0 -- ascending t, the tie +// broken by record order -- and ends on the opaque prim 4. A lane that accepts +// prim P is offered the prefix of that sequence up to P and ends on P. + +#include +#include +#include +#include +#include +#include + +#include +#include +#include "common.h" + +#define RT_CHECK(_expr) \ + do { \ + int _ret = _expr; \ + if (0 == _ret) break; \ + printf("Error: '%s' returned %d!\n", #_expr, (int)_ret); \ + cleanup(); \ + exit(-1); \ + } while (false) + +const char* kernel_file = "kernel.vxbin"; +uint32_t num_lanes = 12; + +vx_device_h device = nullptr; +vx_buffer_h scene_buffer = nullptr; +vx_buffer_h res_buffer = nullptr; +vx_buffer_h tgt_buffer = nullptr; +vx_queue_h queue = nullptr; +vx_module_h module_ = nullptr; +vx_kernel_h kernel = nullptr; +kernel_arg_t kernel_arg = {}; + +static void show_usage() { + std::cout << "RTU multi-candidate any-hit smoke test." << std::endl; + std::cout << "Usage: [-k kernel] [-n lanes] [-h]" << std::endl; +} + +static void parse_args(int argc, char** argv) { + int c; + while ((c = getopt(argc, argv, "n:k:h")) != -1) { + switch (c) { + case 'n': num_lanes = atoi(optarg); break; + case 'k': kernel_file = optarg; break; + case 'h': show_usage(); exit(0); + default: show_usage(); exit(-1); + } + } +} + +void cleanup() { + if (device) { + if (scene_buffer) vx_buffer_release(scene_buffer); + if (res_buffer) vx_buffer_release(res_buffer); + if (tgt_buffer) vx_buffer_release(tgt_buffer); + if (kernel) vx_kernel_release(kernel); + if (module_) vx_module_release(module_); + if (queue) vx_queue_release(queue); + vx_device_release(device); + } +} + +int main(int argc, char* argv[]) { + parse_args(argc, argv); + + RT_CHECK(vx_device_open(0, &device)); + vx_queue_info_t qi = { sizeof(qi), nullptr, VX_QUEUE_PRIORITY_NORMAL, 0 }; + RT_CHECK(vx_queue_create(device, &qi, &queue)); + + static const float tri_t[RTU_MULTI_NUM_TRIS] = {7.f, 3.f, 5.f, 5.f, 8.f, 9.f}; + static const uint32_t tri_fl[RTU_MULTI_NUM_TRIS] = {0, 0, 0, 0, VX_BVH_TRI_FLAG_OPAQUE, 0}; + + std::vector scene_bytes(VX_BVH_SCENE_HDR_BYTES + VX_BVH_LEAF_HDR_BYTES + + RTU_MULTI_NUM_TRIS * VX_BVH_TRI_STRIDE, 0); + uint32_t* sh = reinterpret_cast(scene_bytes.data()); + sh[0] = VX_BVH_SCENE_HDR_BYTES; // root_node_offset + sh[1] = VX_BVH_SCENE_KIND; + sh[2] = (uint32_t)scene_bytes.size(); + sh[3] = 1; // leaf_count + uint32_t* lh = reinterpret_cast(scene_bytes.data() + VX_BVH_SCENE_HDR_BYTES); + lh[0] = VX_BVH_KIND_LEAF_TRI | (RTU_MULTI_NUM_TRIS << VX_BVH_COUNT_SHIFT); + for (uint32_t i = 0; i < RTU_MULTI_NUM_TRIS; ++i) { + uint8_t* rec = scene_bytes.data() + VX_BVH_SCENE_HDR_BYTES + VX_BVH_LEAF_HDR_BYTES + + i * VX_BVH_TRI_STRIDE; + float v[9] = {0.f, 0.f, tri_t[i], 1.f, 0.f, tri_t[i], 0.f, 1.f, tri_t[i]}; + memcpy(rec, v, sizeof(v)); + memcpy(rec + VX_BVH_TRI_FLAGS_OFFSET, &tri_fl[i], sizeof(uint32_t)); + } + + // lane targets cycle through: none, then each primitive in turn + static const uint8_t target_cycle[] = {RTU_MULTI_NONE, 1, 2, 3, 0, 5}; + const uint32_t ncycle = sizeof(target_cycle) / sizeof(target_cycle[0]); + std::vector targets(num_lanes); + for (uint32_t i = 0; i < num_lanes; ++i) + targets[i] = target_cycle[i % ncycle]; + + uint32_t scene_sz = (uint32_t)scene_bytes.size(); + uint32_t res_size = num_lanes * sizeof(rtu_result_t); + RT_CHECK(vx_buffer_create(device, scene_sz, VX_MEM_READ, &scene_buffer)); + RT_CHECK(vx_buffer_address(scene_buffer, &kernel_arg.scene_addr)); + RT_CHECK(vx_buffer_create(device, res_size, VX_MEM_READ_WRITE, &res_buffer)); + RT_CHECK(vx_buffer_address(res_buffer, &kernel_arg.results_addr)); + RT_CHECK(vx_buffer_create(device, num_lanes, VX_MEM_READ, &tgt_buffer)); + RT_CHECK(vx_buffer_address(tgt_buffer, &kernel_arg.targets_addr)); + + kernel_arg.num_lanes = num_lanes; + kernel_arg.ray_origin[0] = 0.25f; + kernel_arg.ray_origin[1] = 0.25f; + kernel_arg.ray_origin[2] = 0.0f; + kernel_arg.ray_direction[2] = 1.0f; + kernel_arg.tmin = 0.001f; + kernel_arg.tmax = 1e30f; + + std::vector zero(num_lanes); + memset(zero.data(), 0, res_size); + RT_CHECK(vx_enqueue_write(queue, scene_buffer, 0, scene_bytes.data(), scene_sz, 0, nullptr, nullptr)); + RT_CHECK(vx_enqueue_write(queue, tgt_buffer, 0, targets.data(), num_lanes, 0, nullptr, nullptr)); + RT_CHECK(vx_enqueue_write(queue, res_buffer, 0, zero.data(), res_size, 0, nullptr, nullptr)); + RT_CHECK(vx_module_load_file(device, kernel_file, &module_)); + RT_CHECK(vx_module_get_kernel(module_, "main", &kernel)); + + std::cout << "bvh4: 1 leaf, " << RTU_MULTI_NUM_TRIS << " tris, lanes=" << num_lanes << std::endl; + + vx_event_h launch_ev = nullptr, read_ev = nullptr; + { + vx_launch_info_t li = {}; + li.struct_size = sizeof(li); + li.kernel = kernel; + li.args_host = &kernel_arg; + li.args_size = sizeof(kernel_arg); + li.ndim = 1; + li.grid_dim[0] = 1; + li.block_dim[0] = num_lanes; + RT_CHECK(vx_enqueue_launch(queue, &li, 0, nullptr, &launch_ev)); + } + std::vector results(num_lanes); + RT_CHECK(vx_enqueue_read(queue, results.data(), res_buffer, 0, res_size, 1, &launch_ev, &read_ev)); + RT_CHECK(vx_event_wait_value(read_ev, 1, VX_TIMEOUT_INFINITE)); + vx_event_release(read_ev); + vx_event_release(launch_ev); + + // oracle: the offer order when every candidate is ignored + static const uint8_t order[] = {1, 2, 3, 0}; + const uint32_t norder = sizeof(order); + + int errors = 0; + for (uint32_t i = 0; i < num_lanes; ++i) { + uint32_t tgt = targets[i]; + uint32_t exp_n = norder, exp_prim = 4; + for (uint32_t k = 0; k < norder; ++k) { + if (order[k] == tgt) { exp_n = k + 1; exp_prim = tgt; break; } + } + float exp_t = tri_t[exp_prim]; + const rtu_result_t& r = results[i]; + bool ok = (r.status == VX_RT_STS_DONE_HIT) + && (r.primitive_id == exp_prim) + && (std::fabs(r.hit_t - exp_t) < 1e-4f) + && (r.num_offered == exp_n); + for (uint32_t k = 0; ok && k < exp_n; ++k) + ok = (r.offered[k] == order[k]); + if (!ok) { + std::cout << "lane " << i << " (accepts " << tgt << "): status=" << r.status + << " prim=" << r.primitive_id << " t=" << r.hit_t << " offered=["; + for (uint32_t k = 0; k < r.num_offered && k < RTU_MULTI_MAX_OFFER; ++k) + std::cout << (k ? "," : "") << uint32_t(r.offered[k]); + std::cout << "] expected prim=" << exp_prim << " t=" << exp_t + << " offered " << exp_n << std::endl; + ++errors; + } + } + + cleanup(); + if (errors != 0) { + std::cout << "FAILED with " << errors << " errors" << std::endl; + return 1; + } + std::cout << "PASSED!" << std::endl; + return 0; +} From dd1188c523e912a1d729604a6bba43197c0cad31 Mon Sep 17 00:00:00 2001 From: Blaise Tine Date: Mon, 21 Sep 2026 22:04:47 -0700 Subject: [PATCH 04/31] simx: WGATHER writes every lane of the warp The gather's lane loop started at the warp's first active lane, so a warp whose low lanes were masked never wrote those lanes' gathered values. A consumer reading fixed lanes then saw stale data: the RTU takes a trace's scene, payload and flags|cull from lanes 1-3 of the config register, so a trace issued with lane 0 inactive ran on whatever an earlier trace had left there (in LumiBench, shadow rays traced against a stale scene with cull mask 0 and missed every occluder). Visit every lane, falling back to the warp's last active lane for a masked source, as the RTL does. New test rt_smoke_partial_mask: an all-lane trace of an empty scene, then a trace from the last lane alone; fails without the fix, passes on simx and rtlsim. Co-Authored-By: Claude Opus 5 (1M context) --- sim/simx/alu_unit.cpp | 12 +- tests/raytracing/Makefile | 1 + .../raytracing/rt_smoke_partial_mask/Makefile | 20 +++ .../raytracing/rt_smoke_partial_mask/common.h | 51 ++++++ .../rt_smoke_partial_mask/kernel.cpp | 44 +++++ .../raytracing/rt_smoke_partial_mask/main.cpp | 150 ++++++++++++++++++ 6 files changed, 273 insertions(+), 5 deletions(-) create mode 100644 tests/raytracing/rt_smoke_partial_mask/Makefile create mode 100644 tests/raytracing/rt_smoke_partial_mask/common.h create mode 100644 tests/raytracing/rt_smoke_partial_mask/kernel.cpp create mode 100644 tests/raytracing/rt_smoke_partial_mask/main.cpp diff --git a/sim/simx/alu_unit.cpp b/sim/simx/alu_unit.cpp index a866e2f1ca..a48fc3ebe5 100644 --- a/sim/simx/alu_unit.cpp +++ b/sim/simx/alu_unit.cpp @@ -345,14 +345,16 @@ void AluUnit::execute(instr_trace_t* trace) { // (which `tmask` aliases) to suppress source lanes, so source-lane liveness // must be judged against the pre-suppression mask, not the live one. auto active = tmask; - uint32_t last_tid = thread_start; - for (uint32_t t = thread_start; t < num_threads; ++t) - if (active.test(t)) last_tid = t; + uint32_t last_tid = (thread_last >= 0) ? uint32_t(thread_last) : 0; // WGATHER writes the FULL nibble (every non-source lane) regardless of // the active mask, so the gathered value is materialised even in masked // lanes; source lanes stay suppressed (keep their self value). Reads fall - // back to the last active lane when the nominal source is masked. - for (uint32_t t = thread_start; t < num_threads; ++t) { + // back to the last active lane when the nominal source is masked. Every + // lane of the warp is visited -- not just those from the first active + // one on: a warp whose low lanes are masked still gets its nibbles, which + // is what a consumer reading fixed lanes (the RTU's config in lanes 1-3) + // relies on. + for (uint32_t t = 0; t < num_threads; ++t) { if ((t & 0x3u) == src_offset) { trace->tmask.reset(t); // suppress writeback for source lane continue; diff --git a/tests/raytracing/Makefile b/tests/raytracing/Makefile index dac035b0bd..28ac30176a 100644 --- a/tests/raytracing/Makefile +++ b/tests/raytracing/Makefile @@ -12,6 +12,7 @@ TESTS := \ rt_smoke_ahs_mixed rt_smoke_recursive \ rt_smoke_bvh_basic rt_smoke_bvh_multilevel \ rt_smoke_bvh_instanced rt_smoke_ahs_bvh rt_smoke_ahs_multi \ + rt_smoke_partial_mask \ rt_smoke_shadow rt_smoke_cull_back \ rt_smoke_async_batch rt_smoke_cull_mask \ rt_smoke_proc rt_smoke_bvh6 rt_bvh_multinode rt_smoke_numctx \ diff --git a/tests/raytracing/rt_smoke_partial_mask/Makefile b/tests/raytracing/rt_smoke_partial_mask/Makefile new file mode 100644 index 0000000000..0432b3400c --- /dev/null +++ b/tests/raytracing/rt_smoke_partial_mask/Makefile @@ -0,0 +1,20 @@ +ROOT_DIR := $(realpath ../../..) +include $(ROOT_DIR)/config.mk + +CONFIGS := $(if $(findstring -DVX_CFG_EXT_RTU_ENABLE,$(CONFIGS)),$(CONFIGS),$(CONFIGS) -DVX_CFG_EXT_RTU_ENABLE) +# CW-BVH4 scenes -> build the RTU as a CW-BVH4 walker. +CONFIGS += -DVX_CFG_RTU_BVH_WIDTH=4 + +PROJECT := rt_smoke_partial_mask + +SRC_DIR := $(VORTEX_HOME)/tests/raytracing/$(PROJECT) + +SRCS := $(SRC_DIR)/main.cpp + +VX_SRCS := $(SRC_DIR)/kernel.cpp + +OPTS ?= + +KERNEL_LIB := vortex2 + +include ../common.mk diff --git a/tests/raytracing/rt_smoke_partial_mask/common.h b/tests/raytracing/rt_smoke_partial_mask/common.h new file mode 100644 index 0000000000..982d6cb0f9 --- /dev/null +++ b/tests/raytracing/rt_smoke_partial_mask/common.h @@ -0,0 +1,51 @@ +// Copyright © 2019-2023 +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. +// +// RTU partial-warp trace smoke. +// +// A trace's warp-uniform config (scene, payload, flags|cull) is gathered into +// lanes 1-3 of its config register. A trace issued by a warp whose low lanes +// are masked off must still deliver its own config: the kernel first traces an +// EMPTY scene from every lane (leaving that scene in the register's lanes), +// then traces the triangle scene from the warp's last lane only. + +#ifndef _RTU_SMOKE_PARTIAL_MASK_COMMON_H_ +#define _RTU_SMOKE_PARTIAL_MASK_COMMON_H_ + +#include + +#define VX_BVH_SCENE_KIND 2 +#define VX_BVH_SCENE_HDR_BYTES 16 +#define VX_BVH_LEAF_HDR_BYTES 16 +#define VX_BVH_TRI_STRIDE 40 +#define VX_BVH_TRI_FLAGS_OFFSET 36 +#define VX_BVH_KIND_LEAF_TRI 1 +#define VX_BVH_COUNT_SHIFT 8 +#define VX_BVH_TRI_FLAG_OPAQUE 0x1u + +typedef struct { + uint32_t first_status; // the all-lane trace of the empty scene + uint32_t second_status; // the last-lane trace of the triangle scene + float second_t; + uint32_t pad; +} rtu_result_t; + +typedef struct { + uint64_t empty_scene_addr; + uint64_t tri_scene_addr; + uint64_t results_addr; + uint32_t num_lanes; + uint32_t pad; +} kernel_arg_t; + +#endif // _RTU_SMOKE_PARTIAL_MASK_COMMON_H_ diff --git a/tests/raytracing/rt_smoke_partial_mask/kernel.cpp b/tests/raytracing/rt_smoke_partial_mask/kernel.cpp new file mode 100644 index 0000000000..157f3533af --- /dev/null +++ b/tests/raytracing/rt_smoke_partial_mask/kernel.cpp @@ -0,0 +1,44 @@ +// Copyright © 2019-2023 +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. +// +// RTU partial-warp trace smoke kernel: an all-lane trace of the empty scene, +// then a trace of the triangle scene from the warp's last lane alone. + +#include +#include +#include "common.h" + +static uint32_t trace_one(uint32_t scene, float* t_out) { + vx_ray_t ray = {{0.25f, 0.25f, 0.f}, {0.f, 0.f, 1.f}, 0.001f, 1e30f}; + uint32_t h = vx_rt_wtrace(scene, 0u, 0u, 0xffu, &ray); + vx_hit_t hit; + uint32_t sts = vx_rt_wait(h, &hit); + while (vx_rt_sts_is_yield(sts)) + sts = vx_rt_continue(h, VX_RT_CB_ACCEPT, hit.t, 0u, &hit); + *t_out = hit.t; + return sts; +} + +__kernel void kernel_main(kernel_arg_t* arg) { + uint32_t tid = threadIdx.x; + if (tid >= arg->num_lanes) return; + rtu_result_t* res = (rtu_result_t*)((uintptr_t)arg->results_addr) + tid; + + float t; + res->first_status = trace_one((uint32_t)arg->empty_scene_addr, &t); + res->second_status = 0xffffffffu; + if (tid == arg->num_lanes - 1) { + res->second_status = trace_one((uint32_t)arg->tri_scene_addr, &t); + res->second_t = t; + } +} diff --git a/tests/raytracing/rt_smoke_partial_mask/main.cpp b/tests/raytracing/rt_smoke_partial_mask/main.cpp new file mode 100644 index 0000000000..e429da77dd --- /dev/null +++ b/tests/raytracing/rt_smoke_partial_mask/main.cpp @@ -0,0 +1,150 @@ +// Copyright © 2019-2023 +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. +// +// RTU partial-warp trace smoke -- host driver. One warp: every lane traces an +// empty CW-BVH4 scene (a miss), then the last lane alone traces a scene holding +// one opaque triangle at t=5 (a hit). + +#include +#include +#include +#include +#include + +#include +#include +#include "common.h" + +#define RT_CHECK(_expr) \ + do { \ + int _ret = _expr; \ + if (0 == _ret) break; \ + printf("Error: '%s' returned %d!\n", #_expr, (int)_ret); \ + cleanup(); \ + exit(-1); \ + } while (false) + +const char* kernel_file = "kernel.vxbin"; + +vx_device_h device = nullptr; +vx_buffer_h scene_buffer = nullptr; +vx_buffer_h res_buffer = nullptr; +vx_queue_h queue = nullptr; +vx_module_h module_ = nullptr; +vx_kernel_h kernel = nullptr; +kernel_arg_t kernel_arg = {}; + +void cleanup() { + if (device) { + if (scene_buffer) vx_buffer_release(scene_buffer); + if (res_buffer) vx_buffer_release(res_buffer); + if (kernel) vx_kernel_release(kernel); + if (module_) vx_module_release(module_); + if (queue) vx_queue_release(queue); + vx_device_release(device); + } +} + +int main(int argc, char* argv[]) { + int c; + while ((c = getopt(argc, argv, "k:h")) != -1) { + if (c == 'k') kernel_file = optarg; + else { std::cout << "Usage: [-k kernel] [-h]" << std::endl; return c == 'h' ? 0 : -1; } + } + + RT_CHECK(vx_device_open(0, &device)); + vx_queue_info_t qi = { sizeof(qi), nullptr, VX_QUEUE_PRIORITY_NORMAL, 0 }; + RT_CHECK(vx_queue_create(device, &qi, &queue)); + + uint64_t num_threads = 0; + RT_CHECK(vx_device_query(device, VX_CAPS_NUM_THREADS, &num_threads)); + const uint32_t num_lanes = (uint32_t)num_threads; // exactly one warp + + // Two scenes in one buffer, 64-B aligned: the empty one (a leaf of zero + // triangles) at 0, the one-triangle one at 128. + const uint32_t kTriOff = 128; + std::vector bytes(kTriOff + VX_BVH_SCENE_HDR_BYTES + VX_BVH_LEAF_HDR_BYTES + + VX_BVH_TRI_STRIDE, 0); + for (uint32_t s = 0; s < 2; ++s) { + uint8_t* base = bytes.data() + (s ? kTriOff : 0); + uint32_t ntri = s; // empty, then one triangle + uint32_t sh[4] = {VX_BVH_SCENE_HDR_BYTES, VX_BVH_SCENE_KIND, + VX_BVH_SCENE_HDR_BYTES + VX_BVH_LEAF_HDR_BYTES + ntri * VX_BVH_TRI_STRIDE, 1}; + memcpy(base, sh, sizeof(sh)); + uint32_t lh[4] = {VX_BVH_KIND_LEAF_TRI | (ntri << VX_BVH_COUNT_SHIFT), 0, 0, 0}; + memcpy(base + VX_BVH_SCENE_HDR_BYTES, lh, sizeof(lh)); + if (ntri) { + uint8_t* rec = base + VX_BVH_SCENE_HDR_BYTES + VX_BVH_LEAF_HDR_BYTES; + float v[9] = {0.f, 0.f, 5.f, 1.f, 0.f, 5.f, 0.f, 1.f, 5.f}; + uint32_t fl = VX_BVH_TRI_FLAG_OPAQUE; + memcpy(rec, v, sizeof(v)); + memcpy(rec + VX_BVH_TRI_FLAGS_OFFSET, &fl, sizeof(fl)); + } + } + + uint32_t res_size = num_lanes * sizeof(rtu_result_t); + uint64_t scene_addr = 0; + RT_CHECK(vx_buffer_create(device, bytes.size(), VX_MEM_READ, &scene_buffer)); + RT_CHECK(vx_buffer_address(scene_buffer, &scene_addr)); + RT_CHECK(vx_buffer_create(device, res_size, VX_MEM_READ_WRITE, &res_buffer)); + RT_CHECK(vx_buffer_address(res_buffer, &kernel_arg.results_addr)); + kernel_arg.empty_scene_addr = scene_addr; + kernel_arg.tri_scene_addr = scene_addr + kTriOff; + kernel_arg.num_lanes = num_lanes; + + RT_CHECK(vx_enqueue_write(queue, scene_buffer, 0, bytes.data(), bytes.size(), 0, nullptr, nullptr)); + RT_CHECK(vx_module_load_file(device, kernel_file, &module_)); + RT_CHECK(vx_module_get_kernel(module_, "main", &kernel)); + + std::cout << "one warp of " << num_lanes << " lanes; last lane traces alone" << std::endl; + + vx_event_h launch_ev = nullptr, read_ev = nullptr; + { + vx_launch_info_t li = {}; + li.struct_size = sizeof(li); + li.kernel = kernel; + li.args_host = &kernel_arg; + li.args_size = sizeof(kernel_arg); + li.ndim = 1; + li.grid_dim[0] = 1; + li.block_dim[0] = num_lanes; + RT_CHECK(vx_enqueue_launch(queue, &li, 0, nullptr, &launch_ev)); + } + std::vector results(num_lanes); + RT_CHECK(vx_enqueue_read(queue, results.data(), res_buffer, 0, res_size, 1, &launch_ev, &read_ev)); + RT_CHECK(vx_event_wait_value(read_ev, 1, VX_TIMEOUT_INFINITE)); + vx_event_release(read_ev); + vx_event_release(launch_ev); + + int errors = 0; + for (uint32_t i = 0; i < num_lanes; ++i) { + const rtu_result_t& r = results[i]; + bool last = (i == num_lanes - 1); + bool ok = (r.first_status == VX_RT_STS_DONE_MISS) + && (last ? (r.second_status == VX_RT_STS_DONE_HIT && std::fabs(r.second_t - 5.f) < 1e-4f) + : (r.second_status == 0xffffffffu)); + if (!ok) { + std::cout << "lane " << i << ": first=" << r.first_status << " second=" << r.second_status + << " t=" << r.second_t << std::endl; + ++errors; + } + } + + cleanup(); + if (errors != 0) { + std::cout << "FAILED with " << errors << " errors" << std::endl; + return 1; + } + std::cout << "PASSED!" << std::endl; + return 0; +} From 43ec716602173c56a6141312b7457f750600d733 Mon Sep 17 00:00:00 2001 From: Blaise Tine Date: Mon, 21 Sep 2026 22:04:47 -0700 Subject: [PATCH 05/31] raytrace.h: value-initialize the BVH builder's child array GCC 13 flags ch[] as maybe-uninitialized under -Werror, which broke the host build of seven tests/raytracing apps. Co-Authored-By: Claude Opus 5 (1M context) --- sw/runtime/include/raytrace.h | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/sw/runtime/include/raytrace.h b/sw/runtime/include/raytrace.h index 82860dfe3b..6febe1c687 100644 --- a/sw/runtime/include/raytrace.h +++ b/sw/runtime/include/raytrace.h @@ -310,7 +310,7 @@ class BvhBuilder { groups.push_back(std::move(b)); } - NodeRef ch[6]; + NodeRef ch[6] = {}; uint32_t n = 0; for (auto& g : groups) if (!g.empty()) ch[n++] = build_node(g); From 2391b0657110d07e5a12a4f77c4b8654dec2acc2 Mon Sep 17 00:00:00 2001 From: Blaise Tine Date: Mon, 21 Sep 2026 22:13:28 -0700 Subject: [PATCH 06/31] rtu: require a full-warp ALU for the WGATHER'd trace config A trace's scene/payload/flags ride lanes 1-3 of a WGATHER'd register. The ALU writes those lanes under a partial warp only when it runs the whole warp in one packet; a narrower ALU skips all-masked packets and picks its fallback source lane per packet, so the RTU would trace a stale config. Fail the build for such a config instead. Co-Authored-By: Claude Opus 5 (1M context) --- hw/rtl/rtu/VX_rtu_core.sv | 9 +++++++++ 1 file changed, 9 insertions(+) diff --git a/hw/rtl/rtu/VX_rtu_core.sv b/hw/rtl/rtu/VX_rtu_core.sv index 738f92085d..d2615404d5 100644 --- a/hw/rtl/rtu/VX_rtu_core.sv +++ b/hw/rtl/rtu/VX_rtu_core.sv @@ -126,6 +126,15 @@ module VX_rtu_core import VX_gpu_pkg::*, VX_rtu_pkg::*; #( `STATIC_ASSERT((`VX_CFG_RTU_MERGE_DEPTH == 0), ("VX_CFG_RTU_MERGE_DEPTH > 0 is not implemented: this core does not merge node fetches")) + // A trace's warp-uniform config (scene, payload, flags|cull) arrives in + // lanes 1-3 of a WGATHER'd register, which must be written even when the + // warp's low lanes are masked. The ALU guarantees that only when it runs + // the whole warp in one packet: a narrower ALU skips all-masked packets and + // resolves its fallback source lane per packet, so the RTU would read a + // stale config. Fail the build instead of tracing the wrong scene. + `STATIC_ASSERT((`VX_CFG_NUM_ALU_LANES == `VX_CFG_NUM_THREADS), + ("the RTU needs a full-warp ALU (VX_CFG_NUM_ALU_LANES == VX_CFG_NUM_THREADS) for its WGATHER'd config")) + // ── ray staging: one entry per {src, wid} ───────────────────────── localparam NUM_STG = NUM_SRCS * NUM_WARPS; localparam STG_IDX_W = `LOG2UP(NUM_STG); From 8ba2724481f8786125b19895a2376b0dce803b48 Mon Sep 17 00:00:00 2001 From: Blaise Tine Date: Mon, 21 Sep 2026 22:30:27 -0700 Subject: [PATCH 07/31] rtu: reject only degenerate triangles, not small ones The ray-triangle test dropped any triangle with |det| < 1e-6. det scales with the triangle's area, so small (or finely tessellated) geometry vanished: in LumiBench's Spring scene the character's eyes and mouth were missing and their shadows with them. Reject only |det| < FLT_MIN (edge-on or zero-area, where 1/det overflows) in both SimX and the RTL tri PE. Co-Authored-By: Claude Opus 5 (1M context) --- hw/rtl/rtu/VX_rtu_tri_pe.sv | 9 ++++++--- sim/simx/rtu/rtu_isect.cpp | 7 +++++-- 2 files changed, 11 insertions(+), 5 deletions(-) diff --git a/hw/rtl/rtu/VX_rtu_tri_pe.sv b/hw/rtl/rtu/VX_rtu_tri_pe.sv index 341e299437..d8f64f7156 100644 --- a/hw/rtl/rtu/VX_rtu_tri_pe.sv +++ b/hw/rtl/rtu/VX_rtu_tri_pe.sv @@ -20,7 +20,7 @@ // u = (T · P) * invDet // Q = T × e1 v = (dir · Q) * invDet // t = (e2 · Q) * invDet -// hit = |det| >= EPS && 0<=u<=1 && 0<=v && u+v<=1 && tmin<=t<=tmax +// hit = |det| >= FLT_MIN && 0<=u<=1 && 0<=v && u+v<=1 && tmin<=t<=tmax // back_facing = det < 0 // // The FP datapath reuses VX_fma_unit (a*b±c), VX_fdiv_unit (1/det) and @@ -72,8 +72,11 @@ module VX_rtu_tri_pe import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( localparam [31:0] FP_ZERO = 32'h00000000; localparam [31:0] FP_ONE = 32'h3F800000; - localparam [31:0] FP_EPS = 32'h358637BD; // 1e-6 - localparam [31:0] FP_NEG_EPS = 32'hB58637BD; // -1e-6 + // |det| scales with the triangle's area, so the degenerate test rejects + // only |det| < FLT_MIN (edge-on / zero-area, where 1/det overflows): any + // larger fixed epsilon drops small triangles. + localparam [31:0] FP_EPS = 32'h00800000; // FLT_MIN + localparam [31:0] FP_NEG_EPS = 32'h80800000; // -FLT_MIN // ── stage e (@F): edge vectors and ray-origin offset ────────────── wire [2:0][31:0] e1, e2, tvec; diff --git a/sim/simx/rtu/rtu_isect.cpp b/sim/simx/rtu/rtu_isect.cpp index d87ab078db..140d643a70 100644 --- a/sim/simx/rtu/rtu_isect.cpp +++ b/sim/simx/rtu/rtu_isect.cpp @@ -12,6 +12,7 @@ // limitations under the License. #include "rtu_isect.h" +#include #include namespace vortex { namespace rtu { @@ -31,8 +32,10 @@ bool ray_triangle(const float ro[3], const float rd[3], Vec3 e2 = V2 - V0; Vec3 P = cross(D, e2); float det = dot(e1, P); - constexpr float EPS = 1e-6f; - if (det > -EPS && det < EPS) return false; + // Reject only a degenerate (edge-on or zero-area) triangle: |det| scales + // with the triangle's area, so any fixed epsilon above the float range would + // drop small triangles. Below FLT_MIN the reciprocal overflows. + if (!(std::fabs(det) >= FLT_MIN)) return false; float invDet = 1.0f / det; Vec3 T = O - V0; float u = dot(T, P) * invDet; From 3118776065ac8e26d914330f4de8ea6834130e0f Mon Sep 17 00:00:00 2001 From: Blaise Tine Date: Tue, 22 Sep 2026 04:22:41 -0700 Subject: [PATCH 08/31] rtu: stage instance ids and the object ray for instanced candidates Two RTL-only defects in the candidate record the RTU hands a warp, both invisible on SimX: 1. A procedural (IS) candidate never staged instance_id or the instance custom index: the commit engine skipped those rows for CK_YLDP, so the intersection shader -- and, after an accept, the committed hit -- read whatever an earlier record left there. Every sphere of LumiBench WKND is a procedural instance whose shaders index per-sphere data by instance id, so WKND rendered only sky (RTLsim) or sky and ground (V80). Both candidate kinds now stage the full t/u/v/prim/inst/geom/custom record. 2. The object-ray slots (gl_ObjectRayOrigin/DirectionEXT) of any candidate were filled from the world-ray staging, so an any-hit or intersection shader inside a translated instance tested the wrong ray. A candidate from inside a BLAS now carries the walker's transformed ray through the commit queue into six window-store rows, flagged per context (obj_vld). The core picks those rows per lane. A candidate outside any BLAS keeps using the staging, where object space is world space, so it costs no extra cycles, and its commit-queue walk still ends at the SBT row. New rt_smoke_proc_inst: two lanes of one warp hit two instances (non-zero ids) of one procedural BLAS; it checks the ids the IS sees, the committed ids and t. It fails on the old RTL (garbage ids, IS rejects) and passes on SimX and the fixed RTL. RT suite: SimX 37/37, RTLsim 35/35. Co-Authored-By: Claude Opus 5 (1M context) --- hw/rtl/rtu/VX_rtu_core.sv | 14 +- hw/rtl/rtu/VX_rtu_pkg.sv | 1 + hw/rtl/rtu/VX_rtu_scheduler.sv | 73 ++++--- tests/raytracing/Makefile | 2 +- tests/raytracing/rt_smoke_proc_inst/Makefile | 20 ++ tests/raytracing/rt_smoke_proc_inst/common.h | 59 ++++++ .../raytracing/rt_smoke_proc_inst/kernel.cpp | 72 +++++++ tests/raytracing/rt_smoke_proc_inst/main.cpp | 194 ++++++++++++++++++ 8 files changed, 400 insertions(+), 35 deletions(-) create mode 100644 tests/raytracing/rt_smoke_proc_inst/Makefile create mode 100644 tests/raytracing/rt_smoke_proc_inst/common.h create mode 100644 tests/raytracing/rt_smoke_proc_inst/kernel.cpp create mode 100644 tests/raytracing/rt_smoke_proc_inst/main.cpp diff --git a/hw/rtl/rtu/VX_rtu_core.sv b/hw/rtl/rtu/VX_rtu_core.sv index d2615404d5..d2dedd1238 100644 --- a/hw/rtl/rtu/VX_rtu_core.sv +++ b/hw/rtl/rtu/VX_rtu_core.sv @@ -174,7 +174,7 @@ module VX_rtu_core import VX_gpu_pkg::*, VX_rtu_pkg::*; #( wire [RTU_RAY_BEATS*32-1:0] rw_data; wire [NUM_SLOTS-1:0] sch_busy, sch_done, sch_yield; - wire [NUM_CTX-1:0] sch_hit, sch_yld, sch_attrv; + wire [NUM_CTX-1:0] sch_hit, sch_yld, sch_attrv, sch_objv; wire [NUM_CTX-1:0][RTU_CB_TYPE_BITS-1:0] sch_cbtype; wire [NUM_SLOTS-1:0] sch_resume; wire [NUM_CTX-1:0][RTU_CB_ACTION_BITS-1:0] sch_action; @@ -218,6 +218,7 @@ module VX_rtu_core import VX_gpu_pkg::*, VX_rtu_pkg::*; #( .hit_bits (sch_hit), .yld_bits (sch_yld), .cb_types (sch_cbtype), + .obj_vld (sch_objv), .attr_vld (sch_attrv), .resume (sch_resume), .action (sch_action), @@ -511,10 +512,15 @@ module VX_rtu_core import VX_gpu_pkg::*, VX_rtu_pkg::*; #( wire wr_objray = is_cand[ws] && !wr_hitspan && (wr_idx < RTU_IDX_BITS'(13)); wire wr_cbtype = is_cand[ws] && (wr_idx == RTU_IDX_BITS'(13)); wire wr_sbt = is_cand[ws] && (wr_idx == RTU_IDX_BITS'(14)); + // Object ray: a candidate from inside a BLAS staged its transformed ray in + // the window store; any other lane's object ray is its world ray (staging). + wire [NUM_LANES-1:0] ws_objv = sch_objv[32'(ws)*NUM_LANES +: NUM_LANES] + & sch_yld[32'(ws)*NUM_LANES +: NUM_LANES]; + wire wr_objrow = wr_objray && (ws_objv != '0); // reads this word needs wire wb_need_w1 = (wb_state == WB_RD1) - && (wr_hitspan || wr_sbt || wr_attr) && !wr_status && !wr_payload; + && (wr_hitspan || wr_sbt || wr_attr || wr_objrow) && !wr_status && !wr_payload; wire wb_need_w2 = is_cand[ws] && wr_hitspan; // the hit row wire wb_need_st = (wr_hitspan && (wr_idx == '0)) || wr_objray; // t_max / object ray @@ -527,6 +533,8 @@ module VX_rtu_core import VX_gpu_pkg::*, VX_rtu_pkg::*; #( : RTU_WS_WORD_BITS'(RTU_WS_HIT_BASE)) + RTU_WS_WORD_BITS'(wr_idx)} : wr_sbt ? {ws, RTU_WS_WORD_BITS'(RTU_WS_YLD_SBT)} + : wr_objray ? {ws, RTU_WS_WORD_BITS'(RTU_WS_YLD_OBJ) + + RTU_WS_WORD_BITS'(wr_idx - RTU_IDX_BITS'(RTU_RES_HIT))} : {ws, RTU_WS_WORD_BITS'(RTU_WS_RES_ATTR)}; assign wb_stg_req = (wb_state == WB_RD1) && wb_need_st; @@ -576,7 +584,7 @@ module VX_rtu_core import VX_gpu_pkg::*, VX_rtu_pkg::*; #( win_word[i] = (wr_idx == '0) ? wb_ds[i] : 32'd0; end end else if (wr_objray) begin - win_word[i] = wb_ds[i]; + win_word[i] = ws_objv[i] ? wb_d1[i*32 +: 32] : wb_ds[i]; end else if (wr_cbtype) begin win_word[i] = {{(32-RTU_CB_TYPE_BITS){1'b0}}, sch_cbtype[wctx[i]]}; end else if (wr_sbt) begin diff --git a/hw/rtl/rtu/VX_rtu_pkg.sv b/hw/rtl/rtu/VX_rtu_pkg.sv index d21d89be69..e0c84d72c5 100644 --- a/hw/rtl/rtu/VX_rtu_pkg.sv +++ b/hw/rtl/rtu/VX_rtu_pkg.sv @@ -299,6 +299,7 @@ package VX_rtu_pkg; localparam RTU_WS_CONT_T = 16; // CONTINUE beat 0: the shader's own t localparam RTU_WS_CONT_ATTR = 17; // CONTINUE beat 1: the shader's hitAttribute localparam RTU_WS_RES_ATTR = 18; // accepted candidate's bound hitAttribute + localparam RTU_WS_YLD_OBJ = 19; // instanced candidate's object ray: o.xyz, d.xyz (19..24) localparam RTU_WS_WORDS = 32; // rows per slot (power of two for addressing) localparam RTU_WS_WORD_BITS = `CLOG2(RTU_WS_WORDS); diff --git a/hw/rtl/rtu/VX_rtu_scheduler.sv b/hw/rtl/rtu/VX_rtu_scheduler.sv index 958251f322..b8f3da9ce7 100644 --- a/hw/rtl/rtu/VX_rtu_scheduler.sv +++ b/hw/rtl/rtu/VX_rtu_scheduler.sv @@ -85,6 +85,7 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( output wire [NUM_CTX-1:0] hit_bits, output wire [NUM_CTX-1:0] yld_bits, output wire [NUM_CTX-1:0][RTU_CB_TYPE_BITS-1:0] cb_types, + output wire [NUM_CTX-1:0] obj_vld, // candidate staged an object ray output wire [NUM_CTX-1:0] attr_vld, // callback resume: the warp's per-lane actions, held stable by the core @@ -245,6 +246,7 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( reg [NUM_CTX-1:0] hit_q; reg [NUM_CTX-1:0] yld_q; reg [NUM_CTX-1:0][RTU_CB_TYPE_BITS-1:0] cbtype_q; + reg [NUM_CTX-1:0] objv_q; // candidate came from inside a BLAS reg [NUM_CTX-1:0] attr_q; reg [NUM_CTX-1:0][RTU_STACK_BITS-1:0] sp_q_arr; reg [NUM_CTX-1:0][LB-1:0] f_slot_q; @@ -848,13 +850,15 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( // one field-row per cycle, per-lane write enable. localparam [1:0] CK_HIT = 2'd0, // committed opaque hit: the 7 hit rows CK_YLDA = 2'd1, // any-hit candidate: 7 yld rows + sbt - CK_YLDP = 2'd2; // IS candidate: t/u/v/prim/geom + sbt + CK_YLDP = 2'd2; // IS candidate: t/u/v/prim/inst/geom/cust + sbt (+ obj ray) typedef struct packed { logic [1:0] kind; logic [CTX_TAG_W-1:0] ctx; logic [RTU_CB_SBT_BITS-1:0] sbt; logic [31:0] t, u, v, prim, inst, geom, cust; + logic objv; // candidate inside a BLAS: stage obj + logic [5:0][31:0] obj; // its object ray: o.xyz, d.xyz } commit_t; wire cf_push, cf_pop, cf_empty, cf_full; @@ -908,7 +912,7 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( assign win_rd_data = ws_rdata; // commit engine sequencing (one row per granted cycle) - reg [2:0] ce_step; + reg [3:0] ce_step; wire ce_active = ~cf_empty; wire [SLOT_W-1:0] ce_slot = SLOT_W'(32'(cf_dout.ctx) / NUM_LANES); wire [NUM_LANES-1:0] ce_lane = NUM_LANES'(1) << (32'(cf_dout.ctx) % NUM_LANES); @@ -917,8 +921,12 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( reg [31:0] ce_data; reg ce_last; always @(*) begin - // per-kind (row, field) walk; CK_YLDP skips inst/custom — an IS - // candidate leaves those rows holding whatever was last staged + // per-kind (row, field) walk. Both candidate kinds stage the full + // attribute record (an IS shader reads gl_InstanceID and + // gl_InstanceCustomIndexEXT too) plus the SBT row; a committed hit + // stops after custom. A candidate from inside a BLAS then stages its + // object-space ray (gl_ObjectRay*): outside one it IS the world ray, + // which the core already holds, so those candidates skip the rows. logic [RTU_WS_WORD_BITS-1:0] base; base = (cf_dout.kind == CK_HIT) ? RTU_WS_WORD_BITS'(RTU_WS_HIT_BASE) : RTU_WS_WORD_BITS'(RTU_WS_YLD_BASE); @@ -926,37 +934,25 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( ce_data = 32'd0; ce_last = 1'b0; case (ce_step) - 3'd0: ce_data = cf_dout.t; - 3'd1: ce_data = cf_dout.u; - 3'd2: ce_data = cf_dout.v; - 3'd3: ce_data = cf_dout.prim; - 3'd4: begin - if (cf_dout.kind == CK_YLDP) begin - ce_word = base + RTU_WS_WORD_BITS'(RTU_WS_F_GEOM); - ce_data = cf_dout.geom; - end else begin - ce_data = cf_dout.inst; - end - end - 3'd5: begin - if (cf_dout.kind == CK_YLDP) begin - ce_word = RTU_WS_WORD_BITS'(RTU_WS_YLD_SBT); - ce_data = 32'(cf_dout.sbt); - ce_last = 1'b1; - end else begin - ce_word = base + RTU_WS_WORD_BITS'(RTU_WS_F_GEOM); - ce_data = cf_dout.geom; - end - end - 3'd6: begin - ce_word = base + RTU_WS_WORD_BITS'(RTU_WS_F_CUST); + 4'd0: ce_data = cf_dout.t; + 4'd1: ce_data = cf_dout.u; + 4'd2: ce_data = cf_dout.v; + 4'd3: ce_data = cf_dout.prim; + 4'd4: ce_data = cf_dout.inst; + 4'd5: ce_data = cf_dout.geom; + 4'd6: begin ce_data = cf_dout.cust; ce_last = (cf_dout.kind == CK_HIT); end - default: begin + 4'd7: begin ce_word = RTU_WS_WORD_BITS'(RTU_WS_YLD_SBT); ce_data = 32'(cf_dout.sbt); - ce_last = 1'b1; + ce_last = ~cf_dout.objv; + end + default: begin // 8..13: object ray + ce_word = RTU_WS_WORD_BITS'(RTU_WS_YLD_OBJ) + RTU_WS_WORD_BITS'(32'(ce_step) - 8); + ce_data = cf_dout.obj[3'(32'(ce_step) - 8)]; + ce_last = (ce_step == 4'd13); end endcase end @@ -1175,7 +1171,7 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( if (reset) begin ce_step <= '0; end else if (ce_wr_gnt) begin - ce_step <= ce_last ? 3'd0 : (ce_step + 3'd1); + ce_step <= ce_last ? 4'd0 : (ce_step + 4'd1); end end @@ -1228,6 +1224,7 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( reg exec_yld_set; reg exec_yld_clr; reg [RTU_CB_TYPE_BITS-1:0] exec_cbtype; + reg exec_objv; reg mem_issue; reg [LB-1:0] mem_fslot; reg box_feed_r, box_raw_r, tri_feed_r, xform_feed_r; @@ -1252,6 +1249,7 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( exec_yld_set = 1'b0; exec_yld_clr = 1'b0; exec_cbtype = '0; + exec_objv = 1'b0; mem_issue = 1'b0; mem_fslot = '0; box_feed_r = 1'b0; @@ -1481,12 +1479,17 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( cf_din_r.kind = CK_YLDP; cf_din_r.t = coll_t0_q; cf_din_r.prim = word_x.prim_base; + cf_din_r.inst = word_x.in_blas ? word_x.inst_id : 32'd0; + cf_din_r.cust = word_x.in_blas ? word_x.inst_cust : 32'd0; cf_din_r.geom = word_x.geom_r; cf_din_r.sbt = word_x.proc_sbt; + cf_din_r.objv = word_x.in_blas; + cf_din_r.obj = {word_x.obj_d, word_x.obj_o}; if (!cf_full) begin cf_push_r = 1'b1; exec_yld_set = 1'b1; exec_cbtype = RTU_CB_TYPE_BITS'(`VX_RT_CB_TYPE_PROC); + exec_objv = word_x.in_blas; word_n.yld_t = coll_t0_q; word_n.yld_ki = cand_ki; word_n.yld_ko = word_x.cur_off; @@ -1587,12 +1590,15 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( cf_din_r.cust = word_x.in_blas ? word_x.inst_cust : 32'd0; cf_din_r.geom = word_x.geom_r; cf_din_r.sbt = cls_sbt; + cf_din_r.objv = word_x.in_blas; + cf_din_r.obj = {word_x.obj_d, word_x.obj_o}; if (cf_full) begin wake_self = 1'b1; end else begin cf_push_r = 1'b1; exec_yld_set = 1'b1; exec_cbtype = cls_cbtype; + exec_objv = word_x.in_blas; word_n.yld_t = trit_q; word_n.yld_ki = cand_ki; word_n.yld_ko = word_x.cur_off; @@ -1848,6 +1854,7 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( mask_q <= '0; hit_q <= '0; yld_q <= '0; + objv_q <= '0; attr_q <= '0; running <= '0; finalised <= '0; @@ -1898,6 +1905,7 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( if (exec_yld_set) begin yld_q[sel_q] <= 1'b1; cbtype_q[sel_q] <= exec_cbtype; + objv_q[sel_q] <= exec_objv; end if (exec_yld_clr) begin yld_q[sel_q] <= 1'b0; @@ -1954,10 +1962,12 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( if (bw_copy_mask[k]) begin yld_q[32'(bw_slot)*NUM_LANES + k] <= 1'b1; cbtype_q[32'(bw_slot)*NUM_LANES + k] <= RTU_CB_TYPE_BITS'(`VX_RT_CB_TYPE_CHS); + objv_q[32'(bw_slot)*NUM_LANES + k] <= 1'b0; end if (bw_miss_mask[k]) begin yld_q[32'(bw_slot)*NUM_LANES + k] <= 1'b1; cbtype_q[32'(bw_slot)*NUM_LANES + k] <= RTU_CB_TYPE_BITS'(`VX_RT_CB_TYPE_MISS); + objv_q[32'(bw_slot)*NUM_LANES + k] <= 1'b0; end end finalised[bw_slot] <= 1'b1; @@ -2159,6 +2169,7 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( assign hit_bits = hit_q; assign yld_bits = yld_q; assign cb_types = cbtype_q; + assign obj_vld = objv_q; assign attr_vld = attr_q; `UNUSED_VAR ({f_aligned, s1_fresh, ray_q, ins_le, ins_here, ins_shift}) diff --git a/tests/raytracing/Makefile b/tests/raytracing/Makefile index 28ac30176a..da11b9fee6 100644 --- a/tests/raytracing/Makefile +++ b/tests/raytracing/Makefile @@ -12,7 +12,7 @@ TESTS := \ rt_smoke_ahs_mixed rt_smoke_recursive \ rt_smoke_bvh_basic rt_smoke_bvh_multilevel \ rt_smoke_bvh_instanced rt_smoke_ahs_bvh rt_smoke_ahs_multi \ - rt_smoke_partial_mask \ + rt_smoke_partial_mask rt_smoke_proc_inst \ rt_smoke_shadow rt_smoke_cull_back \ rt_smoke_async_batch rt_smoke_cull_mask \ rt_smoke_proc rt_smoke_bvh6 rt_bvh_multinode rt_smoke_numctx \ diff --git a/tests/raytracing/rt_smoke_proc_inst/Makefile b/tests/raytracing/rt_smoke_proc_inst/Makefile new file mode 100644 index 0000000000..0379341561 --- /dev/null +++ b/tests/raytracing/rt_smoke_proc_inst/Makefile @@ -0,0 +1,20 @@ +ROOT_DIR := $(realpath ../../..) +include $(ROOT_DIR)/config.mk + +CONFIGS := $(if $(findstring -DVX_CFG_EXT_RTU_ENABLE,$(CONFIGS)),$(CONFIGS),$(CONFIGS) -DVX_CFG_EXT_RTU_ENABLE) +# CW-BVH4 scene -> build the RTU as a CW-BVH4 walker. +CONFIGS += -DVX_CFG_RTU_BVH_WIDTH=4 + +PROJECT := rt_smoke_proc_inst + +SRC_DIR := $(VORTEX_HOME)/tests/raytracing/$(PROJECT) + +SRCS := $(SRC_DIR)/main.cpp + +VX_SRCS := $(SRC_DIR)/kernel.cpp + +OPTS ?= + +KERNEL_LIB := vortex2 + +include ../common.mk diff --git a/tests/raytracing/rt_smoke_proc_inst/common.h b/tests/raytracing/rt_smoke_proc_inst/common.h new file mode 100644 index 0000000000..1c9918f59a --- /dev/null +++ b/tests/raytracing/rt_smoke_proc_inst/common.h @@ -0,0 +1,59 @@ +// Copyright © 2019-2023 +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +#ifndef _RTU_SMOKE_PROC_INST_COMMON_H_ +#define _RTU_SMOKE_PROC_INST_COMMON_H_ + +#include + +// Mirror of sim/simx/rtu/rtu_bvh.h. +#define VX_BVH_SCENE_KIND 2 // kRtuSceneKindBvh4 +#define VX_BVH_SCENE_HDR_BYTES 16 +#define VX_BVH_LEAF_HDR_BYTES 16 +#define VX_BVH_PROC_AABB_BYTES 24 +#define VX_BVH_INSTANCE_STRIDE 64 +#define VX_BVH_INSTANCE_BLAS_OFF 48 +#define VX_BVH_INSTANCE_CUSTOM_ID 52 +#define VX_BVH_INSTANCE_ID_OFFSET 56 +#define VX_BVH_INSTANCE_CULL_MASK 60 +#define VX_BVH_KIND_LEAF_INST 2 +#define VX_BVH_KIND_LEAF_PROC 3 +#define VX_BVH_COUNT_SHIFT 8 + +// Object-space unit sphere the IS intersects, shared by every instance. +#define RTU_SPHERE_CZ 5.0f +#define RTU_SPHERE_R 1.0f + +#define NUM_RAYS 2 + +typedef struct { + uint32_t status; // terminal status + float hit_t; // committed hit distance + uint32_t hit_inst; // committed gl_InstanceID + uint32_t hit_cust; // committed gl_InstanceCustomIndexEXT + uint32_t is_calls; // YIELD_PROC candidates the IS saw + uint32_t is_inst; // gl_InstanceID as the IS saw it + uint32_t is_cust; // gl_InstanceCustomIndexEXT as the IS saw it + float obj_ray[6]; // object-space ray the IS saw (diagnostics) +} rtu_result_t; + +typedef struct { + uint64_t scene_addr; + uint64_t results_addr; + float ray_origin[NUM_RAYS][3]; + float ray_direction[3]; + float tmin; + float tmax; +} kernel_arg_t; + +#endif // _RTU_SMOKE_PROC_INST_COMMON_H_ diff --git a/tests/raytracing/rt_smoke_proc_inst/kernel.cpp b/tests/raytracing/rt_smoke_proc_inst/kernel.cpp new file mode 100644 index 0000000000..41785c2e73 --- /dev/null +++ b/tests/raytracing/rt_smoke_proc_inst/kernel.cpp @@ -0,0 +1,72 @@ +// Copyright © 2019-2023 +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. +// +// Procedural candidates inside instances: the intersection shader must see the +// candidate's own gl_InstanceID / gl_InstanceCustomIndexEXT, and the committed +// hit must carry them too. One ray per lane, each through a different instance. + +#include +#include +#include "common.h" + +__kernel void kernel_main(kernel_arg_t* arg) { + uint32_t tid = threadIdx.x; + if (tid >= NUM_RAYS) return; + + vx_ray_t ray = { + { arg->ray_origin[tid][0], arg->ray_origin[tid][1], arg->ray_origin[tid][2] }, + { arg->ray_direction[0], arg->ray_direction[1], arg->ray_direction[2] }, + arg->tmin, arg->tmax + }; + + uint32_t h = vx_rt_wtrace((uint32_t)arg->scene_addr, 0u, 0u, 0xffu, &ray); + vx_hit_t hit; + uint32_t sts = vx_rt_wait(h, &hit); + uint32_t is_calls = 0, is_inst = ~0u, is_cust = ~0u; + vx_objray_t o = {}; + while (vx_rt_sts_is_yield(sts)) { + uint32_t action = VX_RT_CB_IGNORE; + float hit_t = 0.0f; + if (sts == VX_RT_STS_YIELD_PROC) { + ++is_calls; + is_inst = hit.instance_id; + is_cust = hit.instance_custom; + vx_rt_get_objray(&o); + // |o + t d - C|^2 = r^2 with C = (0,0,CZ) + float ocz = o.origin[2] - RTU_SPHERE_CZ; + float a = o.dir[0]*o.dir[0] + o.dir[1]*o.dir[1] + o.dir[2]*o.dir[2]; + float b = 2.0f * (o.origin[0]*o.dir[0] + o.origin[1]*o.dir[1] + ocz*o.dir[2]); + float c = o.origin[0]*o.origin[0] + o.origin[1]*o.origin[1] + ocz*ocz + - RTU_SPHERE_R*RTU_SPHERE_R; + float disc = b*b - 4.0f*a*c; + if (disc >= 0.0f) { + hit_t = (-b - __builtin_sqrtf(disc)) / (2.0f * a); + action = VX_RT_CB_ACCEPT; + } + } + sts = vx_rt_continue(h, action, hit_t, 0u, &hit); + } + + rtu_result_t* r = (rtu_result_t*)((uintptr_t)arg->results_addr) + tid; + r->status = sts; + r->hit_t = hit.t; + r->hit_inst = hit.instance_id; + r->hit_cust = hit.instance_custom; + r->is_calls = is_calls; + r->is_inst = is_inst; + r->is_cust = is_cust; + for (int i = 0; i < 3; ++i) { + r->obj_ray[i] = o.origin[i]; + r->obj_ray[3 + i] = o.dir[i]; + } +} diff --git a/tests/raytracing/rt_smoke_proc_inst/main.cpp b/tests/raytracing/rt_smoke_proc_inst/main.cpp new file mode 100644 index 0000000000..5cc3fe4e29 --- /dev/null +++ b/tests/raytracing/rt_smoke_proc_inst/main.cpp @@ -0,0 +1,194 @@ +// Copyright © 2019-2023 +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. +// +// PRISM RTU smoke: procedural primitives inside instances. +// +// A TLAS leaf holds two instances of ONE procedural BLAS (a unit sphere at +// object-space (0,0,5)), translated to x=-3 (id 5, custom 0xa5) and x=+3 +// (id 9, custom 0xa9). Two lanes of one warp each fire a +z ray through their +// own instance. Each lane's intersection shader must see the candidate's +// gl_InstanceID / gl_InstanceCustomIndexEXT, and the committed hit must carry +// them: an IS that indexes per-instance data by instance id (every sphere of a +// "Ray Tracing in One Weekend" scene does) reads the wrong sphere otherwise. +// The ids are non-zero on purpose, so a never-written register can't pass. +// +// Scene layout (200 B): +// + 0 VxBvhSceneHeader { root_offset=16, scene_kind=2 } +// + 16 VxBvhLeafHeader { kind=LeafInst|(2<<8) } +// + 32 VxBvhInstance id 5, translate (-3,0,0) -> blas_off=160 +// + 96 VxBvhInstance id 9, translate (+3,0,0) -> blas_off=160 +// +160 VxBvhLeafHeader { kind=LeafProc|(1<<8) } +// +176 VxBvhProcAabb { min=(-1,-1,4), max=(1,1,6) } +// +// Expected per lane: DONE_HIT, t=4, one IS call, IS and hit ids = the lane's +// instance. + +#include +#include +#include + +#include +#include +#include "common.h" + +#define RT_CHECK(_expr) \ + do { \ + int _ret = _expr; \ + if (0 == _ret) break; \ + printf("Error: '%s' returned %d!\n", #_expr, (int)_ret); \ + cleanup(); \ + exit(-1); \ + } while (false) + +const char* kernel_file = "kernel.vxbin"; + +vx_device_h device = nullptr; +vx_buffer_h scene_buffer = nullptr; +vx_buffer_h res_buffer = nullptr; +vx_queue_h queue = nullptr; +vx_module_h module_ = nullptr; +vx_kernel_h kernel = nullptr; +kernel_arg_t kernel_arg = {}; + +void cleanup() { + if (device) { + if (scene_buffer) vx_buffer_release(scene_buffer); + if (res_buffer) vx_buffer_release(res_buffer); + if (kernel) vx_kernel_release(kernel); + if (module_) vx_module_release(module_); + if (queue) vx_queue_release(queue); + vx_device_release(device); + } +} + +static const uint32_t kInstId[NUM_RAYS] = { 5, 9 }; +static const uint32_t kInstCust[NUM_RAYS] = { 0xa5, 0xa9 }; +static const float kInstTx[NUM_RAYS] = { -3.f, 3.f }; + +// Identity rotation + translation (tx,0,0). +static void emit_instance(uint8_t* out, float tx, uint32_t blas_off, + uint32_t custom_id, uint32_t instance_id) { + float* x = reinterpret_cast(out); + x[0] = 1.f; x[1] = 0.f; x[2] = 0.f; x[3] = tx; + x[4] = 0.f; x[5] = 1.f; x[6] = 0.f; x[7] = 0.f; + x[8] = 0.f; x[9] = 0.f; x[10] = 1.f; x[11] = 0.f; + *reinterpret_cast(out + VX_BVH_INSTANCE_BLAS_OFF) = blas_off; + *reinterpret_cast(out + VX_BVH_INSTANCE_CUSTOM_ID) = custom_id; + *reinterpret_cast(out + VX_BVH_INSTANCE_ID_OFFSET) = instance_id; + *reinterpret_cast(out + VX_BVH_INSTANCE_CULL_MASK) = 0xffu; +} + +int main(int /*argc*/, char* /*argv*/[]) { + RT_CHECK(vx_device_open(0, &device)); + vx_queue_info_t qi = { sizeof(qi), nullptr, VX_QUEUE_PRIORITY_NORMAL, 0 }; + RT_CHECK(vx_queue_create(device, &qi, &queue)); + + const uint32_t blas_off = 32 + NUM_RAYS * VX_BVH_INSTANCE_STRIDE; // 160 + std::vector scene(blas_off + VX_BVH_LEAF_HDR_BYTES + VX_BVH_PROC_AABB_BYTES, 0); + + uint32_t* sh = reinterpret_cast(scene.data()); + sh[0] = VX_BVH_SCENE_HDR_BYTES; // root_node_offset = 16 + sh[1] = VX_BVH_SCENE_KIND; // = 2 (BVH4) + sh[2] = (uint32_t)scene.size(); // total scene bytes (pre-fetch) + sh[3] = 2; // leaf_count (1 inst leaf + 1 proc leaf) + + uint32_t* rlh = reinterpret_cast(scene.data() + VX_BVH_SCENE_HDR_BYTES); + rlh[0] = VX_BVH_KIND_LEAF_INST | ((uint32_t)NUM_RAYS << VX_BVH_COUNT_SHIFT); + for (int i = 0; i < NUM_RAYS; ++i) { + emit_instance(scene.data() + 32 + i * VX_BVH_INSTANCE_STRIDE, + kInstTx[i], blas_off, kInstCust[i], kInstId[i]); + } + + uint32_t* blh = reinterpret_cast(scene.data() + blas_off); + blh[0] = VX_BVH_KIND_LEAF_PROC | (1u << VX_BVH_COUNT_SHIFT); + float* aabb = reinterpret_cast(scene.data() + blas_off + VX_BVH_LEAF_HDR_BYTES); + aabb[0] = -1.f; aabb[1] = -1.f; aabb[2] = 4.f; // min + aabb[3] = 1.f; aabb[4] = 1.f; aabb[5] = 6.f; // max + + RT_CHECK(vx_buffer_create(device, (uint32_t)scene.size(), VX_MEM_READ, &scene_buffer)); + RT_CHECK(vx_buffer_address(scene_buffer, &kernel_arg.scene_addr)); + + const uint32_t res_size = NUM_RAYS * sizeof(rtu_result_t); + RT_CHECK(vx_buffer_create(device, res_size, VX_MEM_WRITE, &res_buffer)); + RT_CHECK(vx_buffer_address(res_buffer, &kernel_arg.results_addr)); + + for (int i = 0; i < NUM_RAYS; ++i) { + kernel_arg.ray_origin[i][0] = kInstTx[i]; + kernel_arg.ray_origin[i][1] = 0.f; + kernel_arg.ray_origin[i][2] = 0.f; + } + kernel_arg.ray_direction[0] = 0.f; + kernel_arg.ray_direction[1] = 0.f; + kernel_arg.ray_direction[2] = 1.f; + kernel_arg.tmin = 0.001f; + kernel_arg.tmax = 1e30f; + + std::cout << "scene_addr=0x" << std::hex << kernel_arg.scene_addr << std::dec + << " bvh4 (2 instances of 1 leaf_proc sphere)" << std::endl; + + RT_CHECK(vx_enqueue_write(queue, scene_buffer, 0, scene.data(), + (uint32_t)scene.size(), 0, nullptr, nullptr)); + RT_CHECK(vx_module_load_file(device, kernel_file, &module_)); + RT_CHECK(vx_module_get_kernel(module_, "main", &kernel)); + + std::cout << "launch kernel" << std::endl; + vx_event_h launch_ev = nullptr, read_ev = nullptr; + { + vx_launch_info_t li = {}; + li.struct_size = sizeof(li); + li.kernel = kernel; + li.args_host = &kernel_arg; + li.args_size = sizeof(kernel_arg); + li.ndim = 1; + li.grid_dim[0] = 1; + li.block_dim[0] = NUM_RAYS; // both rays in one warp + RT_CHECK(vx_enqueue_launch(queue, &li, 0, nullptr, &launch_ev)); + } + + rtu_result_t res[NUM_RAYS] = {}; + RT_CHECK(vx_enqueue_read(queue, res, res_buffer, 0, res_size, 1, &launch_ev, &read_ev)); + RT_CHECK(vx_event_wait_value(read_ev, 1, VX_TIMEOUT_INFINITE)); + vx_event_release(read_ev); + vx_event_release(launch_ev); + + int errors = 0; + for (int i = 0; i < NUM_RAYS; ++i) { + const rtu_result_t& r = res[i]; + std::cout << "lane " << i << ": status=" << r.status << " t=" << r.hit_t + << " hit_inst=" << r.hit_inst << " hit_cust=0x" << std::hex << r.hit_cust + << std::dec << " is_calls=" << r.is_calls << " is_inst=" << r.is_inst + << " is_cust=0x" << std::hex << r.is_cust << std::dec + << " obj_ray=(" << r.obj_ray[0] << "," << r.obj_ray[1] << "," << r.obj_ray[2] + << ")+(" << r.obj_ray[3] << "," << r.obj_ray[4] << "," << r.obj_ray[5] << ")" + << std::endl; + bool ok = (r.status == VX_RT_STS_DONE_HIT) + && (std::fabs(r.hit_t - 4.f) <= 1e-4f) + && (r.is_calls == 1) + && (r.is_inst == kInstId[i]) && (r.is_cust == kInstCust[i]) + && (r.hit_inst == kInstId[i]) && (r.hit_cust == kInstCust[i]); + if (!ok) { + std::cout << " expected: status=" << VX_RT_STS_DONE_HIT << " t=4 is_calls=1" + << " inst=" << kInstId[i] << " cust=0x" << std::hex << kInstCust[i] + << std::dec << " (IS and hit)" << std::endl; + ++errors; + } + } + + cleanup(); + if (errors != 0) { + std::cout << "FAILED with " << errors << " errors" << std::endl; + return 1; + } + std::cout << "PASSED!" << std::endl; + return 0; +} From 3da2d0ce85bcfc10ff714bd4d13b8181de62a6a1 Mon Sep 17 00:00:00 2001 From: Blaise Tine Date: Sat, 3 Oct 2026 02:26:21 -0700 Subject: [PATCH 09/31] rtu: match lavapipe's ray/primitive intersection exactly LumiBench images on SimX now have to be bit-identical to lavapipe, so the RTU intersection tests follow lavapipe's arithmetic rather than a generic formulation: - Triangle: watertight test in lavapipe's op order -- fp32 shear, fp64 edge functions and t, det = w0+(w1+w2), t = f32(T/det); accept tmin < t < tmax (strict on both sides, as lvp_build_triangle_case). SimX rtu_isect.cpp and a new pipelined RTL VX_rtu_tri_pe.sv; hw/unittest/rtu_tri_pe checks the RTL against SimX bit-for-bit (1M random cases, 0 mismatches). - Box (SimX): cull against [0, tmax], not [tmin, tmax] -- tmin belongs to the primitive test only. A huge flat triangle whose slab exit lies below tmin was culled although its rounded t passes tmin (CRNVL_PT). Zero direction components use FLT_MAX as reciprocal, as lavapipe. The RTL box PE and the strict tmin compare in the RTL tri PE still need the same change (tracked for the RTLsim step). Co-Authored-By: Claude Opus 5.5 --- hw/rtl/rtu/VX_rtu_pkg.sv | 6 + hw/rtl/rtu/VX_rtu_tri_pe.sv | 986 ++++++++++++++++++-------------- hw/unittest/rtu_tri_pe/Makefile | 34 ++ hw/unittest/rtu_tri_pe/main.cpp | 118 ++++ sim/simx/rtu/rtu_isect.cpp | 125 ++-- sim/simx/rtu/rtu_isect.h | 14 +- sim/simx/rtu/rtu_types.h | 2 + 7 files changed, 824 insertions(+), 461 deletions(-) create mode 100644 hw/unittest/rtu_tri_pe/Makefile create mode 100644 hw/unittest/rtu_tri_pe/main.cpp diff --git a/hw/rtl/rtu/VX_rtu_pkg.sv b/hw/rtl/rtu/VX_rtu_pkg.sv index e0c84d72c5..f94316d891 100644 --- a/hw/rtl/rtu/VX_rtu_pkg.sv +++ b/hw/rtl/rtu/VX_rtu_pkg.sv @@ -98,6 +98,12 @@ package VX_rtu_pkg; // 15). PE delay lines + the scheduler setup window scale with this. localparam RTU_FDIV_LAT = `VX_CFG_RTU_FDIV_LATENCY; + // The tri PE's edge functions and t run in F64 on the soft core, whose + // 53-bit multiply wants a deeper pipe than F32; the F64 divider's depth is + // fixed by its radix-2 recurrence. + localparam RTU_LATENCY_FMA64 = (RTU_LATENCY_FMA > 12) ? RTU_LATENCY_FMA : 12; + localparam RTU_FDIV64_LAT = 32; + // ───────────────────────────────────────────────────────────────── // CW-BVH node-kind tag (low byte of word0) and count field (bits 8..15) // ───────────────────────────────────────────────────────────────── diff --git a/hw/rtl/rtu/VX_rtu_tri_pe.sv b/hw/rtl/rtu/VX_rtu_tri_pe.sv index d8f64f7156..c01949dc94 100644 --- a/hw/rtl/rtu/VX_rtu_tri_pe.sv +++ b/hw/rtl/rtu/VX_rtu_tri_pe.sv @@ -11,23 +11,23 @@ // See the License for the specific language governing permissions and // limitations under the License. -// VX_rtu_tri_pe — pipelined Möller-Trumbore ray-triangle intersector. Streams -// one triangle per cycle and emits {hit, t, u, v, back_facing} after a fixed -// latency. +// VX_rtu_tri_pe — pipelined watertight ray-triangle intersector (Woop, Benthin, +// Wald, JCGT 2013). Streams one triangle per cycle and emits {hit, t, u, v, +// back_facing} after a fixed latency. Mirrors SimX rtu::ray_triangle op for op: // -// e1 = v1 - v0 e2 = v2 - v0 T = origin - v0 -// P = dir × e2 det = e1 · P invDet = 1/det -// u = (T · P) * invDet -// Q = T × e1 v = (dir · Q) * invDet -// t = (e2 · Q) * invDet -// hit = |det| >= FLT_MIN && 0<=u<=1 && 0<=v && u+v<=1 && tmin<=t<=tmax +// kz = argmax|dir|, kx/ky follow (swapped when dir[kz] < 0) +// F32: sz = 1/dir[kz], sx = dir[kx]*sz, sy = dir[ky]*sz +// r = vertex - origin, px = rx - sx*rz, py = ry - sy*rz +// F64: pz = sz*rz (exact), w_i = px_a*py_b - py_a*px_b (one rounding) +// det = w0 + (w1 + w2), T = (w0*pz0 + w1*pz1) + w2*pz2 +// t = f32(T / det), (u, v) = f32(w1, w2) / f32(det) +// hit = !(any w < 0 && any w > 0) && det != 0 && tmin <= t <= tmax // back_facing = det < 0 // -// The FP datapath reuses VX_fma_unit (a*b±c), VX_fdiv_unit (1/det) and -// VX_fncp_unit (compares); the dot/cross products are VX_rtu_fdot3 / -// VX_rtu_fcross3. Side-band operands are delayed through shift registers so -// each stage consumes time-aligned inputs, keeping the whole pipe at a fixed -// latency the scheduler tracks via valid_out. +// A shared edge evaluates to exactly negated weights in its two triangles, so +// the test is watertight; the F64 edge functions and t keep t within half an +// ulp of the exact intersection, in the op order the Vulkan reference +// (lavapipe) uses, so coincident triangles resolve the same way. `include "VX_define.vh" @@ -58,73 +58,404 @@ module VX_rtu_tri_pe import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( output wire [31:0] v, output wire back_facing ); - // VX_fncp_unit result latency is 1 (input pipe reg, OUT_REG=0); the LATENCY - // param only sizes the unused mask pipe, so size it to 2 to avoid a - // degenerate [-1:0] slice while the result still lands after one cycle. - localparam FNCP_LAT = 1; // result latency for alignment - localparam FNCP_SIZE = 2; // mask-pipe sizing param - localparam F = LATENCY_FMA; - localparam V = LATENCY_FDIV; - localparam LATENCY = 8*F + V + 2; - - localparam [INST_FMT_BITS-1:0] FMT_ADD = 2'b00; // F32, a*b + c - localparam [INST_FMT_BITS-1:0] FMT_SUB = 2'b10; // F32, a*b - c - - localparam [31:0] FP_ZERO = 32'h00000000; - localparam [31:0] FP_ONE = 32'h3F800000; - // |det| scales with the triangle's area, so the degenerate test rejects - // only |det| < FLT_MIN (edge-on / zero-area, where 1/det overflows): any - // larger fixed epsilon drops small triangles. - localparam [31:0] FP_EPS = 32'h00800000; // FLT_MIN - localparam [31:0] FP_NEG_EPS = 32'h80800000; // -FLT_MIN - - // ── stage e (@F): edge vectors and ray-origin offset ────────────── - wire [2:0][31:0] e1, e2, tvec; - for (genvar a = 0; a < 3; ++a) begin : g_edges + localparam F = LATENCY_FMA; + localparam V = LATENCY_FDIV; + localparam D = RTU_LATENCY_FMA64; + localparam V64 = RTU_FDIV64_LAT; + + // stage start times (cycles after valid_in) + localparam T_B = 1; // canonical/axis select registered + localparam T_C = T_B + V; // sz ready + localparam T_D = T_C + F; // sx, sy ready + localparam T_E = T_D + F; // sx*rz, sy*rz ready + localparam T_F = T_E + F; // px, py ready + localparam T_G = T_F + 2 * D; // w ready + localparam T_H = T_G + 3 * D; // T ready (det at T_G + 2D) + localparam T_I = T_H + V64 + 1; // t narrowed and registered + localparam LATENCY = T_I + 1; // verdict registered + + `STATIC_ASSERT(V >= F, ("tri PE: FDIV latency must cover the r subtract")) + `STATIC_ASSERT(T_G + 2 * D + 1 + V <= T_I, ("tri PE: bary divide must land before t")) + + localparam [INST_FMT_BITS-1:0] FMT_ADD = 2'b00; + localparam [INST_FMT_BITS-1:0] FMT_SUB = 2'b10; + localparam [31:0] F32_ONE = 32'h3F800000; + + // ── helpers ─────────────────────────────────────────────────────── + // exact F32 -> F64 widening (subnormals normalized) + function automatic [63:0] f32_to_f64(input [31:0] a); + reg [7:0] e; + reg [22:0] m; + reg [4:0] lz; + reg [22:0] mn; + begin + e = a[30:23]; + m = a[22:0]; + if (e == 8'hff) begin + f32_to_f64 = {a[31], 11'h7ff, m, 29'd0}; + end else if (e == 8'd0) begin + if (m == 23'd0) begin + f32_to_f64 = {a[31], 63'd0}; + end else begin + lz = 5'd0; + for (integer i = 22; i >= 0; --i) begin + if (m[i]) begin + lz = 5'(22 - i); + break; + end + end + mn = m << (lz + 5'd1); + f32_to_f64 = {a[31], 11'(11'd896 - 11'(lz)), mn, 29'd0}; + end + end else begin + f32_to_f64 = {a[31], 11'(e) + 11'd896, m, 29'd0}; + end + end + endfunction + + // F64 -> F32, round to nearest even + function automatic [31:0] f64_to_f32(input [63:0] a); + reg s; + reg [10:0] e; + reg [51:0] m; + reg signed [12:0] ue; + reg [52:0] sig; + reg [6:0] sh; + reg [22:0] keep; + reg guard, sticky; + reg [31:0] base; + begin + s = a[63]; + e = a[62:52]; + m = a[51:0]; + ue = 13'(e) - 13'sd896; + if (e == 11'h7ff) begin + f64_to_f32 = {s, 8'hff, (m != 52'd0) ? {1'b1, m[50:29]} : 23'd0}; + end else if (e == 11'd0) begin + f64_to_f32 = {s, 31'd0}; + end else if (ue >= 13'sd255) begin + f64_to_f32 = {s, 8'hff, 23'd0}; + end else if (ue >= 13'sd1) begin + guard = m[28]; + sticky = (m[27:0] != 28'd0); + base = {s, ue[7:0], m[51:29]}; + f64_to_f32 = base + 32'((guard && (sticky || m[29])) ? 1 : 0); + end else begin + // subnormal: mantissa = sig >> (30 - ue), ue <= 0 + sig = {1'b1, m}; + sh = (ue < -13'sd30) ? 7'd61 : 7'(13'sd30 - ue); + if (sh > 7'd54) begin + keep = 23'd0; + guard = 1'b0; + sticky = 1'b1; + end else begin + keep = 23'(sig >> sh); + guard = sig[6'(sh - 7'd1)]; + sticky = (64'(sig) & ((64'd1 << (sh - 7'd1)) - 64'd1)) != 64'd0; + end + base = {s, 8'd0, keep}; + f64_to_f32 = base + 32'((guard && (sticky || keep[0])) ? 1 : 0); + end + end + endfunction + + // IEEE ordering on F32 (+0 == -0); NaN compares false + function automatic f32_le(input [31:0] a, input [31:0] b); + reg a_nan, b_nan; + reg [31:0] ka, kb; + begin + a_nan = (a[30:23] == 8'hff) && (a[22:0] != 23'd0); + b_nan = (b[30:23] == 8'hff) && (b[22:0] != 23'd0); + ka = (a[30:0] == 31'd0) ? 32'h80000000 : (a[31] ? ~a : {1'b1, a[30:0]}); + kb = (b[30:0] == 31'd0) ? 32'h80000000 : (b[31] ? ~b : {1'b1, b[30:0]}); + f32_le = !a_nan && !b_nan && (ka <= kb); + end + endfunction + + // ── stage A (@0 -> @T_B): axis select ───────────────────────────── + wire [30:0] ad0 = dir[0][30:0]; + wire [30:0] ad1 = dir[1][30:0]; + wire [30:0] ad2 = dir[2][30:0]; + wire [1:0] kz_w = (ad0 >= ad1) ? ((ad0 >= ad2) ? 2'd0 : 2'd2) + : ((ad1 >= ad2) ? 2'd1 : 2'd2); + wire [1:0] kx0 = (kz_w == 2'd2) ? 2'd0 : (kz_w + 2'd1); + wire [1:0] ky0 = (kx0 == 2'd2) ? 2'd0 : (kx0 + 2'd1); + wire dz_neg = dir[kz_w][31] && (dir[kz_w][30:0] != 31'd0); + wire [1:0] kx_w = dz_neg ? ky0 : kx0; + wire [1:0] ky_w = dz_neg ? kx0 : ky0; + + // per canonical vertex: (x, y, z) components in the sheared frame's axes + wire [2:0][2:0][31:0] q_w; // [vertex][axis x/y/z] + wire [2:0][2:0][31:0] cvs = {v2, v1, v0}; + for (genvar i = 0; i < 3; ++i) begin : g_q + assign q_w[i][0] = cvs[i][kx_w]; + assign q_w[i][1] = cvs[i][ky_w]; + assign q_w[i][2] = cvs[i][kz_w]; + end + wire [2:0][31:0] o_w = {origin[kz_w], origin[ky_w], origin[kx_w]}; + wire [2:0][31:0] d_w = {dir[kz_w], dir[ky_w], dir[kx_w]}; + + reg [2:0][2:0][31:0] q_a; + reg [2:0][31:0] o_a, d_a; + reg [31:0] tmin_a, tmax_a; + always_ff @(posedge clk) begin + if (enable) begin + q_a <= q_w; + o_a <= o_w; + d_a <= d_w; + tmin_a <= t_min; + tmax_a <= t_max; + end + end + + // ── stage B (@T_B): sz = 1/dir[kz]; r = vertex - origin ─────────── + wire [31:0] sz_c; + VX_fdiv_unit #( + .LATENCY (V), + .FLEN (32), + .USE_DSP (`VX_CFG_RTU_USE_DSP), + .SUBNORM_ENABLE (0), + .EXCEPT_ENABLE (0) + ) fdiv_sz ( + .clk (clk), + .reset (reset), + .enable (enable), + .mask (1'b1), + .fmt ('0), + .frm (INST_FRM_RNE), + .dataa (F32_ONE), + .datab (d_a[2]), + .result (sz_c), + `UNUSED_PIN (fflags) + ); + + wire [2:0][2:0][31:0] r_f; + for (genvar i = 0; i < 3; ++i) begin : g_r + for (genvar a = 0; a < 3; ++a) begin : g_ax + VX_fma_unit #( + .LATENCY (F), + .USE_DSP (`VX_CFG_RTU_USE_DSP), + .SUBNORM_ENABLE (0), + .EXCEPT_ENABLE (0) + ) fsub_r ( + .clk (clk), + .reset (reset), + .enable (enable), + .mask (1'b1), + .op_type (INST_FPU_ADD), + .fmt (FMT_SUB), + .frm (INST_FRM_RNE), + .dataa (q_a[i][a]), + .datab (o_a[a]), + .datac ('0), + .result (r_f[i][a]), + `UNUSED_PIN (fflags) + ); + end + end + + // r from @T_B+F to @T_D (consumed by the sx*rz stage) + wire [2:0][2:0][31:0] r_d; + VX_shift_register #( + .DATAW (9 * 32), + .DEPTH (T_D - (T_B + F)) + ) sr_r ( + .clk (clk), + .reset (reset), + .enable (enable), + .data_in (r_f), + .data_out (r_d) + ); + + wire [1:0][31:0] dxy_c; + VX_shift_register #( + .DATAW (64), + .DEPTH (T_C - T_B) + ) sr_dxy ( + .clk (clk), + .reset (reset), + .enable (enable), + .data_in ({d_a[1], d_a[0]}), + .data_out (dxy_c) + ); + + // ── stage C (@T_C): sx = dir[kx]*sz, sy = dir[ky]*sz ─────────────── + wire [1:0][31:0] sxy_d; + for (genvar a = 0; a < 2; ++a) begin : g_sxy + VX_fma_unit #( + .LATENCY (F), + .USE_DSP (`VX_CFG_RTU_USE_DSP), + .SUBNORM_ENABLE (0), + .EXCEPT_ENABLE (0) + ) fmul_s ( + .clk (clk), + .reset (reset), + .enable (enable), + .mask (1'b1), + .op_type (INST_FPU_MUL), + .fmt (FMT_ADD), + .frm (INST_FRM_RNE), + .dataa (dxy_c[a]), + .datab (sz_c), + .datac ('0), + .result (sxy_d[a]), + `UNUSED_PIN (fflags) + ); + end + + wire [31:0] sz_d; + VX_shift_register #( + .DATAW (32), + .DEPTH (T_D - T_C) + ) sr_sz ( + .clk (clk), + .reset (reset), + .enable (enable), + .data_in (sz_c), + .data_out (sz_d) + ); + + // ── stage D (@T_D): sx*rz, sy*rz (F32); pz = sz*rz (F64, exact) ──── + wire [2:0][1:0][31:0] m_e; + wire [2:0][63:0] pz_x; // @T_D + D + for (genvar i = 0; i < 3; ++i) begin : g_shear + for (genvar a = 0; a < 2; ++a) begin : g_ax + VX_fma_unit #( + .LATENCY (F), + .USE_DSP (`VX_CFG_RTU_USE_DSP), + .SUBNORM_ENABLE (0), + .EXCEPT_ENABLE (0) + ) fmul_m ( + .clk (clk), + .reset (reset), + .enable (enable), + .mask (1'b1), + .op_type (INST_FPU_MUL), + .fmt (FMT_ADD), + .frm (INST_FRM_RNE), + .dataa (sxy_d[a]), + .datab (r_d[i][2]), + .datac ('0), + .result (m_e[i][a]), + `UNUSED_PIN (fflags) + ); + end VX_fma_unit #( - .USE_DSP (`VX_CFG_RTU_USE_DSP), // vendor xil_fma on Vivado (soft in sim), like the FPU - .LATENCY (F), + .LATENCY (D), + .MAN_BITS (52), + .EXP_BITS (11), + .USE_DSP (`VX_CFG_RTU_USE_DSP), .SUBNORM_ENABLE (0), .EXCEPT_ENABLE (0) - ) fma_e1 ( + ) fmul_pz ( .clk (clk), .reset (reset), .enable (enable), .mask (1'b1), - .op_type (INST_FPU_MADD), - .fmt (FMT_SUB), + .op_type (INST_FPU_MUL), + .fmt (FMT_ADD), .frm (INST_FRM_RNE), - .dataa (v1[a]), - .datab (FP_ONE), - .datac (v0[a]), - .result (e1[a]), + .dataa (f32_to_f64(sz_d)), + .datab (f32_to_f64(r_d[i][2])), + .datac ('0), + .result (pz_x[i]), `UNUSED_PIN (fflags) ); + end + + // rx, ry from @T_D to @T_E + wire [2:0][1:0][31:0] rxy_e; + VX_shift_register #( + .DATAW (6 * 32), + .DEPTH (T_E - T_D) + ) sr_rxy ( + .clk (clk), + .reset (reset), + .enable (enable), + .data_in ({r_d[2][1], r_d[2][0], r_d[1][1], r_d[1][0], r_d[0][1], r_d[0][0]}), + .data_out (rxy_e) + ); + + // ── stage E (@T_E): px = rx - sx*rz, py = ry - sy*rz ────────────── + wire [2:0][1:0][31:0] p_f; // [vertex][x/y] + for (genvar i = 0; i < 3; ++i) begin : g_p + for (genvar a = 0; a < 2; ++a) begin : g_ax + VX_fma_unit #( + .LATENCY (F), + .USE_DSP (`VX_CFG_RTU_USE_DSP), + .SUBNORM_ENABLE (0), + .EXCEPT_ENABLE (0) + ) fsub_p ( + .clk (clk), + .reset (reset), + .enable (enable), + .mask (1'b1), + .op_type (INST_FPU_ADD), + .fmt (FMT_SUB), + .frm (INST_FRM_RNE), + .dataa (rxy_e[i][a]), + .datab (m_e[i][a]), + .datac ('0), + .result (p_f[i][a]), + `UNUSED_PIN (fflags) + ); + end + end + + // ── stage F (@T_F): w_i = px_a*py_b - py_a*px_b in F64 ──────────── + // (a, b) per weight: w0 <- (2, 1), w1 <- (0, 2), w2 <- (1, 0) + wire [2:0][1:0][63:0] p64_f; + for (genvar i = 0; i < 3; ++i) begin : g_p64 + assign p64_f[i][0] = f32_to_f64(p_f[i][0]); + assign p64_f[i][1] = f32_to_f64(p_f[i][1]); + end + + wire [2:0][1:0][63:0] p64_g1; // operands delayed D for the fused stage + VX_shift_register #( + .DATAW (6 * 64), + .DEPTH (D) + ) sr_p64 ( + .clk (clk), + .reset (reset), + .enable (enable), + .data_in (p64_f), + .data_out (p64_g1) + ); + + wire [2:0][63:0] w_g; + for (genvar i = 0; i < 3; ++i) begin : g_w + localparam IA = (i == 0) ? 2 : ((i == 1) ? 0 : 1); + localparam IB = (i == 0) ? 1 : ((i == 1) ? 2 : 0); + wire [63:0] cross_q; // py_a * px_b, exact VX_fma_unit #( - .USE_DSP (`VX_CFG_RTU_USE_DSP), // vendor xil_fma on Vivado (soft in sim), like the FPU - .LATENCY (F), + .LATENCY (D), + .MAN_BITS (52), + .EXP_BITS (11), + .USE_DSP (`VX_CFG_RTU_USE_DSP), .SUBNORM_ENABLE (0), .EXCEPT_ENABLE (0) - ) fma_e2 ( + ) fmul_c ( .clk (clk), .reset (reset), .enable (enable), .mask (1'b1), - .op_type (INST_FPU_MADD), - .fmt (FMT_SUB), + .op_type (INST_FPU_MUL), + .fmt (FMT_ADD), .frm (INST_FRM_RNE), - .dataa (v2[a]), - .datab (FP_ONE), - .datac (v0[a]), - .result (e2[a]), + .dataa (p64_f[IA][1]), + .datab (p64_f[IB][0]), + .datac ('0), + .result (cross_q), `UNUSED_PIN (fflags) ); VX_fma_unit #( - .USE_DSP (`VX_CFG_RTU_USE_DSP), // vendor xil_fma on Vivado (soft in sim), like the FPU - .LATENCY (F), + .LATENCY (D), + .MAN_BITS (52), + .EXP_BITS (11), + .USE_DSP (`VX_CFG_RTU_USE_DSP), .SUBNORM_ENABLE (0), .EXCEPT_ENABLE (0) - ) fma_t ( + ) fmsub_w ( .clk (clk), .reset (reset), .enable (enable), @@ -132,449 +463,272 @@ module VX_rtu_tri_pe import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( .op_type (INST_FPU_MADD), .fmt (FMT_SUB), .frm (INST_FRM_RNE), - .dataa (origin[a]), - .datab (FP_ONE), - .datac (v0[a]), - .result (tvec[a]), + .dataa (p64_g1[IA][0]), + .datab (p64_g1[IB][1]), + .datac (cross_q), + .result (w_g[i]), `UNUSED_PIN (fflags) ); end - // dir aligned to the cross/dot consumers - wire [2:0][31:0] dir_f, dir_3f; + // pz from @T_D+D to @T_G + wire [2:0][63:0] pz_g; VX_shift_register #( - .DATAW (96), - .DEPTH (F) - ) sr_dir_f ( + .DATAW (3 * 64), + .DEPTH (T_G - (T_D + D)) + ) sr_pz ( .clk (clk), .reset (reset), .enable (enable), - .data_in (dir), - .data_out (dir_f) + .data_in (pz_x), + .data_out (pz_g) ); + + // ── stage G (@T_G): det = w0 + (w1 + w2); T = (w0 pz0 + w1 pz1) + w2 pz2 + wire [63:0] det12, det_g2, tp0, tp1, tp2, t01, t_num; + wire [63:0] w0_g1, tp2_g2; VX_shift_register #( - .DATAW (96), - .DEPTH (3*F) - ) sr_dir_3f ( + .DATAW (64), + .DEPTH (D) + ) sr_w0 ( .clk (clk), .reset (reset), .enable (enable), - .data_in (dir), - .data_out (dir_3f) + .data_in (w_g[0]), + .data_out (w0_g1) ); - - // ── stage cross (@3F): P = dir × e2, Q = T × e1 ─────────────────── - wire [2:0][31:0] pvec, qvec; - VX_rtu_fcross3 #( - .LATENCY_FMA (F) - ) cross_p ( - .clk (clk), - .reset (reset), - .enable (enable), - .a (dir_f), - .b (e2), - .result (pvec) + VX_fma_unit #(.LATENCY (D), .MAN_BITS (52), .EXP_BITS (11), .USE_DSP (`VX_CFG_RTU_USE_DSP), .SUBNORM_ENABLE (0), .EXCEPT_ENABLE (0)) fadd_det12 ( + .clk (clk), .reset (reset), .enable (enable), .mask (1'b1), + .op_type (INST_FPU_ADD), .fmt (FMT_ADD), .frm (INST_FRM_RNE), + .dataa (w_g[1]), .datab (w_g[2]), .datac ('0), + .result (det12), `UNUSED_PIN (fflags) ); - VX_rtu_fcross3 #( - .LATENCY_FMA (F) - ) cross_q ( - .clk (clk), - .reset (reset), - .enable (enable), - .a (tvec), - .b (e1), - .result (qvec) + VX_fma_unit #(.LATENCY (D), .MAN_BITS (52), .EXP_BITS (11), .USE_DSP (`VX_CFG_RTU_USE_DSP), .SUBNORM_ENABLE (0), .EXCEPT_ENABLE (0)) fadd_det ( + .clk (clk), .reset (reset), .enable (enable), .mask (1'b1), + .op_type (INST_FPU_ADD), .fmt (FMT_ADD), .frm (INST_FRM_RNE), + .dataa (w0_g1), .datab (det12), .datac ('0), + .result (det_g2), `UNUSED_PIN (fflags) ); - - // e1/e2/T aligned from @F to @3F to feed the dot products - wire [2:0][31:0] e1_3f, e2_3f, t_3f; - VX_shift_register #( - .DATAW (96), - .DEPTH (2*F) - ) sr_e1 ( - .clk (clk), - .reset (reset), - .enable (enable), - .data_in (e1), - .data_out (e1_3f) + VX_fma_unit #(.LATENCY (D), .MAN_BITS (52), .EXP_BITS (11), .USE_DSP (`VX_CFG_RTU_USE_DSP), .SUBNORM_ENABLE (0), .EXCEPT_ENABLE (0)) fmul_tp0 ( + .clk (clk), .reset (reset), .enable (enable), .mask (1'b1), + .op_type (INST_FPU_MUL), .fmt (FMT_ADD), .frm (INST_FRM_RNE), + .dataa (w_g[0]), .datab (pz_g[0]), .datac ('0), + .result (tp0), `UNUSED_PIN (fflags) + ); + VX_fma_unit #(.LATENCY (D), .MAN_BITS (52), .EXP_BITS (11), .USE_DSP (`VX_CFG_RTU_USE_DSP), .SUBNORM_ENABLE (0), .EXCEPT_ENABLE (0)) fmul_tp1 ( + .clk (clk), .reset (reset), .enable (enable), .mask (1'b1), + .op_type (INST_FPU_MUL), .fmt (FMT_ADD), .frm (INST_FRM_RNE), + .dataa (w_g[1]), .datab (pz_g[1]), .datac ('0), + .result (tp1), `UNUSED_PIN (fflags) + ); + VX_fma_unit #(.LATENCY (D), .MAN_BITS (52), .EXP_BITS (11), .USE_DSP (`VX_CFG_RTU_USE_DSP), .SUBNORM_ENABLE (0), .EXCEPT_ENABLE (0)) fmul_tp2 ( + .clk (clk), .reset (reset), .enable (enable), .mask (1'b1), + .op_type (INST_FPU_MUL), .fmt (FMT_ADD), .frm (INST_FRM_RNE), + .dataa (w_g[2]), .datab (pz_g[2]), .datac ('0), + .result (tp2), `UNUSED_PIN (fflags) + ); + VX_fma_unit #(.LATENCY (D), .MAN_BITS (52), .EXP_BITS (11), .USE_DSP (`VX_CFG_RTU_USE_DSP), .SUBNORM_ENABLE (0), .EXCEPT_ENABLE (0)) fadd_t01 ( + .clk (clk), .reset (reset), .enable (enable), .mask (1'b1), + .op_type (INST_FPU_ADD), .fmt (FMT_ADD), .frm (INST_FRM_RNE), + .dataa (tp0), .datab (tp1), .datac ('0), + .result (t01), `UNUSED_PIN (fflags) ); VX_shift_register #( - .DATAW (96), - .DEPTH (2*F) - ) sr_e2 ( + .DATAW (64), + .DEPTH (D) + ) sr_tp2 ( .clk (clk), .reset (reset), .enable (enable), - .data_in (e2), - .data_out (e2_3f) + .data_in (tp2), + .data_out (tp2_g2) ); + VX_fma_unit #(.LATENCY (D), .MAN_BITS (52), .EXP_BITS (11), .USE_DSP (`VX_CFG_RTU_USE_DSP), .SUBNORM_ENABLE (0), .EXCEPT_ENABLE (0)) fadd_t ( + .clk (clk), .reset (reset), .enable (enable), .mask (1'b1), + .op_type (INST_FPU_ADD), .fmt (FMT_ADD), .frm (INST_FRM_RNE), + .dataa (t01), .datab (tp2_g2), .datac ('0), + .result (t_num), `UNUSED_PIN (fflags) + ); + + // det from @T_G+2D to @T_H + wire [63:0] det_h; VX_shift_register #( - .DATAW (96), - .DEPTH (2*F) - ) sr_t ( + .DATAW (64), + .DEPTH (T_H - (T_G + 2 * D)) + ) sr_det ( .clk (clk), .reset (reset), .enable (enable), - .data_in (tvec), - .data_out (t_3f) - ); - - // ── stage dot (@6F): det, and the un-scaled u/v/t numerators ────── - wire [31:0] det, u_num, v_num, t_num; - VX_rtu_fdot3 #( - .LATENCY_FMA (F) - ) dot_det ( - .clk (clk), - .reset (reset), - .enable (enable), - .a (e1_3f), - .b (pvec), - .result (det) - ); - VX_rtu_fdot3 #( - .LATENCY_FMA (F) - ) dot_u ( - .clk (clk), - .reset (reset), - .enable (enable), - .a (t_3f), - .b (pvec), - .result (u_num) - ); - VX_rtu_fdot3 #( - .LATENCY_FMA (F) - ) dot_v ( - .clk (clk), - .reset (reset), - .enable (enable), - .a (dir_3f), - .b (qvec), - .result (v_num) - ); - VX_rtu_fdot3 #( - .LATENCY_FMA (F) - ) dot_t ( - .clk (clk), - .reset (reset), - .enable (enable), - .a (e2_3f), - .b (qvec), - .result (t_num) + .data_in (det_g2), + .data_out (det_h) ); - // ── stage recip (@6F+V): invDet = 1/det ─────────────────────────── - wire [31:0] inv_det; + // ── stage H (@T_H): t = f32(T / det) ────────────────────────────── + wire [63:0] t64; VX_fdiv_unit #( - .USE_DSP (`VX_CFG_RTU_USE_DSP), // vendor xil_fdiv on Vivado (soft in sim), like the FPU - .LATENCY (V), + .LATENCY (V64), + .FLEN (64), .SUBNORM_ENABLE (0), .EXCEPT_ENABLE (0) - ) recip ( + ) fdiv_t ( .clk (clk), .reset (reset), .enable (enable), .mask (1'b1), - .fmt ('0), + .fmt (2'b01), .frm (INST_FRM_RNE), - .dataa (FP_ONE), - .datab (det), - .result (inv_det), + .dataa (t_num), + .datab (det_h), + .result (t64), `UNUSED_PIN (fflags) ); - // numerators aligned from @6F to @6F+V - wire [31:0] u_num_d, v_num_d, t_num_d; + reg [31:0] t_i; + always_ff @(posedge clk) begin + if (enable) begin + t_i <= f64_to_f32(t64); + end + end + + // ── barycentrics (@T_G+2D): f32(w1) / f32(det), f32(w2) / f32(det) + wire [1:0][63:0] w_b; // w1, w2 VX_shift_register #( - .DATAW (96), - .DEPTH (V) - ) sr_num ( + .DATAW (2 * 64), + .DEPTH (2 * D) + ) sr_wb ( .clk (clk), .reset (reset), .enable (enable), - .data_in ({u_num, v_num, t_num}), - .data_out ({u_num_d, v_num_d, t_num_d}) + .data_in ({w_g[2], w_g[1]}), + .data_out (w_b) ); + reg [31:0] wu_r, wv_r, det32_r; + always_ff @(posedge clk) begin + if (enable) begin + wu_r <= f64_to_f32(w_b[0]); + wv_r <= f64_to_f32(w_b[1]); + det32_r <= f64_to_f32(det_g2); + end + end - // ── stage scale (@7F+V): u/v/t = numerator * invDet ─────────────── - wire [31:0] u_w, v_w, t_w; - VX_fma_unit #( - .USE_DSP (`VX_CFG_RTU_USE_DSP), // vendor xil_fma on Vivado (soft in sim), like the FPU - .LATENCY (F), - .SUBNORM_ENABLE (0) - ) fma_u ( - .clk (clk), - .reset (reset), - .enable (enable), - .mask (1'b1), - .op_type (INST_FPU_MADD), - .fmt (FMT_ADD), - .frm (INST_FRM_RNE), - .dataa (u_num_d), - .datab (inv_det), - .datac (FP_ZERO), - .result (u_w), - `UNUSED_PIN (fflags) - ); - VX_fma_unit #( - .USE_DSP (`VX_CFG_RTU_USE_DSP), // vendor xil_fma on Vivado (soft in sim), like the FPU - .LATENCY (F), - .SUBNORM_ENABLE (0) - ) fma_v ( - .clk (clk), - .reset (reset), - .enable (enable), - .mask (1'b1), - .op_type (INST_FPU_MADD), - .fmt (FMT_ADD), - .frm (INST_FRM_RNE), - .dataa (v_num_d), - .datab (inv_det), - .datac (FP_ZERO), - .result (v_w), - `UNUSED_PIN (fflags) - ); - VX_fma_unit #( - .USE_DSP (`VX_CFG_RTU_USE_DSP), // vendor xil_fma on Vivado (soft in sim), like the FPU - .LATENCY (F), - .SUBNORM_ENABLE (0) - ) fma_t2 ( + wire [31:0] u_q, v_q; + VX_fdiv_unit #( + .LATENCY (V), + .FLEN (32), + .USE_DSP (`VX_CFG_RTU_USE_DSP), + .SUBNORM_ENABLE (0), + .EXCEPT_ENABLE (0) + ) fdiv_u ( .clk (clk), .reset (reset), .enable (enable), .mask (1'b1), - .op_type (INST_FPU_MADD), - .fmt (FMT_ADD), + .fmt ('0), .frm (INST_FRM_RNE), - .dataa (t_num_d), - .datab (inv_det), - .datac (FP_ZERO), - .result (t_w), + .dataa (wu_r), + .datab (det32_r), + .result (u_q), `UNUSED_PIN (fflags) ); - - // ── stage sum (@8F+V): uv = u + v ───────────────────────────────── - wire [31:0] uv_w; - VX_fma_unit #( - .USE_DSP (`VX_CFG_RTU_USE_DSP), // vendor xil_fma on Vivado (soft in sim), like the FPU - .LATENCY (F), - .SUBNORM_ENABLE (0) - ) fma_uv ( + VX_fdiv_unit #( + .LATENCY (V), + .FLEN (32), + .USE_DSP (`VX_CFG_RTU_USE_DSP), + .SUBNORM_ENABLE (0), + .EXCEPT_ENABLE (0) + ) fdiv_v ( .clk (clk), .reset (reset), .enable (enable), .mask (1'b1), - .op_type (INST_FPU_MADD), - .fmt (FMT_ADD), + .fmt ('0), .frm (INST_FRM_RNE), - .dataa (u_w), - .datab (FP_ONE), - .datac (v_w), - .result (uv_w), + .dataa (wv_r), + .datab (det32_r), + .result (v_q), `UNUSED_PIN (fflags) ); - // u/v/t aligned from @7F+V to @8F+V - wire [31:0] u_c, v_c, t_c; + wire [31:0] u_i, v_i; VX_shift_register #( - .DATAW (96), - .DEPTH (F) - ) sr_uvt ( + .DATAW (64), + .DEPTH (T_I - (T_G + 2 * D + 1 + V)) + ) sr_uv ( .clk (clk), .reset (reset), .enable (enable), - .data_in ({u_w, v_w, t_w}), - .data_out ({u_c, v_c, t_c}) + .data_in ({u_q, v_q}), + .data_out ({u_i, v_i}) ); - // det aligned from @6F to @8F+V - wire [31:0] det_c; + + // ── verdict flags: edge signs (@T_G), det tests (@T_G+2D) ───────── + reg edge_ok_g; + always @(*) begin + reg any_neg, any_pos; + any_neg = 1'b0; + any_pos = 1'b0; + for (integer i = 0; i < 3; ++i) begin + if (w_g[i][62:0] != 63'd0 + && !((w_g[i][62:52] == 11'h7ff) && (w_g[i][51:0] != 52'd0))) begin + any_neg = any_neg | w_g[i][63]; + any_pos = any_pos | ~w_g[i][63]; + end + end + edge_ok_g = !(any_neg && any_pos); + end + + wire edge_ok_d; VX_shift_register #( - .DATAW (32), - .DEPTH (2*F + V) - ) sr_det ( + .DATAW (1), + .DEPTH (2 * D) + ) sr_edge ( .clk (clk), .reset (reset), .enable (enable), - .data_in (det), - .data_out (det_c) + .data_in (edge_ok_g), + .data_out (edge_ok_d) ); - // t_min/t_max aligned from @0 to @8F+V - wire [31:0] tmin_c, tmax_c; + + wire det_nan = (det_g2[62:52] == 11'h7ff) && (det_g2[51:0] != 52'd0); + wire det_ok_d = (det_g2[62:0] != 63'd0) && !det_nan; + wire back_d = det_g2[63]; + + wire [2:0] flags_i; VX_shift_register #( - .DATAW (64), - .DEPTH (8*F + V) - ) sr_tmm ( + .DATAW (3), + .DEPTH (T_I - (T_G + 2 * D)) + ) sr_flags ( .clk (clk), .reset (reset), .enable (enable), - .data_in ({t_min, t_max}), - .data_out ({tmin_c, tmax_c}) - ); - - // ── stage compare (@8F+V+1): bound and determinant tests ────────── - wire [`VX_CFG_XLEN-1:0] cu0, cu1, cv0, cuv, ct0, ct1, cdp, cdn, bfc; - `UNUSED_VAR ({cu0, cu1, cv0, cuv, ct0, ct1, cdp, cdn, bfc}) - VX_fncp_unit #( - .LATENCY (FNCP_SIZE) - ) cmp_u0 ( - .clk (clk), - .reset (reset), - .enable (enable), - .mask (1'b1), - .op_type (INST_FPU_CMP), - .fmt ('0), - .frm (3'd0 /*LE*/), - .dataa (FP_ZERO), - .datab (u_c), - .result (cu0), - `UNUSED_PIN (fflags) - ); - VX_fncp_unit #( - .LATENCY (FNCP_SIZE) - ) cmp_u1 ( - .clk (clk), - .reset (reset), - .enable (enable), - .mask (1'b1), - .op_type (INST_FPU_CMP), - .fmt ('0), - .frm (3'd0), - .dataa (u_c), - .datab (FP_ONE), - .result (cu1), - `UNUSED_PIN (fflags) - ); - VX_fncp_unit #( - .LATENCY (FNCP_SIZE) - ) cmp_v0 ( - .clk (clk), - .reset (reset), - .enable (enable), - .mask (1'b1), - .op_type (INST_FPU_CMP), - .fmt ('0), - .frm (3'd0), - .dataa (FP_ZERO), - .datab (v_c), - .result (cv0), - `UNUSED_PIN (fflags) - ); - VX_fncp_unit #( - .LATENCY (FNCP_SIZE) - ) cmp_uv ( - .clk (clk), - .reset (reset), - .enable (enable), - .mask (1'b1), - .op_type (INST_FPU_CMP), - .fmt ('0), - .frm (3'd0), - .dataa (uv_w), - .datab (FP_ONE), - .result (cuv), - `UNUSED_PIN (fflags) - ); - VX_fncp_unit #( - .LATENCY (FNCP_SIZE) - ) cmp_t0 ( - .clk (clk), - .reset (reset), - .enable (enable), - .mask (1'b1), - .op_type (INST_FPU_CMP), - .fmt ('0), - .frm (3'd0), - .dataa (tmin_c), - .datab (t_c), - .result (ct0), - `UNUSED_PIN (fflags) - ); - VX_fncp_unit #( - .LATENCY (FNCP_SIZE) - ) cmp_t1 ( - .clk (clk), - .reset (reset), - .enable (enable), - .mask (1'b1), - .op_type (INST_FPU_CMP), - .fmt ('0), - .frm (3'd0), - .dataa (t_c), - .datab (tmax_c), - .result (ct1), - `UNUSED_PIN (fflags) - ); - VX_fncp_unit #( - .LATENCY (FNCP_SIZE) - ) cmp_dp ( - .clk (clk), - .reset (reset), - .enable (enable), - .mask (1'b1), - .op_type (INST_FPU_CMP), - .fmt ('0), - .frm (3'd0), - .dataa (FP_EPS), - .datab (det_c), - .result (cdp), - `UNUSED_PIN (fflags) - ); - VX_fncp_unit #( - .LATENCY (FNCP_SIZE) - ) cmp_dn ( - .clk (clk), - .reset (reset), - .enable (enable), - .mask (1'b1), - .op_type (INST_FPU_CMP), - .fmt ('0), - .frm (3'd0), - .dataa (det_c), - .datab (FP_NEG_EPS), - .result (cdn), - `UNUSED_PIN (fflags) - ); - VX_fncp_unit #( - .LATENCY (FNCP_SIZE) - ) cmp_bf ( - .clk (clk), - .reset (reset), - .enable (enable), - .mask (1'b1), - .op_type (INST_FPU_CMP), - .fmt ('0), - .frm (3'd1 /*LT*/), - .dataa (det_c), - .datab (FP_ZERO), - .result (bfc), - `UNUSED_PIN (fflags) + .data_in ({edge_ok_d, det_ok_d, back_d}), + .data_out (flags_i) ); - wire pass_w = cu0[0] & cu1[0] & cv0[0] & cuv[0] - & ct0[0] & ct1[0] & (cdp[0] | cdn[0]); - - // u/v/t aligned from @8F+V to @8F+V+1 (one compare stage) - wire [31:0] u_a, v_a, t_a; + wire [63:0] tmm_i; VX_shift_register #( - .DATAW (96), - .DEPTH (FNCP_LAT) - ) sr_uvt2 ( + .DATAW (64), + .DEPTH (T_I - T_B) + ) sr_tmm ( .clk (clk), .reset (reset), .enable (enable), - .data_in ({u_c, v_c, t_c}), - .data_out ({u_a, v_a, t_a}) + .data_in ({tmin_a, tmax_a}), + .data_out (tmm_i) ); - // ── stage commit (@8F+V+2): register the verdict and attributes ─── + // ── stage I (@T_I): range test and commit ───────────────────────── + wire range_ok = f32_le(tmm_i[63:32], t_i) && f32_le(t_i, tmm_i[31:0]); + reg hit_r, bf_r; reg [31:0] u_r, v_r, t_r; always_ff @(posedge clk) begin if (enable) begin - hit_r <= pass_w; - bf_r <= bfc[0]; - u_r <= u_a; - v_r <= v_a; - t_r <= t_a; + hit_r <= flags_i[2] && flags_i[1] && range_ok; + bf_r <= flags_i[0]; + u_r <= u_i; + v_r <= v_i; + t_r <= t_i; end end @@ -587,8 +741,6 @@ module VX_rtu_tri_pe import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( end end - // carry the caller's tag alongside the datapath so streamed results can be - // routed back to their originating context. wire [TAG_WIDTH-1:0] tag_out_w; VX_shift_register #( .DATAW (TAG_WIDTH), diff --git a/hw/unittest/rtu_tri_pe/Makefile b/hw/unittest/rtu_tri_pe/Makefile new file mode 100644 index 0000000000..3d196e3e5d --- /dev/null +++ b/hw/unittest/rtu_tri_pe/Makefile @@ -0,0 +1,34 @@ +ROOT_DIR := $(realpath ../../..) +include $(ROOT_DIR)/config.mk + +PROJECT := rtu_tri_pe + +RTL_DIR := $(VORTEX_HOME)/hw/rtl +SRC_DIR := $(VORTEX_HOME)/hw/unittest/$(PROJECT) +SIMX_DIR := $(VORTEX_HOME)/sim/simx + +# The reference is the SimX model itself, so the RTU config must match both sides. +CONFIGS += -DSIMULATION -DVX_CFG_EXT_RTU_ENABLE + +CXXFLAGS := -I$(SRC_DIR) -I$(VORTEX_HOME)/hw/unittest/common -I$(SW_COMMON_DIR) +CXXFLAGS += -I$(SIMX_DIR) -I$(SIMX_DIR)/rtu -I$(SIM_COMMON_DIR) +CXXFLAGS += -I$(ROOT_DIR)/sw -I$(ROOT_DIR)/hw +CXXFLAGS += -I$(THIRD_PARTY_DIR)/softfloat/source/include + +SRCS += $(SIMX_DIR)/rtu/rtu_isect.cpp +SRCS += $(SRC_DIR)/main.cpp + +PARAMS := -GTAG_WIDTH=32 + +RTL_PKGS += $(RTL_DIR)/VX_gpu_pkg.sv $(RTL_DIR)/fpu/VX_fpu_pkg.sv $(RTL_DIR)/rtu/VX_rtu_pkg.sv +RTL_INCLUDE := -I$(ROOT_DIR)/sw -I$(RTL_DIR) -I$(RTL_DIR)/libs -I$(RTL_DIR)/interfaces +RTL_INCLUDE += -I$(RTL_DIR)/fpu -I$(RTL_DIR)/rtu -I$(SRC_DIR) + +VL_FLAGS += -I$(ROOT_DIR)/hw + +TOP := VX_rtu_tri_pe + +include ../common.mk + +# the soft FMA's function locals shadow this module's v port once inlined +VL_FLAGS += -Wno-VARHIDDEN diff --git a/hw/unittest/rtu_tri_pe/main.cpp b/hw/unittest/rtu_tri_pe/main.cpp new file mode 100644 index 0000000000..81caf6b42c --- /dev/null +++ b/hw/unittest/rtu_tri_pe/main.cpp @@ -0,0 +1,118 @@ +// VX_rtu_tri_pe against SimX's rtu::ray_triangle: every verdict and, for a hit, +// t / u / v / back_facing must match bit for bit. + +#include +#include +#include +#include +#include +#include "VVX_rtu_tri_pe.h" +#include "verilated.h" +#include "rtu_isect.h" + +namespace { + +struct Case { + float o[3], d[3], v[3][3], tmin, tmax; +}; + +struct Expect { + uint32_t id; + bool hit; + float t, u, v; + bool back; +}; + +uint32_t bits(float f) { uint32_t b; std::memcpy(&b, &f, 4); return b; } + +std::mt19937 rng(7); +float uni(float lo, float hi) { return std::uniform_real_distribution(lo, hi)(rng); } + +// A ray aimed at a point inside (or just outside) the triangle, so hits, +// near-edge misses and shared-edge cases all get exercised. +Case make_case(uint32_t i, const Case* twin_of) { + Case c; + if (twin_of) { + c = *twin_of; + // coincident twin: rotate (even) or mirror (odd) the vertex order + float v[3][3]; + std::memcpy(v, c.v, sizeof v); + if (i & 1) { std::memcpy(c.v[1], v[2], 12); std::memcpy(c.v[2], v[1], 12); } + else { std::memcpy(c.v[0], v[1], 12); std::memcpy(c.v[1], v[2], 12); std::memcpy(c.v[2], v[0], 12); } + return c; + } + const float s = (i % 7 == 0) ? 1e3f : ((i % 5 == 0) ? 1e-2f : 10.f); + for (auto& vv : c.v) for (float& x : vv) x = uni(-s, s); + float a = uni(-0.2f, 1.2f), b = uni(-0.2f, 1.2f); + if ((i % 3) == 0) b = 1.f - a; // on / near an edge + float p[3]; + for (int k = 0; k < 3; ++k) + p[k] = c.v[0][k] + a * (c.v[1][k] - c.v[0][k]) + b * (c.v[2][k] - c.v[0][k]); + for (int k = 0; k < 3; ++k) c.o[k] = uni(-3 * s, 3 * s); + for (int k = 0; k < 3; ++k) c.d[k] = p[k] - c.o[k]; + if ((i % 11) == 0) c.d[i % 3] = 0.f; // axis-aligned component + c.tmin = (i % 13 == 0) ? uni(0.f, 1.f) : 0.f; + c.tmax = (i % 17 == 0) ? uni(0.f, 2.f) : 1e30f; + return c; +} + +} // namespace + +int main(int argc, char** argv) { + Verilated::commandArgs(argc, argv); + const uint32_t N = (argc > 1) ? uint32_t(std::atoi(argv[1])) : 1000000; + VVX_rtu_tri_pe dut; + dut.clk = 0; dut.reset = 1; dut.enable = 1; dut.valid_in = 0; + auto tick = [&] { dut.clk = 0; dut.eval(); dut.clk = 1; dut.eval(); }; + for (int i = 0; i < 4; ++i) tick(); + dut.reset = 0; + + std::deque exp; + uint32_t sent = 0, checked = 0, errors = 0, hits = 0, twins = 0; + Case prev{}; + while (checked < N) { + if (sent < N) { + const bool twin = (sent > 0) && (sent % 4 == 1); + Case c = make_case(sent, twin ? &prev : nullptr); + if (!twin) prev = c; else ++twins; + for (int k = 0; k < 3; ++k) { + dut.origin[k] = bits(c.o[k]); + dut.dir[k] = bits(c.d[k]); + dut.v0[k] = bits(c.v[0][k]); + dut.v1[k] = bits(c.v[1][k]); + dut.v2[k] = bits(c.v[2][k]); + } + dut.t_min = bits(c.tmin); + dut.t_max = bits(c.tmax); + dut.tag_in = sent; + dut.valid_in = 1; + Expect e{sent, false, 0, 0, 0, false}; + e.hit = vortex::rtu::ray_triangle(c.o, c.d, c.v[0], c.v[1], c.v[2], + c.tmin, c.tmax, e.t, e.u, e.v, e.back); + exp.push_back(e); + ++sent; + } else { + dut.valid_in = 0; + } + tick(); + if (dut.valid_out) { + const Expect e = exp.front(); + exp.pop_front(); + bool ok = (dut.tag_out == e.id) && (bool(dut.hit) == e.hit); + if (ok && e.hit) { + ok = dut.t == bits(e.t) && dut.u == bits(e.u) && dut.v == bits(e.v) + && bool(dut.back_facing) == e.back; + ++hits; + } + if (!ok && errors++ < 10) { + std::printf("MISMATCH #%u: ref hit=%d t=%08x u=%08x v=%08x bf=%d | rtl tag=%u hit=%d t=%08x u=%08x v=%08x bf=%d\n", + e.id, e.hit, bits(e.t), bits(e.u), bits(e.v), e.back, + dut.tag_out, dut.hit, dut.t, dut.u, dut.v, dut.back_facing); + } + ++checked; + } + } + std::printf("rtu_tri_pe: %u cases (%u hits, %u twins), %u mismatches\n", checked, hits, twins, errors); + std::printf(errors ? "FAILED!\n" : "PASSED!\n"); + return errors ? 1 : 0; +} diff --git a/sim/simx/rtu/rtu_isect.cpp b/sim/simx/rtu/rtu_isect.cpp index 140d643a70..dc3309a297 100644 --- a/sim/simx/rtu/rtu_isect.cpp +++ b/sim/simx/rtu/rtu_isect.cpp @@ -14,6 +14,7 @@ #include "rtu_isect.h" #include #include +#include namespace vortex { namespace rtu { @@ -22,50 +23,98 @@ bool ray_triangle(const float ro[3], const float rd[3], float tmin, float tmax, float& out_t, float& out_u, float& out_v, bool& out_back_facing) { - Vec3 O = { ro[0], ro[1], ro[2] }; - Vec3 D = { rd[0], rd[1], rd[2] }; - Vec3 V0 = { v0[0], v0[1], v0[2] }; - Vec3 V1 = { v1[0], v1[1], v1[2] }; - Vec3 V2 = { v2[0], v2[1], v2[2] }; - - Vec3 e1 = V1 - V0; - Vec3 e2 = V2 - V0; - Vec3 P = cross(D, e2); - float det = dot(e1, P); - // Reject only a degenerate (edge-on or zero-area) triangle: |det| scales - // with the triangle's area, so any fixed epsilon above the float range would - // drop small triangles. Below FLT_MIN the reciprocal overflows. - if (!(std::fabs(det) >= FLT_MIN)) return false; - float invDet = 1.0f / det; - Vec3 T = O - V0; - float u = dot(T, P) * invDet; - if (u < 0.f || u > 1.f) return false; - Vec3 Q = cross(T, e1); - float v = dot(D, Q) * invDet; - if (v < 0.f || u + v > 1.f) return false; - float t = dot(e2, Q) * invDet; - if (t < tmin || t > tmax) return false; + const float* vin[3] = { v0, v1, v2 }; + + // Watertight ray/triangle test (Woop, Benthin, Wald, JCGT 2013): shear the + // triangle into the ray's frame so the ray runs along +z, then test the 2D + // edge functions. A shared edge evaluates to exactly negated values in its + // two triangles, so no ray slips between them; F64 edge functions and t keep + // t within half an ulp of the exact intersection. The op order is the one + // the Vulkan reference (lavapipe) evaluates, so t matches it bit for bit, + // coincident triangles included. + const float ad[3] = { std::fabs(rd[0]), std::fabs(rd[1]), std::fabs(rd[2]) }; + int kz = (ad[0] >= ad[1]) ? ((ad[0] >= ad[2]) ? 0 : 2) + : ((ad[1] >= ad[2]) ? 1 : 2); + int kx = (kz + 1) % 3; + int ky = (kx + 1) % 3; + if (rd[kz] < 0.f) std::swap(kx, ky); // keep the winding + + const float sz = 1.0f / rd[kz]; + const float sx = rd[kx] * sz; + const float sy = rd[ky] * sz; + + // F32 shear, as each op rounds in the pipeline; F64 from here on, where the + // products of two F32 values are exact. + float px[3], py[3]; + double pz[3]; + for (int i = 0; i < 3; ++i) { + const float* q = vin[i]; + const float rx = q[kx] - ro[kx]; + const float ry = q[ky] - ro[ky]; + const float rz = q[kz] - ro[kz]; + const float mx = sx * rz; + const float my = sy * rz; + px[i] = rx - mx; + py[i] = ry - my; + pz[i] = double(sz) * double(rz); + } + + // Edge functions: w[i] is the weight of vertex i. + double w[3]; + w[0] = double(px[2]) * py[1] - double(py[2]) * px[1]; + w[1] = double(px[0]) * py[2] - double(py[0]) * px[2]; + w[2] = double(px[1]) * py[0] - double(py[1]) * px[0]; + if ((w[0] < 0.0 || w[1] < 0.0 || w[2] < 0.0) + && (w[0] > 0.0 || w[1] > 0.0 || w[2] > 0.0)) + return false; + + const double det = w[0] + (w[1] + w[2]); + // Reject only an edge-on or zero-area triangle: |det| scales with the + // triangle's area, so any epsilon would drop small triangles. + if (!(det != 0.0)) return false; + + const double tp0 = w[0] * pz[0]; + const double tp1 = w[1] * pz[1]; + const double tp2 = w[2] * pz[2]; + const float t = float(((tp0 + tp1) + tp2) / det); + // Open interval, as the reference commits (lvp_build_triangle_case: + // tmin < t and t < tmax). Callers pass the RAY's tmax, never the committed + // t, so an equal-t twin still reaches the walker's tie-break. + if (!(tmin < t && t < tmax)) return false; + + const float det32 = float(det); out_t = t; - out_u = u; - out_v = v; - out_back_facing = (det < 0.f); + out_u = float(w[1]) / det32; + out_v = float(w[2]) / det32; + // det > 0: (v0, v1, v2) winds counter-clockwise as seen by the ray. + out_back_facing = (det < 0.0); return true; } bool ray_aabb_intersect(const float ro[3], const float rd[3], const float mn[3], const float mx[3], float tmin, float tmax, float& t_near) { - float tn = tmin, tf = tmax; + // Slab test in the Vulkan reference's (lavapipe) form: a zero direction + // component uses FLT_MAX as its reciprocal, and the box is culled against + // [0, tmax], NOT [tmin, tmax]. The tmin floor belongs to the primitive test + // alone: a primitive's t carries rounding the slab distances do not (the + // watertight triangle t of a large triangle is off by far more than the + // slabs of its flat box), so a box whose exact exit lies below tmin can + // still hold a hit the primitive test reports past tmin. Culling at tmin + // would drop that hit, which the reference keeps. + float lo = -INFINITY, hi = INFINITY; for (int i = 0; i < 3; ++i) { - float inv = 1.0f / rd[i]; - float t0 = (mn[i] - ro[i]) * inv; - float t1 = (mx[i] - ro[i]) * inv; - if (t0 > t1) { float tmp = t0; t0 = t1; t1 = tmp; } - if (t0 > tn) tn = t0; - if (t1 < tf) tf = t1; - if (tn > tf) return false; + const float inv = (rd[i] == 0.0f) ? FLT_MAX : 1.0f / rd[i]; + const float t0 = (mn[i] - ro[i]) * inv; + const float t1 = (mx[i] - ro[i]) * inv; + lo = std::fmax(lo, std::fmin(t0, t1)); + hi = std::fmin(hi, std::fmax(t0, t1)); } - t_near = tn; + // The upper bound stays inclusive (the reference's is strict): a box + // entered exactly at the committed t can hold an equal-t twin that the + // lowest-(instance, geometry, prim) tie-break must still see. + if (!(hi >= std::fmax(0.0f, lo) && lo <= tmax)) return false; + t_near = std::fmax(tmin, lo); // descent order only return true; } @@ -122,8 +171,10 @@ uint32_t BoxPe::pipe_depth() { } uint32_t TriPe::pipe_depth() { - // 8 FMA stages + 1 reciprocal + 2 = 91. - return 8 * kRtuLatencyFma + kRtuFdivLat + 2; + // input select + 1/dir + 3 F32 stages + 5 F64 stages + F64 divide + narrow + // + verdict (VX_rtu_tri_pe). + return 3 + kRtuFdivLat + 3 * kRtuLatencyFma + 5 * kRtuLatencyFma64 + + kRtuFdiv64Lat; } }} // namespace vortex::rtu diff --git a/sim/simx/rtu/rtu_isect.h b/sim/simx/rtu/rtu_isect.h index 4d2de405a2..8b2f33e871 100644 --- a/sim/simx/rtu/rtu_isect.h +++ b/sim/simx/rtu/rtu_isect.h @@ -36,6 +36,8 @@ namespace vortex { namespace rtu { // triangle's geometric normal (ray-flag face culling). Convention: // triangle front face is the side from which (v0, v1, v2) appear CCW. // Equivalently, det > 0 ↔ ray hits the front face. +// A hit needs tmin < t < tmax (open, as the Vulkan reference); tmax is +// the ray's, not the committed hit's. // ──────────────────────────────────────────────────────────────────── bool ray_triangle(const float ro[3], const float rd[3], const float v0[3], const float v1[3], const float v2[3], @@ -44,13 +46,11 @@ bool ray_triangle(const float ro[3], const float rd[3], bool& out_back_facing); // ──────────────────────────────────────────────────────────────────── -// Ray-vs-AABB slab test. Returns true if the ray's [tmin, tmax] -// interval overlaps the AABB; t_near is the entry parameter (clamped -// to tmin) used by the BVH4 walker to prune descent order. -// -// Assumes well-conditioned rays (no axis-aligned ray with zero -// direction component). A robust branchless ±inf variant is a later -// refinement. +// Ray-vs-AABB slab test. Returns true if the ray's [0, tmax] interval +// overlaps the AABB (the reference's box test: tmin is applied by the +// primitive test only, see rtu_isect.cpp); t_near is the entry +// parameter (clamped to tmin) used by the BVH4 walker to order descent. +// A zero direction component uses FLT_MAX as its reciprocal. // ──────────────────────────────────────────────────────────────────── bool ray_aabb_intersect(const float ro[3], const float rd[3], const float mn[3], const float mx[3], diff --git a/sim/simx/rtu/rtu_types.h b/sim/simx/rtu/rtu_types.h index 1a326407e7..d24c27fbda 100644 --- a/sim/simx/rtu/rtu_types.h +++ b/sim/simx/rtu/rtu_types.h @@ -298,6 +298,8 @@ constexpr uint32_t kRtuImageStatesPerRay = 1; // scene header constexpr uint32_t kRtuSetupLatency = 17; // reciprocal pipe depth constexpr uint32_t kRtuFdivLat = 17; // reciprocal pipe depth constexpr uint32_t kRtuLatencyFma = 9; // FMA pipe depth +constexpr uint32_t kRtuLatencyFma64 = 12; // F64 FMA pipe depth (tri PE) +constexpr uint32_t kRtuFdiv64Lat = 32; // F64 divide pipe depth (tri PE) // Per-instance transform latency = 4 * FMA pipe depth = 36: an (ro-t) subtract // at FMA depth, then a 3-deep dot product. Charged per TLAS instance descent // in the SimX cost model. From 69299a2efbeedc4e1955dbb9d01de8e79ce6af66 Mon Sep 17 00:00:00 2001 From: Blaise Tine Date: Sat, 3 Oct 2026 02:26:21 -0700 Subject: [PATCH 10/31] rtu: report hit facing and pick lavapipe's hit among near-equal ones - gl_HitKindEXT: the back-facing flag rides in bit 31 of the hit geometry index (VX_RT_HIT_BACK_FACING); the geometry index is masked to 28 bits (VX_RT_HIT_GEOMETRY_MASK). SimX walker + RTL scheduler. - Hit choice: lavapipe commits only t < tmax and walks its binary BVH nearest-child-first, so among equal-t hits the first visited wins, and its fp32 box cull can hide a hit a few ulps nearer. Coincident/near-coincident geometry (SHIP sails, PARK body, BATH, PARTY, SPNZA) depends on this. The driver now appends lavapipe's tree as visit-order tables (TLAS in the scene; compact 32 B/node BLAS tables in a separate buffer). For an opaque hit within t*2^-19 of the committed one, the SimX walker climbs both leaves to their LCA, replays lavapipe's box test/order and its cull, and keeps the hit lavapipe would. Scenes without tables keep the static (instance, geometry, prim) key, which the RTL scheduler still uses. Co-Authored-By: Claude Opus 5.5 --- VX_types.toml | 10 +- hw/rtl/rtu/VX_rtu_scheduler.sv | 30 +- sim/simx/rtu/rtu_bvh.h | 13 +- sim/simx/rtu/rtu_walker.cpp | 348 +++++++++++++++++- sw/kernel/include/vx_raytrace.h | 4 +- tests/raytracing/rt_smoke_ahs_geom/kernel.cpp | 3 +- 6 files changed, 387 insertions(+), 21 deletions(-) diff --git a/VX_types.toml b/VX_types.toml index 495afbf7cc..dc0f368b2b 100644 --- a/VX_types.toml +++ b/VX_types.toml @@ -429,7 +429,7 @@ VX_RT_HIT_BARY_U = 11 VX_RT_HIT_BARY_V = 12 VX_RT_HIT_PRIMITIVE_ID = 13 VX_RT_HIT_INSTANCE_ID = 14 -VX_RT_HIT_GEOMETRY_INDEX = 15 +VX_RT_HIT_GEOMETRY_INDEX = 15 # VX_RT_HIT_GEOMETRY_MASK bits: gl_GeometryIndexEXT; bit 31: VX_RT_HIT_BACK_FACING VX_RT_HIT_INSTANCE_CUSTOM = 16 VX_RT_OBJECT_RAY_ORIGIN = 17 # object_ray.origin.{x,y,z} (17..19) VX_RT_OBJECT_RAY_DIRECTION = 20 # object_ray.direction.{x,y,z} (20..22) @@ -477,6 +477,14 @@ VX_RT_FLAG_SKIP_AABBS = 0x200 VX_RT_FLAG_ENABLE_CHS = 0x400 VX_RT_FLAG_ENABLE_MISS = 0x800 +[rtu_hit_bits] +# The HIT_GEOMETRY_INDEX word: the leaf's geometry index in the low 28 bits (the +# Vulkan BVH layout keeps flags above them, which the RTU drops), and +# BACK_FACING set when a triangle hit lies on its back face after the instance's +# FLIP_FACING, i.e. gl_HitKindEXT is BACK_FACING. +VX_RT_HIT_GEOMETRY_MASK = 0x0fffffff +VX_RT_HIT_BACK_FACING = 0x80000000 + [rtu_cb_actions] # vx_rt_cb_ret action codes VX_RT_CB_ACCEPT = 1 diff --git a/hw/rtl/rtu/VX_rtu_scheduler.sv b/hw/rtl/rtu/VX_rtu_scheduler.sv index b8f3da9ce7..84af34c433 100644 --- a/hw/rtl/rtu/VX_rtu_scheduler.sv +++ b/hw/rtl/rtu/VX_rtu_scheduler.sv @@ -184,6 +184,12 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( logic [1:0] setup_axis; logic [2:0][31:0] inv_d; logic [31:0] best_t; + // identity of an opaque hit committed in this walk: an exact t tie + // goes to the lower (instance, geometry, primitive) + logic best_kv; + logic [31:0] best_ki; + logic [27:0] best_kg; + logic [31:0] best_kp; logic [31:0] yld_t; // staged candidate's t (compare copy) logic [31:0] yld_ki; // staged candidate's key: instance id logic [31:0] yld_ko; // ... and record offset @@ -626,6 +632,16 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( && (inst_culldis || !(!eff_back && cull_front)) && !cls_cull; wire tri_committable = tri_pass && (trit_q < word_q.best_t); + wire [31:0] tri_ki = word_q.in_blas ? word_q.inst_id : 32'd0; + wire [27:0] tri_kg = 28'(word_q.geom_r & `VX_RT_HIT_GEOMETRY_MASK); + wire [31:0] tri_kp = word_q.prim_base + word_q.tri_i; + wire tri_t_eq = (trit_q == word_q.best_t) + || ((trit_q[30:0] == 31'd0) && (word_q.best_t[30:0] == 31'd0)); + wire tri_tie = tri_pass && word_q.best_kv && tri_t_eq + && ({tri_ki, tri_kg, tri_kp} < {word_q.best_ki, word_q.best_kg, word_q.best_kp}); + // the reported geometry word: the leaf's index plus the hit's facing + wire [31:0] tri_geom = (word_q.geom_r & `VX_RT_HIT_GEOMETRY_MASK) + | (eff_back ? `VX_RT_HIT_BACK_FACING : 32'd0); // BLAS traversal runs the object-space ray wire [2:0][31:0] walk_ro = word_q.in_blas ? word_q.obj_o : ray_q.origin; @@ -1481,7 +1497,7 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( cf_din_r.prim = word_x.prim_base; cf_din_r.inst = word_x.in_blas ? word_x.inst_id : 32'd0; cf_din_r.cust = word_x.in_blas ? word_x.inst_cust : 32'd0; - cf_din_r.geom = word_x.geom_r; + cf_din_r.geom = word_x.geom_r & `VX_RT_HIT_GEOMETRY_MASK; cf_din_r.sbt = word_x.proc_sbt; cf_din_r.objv = word_x.in_blas; cf_din_r.obj = {word_x.obj_d, word_x.obj_o}; @@ -1546,7 +1562,7 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( CS_TRI_WAIT: begin // woken by the tri PE result (held in its result RAM, so a retry // on a full commit queue re-reads the same result) - if (tri_committable && tri_opaque) begin + if ((tri_committable || tri_tie) && tri_opaque) begin cf_din_r.kind = CK_HIT; cf_din_r.t = trit_q; cf_din_r.u = triu_q; @@ -1554,13 +1570,17 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( cf_din_r.prim = word_x.prim_base + word_x.tri_i; cf_din_r.inst = word_x.in_blas ? word_x.inst_id : 32'd0; cf_din_r.cust = word_x.in_blas ? word_x.inst_cust : 32'd0; - cf_din_r.geom = word_x.geom_r; + cf_din_r.geom = tri_geom; if (cf_full) begin wake_self = 1'b1; end else begin cf_push_r = 1'b1; exec_hit_set = 1'b1; - word_n.best_t = trit_q; + word_n.best_t = trit_q; + word_n.best_kv = 1'b1; + word_n.best_ki = tri_ki; + word_n.best_kg = tri_kg; + word_n.best_kp = tri_kp; // a closer opaque hit occludes a farther candidate if (yld_q[sel_q] && (word_x.yld_t >= trit_q)) begin exec_yld_clr = 1'b1; @@ -1588,7 +1608,7 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( cf_din_r.prim = word_x.prim_base + word_x.tri_i; cf_din_r.inst = word_x.in_blas ? word_x.inst_id : 32'd0; cf_din_r.cust = word_x.in_blas ? word_x.inst_cust : 32'd0; - cf_din_r.geom = word_x.geom_r; + cf_din_r.geom = tri_geom; cf_din_r.sbt = cls_sbt; cf_din_r.objv = word_x.in_blas; cf_din_r.obj = {word_x.obj_d, word_x.obj_o}; diff --git a/sim/simx/rtu/rtu_bvh.h b/sim/simx/rtu/rtu_bvh.h index 6b0b7a5e9c..d2394c4054 100644 --- a/sim/simx/rtu/rtu_bvh.h +++ b/sim/simx/rtu/rtu_bvh.h @@ -198,8 +198,17 @@ inline void decode_bvh6_node(const VxBvh6InternalNode* n, uint32_t count, // uint32 kind : bits 0..7 = kVxBvhKindLeafTri/Inst/Proc // bits 8..15 = prim_count // uint32 geometry_index : Vulkan gl_GeometryIndexEXT for this leaf -// uint32 flags : bit 0 = OPAQUE (all prims), bit 1 = forced -// non-opaque, bits 8..15 = SBT_IDX +// uint32 flags : LeafProc: bit 0 = OPAQUE (all prims), bit 1 = +// forced non-opaque, bits 8..15 = SBT_IDX. +// LeafTri / LeafInst: the leaf's place in the +// source BVH's visit-order table, which settles +// an exact-t tie between opaque hits the way a +// first-visited-wins traversal of the source BVH +// does (rtu_walker.cpp). LeafTri: parent << 1 | +// side in its BLAS's table. LeafInst: the +// instance's visit rank in the TLAS's table, with +// geometry_index = the BLAS's table and prim_base +// = the TLAS's table (scene offsets; 0 = none). // uint32 prim_base : gl_PrimitiveID of this leaf's first // primitive; the walker reports // prim_base + within-leaf index so a diff --git a/sim/simx/rtu/rtu_walker.cpp b/sim/simx/rtu/rtu_walker.cpp index 1c76284fb6..ec4dda580f 100644 --- a/sim/simx/rtu/rtu_walker.cpp +++ b/sim/simx/rtu/rtu_walker.cpp @@ -14,6 +14,7 @@ #include "rtu_walker.h" #include +#include #include #include #include @@ -91,6 +92,12 @@ struct WalkCtx { uint32_t best_instance; uint32_t best_custom; // VK_INSTANCE_CUSTOM_INDEX of the committed instance uint32_t best_geom; // gl_GeometryIndexEXT of the committed leaf + bool best_kv; // an opaque hit committed in this walk + uint64_t best_order; // (instance order << 32) | triangle order of it + uint32_t best_blas_tab; // its BLAS's visit-order table (0: none) + float best_tri_box[6]; // its triangle's vertex min/max + uint32_t tlas_tab; // the TLAS's visit-order table (0: none) + float world_o[3], world_d[3]; bool any_hit; bool yield_pending; float yield_t, yield_u, yield_v; @@ -129,6 +136,291 @@ inline bool cand_takes(const WalkCtx& ctx, float t, uint64_t key) { return t < ctx.yield_t || (t == ctx.yield_t && key < ctx.yield_key); } +// The geometry word's back-facing bit, after the instance's FLIP_FACING: +// what gl_HitKindEXT reports. +inline uint32_t hit_facing_bit(bool back_facing, uint32_t inst_flags) { + if (inst_flags & kRtuInstanceFlagTriFlip) back_facing = !back_facing; + return back_facing ? VX_RT_HIT_BACK_FACING : 0u; +} + +inline uint64_t hit_order(uint32_t inst_order, uint32_t tri_order) { + return (uint64_t(inst_order) << 32) | tri_order; +} + +inline uint32_t scene_u32(SceneView& sv, uint32_t off) { + uint32_t v = 0; + read_scene_bytes(sv, off, sizeof(v), reinterpret_cast(&v)); + return v; +} + +// A source-BVH child box tested exactly as the reference traversal tests it +// (F32, (bound - origin) * 1/dir, 1/0 -> FLT_MAX): `lo` is its entry distance +// (the child-order key), and the box is visited iff it passes against the +// committed t at the time. +struct SrcBox { float lo, hi; }; + +SrcBox src_box(const float box[6], const float o[3], const float inv[3]) { + if (std::isnan(box[0])) return { INFINITY, -INFINITY }; + float b0[3], b1[3]; + for (int i = 0; i < 3; ++i) { + b0[i] = (box[i] - o[i]) * inv[i]; + b1[i] = (box[3 + i] - o[i]) * inv[i]; + } + const float lo = std::fmax(std::fmax(std::fmin(b0[0], b1[0]), + std::fmin(b0[1], b1[1])), + std::fmin(b0[2], b1[2])); + const float hi = std::fmin(std::fmin(std::fmax(b0[0], b1[0]), + std::fmax(b0[1], b1[1])), + std::fmax(b0[2], b1[2])); + return { lo, hi }; +} + +inline bool src_box_hit(const SrcBox& b) { return b.hi >= std::fmax(0.f, b.lo); } +inline float src_box_key(const SrcBox& b) { return src_box_hit(b) ? b.lo : INFINITY; } +inline bool src_box_passes(const SrcBox& b, float tmax) { + return src_box_hit(b) && b.lo < tmax; +} + +// The reference's ray in a table's space. +struct SrcRay { float o[3], d[3], inv[3]; }; + +SrcRay src_ray(const float o[3], const float d[3]) { + SrcRay r; + for (int i = 0; i < 3; ++i) { + r.o[i] = o[i]; r.d[i] = d[i]; + r.inv[i] = (d[i] == 0.f) ? FLT_MAX : 1.0f / d[i]; + } + return r; +} + +// Visit-order tables: the source BVH's binary tree, which the reference walks +// depth-first, nearer child box first, child 0 on equal entry distances, each +// box tested against the hit committed when its parent is visited. +// TLAS table at `tab` (leaves = instances, indexed by visit rank): +// { n_leaves, n_nodes, nodes_off, leaf_stride } +// leaf[i] at tab + 16 + i*leaf_stride: { parent << 1 | side, _, _, _, +// world->object 3x4 } +// node[j] at tab + nodes_off + j*64: { child box 0 min/max, child box 1 +// min/max, parent << 1 | side (~0: root), depth } +// BLAS table at `tab` (leaves = triangles, each naming its parent << 1 | side +// in its leaf header; a triangle's box is its vertices' min/max): +// { n_nodes, _, nodes_off, 32 } +// node[j] at tab + nodes_off + j*32: { own box min/max, parent << 1 | side +// (~0: root), depth } +constexpr uint32_t kSrcRoot = 0xffffffffu; +constexpr uint32_t kSrcTabHdr = 16; +constexpr uint32_t kSrcLeafWto = 16; +constexpr uint32_t kSrcTlasNodeBytes = 64; +constexpr uint32_t kSrcTlasNodeParent = 48; +constexpr uint32_t kSrcTlasNodeDepth = 52; +constexpr uint32_t kSrcBlasNodeBytes = 32; +constexpr uint32_t kSrcBlasNodeParent = 24; +constexpr uint32_t kSrcBlasNodeDepth = 28; + +// One leaf's climb towards the root. `box` is the box of the child the climb +// is at (tested when its parent was visited); each box the climb leaves below +// the lowest common ancestor's child was tested after the other leaf's +// subtree, had that come first, so the climb notes whether one of them fails +// against `tmax` (the other hit's t). +struct SrcClimb { + uint32_t ps; // parent << 1 | side of the current child + uint32_t depth; // depth of node(ps) + float box[6]; + bool culled; +}; + +struct SrcTab { + SceneView& sv; + uint32_t tab, nodes_off, node_bytes, parent_off, depth_off; + bool tlas; + uint32_t node(uint32_t ps) const { return nodes_off + (ps >> 1) * node_bytes; } + uint32_t parent(uint32_t ps) const { return scene_u32(sv, node(ps) + parent_off); } + uint32_t depth(uint32_t ps) const { return scene_u32(sv, node(ps) + depth_off); } + // Box of the child at ps: a TLAS node holds its children's boxes, a BLAS + // node its own. + void child_box(uint32_t ps, uint32_t below, float box[6]) const { + if (tlas) + read_scene_bytes(sv, node(ps) + 24 * (ps & 1u), 24, reinterpret_cast(box)); + else + read_scene_bytes(sv, node(below), 24, reinterpret_cast(box)); + } +}; + +SrcTab src_tab(SceneView& sv, uint32_t tab, bool tlas) { + return { sv, tab, tab + scene_u32(sv, tab + 8), + tlas ? kSrcTlasNodeBytes : kSrcBlasNodeBytes, + tlas ? kSrcTlasNodeParent : kSrcBlasNodeParent, + tlas ? kSrcTlasNodeDepth : kSrcBlasNodeDepth, tlas }; +} + +void src_up(const SrcTab& t, SrcClimb& c, const SrcRay& r, float tmax) { + if (!src_box_passes(src_box(c.box, r.o, r.inv), tmax)) c.culled = true; + const uint32_t below = c.ps; + c.ps = t.parent(c.ps); + --c.depth; + t.child_box(c.ps, below, c.box); +} + +// Climb both leaves to their lowest common ancestor; true when a's side is +// visited first there. a.culled / b.culled report the boxes left on the way. +bool src_lca(const SrcTab& t, SrcClimb& a, SrcClimb& b, const SrcRay& r, + float ta, float tb) { + while (!t.sv.miss && a.depth > b.depth) src_up(t, a, r, tb); + while (!t.sv.miss && b.depth > a.depth) src_up(t, b, r, ta); + while (!t.sv.miss && (a.ps >> 1) != (b.ps >> 1)) { + src_up(t, a, r, tb); + src_up(t, b, r, ta); + } + const float ka = src_box_key(src_box(a.box, r.o, r.inv)); + const float kb = src_box_key(src_box(b.box, r.o, r.inv)); + const float d0 = (a.ps & 1u) ? kb : ka; + const float d1 = (a.ps & 1u) ? ka : kb; + const uint32_t first = (d1 < d0) ? 1u : 0u; + return (a.ps & 1u) == first; +} + +SrcClimb src_blas_leaf(const SrcTab& t, uint32_t ps, const float box[6]) { + SrcClimb c; + c.ps = ps; + c.depth = (ps == kSrcRoot) ? 0 : t.depth(ps) + 1; + std::memcpy(c.box, box, sizeof(c.box)); + c.culled = false; + return c; +} + +// Whether a box on the leaf's whole BLAS path fails against tmax: every box +// below the BLAS root is tested after the instance is entered. +bool src_blas_path_culled(const SrcTab& t, uint32_t ps, const float box[6], + const SrcRay& r, float tmax) { + SrcClimb c = src_blas_leaf(t, ps, box); + while (!t.sv.miss && c.ps != kSrcRoot) src_up(t, c, r, tmax); + return c.culled; +} + +// The reference's object-space ray for TLAS leaf `inst`: its world->object +// matrix applied as translation + x + y + z, in that order, in F32. +SrcRay src_object_ray(SceneView& sv, uint32_t tlas_tab, uint32_t inst, + const float wo[3], const float wd[3]) { + const uint32_t leaf_stride = scene_u32(sv, tlas_tab + 12); + float m[12]; + read_scene_bytes(sv, tlas_tab + kSrcTabHdr + inst * leaf_stride + kSrcLeafWto, + sizeof(m), reinterpret_cast(m)); + float o[3], d[3]; + for (int i = 0; i < 3; ++i) { + float ro = m[i * 4 + 3], rd = 0.f; + for (int j = 0; j < 3; ++j) { + ro = ro + wo[j] * m[i * 4 + j]; + rd = j ? rd + wd[j] * m[i * 4 + j] : wd[j] * m[i * 4 + j]; + } + o[i] = ro; d[i] = rd; + } + return src_ray(o, d); +} + +inline void tri_box(const float* v, float box[6]) { + for (int a = 0; a < 3; ++a) { + box[a] = std::fmin(v[a], std::fmin(v[3 + a], v[6 + a])); + box[3 + a] = std::fmax(v[a], std::fmax(v[3 + a], v[6 + a])); + } +} + +// An opaque triangle hit as the source traversal sees it. +struct SrcHit { + float t; + uint32_t inst; // TLAS visit rank + uint32_t ps; // parent << 1 | side in its BLAS table + uint32_t blas_tab; + float box[6]; +}; + +// Which of two opaque hits within a few ulps of each other the source +// traversal keeps: it commits a hit only nearer than the committed one, and +// tests each box against the committed t, so the first-reached of two equal-t +// hits wins, and a nearer hit reached second is lost when a box on its way +// (below where the two paths part) enters at or past the first one's t. +bool src_keeps_a(SceneView& sv, uint32_t tlas_tab, const float wo[3], + const float wd[3], const SrcHit& a, const SrcHit& b) { + bool a_first; + bool a_culled, b_culled; + if (a.inst == b.inst) { + const SrcTab t = src_tab(sv, a.blas_tab, false); + const SrcRay r = src_object_ray(sv, tlas_tab, a.inst, wo, wd); + if (a.ps == kSrcRoot || b.ps == kSrcRoot) return a.t <= b.t; + SrcClimb ca = src_blas_leaf(t, a.ps, a.box); + SrcClimb cb = src_blas_leaf(t, b.ps, b.box); + a_first = src_lca(t, ca, cb, r, a.t, b.t); + a_culled = ca.culled; b_culled = cb.culled; + } else { + const SrcTab tt = src_tab(sv, tlas_tab, true); + const SrcRay rw = src_ray(wo, wd); + const uint32_t stride = scene_u32(sv, tlas_tab + 12); + auto tlas_leaf = [&](uint32_t inst) { + SrcClimb c; + c.ps = scene_u32(sv, tlas_tab + kSrcTabHdr + inst * stride); + c.depth = (c.ps == kSrcRoot) ? 0 : tt.depth(c.ps) + 1; + tt.child_box(c.ps, 0, c.box); + c.culled = false; + return c; + }; + SrcClimb ca = tlas_leaf(a.inst), cb = tlas_leaf(b.inst); + if (sv.miss) return false; + a_first = src_lca(tt, ca, cb, rw, a.t, b.t); + a_culled = ca.culled + || src_blas_path_culled(src_tab(sv, a.blas_tab, false), a.ps, a.box, + src_object_ray(sv, tlas_tab, a.inst, wo, wd), b.t); + b_culled = cb.culled + || src_blas_path_culled(src_tab(sv, b.blas_tab, false), b.ps, b.box, + src_object_ray(sv, tlas_tab, b.inst, wo, wd), a.t); + } + if (a.t == b.t) return a_first; + if (a.t < b.t) return !(!a_first && a_culled); + return a_first && b_culled; +} + +// A small relative bound past the committed t: within it, a box's slab entry +// (a few ulps of rounding a triangle's watertight t does not carry) can land +// on either side of the committed t, so which of two such hits the source +// traversal keeps depends on its visit order. Beyond it the nearer hit wins. +inline float near_t(float t) { return t + std::fabs(t) * 0x1p-19f; } + +// The bound a child box's entry is culled against: kept open by near_t, so +// every hit the source traversal might keep over the committed one is reached. +inline float box_cull_t(const WalkCtx& ctx) { + return ctx.best_kv ? near_t(ctx.best_t) : ctx.best_t; +} + +// Whether an opaque hit at t replaces the committed one. Within near_t of an +// opaque hit committed in this walk, the scene's visit-order tables settle it +// as the source traversal would; without tables, an exact tie goes to the +// lowest (instance, geometry, primitive) and otherwise the nearer hit wins. +bool commit_takes(SceneView& sv, const WalkCtx& ctx, float t, uint64_t order, + uint32_t blas_tab, const float* tri, uint32_t instance_id, + uint32_t geom, uint32_t prim) { + const bool near = ctx.best_kv + && (t == ctx.best_t + || (t < ctx.best_t && near_t(t) >= ctx.best_t) + || (t > ctx.best_t && t <= near_t(ctx.best_t))); + if (near && ctx.tlas_tab && tri && blas_tab && ctx.best_blas_tab + && order != ctx.best_order) { + SrcHit a, b; + a.t = t; a.inst = uint32_t(order >> 32); a.ps = uint32_t(order); + a.blas_tab = blas_tab; tri_box(tri, a.box); + b.t = ctx.best_t; b.inst = uint32_t(ctx.best_order >> 32); + b.ps = uint32_t(ctx.best_order); b.blas_tab = ctx.best_blas_tab; + std::memcpy(b.box, ctx.best_tri_box, sizeof(b.box)); + const bool keeps_new = src_keeps_a(sv, ctx.tlas_tab, ctx.world_o, + ctx.world_d, a, b); + return !sv.miss && keeps_new; + } + if (t < ctx.best_t) return true; + if (!(ctx.best_kv && t == ctx.best_t)) return false; + const uint32_t best_geom = ctx.best_geom & VX_RT_HIT_GEOMETRY_MASK; + geom &= VX_RT_HIT_GEOMETRY_MASK; + if (instance_id != ctx.best_instance) return instance_id < ctx.best_instance; + if (geom != best_geom) return geom < best_geom; + return prim < ctx.best_prim; +} + // Depth-first walker for one BVH sub-tree under the supplied (object-space) // ray. Recurses on LeafInst so each instance's BLAS gets walked with its // transformed ray. ctx accumulates hits/yields across the whole call tree. @@ -143,6 +435,7 @@ void walk_bvh4_subtree(SceneView& sv, const float ro[3], const float rd[3], uint32_t root_off, uint32_t instance_id, uint32_t custom_id, uint32_t inst_flags, + uint32_t inst_order, uint32_t blas_tab, WalkCtx& ctx, PerfStats& perf) { auto visit_leaf_tri = [&](uint32_t leaf_off, uint32_t count) { uint8_t hdr_buf[kVxBvhLeafHeaderBytes]; @@ -152,6 +445,7 @@ void walk_bvh4_subtree(SceneView& sv, reinterpret_cast(hdr_buf); uint32_t leaf_geom = hdr->geometry_index; uint32_t leaf_prim_base = hdr->prim_base; // Vulkan gl_PrimitiveID base + uint32_t leaf_order = hdr->flags; // LeafTri: source parent/side uint32_t tris_off = leaf_off + kVxBvhLeafHeaderBytes; for (uint32_t i = 0; i < count; ++i) { if (ctx.terminated) return; @@ -167,24 +461,34 @@ void walk_bvh4_subtree(SceneView& sv, float t_hit = 0.f, u = 0.f, v = 0.f; bool back_facing = false; ++perf.bvh_tri_tests; - if (!ray_triangle(ro, rd, &tri[0], &tri[3], &tri[6], + const bool tri_hit = ray_triangle(ro, rd, &tri[0], &tri[3], &tri[6], ctx.tmin, ctx.tmax, - t_hit, u, v, back_facing)) { - continue; - } + t_hit, u, v, back_facing); + if (!tri_hit) continue; TriClassify cls = classify_tri_hit(ctx.ray_flags, tri_flags, inst_flags, back_facing); + const uint32_t hit_geom = (leaf_geom & VX_RT_HIT_GEOMETRY_MASK) + | hit_facing_bit(back_facing, inst_flags); if (cls.action == TriAction::Ignore) continue; if (cls.action == TriAction::Commit) { - if (t_hit < ctx.best_t) { + const uint64_t order = hit_order(inst_order, leaf_order + i); + const bool takes = commit_takes(sv, ctx, t_hit, order, blas_tab, tri, + instance_id, leaf_geom, + leaf_prim_base + i); + if (sv.miss) return; + if (takes) { ctx.best_t = t_hit; ctx.best_u = u; ctx.best_v = v; + ctx.best_order = order; + ctx.best_blas_tab = blas_tab; + tri_box(tri, ctx.best_tri_box); ctx.best_prim = leaf_prim_base + i; ctx.best_instance = instance_id; ctx.best_custom = custom_id; - ctx.best_geom = leaf_geom; + ctx.best_geom = hit_geom; ctx.any_hit = true; + ctx.best_kv = true; vcopy3(ctx.best_obj_o, ro); // object-space ray of this BLAS vcopy3(ctx.best_obj_d, rd); if (ctx.yield_pending && ctx.yield_t >= ctx.best_t) { @@ -205,7 +509,7 @@ void walk_bvh4_subtree(SceneView& sv, ctx.yield_prim = leaf_prim_base + i; ctx.yield_instance = instance_id; ctx.yield_custom = custom_id; - ctx.yield_geom = leaf_geom; + ctx.yield_geom = hit_geom; ctx.yield_sbt = cls.yield_sbt_idx; ctx.yield_cb_type = cls.yield_cb_type; vcopy3(ctx.yield_obj_o, ro); // object-space ray for AHS/IS @@ -253,7 +557,7 @@ void walk_bvh4_subtree(SceneView& sv, ctx.yield_prim = hdr->prim_base + i; // gl_PrimitiveID, as for LEAF_TRI ctx.yield_instance = instance_id; ctx.yield_custom = custom_id; - ctx.yield_geom = hdr->geometry_index; + ctx.yield_geom = hdr->geometry_index & VX_RT_HIT_GEOMETRY_MASK; ctx.yield_sbt = leaf_sbt; ctx.yield_cb_type = VX_RT_CB_TYPE_PROC; vcopy3(ctx.yield_obj_o, ro); @@ -263,6 +567,15 @@ void walk_bvh4_subtree(SceneView& sv, }; auto visit_leaf_inst = [&](uint32_t leaf_off, uint32_t count) { + uint8_t hdr_buf[kVxBvhLeafHeaderBytes]; + read_scene_bytes(sv, leaf_off, sizeof(hdr_buf), hdr_buf); + if (sv.miss) return; + // LeafInst header: geometry_index = the BLAS's visit-order table, + // flags = the instance's visit order, prim_base = the TLAS's table. + const VxBvhLeafHeader* ihdr = + reinterpret_cast(hdr_buf); + const uint32_t leaf_order = ihdr->flags; + ctx.tlas_tab = ihdr->prim_base; uint32_t insts_off = leaf_off + kVxBvhLeafHeaderBytes; for (uint32_t i = 0; i < count; ++i) { uint8_t inst_buf[kVxBvhInstanceStride]; @@ -285,6 +598,7 @@ void walk_bvh4_subtree(SceneView& sv, inst->blas_root_byte_offset, inst->instance_id, inst->custom_id, inst_flags2, + leaf_order + i, ihdr->geometry_index, ctx, perf); if (sv.miss) return; } @@ -369,7 +683,7 @@ void walk_bvh4_subtree(SceneView& sv, float t_near = 0.f; ++perf.bvh_box_tests; if (!ray_aabb_intersect(ro, rd, mn, mx, - ctx.tmin, ctx.best_t, t_near)) { + ctx.tmin, box_cull_t(ctx), t_near)) { continue; } hits[hit_count++] = { child_off, t_near }; @@ -491,6 +805,11 @@ WalkCtx init_ctx(const RtuReq& req, uint32_t t, ctx.best_prim = 0; ctx.best_instance = 0; ctx.best_custom = 0; ctx.best_geom = 0; ctx.any_hit = false; + ctx.best_kv = false; + ctx.best_order = 0; + ctx.best_blas_tab = 0; + ctx.tlas_tab = 0; + vcopy3(ctx.world_o, ro); vcopy3(ctx.world_d, rd); ctx.yield_pending = false; ctx.yield_t = ctx.tmax; ctx.yield_u = 0.f; ctx.yield_v = 0.f; ctx.yield_prim = 0; ctx.yield_sbt = 0; @@ -646,15 +965,21 @@ WalkResult FlatWalker::walk_lane(const RtuReq& req, uint32_t t, SceneView& sv, TriClassify cls = classify_tri_hit(ctx.ray_flags, tri_flags, cur_inst_flags, back_facing); + const uint32_t hit_geom = hit_facing_bit(back_facing, cur_inst_flags); if (cls.action == TriAction::Ignore) continue; if (cls.action == TriAction::Commit) { - if (t_hit < ctx.best_t) { + const bool takes = commit_takes(sv, ctx, t_hit, hit_order(inst_idx, i), + 0, nullptr, inst_idx, 0, i); + if (takes) { ctx.best_t = t_hit; ctx.best_u = u; ctx.best_v = v; + ctx.best_order = hit_order(inst_idx, i); ctx.best_prim = i; ctx.best_instance = inst_idx; ctx.best_custom = cur_custom; + ctx.best_geom = hit_geom; ctx.any_hit = true; + ctx.best_kv = true; vcopy3(ctx.best_obj_o, ray_o); // this instance's object ray vcopy3(ctx.best_obj_d, ray_d); if (ctx.yield_pending && ctx.yield_t >= ctx.best_t) { @@ -679,6 +1004,7 @@ WalkResult FlatWalker::walk_lane(const RtuReq& req, uint32_t t, SceneView& sv, ctx.yield_cb_type = cls.yield_cb_type; ctx.yield_instance = inst_idx; ctx.yield_custom = cur_custom; + ctx.yield_geom = hit_geom; vcopy3(ctx.yield_obj_o, ray_o); // object ray for AHS/IS vcopy3(ctx.yield_obj_d, ray_d); } @@ -709,7 +1035,7 @@ WalkResult Bvh4Walker::walk_lane(const RtuReq& req, uint32_t t, SceneView& sv, WalkCtx ctx = init_ctx(req, t, ro, rd, l); // Top-level (non-instanced) triangles carry no instance flags. - walk_bvh4_subtree(sv, ro, rd, root_off, 0, 0, 0, ctx, perf); + walk_bvh4_subtree(sv, ro, rd, root_off, 0, 0, 0, 0, 0, ctx, perf); if (sv.miss) return {true, false}; return {false, emit_lane_result(req, l, t, ctx)}; diff --git a/sw/kernel/include/vx_raytrace.h b/sw/kernel/include/vx_raytrace.h index 6b415edec4..de4d87ec47 100644 --- a/sw/kernel/include/vx_raytrace.h +++ b/sw/kernel/include/vx_raytrace.h @@ -119,6 +119,7 @@ typedef struct { uint32_t geometry_index; uint32_t instance_id; uint32_t instance_custom; // gl_InstanceCustomIndexEXT (VK_INSTANCE_CUSTOM_INDEX) + uint32_t back_facing; // gl_HitKindEXT == BACK_FACING } vx_hit_t; // The struct field order (memory layout) is intentionally NOT the RTU register- @@ -225,8 +226,9 @@ uint32_t vx_rt_wait(uint32_t handle, vx_hit_t* hit) { hit->v = hv; hit->primitive_id = hp; hit->instance_id = hi; - hit->geometry_index = hg; + hit->geometry_index = hg & VX_RT_HIT_GEOMETRY_MASK; hit->instance_custom = hc; + hit->back_facing = (hg & VX_RT_HIT_BACK_FACING) != 0; return status; } diff --git a/tests/raytracing/rt_smoke_ahs_geom/kernel.cpp b/tests/raytracing/rt_smoke_ahs_geom/kernel.cpp index f3bf38da34..cd33ff2f2b 100644 --- a/tests/raytracing/rt_smoke_ahs_geom/kernel.cpp +++ b/tests/raytracing/rt_smoke_ahs_geom/kernel.cpp @@ -44,7 +44,8 @@ __kernel void kernel_main(kernel_arg_t* arg) { // Read the candidate geometry/instance attributes and the capture buffer // pointer (staged as the trace payload) from the register window, stash. uint32_t cand_ptr = vx_rt_get_attr(VX_RT_PAYLOAD_PTR_LO, sts); - uint32_t cand_geom = vx_rt_get_attr(VX_RT_HIT_GEOMETRY_INDEX, sts); + uint32_t cand_geom = vx_rt_get_attr(VX_RT_HIT_GEOMETRY_INDEX, sts) + & VX_RT_HIT_GEOMETRY_MASK; uint32_t cand_inst = vx_rt_get_attr(VX_RT_HIT_INSTANCE_ID, sts); uint32_t* cand = (uint32_t*)(uintptr_t)cand_ptr; cand[0] = cand_geom; // cand->cand_geometry From 148baa21055f540159749cc3eb4f1153d5755153 Mon Sep 17 00:00:00 2001 From: Blaise Tine Date: Sat, 3 Oct 2026 02:50:55 -0700 Subject: [PATCH 11/31] rtu: SimX ray log and an RTL-vs-SimX ray replay harness VX_RTU_RAYLOG= makes the SimX RTU append every traced ray (scene root, flags, cull mask, origin/dir/tmin/tmax bits) with its terminal result (status, t/u/v bits, primitive, geometry index incl. facing bit, instance id/custom), plus every scene line a completed walk read, at its device address. Lines are re-emitted under a new epoch if the device rewrites them. Rays whose result was decided by an any-hit or intersection shader are flagged non-replayable. VX_RTU_RAYLOG_MAX / VX_RTU_RAYLOG_EVERY bound the log. tests/raytracing/rt_replay restores the logged lines at their original addresses (vx_buffer_reserve before any other allocation), re-traces the rays one per thread in warp-uniform (scene, flags, cull) groups, and compares each result bit-exactly with the log, classifying mismatches (hit/miss, near tie, same-primitive precision, terminate-on-first-hit, ...). Ray selection by index range, list file (its own mismatch output is accepted) or stride. Co-Authored-By: Claude Opus 5.5 --- sim/simx/Makefile | 1 + sim/simx/rtu/rtu_core.cpp | 5 + sim/simx/rtu/rtu_raylog.cpp | 187 +++++++++++ sim/simx/rtu/rtu_raylog.h | 112 +++++++ tests/raytracing/rt_replay/Makefile | 23 ++ tests/raytracing/rt_replay/common.h | 81 +++++ tests/raytracing/rt_replay/kernel.cpp | 50 +++ tests/raytracing/rt_replay/main.cpp | 430 ++++++++++++++++++++++++++ 8 files changed, 889 insertions(+) create mode 100644 sim/simx/rtu/rtu_raylog.cpp create mode 100644 sim/simx/rtu/rtu_raylog.h create mode 100644 tests/raytracing/rt_replay/Makefile create mode 100644 tests/raytracing/rt_replay/common.h create mode 100644 tests/raytracing/rt_replay/kernel.cpp create mode 100644 tests/raytracing/rt_replay/main.cpp diff --git a/sim/simx/Makefile b/sim/simx/Makefile index 1198844b29..63581b4b8f 100644 --- a/sim/simx/Makefile +++ b/sim/simx/Makefile @@ -100,6 +100,7 @@ ifneq ($(filter -DVX_CFG_EXT_RTU_ENABLE, $(XCONFIGS)),) SRCS += $(SRC_DIR)/rtu/rtu_classifier.cpp SRCS += $(SRC_DIR)/rtu/rtu_walker.cpp SRCS += $(SRC_DIR)/rtu/rtu_memory.cpp + SRCS += $(SRC_DIR)/rtu/rtu_raylog.cpp endif # SST integration: build libvortex.so as the SST element library and diff --git a/sim/simx/rtu/rtu_core.cpp b/sim/simx/rtu/rtu_core.cpp index c3dce11781..4f75f5b1a6 100644 --- a/sim/simx/rtu/rtu_core.cpp +++ b/sim/simx/rtu/rtu_core.cpp @@ -26,6 +26,7 @@ #include "rtu_isect.h" // BoxPe / TriPe pipeline depths #include "rtu_walker.h" // FlatWalker / Bvh4Walker #include "rtu_memory.h" // MemoryEngine +#include "rtu_raylog.h" #include "socket.h" #include "constants.h" #include "debug.h" @@ -356,6 +357,7 @@ class RtuCore::Impl { if (s.req.dir_z[first_active] < 0.f) sig |= 0x4; s.coh_signature = sig; } + if (raylog::enabled()) raylog::on_accept(this, idx); ch.pop(); ++perf_stats_.rays_issued; DT(3, "rtu-core accept: tag=" << s.req.tag << ", slot=" << idx); @@ -583,6 +585,7 @@ class RtuCore::Impl { cx.next_state = CtxState::REQ; } else { s.lanes[cx.lane] = result; // carries cb_pending if the ray yielded + if (raylog::enabled()) raylog::on_walk_done(cx.lines); cx.next_state = CtxState::DONE; } cx.state = (cx.fsm_states || lat) ? CtxState::PE : cx.next_state; @@ -655,6 +658,7 @@ class RtuCore::Impl { const LaneState& l = s.lanes[t]; if (!l.active || !l.cb_pending) continue; any_cb = true; + if (raylog::enabled()) raylog::on_callback(this, i, t, l.cb_type); QueueEntry e{i, s.req.warp_id, uint8_t(t), l.sbt_idx, l.cb_type, l.cand_t, l.cand_u, l.cand_v, l.cand_prim, @@ -710,6 +714,7 @@ class RtuCore::Impl { // so it stays live until the WAIT that consumes the record calls // free_slot(). Until then it sits in EMITTED and is not re-sent. rsp.slot_idx = i; + if (raylog::enabled()) raylog::on_terminal(this, i, s.req, s.lanes); port.send(rsp); DT(3, "rtu-core complete: tag=" << s.req.tag << ", slot=" << i); s.state = SlotState::EMITTED; diff --git a/sim/simx/rtu/rtu_raylog.cpp b/sim/simx/rtu/rtu_raylog.cpp new file mode 100644 index 0000000000..e42e23d36f --- /dev/null +++ b/sim/simx/rtu/rtu_raylog.cpp @@ -0,0 +1,187 @@ +// Copyright © 2019-2023 +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +#include "rtu_raylog.h" +#include +#include +#include +#include +#include +#include + +namespace vortex { namespace rtu { namespace raylog { + +namespace { + +template uint32_t bits(T v) { + static_assert(sizeof(T) == 4, "32-bit field"); + uint32_t u; + std::memcpy(&u, &v, 4); + return u; +} + +class Logger { +public: + Logger() { + const char* path = std::getenv("VX_RTU_RAYLOG"); + if (path == nullptr || path[0] == '\0') return; + fp_ = std::fopen(path, "wb"); + if (fp_ == nullptr) { + std::fprintf(stderr, "[rtu-raylog] cannot open %s\n", path); + return; + } + static char buf[1 << 20]; + std::setvbuf(fp_, buf, _IOFBF, sizeof(buf)); + if (const char* s = std::getenv("VX_RTU_RAYLOG_MAX")) max_rays_ = std::strtoull(s, nullptr, 0); + if (const char* s = std::getenv("VX_RTU_RAYLOG_EVERY")) every_ = std::max(1ull, std::strtoull(s, nullptr, 0)); + RaylogHeader h{}; + h.magic = kMagic; + h.version = kVersion; + h.num_threads = VX_CFG_NUM_THREADS; + h.line_bytes = VX_CFG_MEM_BLOCK_SIZE; + h.bvh_width = VX_CFG_RTU_BVH_WIDTH; + h.ray_bytes = sizeof(RaylogRay); + std::fwrite(&h, sizeof(h), 1, fp_); + } + + ~Logger() { + if (fp_ == nullptr) return; + std::fclose(fp_); + std::fprintf(stderr, "[rtu-raylog] rays=%llu logged=%llu replayable=%llu lines=%zu epochs=%u\n", + (unsigned long long)seen_, (unsigned long long)logged_, + (unsigned long long)replayable_, image_.size(), epoch_ + 1); + } + + bool on() const { return fp_ != nullptr; } + + void accept(const void* owner, uint32_t slot) { + std::lock_guard g(mu_); + cb_[{owner, slot}].fill(0); + } + + void callback(const void* owner, uint32_t slot, uint32_t lane, uint32_t cb_type) { + std::lock_guard g(mu_); + cb_[{owner, slot}].at(lane) |= 1u << (cb_type & 7); + } + + void walk_done(const std::unordered_map& lines) { + std::lock_guard g(mu_); + if (full()) return; + for (const auto& kv : lines) { + auto it = image_.find(kv.first); + if (it != image_.end()) { + if (it->second == kv.second) continue; + // The device rewrote a line the scene image already holds: the rays + // from here on see a different memory image. + ++epoch_; + it->second = kv.second; + } else { + image_.emplace(kv.first, kv.second); + } + RaylogLine r{}; + r.type = REC_LINE; + r.epoch = epoch_; + r.addr = kv.first; + std::memcpy(r.data, kv.second.data(), sizeof(r.data)); + std::fwrite(&r, sizeof(r), 1, fp_); + } + } + + void terminal(const void* owner, uint32_t slot, const RtuReq& req, + const std::array& lanes) { + std::lock_guard g(mu_); + auto& cbm = cb_[{owner, slot}]; + for (uint32_t t = 0; t < VX_CFG_NUM_THREADS; ++t) { + const LaneState& l = lanes[t]; + if (!l.active) continue; + uint64_t seq = seen_++; + if (full() || (seq % every_) != 0) continue; + RaylogRay r{}; + r.type = REC_RAY; + const uint32_t cbs = cbm[t]; + const bool decided_by_shader = cbs & ((1u << VX_RT_CB_TYPE_ANYHIT) | (1u << VX_RT_CB_TYPE_PROC)); + r.info = (cbs << kInfoCbShift) | (decided_by_shader ? 0u : kInfoReplayable); + r.seq = uint32_t(seq); + r.epoch = epoch_; + r.scene_root = req.scene_root[t]; + r.ray_flags = req.flags[t]; + r.cull_mask = req.cull_mask[t]; + r.warp_lane = t | (req.warp_id << 8) | (slot << 16); + r.origin[0] = bits(req.origin_x[t]); r.origin[1] = bits(req.origin_y[t]); r.origin[2] = bits(req.origin_z[t]); + r.dir[0] = bits(req.dir_x[t]); r.dir[1] = bits(req.dir_y[t]); r.dir[2] = bits(req.dir_z[t]); + r.tmin = bits(req.tmin[t]); + r.tmax = bits(req.tmax[t]); + if (l.hit) { + r.status = VX_RT_STS_DONE_HIT; + r.hit_t = bits(l.hit_t); + r.hit_u = bits(l.hit_u); + r.hit_v = bits(l.hit_v); + r.prim = l.hit_prim; + r.geom = l.hit_geometry; + r.inst_id = l.hit_instance_id; + r.inst_custom = l.hit_instance_custom; + r.hit_attr = l.hit_attr; + } else { + r.status = VX_RT_STS_DONE_MISS; + } + std::fwrite(&r, sizeof(r), 1, fp_); + ++logged_; + if (!decided_by_shader) ++replayable_; + } + } + +private: + bool full() const { return logged_ >= max_rays_; } + + std::mutex mu_; + FILE* fp_ = nullptr; + unsigned long long max_rays_ = 4ull << 20; + unsigned long long every_ = 1; + uint64_t seen_ = 0; + uint64_t logged_ = 0; + uint64_t replayable_ = 0; + uint32_t epoch_ = 0; + std::unordered_map image_; + std::map, std::array> cb_; +}; + +Logger& logger() { + static Logger inst; + return inst; +} + +} // namespace + +bool enabled() { + static const bool on = logger().on(); + return on; +} + +void on_accept(const void* owner, uint32_t slot) { + logger().accept(owner, slot); +} + +void on_callback(const void* owner, uint32_t slot, uint32_t lane, uint32_t cb_type) { + logger().callback(owner, slot, lane, cb_type); +} + +void on_walk_done(const std::unordered_map& lines) { + logger().walk_done(lines); +} + +void on_terminal(const void* owner, uint32_t slot, const RtuReq& req, + const std::array& lanes) { + logger().terminal(owner, slot, req, lanes); +} + +}}} // namespace vortex::rtu::raylog diff --git a/sim/simx/rtu/rtu_raylog.h b/sim/simx/rtu/rtu_raylog.h new file mode 100644 index 0000000000..c2d0ba2a7f --- /dev/null +++ b/sim/simx/rtu/rtu_raylog.h @@ -0,0 +1,112 @@ +// Copyright © 2019-2023 +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. +// +// PRISM RTU — ray log (opt-in, VX_RTU_RAYLOG=). +// +// Records every traced ray with its terminal result, plus every scene line a +// walk read, so a ray can be re-traced outside the workload that issued it +// (tests/raytracing/rt_replay) and compared bit-exactly against another model. +// +// Stream layout (little-endian): a RaylogHeader, then records tagged by their +// first word. A LINE record is emitted the first time a walk reads a line, and +// again, with the epoch bumped, if a later walk reads different bytes there; +// every RAY record names the epoch whose memory image it was traced against. +// +// Optional knobs: +// VX_RTU_RAYLOG_MAX= stop after n RAY records (default 4M) +// VX_RTU_RAYLOG_EVERY= keep one terminal ray in n (default 1) + +#ifndef _VX_RTU_RAYLOG_H_ +#define _VX_RTU_RAYLOG_H_ + +#include +#include +#include +#include "rtu_types.h" + +namespace vortex { namespace rtu { namespace raylog { + +constexpr uint32_t kMagic = 0x4C525856; // "VXRL" +constexpr uint32_t kVersion = 1; + +enum RecType : uint32_t { + REC_LINE = 1, + REC_RAY = 2, +}; + +// RaylogRay::info bits. +constexpr uint32_t kInfoCbShift = 0; // bit (cb_type) set per callback yielded +constexpr uint32_t kInfoReplayable = 1u << 8; // no any-hit / intersection callback decided it + +struct RaylogHeader { + uint32_t magic; + uint32_t version; + uint32_t num_threads; + uint32_t line_bytes; + uint32_t bvh_width; + uint32_t ray_bytes; + uint32_t reserved[2]; +}; + +struct RaylogLine { + uint32_t type; // REC_LINE + uint32_t epoch; + uint64_t addr; + uint8_t data[VX_CFG_MEM_BLOCK_SIZE]; +}; + +struct RaylogRay { + uint32_t type; // REC_RAY + uint32_t info; + uint32_t seq; + uint32_t epoch; + uint32_t scene_root; + uint32_t ray_flags; + uint32_t cull_mask; + uint32_t warp_lane; // lane | warp << 8 | slot << 16 + uint32_t origin[3]; // float bits + uint32_t dir[3]; + uint32_t tmin; + uint32_t tmax; + uint32_t status; // VX_RT_STS_DONE_HIT / DONE_MISS + uint32_t hit_t; // float bits (0 on a miss) + uint32_t hit_u; + uint32_t hit_v; + uint32_t prim; + uint32_t geom; // geometry index | VX_RT_HIT_BACK_FACING + uint32_t inst_id; + uint32_t inst_custom; + uint32_t hit_attr; + uint32_t reserved[3]; +}; +static_assert(sizeof(RaylogRay) == 112, "RaylogRay layout"); + +bool enabled(); + +// A trace landed in `slot` of the RTU at `owner`: forget the callbacks the +// slot's previous trace raised. +void on_accept(const void* owner, uint32_t slot); + +// A lane of `slot` yielded a callback of `cb_type`. +void on_callback(const void* owner, uint32_t slot, uint32_t lane, uint32_t cb_type); + +// A walk completed against `lines`. +void on_walk_done(const std::unordered_map& lines); + +// `slot` emitted its terminal record. +void on_terminal(const void* owner, uint32_t slot, const RtuReq& req, + const std::array& lanes); + +}}} // namespace vortex::rtu::raylog + +#endif // _VX_RTU_RAYLOG_H_ diff --git a/tests/raytracing/rt_replay/Makefile b/tests/raytracing/rt_replay/Makefile new file mode 100644 index 0000000000..0793d00c5a --- /dev/null +++ b/tests/raytracing/rt_replay/Makefile @@ -0,0 +1,23 @@ +ROOT_DIR := $(realpath ../../..) +include $(ROOT_DIR)/config.mk + +# The scene format the RTU walks is a build-time choice (VX_CFG_RTU_BVH_WIDTH): +# replay against the configuration that recorded the log, so CONFIGS is used +# as given, plus the RTU itself. +CONFIGS := $(if $(findstring -DVX_CFG_EXT_RTU_ENABLE,$(CONFIGS)),$(CONFIGS),$(CONFIGS) -DVX_CFG_EXT_RTU_ENABLE) + +PROJECT := rt_replay + +SRC_DIR := $(VORTEX_HOME)/tests/raytracing/$(PROJECT) + +SRCS := $(SRC_DIR)/main.cpp +HDRS := $(SRC_DIR)/common.h + +VX_SRCS := $(SRC_DIR)/kernel.cpp +VX_HDRS := $(SRC_DIR)/common.h + +OPTS ?= + +KERNEL_LIB := vortex2 + +include ../common.mk diff --git a/tests/raytracing/rt_replay/common.h b/tests/raytracing/rt_replay/common.h new file mode 100644 index 0000000000..8f94077e7b --- /dev/null +++ b/tests/raytracing/rt_replay/common.h @@ -0,0 +1,81 @@ +// Copyright © 2019-2023 +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. +// +// rt_replay — re-traces rays recorded by the SimX RTU ray log +// (VX_RTU_RAYLOG, sim/simx/rtu/rtu_raylog.h) against the scene lines the +// recording walks read, restored at their original device addresses. + +#ifndef _RT_REPLAY_COMMON_H_ +#define _RT_REPLAY_COMMON_H_ + +#include + +// One ray as the kernel consumes it. Each warp's rays share scene / flags / +// cull: those ride the warp-uniform half of vx_rt_wtrace. +typedef struct { + uint32_t scene; + uint32_t flags; + uint32_t cull; + uint32_t pad; + float origin[3]; + float dir[3]; + float tmin; + float tmax; +} replay_ray_t; + +// Status bit set when the trace yielded a candidate (the recorded ray did not). +#define REPLAY_STS_YIELDED 0x100u + +typedef struct { + uint32_t status; + uint32_t t; // float bits + uint32_t u; + uint32_t v; + uint32_t prim; + uint32_t geom; // geometry index | VX_RT_HIT_BACK_FACING + uint32_t inst_id; + uint32_t inst_custom; +} replay_result_t; + +typedef struct { + uint64_t rays_addr; + uint64_t results_addr; + uint32_t count; + uint32_t pad; +} kernel_arg_t; + +#ifdef __cplusplus +// Ray-log stream layout (mirror of sim/simx/rtu/rtu_raylog.h). +#define RAYLOG_MAGIC 0x4C525856u +#define RAYLOG_VERSION 1u +#define RAYLOG_REC_LINE 1u +#define RAYLOG_REC_RAY 2u +#define RAYLOG_INFO_REPLAYABLE (1u << 8) + +struct RaylogHeader { + uint32_t magic, version, num_threads, line_bytes, bvh_width, ray_bytes; + uint32_t reserved[2]; +}; + +struct RaylogRay { + uint32_t type, info, seq, epoch; + uint32_t scene_root, ray_flags, cull_mask, warp_lane; + uint32_t origin[3], dir[3]; + uint32_t tmin, tmax; + uint32_t status, hit_t, hit_u, hit_v, prim, geom, inst_id, inst_custom, hit_attr; + uint32_t reserved[3]; +}; +static_assert(sizeof(RaylogRay) == 112, "RaylogRay layout"); +#endif + +#endif // _RT_REPLAY_COMMON_H_ diff --git a/tests/raytracing/rt_replay/kernel.cpp b/tests/raytracing/rt_replay/kernel.cpp new file mode 100644 index 0000000000..a6cf6922c5 --- /dev/null +++ b/tests/raytracing/rt_replay/kernel.cpp @@ -0,0 +1,50 @@ +// Copyright © 2019-2023 +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. +// +// rt_replay kernel — one recorded ray per thread. The host pads every +// warp-sized group to one (scene, flags, cull), so the trace config is uniform. + +#include +#include +#include "common.h" + +__kernel void kernel_main(kernel_arg_t* arg) { + uint32_t i = blockIdx.x * blockDim.x + threadIdx.x; + if (i >= arg->count) return; + + const replay_ray_t* rays = (const replay_ray_t*)((uintptr_t)arg->rays_addr); + const replay_ray_t& r = rays[i]; + vx_ray_t ray = { {r.origin[0], r.origin[1], r.origin[2]}, + {r.dir[0], r.dir[1], r.dir[2]}, + r.tmin, r.tmax }; + uint32_t h = vx_rt_wtrace(r.scene, 0u, r.flags, r.cull, &ray); + vx_hit_t hit; + uint32_t sts = vx_rt_wait(h, &hit); + uint32_t yielded = 0; + // A replayed ray was never decided by a shader; a candidate here is itself a + // divergence. Resolve it as a closest-hit/miss dispatch would, and flag it. + while (vx_rt_sts_is_yield(sts)) { + yielded = REPLAY_STS_YIELDED; + sts = vx_rt_continue(h, VX_RT_CB_IGNORE, hit.t, 0u, &hit); + } + + replay_result_t* res = (replay_result_t*)((uintptr_t)arg->results_addr) + i; + res->status = sts | yielded; + res->t = __builtin_bit_cast(uint32_t, hit.t); + res->u = __builtin_bit_cast(uint32_t, hit.u); + res->v = __builtin_bit_cast(uint32_t, hit.v); + res->prim = hit.primitive_id; + res->geom = hit.geometry_index | (hit.back_facing ? VX_RT_HIT_BACK_FACING : 0u); + res->inst_id = hit.instance_id; + res->inst_custom = hit.instance_custom; +} diff --git a/tests/raytracing/rt_replay/main.cpp b/tests/raytracing/rt_replay/main.cpp new file mode 100644 index 0000000000..d7262fbad5 --- /dev/null +++ b/tests/raytracing/rt_replay/main.cpp @@ -0,0 +1,430 @@ +// Copyright © 2019-2023 +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. +// +// rt_replay — host driver. +// +// Loads a SimX RTU ray log (VX_RTU_RAYLOG), restores every scene line the +// recorded walks read at its original device address (absolute pointers inside +// the scene stay valid), re-traces the recorded rays in batches and compares +// each result bit-exactly with the recorded one. + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include +#include +#include "common.h" + +#define RT_CHECK(_expr) \ + do { \ + int _ret = _expr; \ + if (0 == _ret) break; \ + printf("Error: '%s' returned %d!\n", #_expr, (int)_ret); \ + cleanup(); \ + exit(-1); \ + } while (false) + +namespace { + +const char* kernel_file = "kernel.vxbin"; + +vx_device_h device = nullptr; +vx_queue_h queue = nullptr; +vx_module_h module_ = nullptr; +vx_kernel_h kernel = nullptr; +vx_buffer_h rays_buf = nullptr; +vx_buffer_h res_buf = nullptr; +std::vector scene_bufs; + +void cleanup() { + if (!device) return; + for (auto b : scene_bufs) vx_buffer_release(b); + if (rays_buf) vx_buffer_release(rays_buf); + if (res_buf) vx_buffer_release(res_buf); + if (kernel) vx_kernel_release(kernel); + if (module_) vx_module_release(module_); + if (queue) vx_queue_release(queue); + vx_device_release(device); + device = nullptr; +} + +struct Line { + uint64_t addr; + uint32_t epoch; + std::vector data; +}; + +struct Range { + uint64_t base; + uint64_t size; + vx_buffer_h buf; +}; + +const char* log_path = nullptr; +const char* list_path = nullptr; +const char* out_path = nullptr; +uint32_t batch = 4096; +uint64_t range_lo = 0; +uint64_t range_hi = UINT64_MAX; +uint64_t stride = 1; +int64_t only_epoch = -1; +uint32_t max_print = 20; +bool all_rays = false; +bool verbose = false; + +void show_usage() { + printf("Usage: rt_replay -f [-b batch] [-r lo:hi] [-l listfile] [-s stride]\n" + " [-e epoch] [-m maxprint] [-o mismatch_file] [-a] [-v]\n" + " -r lo:hi replay ray indices [lo, hi) (index = RAY record order in the log)\n" + " -l file replay only the ray indices listed (one per line, '#' comments;\n" + " a mismatch file from -o is accepted as is)\n" + " -s n replay every n-th selected ray\n" + " -a also replay rays a shader decided (any-hit / intersection)\n"); +} + +void parse_args(int argc, char** argv) { + int c; + while ((c = getopt(argc, argv, "f:b:r:l:s:e:m:o:avh")) != -1) { + switch (c) { + case 'f': log_path = optarg; break; + case 'b': batch = std::max(1u, (uint32_t)strtoul(optarg, nullptr, 0)); break; + case 'r': { + const char* colon = strchr(optarg, ':'); + range_lo = strtoull(optarg, nullptr, 0); + if (colon && colon[1]) range_hi = strtoull(colon + 1, nullptr, 0); + else if (!colon) range_hi = range_lo + 1; + break; + } + case 'l': list_path = optarg; break; + case 's': stride = std::max(1, strtoull(optarg, nullptr, 0)); break; + case 'e': only_epoch = strtoll(optarg, nullptr, 0); break; + case 'm': max_print = strtoul(optarg, nullptr, 0); break; + case 'o': out_path = optarg; break; + case 'a': all_rays = true; break; + case 'v': verbose = true; break; + default: show_usage(); exit(c == 'h' ? 0 : -1); + } + } + if (!log_path) { show_usage(); exit(-1); } +} + +float fbits(uint32_t u) { float f; memcpy(&f, &u, 4); return f; } + +float near_t(float t) { return t + std::fabs(t) * 0x1p-19f; } + +// Why a replayed ray disagrees with its record. +const char* classify(const RaylogRay& ref, const replay_result_t& got) { + if (got.status & REPLAY_STS_YIELDED) return "yielded"; + const bool rh = ref.status == VX_RT_STS_DONE_HIT; + const bool gh = got.status == VX_RT_STS_DONE_HIT; + if (rh && !gh) return "ref_hit_got_miss"; + if (!rh && gh) return "ref_miss_got_hit"; + if (!rh) return "status"; + const bool same_prim = ref.prim == got.prim && ref.inst_id == got.inst_id + && (ref.geom & VX_RT_HIT_GEOMETRY_MASK) == (got.geom & VX_RT_HIT_GEOMETRY_MASK); + if (same_prim) { + if (ref.hit_t != got.t || ref.hit_u != got.u || ref.hit_v != got.v) return "same_prim_bits"; + return "attr"; + } + if (ref.ray_flags & VX_RT_FLAG_TERMINATE_ON_FIRST_HIT) return "tofh_other_hit"; + const float a = fbits(ref.hit_t), b = fbits(got.t); + if (a == b) return "tie_exact"; + if ((a < b && near_t(a) >= b) || (b < a && near_t(b) >= a)) return "near_tie"; + return b < a ? "got_nearer_hit" : "got_farther_hit"; +} + +bool matches(const RaylogRay& ref, const replay_result_t& got) { + if (got.status != ref.status) return false; + if (ref.status != VX_RT_STS_DONE_HIT) return true; + return got.t == ref.hit_t && got.u == ref.hit_u && got.v == ref.hit_v + && got.prim == ref.prim && got.geom == ref.geom + && got.inst_id == ref.inst_id && got.inst_custom == ref.inst_custom; +} + +void print_ray(FILE* fp, uint64_t idx, const RaylogRay& r, const replay_result_t& g, const char* cat) { + fprintf(fp, "%llu seq=%u cat=%s epoch=%u scene=0x%x flags=0x%x cull=0x%x" + " o=(%a,%a,%a) d=(%a,%a,%a) tmin=%a tmax=%a\n", + (unsigned long long)idx, r.seq, cat, r.epoch, r.scene_root, r.ray_flags, r.cull_mask, + fbits(r.origin[0]), fbits(r.origin[1]), fbits(r.origin[2]), + fbits(r.dir[0]), fbits(r.dir[1]), fbits(r.dir[2]), + fbits(r.tmin), fbits(r.tmax)); + fprintf(fp, "# ref: sts=%u t=%a (0x%08x) u=0x%08x v=0x%08x prim=%u geom=0x%08x inst=%u custom=%u\n", + r.status, fbits(r.hit_t), r.hit_t, r.hit_u, r.hit_v, r.prim, r.geom, r.inst_id, r.inst_custom); + fprintf(fp, "# got: sts=%u t=%a (0x%08x) u=0x%08x v=0x%08x prim=%u geom=0x%08x inst=%u custom=%u\n", + g.status, fbits(g.t), g.t, g.u, g.v, g.prim, g.geom, g.inst_id, g.inst_custom); +} + +} // namespace + +int main(int argc, char** argv) { + parse_args(argc, argv); + + // ── load the log ────────────────────────────────────────────────── + FILE* fp = fopen(log_path, "rb"); + if (!fp) { printf("Error: cannot open %s\n", log_path); return -1; } + RaylogHeader hdr{}; + if (fread(&hdr, sizeof(hdr), 1, fp) != 1 || hdr.magic != RAYLOG_MAGIC + || hdr.version != RAYLOG_VERSION || hdr.ray_bytes != sizeof(RaylogRay)) { + printf("Error: %s is not a version-%u ray log\n", log_path, RAYLOG_VERSION); + return -1; + } +#ifdef VX_CFG_RTU_BVH_WIDTH + if (hdr.bvh_width != VX_CFG_RTU_BVH_WIDTH) { + printf("Error: log recorded with RTU_BVH_WIDTH=%u, replay built with %u\n", + hdr.bvh_width, (uint32_t)VX_CFG_RTU_BVH_WIDTH); + return -1; + } +#endif + const uint32_t lb = hdr.line_bytes; + std::vector lines; + std::vector rays; + { + std::vector rec(std::max(sizeof(RaylogRay), 16 + lb)); + uint32_t type; + while (fread(&type, 4, 1, fp) == 1) { + if (type == RAYLOG_REC_LINE) { + if (fread(rec.data() + 4, 12 + lb, 1, fp) != 1) break; + Line l; + memcpy(&l.epoch, rec.data() + 4, 4); + memcpy(&l.addr, rec.data() + 8, 8); + l.data.assign(rec.data() + 16, rec.data() + 16 + lb); + lines.push_back(std::move(l)); + } else if (type == RAYLOG_REC_RAY) { + RaylogRay r; + r.type = type; + if (fread(reinterpret_cast(&r) + 4, sizeof(r) - 4, 1, fp) != 1) break; + rays.push_back(r); + } else { + printf("Error: corrupt record type %u in %s\n", type, log_path); + return -1; + } + } + } + fclose(fp); + uint32_t num_epochs = 1; + for (auto& l : lines) num_epochs = std::max(num_epochs, l.epoch + 1); + printf("raylog: %zu rays, %zu lines (%u B), %u epoch(s), bvh_width=%u, recorded NT=%u\n", + rays.size(), lines.size(), lb, num_epochs, hdr.bvh_width, hdr.num_threads); + + // ── select rays ─────────────────────────────────────────────────── + std::vector sel; + { + std::vector cand; + if (list_path) { + FILE* lf = fopen(list_path, "r"); + if (!lf) { printf("Error: cannot open %s\n", list_path); return -1; } + char buf[4096]; + while (fgets(buf, sizeof(buf), lf)) { + if (buf[0] == '#' || buf[0] == '\n') continue; + cand.push_back(strtoull(buf, nullptr, 0)); + } + fclose(lf); + } else { + for (uint64_t i = range_lo; i < std::min(range_hi, rays.size()); ++i) cand.push_back(i); + } + uint64_t skipped_cb = 0, n = 0; + for (uint64_t i : cand) { + if (i >= rays.size()) continue; + if (range_lo > i || i >= range_hi) continue; + const RaylogRay& r = rays[i]; + if (only_epoch >= 0 && r.epoch != (uint32_t)only_epoch) continue; + if (!all_rays && !(r.info & RAYLOG_INFO_REPLAYABLE)) { ++skipped_cb; continue; } + if ((n++ % stride) != 0) continue; + sel.push_back(i); + } + std::stable_sort(sel.begin(), sel.end(), [&](uint64_t a, uint64_t b) { + return rays[a].epoch < rays[b].epoch; + }); + printf("selected %zu rays (%llu skipped: decided by a shader callback)\n", + sel.size(), (unsigned long long)skipped_cb); + } + if (sel.empty()) { printf("nothing to replay\n"); return 0; } + + // ── device + scene reservation ──────────────────────────────────── + RT_CHECK(vx_device_open(0, &device)); + vx_queue_info_t qi = { sizeof(qi), nullptr, VX_QUEUE_PRIORITY_NORMAL, 0 }; + RT_CHECK(vx_queue_create(device, &qi, &queue)); + + // Claim the recorded scene addresses before anything else is allocated, in + // page-aligned ranges (nearby pages merged to keep the buffer count low). + std::vector ranges; + { + const uint64_t page = 4096, merge_gap = 64 * 1024; + std::vector pages; + for (auto& l : lines) { + for (uint64_t p = l.addr & ~(page - 1); p < l.addr + lb; p += page) pages.push_back(p); + } + std::sort(pages.begin(), pages.end()); + pages.erase(std::unique(pages.begin(), pages.end()), pages.end()); + for (uint64_t p : pages) { + if (!ranges.empty() && p <= ranges.back().base + ranges.back().size + merge_gap) { + ranges.back().size = p + page - ranges.back().base; + } else { + ranges.push_back({p, page, nullptr}); + } + } + uint64_t total = 0; + for (auto& r : ranges) { + RT_CHECK(vx_buffer_reserve(device, r.base, r.size, VX_MEM_READ, &r.buf)); + scene_bufs.push_back(r.buf); + total += r.size; + } + printf("scene: %zu reserved range(s), %llu KB\n", ranges.size(), (unsigned long long)(total >> 10)); + } + + RT_CHECK(vx_module_load_file(device, kernel_file, &module_)); + RT_CHECK(vx_module_get_kernel(module_, "main", &kernel)); + + const uint32_t NT = VX_CFG_NUM_THREADS; + uint64_t cap = 0; + + std::map cats; + uint64_t replayed = 0, mismatched = 0, printed = 0; + FILE* of = out_path ? fopen(out_path, "w") : nullptr; + if (of) fprintf(of, "# rt_replay mismatches of %s: then ref/got detail\n", log_path); + double kernel_s = 0; + int32_t loaded_epoch = -1; + + for (size_t pos = 0; pos < sel.size();) { + // A batch never spans epochs: each epoch replays against its own image. + const uint32_t epoch = rays[sel[pos]].epoch; + size_t end = pos; + while (end < sel.size() && end - pos < batch && rays[sel[end]].epoch == epoch) ++end; + + if ((int32_t)epoch != loaded_epoch) { + // Image of `epoch`: per address, the newest version recorded at or before + // it, else the first one recorded (the line was not rewritten before then). + std::map img; + for (auto& l : lines) { + auto it = img.find(l.addr); + if (it == img.end()) { img[l.addr] = &l; continue; } + if (l.epoch <= epoch && (it->second->epoch > epoch || l.epoch >= it->second->epoch)) it->second = &l; + } + std::vector> hosts; // alive until the queue drains + for (auto& r : ranges) { + hosts.emplace_back(r.size, 0); + auto& host = hosts.back(); + for (auto it = img.lower_bound(r.base); it != img.end() && it->first < r.base + r.size; ++it) { + uint64_t off = it->first - r.base; + memcpy(host.data() + off, it->second->data.data(), std::min(lb, r.size - off)); + } + RT_CHECK(vx_enqueue_write(queue, r.buf, 0, host.data(), r.size, 0, nullptr, nullptr)); + } + RT_CHECK(vx_queue_finish(queue, VX_TIMEOUT_INFINITE)); + loaded_epoch = epoch; + } + + // Group by the warp-uniform trace config; pad each group to whole warps + // with copies of its last ray (their results are dropped). + std::map, std::vector> groups; + for (size_t k = pos; k < end; ++k) { + const RaylogRay& r = rays[sel[k]]; + groups[{r.scene_root, r.ray_flags & ~(VX_RT_FLAG_ENABLE_CHS | VX_RT_FLAG_ENABLE_MISS), r.cull_mask}].push_back(sel[k]); + } + std::vector dev_rays; + std::vector owner; // ray index, or -1 for padding + for (auto& g : groups) { + auto& v = g.second; + size_t padded = (v.size() + NT - 1) / NT * NT; + for (size_t k = 0; k < padded; ++k) { + uint64_t idx = v[std::min(k, v.size() - 1)]; + const RaylogRay& r = rays[idx]; + replay_ray_t d{}; + d.scene = r.scene_root; + d.flags = std::get<1>(g.first); + d.cull = r.cull_mask; + for (int j = 0; j < 3; ++j) { d.origin[j] = fbits(r.origin[j]); d.dir[j] = fbits(r.dir[j]); } + d.tmin = fbits(r.tmin); + d.tmax = fbits(r.tmax); + dev_rays.push_back(d); + owner.push_back(k < v.size() ? (int64_t)idx : -1); + } + } + const uint32_t count = (uint32_t)dev_rays.size(); + if (count > cap) { + if (rays_buf) vx_buffer_release(rays_buf); + if (res_buf) vx_buffer_release(res_buf); + cap = std::max(count, batch + NT); + RT_CHECK(vx_buffer_create(device, cap * sizeof(replay_ray_t), VX_MEM_READ, &rays_buf)); + RT_CHECK(vx_buffer_create(device, cap * sizeof(replay_result_t), VX_MEM_WRITE, &res_buf)); + } + kernel_arg_t arg{}; + RT_CHECK(vx_buffer_address(rays_buf, &arg.rays_addr)); + RT_CHECK(vx_buffer_address(res_buf, &arg.results_addr)); + arg.count = count; + RT_CHECK(vx_enqueue_write(queue, rays_buf, 0, dev_rays.data(), count * sizeof(replay_ray_t), 0, nullptr, nullptr)); + RT_CHECK(vx_queue_finish(queue, VX_TIMEOUT_INFINITE)); + + auto t0 = std::chrono::steady_clock::now(); + vx_event_h lev = nullptr, rev = nullptr; + vx_launch_info_t li = {}; + li.struct_size = sizeof(li); + li.kernel = kernel; + li.args_host = &arg; + li.args_size = sizeof(arg); + li.ndim = 1; + li.grid_dim[0] = count / NT; + li.block_dim[0] = NT; + RT_CHECK(vx_enqueue_launch(queue, &li, 0, nullptr, &lev)); + std::vector res(count); + RT_CHECK(vx_enqueue_read(queue, res.data(), res_buf, 0, count * sizeof(replay_result_t), 1, &lev, &rev)); + RT_CHECK(vx_event_wait_value(rev, 1, VX_TIMEOUT_INFINITE)); + vx_event_release(rev); + vx_event_release(lev); + double dt = std::chrono::duration(std::chrono::steady_clock::now() - t0).count(); + kernel_s += dt; + + uint64_t bad = 0; + for (uint32_t k = 0; k < count; ++k) { + if (owner[k] < 0) continue; + const RaylogRay& r = rays[owner[k]]; + ++replayed; + if (matches(r, res[k])) continue; + ++bad; + const char* cat = classify(r, res[k]); + ++cats[cat]; + if (of) print_ray(of, owner[k], r, res[k], cat); + if (printed < max_print) { print_ray(stdout, owner[k], r, res[k], cat); ++printed; } + } + mismatched += bad; + if (verbose || sel.size() > batch) { + printf("batch [%zu, %zu) epoch %u: %llu rays, %llu mismatches, %.1fs\n", + pos, end, epoch, (unsigned long long)(end - pos), (unsigned long long)bad, dt); + fflush(stdout); + } + pos = end; + } + if (of) fclose(of); + + printf("replayed %llu rays in %.1fs: %llu match, %llu mismatch\n", + (unsigned long long)replayed, kernel_s, + (unsigned long long)(replayed - mismatched), (unsigned long long)mismatched); + for (auto& c : cats) printf(" %-18s %llu\n", c.first.c_str(), (unsigned long long)c.second); + + cleanup(); + if (mismatched) { + printf("FAILED!\n"); + return 1; + } + printf("PASSED!\n"); + return 0; +} From 36e11c037e20303e2b2bfa9f4b38271264027758 Mon Sep 17 00:00:00 2001 From: Blaise Tine Date: Sat, 3 Oct 2026 03:15:49 -0700 Subject: [PATCH 12/31] rtu: RTL tri/box PEs match SimX's intersection tests bit for bit - Tri PE: accept tmin < t < tmax (strict on both sides, as SimX ray_triangle). The FP units now run with EXCEPT_ENABLE=1: with it off the soft FMA/FDIV treat NaN/inf operands as finite, so a NaN vertex (an inactive triangle) reported a hit. - Box PE: rebuilt to mirror SimX reconstruct_child_aabb + ray_aabb_intersect: mn = origin + q*2^e (exact product, one add; the old q*2^e + (origin - ro) FMA rounds differently in ~3% of slabs), then (mn - ro) * inv_d; lo/hi reduced with fmin/fmax NaN semantics seeded with -/+inf; accept hi >= max(0, lo) && lo <= t_max (culled against [0, t_max], not t_min); t_near = max(t_min, lo), +0 for a zero so the collector's unsigned order holds. Same latency; IEEE specials on. - Recip: a zero (or flushed subnormal) direction component returns +FLT_MAX, as SimX; the divider handles inf/NaN. - hw/unittest/rtu_box_pe (recip + box PE vs SimX: inv_d, accept, t_near and per-node child order) and directed cases in rtu_tri_pe (t on tmin/tmax, origin on the triangle, NaN/inf/-0/subnormal operands). Both report 0 mismatches; the only differences left are subnormal flushes (the PEs are FTZ/DAZ by design), each confirmed equal to SimX run under FTZ/DAZ. The scheduler must now pass the ray's t_max (not the committed t) to the tri PE, as SimX does, or equal-t twins never reach the tie-break. Co-Authored-By: Claude Opus 5.5 --- hw/rtl/rtu/VX_rtu_box_pe.sv | 476 ++++++++------------ hw/rtl/rtu/VX_rtu_recip.sv | 53 ++- hw/rtl/rtu/VX_rtu_tri_pe.sv | 46 +- hw/unittest/rtu_box_pe/Makefile | 31 ++ hw/unittest/rtu_box_pe/VX_rtu_box_pe_tb.sv | 111 +++++ hw/unittest/rtu_box_pe/main.cpp | 492 +++++++++++++++++++++ hw/unittest/rtu_tri_pe/main.cpp | 174 +++++++- 7 files changed, 1034 insertions(+), 349 deletions(-) create mode 100644 hw/unittest/rtu_box_pe/Makefile create mode 100644 hw/unittest/rtu_box_pe/VX_rtu_box_pe_tb.sv create mode 100644 hw/unittest/rtu_box_pe/main.cpp diff --git a/hw/rtl/rtu/VX_rtu_box_pe.sv b/hw/rtl/rtu/VX_rtu_box_pe.sv index 456b05e50d..55f34f9efa 100644 --- a/hw/rtl/rtu/VX_rtu_box_pe.sv +++ b/hw/rtl/rtu/VX_rtu_box_pe.sv @@ -12,22 +12,21 @@ // limitations under the License. // VX_rtu_box_pe — pipelined ray-vs-AABB slab intersector for one child box. -// Streams one box per cycle; emits {hit, t_near} after a fixed latency. +// Streams one box per cycle; emits {hit, t_near} after a fixed latency. Mirrors +// SimX reconstruct_child_aabb + rtu::ray_aabb_intersect op for op, so every +// accept decision and t_near match it: // -// dequant mn[a] = origin[a] + qmin[a] * 2^exp[a] (qmax symmetric) -// slab t0[a] = (mn[a] - ro[a]) * inv_d[a] (t1 from mx) -// lo[a] = min(t0,t1) hi[a] = max(t0,t1) -// reduce t_near = max(t_min, lo[x], lo[y], lo[z]) -// t_far = min(t_max, hi[x], hi[y], hi[z]) -// hit = (t_near <= t_far) +// dequant mn[a] = origin[a] + q[a]*2^exp[a] (product exact, one add) +// slab t0[a] = (mn[a] - ro[a]) * inv_d[a] (t1 from mx) +// reduce lo = max(-inf, min(t0,t1)[*]) hi = min(+inf, max(t0,t1)[*]) +// (fmin/fmax drop a NaN operand, so lo and hi are never NaN) +// hit = hi >= max(0, lo) && lo <= t_max +// t_near = max(t_min, lo) (descent order only) // -// The slab subtracts the ray origin before multiplying by inv_d (rather than -// the algebraically-equal mn*inv_d - ro*inv_d) so axis-aligned rays — where -// inv_d is +/-inf — stay numerically correct: (mn-ro) is finite, so (mn-ro)*inf -// is a signed infinity that the min/max reduction treats as a non-constraining -// slab, instead of inf-inf = NaN. The uint8->fp32 and 2^exp dequant terms are -// combinational; the FP add/mul use VX_fma_unit (a*b±c) and the min/max/compare -// use VX_fncp_unit, register-balanced to the configured latencies. +// The box is culled against [0, t_max], not [t_min, t_max]: t_min belongs to +// the primitive test alone, whose t carries rounding the slab distances do not. +// inv_d is the ray-setup reciprocal (VX_rtu_recip), FLT_MAX for a zero +// direction component. The FP units flush subnormals. `include "VX_define.vh" @@ -48,8 +47,7 @@ module VX_rtu_box_pe import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( input wire [2:0][7:0] qmin, input wire [2:0][7:0] qmax, // raw (unquantized) AABB path — procedural-leaf boxes carry float min/max - // directly instead of node-relative quantized corners. raw=0 is bit- - // identical to the quantized path (BVH internal-node box tests). + // directly instead of node-relative quantized corners. input wire raw, input wire [2:0][31:0] raw_min, input wire [2:0][31:0] raw_max, @@ -67,162 +65,191 @@ module VX_rtu_box_pe import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( output wire hit, output wire [31:0] t_near ); - // VX_fncp_unit result latency is 1 (one input pipe reg, OUT_REG=0); its - // LATENCY param only sizes the internal mask pipe, not the result path, so - // size it to 2 to avoid a degenerate [-1:0] mask-pipe slice while the result - // still lands after one cycle. - localparam FNCP_LAT = 1; // result latency for alignment - localparam FNCP_SIZE = 2; // mask-pipe sizing param - localparam LAT_ORIGIN = LATENCY_FMA; // origin - ro - localparam LAT_DEQUANT = LATENCY_FMA; // q*scale + (origin - ro) - localparam LAT_SLAB = LATENCY_FMA; // (mn - ro)*inv_d - localparam LAT_MINMAX = FNCP_LAT; // lo/hi per axis - localparam LAT_REDUCE = 2 * FNCP_LAT; // 4-input min/max tree - localparam LAT_CMP = FNCP_LAT; // t_near <= t_far - localparam LATENCY = LAT_ORIGIN + LAT_DEQUANT + LAT_SLAB + LAT_MINMAX + LAT_REDUCE + LAT_CMP; + localparam F = LATENCY_FMA; + localparam LAT_FMA = 3 * F; // mn, mn - ro, * inv_d + localparam LATENCY = LAT_FMA + 4; // + per-axis, 2 reduce, verdict localparam [INST_FMT_BITS-1:0] FMT_ADD = 2'b00; // F32, a*b + c localparam [INST_FMT_BITS-1:0] FMT_SUB = 2'b10; // F32, a*b - c + localparam [31:0] F32_ONE = 32'h3F800000; + localparam [31:0] F32_NEG0 = 32'h80000000; // x + -0 == x, signs kept + localparam [31:0] F32_PINF = 32'h7F800000; + localparam [31:0] F32_NINF = 32'hFF800000; - // ── combinational uint8 -> fp32 ─────────────────────────────────── - function automatic logic [31:0] u8_to_f32(input logic [7:0] n); + // ── helpers ─────────────────────────────────────────────────────── + // q * 2^e for an 8-bit integer q and an int8 e, as the F32 product rounds: + // exact, +inf past the range, 0 below it (subnormals flush). + function automatic logic [31:0] q_scale(input logic [7:0] q, input logic [7:0] e); logic [2:0] msb; - logic [6:0] shifted; - logic [22:0] man; - if (n == 8'd0) begin - u8_to_f32 = 32'd0; + logic [6:0] frac; + logic signed [9:0] be; + if (q == 8'd0) begin + q_scale = 32'd0; end else begin msb = 3'd0; for (integer b = 0; b < 8; ++b) begin - if (n[b]) begin + if (q[b]) begin msb = b[2:0]; end end - // normalize so the leading 1 sits at bit 7, then the 7 bits - // below it become the top of the fp32 mantissa. - shifted = 7'(n << (3'd7 - msb)); - man = {shifted, 16'd0}; - u8_to_f32 = {1'b0, (8'd127 + 8'(msb)), man}; + frac = 7'(q << (3'd7 - msb)); + be = 10'sd127 + 10'(msb) + 10'($signed(e)); + if (be >= 10'sd255) begin + q_scale = F32_PINF; + end else if (be <= 10'sd0) begin + q_scale = 32'd0; + end else begin + q_scale = {1'b0, be[7:0], frac, 16'd0}; + end end endfunction - // ── combinational 2^exp as fp32 (well-conditioned exponents) ────── - function automatic logic [31:0] pow2_f32(input logic [7:0] e); - logic [7:0] biased; - biased = 8'(9'sd127 + {e[7], e}); // sign-extend int8 exponent - pow2_f32 = {1'b0, biased, 23'd0}; + function automatic logic f32_is_nan(input logic [30:0] a); + f32_is_nan = (a[30:23] == 8'hff) && (a[22:0] != 23'd0); + endfunction + + // monotone integer key of a non-NaN F32 (+0 and -0 share one key) + function automatic logic [31:0] f32_key(input logic [31:0] a); + f32_key = (a[30:0] == 31'd0) ? 32'h80000000 : (a[31] ? ~a : {1'b1, a[30:0]}); + endfunction + + // IEEE a <= b; false on NaN + function automatic logic f32_le(input logic [31:0] a, input logic [31:0] b); + f32_le = !f32_is_nan(a[30:0]) && !f32_is_nan(b[30:0]) && (f32_key(a) <= f32_key(b)); endfunction - // ── stage 0: prep per-axis float operands ───────────────────────── - wire [2:0][31:0] qmin_f, qmax_f, scale; + // fmin / fmax: a NaN operand yields the other one; two NaNs yield `none` + function automatic logic [31:0] f32_min(input logic [31:0] a, input logic [31:0] b, + input logic [31:0] none); + if (f32_is_nan(a[30:0])) begin + f32_min = f32_is_nan(b[30:0]) ? none : b; + end else if (f32_is_nan(b[30:0])) begin + f32_min = a; + end else begin + f32_min = (f32_key(b) < f32_key(a)) ? b : a; + end + endfunction + + function automatic logic [31:0] f32_max(input logic [31:0] a, input logic [31:0] b, + input logic [31:0] none); + if (f32_is_nan(a[30:0])) begin + f32_max = f32_is_nan(b[30:0]) ? none : b; + end else if (f32_is_nan(b[30:0])) begin + f32_max = a; + end else begin + f32_max = (f32_key(b) > f32_key(a)) ? b : a; + end + endfunction + + // ── stage 1: box corners mn = origin + q*2^exp (raw: the corner itself) ── + wire [2:0][31:0] mn_a, mx_a, mnx_c; for (genvar a = 0; a < 3; ++a) begin : g_prep - assign qmin_f[a] = u8_to_f32(qmin[a]); - assign qmax_f[a] = u8_to_f32(qmax[a]); - assign scale[a] = pow2_f32(exp[a]); + assign mn_a[a] = raw ? raw_min[a] : q_scale(qmin[a], exp[a]); + assign mx_a[a] = raw ? raw_max[a] : q_scale(qmax[a], exp[a]); + assign mnx_c[a] = raw ? F32_NEG0 : origin[a]; end - // ── stage 1: origin - ro (per axis) ─────────────────────────────── - wire [2:0][31:0] oro; - for (genvar a = 0; a < 3; ++a) begin : g_origin + wire [2:0][31:0] mn, mx; + for (genvar a = 0; a < 3; ++a) begin : g_corner VX_fma_unit #( - .USE_DSP (`VX_CFG_RTU_USE_DSP), // vendor xil_fma on Vivado (soft in sim), like the FPU - .LATENCY (LAT_ORIGIN), + .USE_DSP (`VX_CFG_RTU_USE_DSP), + .LATENCY (F), .SUBNORM_ENABLE (0), - .EXCEPT_ENABLE (0) - ) fma_oro ( + .EXCEPT_ENABLE (1) + ) fma_mn ( .clk (clk), .reset (reset), .enable (enable), .mask (valid_in), .op_type (INST_FPU_MADD), - .fmt (FMT_SUB), + .fmt (FMT_ADD), + .frm (INST_FRM_RNE), + .dataa (mn_a[a]), + .datab (F32_ONE), + .datac (mnx_c[a]), + .result (mn[a]), + `UNUSED_PIN (fflags) + ); + VX_fma_unit #( + .USE_DSP (`VX_CFG_RTU_USE_DSP), + .LATENCY (F), + .SUBNORM_ENABLE (0), + .EXCEPT_ENABLE (1) + ) fma_mx ( + .clk (clk), + .reset (reset), + .enable (enable), + .mask (valid_in), + .op_type (INST_FPU_MADD), + .fmt (FMT_ADD), .frm (INST_FRM_RNE), - .dataa (origin[a]), - .datab (32'h3F800000 /*1.0*/), - .datac (ro[a]), - .result (oro[a]), + .dataa (mx_a[a]), + .datab (F32_ONE), + .datac (mnx_c[a]), + .result (mx[a]), `UNUSED_PIN (fflags) ); end - // quantized corners delayed to align with origin-ro - wire [2:0][31:0] qmin_f_q, qmax_f_q, scale_q; + wire [2:0][31:0] ro_d; VX_shift_register #( - .DATAW (3*32*3), - .DEPTH (LAT_ORIGIN) - ) sr_q ( - .clk (clk), - .reset (reset), - .enable (enable), - .data_in ({qmin_f, qmax_f, scale}), - .data_out ({qmin_f_q, qmax_f_q, scale_q}) - ); - - // raw-path operands delayed to align with the dequant-FMA inputs. - wire raw_d; - wire [2:0][31:0] raw_min_d, raw_max_d, ro_d; - VX_shift_register #( - .DATAW (1 + 3*32*3), - .DEPTH (LAT_ORIGIN) - ) sr_raw ( + .DATAW (3*32), + .DEPTH (F) + ) sr_ro ( .clk (clk), .reset (reset), .enable (enable), - .data_in ({raw, raw_min, raw_max, ro}), - .data_out ({raw_d, raw_min_d, raw_max_d, ro_d}) + .data_in (ro), + .data_out (ro_d) ); - // ── stage 2: corners relative to the ray origin (mn-ro, mx-ro). Quantized: - // q*scale + (origin-ro). Raw procedural box: (min*1.0 - ro) directly, - // reusing the same FMAs (FMT_SUB). ── - localparam [31:0] FP_ONE = 32'h3F800000; + // ── stage 2: corners relative to the ray origin, mn - ro / mx - ro ── wire [2:0][31:0] dmn, dmx; - for (genvar a = 0; a < 3; ++a) begin : g_dequant + for (genvar a = 0; a < 3; ++a) begin : g_rel VX_fma_unit #( - .USE_DSP (`VX_CFG_RTU_USE_DSP), // vendor xil_fma on Vivado (soft in sim), like the FPU - .LATENCY (LAT_DEQUANT), + .USE_DSP (`VX_CFG_RTU_USE_DSP), + .LATENCY (F), .SUBNORM_ENABLE (0), - .EXCEPT_ENABLE (0) - ) fma_mn ( + .EXCEPT_ENABLE (1) + ) fma_dmn ( .clk (clk), .reset (reset), .enable (enable), .mask (1'b1), .op_type (INST_FPU_MADD), - .fmt (raw_d ? FMT_SUB : FMT_ADD), + .fmt (FMT_SUB), .frm (INST_FRM_RNE), - .dataa (raw_d ? raw_min_d[a] : qmin_f_q[a]), - .datab (raw_d ? FP_ONE : scale_q[a]), - .datac (raw_d ? ro_d[a] : oro[a]), + .dataa (mn[a]), + .datab (F32_ONE), + .datac (ro_d[a]), .result (dmn[a]), `UNUSED_PIN (fflags) ); VX_fma_unit #( - .USE_DSP (`VX_CFG_RTU_USE_DSP), // vendor xil_fma on Vivado (soft in sim), like the FPU - .LATENCY (LAT_DEQUANT), + .USE_DSP (`VX_CFG_RTU_USE_DSP), + .LATENCY (F), .SUBNORM_ENABLE (0), - .EXCEPT_ENABLE (0) - ) fma_mx ( + .EXCEPT_ENABLE (1) + ) fma_dmx ( .clk (clk), .reset (reset), .enable (enable), .mask (1'b1), .op_type (INST_FPU_MADD), - .fmt (raw_d ? FMT_SUB : FMT_ADD), + .fmt (FMT_SUB), .frm (INST_FRM_RNE), - .dataa (raw_d ? raw_max_d[a] : qmax_f_q[a]), - .datab (raw_d ? FP_ONE : scale_q[a]), - .datac (raw_d ? ro_d[a] : oro[a]), + .dataa (mx[a]), + .datab (F32_ONE), + .datac (ro_d[a]), .result (dmx[a]), `UNUSED_PIN (fflags) ); end - // inv_d delayed to align with the origin-relative corners wire [2:0][31:0] inv_d_q; VX_shift_register #( .DATAW (3*32), - .DEPTH (LAT_ORIGIN + LAT_DEQUANT) + .DEPTH (2 * F) ) sr_invd ( .clk (clk), .reset (reset), @@ -235,10 +262,10 @@ module VX_rtu_box_pe import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( wire [2:0][31:0] t0, t1; for (genvar a = 0; a < 3; ++a) begin : g_slab VX_fma_unit #( - .USE_DSP (`VX_CFG_RTU_USE_DSP), // vendor xil_fma on Vivado (soft in sim), like the FPU - .LATENCY (LAT_SLAB), + .USE_DSP (`VX_CFG_RTU_USE_DSP), + .LATENCY (F), .SUBNORM_ENABLE (0), - .EXCEPT_ENABLE (0) + .EXCEPT_ENABLE (1) ) fma_t0 ( .clk (clk), .reset (reset), @@ -249,15 +276,15 @@ module VX_rtu_box_pe import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( .frm (INST_FRM_RNE), .dataa (dmn[a]), .datab (inv_d_q[a]), - .datac (32'h0), + .datac (F32_NEG0), .result (t0[a]), `UNUSED_PIN (fflags) ); VX_fma_unit #( - .USE_DSP (`VX_CFG_RTU_USE_DSP), // vendor xil_fma on Vivado (soft in sim), like the FPU - .LATENCY (LAT_SLAB), + .USE_DSP (`VX_CFG_RTU_USE_DSP), + .LATENCY (F), .SUBNORM_ENABLE (0), - .EXCEPT_ENABLE (0) + .EXCEPT_ENABLE (1) ) fma_t1 ( .clk (clk), .reset (reset), @@ -268,59 +295,17 @@ module VX_rtu_box_pe import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( .frm (INST_FRM_RNE), .dataa (dmx[a]), .datab (inv_d_q[a]), - .datac (32'h0), + .datac (F32_NEG0), .result (t1[a]), `UNUSED_PIN (fflags) ); end - // ── stage 4: per-axis lo/hi ─────────────────────────────────────── - // VX_fncp_unit returns an XLEN-wide result (it also serves the - // integer-returning compare and class ops); the traversal math is fp32, so - // every min/max result is taken from the low word. - wire [2:0][`VX_CFG_XLEN-1:0] lo_res, hi_res; - `UNUSED_VAR ({lo_res, hi_res}) - wire [2:0][31:0] lo, hi; - for (genvar a = 0; a < 3; ++a) begin : g_minmax - VX_fncp_unit #( - .LATENCY (FNCP_SIZE) - ) fncp_lo ( - .clk (clk), - .reset (reset), - .enable (enable), - .mask (1'b1), - .op_type (INST_FPU_MISC), - .fmt ('0), - .frm (3'd6 /*FMIN*/), - .dataa (t0[a]), - .datab (t1[a]), - .result (lo_res[a]), - `UNUSED_PIN (fflags) - ); - VX_fncp_unit #( - .LATENCY (FNCP_SIZE) - ) fncp_hi ( - .clk (clk), - .reset (reset), - .enable (enable), - .mask (1'b1), - .op_type (INST_FPU_MISC), - .fmt ('0), - .frm (3'd7 /*FMAX*/), - .dataa (t0[a]), - .datab (t1[a]), - .result (hi_res[a]), - `UNUSED_PIN (fflags) - ); - assign lo[a] = lo_res[a][31:0]; - assign hi[a] = hi_res[a][31:0]; - end - - // t_min/t_max delayed to align with lo/hi + // t_min/t_max delayed to the verdict stage wire [31:0] tmin_r, tmax_r; VX_shift_register #( .DATAW (64), - .DEPTH (LAT_ORIGIN + LAT_DEQUANT + LAT_SLAB + LAT_MINMAX) + .DEPTH (LAT_FMA + 3) ) sr_t ( .clk (clk), .reset (reset), @@ -329,140 +314,45 @@ module VX_rtu_box_pe import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( .data_out ({tmin_r, tmax_r}) ); - // ── stage 5: reduce — t_near = max(tmin, lo[*]), t_far = min(tmax, hi[*]) ── - wire [`VX_CFG_XLEN-1:0] near_a_res, near_b_res, far_a_res, far_b_res; - `UNUSED_VAR ({near_a_res, near_b_res, far_a_res, far_b_res}) - wire [31:0] near_a = near_a_res[31:0]; // first reduce level - wire [31:0] near_b = near_b_res[31:0]; - wire [31:0] far_a = far_a_res[31:0]; - wire [31:0] far_b = far_b_res[31:0]; - VX_fncp_unit #( - .LATENCY (FNCP_SIZE) - ) r_near_a ( - .clk (clk), - .reset (reset), - .enable (enable), - .mask (1'b1), - .op_type (INST_FPU_MISC), - .fmt ('0), - .frm (3'd7), - .dataa (lo[0]), - .datab (lo[1]), - .result (near_a_res), - `UNUSED_PIN (fflags) - ); - VX_fncp_unit #( - .LATENCY (FNCP_SIZE) - ) r_near_b ( - .clk (clk), - .reset (reset), - .enable (enable), - .mask (1'b1), - .op_type (INST_FPU_MISC), - .fmt ('0), - .frm (3'd7), - .dataa (lo[2]), - .datab (tmin_r), - .result (near_b_res), - `UNUSED_PIN (fflags) - ); - VX_fncp_unit #( - .LATENCY (FNCP_SIZE) - ) r_far_a ( - .clk (clk), - .reset (reset), - .enable (enable), - .mask (1'b1), - .op_type (INST_FPU_MISC), - .fmt ('0), - .frm (3'd6), - .dataa (hi[0]), - .datab (hi[1]), - .result (far_a_res), - `UNUSED_PIN (fflags) - ); - VX_fncp_unit #( - .LATENCY (FNCP_SIZE) - ) r_far_b ( - .clk (clk), - .reset (reset), - .enable (enable), - .mask (1'b1), - .op_type (INST_FPU_MISC), - .fmt ('0), - .frm (3'd6), - .dataa (hi[2]), - .datab (tmax_r), - .result (far_b_res), - `UNUSED_PIN (fflags) - ); - - wire [`VX_CFG_XLEN-1:0] t_near_res, t_far_res; - `UNUSED_VAR ({t_near_res, t_far_res}) - wire [31:0] t_near_w = t_near_res[31:0]; // second reduce level - wire [31:0] t_far_w = t_far_res[31:0]; - VX_fncp_unit #( - .LATENCY (FNCP_SIZE) - ) r_near ( - .clk (clk), - .reset (reset), - .enable (enable), - .mask (1'b1), - .op_type (INST_FPU_MISC), - .fmt ('0), - .frm (3'd7), - .dataa (near_a), - .datab (near_b), - .result (t_near_res), - `UNUSED_PIN (fflags) - ); - VX_fncp_unit #( - .LATENCY (FNCP_SIZE) - ) r_far ( - .clk (clk), - .reset (reset), - .enable (enable), - .mask (1'b1), - .op_type (INST_FPU_MISC), - .fmt ('0), - .frm (3'd6), - .dataa (far_a), - .datab (far_b), - .result (t_far_res), - `UNUSED_PIN (fflags) - ); + // ── stage 4: per-axis lo/hi; an axis whose slabs are both NaN drops out ── + reg [2:0][31:0] lo_r, hi_r; + always_ff @(posedge clk) begin + if (enable) begin + for (integer a = 0; a < 3; ++a) begin + lo_r[a] <= f32_min(t0[a], t1[a], F32_NINF); + hi_r[a] <= f32_max(t0[a], t1[a], F32_PINF); + end + end + end - // ── stage 6: hit = (t_near <= t_far) ────────────────────────────── - wire [`VX_CFG_XLEN-1:0] cmp_res; - `UNUSED_VAR (cmp_res) - VX_fncp_unit #( - .LATENCY (FNCP_SIZE) - ) fncp_cmp ( - .clk (clk), - .reset (reset), - .enable (enable), - .mask (1'b1), - .op_type (INST_FPU_CMP), - .fmt ('0), - .frm (3'd0 /*LE*/), - .dataa (t_near_w), - .datab (t_far_w), - .result (cmp_res), - `UNUSED_PIN (fflags) - ); + // ── stage 5/6: lo = max over axes, hi = min over axes ───────────── + reg [31:0] near_a_r, near_b_r, far_a_r, far_b_r; + reg [31:0] lo_all_r, hi_all_r; + always_ff @(posedge clk) begin + if (enable) begin + near_a_r <= f32_max(lo_r[0], lo_r[1], F32_NINF); + near_b_r <= lo_r[2]; + far_a_r <= f32_min(hi_r[0], hi_r[1], F32_PINF); + far_b_r <= hi_r[2]; + lo_all_r <= f32_max(near_a_r, near_b_r, F32_NINF); + hi_all_r <= f32_min(far_a_r, far_b_r, F32_PINF); + end + end - // carry t_near alongside the compare result, plus the overall valid pipe - wire [31:0] t_near_cmp; - VX_shift_register #( - .DATAW (32), - .DEPTH (LAT_CMP) - ) sr_tnear ( - .clk (clk), - .reset (reset), - .enable (enable), - .data_in (t_near_w), - .data_out (t_near_cmp) - ); + // ── stage 7: hit = hi >= max(0, lo) && lo <= t_max; t_near = max(t_min, lo) + // lo and hi are never NaN; t_near is non-negative for t_min >= 0 and +0 for + // a zero, so the consumer may order it as an unsigned integer. + wire hi_ge0 = !hi_all_r[31] || (hi_all_r[30:0] == 31'd0); + wire hit_w = hi_ge0 && f32_le(lo_all_r, hi_all_r) && f32_le(lo_all_r, tmax_r); + wire [31:0] tnear_w = f32_max(tmin_r, lo_all_r, lo_all_r); + reg hit_r; + reg [31:0] t_near_r; + always_ff @(posedge clk) begin + if (enable) begin + hit_r <= hit_w; + t_near_r <= (tnear_w[30:0] == 31'd0) ? 32'd0 : tnear_w; + end + end reg [LATENCY-1:0] valid_pipe_r; always_ff @(posedge clk) begin @@ -503,7 +393,7 @@ module VX_rtu_box_pe import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( assign valid_out = valid_pipe_r[LATENCY-1]; assign tag_out = tag_out_w; assign tag_out_pre = tag_out_pre_w; - assign hit = cmp_res[0]; - assign t_near = t_near_cmp; + assign hit = hit_r; + assign t_near = t_near_r; endmodule diff --git a/hw/rtl/rtu/VX_rtu_recip.sv b/hw/rtl/rtu/VX_rtu_recip.sv index d5c2094cc3..29440da925 100644 --- a/hw/rtl/rtu/VX_rtu_recip.sv +++ b/hw/rtl/rtu/VX_rtu_recip.sv @@ -21,6 +21,10 @@ // map to DSP48. Trades ~2K LUT/unit onto the idle BRAM + DSP blocks. // ~9e-8 max relative error (well inside the RTU's 1e-4 tolerance). // +// A zero (or, flushed, subnormal) operand returns +FLT_MAX rather than inf, as +// the box test's reference does: a slab along a zero direction component then +// stays finite, (b - o) * FLT_MAX, instead of 0 * inf = NaN. +// // The input is presented combinationally and held stable for the whole setup // span by the scheduler; the result is a fixed-latency pipeline output, valid // after the backend's pipeline depth (<= the scheduler's SETUP_LAT wait). @@ -36,8 +40,10 @@ module VX_rtu_recip import VX_gpu_pkg::*, VX_fpu_pkg::*; #( input wire enable, input wire mask, input wire [31:0] x, // operand (dir component) - output wire [31:0] result // 1 / x + output wire [31:0] result // 1 / x (+FLT_MAX for x == 0) ); + localparam [31:0] F32_MAX = 32'h7F7FFFFF; + if (DSP_SEED != 0) begin : g_dsp_seed `UNUSED_VAR (mask) // ── seed ROM: 1/a for a = 1.fraction in [1,2), indexed by the top 10 @@ -68,16 +74,17 @@ module VX_rtu_recip import VX_gpu_pkg::*, VX_fpu_pkg::*; #( wire [22:0] s0_frac = x[22:0]; wire [23:0] s0_A = {1'b1, s0_frac}; // significand a*2^23 wire [30:0] s0_afx = {s0_A, 7'b0}; // a in Q2.30 - wire s0_inf = (s0_exp == 8'h00); // 1/0 -> inf - wire s0_zero = (s0_exp == 8'hFF); // 1/inf -> 0 + wire s0_inf = (s0_exp == 8'h00); // 1/0 -> FLT_MAX + wire s0_zero = (s0_exp == 8'hFF); // 1/inf -> 0, 1/NaN -> NaN + wire s0_nan = s0_zero && (s0_frac != 23'd0); wire [KIDX-1:0] s0_idx = s0_frac[22 -: KIDX]; - reg s1_sign, s1_inf, s1_zero; + reg s1_sign, s1_inf, s1_zero, s1_nan; reg [7:0] s1_exp; reg [30:0] s1_afx; reg [31:0] s1_y; // seed, Q1.31 always_ff @(posedge clk) if (enable) begin - s1_sign <= s0_sign; s1_inf <= s0_inf; s1_zero <= s0_zero; + s1_sign <= s0_sign; s1_inf <= s0_inf; s1_zero <= s0_zero; s1_nan <= s0_nan; s1_exp <= s0_exp; s1_afx <= s0_afx; s1_y <= seed_rom[s0_idx]; // registered ROM read -> BRAM end @@ -86,23 +93,23 @@ module VX_rtu_recip import VX_gpu_pkg::*, VX_fpu_pkg::*; #( wire [62:0] s1_ay = s1_afx * s1_y; // -> DSP (31b * 32b) wire [31:0] s1_p = 32'(s1_ay >> 31); // a*y (Q2.30) wire [31:0] s1_t = 32'h8000_0000 - s1_p; // 2 - p (2 == 2^31 in Q2.30) - reg s2_sign, s2_inf, s2_zero; + reg s2_sign, s2_inf, s2_zero, s2_nan; reg [7:0] s2_exp; reg [30:0] s2_afx; reg [31:0] s2_y0, s2_t; always_ff @(posedge clk) if (enable) begin - s2_sign <= s1_sign; s2_inf <= s1_inf; s2_zero <= s1_zero; + s2_sign <= s1_sign; s2_inf <= s1_inf; s2_zero <= s1_zero; s2_nan <= s1_nan; s2_exp <= s1_exp; s2_afx <= s1_afx; s2_y0 <= s1_y; s2_t <= s1_t; end wire [63:0] s2_yt = s2_y0 * s2_t; // -> DSP (32b * 32b) wire [31:0] s2_y1 = 32'(s2_yt >> 30); // y*(2-a*y) (Q1.31) - reg s3_sign, s3_inf, s3_zero; + reg s3_sign, s3_inf, s3_zero, s3_nan; reg [7:0] s3_exp; reg [30:0] s3_afx; reg [31:0] s3_y1; always_ff @(posedge clk) if (enable) begin - s3_sign <= s2_sign; s3_inf <= s2_inf; s3_zero <= s2_zero; + s3_sign <= s2_sign; s3_inf <= s2_inf; s3_zero <= s2_zero; s3_nan <= s2_nan; s3_exp <= s2_exp; s3_afx <= s2_afx; s3_y1 <= s2_y1; end @@ -110,20 +117,20 @@ module VX_rtu_recip import VX_gpu_pkg::*, VX_fpu_pkg::*; #( wire [62:0] s3_ay = s3_afx * s3_y1; // -> DSP (31b * 32b) wire [31:0] s3_p = 32'(s3_ay >> 31); wire [31:0] s3_t = 32'h8000_0000 - s3_p; - reg s4_sign, s4_inf, s4_zero; + reg s4_sign, s4_inf, s4_zero, s4_nan; reg [7:0] s4_exp; reg [31:0] s4_y1, s4_t; always_ff @(posedge clk) if (enable) begin - s4_sign <= s3_sign; s4_inf <= s3_inf; s4_zero <= s3_zero; + s4_sign <= s3_sign; s4_inf <= s3_inf; s4_zero <= s3_zero; s4_nan <= s3_nan; s4_exp <= s3_exp; s4_y1 <= s3_y1; s4_t <= s3_t; end wire [63:0] s4_yt = s4_y1 * s4_t; // -> DSP (32b * 32b) wire [31:0] s4_y2 = 32'(s4_yt >> 30); // 1/a in Q1.31 - reg s5_sign, s5_inf, s5_zero; + reg s5_sign, s5_inf, s5_zero, s5_nan; reg [7:0] s5_exp; reg [31:0] s5_y2; always_ff @(posedge clk) if (enable) begin - s5_sign <= s4_sign; s5_inf <= s4_inf; s5_zero <= s4_zero; + s5_sign <= s4_sign; s5_inf <= s4_inf; s5_zero <= s4_zero; s5_nan <= s4_nan; s5_exp <= s4_exp; s5_y2 <= s4_y2; end @@ -135,18 +142,20 @@ module VX_rtu_recip import VX_gpu_pkg::*, VX_fpu_pkg::*; #( wire s5_ovf = s5_fr[23]; // 2y rounded up to 2.0 wire [7:0] s5_expf = s5_ovf ? (s5_exf[7:0] + 8'd1) : s5_exf[7:0]; wire [22:0] s5_frac = s5_ovf ? 23'd0 : s5_fr[22:0]; - assign result = s5_inf ? {s5_sign, 8'hFF, 23'd0} + assign result = s5_inf ? F32_MAX + : s5_nan ? 32'h7FC00000 : s5_zero ? {s5_sign, 8'h00, 23'd0} : {s5_sign, s5_expf, s5_frac}; `UNUSED_PARAM (LATENCY) end else begin : g_lut_nr // portable baseline: 1.0 / x via the shared divide unit — vendor xil_fdiv // on Vivado (USE_DSP=VX_CFG_RTU_USE_DSP, LATENCY 28), soft NR in sim (17). + wire [31:0] quot; VX_fdiv_unit #( .USE_DSP (`VX_CFG_RTU_USE_DSP), .LATENCY (LATENCY), .SUBNORM_ENABLE (0), - .EXCEPT_ENABLE (0) + .EXCEPT_ENABLE (1) ) u_recip ( .clk (clk), .reset (reset), @@ -156,9 +165,21 @@ module VX_rtu_recip import VX_gpu_pkg::*, VX_fpu_pkg::*; #( .frm (INST_FRM_RNE), .dataa (32'h3F800000 /*1.0*/), .datab (x), - .result (result), + .result (quot), `UNUSED_PIN (fflags) ); + wire x_zero_q; + VX_shift_register #( + .DATAW (1), + .DEPTH (LATENCY) + ) sr_zero ( + .clk (clk), + .reset (reset), + .enable (enable), + .data_in (x[30:23] == 8'd0), + .data_out (x_zero_q) + ); + assign result = x_zero_q ? F32_MAX : quot; end endmodule diff --git a/hw/rtl/rtu/VX_rtu_tri_pe.sv b/hw/rtl/rtu/VX_rtu_tri_pe.sv index c01949dc94..8b96f1ae97 100644 --- a/hw/rtl/rtu/VX_rtu_tri_pe.sv +++ b/hw/rtl/rtu/VX_rtu_tri_pe.sv @@ -21,7 +21,7 @@ // F64: pz = sz*rz (exact), w_i = px_a*py_b - py_a*px_b (one rounding) // det = w0 + (w1 + w2), T = (w0*pz0 + w1*pz1) + w2*pz2 // t = f32(T / det), (u, v) = f32(w1, w2) / f32(det) -// hit = !(any w < 0 && any w > 0) && det != 0 && tmin <= t <= tmax +// hit = !(any w < 0 && any w > 0) && det != 0 && tmin < t < tmax // back_facing = det < 0 // // A shared edge evaluates to exactly negated weights in its two triangles, so @@ -172,6 +172,11 @@ module VX_rtu_tri_pe import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( end endfunction + // strict IEEE a < b (+0 == -0); NaN compares false + function automatic f32_lt(input [31:0] a, input [31:0] b); + f32_lt = f32_le(a, b) && !f32_le(b, a); + endfunction + // ── stage A (@0 -> @T_B): axis select ───────────────────────────── wire [30:0] ad0 = dir[0][30:0]; wire [30:0] ad1 = dir[1][30:0]; @@ -215,7 +220,7 @@ module VX_rtu_tri_pe import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( .FLEN (32), .USE_DSP (`VX_CFG_RTU_USE_DSP), .SUBNORM_ENABLE (0), - .EXCEPT_ENABLE (0) + .EXCEPT_ENABLE (1) ) fdiv_sz ( .clk (clk), .reset (reset), @@ -236,7 +241,7 @@ module VX_rtu_tri_pe import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( .LATENCY (F), .USE_DSP (`VX_CFG_RTU_USE_DSP), .SUBNORM_ENABLE (0), - .EXCEPT_ENABLE (0) + .EXCEPT_ENABLE (1) ) fsub_r ( .clk (clk), .reset (reset), @@ -286,7 +291,7 @@ module VX_rtu_tri_pe import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( .LATENCY (F), .USE_DSP (`VX_CFG_RTU_USE_DSP), .SUBNORM_ENABLE (0), - .EXCEPT_ENABLE (0) + .EXCEPT_ENABLE (1) ) fmul_s ( .clk (clk), .reset (reset), @@ -324,7 +329,7 @@ module VX_rtu_tri_pe import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( .LATENCY (F), .USE_DSP (`VX_CFG_RTU_USE_DSP), .SUBNORM_ENABLE (0), - .EXCEPT_ENABLE (0) + .EXCEPT_ENABLE (1) ) fmul_m ( .clk (clk), .reset (reset), @@ -346,7 +351,7 @@ module VX_rtu_tri_pe import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( .EXP_BITS (11), .USE_DSP (`VX_CFG_RTU_USE_DSP), .SUBNORM_ENABLE (0), - .EXCEPT_ENABLE (0) + .EXCEPT_ENABLE (1) ) fmul_pz ( .clk (clk), .reset (reset), @@ -384,7 +389,7 @@ module VX_rtu_tri_pe import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( .LATENCY (F), .USE_DSP (`VX_CFG_RTU_USE_DSP), .SUBNORM_ENABLE (0), - .EXCEPT_ENABLE (0) + .EXCEPT_ENABLE (1) ) fsub_p ( .clk (clk), .reset (reset), @@ -433,7 +438,7 @@ module VX_rtu_tri_pe import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( .EXP_BITS (11), .USE_DSP (`VX_CFG_RTU_USE_DSP), .SUBNORM_ENABLE (0), - .EXCEPT_ENABLE (0) + .EXCEPT_ENABLE (1) ) fmul_c ( .clk (clk), .reset (reset), @@ -454,7 +459,7 @@ module VX_rtu_tri_pe import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( .EXP_BITS (11), .USE_DSP (`VX_CFG_RTU_USE_DSP), .SUBNORM_ENABLE (0), - .EXCEPT_ENABLE (0) + .EXCEPT_ENABLE (1) ) fmsub_w ( .clk (clk), .reset (reset), @@ -497,37 +502,37 @@ module VX_rtu_tri_pe import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( .data_in (w_g[0]), .data_out (w0_g1) ); - VX_fma_unit #(.LATENCY (D), .MAN_BITS (52), .EXP_BITS (11), .USE_DSP (`VX_CFG_RTU_USE_DSP), .SUBNORM_ENABLE (0), .EXCEPT_ENABLE (0)) fadd_det12 ( + VX_fma_unit #(.LATENCY (D), .MAN_BITS (52), .EXP_BITS (11), .USE_DSP (`VX_CFG_RTU_USE_DSP), .SUBNORM_ENABLE (0), .EXCEPT_ENABLE (1)) fadd_det12 ( .clk (clk), .reset (reset), .enable (enable), .mask (1'b1), .op_type (INST_FPU_ADD), .fmt (FMT_ADD), .frm (INST_FRM_RNE), .dataa (w_g[1]), .datab (w_g[2]), .datac ('0), .result (det12), `UNUSED_PIN (fflags) ); - VX_fma_unit #(.LATENCY (D), .MAN_BITS (52), .EXP_BITS (11), .USE_DSP (`VX_CFG_RTU_USE_DSP), .SUBNORM_ENABLE (0), .EXCEPT_ENABLE (0)) fadd_det ( + VX_fma_unit #(.LATENCY (D), .MAN_BITS (52), .EXP_BITS (11), .USE_DSP (`VX_CFG_RTU_USE_DSP), .SUBNORM_ENABLE (0), .EXCEPT_ENABLE (1)) fadd_det ( .clk (clk), .reset (reset), .enable (enable), .mask (1'b1), .op_type (INST_FPU_ADD), .fmt (FMT_ADD), .frm (INST_FRM_RNE), .dataa (w0_g1), .datab (det12), .datac ('0), .result (det_g2), `UNUSED_PIN (fflags) ); - VX_fma_unit #(.LATENCY (D), .MAN_BITS (52), .EXP_BITS (11), .USE_DSP (`VX_CFG_RTU_USE_DSP), .SUBNORM_ENABLE (0), .EXCEPT_ENABLE (0)) fmul_tp0 ( + VX_fma_unit #(.LATENCY (D), .MAN_BITS (52), .EXP_BITS (11), .USE_DSP (`VX_CFG_RTU_USE_DSP), .SUBNORM_ENABLE (0), .EXCEPT_ENABLE (1)) fmul_tp0 ( .clk (clk), .reset (reset), .enable (enable), .mask (1'b1), .op_type (INST_FPU_MUL), .fmt (FMT_ADD), .frm (INST_FRM_RNE), .dataa (w_g[0]), .datab (pz_g[0]), .datac ('0), .result (tp0), `UNUSED_PIN (fflags) ); - VX_fma_unit #(.LATENCY (D), .MAN_BITS (52), .EXP_BITS (11), .USE_DSP (`VX_CFG_RTU_USE_DSP), .SUBNORM_ENABLE (0), .EXCEPT_ENABLE (0)) fmul_tp1 ( + VX_fma_unit #(.LATENCY (D), .MAN_BITS (52), .EXP_BITS (11), .USE_DSP (`VX_CFG_RTU_USE_DSP), .SUBNORM_ENABLE (0), .EXCEPT_ENABLE (1)) fmul_tp1 ( .clk (clk), .reset (reset), .enable (enable), .mask (1'b1), .op_type (INST_FPU_MUL), .fmt (FMT_ADD), .frm (INST_FRM_RNE), .dataa (w_g[1]), .datab (pz_g[1]), .datac ('0), .result (tp1), `UNUSED_PIN (fflags) ); - VX_fma_unit #(.LATENCY (D), .MAN_BITS (52), .EXP_BITS (11), .USE_DSP (`VX_CFG_RTU_USE_DSP), .SUBNORM_ENABLE (0), .EXCEPT_ENABLE (0)) fmul_tp2 ( + VX_fma_unit #(.LATENCY (D), .MAN_BITS (52), .EXP_BITS (11), .USE_DSP (`VX_CFG_RTU_USE_DSP), .SUBNORM_ENABLE (0), .EXCEPT_ENABLE (1)) fmul_tp2 ( .clk (clk), .reset (reset), .enable (enable), .mask (1'b1), .op_type (INST_FPU_MUL), .fmt (FMT_ADD), .frm (INST_FRM_RNE), .dataa (w_g[2]), .datab (pz_g[2]), .datac ('0), .result (tp2), `UNUSED_PIN (fflags) ); - VX_fma_unit #(.LATENCY (D), .MAN_BITS (52), .EXP_BITS (11), .USE_DSP (`VX_CFG_RTU_USE_DSP), .SUBNORM_ENABLE (0), .EXCEPT_ENABLE (0)) fadd_t01 ( + VX_fma_unit #(.LATENCY (D), .MAN_BITS (52), .EXP_BITS (11), .USE_DSP (`VX_CFG_RTU_USE_DSP), .SUBNORM_ENABLE (0), .EXCEPT_ENABLE (1)) fadd_t01 ( .clk (clk), .reset (reset), .enable (enable), .mask (1'b1), .op_type (INST_FPU_ADD), .fmt (FMT_ADD), .frm (INST_FRM_RNE), .dataa (tp0), .datab (tp1), .datac ('0), @@ -543,7 +548,7 @@ module VX_rtu_tri_pe import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( .data_in (tp2), .data_out (tp2_g2) ); - VX_fma_unit #(.LATENCY (D), .MAN_BITS (52), .EXP_BITS (11), .USE_DSP (`VX_CFG_RTU_USE_DSP), .SUBNORM_ENABLE (0), .EXCEPT_ENABLE (0)) fadd_t ( + VX_fma_unit #(.LATENCY (D), .MAN_BITS (52), .EXP_BITS (11), .USE_DSP (`VX_CFG_RTU_USE_DSP), .SUBNORM_ENABLE (0), .EXCEPT_ENABLE (1)) fadd_t ( .clk (clk), .reset (reset), .enable (enable), .mask (1'b1), .op_type (INST_FPU_ADD), .fmt (FMT_ADD), .frm (INST_FRM_RNE), .dataa (t01), .datab (tp2_g2), .datac ('0), @@ -569,7 +574,7 @@ module VX_rtu_tri_pe import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( .LATENCY (V64), .FLEN (64), .SUBNORM_ENABLE (0), - .EXCEPT_ENABLE (0) + .EXCEPT_ENABLE (1) ) fdiv_t ( .clk (clk), .reset (reset), @@ -617,7 +622,7 @@ module VX_rtu_tri_pe import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( .FLEN (32), .USE_DSP (`VX_CFG_RTU_USE_DSP), .SUBNORM_ENABLE (0), - .EXCEPT_ENABLE (0) + .EXCEPT_ENABLE (1) ) fdiv_u ( .clk (clk), .reset (reset), @@ -635,7 +640,7 @@ module VX_rtu_tri_pe import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( .FLEN (32), .USE_DSP (`VX_CFG_RTU_USE_DSP), .SUBNORM_ENABLE (0), - .EXCEPT_ENABLE (0) + .EXCEPT_ENABLE (1) ) fdiv_v ( .clk (clk), .reset (reset), @@ -718,7 +723,8 @@ module VX_rtu_tri_pe import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( ); // ── stage I (@T_I): range test and commit ───────────────────────── - wire range_ok = f32_le(tmm_i[63:32], t_i) && f32_le(t_i, tmm_i[31:0]); + // open interval, as the Vulkan reference commits a triangle hit + wire range_ok = f32_lt(tmm_i[63:32], t_i) && f32_lt(t_i, tmm_i[31:0]); reg hit_r, bf_r; reg [31:0] u_r, v_r, t_r; diff --git a/hw/unittest/rtu_box_pe/Makefile b/hw/unittest/rtu_box_pe/Makefile new file mode 100644 index 0000000000..102b190d2c --- /dev/null +++ b/hw/unittest/rtu_box_pe/Makefile @@ -0,0 +1,31 @@ +ROOT_DIR := $(realpath ../../..) +include $(ROOT_DIR)/config.mk + +PROJECT := rtu_box_pe + +RTL_DIR := $(VORTEX_HOME)/hw/rtl +SRC_DIR := $(VORTEX_HOME)/hw/unittest/$(PROJECT) +SIMX_DIR := $(VORTEX_HOME)/sim/simx + +# The reference is the SimX model itself, so the RTU config must match both sides. +CONFIGS += -DSIMULATION -DVX_CFG_EXT_RTU_ENABLE + +CXXFLAGS := -I$(SRC_DIR) -I$(VORTEX_HOME)/hw/unittest/common -I$(SW_COMMON_DIR) +CXXFLAGS += -I$(SIMX_DIR) -I$(SIMX_DIR)/rtu -I$(SIM_COMMON_DIR) +CXXFLAGS += -I$(ROOT_DIR)/sw -I$(ROOT_DIR)/hw +CXXFLAGS += -I$(THIRD_PARTY_DIR)/softfloat/source/include + +SRCS += $(SIMX_DIR)/rtu/rtu_isect.cpp +SRCS += $(SRC_DIR)/main.cpp + +PARAMS := -GTAG_WIDTH=32 + +RTL_PKGS += $(RTL_DIR)/VX_gpu_pkg.sv $(RTL_DIR)/fpu/VX_fpu_pkg.sv $(RTL_DIR)/rtu/VX_rtu_pkg.sv +RTL_INCLUDE := -I$(ROOT_DIR)/sw -I$(RTL_DIR) -I$(RTL_DIR)/libs -I$(RTL_DIR)/interfaces +RTL_INCLUDE += -I$(RTL_DIR)/fpu -I$(RTL_DIR)/rtu -I$(SRC_DIR) + +VL_FLAGS += -I$(ROOT_DIR)/hw + +TOP := VX_rtu_box_pe_tb + +include ../common.mk diff --git a/hw/unittest/rtu_box_pe/VX_rtu_box_pe_tb.sv b/hw/unittest/rtu_box_pe/VX_rtu_box_pe_tb.sv new file mode 100644 index 0000000000..f89563d573 --- /dev/null +++ b/hw/unittest/rtu_box_pe/VX_rtu_box_pe_tb.sv @@ -0,0 +1,111 @@ +// Copyright © 2019-2023 +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +// Ray setup + box PE as the scheduler composes them: inv_d comes from +// VX_rtu_recip on the ray direction, the rest of the box request waits for it. + +`include "VX_define.vh" + +module VX_rtu_box_pe_tb import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( + parameter TAG_WIDTH = 32 +) ( + input wire clk, + input wire reset, + input wire valid_in, + input wire [TAG_WIDTH-1:0] tag_in, + input wire [2:0][31:0] origin, + input wire [2:0][7:0] exp, + input wire [2:0][7:0] qmin, + input wire [2:0][7:0] qmax, + input wire raw, + input wire [2:0][31:0] raw_min, + input wire [2:0][31:0] raw_max, + input wire [2:0][31:0] ro, + input wire [2:0][31:0] dir, + input wire [31:0] t_min, + input wire [31:0] t_max, + + // the reciprocal stage, observable on its own + output wire inv_valid, + output wire [TAG_WIDTH-1:0] inv_tag, + output wire [2:0][31:0] inv_d, + + output wire valid_out, + output wire [TAG_WIDTH-1:0] tag_out, + output wire hit, + output wire [31:0] t_near +); + localparam LAT = RTU_FDIV_LAT; + localparam REQW = 1 + TAG_WIDTH + 3*32 + 3*8*3 + 1 + 3*32*3 + 2*32; + + for (genvar a = 0; a < 3; ++a) begin : g_recip + VX_rtu_recip #( + .LATENCY (LAT), + .DSP_SEED (`VX_CFG_RTU_RECIP_DSP_SEED) + ) recip ( + .clk (clk), + .reset (reset), + .enable (1'b1), + .mask (1'b1), + .x (dir[a]), + .result (inv_d[a]) + ); + end + + wire valid_d, raw_d; + wire [TAG_WIDTH-1:0] tag_d; + wire [2:0][31:0] origin_d, raw_min_d, raw_max_d, ro_d; + wire [2:0][7:0] exp_d, qmin_d, qmax_d; + wire [31:0] t_min_d, t_max_d; + VX_shift_register #( + .DATAW (REQW), + .RESETW (1), + .DEPTH (LAT) + ) sr_req ( + .clk (clk), + .reset (reset), + .enable (1'b1), + .data_in ({valid_in, tag_in, origin, exp, qmin, qmax, raw, raw_min, raw_max, ro, t_min, t_max}), + .data_out ({valid_d, tag_d, origin_d, exp_d, qmin_d, qmax_d, raw_d, raw_min_d, raw_max_d, ro_d, t_min_d, t_max_d}) + ); + + assign inv_valid = valid_d; + assign inv_tag = tag_d; + + VX_rtu_box_pe #( + .TAG_WIDTH (TAG_WIDTH) + ) box_pe ( + .clk (clk), + .reset (reset), + .enable (1'b1), + .valid_in (valid_d), + .tag_in (tag_d), + .origin (origin_d), + .exp (exp_d), + .qmin (qmin_d), + .qmax (qmax_d), + .raw (raw_d), + .raw_min (raw_min_d), + .raw_max (raw_max_d), + .ro (ro_d), + .inv_d (inv_d), + .t_min (t_min_d), + .t_max (t_max_d), + .valid_out (valid_out), + .tag_out (tag_out), + `UNUSED_PIN (tag_out_pre), + .hit (hit), + .t_near (t_near) + ); + +endmodule diff --git a/hw/unittest/rtu_box_pe/main.cpp b/hw/unittest/rtu_box_pe/main.cpp new file mode 100644 index 0000000000..9bcaef590a --- /dev/null +++ b/hw/unittest/rtu_box_pe/main.cpp @@ -0,0 +1,492 @@ +// VX_rtu_recip + VX_rtu_box_pe against SimX's box test (reconstruct_child_aabb +// + rtu::ray_aabb_intersect): inv_d bit for bit, every accept decision, t_near +// by value, and the order the scheduler's insertion collector gives a node's +// accepted children against SimX's nearest-first sort. +// +// The PE's FP units flush subnormals (FTZ/DAZ) while SimX runs IEEE. A case +// whose RTL result differs from SimX but equals SimX evaluated under the host's +// FTZ/DAZ mode, and only there, is counted as a subnormal flush, not a mismatch. + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include "VVX_rtu_box_pe_tb.h" +#include "verilated.h" +#include "rtu_isect.h" + +namespace { + +constexpr int kMaxChildren = 6; + +struct Ray { + float o[3], d[3], tmin, tmax; +}; + +struct Child { + uint8_t qmin[3], qmax[3]; + float rmin[3], rmax[3]; +}; + +struct Node { + Ray ray; + float origin[3]; + int8_t exp[3]; + bool raw = false; + int n = 0; + Child ch[kMaxChildren]; + int cat = 0; +}; + +const char* const kCatNames[] = { + "random", "random raw", "zero dir on slab", "zero dir off slab", "flat box", + "tmin past exit", "tmax == entry", "tmax < entry", "exit == 0", "-0 operand", + "special origin", "special ro", "special dir", "special raw", "special tmin", + "special tmax", "exp extremes", +}; +constexpr int kNumCats = int(sizeof(kCatNames) / sizeof(kCatNames[0])); + +uint32_t bits(float f) { uint32_t b; std::memcpy(&b, &f, 4); return b; } +float fbits(uint32_t b) { float f; std::memcpy(&f, &b, 4); return f; } + +bool is_nan_bits(uint32_t b) { return ((b >> 23) & 0xff) == 0xff && (b & 0x7fffff) != 0; } + +// bit-exact, except that any NaN matches any NaN (the payload is not specified) +bool same_bits(uint32_t rtl, float ref) { + return std::isnan(ref) ? is_nan_bits(rtl) : rtl == bits(ref); +} + +// IEEE value equality (+0 == -0), any NaN matches any NaN +bool same_value(uint32_t rtl, float ref) { + return std::isnan(ref) ? is_nan_bits(rtl) : fbits(rtl) == ref; +} + +// SimX rtu_walker.cpp reconstruct_child_aabb +void reconstruct_child_aabb(const float origin[3], const int8_t exp[3], + const uint8_t qmin[3], const uint8_t qmax[3], + float out_mn[3], float out_mx[3]) { + for (int i = 0; i < 3; ++i) { + float scale = std::ldexp(1.0f, exp[i]); + out_mn[i] = origin[i] + static_cast(qmin[i]) * scale; + out_mx[i] = origin[i] + static_cast(qmax[i]) * scale; + } +} + +struct BoxResult { + bool hit = false; + float t_near = 0; +}; + +struct Ref { + float inv[3]; + BoxResult box; +}; + +Ref reference(const Node& nd, int c, bool ftz) { + const unsigned csr = _mm_getcsr(); + if (ftz) _mm_setcsr(csr | 0x8040); // FTZ | DAZ + Ref r; + for (int i = 0; i < 3; ++i) + r.inv[i] = (nd.ray.d[i] == 0.0f) ? FLT_MAX : 1.0f / nd.ray.d[i]; + float mn[3], mx[3]; + if (nd.raw) { + std::memcpy(mn, nd.ch[c].rmin, sizeof mn); + std::memcpy(mx, nd.ch[c].rmax, sizeof mx); + } else { + reconstruct_child_aabb(nd.origin, nd.exp, nd.ch[c].qmin, nd.ch[c].qmax, mn, mx); + } + r.box.hit = vortex::rtu::ray_aabb_intersect(nd.ray.o, nd.ray.d, mn, mx, + nd.ray.tmin, nd.ray.tmax, r.box.t_near); + _mm_setcsr(csr); + return r; +} + +bool same_box(bool hit, uint32_t t_near, const BoxResult& r) { + return hit == r.hit && (!r.hit || same_value(t_near, r.t_near)); +} + +bool same_ref(const Ref& a, const Ref& b) { + for (int i = 0; i < 3; ++i) + if (!same_bits(bits(a.inv[i]), b.inv[i])) return false; + return same_box(a.box.hit, bits(a.box.t_near), b.box); +} + +// SimX walker: accepted children, insertion-sorted by t_near (stable). +std::vector simx_order(const std::vector& r) { + std::vector o; + for (int i = 0; i < int(r.size()); ++i) { + if (!r[i].hit) continue; + size_t j = o.size(); + o.push_back(i); + while (j > 0 && r[o[j - 1]].t_near > r[i].t_near) { o[j] = o[j - 1]; --j; } + o[j] = i; + } + return o; +} + +// RTL scheduler collector: results arrive in child order; each goes after every +// collected entry whose t_near is <= its own as unsigned bits. +std::vector rtl_order(const std::vector& hit, const std::vector& t) { + std::vector o; + for (int i = 0; i < int(hit.size()); ++i) { + if (!hit[i]) continue; + size_t j = 0; + while (j < o.size() && t[o[j]] <= t[i]) ++j; + o.insert(o.begin() + j, i); + } + return o; +} + +std::mt19937 rng(11); +float uni(float lo, float hi) { return std::uniform_real_distribution(lo, hi)(rng); } +int irand(int lo, int hi) { return std::uniform_int_distribution(lo, hi)(rng); } +bool chance(int n) { return irand(0, n - 1) == 0; } + +void random_children(Node& nd, int n) { + nd.n = n; + for (int c = 0; c < n; ++c) { + for (int a = 0; a < 3; ++a) { + int lo = irand(0, 255), hi = chance(10) ? lo : irand(lo, 255); + nd.ch[c].qmin[a] = uint8_t(lo); + nd.ch[c].qmax[a] = uint8_t(hi); + float r0 = uni(-100, 100), r1 = chance(10) ? r0 : r0 + uni(0, 50); + nd.ch[c].rmin[a] = r0; + nd.ch[c].rmax[a] = r1; + } + } +} + +// node box corner along axis a, at quantized coordinate q +float node_coord(const Node& nd, int a, float q) { + return nd.origin[a] + q * std::ldexp(1.0f, nd.exp[a]); +} + +Node make_random(uint32_t i) { + Node nd; + nd.raw = (i % 10 == 0); + nd.cat = nd.raw ? 1 : 0; + const float S = (i % 3 == 0) ? 1.f : ((i % 3 == 1) ? 100.f : 1e4f); + for (int a = 0; a < 3; ++a) { + nd.origin[a] = uni(-S, S); + nd.exp[a] = int8_t(chance(50) ? irand(-128, 127) : irand(-15, 5)); + } + random_children(nd, irand(1, kMaxChildren)); + if (chance(3)) { // overlapping siblings, so several are accepted together + for (int c = 0; c < nd.n; ++c) + for (int a = 0; a < 3; ++a) { + nd.ch[c].qmin[a] = uint8_t(irand(0, 120)); + nd.ch[c].qmax[a] = uint8_t(irand(130, 255)); + nd.ch[c].rmin[a] = uni(-100, -10); + nd.ch[c].rmax[a] = uni(10, 100); + } + } + Ray& r = nd.ray; + // half the rays aim into one child, the rest anywhere in the node + float target[3]; + const int aim = chance(2) ? irand(0, nd.n - 1) : -1; + for (int a = 0; a < 3; ++a) { + const Child& ch = nd.ch[aim < 0 ? 0 : aim]; + const float f = uni(0, 1); + if (aim < 0) + target[a] = nd.raw ? uni(-100, 150) : node_coord(nd, a, uni(0, 255)); + else + target[a] = nd.raw ? ch.rmin[a] + f * (ch.rmax[a] - ch.rmin[a]) + : node_coord(nd, a, ch.qmin[a] + f * float(ch.qmax[a] - ch.qmin[a])); + const float ext = nd.raw ? 150.f : 255.f * std::ldexp(1.0f, nd.exp[a]); + r.o[a] = chance(3) ? target[a] + uni(-0.3f, 0.3f) * ext // inside / near + : target[a] + uni(-3.f, 3.f) * ext; + } + for (int a = 0; a < 3; ++a) { + r.d[a] = target[a] - r.o[a]; + if (chance(4)) r.d[a] = uni(-1, 1); + if (chance(8)) r.d[a] = chance(2) ? 0.f : -0.f; + } + r.tmin = chance(4) ? uni(0, 2) : 0.f; + const int tm = irand(0, 3); + r.tmax = (tm == 0) ? INFINITY : (tm == 1) ? 1e30f : uni(0, 3); + return nd; +} + +std::vector directed_nodes() { + std::vector out; + const float inf = INFINITY, nan = NAN, sub = 1e-40f; + auto one = [](Node nd, int cat) { nd.n = 1; nd.cat = cat; return nd; }; + for (uint32_t i = 0; i < 3000; ++i) { + Node base = make_random(i * 7 + 1); + base.raw = (i % 5 == 0); + random_children(base, 1); + const Child& ch = base.ch[0]; + float mn[3], mx[3]; + if (base.raw) { + std::memcpy(mn, ch.rmin, sizeof mn); std::memcpy(mx, ch.rmax, sizeof mx); + } else { + reconstruct_child_aabb(base.origin, base.exp, ch.qmin, ch.qmax, mn, mx); + } + const int a = int(i % 3); + // zero direction component with the origin exactly on / off a slab plane + { + Node nd = base; + nd.ray.d[a] = (i & 8) ? -0.f : 0.f; + nd.ray.o[a] = (i & 16) ? mx[a] : mn[a]; + out.push_back(one(nd, 2)); + nd.ray.o[a] = (i & 16) ? std::nextafter(mx[a], inf) : std::nextafter(mn[a], -inf); + out.push_back(one(nd, 3)); + } + // flat box crossed by the ray + { + Node nd = base; + nd.ch[0].qmax[a] = nd.ch[0].qmin[a]; + nd.ch[0].rmax[a] = nd.ch[0].rmin[a]; + for (int k = 0; k < 3; ++k) + nd.ray.d[k] = 0.5f * ((k == a ? mn[k] : 0.5f * (mn[k] + mx[k])) - nd.ray.o[k]); + out.push_back(one(nd, 4)); + } + // aimed at the box centre; tmin / tmax placed on the slab interval ends + { + Node nd = base; + for (int k = 0; k < 3; ++k) nd.ray.d[k] = 0.5f * (mn[k] + mx[k]) - nd.ray.o[k]; + nd.ray.tmin = 0.f; nd.ray.tmax = inf; + Ref r = reference(nd, 0, false); + if (r.box.hit) { + // the slab exit: shrink tmax until the box drops out + Node e = nd; + e.ray.tmin = 1e30f; // past the exit + out.push_back(one(e, 5)); + e = nd; e.ray.tmax = r.box.t_near; // entry, tmin = 0 + out.push_back(one(e, 6)); + e.ray.tmax = std::nextafter(r.box.t_near, -inf); + out.push_back(one(e, 7)); + } + // ray leaving the box through the face it starts on: exit at t == 0 + Node x = nd; + x.ray.o[a] = mx[a]; + for (int k = 0; k < 3; ++k) if (k != a) x.ray.o[k] = 0.5f * (mn[k] + mx[k]); + x.ray.d[a] = 1.f; + out.push_back(one(x, 8)); + x.ray.d[a] = -1.f; + out.push_back(one(x, 8)); + } + // -0 in each operand + for (int slot = 0; slot < 6; ++slot) { + Node nd = base; + if (slot == 0) nd.origin[a] = -0.f; + if (slot == 1) nd.ray.o[a] = -0.f; + if (slot == 2) nd.ray.d[a] = -0.f; + if (slot == 3) { nd.ch[0].rmin[a] = -0.f; nd.ch[0].rmax[a] = 0.f; } + if (slot == 4) nd.ray.tmin = -0.f; + if (slot == 5) nd.ray.tmax = -0.f; + out.push_back(one(nd, 9)); + } + } + // special operands in every slot + const float specials[] = { nan, -nan, inf, -inf, 0.f, -0.f, sub, -sub, FLT_MAX, -FLT_MAX }; + for (uint32_t i = 0; i < 300; ++i) { + Node base = make_random(i * 13 + 2); + random_children(base, 1); + for (float sp : specials) { + for (int a = 0; a < 3; ++a) { + Node nd = base; + nd.raw = false; nd.origin[a] = sp; out.push_back(one(nd, 10)); + nd = base; nd.ray.o[a] = sp; out.push_back(one(nd, 11)); + nd = base; nd.ray.d[a] = sp; out.push_back(one(nd, 12)); + nd = base; nd.raw = true; nd.ch[0].rmin[a] = sp; out.push_back(one(nd, 13)); + nd = base; nd.raw = true; nd.ch[0].rmax[a] = sp; out.push_back(one(nd, 13)); + } + Node nd = base; + nd.ray.tmin = sp; out.push_back(one(nd, 14)); + nd = base; nd.ray.tmax = sp; out.push_back(one(nd, 15)); + } + } + // dequant exponents at and past the F32 range + const int exps[] = { -128, -127, -126, -125, -120, 100, 120, 126, 127 }; + const int qs[] = { 0, 1, 2, 3, 128, 255 }; + for (uint32_t i = 0; i < 200; ++i) { + Node base = make_random(i * 17 + 3); + base.raw = false; + random_children(base, 1); + for (int e : exps) { + for (int q : qs) { + Node nd = base; + const int a = int(i % 3); + nd.exp[a] = int8_t(e); + nd.ch[0].qmin[a] = uint8_t(i & 1 ? 0 : q); + nd.ch[0].qmax[a] = uint8_t(q); + if (i & 2) nd.origin[a] = (i & 4) ? -FLT_MAX : 0.f; + out.push_back(one(nd, 16)); + } + } + } + return out; +} + +} // namespace + +int main(int argc, char** argv) { + Verilated::commandArgs(argc, argv); + const uint32_t N = (argc > 1) ? uint32_t(std::atoi(argv[1])) : 1000000; + VVX_rtu_box_pe_tb dut; + dut.clk = 0; dut.reset = 1; dut.valid_in = 0; + auto tick = [&] { dut.clk = 0; dut.eval(); dut.clk = 1; dut.eval(); }; + for (int i = 0; i < 4; ++i) tick(); + dut.reset = 0; + + std::vector nodes = directed_nodes(); + const uint32_t ND = uint32_t(nodes.size()); + uint32_t random_boxes = 0; + for (uint32_t i = 0; random_boxes < N; ++i) { + nodes.push_back(make_random(i)); + random_boxes += uint32_t(nodes.back().n); + } + + struct Item { uint32_t node; int child; Ref ieee, ftz; }; + std::vector items; + for (uint32_t k = 0; k < nodes.size(); ++k) + for (int c = 0; c < nodes[k].n; ++c) + items.push_back({k, c, reference(nodes[k], c, false), reference(nodes[k], c, true)}); + + // per node: RTL results, and whether each child matched only the FTZ reference + std::vector> rtl_hit(nodes.size()); + std::vector> rtl_t(nodes.size()); + std::vector node_ftz(nodes.size(), false); + for (uint32_t k = 0; k < nodes.size(); ++k) { + rtl_hit[k].assign(size_t(nodes[k].n), false); + rtl_t[k].assign(size_t(nodes[k].n), 0); + } + + uint32_t cat_cases[kNumCats] = {}, cat_err[kNumCats] = {}, cat_ftz[kNumCats] = {}; + uint32_t errors = 0, flushed = 0, inv_errors = 0, hits = 0; + uint32_t order_multi = 0, order_checked = 0, order_errors = 0, order_ftz = 0; + std::deque inv_q, out_q; + std::vector inv_ok(items.size(), 0); + uint32_t sent = 0, checked = 0; + const uint32_t total = uint32_t(items.size()); + + auto report_case = [&](const Item& it, const char* what) { + const Node& nd = nodes[it.node]; + const Child& ch = nd.ch[it.child]; + std::printf(" [%s] %s raw=%d origin=(%a %a %a) exp=(%d %d %d) q=(%u %u %u)-(%u %u %u) " + "rmin=(%a %a %a) rmax=(%a %a %a) o=(%a %a %a) d=(%a %a %a) tmin=%a tmax=%a\n", + kCatNames[nd.cat], what, nd.raw, nd.origin[0], nd.origin[1], nd.origin[2], + nd.exp[0], nd.exp[1], nd.exp[2], ch.qmin[0], ch.qmin[1], ch.qmin[2], + ch.qmax[0], ch.qmax[1], ch.qmax[2], ch.rmin[0], ch.rmin[1], ch.rmin[2], + ch.rmax[0], ch.rmax[1], ch.rmax[2], nd.ray.o[0], nd.ray.o[1], nd.ray.o[2], + nd.ray.d[0], nd.ray.d[1], nd.ray.d[2], nd.ray.tmin, nd.ray.tmax); + }; + + while (checked < total) { + if (sent < total) { + const Item& it = items[sent]; + const Node& nd = nodes[it.node]; + const Child& ch = nd.ch[it.child]; + for (int a = 0; a < 3; ++a) { + dut.origin[a] = bits(nd.origin[a]); + dut.raw_min[a] = bits(ch.rmin[a]); + dut.raw_max[a] = bits(ch.rmax[a]); + dut.ro[a] = bits(nd.ray.o[a]); + dut.dir[a] = bits(nd.ray.d[a]); + } + dut.exp = uint32_t(uint8_t(nd.exp[0])) | uint32_t(uint8_t(nd.exp[1])) << 8 + | uint32_t(uint8_t(nd.exp[2])) << 16; + dut.qmin = uint32_t(ch.qmin[0]) | uint32_t(ch.qmin[1]) << 8 | uint32_t(ch.qmin[2]) << 16; + dut.qmax = uint32_t(ch.qmax[0]) | uint32_t(ch.qmax[1]) << 8 | uint32_t(ch.qmax[2]) << 16; + dut.raw = nd.raw; + dut.t_min = bits(nd.ray.tmin); + dut.t_max = bits(nd.ray.tmax); + dut.tag_in = sent; + dut.valid_in = 1; + inv_q.push_back(sent); + out_q.push_back(sent); + ++sent; + } else { + dut.valid_in = 0; + } + tick(); + if (dut.inv_valid) { + const uint32_t id = inv_q.front(); + inv_q.pop_front(); + const Item& it = items[id]; + bool ok = (dut.inv_tag == id), ftz_ok = ok; + for (int a = 0; a < 3; ++a) { + ok = ok && same_bits(dut.inv_d[a], it.ieee.inv[a]); + ftz_ok = ftz_ok && same_bits(dut.inv_d[a], it.ftz.inv[a]); + } + inv_ok[id] = ok ? 1 : (ftz_ok ? 2 : 0); + } + if (dut.valid_out) { + const uint32_t id = out_q.front(); + out_q.pop_front(); + const Item& it = items[id]; + const int cat = nodes[it.node].cat; + ++cat_cases[cat]; + bool ok = (dut.tag_out == id) && inv_ok[id] == 1 + && same_box(dut.hit, dut.t_near, it.ieee.box); + if (it.ieee.box.hit) ++hits; + if (!ok && dut.tag_out == id && inv_ok[id] != 0 + && same_box(dut.hit, dut.t_near, it.ftz.box) && !same_ref(it.ieee, it.ftz)) { + ok = true; + ++cat_ftz[cat]; + ++flushed; + node_ftz[it.node] = true; + } + if (!ok) { + ++errors; + if (inv_ok[id] == 0) ++inv_errors; + if (cat_err[cat]++ < 3) { + report_case(it, "MISMATCH"); + std::printf(" ref hit=%d t_near=%08x inv=(%08x %08x %08x) | rtl tag=%u hit=%d t_near=%08x inv_ok=%d\n", + it.ieee.box.hit, bits(it.ieee.box.t_near), bits(it.ieee.inv[0]), + bits(it.ieee.inv[1]), bits(it.ieee.inv[2]), dut.tag_out, dut.hit, + dut.t_near, inv_ok[id]); + } + } + rtl_hit[it.node][size_t(it.child)] = dut.hit; + rtl_t[it.node][size_t(it.child)] = dut.t_near; + ++checked; + } + } + + // child visit order per node + { + size_t base = 0; + for (uint32_t k = 0; k < nodes.size(); ++k) { + const int n = nodes[k].n; + if (n >= 2) { + std::vector ieee, ftz; + for (int c = 0; c < n; ++c) { + ieee.push_back(items[base + size_t(c)].ieee.box); + ftz.push_back(items[base + size_t(c)].ftz.box); + } + ++order_checked; + int acc = 0; + for (const BoxResult& b : ieee) acc += b.hit; + if (acc >= 2) ++order_multi; + const std::vector got = rtl_order(rtl_hit[k], rtl_t[k]); + if (got != simx_order(ieee)) { + if (node_ftz[k] && got == simx_order(ftz)) { + ++order_ftz; + } else { + ++order_errors; + } + } + } + base += size_t(n); + } + } + + std::printf("rtu_box_pe: %u boxes (%u directed), %u accepted, %u mismatches " + "(%u in inv_d), %u subnormal flushes\n", + checked, ND, hits, errors, inv_errors, flushed); + std::printf(" child order: %u nodes (%u with 2+ accepted), %u mismatches, %u subnormal flushes\n", + order_checked, order_multi, order_errors, order_ftz); + for (int k = 0; k < kNumCats; ++k) + std::printf(" %-20s %8u boxes %6u mismatches %6u subnormal flushes\n", + kCatNames[k], cat_cases[k], cat_err[k], cat_ftz[k]); + const bool fail = errors || order_errors; + std::printf(fail ? "FAILED!\n" : "PASSED!\n"); + return fail ? 1 : 0; +} diff --git a/hw/unittest/rtu_tri_pe/main.cpp b/hw/unittest/rtu_tri_pe/main.cpp index 81caf6b42c..b7a16a0195 100644 --- a/hw/unittest/rtu_tri_pe/main.cpp +++ b/hw/unittest/rtu_tri_pe/main.cpp @@ -1,11 +1,18 @@ // VX_rtu_tri_pe against SimX's rtu::ray_triangle: every verdict and, for a hit, -// t / u / v / back_facing must match bit for bit. +// t / u / v / back_facing must match bit for bit (any NaN matches any NaN). +// +// The PE's FP units flush subnormals (FTZ/DAZ) while SimX runs IEEE. A case +// whose RTL result differs from SimX but equals SimX evaluated under the host's +// FTZ/DAZ mode, and only there, is counted as a subnormal flush, not a mismatch. #include #include #include #include #include +#include +#include +#include #include "VVX_rtu_tri_pe.h" #include "verilated.h" #include "rtu_isect.h" @@ -14,17 +21,59 @@ namespace { struct Case { float o[3], d[3], v[3][3], tmin, tmax; + int cat = 0; // directed-case family (0: random) +}; + +const char* const kCatNames[] = { + "random", "t==tmin", "t==tmax", "t==tmin==tmax", "t in (t-,t+)", + "tmin=t+", "tmax=t-", "origin on triangle", "zero dir", + "special origin", "special dir", "special vertex", "special tmin", "special tmax", +}; +constexpr int kNumCats = int(sizeof(kCatNames) / sizeof(kCatNames[0])); + +struct Result { + bool hit = false; + float t = 0, u = 0, v = 0; + bool back = false; }; struct Expect { uint32_t id; - bool hit; - float t, u, v; - bool back; + int cat; + Result ieee, ftz; }; uint32_t bits(float f) { uint32_t b; std::memcpy(&b, &f, 4); return b; } +// bit-exact, except that any NaN matches any NaN (the payload is not specified) +bool same(uint32_t rtl, float ref) { + const bool rtl_nan = ((rtl >> 23) & 0xff) == 0xff && (rtl & 0x7fffff) != 0; + return std::isnan(ref) ? rtl_nan : rtl == bits(ref); +} + +Result reference(const Case& c, bool ftz) { + const unsigned csr = _mm_getcsr(); + if (ftz) _mm_setcsr(csr | 0x8040); // FTZ | DAZ + Result r; + r.hit = vortex::rtu::ray_triangle(c.o, c.d, c.v[0], c.v[1], c.v[2], + c.tmin, c.tmax, r.t, r.u, r.v, r.back); + _mm_setcsr(csr); + return r; +} + +template +bool matches(const Dut& dut, const Result& r) { + if (bool(dut.hit) != r.hit) return false; + return !r.hit || (same(dut.t, r.t) && same(dut.u, r.u) && same(dut.v, r.v) + && bool(dut.back_facing) == r.back); +} + +bool same_result(const Result& a, const Result& b) { + return a.hit == b.hit + && (!a.hit || (same(bits(a.t), b.t) && same(bits(a.u), b.u) + && same(bits(a.v), b.v) && a.back == b.back)); +} + std::mt19937 rng(7); float uni(float lo, float hi) { return std::uniform_real_distribution(lo, hi)(rng); } @@ -56,6 +105,67 @@ Case make_case(uint32_t i, const Case* twin_of) { return c; } +// Directed cases: t landing exactly on tmin / tmax, a zero t from an origin on +// the triangle, and NaN / inf / -0 / subnormal operands. +std::vector directed_cases() { + std::vector out; + const float inf = INFINITY, nan = NAN; + const float sub = 1e-40f; + for (uint32_t i = 0; i < 4000; ++i) { + Case c = make_case(i * 3 + 1, nullptr); // edge-biased family excluded + c.tmin = 0.f; c.tmax = 1e30f; + float t, u, v; bool bf; + if (!vortex::rtu::ray_triangle(c.o, c.d, c.v[0], c.v[1], c.v[2], + -inf, inf, t, u, v, bf)) + continue; + Case e = c; + e.cat = 1; e.tmin = t; e.tmax = inf; out.push_back(e); + e.cat = 2; e.tmin = -inf; e.tmax = t; out.push_back(e); + e.cat = 3; e.tmin = t; e.tmax = t; out.push_back(e); + e.cat = 4; e.tmin = std::nextafter(t, -inf); e.tmax = std::nextafter(t, inf); + out.push_back(e); + e.cat = 5; e.tmin = std::nextafter(t, inf); e.tmax = inf; out.push_back(e); + e.cat = 6; e.tmin = -inf; e.tmax = std::nextafter(t, -inf); out.push_back(e); + } + // origin on the triangle's plane: every rz is 0, so t is exactly +-0 + for (uint32_t i = 0; i < 2000; ++i) { + Case c; + c.cat = 7; + for (auto& vv : c.v) { vv[0] = uni(-5, 5); vv[1] = uni(-5, 5); vv[2] = 0.f; } + const float a = uni(0.f, 0.5f), b = uni(0.f, 0.5f); + for (int k = 0; k < 2; ++k) + c.o[k] = c.v[0][k] + a * (c.v[1][k] - c.v[0][k]) + b * (c.v[2][k] - c.v[0][k]); + c.o[2] = (i & 1) ? -0.f : 0.f; + c.d[0] = uni(-1, 1); c.d[1] = uni(-1, 1); c.d[2] = (i & 2) ? uni(0.5f, 2) : -uni(0.5f, 2); + const float tmins[] = { 0.f, -0.f, -1.f, sub, -sub }; + c.tmin = tmins[i % 5]; + c.tmax = (i % 7 == 0) ? 0.f : 1e30f; + out.push_back(c); + } + // special operands in every input slot + const float specials[] = { nan, -nan, inf, -inf, 0.f, -0.f, sub, -sub, 3.4e38f }; + for (uint32_t i = 0; i < 400; ++i) { + const Case base = make_case(i * 5 + 2, nullptr); + for (float sp : specials) { + for (int slot = 0; slot < 17; ++slot) { + Case c = base; + c.cat = (slot < 3) ? 9 : (slot < 6) ? 10 : (slot < 15) ? 11 : (slot == 15) ? 12 : 13; + if (slot < 3) c.o[slot] = sp; + else if (slot < 6) c.d[slot - 3] = sp; + else if (slot < 15) c.v[(slot - 6) / 3][(slot - 6) % 3] = sp; + else if (slot == 15) c.tmin = sp; + else c.tmax = sp; + out.push_back(c); + } + } + Case z = base; // zero direction + z.cat = 8; + z.d[0] = z.d[1] = z.d[2] = (i & 1) ? -0.f : 0.f; + out.push_back(z); + } + return out; +} + } // namespace int main(int argc, char** argv) { @@ -69,12 +179,24 @@ int main(int argc, char** argv) { std::deque exp; uint32_t sent = 0, checked = 0, errors = 0, hits = 0, twins = 0; + uint32_t cat_cases[kNumCats] = {}, cat_errors[kNumCats] = {}, cat_ftz[kNumCats] = {}; + uint32_t flushed = 0; + std::deque sent_cases; Case prev{}; - while (checked < N) { - if (sent < N) { - const bool twin = (sent > 0) && (sent % 4 == 1); - Case c = make_case(sent, twin ? &prev : nullptr); - if (!twin) prev = c; else ++twins; + const std::vector directed = directed_cases(); + const uint32_t ND = uint32_t(directed.size()); + const uint32_t total = N + ND; + while (checked < total) { + if (sent < total) { + Case c; + if (sent < ND) { + c = directed[sent]; + } else { + const uint32_t r = sent - ND; + const bool twin = (r > 0) && (r % 4 == 1); + c = make_case(r, twin ? &prev : nullptr); + if (!twin) prev = c; else ++twins; + } for (int k = 0; k < 3; ++k) { dut.origin[k] = bits(c.o[k]); dut.dir[k] = bits(c.d[k]); @@ -86,10 +208,9 @@ int main(int argc, char** argv) { dut.t_max = bits(c.tmax); dut.tag_in = sent; dut.valid_in = 1; - Expect e{sent, false, 0, 0, 0, false}; - e.hit = vortex::rtu::ray_triangle(c.o, c.d, c.v[0], c.v[1], c.v[2], - c.tmin, c.tmax, e.t, e.u, e.v, e.back); + Expect e{sent, c.cat, reference(c, false), reference(c, true)}; exp.push_back(e); + sent_cases.push_back(c); ++sent; } else { dut.valid_in = 0; @@ -98,21 +219,34 @@ int main(int argc, char** argv) { if (dut.valid_out) { const Expect e = exp.front(); exp.pop_front(); - bool ok = (dut.tag_out == e.id) && (bool(dut.hit) == e.hit); - if (ok && e.hit) { - ok = dut.t == bits(e.t) && dut.u == bits(e.u) && dut.v == bits(e.v) - && bool(dut.back_facing) == e.back; - ++hits; + const Case c = sent_cases.front(); + sent_cases.pop_front(); + ++cat_cases[e.cat]; + bool ok = (dut.tag_out == e.id) && matches(dut, e.ieee); + if (e.ieee.hit) ++hits; + if (!ok && dut.tag_out == e.id && matches(dut, e.ftz) && !same_result(e.ieee, e.ftz)) { + ++cat_ftz[e.cat]; + ++flushed; + ok = true; } - if (!ok && errors++ < 10) { + if (!ok) { ++cat_errors[e.cat]; ++errors; } + if (!ok && cat_errors[e.cat] <= 3) { + std::printf(" [%s] o=(%a %a %a) d=(%a %a %a) v0=(%a %a %a) v1=(%a %a %a) v2=(%a %a %a) tmin=%a tmax=%a\n", + kCatNames[e.cat], c.o[0], c.o[1], c.o[2], c.d[0], c.d[1], c.d[2], + c.v[0][0], c.v[0][1], c.v[0][2], c.v[1][0], c.v[1][1], c.v[1][2], + c.v[2][0], c.v[2][1], c.v[2][2], c.tmin, c.tmax); std::printf("MISMATCH #%u: ref hit=%d t=%08x u=%08x v=%08x bf=%d | rtl tag=%u hit=%d t=%08x u=%08x v=%08x bf=%d\n", - e.id, e.hit, bits(e.t), bits(e.u), bits(e.v), e.back, + e.id, e.ieee.hit, bits(e.ieee.t), bits(e.ieee.u), bits(e.ieee.v), e.ieee.back, dut.tag_out, dut.hit, dut.t, dut.u, dut.v, dut.back_facing); } ++checked; } } - std::printf("rtu_tri_pe: %u cases (%u hits, %u twins), %u mismatches\n", checked, hits, twins, errors); + std::printf("rtu_tri_pe: %u cases (%u directed, %u hits, %u twins), %u mismatches, %u subnormal flushes\n", + checked, ND, hits, twins, errors, flushed); + for (int k = 0; k < kNumCats; ++k) + std::printf(" %-20s %8u cases %6u mismatches %6u subnormal flushes\n", + kCatNames[k], cat_cases[k], cat_errors[k], cat_ftz[k]); std::printf(errors ? "FAILED!\n" : "PASSED!\n"); return errors ? 1 : 0; } From ce1472962debcbfe3aab810d9ee0f4234ddb217a Mon Sep 17 00:00:00 2001 From: Blaise Tine Date: Sat, 3 Oct 2026 04:13:12 -0700 Subject: [PATCH 13/31] rt_replay: flush mismatches per batch; document the rtlsim stall-timeout define A model that hangs or aborts mid-run now keeps the mismatches it already reported. On rtlsim, one hard ray can hold every warp in vx_rt_wait past the scheduler's all-warps-stalled watchdog, so the replay must be built with -DVX_DBG_STALL_TIMEOUT=2000000000. Co-Authored-By: Claude Opus 5.5 --- tests/raytracing/rt_replay/Makefile | 6 ++++++ tests/raytracing/rt_replay/main.cpp | 10 +++++++++- 2 files changed, 15 insertions(+), 1 deletion(-) diff --git a/tests/raytracing/rt_replay/Makefile b/tests/raytracing/rt_replay/Makefile index 0793d00c5a..0250d63b17 100644 --- a/tests/raytracing/rt_replay/Makefile +++ b/tests/raytracing/rt_replay/Makefile @@ -16,6 +16,12 @@ HDRS := $(SRC_DIR)/common.h VX_SRCS := $(SRC_DIR)/kernel.cpp VX_HDRS := $(SRC_DIR)/common.h +# Usage (from the build dir), e.g. for the LumiBench RTU config on rtlsim: +# CONFIGS="$(cat /CONFIGS.txt) -DVX_DBG_STALL_TIMEOUT=2000000000" \ +# make -C tests/raytracing/rt_replay run-rtlsim OPTS="-f -r 0:1024" +# The stall-timeout define is required on rtlsim: one hard ray can keep every +# warp waiting in vx_rt_wait longer than the scheduler watchdog's default. +# Logs come from SimX with VX_RTU_RAYLOG=; ./rt_replay -h lists options. OPTS ?= KERNEL_LIB := vortex2 diff --git a/tests/raytracing/rt_replay/main.cpp b/tests/raytracing/rt_replay/main.cpp index d7262fbad5..e326bcecfe 100644 --- a/tests/raytracing/rt_replay/main.cpp +++ b/tests/raytracing/rt_replay/main.cpp @@ -98,7 +98,10 @@ void show_usage() { " -l file replay only the ray indices listed (one per line, '#' comments;\n" " a mismatch file from -o is accepted as is)\n" " -s n replay every n-th selected ray\n" - " -a also replay rays a shader decided (any-hit / intersection)\n"); + " -a also replay rays a shader decided (any-hit / intersection)\n" + "Build with the CONFIGS that recorded the log. For rtlsim also pass\n" + "-DVX_DBG_STALL_TIMEOUT=2000000000: one hard ray can hold every warp in\n" + "vx_rt_wait past the scheduler's default all-warps-stalled watchdog.\n"); } void parse_args(int argc, char** argv) { @@ -374,6 +377,10 @@ int main(int argc, char** argv) { RT_CHECK(vx_enqueue_write(queue, rays_buf, 0, dev_rays.data(), count * sizeof(replay_ray_t), 0, nullptr, nullptr)); RT_CHECK(vx_queue_finish(queue, VX_TIMEOUT_INFINITE)); + if (verbose) { + printf("launch [%zu, %zu): %u threads\n", pos, end, count); + fflush(stdout); + } auto t0 = std::chrono::steady_clock::now(); vx_event_h lev = nullptr, rev = nullptr; vx_launch_info_t li = {}; @@ -406,6 +413,7 @@ int main(int argc, char** argv) { if (printed < max_print) { print_ray(stdout, owner[k], r, res[k], cat); ++printed; } } mismatched += bad; + if (of) fflush(of); // a model that hangs or aborts later keeps what it reported if (verbose || sel.size() > batch) { printf("batch [%zu, %zu) epoch %u: %llu rays, %llu mismatches, %.1fs\n", pos, end, epoch, (unsigned long long)(end - pos), (unsigned long long)bad, dt); From a8861312d17b56626c4c0bd21979afc6ba0fa567 Mon Sep 17 00:00:00 2001 From: Blaise Tine Date: Sat, 3 Oct 2026 04:36:24 -0700 Subject: [PATCH 14/31] rtu: RTL near-tie oracle picks the hit lavapipe keeps, as SimX does Ports the SimX visit-order oracle (6804d1b0c) to the RTL RTU, so that among equal / near-equal-t opaque hits the RTL commits exactly the hit SimX (and lavapipe) commits. - Scheduler: an opaque hit within t*2^-19 of the hit committed in this walk, with visit-order tables for both and a different triangle, hands both hits to the oracle and parks the context (CS_ORC_REQ / CS_ORC_WAIT); otherwise the static (instance, geometry, prim) key still settles exact ties, so scenes without tables are unchanged. The context word carries the committed hit's near bound, TLAS rank, parent/side, BLAS table and vertices, and the walk's TLAS table / BLAS table / rank / parent/side from the leaf headers. near_t(t) is computed off the tri PE result into its result RAM and the window compares at ALIGN, so EXEC gains no logic depth and the common path no cycles. - Culling as SimX: node children are culled against near_t(best_t) once an opaque hit is committed (procedural AABBs still against best_t), and the tri PE tests the ray's own [tmin, tmax], so a hit at or past the committed t reaches the tie-break. - VX_rtu_oracle: climbs both leaves to their lowest common ancestor in the BLAS table (same instance) or the TLAS table with the world ray (across instances), testing every box on the way in lavapipe's F32 slab test against the other hit's t, orders the two LCA children by entry distance (child 0 on ties), and on a lost-by-cull verdict climbs the nearer hit's BLAS path with lavapipe's object ray rebuilt from the table's world->object matrix in its op order. One VX_fma_unit (IEEE, subnormals), three serial correctly-rounded reciprocals (1/0 -> FLT_MAX), a 64-word LUTRAM register file, and a fetch unit that overlaps a climb step's parent fetch with its box test; table lines come over the scheduler's memory port under the context's tag (oracle first; a context's fetch retries). A climb cap keeps a malformed (cyclic) table from hanging a context. - VX_rtu_near_t: t + |t|*2^-19 with both F32 roundings, exact for all 2^32 inputs; VX_rtu_f32_round: the shared integer-to-F32 rounder. - rt_smoke_tie: coincident and near-coincident triangles within a BLAS and across instances, with lavapipe-style visit-order tables (TLAS in the scene, BLAS tables in a separate buffer), traced with and without the tables; every hit field is compared bit-exactly against SimX's results (golden.h, regenerated with -g), and the tables must change some verdicts. Co-Authored-By: Claude Opus 5.5 --- hw/rtl/rtu/VX_rtu_f32_round.sv | 66 ++ hw/rtl/rtu/VX_rtu_near_t.sv | 71 ++ hw/rtl/rtu/VX_rtu_oracle.sv | 1098 ++++++++++++++++++++++ hw/rtl/rtu/VX_rtu_scheduler.sv | 218 ++++- tests/raytracing/Makefile | 2 +- tests/raytracing/rt_smoke_tie/Makefile | 22 + tests/raytracing/rt_smoke_tie/common.h | 44 + tests/raytracing/rt_smoke_tie/golden.h | 522 ++++++++++ tests/raytracing/rt_smoke_tie/kernel.cpp | 41 + tests/raytracing/rt_smoke_tie/main.cpp | 672 +++++++++++++ 10 files changed, 2737 insertions(+), 19 deletions(-) create mode 100644 hw/rtl/rtu/VX_rtu_f32_round.sv create mode 100644 hw/rtl/rtu/VX_rtu_near_t.sv create mode 100644 hw/rtl/rtu/VX_rtu_oracle.sv create mode 100644 tests/raytracing/rt_smoke_tie/Makefile create mode 100644 tests/raytracing/rt_smoke_tie/common.h create mode 100644 tests/raytracing/rt_smoke_tie/golden.h create mode 100644 tests/raytracing/rt_smoke_tie/kernel.cpp create mode 100644 tests/raytracing/rt_smoke_tie/main.cpp diff --git a/hw/rtl/rtu/VX_rtu_f32_round.sv b/hw/rtl/rtu/VX_rtu_f32_round.sv new file mode 100644 index 0000000000..d5004da891 --- /dev/null +++ b/hw/rtl/rtu/VX_rtu_f32_round.sv @@ -0,0 +1,66 @@ +// Copyright © 2019-2023 +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +// VX_rtu_f32_round — rounds the positive value (mag + sticky) * 2^exp to F32, +// nearest even, subnormals and overflow to infinity included (combinational). +// `sticky` stands for a nonzero remainder below mag's LSB; callers keep at +// least two bits of mag below the result's LSB whenever it is set. Returns the +// magnitude bits {exponent, fraction}; mag == 0 gives +0. + +`include "VX_define.vh" + +module VX_rtu_f32_round #( + parameter WB = 44, // mag width + parameter EW = 11 // signed exponent width +) ( + input wire [WB-1:0] mag, + input wire signed [EW-1:0] exp, + input wire sticky, + output wire [30:0] result +); + localparam IW = `CLOG2(WB); + localparam BW = WB + 12; + + // index of mag's leading one + reg [IW-1:0] msb; + always @(*) begin + msb = '0; + for (integer i = 0; i < WB; ++i) begin + if (mag[i]) msb = IW'(i); + end + end + + // the result LSB's exponent, clamped at the subnormal LSB 2^-149 + wire signed [EW+1:0] lsb_raw = (EW+2)'(exp) + (EW+2)'($signed({1'b0, msb})) - (EW+2)'(23); + wire signed [EW+1:0] lsb = (lsb_raw < -(EW+2)'(149)) ? -(EW+2)'(149) : lsb_raw; + wire signed [EW+1:0] sh = lsb - (EW+2)'(exp); + + // sh <= 0: exact, shifted up; sh > 0: drop sh bits, guard + sticky + wire exact = (sh <= 0); + wire [IW-1:0] ush = exact ? IW'(-sh) : IW'(sh); + wire [IW-1:0] gi = exact ? '0 : IW'(sh - (EW+2)'(1)); + wire [WB-1:0] q = mag >> ush; + wire [WB-1:0] low = mag & ((WB'(1) << gi) - WB'(1)); + wire g = mag[gi]; + wire st = (low != '0) || sticky; + wire [BW-1:0] sig = exact ? (BW'(mag) << ush) + : (BW'(q) + BW'(g && (st || q[0]))); + + // {biased exponent of the LSB position, significand}: a carry out of the + // significand bumps the exponent, a subnormal reaching 2^23 becomes normal + wire [BW-1:0] bits = (BW'(lsb + (EW+2)'(149)) << 23) + sig; + assign result = (mag == '0) ? 31'd0 + : (bits >= BW'(32'h7f800000)) ? 31'h7f800000 + : bits[30:0]; + +endmodule diff --git a/hw/rtl/rtu/VX_rtu_near_t.sv b/hw/rtl/rtu/VX_rtu_near_t.sv new file mode 100644 index 0000000000..b46a3efdb4 --- /dev/null +++ b/hw/rtl/rtu/VX_rtu_near_t.sv @@ -0,0 +1,71 @@ +// Copyright © 2019-2023 +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +// VX_rtu_near_t — the near-hit window bound t + |t| * 2^-19 in F32, both +// operations rounded to nearest even, subnormals included (combinational). +// Two opaque hits within it of each other may be ordered either way by the +// source BVH's F32 box cull, so the walker settles them by its visit order. +// +// |t| * 2^-19 is exact unless it lands in the subnormal range, so the sum is +// formed as one integer W * 2^p (the scaled term pre-rounded where it is not +// exact) and rounded once. + +`include "VX_define.vh" + +module VX_rtu_near_t ( + input wire [31:0] t, + output wire [31:0] result +); + localparam WB = 44; + + wire s = t[31]; + wire [7:0] e = t[30:23]; + wire [22:0] f = t[22:0]; + + wire is_nan = (e == 8'hff) && (f != 23'd0); + wire is_inf = (e == 8'hff) && (f == 23'd0); + wire is_zero = (e == 8'h00) && (f == 23'd0); + + // |t| = m * 2^q, q = e - 150 (normal) or -149 (subnormal) + wire [23:0] m = {(e != 8'h00), f}; + // |t| * 2^-19 = (m / 2^d) * 2^p: d > 0 only when it is subnormal, where + // its integer m / 2^d is rounded as the F32 multiply rounds it + wire [4:0] d = (e >= 8'd20) ? 5'd0 : ((e == 8'h00) ? 5'd19 : 5'(8'd20 - e)); + wire signed [10:0] p = (d == 5'd0) ? (11'($signed({1'b0, e})) - 11'sd169) : -11'sd149; + + wire [23:0] q0 = m >> d; + wire [4:0] gi = (d == 5'd0) ? 5'd0 : (d - 5'd1); + wire g = (d != 5'd0) && m[gi]; + wire st = (m & ((24'd1 << gi) - 24'd1)) != 24'd0; + wire [23:0] y_int = q0 + 24'(g && (st || q0[0])); + + wire [WB-1:0] m_al = WB'(m) << (5'd19 - d); + wire [WB-1:0] w = s ? (m_al - WB'(y_int)) : (m_al + WB'(y_int)); + + wire [30:0] mag; + VX_rtu_f32_round #( + .WB (WB), + .EW (11) + ) round ( + .mag (w), + .exp (p), + .sticky (1'b0), + .result (mag) + ); + + assign result = is_nan ? (t | 32'h00400000) + : is_inf ? (s ? 32'hffc00000 : 32'h7f800000) + : is_zero ? 32'h00000000 + : {s, mag}; + +endmodule diff --git a/hw/rtl/rtu/VX_rtu_oracle.sv b/hw/rtl/rtu/VX_rtu_oracle.sv new file mode 100644 index 0000000000..552310f06f --- /dev/null +++ b/hw/rtl/rtu/VX_rtu_oracle.sv @@ -0,0 +1,1098 @@ +// Copyright © 2019-2023 +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +// VX_rtu_oracle — the visit-order oracle: of two opaque triangle hits within +// the near window of each other (VX_rtu_near_t), which one the source BVH's +// traversal keeps. The source traversal (the Vulkan reference) walks its +// binary tree depth-first, nearer child box first, child 0 on equal entry +// distances, tests each box in F32 against the hit committed when the box's +// parent is visited, and commits only a strictly nearer hit. So of two such +// hits the first-reached wins at equal t, and a nearer hit reached second is +// lost when a box on its own path, below where the two paths part, enters at +// or past the first one's t. The driver lays the source trees out as +// visit-order tables (TLAS in the scene, BLAS tables reached by scene offset): +// +// TLAS { n_leaves, n_nodes, nodes_off, leaf_stride } +// leaf[i] @ +16 + i*stride : { parent << 1 | side, _, _, _, world->object 3x4 } +// node[j] @ +nodes_off + j*64 : { child box 0, child box 1, parent << 1 | side, depth } +// BLAS { n_nodes, _, nodes_off, 32 } +// node[j] @ +nodes_off + j*32 : { own box, parent << 1 | side, depth } +// +// a triangle's box being its vertices' F32 min/max. Both leaves climb to +// their lowest common ancestor, every box on the way tested against the other +// hit's t; the two child boxes under it are ordered by their entry distance. +// Across instances the TLAS is climbed with the world ray, and a lost-by-cull +// verdict also climbs the nearer hit's own BLAS path with the reference's +// object-space ray, rebuilt from the table's matrix in its own op order. +// +// One request at a time; the requesting context parks until `done`. The +// engine is a small sequencer around one F32 add/mul unit (IEEE, subnormals), +// three serial correctly-rounded reciprocals and a 64-word LUTRAM register +// file. Table words are fetched a line at a time on the scheduler's memory +// port, under the requesting context's tag, by a fetch unit that runs beside +// the sequencer, so a climb step's parent fetch overlaps its box test. Cost is +// O(tree depth) table reads and box tests, paid only on near ties. + +`include "VX_define.vh" + +module VX_rtu_oracle import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( + parameter CTX_TAG_W = 1, + parameter ADDRW = `VX_CFG_MEM_ADDR_WIDTH, + parameter LINE_BITS = `VX_CFG_MEM_BLOCK_SIZE * 8, + parameter FMA_LAT = RTU_LATENCY_FMA +) ( + input wire clk, + input wire reset, + + // request: hit a (the new one) against hit b (the committed one) + input wire req_valid, + output wire req_ready, + input wire [CTX_TAG_W-1:0] req_ctx, + input wire [ADDRW-1:0] req_scene, + input wire [2:0][31:0] req_wo, // world ray + input wire [2:0][31:0] req_wd, + input wire [31:0] req_tlas, // TLAS table (scene offset) + input wire [31:0] req_a_t, + input wire [31:0] req_a_inst, // TLAS leaf rank + input wire [31:0] req_a_ps, // parent << 1 | side in its BLAS table + input wire [31:0] req_a_tab, // its BLAS table (scene offset) + input wire [8:0][31:0] req_a_v, // v0.xyz, v1.xyz, v2.xyz + input wire [31:0] req_b_t, + input wire [31:0] req_b_inst, + input wire [31:0] req_b_ps, + input wire [31:0] req_b_tab, + input wire [8:0][31:0] req_b_v, + + // verdict: true when the source traversal keeps hit a + output wire done_valid, + output wire [CTX_TAG_W-1:0] done_ctx, + output wire done_keep, + + // table fetch, one line in flight + output wire mem_req_valid, + output wire [ADDRW-1:0] mem_req_addr, + input wire mem_req_ready, + input wire mem_rsp_valid, + input wire [LINE_BITS-1:0] mem_rsp_data +); + localparam LINE_BYTES = LINE_BITS / 8; + localparam WPL = LINE_BYTES / 4; // words per line + localparam WIDXW = `CLOG2(2 * WPL); + localparam LSELW = `CLOG2(LINE_BYTES); + + `STATIC_ASSERT((LINE_BYTES >= 64), ("table fetches assume >= 64-byte lines")) + + localparam [31:0] ROOT = 32'hffffffff; + localparam [31:0] F_INF = 32'h7f800000; + localparam [31:0] F_MAX = 32'h7f7fffff; + + // Climb steps one verdict may take. Valid tables reach the root in tree + // depth steps; a malformed table (a parent cycle) must still not hang the + // context, so past the cap the verdict falls to the nearer hit. + localparam MAX_CLIMBS = 4096; + localparam CLIMBW = `CLOG2(MAX_CLIMBS + 1); + + // ── register file map ───────────────────────────────────────────── + localparam [5:0] RA_WO = 6'd0, // world ray o, d + RA_WD = 6'd3, + RA_RO = 6'd6, // current ray o, d, 1/d + RA_RD = 6'd9, + RA_RI = 6'd12, + RA_BA = 6'd15, // hit a / b vertex boxes (min.xyz, max.xyz) + RA_BB = 6'd21, + RA_CB0 = 6'd27, // climb boxes + RA_CB1 = 6'd33, + RA_M = 6'd27, // object-ray matrix (aliases the climb boxes) + RA_S = 6'd39, // slab distances + RA_X = 6'd45; // the object ray's products + + // ── sequencer states ────────────────────────────────────────────── + localparam [6:0] + S_IDLE = 7'd0, S_CAP = 7'd1, S_BOX1 = 7'd2, S_BOX2 = 7'd3, + S_DISP = 7'd5, S_DONE = 7'd6, + // same instance + T_S1 = 7'd8, T_S2 = 7'd9, T_S3 = 7'd10, T_S4 = 7'd11, + T_S5 = 7'd12, T_S6 = 7'd13, T_S7 = 7'd14, + // different instances + T_D0 = 7'd16, T_D1 = 7'd17, T_D2 = 7'd18, T_D3 = 7'd19, + T_D3B = 7'd20, T_D4 = 7'd21, T_D5 = 7'd22, T_D6 = 7'd23, + T_D7 = 7'd24, T_D8 = 7'd25, T_D9 = 7'd26, T_D10 = 7'd27, + T_D11 = 7'd28, T_D12 = 7'd29, + // a hit's BLAS path cull + P_0 = 7'd32, P_1 = 7'd33, P_2 = 7'd34, P_3 = 7'd35, + // lowest common ancestor + L_0 = 7'd40, L_1 = 7'd41, L_2 = 7'd42, + // one climb step + U_0 = 7'd44, U_1 = 7'd45, U_2 = 7'd46, U_3 = 7'd47, + // box test + B_SUB = 7'd48, B_MUL = 7'd49, B_RED = 7'd50, B_FIN = 7'd51, + // object ray + O_0 = 7'd56, O_1 = 7'd57, O_2 = 7'd58, O_3 = 7'd59, + O_4 = 7'd60, O_5 = 7'd61, O_6 = 7'd62, O_7 = 7'd63, + O_8 = 7'd64, + // reciprocals + R_LD0 = 7'd68, R_LD1 = 7'd69, R_IT = 7'd70, R_WR = 7'd71, + // fetch wait / move / copy / multiply / drain + F_WAIT = 7'd72, M_MV = 7'd76, C_CP = 7'd77, X_MUL = 7'd78, + W_DRAIN = 7'd79; + + // ── F32 helpers (C fmin/fmax, IEEE ordered compares) ────────────── + function automatic logic f_nan(input logic [30:0] a); + f_nan = (a[30:23] == 8'hff) && (a[22:0] != 23'd0); + endfunction + function automatic logic f_eq(input logic [31:0] a, input logic [31:0] b); + f_eq = !f_nan(a[30:0]) && !f_nan(b[30:0]) + && ((a == b) || ((a[30:0] == 31'd0) && (b[30:0] == 31'd0))); + endfunction + function automatic logic f_lt(input logic [31:0] a, input logic [31:0] b); + if (f_nan(a[30:0]) || f_nan(b[30:0]) || ((a[30:0] == 31'd0) && (b[30:0] == 31'd0))) begin + f_lt = 1'b0; + end else if (a[31] != b[31]) begin + f_lt = a[31]; + end else if (!a[31]) begin + f_lt = (a[30:0] < b[30:0]); + end else begin + f_lt = (a[30:0] > b[30:0]); + end + endfunction + function automatic logic [31:0] f_min(input logic [31:0] a, input logic [31:0] b); + f_min = f_nan(a[30:0]) ? b : (f_nan(b[30:0]) ? a : (f_lt(b, a) ? b : a)); + endfunction + function automatic logic [31:0] f_max(input logic [31:0] a, input logic [31:0] b); + f_max = f_nan(a[30:0]) ? b : (f_nan(b[30:0]) ? a : (f_lt(a, b) ? b : a)); + endfunction + function automatic logic [5:0] mod3(input logic [4:0] k); + mod3 = (k >= 5'd3) ? 6'(k - 5'd3) : 6'(k); + endfunction + + // the scene offset of a table node + function automatic logic [31:0] node_off(input logic [31:0] tab, input logic [31:0] noff, + input logic tl, input logic [31:0] ps); + node_off = tab + noff + (tl ? ((ps >> 1) << 6) : ((ps >> 1) << 5)); + endfunction + + // ── state ───────────────────────────────────────────────────────── + reg [6:0] state; + reg [6:0] f_ret, b_ret, u_ret, l_ret, o_ret, r_ret, w_ret, m_ret, c_ret, x_ret; + + reg [CTX_TAG_W-1:0] ctx_r; + reg [ADDRW-1:0] scene_r; + reg [23:0][31:0] cap; + reg [31:0] tlas_r, ta, tb, ia, ib, psa, psb, taba, tabb; + reg [31:0] tab_r, noff_r, stride_r; + reg is_tlas; + reg [31:0] ps0, ps1, dep0, dep1; + reg cul0, cul1; + reg cur; // the climb UP moves + reg [31:0] upt; // ... and the t its boxes are culled against + reg a_first, keep, ph; + reg [31:0] key0; + reg [CLIMBW-1:0] climbs; + + // box test + reg [5:0] bt_base; + reg [31:0] bt_tmax; + reg bt_nan; + reg [31:0] bt_lo, bt_hi, bt_key; + reg bt_pass; + reg [2:0] bt_wbc; // slab differences written back so far + reg [4:0] k; // the running routine counter + + // object ray / multiply + reg [31:0] o_inst; + reg [31:0] mul_a, mul_b, mul_p; + + // move / copy + reg [5:0] mv_dst, cp_src; + reg [4:0] mv_k0, mv_n; + + // ── register file: two LUTRAM copies for two read ports ────────── + reg [5:0] ra, rb; + wire [31:0] rda, rdb; + reg fsm_we; + reg [5:0] fsm_wa; + reg [31:0] fsm_wd; + wire wb_v; + wire [5:0] wb_dst; + wire [31:0] fma_res; + wire rf_we = wb_v || fsm_we; + wire [5:0] rf_wa = wb_v ? wb_dst : fsm_wa; + wire [31:0] rf_wd = wb_v ? fma_res : fsm_wd; + + VX_dp_ram #( + .DATAW (32), + .SIZE (64), + .LUTRAM (1), + .OUT_REG (0) + ) rf_a ( + .clk (clk), + .reset (reset), + .read (1'b1), + .write (rf_we), + .wren (1'b1), + .waddr (rf_wa), + .wdata (rf_wd), + .raddr (ra), + .rdata (rda) + ); + VX_dp_ram #( + .DATAW (32), + .SIZE (64), + .LUTRAM (1), + .OUT_REG (0) + ) rf_b ( + .clk (clk), + .reset (reset), + .read (1'b1), + .write (rf_we), + .wren (1'b1), + .waddr (rf_wa), + .wdata (rf_wd), + .raddr (rb), + .rdata (rdb) + ); + + // ── F32 add / sub / mul (IEEE, subnormals) ─────────────────────── + reg fma_issue; + reg [1:0] fma_kind; // 0: add, 1: sub, 2: mul + reg [5:0] fma_dst; + reg [5:0] fma_pend; + + VX_fma_unit #( + .LATENCY (FMA_LAT), + .USE_DSP (`VX_CFG_RTU_USE_DSP), + .SUBNORM_ENABLE (1), + .EXCEPT_ENABLE (1) + ) fma ( + .clk (clk), + .reset (reset), + .enable (1'b1), + .mask (fma_issue), + .op_type ((fma_kind == 2'd2) ? INST_FPU_MUL : INST_FPU_ADD), + .fmt ((fma_kind == 2'd1) ? INST_FMT_BITS'(2'b10) : INST_FMT_BITS'(2'b00)), + .frm (INST_FRM_RNE), + .dataa (rda), + .datab (rdb), + .datac (32'd0), + .result (fma_res), + `UNUSED_PIN (fflags) + ); + + VX_shift_register #( + .DATAW (1 + 6), + .RESETW (1), + .DEPTH (FMA_LAT) + ) fma_tags ( + .clk (clk), + .reset (reset), + .enable (1'b1), + .data_in ({fma_issue, fma_dst}), + .data_out ({wb_v, wb_dst}) + ); + + // ── fetch unit: f_n words at scene offset f_off, beside the sequencer ── + localparam [2:0] FS_IDLE = 3'd0, FS_REQ0 = 3'd1, FS_RSP0 = 3'd2, + FS_REQ1 = 3'd3, FS_RSP1 = 3'd4; + reg [2:0] fs_state; + reg fs_start; + reg [31:0] f_off; + reg [4:0] f_n; + reg [ADDRW-1:0] f_line; + reg [LSELW-3:0] f_w0; + reg f_two; + reg [LINE_BITS-1:0] lb0, lb1; + + // tables are word aligned: the low address bits are always zero + wire [ADDRW-1:0] f_addr = scene_r + ADDRW'(f_off); + `UNUSED_VAR (f_addr[1:0]) + + always_ff @(posedge clk) begin + if (reset) begin + fs_state <= FS_IDLE; + end else begin + case (fs_state) + FS_IDLE: begin + if (fs_start) begin + f_line <= {f_addr[ADDRW-1:LSELW], LSELW'(0)}; + f_w0 <= f_addr[LSELW-1:2]; + f_two <= (32'(f_addr[LSELW-1:2]) + 32'(f_n)) > WPL; + fs_state <= FS_REQ0; + end + end + FS_REQ0: if (mem_req_ready) fs_state <= FS_RSP0; + FS_RSP0: begin + if (mem_rsp_valid) begin + lb0 <= mem_rsp_data; + fs_state <= f_two ? FS_REQ1 : FS_IDLE; + end + end + FS_REQ1: if (mem_req_ready) fs_state <= FS_RSP1; + FS_RSP1: begin + if (mem_rsp_valid) begin + lb1 <= mem_rsp_data; + fs_state <= FS_IDLE; + end + end + default: fs_state <= FS_IDLE; + endcase + end + end + wire fs_done = (fs_state == FS_IDLE) && !fs_start; + + assign mem_req_valid = (fs_state == FS_REQ0) || (fs_state == FS_REQ1); + assign mem_req_addr = (fs_state == FS_REQ0) ? f_line : (f_line + ADDRW'(LINE_BYTES)); + + // fetched words: one read port over the two lines + reg [4:0] fw_k; + wire [2*LINE_BITS-1:0] lbs = {lb1, lb0}; + wire [WIDXW-1:0] fw_idx = WIDXW'(f_w0) + WIDXW'(fw_k); + wire [31:0] fw = lbs[32'(fw_idx) * 32 +: 32]; + + // ── reciprocals: 2^50 / significand, 28 quotient bits, three lanes ─ + reg [2:0] rc_spec; // lane result is special (rc_sval) + reg [2:0][31:0] rc_sval; + reg [2:0][23:0] dv_m; + reg [2:0][10:0] dv_e; // signed exponent of the operand's significand LSB + reg [2:0] dv_s; + reg [2:0][24:0] dv_rem; + reg [2:0][27:0] dv_q; + reg [4:0] dv_it; + + // operand decode: 1/0 -> FLT_MAX (the reference's zero-direction + // reciprocal), 1/inf -> 0, 1/NaN -> NaN + function automatic logic [68:0] rc_decode(input logic [31:0] x); + logic [7:0] e; + logic [22:0] f; + logic [4:0] msb; + logic spec; + logic [31:0] sval; + logic [23:0] m; + logic [10:0] ex; + e = x[30:23]; + f = x[22:0]; + msb = '0; + for (integer i = 0; i < 23; ++i) begin + if (f[i]) msb = 5'(i); + end + spec = (x[30:0] == 31'd0) || (e == 8'hff); + sval = (x[30:0] == 31'd0) ? F_MAX + : ((f != 23'd0) ? (x | 32'h00400000) : {x[31], 31'd0}); + if (e == 8'h00) begin + m = 24'(f) << (5'd23 - msb); + ex = 11'({6'd0, msb}) - 11'd172; + end else begin + m = {1'b1, f}; + ex = 11'({3'd0, e}) - 11'd150; + end + rc_decode = {spec, sval, x[31], m, ex}; + endfunction + wire [68:0] rc_dec_a = rc_decode(rda); + wire [68:0] rc_dec_b = rc_decode(rdb); + + wire [1:0] rc_w = 2'(k); + wire [30:0] rc_mag; + VX_rtu_f32_round #( + .WB (28), + .EW (11) + ) rc_round ( + .mag (dv_q[rc_w]), + .exp (-11'sd50 - $signed(dv_e[rc_w])), + .sticky (dv_rem[rc_w] != 25'd0), + .result (rc_mag) + ); + + // ── sequencer ───────────────────────────────────────────────────── + // O_4's product index k = (d ? 9 : 0) + i*3 + j + wire [4:0] o4_kk = (k >= 5'd9) ? (k - 5'd9) : k; + wire [5:0] o4_i = (o4_kk >= 5'd6) ? 6'd2 : ((o4_kk >= 5'd3) ? 6'd1 : 6'd0); + wire [5:0] o4_j = 6'(o4_kk) - o4_i * 6'd3; + + // S_BOX: hit k/3, axis k%3, straight off the captured vertices + wire [4:0] bx_h = (k >= 5'd3) ? 5'd9 : 5'd0; + wire [4:0] bx_a = 5'(mod3(k)); + wire [31:0] bx_v0 = cap[bx_h + bx_a]; + wire [31:0] bx_v1 = cap[bx_h + 5'd3 + bx_a]; + wire [31:0] bx_v2 = cap[bx_h + 5'd6 + bx_a]; + + // combinational controls + always @(*) begin + ra = '0; + rb = '0; + fma_issue = 1'b0; + fma_kind = 2'd0; + fma_dst = '0; + fsm_we = 1'b0; + fsm_wa = '0; + fsm_wd = '0; + fw_k = '0; + case (state) + S_CAP: begin + fsm_we = 1'b1; + fsm_wa = RA_WO + 6'(k); + fsm_wd = cap[0]; + end + S_BOX1: begin + fsm_we = 1'b1; + fsm_wa = ((k >= 5'd3) ? RA_BB : RA_BA) + 6'(bx_a); + fsm_wd = f_min(bx_v0, f_min(bx_v1, bx_v2)); + end + S_BOX2: begin + fsm_we = 1'b1; + fsm_wa = ((k >= 5'd3) ? RA_BB : RA_BA) + 6'd3 + 6'(bx_a); + fsm_wd = f_max(bx_v0, f_max(bx_v1, bx_v2)); + end + B_SUB: begin + ra = bt_base + 6'(k); + rb = RA_RO + mod3(k); + fma_issue = 1'b1; + fma_kind = 2'd1; + fma_dst = RA_S + 6'(k); + end + B_MUL: begin + // each product issues as soon as its difference is written back + ra = RA_S + 6'(k); + rb = RA_RI + mod3(k); + fma_issue = (5'(bt_wbc) > k); + fma_kind = 2'd2; + fma_dst = RA_S + 6'(k); + end + B_RED: begin + ra = RA_S + 6'(k); + rb = RA_S + 6'd3 + 6'(k); + end + O_4: begin + // products: X[i*3+j] = wo[j] * m[i][j], X[9+i*3+j] = wd[j] * m[i][j] + ra = ((k >= 5'd9) ? RA_WD : RA_WO) + o4_j; + rb = RA_M + o4_i * 6'd4 + o4_j; + fma_issue = 1'b1; + fma_kind = 2'd2; + fma_dst = RA_X + 6'(k); + end + O_5: begin + // ro[i] = m[i][3] + P[i][0]; rd[i] = Q[i][0] + Q[i][1] + if (k < 5'd3) begin + ra = RA_M + 6'(k) * 6'd4 + 6'd3; + rb = RA_X + 6'(k) * 6'd3; + fma_dst = RA_RO + 6'(k); + end else begin + ra = RA_X + 6'd9 + 6'(k - 5'd3) * 6'd3; + rb = RA_X + 6'd9 + 6'(k - 5'd3) * 6'd3 + 6'd1; + fma_dst = RA_RD + 6'(k - 5'd3); + end + fma_issue = 1'b1; + end + O_6: begin + // ro[i] += P[i][1]; rd[i] += Q[i][2] + if (k < 5'd3) begin + ra = RA_RO + 6'(k); + rb = RA_X + 6'(k) * 6'd3 + 6'd1; + fma_dst = RA_RO + 6'(k); + end else begin + ra = RA_RD + 6'(k - 5'd3); + rb = RA_X + 6'd9 + 6'(k - 5'd3) * 6'd3 + 6'd2; + fma_dst = RA_RD + 6'(k - 5'd3); + end + fma_issue = 1'b1; + end + O_7: begin + // ro[i] += P[i][2] + ra = RA_RO + 6'(k); + rb = RA_X + 6'(k) * 6'd3 + 6'd2; + fma_dst = RA_RO + 6'(k); + fma_issue = 1'b1; + end + R_LD0: begin + ra = RA_RD; + rb = RA_RD + 6'd1; + end + R_LD1: begin + ra = RA_RD + 6'd2; + end + R_WR: begin + fsm_we = 1'b1; + fsm_wa = RA_RI + 6'(k); + fsm_wd = rc_spec[rc_w] ? rc_sval[rc_w] : {dv_s[rc_w], rc_mag}; + end + M_MV: begin + fw_k = mv_k0 + k; + fsm_we = 1'b1; + fsm_wa = mv_dst + 6'(k); + fsm_wd = fw; + end + C_CP: begin + ra = cp_src + 6'(k); + fsm_we = 1'b1; + fsm_wa = mv_dst + 6'(k); + fsm_wd = rda; + end + T_D3B: fw_k = 5'd1; + U_2: fw_k = is_tlas ? 5'd12 : 5'd6; + T_D6, T_D10: fw_k = 5'd13; + default:; + endcase + end + + `RUNTIME_ASSERT(!(wb_v && fsm_we), ("%t: rtu oracle: register-file write conflict", $time)) + + wire [31:0] mul_sum = mul_p + (mul_a[0] ? mul_b : 32'd0); + + always_ff @(posedge clk) begin + fs_start <= 1'b0; + if (reset) begin + state <= S_IDLE; + fma_pend <= '0; + end else begin + fma_pend <= fma_pend + 6'(fma_issue) - 6'(wb_v); + if ((state == B_SUB) || (state == B_MUL)) begin + bt_wbc <= bt_wbc + 3'(wb_v); + end else begin + bt_wbc <= '0; + end + + case (state) + S_IDLE: begin + if (req_valid) begin + ctx_r <= req_ctx; + scene_r <= req_scene; + tlas_r <= req_tlas; + ta <= req_a_t; tb <= req_b_t; + ia <= req_a_inst; ib <= req_b_inst; + psa <= req_a_ps; psb <= req_b_ps; + taba <= req_a_tab; tabb <= req_b_tab; + cap <= {req_b_v, req_a_v, req_wd, req_wo}; + k <= '0; + climbs <= '0; + state <= S_CAP; + end + end + S_CAP: begin + // world ray into the register file; the vertices stay captured + cap <= cap >> 32; + k <= k + 5'd1; + if (k == 5'd5) begin + k <= '0; + state <= S_BOX1; + end + end + S_BOX1: begin + state <= S_BOX2; + end + S_BOX2: begin + k <= k + 5'd1; + state <= S_BOX1; + if (k == 5'd5) begin + k <= '0; + state <= S_DISP; + end + end + S_DISP: begin + state <= (ia == ib) ? T_S1 : T_D0; + end + + // ── same instance: climb its BLAS table ────────────────── + T_S1: begin + if ((psa == ROOT) || (psb == ROOT)) begin + keep <= f_lt(ta, tb) || f_eq(ta, tb); + state <= S_DONE; + end else begin + o_inst <= ia; + o_ret <= T_S2; + state <= O_0; + end + end + T_S2: begin + is_tlas <= 1'b0; + tab_r <= taba; + f_off <= taba + 32'd8; + f_n <= 5'd1; + fs_start <= 1'b1; + f_ret <= T_S3; + state <= F_WAIT; + end + T_S3: begin + noff_r <= fw; + ps0 <= psa; + f_off <= node_off(tab_r, fw, 1'b0, psa) + 32'd28; + fs_start <= 1'b1; + f_ret <= T_S4; + state <= F_WAIT; + end + T_S4: begin + dep0 <= fw + 32'd1; + ps1 <= psb; + f_off <= node_off(tab_r, noff_r, 1'b0, psb) + 32'd28; + fs_start <= 1'b1; + f_ret <= T_S5; + state <= F_WAIT; + end + T_S5: begin + dep1 <= fw + 32'd1; + cul0 <= 1'b0; + cul1 <= 1'b0; + mv_dst <= RA_CB0; + cp_src <= RA_BA; + mv_n <= 5'd12; + c_ret <= T_S6; + state <= C_CP; + end + T_S6: begin + l_ret <= T_S7; + state <= L_0; + end + T_S7: begin + keep <= f_eq(ta, tb) ? a_first + : f_lt(ta, tb) ? (a_first || !cul0) + : (a_first && cul1); + state <= S_DONE; + end + + // ── different instances: climb the TLAS table ──────────── + T_D0: begin + mv_dst <= RA_RO; // world ray o, d -> current ray + cp_src <= RA_WO; + mv_n <= 5'd6; + c_ret <= T_D1; + state <= C_CP; + end + T_D1: begin + // the TLAS header fetch runs beside the reciprocals + is_tlas <= 1'b1; + tab_r <= tlas_r; + f_off <= tlas_r + 32'd8; + f_n <= 5'd2; + fs_start <= 1'b1; + r_ret <= T_D2; + state <= R_LD0; + end + T_D2: begin + f_ret <= T_D3; + state <= F_WAIT; + end + T_D3: begin + noff_r <= fw; + state <= T_D3B; + end + T_D3B: begin + stride_r <= fw; + mul_a <= ia; + mul_b <= fw; + mul_p <= '0; + x_ret <= T_D4; + state <= X_MUL; + end + T_D4: begin + f_off <= tlas_r + 32'd16 + mul_p; + f_n <= 5'd1; + fs_start <= 1'b1; + f_ret <= T_D5; + state <= F_WAIT; + end + T_D5: begin + ps0 <= fw; + if (fw == ROOT) begin + dep0 <= '0; + state <= T_D7; + end else begin + f_off <= node_off(tab_r, noff_r, 1'b1, fw); + f_n <= 5'd14; + fs_start <= 1'b1; + f_ret <= T_D6; + state <= F_WAIT; + end + end + T_D6: begin + dep0 <= fw + 32'd1; + mv_dst <= RA_CB0; + mv_k0 <= ps0[0] ? 5'd6 : 5'd0; + mv_n <= 5'd6; + m_ret <= T_D7; + state <= M_MV; + end + T_D7: begin + mul_a <= ib; + mul_b <= stride_r; + mul_p <= '0; + x_ret <= T_D8; + state <= X_MUL; + end + T_D8: begin + f_off <= tlas_r + 32'd16 + mul_p; + f_n <= 5'd1; + fs_start <= 1'b1; + f_ret <= T_D9; + state <= F_WAIT; + end + T_D9: begin + ps1 <= fw; + if (fw == ROOT) begin + dep1 <= '0; + state <= T_D11; + end else begin + f_off <= node_off(tab_r, noff_r, 1'b1, fw); + f_n <= 5'd14; + fs_start <= 1'b1; + f_ret <= T_D10; + state <= F_WAIT; + end + end + T_D10: begin + dep1 <= fw + 32'd1; + mv_dst <= RA_CB1; + mv_k0 <= ps1[0] ? 5'd6 : 5'd0; + mv_n <= 5'd6; + m_ret <= T_D11; + state <= M_MV; + end + T_D11: begin + cul0 <= 1'b0; + cul1 <= 1'b0; + l_ret <= T_D12; + state <= L_0; + end + T_D12: begin + state <= S_DONE; + if (f_eq(ta, tb)) begin + keep <= a_first; + end else if (f_lt(ta, tb)) begin + if (a_first) begin + keep <= 1'b1; + end else if (cul0) begin + keep <= 1'b0; + end else begin + ph <= 1'b0; + state <= P_0; + end + end else begin + if (!a_first) begin + keep <= 1'b0; + end else if (cul1) begin + keep <= 1'b1; + end else begin + ph <= 1'b1; + state <= P_0; + end + end + end + + // ── the nearer hit's BLAS path, culled against the other t ── + P_0: begin + o_inst <= ph ? ib : ia; + o_ret <= P_1; + state <= O_0; + end + P_1: begin + is_tlas <= 1'b0; + tab_r <= ph ? tabb : taba; + f_off <= (ph ? tabb : taba) + 32'd8; + f_n <= 5'd1; + fs_start <= 1'b1; + f_ret <= P_2; + state <= F_WAIT; + end + P_2: begin + noff_r <= fw; + ps0 <= ph ? psb : psa; + cul0 <= 1'b0; + cur <= 1'b0; + upt <= ph ? ta : tb; + mv_dst <= RA_CB0; + cp_src <= ph ? RA_BB : RA_BA; + mv_n <= 5'd6; + c_ret <= P_3; + state <= C_CP; + end + P_3: begin + // the verdict only needs whether some box fails + if ((ps0 == ROOT) || cul0) begin + keep <= ph ? cul0 : !cul0; + state <= S_DONE; + end else begin + u_ret <= P_3; + state <= U_0; + end + end + + // ── climb both leaves to their lowest common ancestor ──── + L_0: begin + u_ret <= L_0; + if (dep0 > dep1) begin + cur <= 1'b0; + upt <= tb; + state <= U_0; + end else if (dep1 > dep0) begin + cur <= 1'b1; + upt <= ta; + state <= U_0; + end else if (ps0[31:1] != ps1[31:1]) begin + cur <= 1'b0; + upt <= tb; + state <= U_0; + end else begin + bt_base <= RA_CB0; + b_ret <= L_1; + state <= B_SUB; + end + end + L_1: begin + key0 <= bt_key; + bt_base <= RA_CB1; + b_ret <= L_2; + state <= B_SUB; + end + L_2: begin + // child 1 first only on a strictly nearer entry + a_first <= (ps0[0] == f_lt(ps0[0] ? key0 : bt_key, ps0[0] ? bt_key : key0)); + state <= l_ret; + end + + // ── one climb step: test the box, move to the parent ───── + U_0: begin + // the parent link's fetch runs beside the box test + climbs <= climbs + CLIMBW'(1); + if (climbs == CLIMBW'(MAX_CLIMBS)) begin + keep <= f_lt(ta, tb); + end + bt_base <= cur ? RA_CB1 : RA_CB0; + bt_tmax <= upt; + f_off <= node_off(tab_r, noff_r, is_tlas, cur ? ps1 : ps0); + f_n <= is_tlas ? 5'd14 : 5'd8; + fs_start <= (climbs != CLIMBW'(MAX_CLIMBS)); + b_ret <= U_1; + state <= (climbs == CLIMBW'(MAX_CLIMBS)) ? S_DONE : B_SUB; + end + U_1: begin + if (!bt_pass) begin + if (cur) cul1 <= 1'b1; else cul0 <= 1'b1; + end + f_ret <= U_2; + state <= F_WAIT; + end + U_2: begin + if (cur) begin + ps1 <= fw; + dep1 <= dep1 - 32'd1; + end else begin + ps0 <= fw; + dep0 <= dep0 - 32'd1; + end + mv_dst <= cur ? RA_CB1 : RA_CB0; + mv_n <= 5'd6; + m_ret <= u_ret; + if (!is_tlas) begin + // a BLAS node holds its own box, read with its parent link + mv_k0 <= 5'd0; + state <= M_MV; + end else if (fw == ROOT) begin + state <= u_ret; + end else begin + mv_k0 <= fw[0] ? 5'd6 : 5'd0; + f_off <= node_off(tab_r, noff_r, 1'b1, fw); + f_n <= 5'd12; + fs_start <= 1'b1; + f_ret <= U_3; + state <= F_WAIT; + end + end + U_3: begin + state <= M_MV; + end + + // ── box test (the reference's slab test) ───────────────── + B_SUB: begin + if (k == 5'd0) begin + bt_nan <= f_nan(rda[30:0]); + end + k <= k + 5'd1; + if (k == 5'd5) begin + k <= '0; + state <= B_MUL; + end + end + B_MUL: begin + if (fma_issue) begin + k <= k + 5'd1; + if (k == 5'd5) begin + k <= '0; + w_ret <= B_RED; + state <= W_DRAIN; + end + end + end + B_RED: begin + bt_lo <= (k == 5'd0) ? f_min(rda, rdb) : f_max(bt_lo, f_min(rda, rdb)); + bt_hi <= (k == 5'd0) ? f_max(rda, rdb) : f_min(bt_hi, f_max(rda, rdb)); + k <= k + 5'd1; + if (k == 5'd2) begin + k <= '0; + state <= B_FIN; + end + end + B_FIN: begin + // hi >= fmax(0, lo); an empty (NaN) box is never entered + logic hit; + logic [31:0] lo0; + lo0 = f_max(32'd0, bt_lo); + hit = !bt_nan && (f_lt(lo0, bt_hi) || f_eq(lo0, bt_hi)); + bt_key <= hit ? bt_lo : F_INF; + bt_pass <= hit && f_lt(bt_lo, bt_tmax); + state <= b_ret; + end + + // ── the reference's object-space ray for TLAS leaf o_inst ─ + O_0: begin + f_off <= tlas_r + 32'd12; + f_n <= 5'd1; + fs_start <= 1'b1; + f_ret <= O_1; + state <= F_WAIT; + end + O_1: begin + stride_r <= fw; + mul_a <= o_inst; + mul_b <= fw; + mul_p <= '0; + x_ret <= O_2; + state <= X_MUL; + end + O_2: begin + f_off <= tlas_r + 32'd32 + mul_p; + f_n <= 5'd12; + fs_start <= 1'b1; + f_ret <= O_3; + state <= F_WAIT; + end + O_3: begin + mv_dst <= RA_M; + mv_k0 <= '0; + mv_n <= 5'd12; + m_ret <= O_4; + state <= M_MV; + end + O_4: begin + k <= k + 5'd1; + if (k == 5'd17) begin + k <= '0; + w_ret <= O_5; + state <= W_DRAIN; + end + end + O_5: begin + k <= k + 5'd1; + if (k == 5'd5) begin + k <= '0; + w_ret <= O_6; + state <= W_DRAIN; + end + end + O_6: begin + k <= k + 5'd1; + if (k == 5'd5) begin + k <= '0; + w_ret <= O_7; + state <= W_DRAIN; + end + end + O_7: begin + k <= k + 5'd1; + if (k == 5'd2) begin + k <= '0; + w_ret <= O_8; + state <= W_DRAIN; + end + end + O_8: begin + r_ret <= o_ret; + state <= R_LD0; + end + + // ── 1/d per axis, correctly rounded ────────────────────── + R_LD0: begin + {rc_spec[0], rc_sval[0], dv_s[0], dv_m[0], dv_e[0]} <= rc_dec_a; + {rc_spec[1], rc_sval[1], dv_s[1], dv_m[1], dv_e[1]} <= rc_dec_b; + state <= R_LD1; + end + R_LD1: begin + {rc_spec[2], rc_sval[2], dv_s[2], dv_m[2], dv_e[2]} <= rc_dec_a; + for (integer i = 0; i < 3; ++i) begin + dv_rem[i] <= 25'h400000; // 2^22: the dividend's leading bits + dv_q[i] <= '0; + end + dv_it <= '0; + state <= R_IT; + end + R_IT: begin + for (integer i = 0; i < 3; ++i) begin + if ({dv_rem[i][23:0], 1'b0} >= 25'(dv_m[i])) begin + dv_rem[i] <= {dv_rem[i][23:0], 1'b0} - 25'(dv_m[i]); + dv_q[i] <= {dv_q[i][26:0], 1'b1}; + end else begin + dv_rem[i] <= {dv_rem[i][23:0], 1'b0}; + dv_q[i] <= {dv_q[i][26:0], 1'b0}; + end + end + dv_it <= dv_it + 5'd1; + if (dv_it == 5'd27) begin + k <= '0; + state <= R_WR; + end + end + R_WR: begin + k <= k + 5'd1; + if (k == 5'd2) begin + k <= '0; + state <= r_ret; + end + end + + // ── leaf routines ──────────────────────────────────────── + F_WAIT: begin + if (fs_done) begin + state <= f_ret; + end + end + M_MV, C_CP: begin + k <= k + 5'd1; + if ((k + 5'd1) == mv_n) begin + k <= '0; + state <= (state == M_MV) ? m_ret : c_ret; + end + end + X_MUL: begin + mul_p <= mul_sum; + mul_a <= mul_a >> 1; + mul_b <= mul_b << 1; + if (mul_a[31:1] == 31'd0) begin + state <= x_ret; + end + end + W_DRAIN: begin + if (fma_pend == 6'd0) begin + state <= w_ret; + end + end + S_DONE: begin + state <= S_IDLE; + end + default: begin + state <= S_IDLE; + end + endcase + end + end + + assign req_ready = (state == S_IDLE); + assign done_valid = (state == S_DONE); + assign done_ctx = ctx_r; + assign done_keep = keep; + +endmodule diff --git a/hw/rtl/rtu/VX_rtu_scheduler.sv b/hw/rtl/rtu/VX_rtu_scheduler.sv index 84af34c433..2dba1c5815 100644 --- a/hw/rtl/rtu/VX_rtu_scheduler.sv +++ b/hw/rtl/rtu/VX_rtu_scheduler.sv @@ -173,7 +173,9 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( CS_OBJ_SETUP_WT = 5'd26, CS_INST_NEXT = 5'd27, CS_BHDR_REQ = 5'd28, // flat TLAS: BLAS header fetch - CS_BHDR_WAIT = 5'd29; + CS_BHDR_WAIT = 5'd29, + CS_ORC_REQ = 5'd30, // hand a near tie to the oracle + CS_ORC_WAIT = 5'd31; // park: the oracle's verdict // ── the context word: everything only the walker's EXEC touches ─── // One row of the context store. State an async producer writes (fetched @@ -190,6 +192,20 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( logic [31:0] best_ki; logic [27:0] best_kg; logic [31:0] best_kp; + // ... and its place in the source BVH's visit-order tables, which + // settle a near tie with it (VX_rtu_oracle): near_t(best_t), its + // instance rank, parent/side, BLAS table and vertices + logic [31:0] best_tn; + logic [31:0] best_iord; + logic [31:0] best_tord; + logic [31:0] best_btab; + logic [8:0][31:0] best_v; + // the walk's position in those tables: the TLAS table, the current + // instance's BLAS table and rank, the current triangle's parent/side + logic [31:0] tlas_tab; + logic [31:0] blas_tab; + logic [31:0] iord; + logic [31:0] tord; logic [31:0] yld_t; // staged candidate's t (compare copy) logic [31:0] yld_ki; // staged candidate's key: instance id logic [31:0] yld_ko; // ... and record offset @@ -257,6 +273,8 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( reg [NUM_CTX-1:0][RTU_STACK_BITS-1:0] sp_q_arr; reg [NUM_CTX-1:0][LB-1:0] f_slot_q; reg [NUM_CTX-1:0][RTU_CB_ACTION_BITS-1:0] act_q; + reg [NUM_CTX-1:0] orc_q; // the oracle holds the context's memory tag + reg [NUM_CTX-1:0] orc_res_q; // its verdict: the new hit replaces the committed one // ── per-slot state ──────────────────────────────────────────────── reg [NUM_SLOTS-1:0] running; @@ -307,6 +325,11 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( // the staged candidate, so EXEC only adds the t compares reg key_gt_floor_q; reg key_lt_yld_q; + // near-tie classification of the tri result against the committed hit, + // also precomputed at ALIGN: within the near window, and the oracle has + // tables for both hits and they are different triangles + reg near_q; + reg orc_ok_q; ctx_state_t word_q; lane_ray_t ray_q; reg [BUF_BITS-1:0] fbuf_q; @@ -316,7 +339,7 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( reg [15:0] flags_q; reg [15:0] cull_q; reg trihit_q, triback_q; - reg [31:0] trit_q, triu_q, triv_q; + reg [31:0] trit_q, triu_q, triv_q, trin_q; reg [2:0][31:0] xfo_q, xfd_q; reg [2:0][31:0] recip_q; // collector head sampled at ALIGN: EXEC reads these registers instead of @@ -385,7 +408,7 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( .clk (clk), .reset (reset), .read (g1_valid), - .write (mem_rsp_valid && (f_slot_q[mem_rsp_tag] == LB'(s))), + .write (mem_rsp_valid && !orc_q[mem_rsp_tag] && (f_slot_q[mem_rsp_tag] == LB'(s))), .wren (1'b1), .waddr (mem_rsp_tag), .wdata (mem_rsp_data), @@ -400,9 +423,15 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( wire tri_valid_out, tri_hit, tri_back; wire [CTX_TAG_W-1:0] tri_tag_out; wire [31:0] tri_t, tri_u, tri_v; - wire [97:0] trires_rdata; + // the near window's bound rides with the result, off the EXEC path + wire [31:0] tri_tn; + VX_rtu_near_t tri_near ( + .t (tri_t), + .result (tri_tn) + ); + wire [129:0] trires_rdata; VX_dp_ram #( - .DATAW (98), + .DATAW (130), .SIZE (NUM_CTX), .OUT_REG (1), .RDW_MODE ("W") @@ -413,7 +442,7 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( .write (tri_valid_out), .wren (1'b1), .waddr (tri_tag_out), - .wdata ({tri_hit, tri_back, tri_t, tri_u, tri_v}), + .wdata ({tri_hit, tri_back, tri_t, tri_u, tri_v, tri_tn}), .raddr (g1_idx), .rdata (trires_rdata) ); @@ -486,6 +515,42 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( wire [31:0] cand_ki_al = cs_word.in_blas ? cs_word.inst_id : 32'd0; + // IEEE F32 ordered compares (NaN false, +0 == -0) + function automatic logic f_nan(input logic [30:0] a); + f_nan = (a[30:23] == 8'hff) && (a[22:0] != 23'd0); + endfunction + function automatic logic f_eq(input logic [31:0] a, input logic [31:0] b); + f_eq = !f_nan(a[30:0]) && !f_nan(b[30:0]) + && ((a == b) || ((a[30:0] == 31'd0) && (b[30:0] == 31'd0))); + endfunction + function automatic logic f_lt(input logic [31:0] a, input logic [31:0] b); + if (f_nan(a[30:0]) || f_nan(b[30:0]) || ((a[30:0] == 31'd0) && (b[30:0] == 31'd0))) begin + f_lt = 1'b0; + end else if (a[31] != b[31]) begin + f_lt = a[31]; + end else if (!a[31]) begin + f_lt = (a[30:0] < b[30:0]); + end else begin + f_lt = (a[30:0] > b[30:0]); + end + endfunction + + // near window: t == best, or nearer with near_t(t) >= best, or farther + // with t <= near_t(best) + wire [31:0] al_t = trires_rdata[127:96]; + wire [31:0] al_tn = trires_rdata[31:0]; + wire al_near = cs_word.best_kv + && (f_eq(al_t, cs_word.best_t) + || (f_lt(al_t, cs_word.best_t) + && (f_lt(cs_word.best_t, al_tn) || f_eq(cs_word.best_t, al_tn))) + || (f_lt(cs_word.best_t, al_t) + && (f_lt(al_t, cs_word.best_tn) || f_eq(al_t, cs_word.best_tn)))); + wire [31:0] al_iord = cs_word.in_blas ? cs_word.iord : 32'd0; + wire [31:0] al_btab = cs_word.in_blas ? cs_word.blas_tab : 32'd0; + wire al_orc_ok = (FLAT == 0) + && (cs_word.tlas_tab != 32'd0) && (al_btab != 32'd0) && (cs_word.best_btab != 32'd0) + && ({al_iord, cs_word.tord} != {cs_word.best_iord, cs_word.best_tord}); + // ═══════════════════════ stage advance ════════════════════════════ always_ff @(posedge clk) begin if (reset) begin @@ -508,6 +573,8 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( rewalk_q <= s1_rewalk; key_gt_floor_q <= {cand_ki_al, cs_word.cur_off} > {cs_word.floor_ki, cs_word.floor_ko}; key_lt_yld_q <= {cand_ki_al, cs_word.cur_off} < {cs_word.yld_ki, cs_word.yld_ko}; + near_q <= al_near; + orc_ok_q <= al_orc_ok; word_q <= cs_word; ray_q <= lane_ray_t'(ray_rdata); fbuf_q <= fbuf; @@ -519,7 +586,7 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( sp_q <= sp_q_arr[s1_sel]; flags_q <= slot_flags[s1_slot]; cull_q <= slot_cull[s1_slot]; - {trihit_q, triback_q, trit_q, triu_q, triv_q} <= trires_rdata; + {trihit_q, triback_q, trit_q, triu_q, triv_q, trin_q} <= trires_rdata; {xfo_q, xfd_q} <= xfres_rdata; recip_q <= recip_rdata; coll_hit_q <= coll_prochit[cs_word.coll_id]; @@ -694,7 +761,11 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( .ro (walk_ro), .inv_d (walk_inv_d), .t_min (ray_q.t_min), - .t_max (word_q.best_t), + // a node's children are culled with slack past an opaque hit + // committed in this walk, so every hit the near-tie oracle may + // prefer is still reached; a procedural AABB's entry is a + // candidate t, culled at the committed hit itself + .t_max ((box_feed_raw || !word_q.best_kv) ? word_q.best_t : word_q.best_tn), .valid_out (box_valid_out), .tag_out (box_tag_out), .tag_out_pre (box_tag_pre), @@ -779,7 +850,9 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( .v1 (ltri_v1), .v2 (ltri_v2), .t_min (ray_q.t_min), - .t_max (word_q.best_t), + // the ray's own interval: a hit at or past the committed t still + // reaches the tie-break and the near-tie oracle + .t_max (ray_q.t_max), .valid_out (tri_valid_out), .tag_out (tri_tag_out), .hit (tri_hit), @@ -808,6 +881,65 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( .obj_rd (xform_obj_d) ); + // ── near-tie oracle (BVH only) ──────────────────────────────────── + // A context whose opaque hit lands within the near window of the one it + // committed hands both to the oracle and parks. The oracle fetches table + // lines under the context's tag; their responses are its own, never the + // context's fetched-line buffer (which still holds the triangle record). + wire orc_start; + wire orc_req_ready; + wire orc_done_valid, orc_done_keep; + wire [CTX_TAG_W-1:0] orc_done_ctx; + wire orc_mreq_valid; + wire [ADDRW-1:0] orc_mreq_addr; + wire [SLOT_W-1:0] x_slot = SLOT_W'(32'(sel_q) / NUM_LANES); + wire [31:0] cur_iord = word_q.in_blas ? word_q.iord : 32'd0; + wire [31:0] cur_btab = word_q.in_blas ? word_q.blas_tab : 32'd0; + wire [8:0][31:0] ltri_v = {ltri_v2, ltri_v1, ltri_v0}; + if (!FLAT) begin : g_oracle + VX_rtu_oracle #( + .CTX_TAG_W (CTX_TAG_W), + .ADDRW (ADDRW), + .LINE_BITS (LINE_BITS) + ) oracle ( + .clk (clk), + .reset (reset), + .req_valid (orc_start), + .req_ready (orc_req_ready), + .req_ctx (sel_q), + .req_scene (slot_scene[x_slot]), + .req_wo (ray_q.origin), + .req_wd (ray_q.dir), + .req_tlas (word_q.tlas_tab), + .req_a_t (trit_q), + .req_a_inst (cur_iord), + .req_a_ps (word_q.tord), + .req_a_tab (cur_btab), + .req_a_v (ltri_v), + .req_b_t (word_q.best_t), + .req_b_inst (word_q.best_iord), + .req_b_ps (word_q.best_tord), + .req_b_tab (word_q.best_btab), + .req_b_v (word_q.best_v), + .done_valid (orc_done_valid), + .done_ctx (orc_done_ctx), + .done_keep (orc_done_keep), + .mem_req_valid (orc_mreq_valid), + .mem_req_addr (orc_mreq_addr), + .mem_req_ready (mem_req_ready), + .mem_rsp_valid (mem_rsp_valid && orc_q[mem_rsp_tag]), + .mem_rsp_data (mem_rsp_data) + ); + end else begin : g_no_oracle + assign orc_req_ready = 1'b0; + assign orc_done_valid = 1'b0; + assign orc_done_keep = 1'b0; + assign orc_done_ctx = '0; + assign orc_mreq_valid = 1'b0; + assign orc_mreq_addr = '0; + `UNUSED_VAR ({orc_start, x_slot, cur_iord, cur_btab, ltri_v}) + end + // ── reciprocal datapath: pipelined, one axis per issue ──────────── wire [1:0] recip_axis; wire [31:0] recip_din = recip_obj ? word_q.obj_d[recip_axis] : ray_q.dir[recip_axis]; @@ -1248,12 +1380,20 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( reg coll_alloc_r; reg coll_free_r; reg cf_push_r; + reg orc_start_r; commit_t cf_din_r; reg sp_inc, sp_dec; reg stk_wr_r; reg [31:0] stk_wdata_r; - wire mem_fire = x_valid && mem_issue && mem_req_ready; + // the oracle's table fetches go first; a context's fetch retries + wire mem_fire = x_valid && mem_issue && mem_req_ready && !orc_mreq_valid; + + // the TRI_WAIT / ORC_WAIT verdict on an opaque hit + wire in_orc_wait = (word_x.cstate == CS_ORC_WAIT); + wire to_oracle = !in_orc_wait && tri_pass && tri_opaque && near_q && orc_ok_q; + wire opq_take = in_orc_wait ? orc_res_q[sel_q] + : ((tri_committable || tri_tie) && tri_opaque); wire [RTU_CHILD_BITS-1:0] last_child = node.n_children - RTU_CHILD_BITS'(1); @@ -1277,6 +1417,7 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( coll_alloc_r = 1'b0; coll_free_r = 1'b0; cf_push_r = 1'b0; + orc_start_r = 1'b0; cf_din_r = '0; sp_inc = 1'b0; sp_dec = 1'b0; @@ -1409,6 +1550,7 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( end else if (node_kind == RTU_KIND_LEAF_TRI) begin word_n.prim_base = leaf_prim; word_n.geom_r = leaf_geom; + word_n.tord = leaf_flags; // parent/side in its BLAS table word_n.tri_n = 32'(leaf_count); word_n.tri_i = '0; if (leaf_count == 8'd0) begin @@ -1432,6 +1574,11 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( wake_self = 1'b1; end end else if (node_kind == RTU_KIND_LEAF_INST && leaf_count != 8'd0) begin + // header: the BLAS table, the first instance's TLAS rank, + // the TLAS table + word_n.blas_tab = leaf_geom; + word_n.iord = leaf_flags; + word_n.tlas_tab = leaf_prim; word_n.inst_cnt = {24'd0, leaf_count}; word_n.inst_idx = '0; word_n.inst_base = word_x.cur_off + 32'(RTU_LEAF_HDR_BYTES); @@ -1559,10 +1706,14 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( wake_self = 1'b1; end end - CS_TRI_WAIT: begin + CS_TRI_WAIT, CS_ORC_WAIT: begin // woken by the tri PE result (held in its result RAM, so a retry - // on a full commit queue re-reads the same result) - if ((tri_committable || tri_tie) && tri_opaque) begin + // on a full commit queue re-reads the same result), or by the + // oracle's verdict on it + if (to_oracle) begin + word_n.cstate = CS_ORC_REQ; + wake_self = 1'b1; + end else if (opq_take) begin cf_din_r.kind = CK_HIT; cf_din_r.t = trit_q; cf_din_r.u = triu_q; @@ -1581,6 +1732,11 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( word_n.best_ki = tri_ki; word_n.best_kg = tri_kg; word_n.best_kp = tri_kp; + word_n.best_tn = trin_q; + word_n.best_iord = cur_iord; + word_n.best_tord = word_x.tord; + word_n.best_btab = cur_btab; + word_n.best_v = ltri_v; // a closer opaque hit occludes a farther candidate if (yld_q[sel_q] && (word_x.yld_t >= trit_q)) begin exec_yld_clr = 1'b1; @@ -1590,6 +1746,7 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( exec_done = 1'b1; end else if ((word_x.tri_i + 32'd1) < word_x.tri_n) begin word_n.tri_i = word_x.tri_i + 32'd1; + word_n.tord = word_x.tord + 32'd1; word_n.cur_off = word_x.cur_off + 32'(RTU_TRI_STRIDE); word_n.cstate = CS_LTRI_REQ0; wake_self = 1'b1; @@ -1598,7 +1755,8 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( wake_self = 1'b1; end end - end else if (tri_committable + end else if (!in_orc_wait + && tri_committable && above_floor(trit_q) && before_yld(trit_q)) begin cf_din_r.kind = CK_YLDA; @@ -1624,6 +1782,7 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( word_n.yld_ko = word_x.cur_off; if ((word_x.tri_i + 32'd1) < word_x.tri_n) begin word_n.tri_i = word_x.tri_i + 32'd1; + word_n.tord = word_x.tord + 32'd1; word_n.cur_off = word_x.cur_off + 32'(RTU_TRI_STRIDE); word_n.cstate = CS_LTRI_REQ0; end else begin @@ -1634,6 +1793,7 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( end else begin if ((word_x.tri_i + 32'd1) < word_x.tri_n) begin word_n.tri_i = word_x.tri_i + 32'd1; + word_n.tord = word_x.tord + 32'd1; word_n.cur_off = word_x.cur_off + 32'(RTU_TRI_STRIDE); word_n.cstate = CS_LTRI_REQ0; end else begin @@ -1642,6 +1802,15 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( wake_self = 1'b1; end end + CS_ORC_REQ: begin + // the triangle record stays in the line buffer while parked + if (orc_req_ready) begin + orc_start_r = 1'b1; + word_n.cstate = CS_ORC_WAIT; + end else begin + wake_self = 1'b1; + end + end CS_POP: begin if (FLAT) begin if (word_x.in_blas) begin @@ -1774,6 +1943,7 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( end end else begin word_n.inst_idx = word_x.inst_idx + 32'd1; + word_n.iord = word_x.iord + 32'd1; word_n.cur_off = word_x.inst_base + ((word_x.inst_idx + 32'd1) * 32'(RTU_INST_STRIDE)); word_n.cstate = CS_INST_REQ; @@ -1820,9 +1990,11 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( assign cs_wdata = word_n; assign stk_wr = x_valid && stk_wr_r; assign stk_wdata = stk_wdata_r; - assign mem_req_valid = x_valid && mem_issue; - assign mem_req_addr = structaddr_q + (ADDRW'(mem_fslot) << RTU_LINE_SEL_BITS); - assign mem_req_tag = sel_q; + assign orc_start = x_valid && orc_start_r; + assign mem_req_valid = orc_mreq_valid || (x_valid && mem_issue); + assign mem_req_addr = orc_mreq_valid ? orc_mreq_addr + : (structaddr_q + (ADDRW'(mem_fslot) << RTU_LINE_SEL_BITS)); + assign mem_req_tag = orc_mreq_valid ? orc_done_ctx : sel_q; // ═══════════════════════ hot-state update ═════════════════════════ // The wake vector's next state, fed to the SELECT arbiter. Wake events @@ -1852,7 +2024,8 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( wire [NUM_CTX-1:0] rdy_wake_mask = (ray_wr_valid ? NUM_CTX'(1) << ray_wr_ctx : NUM_CTX'(0)) - | (mem_rsp_valid ? NUM_CTX'(1) << mem_rsp_tag : NUM_CTX'(0)) + | ((mem_rsp_valid && !orc_q[mem_rsp_tag]) ? NUM_CTX'(1) << mem_rsp_tag : NUM_CTX'(0)) + | (orc_done_valid ? NUM_CTX'(1) << orc_done_ctx : NUM_CTX'(0)) | (tri_valid_out ? NUM_CTX'(1) << tri_tag_out : NUM_CTX'(0)) | (xform_valid_out ? NUM_CTX'(1) << xform_tag_out : NUM_CTX'(0)) | ((recip_valid_out && recip_last_out) ? NUM_CTX'(1) << recip_tag_out : NUM_CTX'(0)) @@ -1876,6 +2049,7 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( yld_q <= '0; objv_q <= '0; attr_q <= '0; + orc_q <= '0; running <= '0; finalised <= '0; done_r <= '0; @@ -1940,6 +2114,14 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( if (mem_fire) begin f_slot_q[sel_q] <= mem_fslot; end + if (orc_start_r) begin + orc_q[sel_q] <= 1'b1; + end + end + + if (orc_done_valid) begin + orc_q[orc_done_ctx] <= 1'b0; + orc_res_q[orc_done_ctx] <= orc_done_keep; end // resume: capture the actions, queue the walker job diff --git a/tests/raytracing/Makefile b/tests/raytracing/Makefile index da11b9fee6..2b2857e18c 100644 --- a/tests/raytracing/Makefile +++ b/tests/raytracing/Makefile @@ -18,7 +18,7 @@ TESTS := \ rt_smoke_proc rt_smoke_bvh6 rt_bvh_multinode rt_smoke_numctx \ rt_smoke_inst_flags rt_smoke_tlas_builder rt_smoke_deep_stack \ rt_smoke_ahs_custom rt_smoke_fat_leaf rt_smoke_ahs_geom \ - rt_smoke_host_cfg rt_smoke_deep_tlas rt_raycast + rt_smoke_host_cfg rt_smoke_deep_tlas rt_smoke_tie rt_raycast # --- common exclude list --------------------------------------------- EXCLUDE := diff --git a/tests/raytracing/rt_smoke_tie/Makefile b/tests/raytracing/rt_smoke_tie/Makefile new file mode 100644 index 0000000000..db41a4f2ea --- /dev/null +++ b/tests/raytracing/rt_smoke_tie/Makefile @@ -0,0 +1,22 @@ +ROOT_DIR := $(realpath ../../..) +include $(ROOT_DIR)/config.mk + +CONFIGS := $(if $(findstring -DVX_CFG_EXT_RTU_ENABLE,$(CONFIGS)),$(CONFIGS),$(CONFIGS) -DVX_CFG_EXT_RTU_ENABLE) +# CW-BVH4 scene -> build the RTU as a CW-BVH4 walker. +CONFIGS += -DVX_CFG_RTU_BVH_WIDTH=4 + +PROJECT := rt_smoke_tie + +SRC_DIR := $(VORTEX_HOME)/tests/raytracing/$(PROJECT) + +SRCS := $(SRC_DIR)/main.cpp +HDRS := $(SRC_DIR)/common.h $(SRC_DIR)/golden.h + +VX_SRCS := $(SRC_DIR)/kernel.cpp +VX_HDRS := $(SRC_DIR)/common.h + +OPTS ?= + +KERNEL_LIB := vortex2 + +include ../common.mk diff --git a/tests/raytracing/rt_smoke_tie/common.h b/tests/raytracing/rt_smoke_tie/common.h new file mode 100644 index 0000000000..6c8dec3582 --- /dev/null +++ b/tests/raytracing/rt_smoke_tie/common.h @@ -0,0 +1,44 @@ +// Copyright © 2019-2023 +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +#ifndef _RT_SMOKE_TIE_COMMON_H_ +#define _RT_SMOKE_TIE_COMMON_H_ + +#include + +typedef struct { + float origin[3]; + float dir[3]; + float tmin; + float tmax; +} tie_ray_t; + +typedef struct { + uint32_t status; + uint32_t t; // float bits + uint32_t u; + uint32_t v; + uint32_t prim; + uint32_t geom; // geometry index | VX_RT_HIT_BACK_FACING + uint32_t inst_id; + uint32_t inst_custom; +} tie_result_t; + +typedef struct { + uint64_t rays_addr; + uint64_t results_addr; + uint32_t scene; + uint32_t count; +} kernel_arg_t; + +#endif // _RT_SMOKE_TIE_COMMON_H_ diff --git a/tests/raytracing/rt_smoke_tie/golden.h b/tests/raytracing/rt_smoke_tie/golden.h new file mode 100644 index 0000000000..60fdac6ed0 --- /dev/null +++ b/tests/raytracing/rt_smoke_tie/golden.h @@ -0,0 +1,522 @@ +// Generated by rt_smoke_tie -g on the SimX reference. Do not edit. +#pragma once +#include +static const uint32_t kGoldenCount = 256; +static const uint32_t kGolden[2][256][8] = { + { + { 0x0, 0x3ffffffc, 0x80000000, 0x3e800000, 96, 0x80000003, 0, 0x100 }, + { 0x0, 0x403ffffb, 0x80000000, 0x3e800000, 67, 0x00000002, 5, 0x105 }, + { 0x0, 0x3ffffffc, 0x3e800000, 0x3f000000, 113, 0x80000003, 2, 0x102 }, + { 0x0, 0x403ffffc, 0x80000000, 0x3e800000, 101, 0x00000003, 5, 0x105 }, + { 0x0, 0x3ffffffc, 0x3e800000, 0x3f000000, 105, 0x80000003, 2, 0x102 }, + { 0x0, 0x403ffffb, 0x80000000, 0x3e800000, 71, 0x00000002, 5, 0x105 }, + { 0x0, 0x3ffffffc, 0x3e800000, 0x3f000000, 97, 0x80000003, 2, 0x102 }, + { 0x0, 0x403fffff, 0x3e800000, 0x3e800000, 102, 0x00000003, 4, 0x104 }, + { 0x0, 0x3fe00000, 0x80000000, 0x80000000, 0, 0x80000000, 3, 0x103 }, + { 0x0, 0x403ffffc, 0x00000000, 0x3f400000, 64, 0x00000002, 5, 0x105 }, + { 0x0, 0x3ff55555, 0x3eaaaaab, 0x80000000, 2, 0x80000000, 3, 0x103 }, + { 0x0, 0x403ffffb, 0x00000000, 0x3f400000, 66, 0x00000002, 5, 0x105 }, + { 0x0, 0x3ffffffe, 0x80000000, 0x3f000000, 107, 0x80000003, 2, 0x102 }, + { 0x0, 0x403ffffb, 0x00000000, 0x3f400000, 100, 0x00000003, 5, 0x105 }, + { 0x0, 0x3ffffffe, 0x80000000, 0x3f000000, 99, 0x80000003, 2, 0x102 }, + { 0x0, 0x403ffffa, 0x00000000, 0x3f400000, 102, 0x00000003, 5, 0x105 }, + { 0x0, 0x3ffffffc, 0x80000000, 0x3e800000, 104, 0x80000003, 0, 0x100 }, + { 0x0, 0x403aaaab, 0x3f0aaaab, 0x3e000000, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3fe00000, 0x80000000, 0x3e000000, 3, 0x80000000, 3, 0x103 }, + { 0x0, 0x40300000, 0x3f600000, 0x3e000000, 2, 0x00000000, 3, 0x103 }, + { 0x0, 0x3ff55555, 0x3e555555, 0x3e000000, 4, 0x80000000, 3, 0x103 }, + { 0x0, 0x403ffffb, 0x80000000, 0x3e800000, 111, 0x00000003, 5, 0x105 }, + { 0x0, 0x3ffffffe, 0x3e800000, 0x3f000000, 99, 0x80000003, 2, 0x102 }, + { 0x0, 0x403ffffe, 0x3e800000, 0x3e800000, 110, 0x00000003, 4, 0x104 }, + { 0x0, 0x3fe00000, 0x80000000, 0x3f000000, 1, 0x80000000, 3, 0x103 }, + { 0x0, 0x403ffffb, 0x00000000, 0x3f400000, 72, 0x00000002, 5, 0x105 }, + { 0x0, 0x3ff55555, 0x3eaaaaab, 0x3e2aaaab, 3, 0x80000000, 3, 0x103 }, + { 0x0, 0x403aaaab, 0x3e955555, 0x3ec00000, 2, 0x00000000, 3, 0x103 }, + { 0x0, 0x40000000, 0x80000000, 0x80000000, 20, 0x80000000, 4, 0x104 }, + { 0x0, 0x40300000, 0x3f200000, 0x3ec00000, 4, 0x00000000, 3, 0x103 }, + { 0x0, 0x40000000, 0x80000000, 0x80000000, 22, 0x80000000, 4, 0x104 }, + { 0x0, 0x403ffffa, 0x00000000, 0x3f400000, 110, 0x00000003, 5, 0x105 }, + { 0x0, 0x3ffffffc, 0x80000000, 0x3e800000, 112, 0x80000003, 0, 0x100 }, + { 0x0, 0x403aaaab, 0x3d2aaaab, 0x3f200000, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3fe00000, 0x80000000, 0x3f200000, 3, 0x80000000, 3, 0x103 }, + { 0x0, 0x40300000, 0x3ec00000, 0x3f200000, 2, 0x00000000, 3, 0x103 }, + { 0x0, 0x3ff55555, 0x3eaaaaab, 0x3e955555, 5, 0x80000000, 3, 0x103 }, + { 0x0, 0x403ffffb, 0x80000000, 0x3e800000, 119, 0x00000003, 5, 0x105 }, + { 0x0, 0x40000000, 0x3f000000, 0x3e800000, 20, 0x80000000, 4, 0x104 }, + { 0x0, 0x403ffffe, 0x3e800000, 0x3e800000, 118, 0x00000003, 4, 0x104 }, + { 0x0, 0x3fe00000, 0x80000000, 0x3f800000, 1, 0x80000000, 3, 0x103 }, + { 0x0, 0x403ffffc, 0x00000000, 0x3f400000, 80, 0x00000002, 5, 0x105 }, + { 0x0, 0x3ff55555, 0x3eaaaaab, 0x3f2aaaab, 3, 0x80000000, 3, 0x103 }, + { 0x0, 0x403aaaab, 0x3f2aaaab, 0x3e555555, 3, 0x00000000, 3, 0x103 }, + { 0x0, 0x40000000, 0x80000000, 0x80000000, 28, 0x80000000, 4, 0x104 }, + { 0x0, 0x40300000, 0x3e000000, 0x3f600000, 4, 0x00000000, 3, 0x103 }, + { 0x0, 0x40000000, 0x80000000, 0x80000000, 30, 0x80000000, 4, 0x104 }, + { 0x0, 0x403ffffa, 0x00000000, 0x3f400000, 118, 0x00000003, 5, 0x105 }, + { 0x0, 0x3ffffffd, 0x80000000, 0x3e800000, 120, 0x80000003, 0, 0x100 }, + { 0x0, 0x403ffffc, 0x80000000, 0x3e800000, 27, 0x00000000, 5, 0x105 }, + { 0x0, 0x3ffffffe, 0x3f000000, 0x3e800000, 120, 0x80000003, 4, 0x104 }, + { 0x0, 0x403ffffb, 0x80000000, 0x3e800000, 93, 0x00000002, 5, 0x105 }, + { 0x0, 0x40000000, 0x3f000000, 0x3e800000, 122, 0x80000003, 4, 0x104 }, + { 0x0, 0x403ffffb, 0x80000000, 0x3e800000, 127, 0x00000003, 5, 0x105 }, + { 0x0, 0x40000000, 0x3f000000, 0x3e800000, 28, 0x80000000, 4, 0x104 }, + { 0x0, 0x403ffffe, 0x3e800000, 0x3e800000, 126, 0x00000003, 4, 0x104 }, + { 0x0, 0x3ffffffd, 0x80000000, 0x3f800000, 121, 0x80000003, 4, 0x104 }, + { 0x0, 0x403ffffb, 0x00000000, 0x3f400000, 88, 0x00000002, 5, 0x105 }, + { 0x0, 0x3fffffff, 0x80000000, 0x3f800000, 120, 0x80000003, 4, 0x104 }, + { 0x0, 0x403ffffc, 0x00000000, 0x3f400000, 26, 0x00000000, 5, 0x105 }, + { 0x0, 0x40000000, 0x80000000, 0x3f800000, 122, 0x80000003, 4, 0x104 }, + { 0x0, 0x403ffffa, 0x00000000, 0x3f400000, 124, 0x00000003, 5, 0x105 }, + { 0x0, 0x40000000, 0x80000000, 0x3f800000, 28, 0x80000000, 4, 0x104 }, + { 0x0, 0x403ffffa, 0x00000000, 0x3f400000, 126, 0x00000003, 5, 0x105 }, + { 0x0, 0x3f7ffffd, 0x33800000, 0x3d878f51, 97, 0x80000003, 0, 0x100 }, + { 0x0, 0x3f7ffffa, 0x3f000003, 0x3ef794dc, 73, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f7ffff9, 0x359f34db, 0x3f79a6dc, 81, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f6ba787, 0x3e658a10, 0x3f3c1362, 2, 0x80000000, 6, 0x106 }, + { 0x0, 0x3f800000, 0x3f000000, 0x3e1e8470, 23, 0x80000000, 4, 0x104 }, + { 0x0, 0x3f7ffffa, 0x3effffe2, 0x3e4e1b00, 83, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f7ffff8, 0x3f78c284, 0x3ce7af42, 64, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f7ffffe, 0x3ee053bc, 0x3d7d625e, 102, 0x00000003, 4, 0x104 }, + { 0x0, 0x3f800000, 0x3ea018d0, 0x3e3fce60, 22, 0x80000000, 4, 0x104 }, + { 0x0, 0x3f7ffff9, 0x3f000008, 0x3e5d7235, 89, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f760638, 0x3f30bc73, 0x3e87ee1f, 1, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f7ffff9, 0x3f000007, 0x3ea7d5e3, 73, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f800000, 0x331848a2, 0x3f33dbae, 123, 0x80000003, 4, 0x104 }, + { 0x0, 0x3f7ffff9, 0x3d8967f0, 0x3edda620, 10, 0x00000000, 5, 0x105 }, + { 0x0, 0x3f7ffff9, 0x3f000010, 0x3dee19ee, 73, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f6052c3, 0x3f3318d4, 0x3dc04463, 1, 0x80000000, 6, 0x106 }, + { 0x0, 0x3f800000, 0x3e6c8d40, 0x3f44dcb0, 20, 0x80000000, 4, 0x104 }, + { 0x0, 0x3f7ffffd, 0x3f000004, 0x3d11e150, 111, 0x00000003, 4, 0x104 }, + { 0x0, 0x3f64fc82, 0x3bc646d0, 0x3f4c7ef1, 5, 0x00000000, 6, 0x106 }, + { 0x0, 0x3f7ffff8, 0x35593b39, 0x3dd898f2, 73, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f711e1a, 0x3e072700, 0x3d39ae65, 0, 0x80000000, 3, 0x103 }, + { 0x0, 0x3f7ffff8, 0x35400002, 0x3eb95dc0, 103, 0x00000003, 5, 0x105 }, + { 0x0, 0x3f7ffff9, 0x3efffff8, 0x3de88e01, 89, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f7ffff4, 0x3effffb8, 0x3effd986, 127, 0x00000003, 5, 0x105 }, + { 0x0, 0x3f7ffffe, 0x32480a29, 0x3edfd75d, 99, 0x80000003, 4, 0x104 }, + { 0x0, 0x3f5cae6d, 0x3bd39700, 0x3f4d2acb, 0, 0x80000000, 6, 0x106 }, + { 0x0, 0x3f7ffffa, 0x3f17030b, 0x3ed1f9e6, 84, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f7ffff7, 0x3ec09e8d, 0x3f1fb0b7, 116, 0x00000003, 5, 0x105 }, + { 0x0, 0x3f7a839a, 0x3e549290, 0x3e4e945d, 2, 0x80000000, 3, 0x103 }, + { 0x0, 0x3f76e818, 0x3f26eea0, 0x3e667111, 5, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f7ffff9, 0x3db5e363, 0x3ed2872d, 80, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f681e3d, 0x3e549f8d, 0x3f3039b5, 2, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f7ffffc, 0x3f000002, 0x3e71bbd9, 97, 0x80000003, 4, 0x104 }, + { 0x0, 0x3f7ffff6, 0x3f00001a, 0x3efda1b0, 103, 0x00000003, 5, 0x105 }, + { 0x0, 0x3f7ffffa, 0x3f792cae, 0x3cda68fc, 66, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f7ffff8, 0x3ee56754, 0x3d54c4c7, 64, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f6c4aea, 0x3dc95bac, 0x3f42ada5, 7, 0x80000000, 3, 0x103 }, + { 0x0, 0x3f7ffff9, 0x3effffe0, 0x3d8aba81, 73, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f79d359, 0x3daefdf9, 0x3f0be1b6, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f7ffff6, 0x35100001, 0x3f0311eb, 103, 0x00000003, 5, 0x105 }, + { 0x0, 0x3f52c715, 0x3e0ee67a, 0x3be60824, 9, 0x80000001, 6, 0x106 }, + { 0x0, 0x3f7a6dcc, 0x3e9dc06c, 0x3e9eb79b, 4, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f7ffff8, 0x3f141d4a, 0x3ed7c558, 124, 0x00000003, 5, 0x105 }, + { 0x0, 0x3f7ffffe, 0x3f000000, 0x3dc5f281, 103, 0x00000003, 4, 0x104 }, + { 0x0, 0x3f7ffffd, 0x3e4fea82, 0x3e980ac5, 104, 0x80000003, 4, 0x104 }, + { 0x0, 0x3f7ffffa, 0x34b55665, 0x3eaa99c0, 93, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f7ffffd, 0x3f000006, 0x3eb315b8, 111, 0x00000003, 4, 0x104 }, + { 0x0, 0x3f800000, 0x80000000, 0x3eebaa80, 25, 0x00000000, 0, 0x100 }, + { 0x0, 0x3f7fffff, 0x3e6973a4, 0x3effffff, 96, 0x80000003, 2, 0x102 }, + { 0x0, 0x3f7ffff9, 0x3f000003, 0x3e611d40, 65, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f70d836, 0x3ed96353, 0x3e57f246, 5, 0x80000000, 6, 0x106 }, + { 0x0, 0x3f7ffff8, 0x3e374285, 0x3ea45ea0, 122, 0x00000003, 5, 0x105 }, + { 0x0, 0x3f7fffff, 0x3c80fb20, 0x3ef7f04c, 106, 0x80000003, 4, 0x104 }, + { 0x0, 0x3f7ffffa, 0x3efffff1, 0x3e944807, 89, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f7ffff8, 0x3d35a890, 0x3ee94adc, 126, 0x00000003, 5, 0x105 }, + { 0x0, 0x3f7ffff7, 0x3effffd4, 0x3e582798, 125, 0x00000003, 5, 0x105 }, + { 0x0, 0x3f719475, 0x3e7218da, 0x3eaa6b9e, 3, 0x80000000, 3, 0x103 }, + { 0x0, 0x3f6494a2, 0x3f1fb7ed, 0x3ea1fde7, 6, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f6d25ea, 0x3f445833, 0x3e11348c, 2, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f7ffff8, 0x3efffffd, 0x3e5c5ea7, 103, 0x00000003, 5, 0x105 }, + { 0x0, 0x3f7ffffe, 0x343807f7, 0x3f0ff00f, 121, 0x80000003, 4, 0x104 }, + { 0x0, 0x3f7ffff9, 0x344380da, 0x3f63f931, 73, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f7ffffa, 0x3f4152a4, 0x3e7ab55f, 90, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f7ffff7, 0x3effffe3, 0x3e852084, 93, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f7fffff, 0x3f0da5a0, 0x3ee4b4c0, 122, 0x80000003, 4, 0x104 }, + { 0x0, 0x3f7ffffa, 0x34a00000, 0x3f1de4e1, 9, 0x00000000, 5, 0x105 }, + { 0x0, 0x3f5c2c0e, 0x3e2ef1fa, 0x3f3d53a4, 1, 0x80000000, 6, 0x106 }, + { 0x0, 0x3f7ffff8, 0x3ef77fe1, 0x3f04400c, 90, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f800000, 0x3f5f3c5c, 0x3e030e90, 28, 0x80000000, 4, 0x104 }, + { 0x0, 0x3f6d1af5, 0x3e33a054, 0x3f33789d, 4, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f7ffffa, 0x34c00000, 0x3e1a338f, 69, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f66463e, 0x3f1cf9a6, 0x3ea4a508, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f6d250b, 0x3db836a6, 0x3f67868d, 5, 0x80000000, 3, 0x103 }, + { 0x0, 0x3f7ffff9, 0x3e3f4086, 0x3ea05f96, 88, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f79948f, 0x3f23bab4, 0x3e656df9, 1, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f7ffff7, 0x3e60fbae, 0x3f47c0fa, 66, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f800000, 0x3efffffd, 0x3ea2b2aa, 79, 0x80000002, 4, 0x104 }, + { 0x0, 0x3f7ffff8, 0x3ed3c057, 0x3db0feed, 72, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f7ffffe, 0x3edc8264, 0x3d8df67a, 102, 0x00000003, 4, 0x104 }, + { 0x0, 0x3f6fc2c9, 0x3f587690, 0x3e075479, 7, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f7ffffe, 0x3e9236bf, 0x3f000004, 104, 0x80000003, 2, 0x102 }, + { 0x0, 0x3f7b01d8, 0x3ecf4c13, 0x3e5732ce, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f7ffffa, 0x3ed963a9, 0x3d9a71ca, 68, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f7ffff9, 0x3f4cfd30, 0x3e4c0af0, 24, 0x00000000, 5, 0x105 }, + { 0x0, 0x3f7ffffd, 0x3dc53d51, 0x34b3ac29, 112, 0x80000003, 2, 0x102 }, + { 0x0, 0x3f7ffff8, 0x3c88b1f1, 0x3f7bba6f, 88, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f7ffff8, 0x345d935e, 0x3eec9aed, 69, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f763bbb, 0x3eb37160, 0x3eab065a, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f7fffff, 0x3e397208, 0x3f51a37e, 28, 0x80000000, 4, 0x104 }, + { 0x0, 0x3f7b5ec8, 0x3f058734, 0x3da57d7b, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f7ffff8, 0x33debb08, 0x3f75d83c, 127, 0x00000003, 5, 0x105 }, + { 0x0, 0x3f78d7b2, 0x3e9dcb1f, 0x3eaf3a81, 4, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f7ffffd, 0x34fffffc, 0x3e976a28, 97, 0x80000003, 0, 0x100 }, + { 0x0, 0x3f7ffff9, 0x3eb63614, 0x3e1393eb, 66, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f7ffffd, 0x3e965965, 0x3e534d4c, 126, 0x00000003, 4, 0x104 }, + { 0x0, 0x3f7ffff8, 0x3e06ad8a, 0x3ebca923, 88, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f800000, 0x3ed7580c, 0x3da29fd0, 14, 0x80000000, 4, 0x104 }, + { 0x0, 0x3f66a751, 0x3e0696c8, 0x3f5b5f23, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f7ffffe, 0x3f000002, 0x3eb07957, 111, 0x00000003, 4, 0x104 }, + { 0x0, 0x3f7ffff9, 0x3eba09a4, 0x3e0bec7a, 72, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f7fffff, 0x3f000000, 0x3ecf308c, 29, 0x80000000, 4, 0x104 }, + { 0x0, 0x3f7ffffa, 0x350ed981, 0x3eed9810, 65, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f7ffff7, 0x3e9b0a14, 0x3f327af5, 88, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f7ffff7, 0x35a00003, 0x3f188ac0, 119, 0x00000003, 5, 0x105 }, + { 0x0, 0x3f77a9cc, 0x3c9c7f58, 0x3ed2f161, 0, 0x80000000, 6, 0x106 }, + { 0x0, 0x3f7ffff7, 0x35819393, 0x3dc9c931, 127, 0x00000003, 5, 0x105 }, + { 0x0, 0x3f7ffff8, 0x3599f5c4, 0x3f4fae1d, 81, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f7ffff8, 0x35600002, 0x3f1d128e, 111, 0x00000003, 5, 0x105 }, + { 0x0, 0x3f800000, 0x3f648878, 0x3ddbbc40, 28, 0x80000000, 4, 0x104 }, + { 0x0, 0x3f7ffffa, 0x35200000, 0x3ccd882c, 27, 0x00000000, 5, 0x105 }, + { 0x0, 0x3f7ffffe, 0x3e9a0220, 0x3e4bfbd0, 102, 0x00000003, 4, 0x104 }, + { 0x0, 0x3f7ffffd, 0x3f000003, 0x3da72fcc, 111, 0x00000003, 4, 0x104 }, + { 0x0, 0x3f800000, 0x3e151f80, 0x3eb57040, 22, 0x80000000, 4, 0x104 }, + { 0x0, 0x3f7ffffe, 0x3f00000a, 0x3edd2322, 111, 0x00000003, 4, 0x104 }, + { 0x0, 0x3f5a6702, 0x3c4d5104, 0x3f6cfda3, 1, 0x80000000, 6, 0x106 }, + { 0x0, 0x3f64eaef, 0x3d8ea9a7, 0x3f640db7, 6, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f7ffffd, 0x3efffff6, 0x3d944b98, 113, 0x80000003, 0, 0x100 }, + { 0x0, 0x3f7ffffa, 0x3cadbb04, 0x3ef52438, 124, 0x00000003, 5, 0x105 }, + { 0x0, 0x3f7ffff9, 0x3d556639, 0x3ee55350, 88, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f7ffff8, 0x3effffda, 0x3e9cec8b, 83, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f7ffffd, 0x3dbac67d, 0x3ed14e55, 97, 0x80000003, 2, 0x102 }, + { 0x0, 0x3f7ffff9, 0x34de28f9, 0x3ef147c6, 85, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f738af7, 0x3e740404, 0x3f1a12dd, 2, 0x80000000, 6, 0x106 }, + { 0x0, 0x3f7ffffa, 0x3f1b3c9e, 0x3ec986bc, 84, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f800000, 0x3bf5c441, 0x3efc28f1, 122, 0x80000003, 4, 0x104 }, + { 0x0, 0x3f695785, 0x3e11de66, 0x3f3e2607, 2, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f6bfc68, 0x3de2b043, 0x3f15b4a2, 5, 0x00000000, 6, 0x106 }, + { 0x0, 0x3f7ffff9, 0x3f4163b8, 0x3e7a70ce, 108, 0x00000003, 5, 0x105 }, + { 0x0, 0x3f4ff44b, 0x3e293406, 0x3ec7959a, 1, 0x00000000, 6, 0x106 }, + { 0x0, 0x3f532d8b, 0x3f196c52, 0x3eb804a3, 1, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f7cdcd6, 0x3e64bb34, 0x3f33fe34, 66, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f7e2de1, 0x3ef9c2c5, 0x3ef063c4, 90, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f531696, 0x3f1291a6, 0x3cc460ca, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f7eac1d, 0x3f53136e, 0x3e13d4fd, 90, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f535e35, 0x3ea5d7ee, 0x3e900b53, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f5373f3, 0x3ea61344, 0x3e912be7, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f53b04a, 0x3e4a6ad8, 0x3ed5cf27, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f537b21, 0x3f1bd906, 0x3ebe90d2, 1, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f537ef1, 0x3f1bf784, 0x3e298ebf, 1, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f8086e6, 0x3cca58cb, 0x3f02df2f, 93, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f7d37fe, 0x3ed975f9, 0x3f0294f7, 90, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f532e5c, 0x3e4ad678, 0x3ecd7a7d, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f817b05, 0x3d8e21db, 0x3eae59d9, 93, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f53644a, 0x3f1b2249, 0x3e6c1301, 1, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f536ada, 0x3e6d9970, 0x3ebfe0e0, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f7d1615, 0x3f643c3c, 0x3d24840f, 90, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f80641f, 0x3c962e49, 0x3f36282b, 125, 0x00000003, 5, 0x105 }, + { 0x0, 0x3f7cfc7f, 0x3f34250b, 0x3e6717bc, 90, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f7e0010, 0x3ee8bc6a, 0x3eff4456, 66, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f811cfb, 0x3d55bc6f, 0x3dbe83d4, 93, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f53b277, 0x3f089ff1, 0x3da79e02, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f534ae7, 0x3eeb3474, 0x3e12f3ee, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f537348, 0x3cf986e6, 0x3f13ce02, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f4b6fe1, 0x3cc535ac, 0x3f081940, 1, 0x00000000, 6, 0x106 }, + { 0x0, 0x3f80084c, 0x3ac71c69, 0x3db9cd41, 69, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f7edddd, 0x3e77aab9, 0x3f3b487b, 122, 0x00000003, 5, 0x105 }, + { 0x0, 0x3f536fc6, 0x3c0b33db, 0x3f195160, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f80bcc7, 0x3d0d94d9, 0x3e01d49c, 69, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f5334eb, 0x3f099f0e, 0x3d80423a, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f7cabe8, 0x3e41b8e7, 0x3f3b9934, 66, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f532ef9, 0x3f1977c6, 0x3e85db3e, 1, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f535503, 0x3cc2bcc1, 0x3f14922d, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f538608, 0x3ecabf0b, 0x3e5b42e7, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f80a973, 0x3cfe2caa, 0x3ef96bc0, 125, 0x00000003, 5, 0x105 }, + { 0x0, 0x3f535d6c, 0x3f03219a, 0x3dbe4dfd, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f5343a5, 0x3ea800d4, 0x3e8c3979, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f81d8a8, 0x3d8fb50c, 0x3c8627a9, 92, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f5369f1, 0x3e00b76b, 0x3ef64354, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f80688e, 0x3c9cd3f8, 0x3f75bcda, 69, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f5355c1, 0x3ed87cdc, 0x3e39be66, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f7e3aa2, 0x3f735972, 0x3c019519, 90, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f7f9177, 0x3f1a20d9, 0x3ec68fde, 90, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f4facc6, 0x3e20435d, 0x3ecd6d43, 1, 0x00000000, 6, 0x106 }, + { 0x0, 0x3f532fe7, 0x3de1b188, 0x3efa920c, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f539dbd, 0x3f1cede8, 0x3e1d79f8, 1, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f7f6244, 0x3f48709f, 0x3e4f73d8, 90, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f809a25, 0x3ce73741, 0x3f00af59, 69, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f7d9e68, 0x3f05869f, 0x3ed85fa1, 90, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f53ae22, 0x3e2a8867, 0x3ee59de4, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f538038, 0x3e9d4a69, 0x3e9ab90a, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f80e2ce, 0x3d2a1a42, 0x3ef7c7d2, 93, 0x00000002, 5, 0x105 }, + { 0x1, 0x00000000, 0x00000000, 0x00000000, 0, 0x00000000, 0, 0x0 }, + { 0x0, 0x3f53901d, 0x3ee09e4c, 0x3e30c6eb, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f80e782, 0x3d2da155, 0x3e649458, 69, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f53777d, 0x3eb4a8ba, 0x3e82cf14, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f805d9e, 0x3c8c6d1c, 0x3f03f171, 125, 0x00000003, 5, 0x105 }, + { 0x0, 0x3f80f545, 0x3d37f3cd, 0x3e503679, 69, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f538794, 0x3ef8afd2, 0x3dff25a1, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f7f6e29, 0x3c89b8c4, 0x3f78472b, 66, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f5355c8, 0x3f0d7a08, 0x3d53432d, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f531799, 0x3dc676f8, 0x3effdbcb, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f7df3f6, 0x3eb0e7aa, 0x3f1b43ee, 66, 0x00000002, 5, 0x105 }, + }, + { + { 0x0, 0x3ffffffc, 0x80000000, 0x3e800000, 96, 0x80000003, 0, 0x100 }, + { 0x0, 0x403ffffb, 0x3f400000, 0x3e800000, 64, 0x00000002, 5, 0x105 }, + { 0x0, 0x3ffffffc, 0x3e800000, 0x3f000000, 113, 0x80000003, 2, 0x102 }, + { 0x0, 0x403ffffc, 0x3f400000, 0x3e800000, 2, 0x00000000, 5, 0x105 }, + { 0x0, 0x3ffffffc, 0x3e800000, 0x3f000000, 105, 0x80000003, 2, 0x102 }, + { 0x0, 0x403ffffb, 0x3f400000, 0x3e800000, 68, 0x00000002, 5, 0x105 }, + { 0x0, 0x3ffffffc, 0x3e800000, 0x3f000000, 97, 0x80000003, 2, 0x102 }, + { 0x0, 0x403ffffa, 0x3f400000, 0x3e800000, 102, 0x00000003, 5, 0x105 }, + { 0x0, 0x3fe00000, 0x80000000, 0x80000000, 0, 0x80000000, 3, 0x103 }, + { 0x0, 0x403ffffc, 0x00000000, 0x3f400000, 0, 0x00000000, 5, 0x105 }, + { 0x0, 0x3ff55555, 0x3eaaaaab, 0x80000000, 2, 0x80000000, 3, 0x103 }, + { 0x0, 0x403ffffb, 0x00000000, 0x3f400000, 66, 0x00000002, 5, 0x105 }, + { 0x0, 0x3ffffffe, 0x3f000000, 0x3f000000, 104, 0x80000003, 2, 0x102 }, + { 0x0, 0x403ffffb, 0x00000000, 0x3f400000, 100, 0x00000003, 5, 0x105 }, + { 0x0, 0x3ffffffe, 0x3f000000, 0x3f000000, 96, 0x80000003, 2, 0x102 }, + { 0x0, 0x403ffffa, 0x00000000, 0x3f400000, 102, 0x00000003, 5, 0x105 }, + { 0x0, 0x3ffffffc, 0x80000000, 0x3e800000, 104, 0x80000003, 0, 0x100 }, + { 0x0, 0x403aaaab, 0x3f0aaaab, 0x3e000000, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3fe00000, 0x80000000, 0x3e000000, 3, 0x80000000, 3, 0x103 }, + { 0x0, 0x40300000, 0x3f600000, 0x3e000000, 2, 0x00000000, 3, 0x103 }, + { 0x0, 0x3ff55555, 0x3e555555, 0x3e000000, 4, 0x80000000, 3, 0x103 }, + { 0x0, 0x403ffffb, 0x3f400000, 0x3e800000, 108, 0x00000003, 5, 0x105 }, + { 0x0, 0x3ffffffe, 0x3e800000, 0x3f000000, 99, 0x80000003, 2, 0x102 }, + { 0x0, 0x403ffffa, 0x3f400000, 0x3e800000, 110, 0x00000003, 5, 0x105 }, + { 0x0, 0x3fe00000, 0x80000000, 0x3f000000, 1, 0x80000000, 3, 0x103 }, + { 0x0, 0x403ffffb, 0x00000000, 0x3f400000, 72, 0x00000002, 5, 0x105 }, + { 0x0, 0x3ff55555, 0x3eaaaaab, 0x3e2aaaab, 3, 0x80000000, 3, 0x103 }, + { 0x0, 0x403aaaab, 0x3e955555, 0x3ec00000, 2, 0x00000000, 3, 0x103 }, + { 0x0, 0x40000000, 0x3f000000, 0x3f000000, 13, 0x80000000, 0, 0x100 }, + { 0x0, 0x40300000, 0x3f200000, 0x3ec00000, 4, 0x00000000, 3, 0x103 }, + { 0x0, 0x40000000, 0x3f000000, 0x3f000000, 15, 0x80000000, 0, 0x100 }, + { 0x0, 0x403ffffa, 0x00000000, 0x3f400000, 110, 0x00000003, 5, 0x105 }, + { 0x0, 0x3ffffffc, 0x80000000, 0x3e800000, 112, 0x80000003, 0, 0x100 }, + { 0x0, 0x403aaaab, 0x3d2aaaab, 0x3f200000, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3fe00000, 0x80000000, 0x3f200000, 3, 0x80000000, 3, 0x103 }, + { 0x0, 0x40300000, 0x3ec00000, 0x3f200000, 2, 0x00000000, 3, 0x103 }, + { 0x0, 0x3ff55555, 0x3eaaaaab, 0x3e955555, 5, 0x80000000, 3, 0x103 }, + { 0x0, 0x403ffffb, 0x3f400000, 0x3e800000, 84, 0x00000002, 5, 0x105 }, + { 0x0, 0x40000000, 0x80000000, 0x3e800000, 22, 0x80000000, 0, 0x100 }, + { 0x0, 0x403ffffa, 0x3f400000, 0x3e800000, 118, 0x00000003, 5, 0x105 }, + { 0x0, 0x3fe00000, 0x80000000, 0x3f800000, 1, 0x80000000, 3, 0x103 }, + { 0x0, 0x403ffffc, 0x00000000, 0x3f400000, 16, 0x00000000, 5, 0x105 }, + { 0x0, 0x3ff55555, 0x3eaaaaab, 0x3f2aaaab, 3, 0x80000000, 3, 0x103 }, + { 0x0, 0x403aaaab, 0x3f2aaaab, 0x3e555555, 3, 0x00000000, 3, 0x103 }, + { 0x0, 0x40000000, 0x3f000000, 0x3f000000, 21, 0x80000000, 0, 0x100 }, + { 0x0, 0x40300000, 0x3e000000, 0x3f600000, 4, 0x00000000, 3, 0x103 }, + { 0x0, 0x40000000, 0x3f000000, 0x3f000000, 23, 0x80000000, 0, 0x100 }, + { 0x0, 0x403ffffa, 0x00000000, 0x3f400000, 118, 0x00000003, 5, 0x105 }, + { 0x0, 0x3ffffffd, 0x80000000, 0x3e800000, 120, 0x80000003, 0, 0x100 }, + { 0x0, 0x403ffffc, 0x3f400000, 0x3e800000, 24, 0x00000000, 5, 0x105 }, + { 0x0, 0x3ffffffe, 0x80000000, 0x3e800000, 122, 0x80000003, 0, 0x100 }, + { 0x0, 0x403ffffb, 0x3f400000, 0x3e800000, 90, 0x00000002, 5, 0x105 }, + { 0x0, 0x40000000, 0x80000000, 0x3e800000, 28, 0x80000000, 0, 0x100 }, + { 0x0, 0x403ffffb, 0x3f400000, 0x3e800000, 124, 0x00000003, 5, 0x105 }, + { 0x0, 0x40000000, 0x80000000, 0x3e800000, 30, 0x80000000, 0, 0x100 }, + { 0x0, 0x403ffffa, 0x3f400000, 0x3e800000, 126, 0x00000003, 5, 0x105 }, + { 0x0, 0x3ffffffd, 0x80000000, 0x3f800000, 121, 0x80000003, 4, 0x104 }, + { 0x0, 0x403ffffb, 0x00000000, 0x3f400000, 88, 0x00000002, 5, 0x105 }, + { 0x0, 0x3fffffff, 0x80000000, 0x3f800000, 120, 0x80000003, 4, 0x104 }, + { 0x0, 0x403ffffc, 0x00000000, 0x3f400000, 26, 0x00000000, 5, 0x105 }, + { 0x0, 0x40000000, 0x3f000000, 0x3f000000, 29, 0x80000000, 0, 0x100 }, + { 0x0, 0x403ffffa, 0x00000000, 0x3f400000, 124, 0x00000003, 5, 0x105 }, + { 0x0, 0x40000000, 0x3f000000, 0x3f000000, 31, 0x80000000, 0, 0x100 }, + { 0x0, 0x403ffffa, 0x00000000, 0x3f400000, 126, 0x00000003, 5, 0x105 }, + { 0x0, 0x3f7ffffd, 0x33800000, 0x3d878f51, 97, 0x80000003, 0, 0x100 }, + { 0x0, 0x3f7ffffa, 0x3f000003, 0x3ef794dc, 73, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f7ffff9, 0x359f34db, 0x3f79a6dc, 81, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f6ba787, 0x3e658a10, 0x3f3c1362, 2, 0x80000000, 6, 0x106 }, + { 0x0, 0x3f800000, 0x3eb0bdc8, 0x3f27a11c, 22, 0x80000000, 0, 0x100 }, + { 0x0, 0x3f7ffffa, 0x3effffe2, 0x3e4e1b00, 83, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f7ffff8, 0x3f78c284, 0x3ce7af42, 64, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f7ffffe, 0x3ee053bc, 0x3d7d625e, 102, 0x00000003, 4, 0x104 }, + { 0x0, 0x3f800000, 0x3f500c68, 0x3e3fce60, 22, 0x80000000, 0, 0x100 }, + { 0x0, 0x3f7ffff9, 0x3f000008, 0x3e5d7235, 89, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f760638, 0x3f30bc73, 0x3e87ee1f, 1, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f7ffff9, 0x3f000007, 0x3ea7d5e3, 73, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f800000, 0x3f000000, 0x3e4f6ec0, 27, 0x80000000, 0, 0x100 }, + { 0x0, 0x3f7ffff9, 0x3d8967f0, 0x3edda620, 10, 0x00000000, 5, 0x105 }, + { 0x0, 0x3f7ffff9, 0x3f000010, 0x3dee19ee, 73, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f6052c3, 0x3f3318d4, 0x3dc04463, 1, 0x80000000, 6, 0x106 }, + { 0x0, 0x3f800000, 0x3f000000, 0x3e89b960, 23, 0x80000000, 0, 0x100 }, + { 0x0, 0x3f7ffffd, 0x3f000004, 0x3d11e150, 111, 0x00000003, 4, 0x104 }, + { 0x0, 0x3f64fc82, 0x3bc646d0, 0x3f4c7ef1, 5, 0x00000000, 6, 0x106 }, + { 0x0, 0x3f7ffff8, 0x35593b39, 0x3dd898f2, 73, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f711e1a, 0x3e072700, 0x3d39ae65, 0, 0x80000000, 3, 0x103 }, + { 0x0, 0x3f7ffff8, 0x35400002, 0x3eb95dc0, 103, 0x00000003, 5, 0x105 }, + { 0x0, 0x3f7ffff9, 0x3efffff8, 0x3de88e01, 89, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f7ffff4, 0x3effffb8, 0x3effd986, 127, 0x00000003, 5, 0x105 }, + { 0x0, 0x3f7ffffe, 0x3d80a2a0, 0x3edfd758, 98, 0x80000003, 0, 0x100 }, + { 0x0, 0x3f5cae6d, 0x3bd39700, 0x3f4d2acb, 0, 0x80000000, 6, 0x106 }, + { 0x0, 0x3f7ffffa, 0x3f17030b, 0x3ed1f9e6, 84, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f7ffff7, 0x3ec09e8d, 0x3f1fb0b7, 116, 0x00000003, 5, 0x105 }, + { 0x0, 0x3f7a839a, 0x3e549290, 0x3e4e945d, 2, 0x80000000, 3, 0x103 }, + { 0x0, 0x3f76e818, 0x3f26eea0, 0x3e667111, 5, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f7ffff9, 0x3db5e363, 0x3ed2872d, 80, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f681e3d, 0x3e549f8d, 0x3f3039b5, 2, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f7ffffc, 0x3f000002, 0x3e71bbd9, 97, 0x80000003, 4, 0x104 }, + { 0x0, 0x3f7ffff6, 0x3f00001a, 0x3efda1b0, 103, 0x00000003, 5, 0x105 }, + { 0x0, 0x3f7ffffa, 0x3f792cae, 0x3cda6900, 2, 0x00000000, 5, 0x105 }, + { 0x0, 0x3f7ffff8, 0x3ee56754, 0x3d54c4c7, 64, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f6c4aea, 0x3dc95bac, 0x3f42ada5, 7, 0x80000000, 3, 0x103 }, + { 0x0, 0x3f7ffff9, 0x3effffe0, 0x3d8aba81, 73, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f79d359, 0x3daefdf9, 0x3f0be1b6, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f7ffff6, 0x35100001, 0x3f0311eb, 103, 0x00000003, 5, 0x105 }, + { 0x0, 0x3f52c715, 0x3e0ee67a, 0x3be60824, 9, 0x80000001, 6, 0x106 }, + { 0x0, 0x3f7a6dcc, 0x3e9dc06c, 0x3e9eb79b, 4, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f7ffff8, 0x3f141d4a, 0x3ed7c558, 124, 0x00000003, 5, 0x105 }, + { 0x0, 0x3f7ffffe, 0x3f000000, 0x3dc5f27d, 71, 0x00000002, 4, 0x104 }, + { 0x0, 0x3f7ffffd, 0x3e4fea82, 0x3e980ac5, 104, 0x80000003, 4, 0x104 }, + { 0x0, 0x3f7ffffa, 0x34b55665, 0x3eaa99c0, 93, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f7ffffd, 0x3f000006, 0x3eb315b8, 111, 0x00000003, 4, 0x104 }, + { 0x0, 0x3f800000, 0x80000000, 0x3eebaa80, 25, 0x00000000, 0, 0x100 }, + { 0x0, 0x3f7fffff, 0x3e6973a4, 0x3effffff, 96, 0x80000003, 2, 0x102 }, + { 0x0, 0x3f7ffff9, 0x3f000003, 0x3e611d40, 65, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f70d836, 0x3ed96353, 0x3e57f246, 5, 0x80000000, 6, 0x106 }, + { 0x0, 0x3f7ffff8, 0x3e374280, 0x3ea45ea0, 26, 0x00000000, 5, 0x105 }, + { 0x0, 0x3f7fffff, 0x3ef7f04a, 0x338407da, 114, 0x80000003, 2, 0x102 }, + { 0x0, 0x3f7ffffa, 0x3efffff1, 0x3e944807, 89, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f7ffff8, 0x3d35a890, 0x3ee94adc, 126, 0x00000003, 5, 0x105 }, + { 0x0, 0x3f7ffff7, 0x3effffda, 0x3e582790, 93, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f719475, 0x3e7218da, 0x3eaa6b9e, 3, 0x80000000, 3, 0x103 }, + { 0x0, 0x3f6494a2, 0x3f1fb7ed, 0x3ea1fde7, 6, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f6d25ea, 0x3f445833, 0x3e11348c, 2, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f7ffff8, 0x3efffffd, 0x3e5c5ea8, 71, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f7ffffe, 0x3f000002, 0x3d7f0106, 121, 0x80000003, 0, 0x100 }, + { 0x0, 0x3f7ffff9, 0x344380da, 0x3f63f931, 73, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f7ffffa, 0x3f4152a4, 0x3e7ab55f, 90, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f7ffff7, 0x3effffe3, 0x3e852084, 93, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f7fffff, 0x3d5a5a00, 0x3ee4b4c0, 28, 0x80000000, 0, 0x100 }, + { 0x0, 0x3f7ffffa, 0x34a00000, 0x3f1de4e1, 9, 0x00000000, 5, 0x105 }, + { 0x0, 0x3f5c2c0e, 0x3e2ef1fa, 0x3f3d53a4, 1, 0x80000000, 6, 0x106 }, + { 0x0, 0x3f7ffff8, 0x3ef77fe1, 0x3f04400c, 90, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f800000, 0x3ebe78b8, 0x3e030e90, 30, 0x80000000, 0, 0x100 }, + { 0x0, 0x3f6d1af5, 0x3e33a054, 0x3f33789d, 4, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f7ffffa, 0x34c00000, 0x3e1a338f, 69, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f66463e, 0x3f1cf9a6, 0x3ea4a508, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f6d250b, 0x3db836a6, 0x3f67868d, 5, 0x80000000, 3, 0x103 }, + { 0x0, 0x3f7ffff9, 0x3e3f4086, 0x3ea05f96, 88, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f79948f, 0x3f23bab4, 0x3e656df9, 1, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f7ffff7, 0x3e60fbae, 0x3f47c0fa, 66, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f7fffff, 0x3f515954, 0x80000000, 98, 0x80000003, 2, 0x102 }, + { 0x0, 0x3f7ffff8, 0x3ed3c057, 0x3db0feed, 72, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f7ffffe, 0x3edc8264, 0x3d8df67a, 102, 0x00000003, 4, 0x104 }, + { 0x0, 0x3f6fc2c9, 0x3f587690, 0x3e075479, 7, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f7ffffe, 0x3e9236bf, 0x3f000004, 104, 0x80000003, 2, 0x102 }, + { 0x0, 0x3f7b01d8, 0x3ecf4c13, 0x3e5732ce, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f7ffffa, 0x3ed963a9, 0x3d9a71ca, 68, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f7ffff9, 0x3f4cfd30, 0x3e4c0af0, 24, 0x00000000, 5, 0x105 }, + { 0x0, 0x3f7ffffd, 0x3dc53d51, 0x34b3ac29, 112, 0x80000003, 2, 0x102 }, + { 0x0, 0x3f7ffff8, 0x3c88b1f1, 0x3f7bba6f, 88, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f7ffff8, 0x345d935e, 0x3eec9aed, 69, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f763bbb, 0x3eb37160, 0x3eab065a, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f7fffff, 0x3f000000, 0x3ea346fc, 31, 0x80000000, 0, 0x100 }, + { 0x0, 0x3f7b5ec8, 0x3f058734, 0x3da57d7b, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f7ffff8, 0x33debb08, 0x3f75d83c, 127, 0x00000003, 5, 0x105 }, + { 0x0, 0x3f78d7b2, 0x3e9dcb1f, 0x3eaf3a81, 4, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f7ffffd, 0x34fffffc, 0x3e976a28, 97, 0x80000003, 0, 0x100 }, + { 0x0, 0x3f7ffff9, 0x3eb63614, 0x3e1393eb, 66, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f7ffffd, 0x3e965965, 0x3e534d4c, 126, 0x00000003, 4, 0x104 }, + { 0x0, 0x3f7ffff8, 0x3e06ad8a, 0x3ebca923, 88, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f800000, 0x3f6bac06, 0x3da29fd0, 14, 0x80000000, 0, 0x100 }, + { 0x0, 0x3f66a751, 0x3e0696c8, 0x3f5b5f23, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f7ffffe, 0x3f000002, 0x3eb07957, 111, 0x00000003, 4, 0x104 }, + { 0x0, 0x3f7ffff9, 0x3eba09a4, 0x3e0bec7a, 72, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f7fffff, 0x3dc33dd0, 0x3f679846, 28, 0x80000000, 0, 0x100 }, + { 0x0, 0x3f7ffffa, 0x35000000, 0x3eed9814, 1, 0x00000000, 5, 0x105 }, + { 0x0, 0x3f7ffff7, 0x3e9b0a14, 0x3f327af5, 88, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f7ffff7, 0x35a00003, 0x3f188ac0, 119, 0x00000003, 5, 0x105 }, + { 0x0, 0x3f77a9cc, 0x3c9c7f58, 0x3ed2f161, 0, 0x80000000, 6, 0x106 }, + { 0x0, 0x3f7ffff7, 0x35819393, 0x3dc9c931, 127, 0x00000003, 5, 0x105 }, + { 0x0, 0x3f7ffff8, 0x3599f5c4, 0x3f4fae1d, 81, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f7ffff8, 0x35600002, 0x3f1d128e, 111, 0x00000003, 5, 0x105 }, + { 0x0, 0x3f800000, 0x3ec910f0, 0x3ddbbc40, 30, 0x80000000, 0, 0x100 }, + { 0x0, 0x3f7ffffa, 0x35200000, 0x3ccd882c, 27, 0x00000000, 5, 0x105 }, + { 0x0, 0x3f7ffffe, 0x3e9a0220, 0x3e4bfbd0, 102, 0x00000003, 4, 0x104 }, + { 0x0, 0x3f7ffffd, 0x3f000003, 0x3da72fcc, 111, 0x00000003, 4, 0x104 }, + { 0x0, 0x3f800000, 0x3f2547e0, 0x3eb57040, 22, 0x80000000, 0, 0x100 }, + { 0x0, 0x3f7ffffe, 0x3f00000a, 0x3edd2322, 111, 0x00000003, 4, 0x104 }, + { 0x0, 0x3f5a6702, 0x3c4d5104, 0x3f6cfda3, 1, 0x80000000, 6, 0x106 }, + { 0x0, 0x3f64eaef, 0x3d8ea9a7, 0x3f640db7, 6, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f7ffffd, 0x3efffff6, 0x3d944b98, 113, 0x80000003, 0, 0x100 }, + { 0x0, 0x3f7ffffa, 0x3cadbb04, 0x3ef52438, 92, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f7ffff9, 0x3d556639, 0x3ee55350, 88, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f7ffff8, 0x3effffda, 0x3e9cec8b, 83, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f7ffffd, 0x3dbac67d, 0x3ed14e55, 97, 0x80000003, 2, 0x102 }, + { 0x0, 0x3f7ffff9, 0x34de28f9, 0x3ef147c6, 85, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f738af7, 0x3e740404, 0x3f1a12dd, 2, 0x80000000, 6, 0x106 }, + { 0x0, 0x3f7ffffa, 0x3f1b3c9e, 0x3ec986bc, 84, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f800000, 0x3f01eb89, 0x3efc28ee, 26, 0x80000000, 0, 0x100 }, + { 0x0, 0x3f695785, 0x3e11de66, 0x3f3e2607, 2, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f6bfc68, 0x3de2b043, 0x3f15b4a2, 5, 0x00000000, 6, 0x106 }, + { 0x0, 0x3f7ffff9, 0x3f4163b8, 0x3e7a70ce, 108, 0x00000003, 5, 0x105 }, + { 0x0, 0x3f4ff44b, 0x3e293406, 0x3ec7959a, 1, 0x00000000, 6, 0x106 }, + { 0x0, 0x3f532d8b, 0x3f196c52, 0x3eb804a3, 1, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f7cdcd6, 0x3e64bb34, 0x3f33fe34, 66, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f7e2de1, 0x3ef9c2c5, 0x3ef063c4, 90, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f531696, 0x3f1291a6, 0x3cc460ca, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f7eac1d, 0x3f53136e, 0x3e13d4fd, 90, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f535e35, 0x3ea5d7ee, 0x3e900b53, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f5373f3, 0x3ea61344, 0x3e912be7, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f53b04a, 0x3e4a6ad8, 0x3ed5cf27, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f537b21, 0x3f1bd906, 0x3ebe90d2, 1, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f537ef1, 0x3f1bf784, 0x3e298ebf, 1, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f8086e6, 0x3cca58cb, 0x3f02df2f, 93, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f7d37fe, 0x3ed975f9, 0x3f0294f7, 90, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f532e5c, 0x3e4ad678, 0x3ecd7a7d, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f817b05, 0x3d8e21db, 0x3eae59d9, 93, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f53644a, 0x3f1b2249, 0x3e6c1301, 1, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f536ada, 0x3e6d9970, 0x3ebfe0e0, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f7d1615, 0x3f643c3c, 0x3d24840f, 90, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f80641f, 0x3c962e49, 0x3f36282b, 125, 0x00000003, 5, 0x105 }, + { 0x0, 0x3f7cfc7f, 0x3f34250b, 0x3e6717bc, 90, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f7e0010, 0x3ee8bc6a, 0x3eff4456, 66, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f811cfb, 0x3d55bc6f, 0x3dbe83d4, 93, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f53b277, 0x3f089ff1, 0x3da79e02, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f534ae7, 0x3eeb3474, 0x3e12f3ee, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f537348, 0x3cf986e6, 0x3f13ce02, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f4b6fe1, 0x3cc535ac, 0x3f081940, 1, 0x00000000, 6, 0x106 }, + { 0x0, 0x3f80084c, 0x3ac71c69, 0x3db9cd41, 69, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f7edddd, 0x3e77aab9, 0x3f3b487b, 122, 0x00000003, 5, 0x105 }, + { 0x0, 0x3f536fc6, 0x3c0b33db, 0x3f195160, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f80bcc7, 0x3d0d94d9, 0x3e01d49c, 69, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f5334eb, 0x3f099f0e, 0x3d80423a, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f7cabe8, 0x3e41b8e7, 0x3f3b9934, 66, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f532ef9, 0x3f1977c6, 0x3e85db3e, 1, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f535503, 0x3cc2bcc1, 0x3f14922d, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f538608, 0x3ecabf0b, 0x3e5b42e7, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f80a973, 0x3cfe2ba5, 0x3ef96bd2, 93, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f535d6c, 0x3f03219a, 0x3dbe4dfd, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f5343a5, 0x3ea800d4, 0x3e8c3979, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f81d8a8, 0x3d8fb50c, 0x3c8627a9, 92, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f5369f1, 0x3e00b76b, 0x3ef64354, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f80688e, 0x3c9cd3f8, 0x3f75bcda, 69, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f5355c1, 0x3ed87cdc, 0x3e39be66, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f7e3aa2, 0x3f735972, 0x3c019519, 90, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f7f9177, 0x3f1a20d9, 0x3ec68fde, 90, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f4facc6, 0x3e20435d, 0x3ecd6d43, 1, 0x00000000, 6, 0x106 }, + { 0x0, 0x3f532fe7, 0x3de1b188, 0x3efa920c, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f539dbd, 0x3f1cede8, 0x3e1d79f8, 1, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f7f6244, 0x3f48709f, 0x3e4f73d8, 90, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f809a25, 0x3ce73741, 0x3f00af59, 69, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f7d9e68, 0x3f05869f, 0x3ed85fa1, 90, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f53ae22, 0x3e2a8867, 0x3ee59de4, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f538038, 0x3e9d4a69, 0x3e9ab90a, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f80e2ce, 0x3d2a1a42, 0x3ef7c7d2, 93, 0x00000002, 5, 0x105 }, + { 0x1, 0x00000000, 0x00000000, 0x00000000, 0, 0x00000000, 0, 0x0 }, + { 0x0, 0x3f53901d, 0x3ee09e4c, 0x3e30c6eb, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f80e782, 0x3d2da155, 0x3e649458, 69, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f53777d, 0x3eb4a8ba, 0x3e82cf14, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f805d9e, 0x3c8c6c45, 0x3f03f177, 93, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f80f545, 0x3d37f3cd, 0x3e503679, 69, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f538794, 0x3ef8afd2, 0x3dff25a1, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f7f6e29, 0x3c89b8c4, 0x3f78472b, 66, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f5355c8, 0x3f0d7a08, 0x3d53432d, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f531799, 0x3dc676f8, 0x3effdbcb, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f7df3f6, 0x3eb0e7aa, 0x3f1b43ee, 66, 0x00000002, 5, 0x105 }, + }, +}; diff --git a/tests/raytracing/rt_smoke_tie/kernel.cpp b/tests/raytracing/rt_smoke_tie/kernel.cpp new file mode 100644 index 0000000000..42aba68f51 --- /dev/null +++ b/tests/raytracing/rt_smoke_tie/kernel.cpp @@ -0,0 +1,41 @@ +// Copyright © 2019-2023 +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +#include +#include +#include "common.h" + +// One ray per thread against one scene: every hit is opaque, so the trace +// ends on the committed hit. +__kernel void kernel_main(kernel_arg_t* arg) { + uint32_t i = blockIdx.x * blockDim.x + threadIdx.x; + if (i >= arg->count) return; + + const tie_ray_t& r = ((const tie_ray_t*)((uintptr_t)arg->rays_addr))[i]; + vx_ray_t ray = { {r.origin[0], r.origin[1], r.origin[2]}, + {r.dir[0], r.dir[1], r.dir[2]}, + r.tmin, r.tmax }; + uint32_t h = vx_rt_wtrace(arg->scene, 0u, VX_RT_FLAG_OPAQUE, 0xffu, &ray); + vx_hit_t hit; + uint32_t sts = vx_rt_wait(h, &hit); + + tie_result_t* res = (tie_result_t*)((uintptr_t)arg->results_addr) + i; + res->status = sts; + res->t = __builtin_bit_cast(uint32_t, hit.t); + res->u = __builtin_bit_cast(uint32_t, hit.u); + res->v = __builtin_bit_cast(uint32_t, hit.v); + res->prim = hit.primitive_id; + res->geom = hit.geometry_index | (hit.back_facing ? VX_RT_HIT_BACK_FACING : 0u); + res->inst_id = hit.instance_id; + res->inst_custom = hit.instance_custom; +} diff --git a/tests/raytracing/rt_smoke_tie/main.cpp b/tests/raytracing/rt_smoke_tie/main.cpp new file mode 100644 index 0000000000..4cf4e6e5cd --- /dev/null +++ b/tests/raytracing/rt_smoke_tie/main.cpp @@ -0,0 +1,672 @@ +// Copyright © 2019-2023 +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. +// +// rt_smoke_tie — host driver: which of several equal / near-equal-t opaque +// hits the RTU commits. +// +// The scene stacks coincident and near-coincident geometry, within one BLAS +// (exact twin triangles, a copy a few ulps off the plane, a copy tilted by a +// hair) and across instances (the same BLAS instanced twice in place, rotated +// a quarter turn onto itself, shifted by half a cell, lifted by 2^-20). The +// RTU walks its own CW-BVH4; which of those hits it keeps is settled by the +// source BVH's visit order, carried as the visit-order tables the Vulkan +// driver appends (vortexpipe vp_launch.c): a TLAS table inside the scene and +// compact BLAS tables in a separate buffer reached by scene offset, triangle +// leaves naming their parent/side, instance leaves naming their BLAS table, +// TLAS rank and the TLAS table. Here the "source" trees are binary +// median-split trees built differently from the CW-BVH4, so their order is +// not the RTU's own. +// +// The same scene is traced twice: with the tables, and with the instance +// leaves' table words zeroed (the static (instance, geometry, primitive) tie +// key). Expected results are the SimX reference's (golden.h, regenerated with +// -g); every field of a hit is compared bit-exactly, and the two runs must disagree on +// some rays, or the tables were never consulted. + +#include +#include +#include +#include +#include +#include +#include + +#include +#include +#include +#include "common.h" +#include "golden.h" + +#define RT_CHECK(_expr) \ + do { \ + int _ret = _expr; \ + if (0 == _ret) break; \ + printf("Error: '%s' returned %d!\n", #_expr, (int)_ret); \ + cleanup(); \ + exit(-1); \ + } while (false) + +namespace { + +const char* kernel_file = "kernel.vxbin"; + +vx_device_h device = nullptr; +vx_queue_h queue = nullptr; +vx_module_h module_ = nullptr; +vx_kernel_h kernel = nullptr; +vx_buffer_h scene_buf[2] = { nullptr, nullptr }; +vx_buffer_h tab_buf = nullptr; +vx_buffer_h rays_buf = nullptr; +vx_buffer_h res_buf = nullptr; + +void cleanup() { + if (!device) return; + for (auto b : scene_buf) if (b) vx_buffer_release(b); + if (tab_buf) vx_buffer_release(tab_buf); + if (rays_buf) vx_buffer_release(rays_buf); + if (res_buf) vx_buffer_release(res_buf); + if (kernel) vx_kernel_release(kernel); + if (module_) vx_module_release(module_); + if (queue) vx_queue_release(queue); + vx_device_release(device); + device = nullptr; +} + +constexpr uint32_t kRoot = 0xffffffffu; + +struct Tri { + float v[9]; + uint32_t geom; +}; + +struct Instance { + float otw[12]; // object -> world + float wto[12]; // world -> object, as the source driver stores it + uint32_t blas; + uint32_t id; + uint32_t custom; +}; + +struct Box { + float mn[3], mx[3]; +}; + +Box tri_box(const Tri& t) { + Box b; + for (int a = 0; a < 3; ++a) { + b.mn[a] = std::fmin(t.v[a], std::fmin(t.v[3 + a], t.v[6 + a])); + b.mx[a] = std::fmax(t.v[a], std::fmax(t.v[3 + a], t.v[6 + a])); + } + return b; +} + +Box box_union(const Box& x, const Box& y) { + Box b; + for (int a = 0; a < 3; ++a) { + b.mn[a] = std::fmin(x.mn[a], y.mn[a]); + b.mx[a] = std::fmax(x.mx[a], y.mx[a]); + } + return b; +} + +// The world box of a BLAS box under an affine map, widened by an ulp. +Box xform_box(const float m[12], const Box& b) { + Box r; + for (int a = 0; a < 3; ++a) { r.mn[a] = INFINITY; r.mx[a] = -INFINITY; } + for (int c = 0; c < 8; ++c) { + const double p[3] = { (c & 1) ? b.mx[0] : b.mn[0], + (c & 2) ? b.mx[1] : b.mn[1], + (c & 4) ? b.mx[2] : b.mn[2] }; + for (int a = 0; a < 3; ++a) { + const double w = m[a * 4] * p[0] + m[a * 4 + 1] * p[1] + m[a * 4 + 2] * p[2] + m[a * 4 + 3]; + r.mn[a] = std::fmin(r.mn[a], std::nextafter(float(w), -INFINITY)); + r.mx[a] = std::fmax(r.mx[a], std::nextafter(float(w), INFINITY)); + } + } + return r; +} + +// ── the source BVH: a binary median-split tree ────────────────────────── +struct SrcNode { + Box box; + int child[2]; // node index, or ~item for a leaf +}; + +struct SrcTree { + std::vector nodes; + int root = 0; + + int build(const std::vector& boxes, std::vector items, int depth) { + if (items.size() == 1) return ~int(items[0]); + // split on the axis that cycles with depth, so the order differs from + // the CW-BVH4's longest-axis split + const int axis = depth % 3; + std::stable_sort(items.begin(), items.end(), [&](uint32_t a, uint32_t b) { + return boxes[a].mn[axis] + boxes[a].mx[axis] < boxes[b].mn[axis] + boxes[b].mx[axis]; + }); + const size_t h = items.size() / 2; + std::vector l(items.begin(), items.begin() + h), r(items.begin() + h, items.end()); + const int me = int(nodes.size()); + nodes.push_back(SrcNode()); + const int c0 = build(boxes, l, depth + 1); + const int c1 = build(boxes, r, depth + 1); + nodes[me].child[0] = c0; + nodes[me].child[1] = c1; + auto cbox = [&](int c) { return c < 0 ? boxes[~c] : nodes[c].box; }; + nodes[me].box = box_union(cbox(c0), cbox(c1)); + return me; + } + + void make(const std::vector& boxes) { + nodes.clear(); + std::vector items(boxes.size()); + for (uint32_t i = 0; i < items.size(); ++i) items[i] = i; + root = build(boxes, items, 0); + } + + Box child_box(const std::vector& boxes, int c) const { + return c < 0 ? boxes[~c] : nodes[c].box; + } +}; + +// The source driver's table order: depth first, child 0 first. Table node +// j records its parent/side and depth; each leaf its parent/side, and leaves +// are ranked in visit order. +struct SrcTable { + struct Node { int src; uint32_t ps, depth; }; + std::vector nodes; + std::vector leaf_ps; // per item + std::vector leaf_rank; // per item + std::vector rank_item; // per rank + + void make(const SrcTree& t, size_t n_items) { + nodes.clear(); + leaf_ps.assign(n_items, kRoot); + leaf_rank.assign(n_items, 0); + rank_item.clear(); + std::vector> st = { { t.root, kRoot } }; + while (!st.empty()) { + auto [c, ps] = st.back(); + st.pop_back(); + if (c < 0) { + leaf_ps[~c] = ps; + leaf_rank[~c] = uint32_t(rank_item.size()); + rank_item.push_back(uint32_t(~c)); + continue; + } + const uint32_t idx = uint32_t(nodes.size()); + nodes.push_back({ c, ps, ps == kRoot ? 0u : nodes[ps >> 1].depth + 1 }); + st.push_back({ t.nodes[c].child[1], idx << 1 | 1u }); + st.push_back({ t.nodes[c].child[0], idx << 1 }); + } + } +}; + +// ── byte image helpers ────────────────────────────────────────────────── +struct Image { + std::vector b; + uint32_t alloc(uint32_t n, uint32_t align = 4) { + const uint32_t off = uint32_t((b.size() + align - 1) & ~size_t(align - 1)); + b.resize(off + n, 0); + return off; + } + void u32(uint32_t off, uint32_t v) { std::memcpy(&b[off], &v, 4); } + void f32(uint32_t off, float v) { std::memcpy(&b[off], &v, 4); } + void box(uint32_t off, const Box& x) { + for (int a = 0; a < 3; ++a) { f32(off + 4 * a, x.mn[a]); f32(off + 12 + 4 * a, x.mx[a]); } + } +}; + +// ── the RTU's CW-BVH4: post-order, children before parents ───────────── +struct RtuRef { uint32_t off; Box box; bool leaf; }; + +RtuRef emit_internal(Image& img, const std::vector& ch) { + Box nb = ch[0].box; + for (size_t i = 1; i < ch.size(); ++i) nb = box_union(nb, ch[i].box); + int8_t ex[3]; + float step[3]; + for (int a = 0; a < 3; ++a) { + const float ext = nb.mx[a] - nb.mn[a]; + int e = -20; + // headroom: 250 steps must cover the extent, so the 255 the quantizer + // may round up to still lies past it + if (ext > 0.f) std::frexp(ext / 250.0f, &e); + ex[a] = int8_t(std::max(-100, std::min(100, e))); + step[a] = std::ldexp(1.0f, ex[a]); + } + const uint32_t off = img.alloc(RTU_BVH_NODE4_BYTES); + img.u32(off, RTU_BVH_KIND_INTERNAL | (uint32_t(ch.size()) << RTU_BVH_COUNT_SHIFT)); + for (int a = 0; a < 3; ++a) img.f32(off + RTU_BVH_NODE_ORIGIN_OFF + 4 * a, nb.mn[a]); + std::memcpy(&img.b[off + RTU_BVH_NODE_EXP_OFF], ex, 3); + const uint32_t qmin_off = RTU_BVH_NODE_CHILD_OFF + 4 * 4; + const uint32_t qmax_off = qmin_off + 3 * 4; + for (size_t i = 0; i < ch.size(); ++i) { + img.u32(off + RTU_BVH_NODE_CHILD_OFF + 4 * uint32_t(i), + ch[i].off | (ch[i].leaf ? RTU_BVH_CHILD_LEAF_FLAG : 0u)); + for (int a = 0; a < 3; ++a) { + // conservative by a step either way + int qmn = int(std::floor((ch[i].box.mn[a] - nb.mn[a]) / step[a])) - 1; + int qmx = int(std::ceil((ch[i].box.mx[a] - nb.mn[a]) / step[a])) + 1; + qmn = std::max(0, std::min(255, qmn)); + qmx = std::max(qmn, std::min(255, qmx)); + img.b[off + qmin_off + 3 * i + a] = uint8_t(qmn); + img.b[off + qmax_off + 3 * i + a] = uint8_t(qmx); + } + } + return { off, nb, false }; +} + +// Groups of up to four along the longest centroid axis, recursively. +template +RtuRef emit_cwbvh(Image& img, const std::vector& boxes, std::vector items, + EmitLeaf&& leaf) { + if (items.size() == 1) return { leaf(items[0]), boxes[items[0]], true }; + float cmn[3] = { INFINITY, INFINITY, INFINITY }, cmx[3] = { -INFINITY, -INFINITY, -INFINITY }; + for (uint32_t i : items) + for (int a = 0; a < 3; ++a) { + const float c = boxes[i].mn[a] + boxes[i].mx[a]; + cmn[a] = std::min(cmn[a], c); + cmx[a] = std::max(cmx[a], c); + } + int axis = 0; + for (int a = 1; a < 3; ++a) + if (cmx[a] - cmn[a] > cmx[axis] - cmn[axis]) axis = a; + std::stable_sort(items.begin(), items.end(), [&](uint32_t a, uint32_t b) { + return boxes[a].mn[axis] + boxes[a].mx[axis] > boxes[b].mn[axis] + boxes[b].mx[axis]; + }); + const size_t g = std::min(4, items.size()); + std::vector ch; + for (size_t k = 0; k < g; ++k) { + const size_t lo = items.size() * k / g, hi = items.size() * (k + 1) / g; + ch.push_back(emit_cwbvh(img, boxes, std::vector(items.begin() + lo, items.begin() + hi), leaf)); + } + return emit_internal(img, ch); +} + +// ── the scene ─────────────────────────────────────────────────────────── +std::vector> blases; +std::vector insts; + +void make_geometry() { + // BLAS 0, the "sail": a 4x4 grid of quads on z = 0 over [-1, 1]^2 (geometry + // 0); an exact twin of it (geometry 1); a copy a few ulps above the plane on + // some vertices (geometry 2); a copy tilted by a hair, crossing the plane + // along x = 0 (geometry 3). + std::vector sail; + auto quad = [&](float x0, float y0, float x1, float y1, uint32_t geom, auto zf) { + Tri a = { { x0, y0, zf(x0, y0), x1, y0, zf(x1, y0), x1, y1, zf(x1, y1) }, geom }; + Tri b = { { x0, y0, zf(x0, y0), x1, y1, zf(x1, y1), x0, y1, zf(x0, y1) }, geom }; + sail.push_back(a); + sail.push_back(b); + }; + for (uint32_t g = 0; g < 4; ++g) { + for (int j = 0; j < 4; ++j) { + for (int i = 0; i < 4; ++i) { + const float x0 = -1.f + 0.5f * i, y0 = -1.f + 0.5f * j; + auto z = [&](float x, float y) -> float { + (void)i; (void)j; + switch (g) { + // every other grid vertex: about an ulp of t for these rays + case 2: return (int((x + 1.f) * 4.f + (y + 1.f) * 4.f) & 2) ? 0x1p-22f : 0.f; + case 3: return x * 0x1p-21f + y * 0x1p-23f; + default: return 0.f; + } + }; + quad(x0, y0, x0 + 0.5f, y0 + 0.5f, g, z); + } + } + } + blases.push_back(sail); + + // BLAS 1: a small plate tilted across the sail, plus an exact copy of two + // sail triangles (so an instance of it coincides with the sail there). + std::vector plate; + for (int i = 0; i < 4; ++i) { + const float x0 = -0.75f + 0.375f * i; + plate.push_back({ { x0, -0.5f, -0.25f, x0 + 0.375f, -0.5f, 0.25f, x0 + 0.375f, 0.5f, 0.25f }, 0 }); + plate.push_back({ { x0, -0.5f, -0.25f, x0 + 0.375f, 0.5f, 0.25f, x0, 0.5f, -0.25f }, 0 }); + } + plate.push_back({ sail[10].v[0], sail[10].v[1], sail[10].v[2], sail[10].v[3], sail[10].v[4], + sail[10].v[5], sail[10].v[6], sail[10].v[7], sail[10].v[8], 1 }); + plate.push_back({ sail[11].v[0], sail[11].v[1], sail[11].v[2], sail[11].v[3], sail[11].v[4], + sail[11].v[5], sail[11].v[6], sail[11].v[7], sail[11].v[8], 1 }); + blases.push_back(plate); + + auto inst = [&](uint32_t blas, uint32_t id, const float otw[12], const float wto[12]) { + Instance in; + std::memcpy(in.otw, otw, sizeof(in.otw)); + std::memcpy(in.wto, wto, sizeof(in.wto)); + in.blas = blas; + in.id = id; + in.custom = 0x100 + id; + insts.push_back(in); + }; + const float I[12] = { 1, 0, 0, 0, 0, 1, 0, 0, 0, 0, 1, 0 }; + // a quarter turn about z maps the grid onto itself + const float R[12] = { 0, -1, 0, 0, 1, 0, 0, 0, 0, 0, 1, 0 }; + const float Ri[12] = { 0, 1, 0, 0, -1, 0, 0, 0, 0, 0, 1, 0 }; + // half a cell along x + const float T[12] = { 1, 0, 0, 0.25f, 0, 1, 0, 0, 0, 0, 1, 0 }; + const float Ti[12] = { 1, 0, 0, -0.25f, 0, 1, 0, 0, 0, 0, 1, 0 }; + // lifted off the plane by 2^-20 + const float U[12] = { 1, 0, 0, 0, 0, 1, 0, 0, 0, 0, 1, 0x1p-20f }; + const float Ui[12] = { 1, 0, 0, 0, 0, 1, 0, 0, 0, 0, 1, -0x1p-20f }; + // the plate: a quarter turn about x, then shifted (the RTU inverts an + // instance transform as a rotation, so instances stay orthonormal) + const float P[12] = { 1, 0, 0, 0.125f, 0, 0, -1, 0, 0, 1, 0, 0 }; + const float Pi[12] = { 1, 0, 0, -0.125f, 0, 0, 1, 0, 0, -1, 0, 0 }; + inst(0, 0, I, I); + inst(0, 1, I, I); + inst(0, 2, R, Ri); + inst(1, 3, I, I); + inst(0, 4, T, Ti); + inst(0, 5, U, Ui); + inst(1, 6, P, Pi); +} + +std::vector make_rays() { + std::vector rays; + auto ray = [&](float ox, float oy, float oz, float dx, float dy, float dz) { + rays.push_back({ { ox, oy, oz }, { dx, dy, dz }, 0.001f, 100.f }); + }; + // straight down and straight up, on cell interiors, edges and vertices + for (int j = 0; j < 8; ++j) + for (int i = 0; i < 8; ++i) { + const float x = -0.875f + 0.25f * i, y = -0.875f + 0.25f * j; + const float ex = ((i + j) & 1) ? 0.125f : 0.f; // half the rays sit on an edge + if ((i + j * 8) & 1) ray(x + ex, y, 3.f, 0.f, 0.f, -1.f); + else ray(x + ex, y + ex, -2.f, 0.f, 0.f, 1.f); + } + // oblique, aimed at grid lines and vertices from scattered origins + uint32_t s = 12345u; + auto rnd = [&]() { s = s * 1664525u + 1013904223u; return float(s >> 8) / float(1u << 24); }; + for (int k = 0; k < 128; ++k) { + const float tx = -1.f + 0.25f * float(int(rnd() * 9.f)); + const float ty = -1.f + 0.5f * rnd() * 4.f; + const float ox = tx + (rnd() - 0.5f) * 3.f, oy = ty + (rnd() - 0.5f) * 3.f; + const float oz = (k & 3) ? 2.f + rnd() : -2.f - rnd(); + ray(ox, oy, oz, tx - ox, ty - oy, -oz); + } + // shallow rays skimming the plate and the sail + for (int k = 0; k < 64; ++k) { + const float ox = -3.f, oy = -0.9f + 1.8f * rnd(); + const float tz = (rnd() - 0.5f) * 0.02f; + ray(ox, oy, 0.3f + tz, 3.f, (rnd() - 0.5f) * 0.4f, -0.3f - tz * 0.5f); + } + return rays; +} + +// Build the scene image for one scene base address. tab_base is the BLAS +// table buffer's address, or 0 to leave the tables out. +struct Built { + Image scene; + Image tabs; + uint32_t root = 0; +}; + +Built build(uint64_t scene_addr, uint64_t tab_addr, bool with_tables) { + Built out; + Image& img = out.scene; + img.alloc(RTU_BVH_SCENE_HDR_BYTES); + + // BLAS: source trees + tables, then the CW-BVH4 with each leaf naming its + // parent/side in the source tree + std::vector blas_root(blases.size()), blas_tab(blases.size(), 0); + std::vector blas_box(blases.size()); + for (size_t b = 0; b < blases.size(); ++b) { + const auto& tris = blases[b]; + std::vector boxes; + for (const auto& t : tris) boxes.push_back(tri_box(t)); + SrcTree st; + st.make(boxes); + SrcTable tab; + tab.make(st, tris.size()); + + // BLAS table: { n_nodes, 0, 64, 32 }, node[j] = { own box, ps, depth } + const uint32_t n = uint32_t(tab.nodes.size()); + const uint32_t toff = out.tabs.alloc(64 + 32 * n, 64); + out.tabs.u32(toff, n); + out.tabs.u32(toff + 8, 64); + out.tabs.u32(toff + 12, 32); + for (uint32_t j = 0; j < n; ++j) { + const auto& nd = tab.nodes[j]; + const uint32_t o = toff + 64 + 32 * j; + if (nd.ps != kRoot) { + const SrcNode& parent = st.nodes[tab.nodes[nd.ps >> 1].src]; + out.tabs.box(o, st.child_box(boxes, parent.child[nd.ps & 1])); + } + out.tabs.u32(o + 24, nd.ps); + out.tabs.u32(o + 28, nd.depth); + } + blas_tab[b] = uint32_t(tab_addr + toff - scene_addr); + + std::vector items(tris.size()); + for (uint32_t i = 0; i < items.size(); ++i) items[i] = i; + RtuRef r = emit_cwbvh(img, boxes, items, [&](uint32_t i) { + const uint32_t off = img.alloc(RTU_BVH_LEAF_HDR_BYTES + RTU_BVH_TRI_STRIDE); + img.u32(off, RTU_BVH_KIND_LEAF_TRI | (1u << RTU_BVH_COUNT_SHIFT)); + img.u32(off + 4, tris[i].geom); + img.u32(off + 8, tab.leaf_ps[i]); + img.u32(off + 12, i); + for (int k = 0; k < 9; ++k) img.f32(off + RTU_BVH_LEAF_HDR_BYTES + 4 * k, tris[i].v[k]); + img.u32(off + RTU_BVH_LEAF_HDR_BYTES + 36, RTU_BVH_FLAG_OPAQUE); + return off; + }); + blas_root[b] = r.off; + blas_box[b] = r.box; + } + + // TLAS: source tree over the instances' world boxes; table in the scene + std::vector wbox; + for (const auto& in : insts) wbox.push_back(xform_box(in.otw, blas_box[in.blas])); + SrcTree tt; + tt.make(wbox); + SrcTable ttab; + ttab.make(tt, insts.size()); + const uint32_t nl = uint32_t(insts.size()), nn = uint32_t(ttab.nodes.size()); + const uint32_t noff = (16 + nl * 64 + 63) & ~63u; + const uint32_t tlas_tab = img.alloc(noff + 64 * nn, 64); + img.u32(tlas_tab, nl); + img.u32(tlas_tab + 4, nn); + img.u32(tlas_tab + 8, noff); + img.u32(tlas_tab + 12, 64); + for (uint32_t r = 0; r < nl; ++r) { + const uint32_t item = ttab.rank_item[r]; + const uint32_t o = tlas_tab + 16 + 64 * r; + img.u32(o, ttab.leaf_ps[item]); + for (int k = 0; k < 12; ++k) img.f32(o + 16 + 4 * k, insts[item].wto[k]); + } + for (uint32_t j = 0; j < nn; ++j) { + const SrcNode& nd = tt.nodes[ttab.nodes[j].src]; + const uint32_t o = tlas_tab + noff + 64 * j; + img.box(o, tt.child_box(wbox, nd.child[0])); + img.box(o + 24, tt.child_box(wbox, nd.child[1])); + img.u32(o + 48, ttab.nodes[j].ps); + img.u32(o + 52, ttab.nodes[j].depth); + } + + std::vector items(insts.size()); + for (uint32_t i = 0; i < items.size(); ++i) items[i] = i; + RtuRef root = emit_cwbvh(img, wbox, items, [&](uint32_t i) { + const Instance& in = insts[i]; + const uint32_t off = img.alloc(RTU_BVH_LEAF_HDR_BYTES + RTU_BVH_INSTANCE_STRIDE); + img.u32(off, RTU_BVH_KIND_LEAF_INST | (1u << RTU_BVH_COUNT_SHIFT)); + if (with_tables) { + img.u32(off + 4, blas_tab[in.blas]); + img.u32(off + 8, ttab.leaf_rank[i]); + img.u32(off + 12, tlas_tab); + } + const uint32_t rec = off + RTU_BVH_LEAF_HDR_BYTES; + for (int k = 0; k < 12; ++k) img.f32(rec + 4 * k, in.otw[k]); + img.u32(rec + RTU_BVH_INSTANCE_BLAS_OFF, blas_root[in.blas]); + img.u32(rec + RTU_BVH_INSTANCE_CUSTOM_OFF, in.custom); + img.u32(rec + RTU_BVH_INSTANCE_ID_OFF, in.id); + img.u32(rec + RTU_BVH_INSTANCE_CULL_OFF, 0xffu); + return off; + }); + out.root = root.off; + img.u32(0, root.off); + img.u32(4, RTU_SCENE_KIND_BVH4); + img.u32(8, uint32_t(img.b.size())); + img.u32(12, uint32_t(insts.size())); + img.alloc(0, 64); + return out; +} + +const char* field_name[8] = { "status", "t", "u", "v", "prim", "geom", "inst", "custom" }; + +} // namespace + +int main(int argc, char* argv[]) { + const char* golden_out = nullptr; + int c; + while ((c = getopt(argc, argv, "g:h")) != -1) { + if (c == 'g') golden_out = optarg; + else { printf("usage: %s [-g golden.h]\n", argv[0]); return 0; } + } + + make_geometry(); + std::vector rays = make_rays(); + const uint32_t NT = VX_CFG_NUM_THREADS; + while (rays.size() % NT) rays.push_back(rays.back()); + const uint32_t count = uint32_t(rays.size()); + + RT_CHECK(vx_device_open(0, &device)); + vx_queue_info_t qi = { sizeof(qi), nullptr, VX_QUEUE_PRIORITY_NORMAL, 0 }; + RT_CHECK(vx_queue_create(device, &qi, &queue)); + RT_CHECK(vx_module_load_file(device, kernel_file, &module_)); + RT_CHECK(vx_module_get_kernel(module_, "main", &kernel)); + + // Sizes do not depend on the addresses: size the buffers from a dry build. + const Built dry = build(0, 0, true); + const uint32_t scene_bytes = uint32_t(dry.scene.b.size()); + const uint32_t tab_bytes = uint32_t(dry.tabs.b.size()); + uint64_t scene_addr[2], tab_addr = 0; + for (int k = 0; k < 2; ++k) { + RT_CHECK(vx_buffer_create(device, scene_bytes, VX_MEM_READ, &scene_buf[k])); + RT_CHECK(vx_buffer_address(scene_buf[k], &scene_addr[k])); + } + RT_CHECK(vx_buffer_create(device, tab_bytes, VX_MEM_READ, &tab_buf)); + RT_CHECK(vx_buffer_address(tab_buf, &tab_addr)); + if (tab_addr <= scene_addr[0] || tab_addr - scene_addr[0] >= (1ull << 31)) { + printf("Error: the BLAS table buffer is not reachable from the scene\n"); + cleanup(); + return 1; + } + + RT_CHECK(vx_buffer_create(device, count * sizeof(tie_ray_t), VX_MEM_READ, &rays_buf)); + RT_CHECK(vx_buffer_create(device, count * sizeof(tie_result_t), VX_MEM_WRITE, &res_buf)); + uint64_t rays_addr = 0, res_addr = 0; + RT_CHECK(vx_buffer_address(rays_buf, &rays_addr)); + RT_CHECK(vx_buffer_address(res_buf, &res_addr)); + RT_CHECK(vx_enqueue_write(queue, rays_buf, 0, rays.data(), count * sizeof(tie_ray_t), + 0, nullptr, nullptr)); + + std::vector got[2]; + for (int k = 0; k < 2; ++k) { + const bool with_tables = (k == 0); + const Built bi = build(scene_addr[k], tab_addr, with_tables); + RT_CHECK(vx_enqueue_write(queue, scene_buf[k], 0, bi.scene.b.data(), scene_bytes, + 0, nullptr, nullptr)); + if (with_tables) + RT_CHECK(vx_enqueue_write(queue, tab_buf, 0, bi.tabs.b.data(), tab_bytes, + 0, nullptr, nullptr)); + + kernel_arg_t arg = {}; + arg.rays_addr = rays_addr; + arg.results_addr = res_addr; + arg.scene = uint32_t(scene_addr[k]); + arg.count = count; + + vx_event_h launch_ev = nullptr, read_ev = nullptr; + vx_launch_info_t li = {}; + li.struct_size = sizeof(li); + li.kernel = kernel; + li.args_host = &arg; + li.args_size = sizeof(arg); + li.ndim = 1; + li.grid_dim[0] = count / NT; + li.block_dim[0] = NT; + RT_CHECK(vx_enqueue_launch(queue, &li, 0, nullptr, &launch_ev)); + got[k].resize(count); + RT_CHECK(vx_enqueue_read(queue, got[k].data(), res_buf, 0, count * sizeof(tie_result_t), + 1, &launch_ev, &read_ev)); + RT_CHECK(vx_event_wait_value(read_ev, 1, VX_TIMEOUT_INFINITE)); + vx_event_release(read_ev); + vx_event_release(launch_ev); + } + + uint32_t hits = 0, differ = 0; + for (uint32_t i = 0; i < count; ++i) { + hits += (got[0][i].status == VX_RT_STS_DONE_HIT); + differ += (std::memcmp(&got[0][i], &got[1][i], sizeof(tie_result_t)) != 0); + } + printf("rays=%u hits=%u, table vs static-key verdicts differ on %u rays\n", count, hits, differ); + + if (golden_out) { + FILE* f = fopen(golden_out, "w"); + if (!f) { printf("Error: cannot write %s\n", golden_out); cleanup(); return 1; } + fprintf(f, "// Generated by rt_smoke_tie -g on the SimX reference. Do not edit.\n"); + fprintf(f, "#pragma once\n#include \n"); + fprintf(f, "static const uint32_t kGoldenCount = %u;\n", count); + fprintf(f, "static const uint32_t kGolden[2][%u][8] = {\n", count); + for (int k = 0; k < 2; ++k) { + fprintf(f, " {\n"); + for (uint32_t i = 0; i < count; ++i) { + const uint32_t* w = reinterpret_cast(&got[k][i]); + fprintf(f, " { 0x%x, 0x%08x, 0x%08x, 0x%08x, %u, 0x%08x, %u, 0x%x },\n", + w[0], w[1], w[2], w[3], w[4], w[5], w[6], w[7]); + } + fprintf(f, " },\n"); + } + fprintf(f, "};\n"); + fclose(f); + printf("wrote %s\n", golden_out); + cleanup(); + return 0; + } + + int errors = 0; + if (count != kGoldenCount) { + printf("Error: %u rays, golden has %u\n", count, kGoldenCount); + ++errors; + } else { + for (int k = 0; k < 2; ++k) { + for (uint32_t i = 0; i < count; ++i) { + const uint32_t* w = reinterpret_cast(&got[k][i]); + // a miss carries no hit attributes + const int nf = (kGolden[k][i][0] == VX_RT_STS_DONE_HIT) ? 8 : 1; + for (int f = 0; f < nf; ++f) { + if (w[f] != kGolden[k][i][f]) { + if (errors < 20) + printf("%s ray %u: %s got 0x%08x expected 0x%08x\n", + k ? "static-key" : "tables", i, field_name[f], w[f], kGolden[k][i][f]); + ++errors; + } + } + } + } + } + if (differ == 0) { + printf("Error: the visit-order tables changed no verdict\n"); + ++errors; + } + + cleanup(); + if (errors != 0) { + printf("FAILED with %d errors\n", errors); + return 1; + } + printf("PASSED!\n"); + return 0; +} From 9c9c6c3e779f0de65d96cf375f88ac417bbef82a Mon Sep 17 00:00:00 2001 From: Blaise Tine Date: Sat, 3 Oct 2026 08:29:11 -0700 Subject: [PATCH 15/31] rtu: pipeline the near-tie window bound and the oracle for 200 MHz The V80 build of a8861312d missed 5 ns by 5.6 ns: near_t(t) was formed combinationally from the tri PE result into the result RAM (53 logic levels), and the oracle had -0.5..-2 ns paths through its register file reads. - VX_rtu_near_t / VX_rtu_f32_round: pipelined (6 / 4 stages; leading one, alignment, round increment and result each registered). Still exact for all 2^32 inputs. - Scheduler: the tri result event (result RAM write and context wake) is delayed through the near_t pipeline, so the bound lands with the result and no PE output reaches a RAM through more than a few levels. The tri PE's SimX cost model gains the same 6 cycles (rtu_isect.cpp). - Oracle: F32 unit operands registered off the LUTRAM reads; the box test's min/max fold split over registered reads; reciprocal operands registered before decode and the rounding pipelined; fetched scalar words captured into a register before they feed address arithmetic. - Oracle: the ray it set up last (world or object ray and reciprocals) is kept, keyed by world ray, TLAS table and instance, so a run of verdicts on one ray skips the setup; it is forgotten whenever the RTU is idle, as it always is between dependent launches. - rt_smoke_tie: thinner coincident stack (12 surfaces instead of 20), so its densest warp trace stays well inside the core's stall watchdog on the one-oracle RTL; the tables now change 117 of 256 verdicts (was 51). Out-of-context Vivado (VX_rtu_scheduler, NUM_CTX=16, V80 part, 5 ns): post-place WNS +0.148 ns, the worst paths all inside the tri PE. Co-Authored-By: Claude Opus 5.5 --- hw/rtl/rtu/VX_rtu_f32_round.sv | 107 ++++-- hw/rtl/rtu/VX_rtu_near_t.sv | 95 ++++- hw/rtl/rtu/VX_rtu_oracle.sv | 288 ++++++++++---- hw/rtl/rtu/VX_rtu_scheduler.sv | 38 +- sim/simx/rtu/rtu_isect.cpp | 5 +- tests/raytracing/rt_smoke_tie/golden.h | 494 ++++++++++++------------- tests/raytracing/rt_smoke_tie/main.cpp | 17 +- 7 files changed, 654 insertions(+), 390 deletions(-) diff --git a/hw/rtl/rtu/VX_rtu_f32_round.sv b/hw/rtl/rtu/VX_rtu_f32_round.sv index d5004da891..7f5ddae7c3 100644 --- a/hw/rtl/rtu/VX_rtu_f32_round.sv +++ b/hw/rtl/rtu/VX_rtu_f32_round.sv @@ -12,26 +12,33 @@ // limitations under the License. // VX_rtu_f32_round — rounds the positive value (mag + sticky) * 2^exp to F32, -// nearest even, subnormals and overflow to infinity included (combinational). -// `sticky` stands for a nonzero remainder below mag's LSB; callers keep at -// least two bits of mag below the result's LSB whenever it is set. Returns the -// magnitude bits {exponent, fraction}; mag == 0 gives +0. +// nearest even, subnormals and overflow to infinity included. `sticky` stands +// for a nonzero remainder below mag's LSB; callers keep at least two bits of +// mag below the result's LSB whenever it is set. Returns the magnitude bits +// {exponent, fraction}; mag == 0 gives +0. LATENCY 0 is combinational; 4 +// registers the leading one, the alignment, the round increment and the +// result. `include "VX_define.vh" module VX_rtu_f32_round #( - parameter WB = 44, // mag width - parameter EW = 11 // signed exponent width + parameter WB = 44, // mag width + parameter EW = 11, // signed exponent width + parameter LATENCY = 0 // 0 or 4 ) ( + input wire clk, + input wire enable, input wire [WB-1:0] mag, input wire signed [EW-1:0] exp, input wire sticky, output wire [30:0] result ); + `STATIC_ASSERT(((LATENCY == 0) || (LATENCY == 4)), ("invalid LATENCY")) localparam IW = `CLOG2(WB); localparam BW = WB + 12; + localparam EX = EW + 2; - // index of mag's leading one + // ── stage 0: leading one, the LSB exponent and the shift ────────── reg [IW-1:0] msb; always @(*) begin msb = '0; @@ -41,26 +48,80 @@ module VX_rtu_f32_round #( end // the result LSB's exponent, clamped at the subnormal LSB 2^-149 - wire signed [EW+1:0] lsb_raw = (EW+2)'(exp) + (EW+2)'($signed({1'b0, msb})) - (EW+2)'(23); - wire signed [EW+1:0] lsb = (lsb_raw < -(EW+2)'(149)) ? -(EW+2)'(149) : lsb_raw; - wire signed [EW+1:0] sh = lsb - (EW+2)'(exp); + wire signed [EX-1:0] lsb_raw = EX'(exp) + EX'($signed({1'b0, msb})) - EX'(23); + wire signed [EX-1:0] s0_lsb = (lsb_raw < -EX'(149)) ? -EX'(149) : lsb_raw; + wire signed [EX-1:0] s0_sh = s0_lsb - EX'(exp); + + // ── stage 1: alignment, guard/sticky ────────────────────────────── + wire [WB-1:0] s1_mag; + wire s1_sticky; + wire signed [EX-1:0] lsb, sh; + VX_pipe_register #( + .DATAW (WB + 1 + 2 * EX), + .DEPTH ((LATENCY != 0) ? 1 : 0) + ) pipe0 ( + .clk (clk), + .reset (1'b0), + .enable (enable), + .data_in ({mag, sticky, s0_lsb, s0_sh}), + .data_out ({s1_mag, s1_sticky, lsb, sh}) + ); // sh <= 0: exact, shifted up; sh > 0: drop sh bits, guard + sticky wire exact = (sh <= 0); wire [IW-1:0] ush = exact ? IW'(-sh) : IW'(sh); - wire [IW-1:0] gi = exact ? '0 : IW'(sh - (EW+2)'(1)); - wire [WB-1:0] q = mag >> ush; - wire [WB-1:0] low = mag & ((WB'(1) << gi) - WB'(1)); - wire g = mag[gi]; - wire st = (low != '0) || sticky; - wire [BW-1:0] sig = exact ? (BW'(mag) << ush) - : (BW'(q) + BW'(g && (st || q[0]))); + wire [IW-1:0] gi = exact ? '0 : IW'(sh - EX'(1)); + wire [WB-1:0] q = s1_mag >> ush; + wire [WB-1:0] low = s1_mag & ((WB'(1) << gi) - WB'(1)); + wire [BW-1:0] s1_sig0 = exact ? (BW'(s1_mag) << ush) : BW'(q); + wire s1_inc = !exact && s1_mag[gi] && ((low != '0) || s1_sticky || q[0]); + wire [BW-1:0] s1_ebits = BW'(lsb + EX'(149)) << 23; + wire s1_zero = (s1_mag == '0); + + // ── stage 2: round increment ────────────────────────────────────── + wire [BW-1:0] s2_sig0, s2_ebits; + wire s2_inc, s2_zero; + VX_pipe_register #( + .DATAW (2 * BW + 2), + .DEPTH ((LATENCY != 0) ? 1 : 0) + ) pipe1 ( + .clk (clk), + .reset (1'b0), + .enable (enable), + .data_in ({s1_sig0, s1_ebits, s1_inc, s1_zero}), + .data_out ({s2_sig0, s2_ebits, s2_inc, s2_zero}) + ); + wire [BW-1:0] s2_sig = s2_sig0 + BW'(s2_inc); + + // ── stage 3: {biased exponent of the LSB position, significand}: a + // carry out of the significand bumps the exponent, a subnormal reaching + // 2^23 becomes normal ───────────────────────────────────────────── + wire [BW-1:0] s3_sig, s3_ebits; + wire s3_zero; + VX_pipe_register #( + .DATAW (2 * BW + 1), + .DEPTH ((LATENCY != 0) ? 1 : 0) + ) pipe2 ( + .clk (clk), + .reset (1'b0), + .enable (enable), + .data_in ({s2_sig, s2_ebits, s2_zero}), + .data_out ({s3_sig, s3_ebits, s3_zero}) + ); + wire [BW-1:0] s3_bits = s3_ebits + s3_sig; + wire [30:0] s3_res = s3_zero ? 31'd0 + : (s3_bits >= BW'(32'h7f800000)) ? 31'h7f800000 + : s3_bits[30:0]; - // {biased exponent of the LSB position, significand}: a carry out of the - // significand bumps the exponent, a subnormal reaching 2^23 becomes normal - wire [BW-1:0] bits = (BW'(lsb + (EW+2)'(149)) << 23) + sig; - assign result = (mag == '0) ? 31'd0 - : (bits >= BW'(32'h7f800000)) ? 31'h7f800000 - : bits[30:0]; + VX_pipe_register #( + .DATAW (31), + .DEPTH ((LATENCY != 0) ? 1 : 0) + ) pipe3 ( + .clk (clk), + .reset (1'b0), + .enable (enable), + .data_in (s3_res), + .data_out (result) + ); endmodule diff --git a/hw/rtl/rtu/VX_rtu_near_t.sv b/hw/rtl/rtu/VX_rtu_near_t.sv index b46a3efdb4..6225e78b23 100644 --- a/hw/rtl/rtu/VX_rtu_near_t.sv +++ b/hw/rtl/rtu/VX_rtu_near_t.sv @@ -12,60 +12,115 @@ // limitations under the License. // VX_rtu_near_t — the near-hit window bound t + |t| * 2^-19 in F32, both -// operations rounded to nearest even, subnormals included (combinational). -// Two opaque hits within it of each other may be ordered either way by the -// source BVH's F32 box cull, so the walker settles them by its visit order. +// operations rounded to nearest even, subnormals included. Two opaque hits +// within it of each other may be ordered either way by the source BVH's F32 +// box cull, so the walker settles them by its visit order. // // |t| * 2^-19 is exact unless it lands in the subnormal range, so the sum is // formed as one integer W * 2^p (the scaled term pre-rounded where it is not -// exact) and rounded once. +// exact) and rounded once. LATENCY 0 is combinational; 6 registers the scaled +// term, the sum and the four rounding steps. `include "VX_define.vh" -module VX_rtu_near_t ( +module VX_rtu_near_t #( + parameter LATENCY = 0 // 0 or 6 +) ( + input wire clk, + input wire enable, input wire [31:0] t, output wire [31:0] result ); + `STATIC_ASSERT(((LATENCY == 0) || (LATENCY == 6)), ("invalid LATENCY")) localparam WB = 44; + localparam PD = (LATENCY != 0) ? 1 : 0; + // ── stage 1: decode, the scaled term ────────────────────────────── wire s = t[31]; wire [7:0] e = t[30:23]; wire [22:0] f = t[22:0]; - wire is_nan = (e == 8'hff) && (f != 23'd0); - wire is_inf = (e == 8'hff) && (f == 23'd0); - wire is_zero = (e == 8'h00) && (f == 23'd0); + // the result when t is not a finite nonzero value + wire s1_spec = (e == 8'hff) || ((e == 8'h00) && (f == 23'd0)); + wire [31:0] s1_sval = (e != 8'hff) ? 32'h00000000 + : (f != 23'd0) ? (t | 32'h00400000) + : (s ? 32'hffc00000 : 32'h7f800000); // |t| = m * 2^q, q = e - 150 (normal) or -149 (subnormal) wire [23:0] m = {(e != 8'h00), f}; // |t| * 2^-19 = (m / 2^d) * 2^p: d > 0 only when it is subnormal, where // its integer m / 2^d is rounded as the F32 multiply rounds it wire [4:0] d = (e >= 8'd20) ? 5'd0 : ((e == 8'h00) ? 5'd19 : 5'(8'd20 - e)); - wire signed [10:0] p = (d == 5'd0) ? (11'($signed({1'b0, e})) - 11'sd169) : -11'sd149; + wire signed [10:0] s1_p = (d == 5'd0) ? (11'($signed({1'b0, e})) - 11'sd169) : -11'sd149; wire [23:0] q0 = m >> d; wire [4:0] gi = (d == 5'd0) ? 5'd0 : (d - 5'd1); wire g = (d != 5'd0) && m[gi]; wire st = (m & ((24'd1 << gi) - 24'd1)) != 24'd0; - wire [23:0] y_int = q0 + 24'(g && (st || q0[0])); + wire [23:0] s1_y = q0 + 24'(g && (st || q0[0])); + wire [WB-1:0] s1_mal = WB'(m) << (5'd19 - d); - wire [WB-1:0] m_al = WB'(m) << (5'd19 - d); - wire [WB-1:0] w = s ? (m_al - WB'(y_int)) : (m_al + WB'(y_int)); + // ── stage 2: the exact sum ──────────────────────────────────────── + wire s2_s, s2_spec; + wire [31:0] s2_sval; + wire signed [10:0] s2_p; + wire [23:0] s2_y; + wire [WB-1:0] s2_mal; + VX_pipe_register #( + .DATAW (1 + 1 + 32 + 11 + 24 + WB), + .DEPTH (PD) + ) pipe1 ( + .clk (clk), + .reset (1'b0), + .enable (enable), + .data_in ({s, s1_spec, s1_sval, s1_p, s1_y, s1_mal}), + .data_out ({s2_s, s2_spec, s2_sval, s2_p, s2_y, s2_mal}) + ); + wire [WB-1:0] s2_w = s2_s ? (s2_mal - WB'(s2_y)) : (s2_mal + WB'(s2_y)); + + wire s3_s, s3_spec; + wire [31:0] s3_sval; + wire signed [10:0] s3_p; + wire [WB-1:0] s3_w; + VX_pipe_register #( + .DATAW (1 + 1 + 32 + 11 + WB), + .DEPTH (PD) + ) pipe2 ( + .clk (clk), + .reset (1'b0), + .enable (enable), + .data_in ({s2_s, s2_spec, s2_sval, s2_p, s2_w}), + .data_out ({s3_s, s3_spec, s3_sval, s3_p, s3_w}) + ); + // ── stages 3..6: the rounding, with the special result alongside ── wire [30:0] mag; VX_rtu_f32_round #( - .WB (WB), - .EW (11) + .WB (WB), + .EW (11), + .LATENCY ((LATENCY != 0) ? 4 : 0) ) round ( - .mag (w), - .exp (p), + .clk (clk), + .enable (enable), + .mag (s3_w), + .exp (s3_p), .sticky (1'b0), .result (mag) ); - assign result = is_nan ? (t | 32'h00400000) - : is_inf ? (s ? 32'hffc00000 : 32'h7f800000) - : is_zero ? 32'h00000000 - : {s, mag}; + wire o_s, o_spec; + wire [31:0] o_sval; + VX_pipe_register #( + .DATAW (1 + 1 + 32), + .DEPTH ((LATENCY != 0) ? 4 : 0) + ) pipe_side ( + .clk (clk), + .reset (1'b0), + .enable (enable), + .data_in ({s3_s, s3_spec, s3_sval}), + .data_out ({o_s, o_spec, o_sval}) + ); + + assign result = o_spec ? o_sval : {o_s, mag}; endmodule diff --git a/hw/rtl/rtu/VX_rtu_oracle.sv b/hw/rtl/rtu/VX_rtu_oracle.sv index 552310f06f..cc2afa728b 100644 --- a/hw/rtl/rtu/VX_rtu_oracle.sv +++ b/hw/rtl/rtu/VX_rtu_oracle.sv @@ -53,6 +53,9 @@ module VX_rtu_oracle import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( ) ( input wire clk, input wire reset, + // the scene may change: forget the ray set up last (asserted while the + // RTU is idle, which it always is between dependent launches) + input wire flush, // request: hit a (the new one) against hit b (the committed one) input wire req_valid, @@ -119,7 +122,7 @@ module VX_rtu_oracle import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( // ── sequencer states ────────────────────────────────────────────── localparam [6:0] S_IDLE = 7'd0, S_CAP = 7'd1, S_BOX1 = 7'd2, S_BOX2 = 7'd3, - S_DISP = 7'd5, S_DONE = 7'd6, + S_BOX3 = 7'd4, S_DISP = 7'd5, S_DONE = 7'd6, // same instance T_S1 = 7'd8, T_S2 = 7'd9, T_S3 = 7'd10, T_S4 = 7'd11, T_S5 = 7'd12, T_S6 = 7'd13, T_S7 = 7'd14, @@ -135,16 +138,18 @@ module VX_rtu_oracle import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( // one climb step U_0 = 7'd44, U_1 = 7'd45, U_2 = 7'd46, U_3 = 7'd47, // box test - B_SUB = 7'd48, B_MUL = 7'd49, B_RED = 7'd50, B_FIN = 7'd51, + B_SUB = 7'd48, B_MUL = 7'd49, B_RED = 7'd50, B_FIN0 = 7'd51, + B_FIN = 7'd52, // object ray O_0 = 7'd56, O_1 = 7'd57, O_2 = 7'd58, O_3 = 7'd59, O_4 = 7'd60, O_5 = 7'd61, O_6 = 7'd62, O_7 = 7'd63, O_8 = 7'd64, // reciprocals - R_LD0 = 7'd68, R_LD1 = 7'd69, R_IT = 7'd70, R_WR = 7'd71, - // fetch wait / move / copy / multiply / drain - F_WAIT = 7'd72, M_MV = 7'd76, C_CP = 7'd77, X_MUL = 7'd78, - W_DRAIN = 7'd79; + R_LD0 = 7'd68, R_LD1 = 7'd69, R_LD2 = 7'd67, R_IT = 7'd70, + R_WR = 7'd71, + // fetch wait / fetched-word capture / move / copy / multiply / drain + F_WAIT = 7'd72, F_CAP = 7'd73, M_MV = 7'd76, C_CP = 7'd77, + X_MUL = 7'd78, W_DRAIN = 7'd79; // ── F32 helpers (C fmin/fmax, IEEE ordered compares) ────────────── function automatic logic f_nan(input logic [30:0] a); @@ -197,6 +202,15 @@ module VX_rtu_oracle import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( reg [31:0] upt; // ... and the t its boxes are culled against reg a_first, keep, ph; reg [31:0] key0; + + // The current ray (RO/RD/RI) as last set up: the world ray, or a TLAS + // leaf's object ray. Verdicts for one walk come in runs on the same ray, + // so a request whose ray is the one already set up skips the setup. + reg [2:0][31:0] cw_o, cw_d; // this request's world ray + reg rk_valid, rk_world; + reg [31:0] rk_inst, rk_tlas; + reg [2:0][31:0] rk_o, rk_d; + reg ray_same; // rk was set up for this request's world ray reg [CLIMBW-1:0] climbs; // box test @@ -207,6 +221,8 @@ module VX_rtu_oracle import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( reg bt_pass; reg [2:0] bt_wbc; // slab differences written back so far reg [4:0] k; // the running routine counter + reg [31:0] red_a, red_b, red_mn, red_mx, bt_lo0; + reg [31:0] tmn, tmx; // vertex box min/max partials // object ray / multiply reg [31:0] o_inst; @@ -268,6 +284,23 @@ module VX_rtu_oracle import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( reg [5:0] fma_dst; reg [5:0] fma_pend; + // operands are registered off the register file's asynchronous read + reg fx_issue; + reg [1:0] fx_kind; + reg [5:0] fx_dst; + reg [31:0] fx_a, fx_b; + always_ff @(posedge clk) begin + if (reset) begin + fx_issue <= 1'b0; + end else begin + fx_issue <= fma_issue; + end + fx_kind <= fma_kind; + fx_dst <= fma_dst; + fx_a <= rda; + fx_b <= rdb; + end + VX_fma_unit #( .LATENCY (FMA_LAT), .USE_DSP (`VX_CFG_RTU_USE_DSP), @@ -277,12 +310,12 @@ module VX_rtu_oracle import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( .clk (clk), .reset (reset), .enable (1'b1), - .mask (fma_issue), - .op_type ((fma_kind == 2'd2) ? INST_FPU_MUL : INST_FPU_ADD), - .fmt ((fma_kind == 2'd1) ? INST_FMT_BITS'(2'b10) : INST_FMT_BITS'(2'b00)), + .mask (fx_issue), + .op_type ((fx_kind == 2'd2) ? INST_FPU_MUL : INST_FPU_ADD), + .fmt ((fx_kind == 2'd1) ? INST_FMT_BITS'(2'b10) : INST_FMT_BITS'(2'b00)), .frm (INST_FRM_RNE), - .dataa (rda), - .datab (rdb), + .dataa (fx_a), + .datab (fx_b), .datac (32'd0), .result (fma_res), `UNUSED_PIN (fflags) @@ -296,7 +329,7 @@ module VX_rtu_oracle import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( .clk (clk), .reset (reset), .enable (1'b1), - .data_in ({fma_issue, fma_dst}), + .data_in ({fx_issue, fx_dst}), .data_out ({wb_v, wb_dst}) ); @@ -354,6 +387,8 @@ module VX_rtu_oracle import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( // fetched words: one read port over the two lines reg [4:0] fw_k; + reg [4:0] f_k0; // the word F_CAP captures into fw_q + reg [31:0] fw_q; wire [2*LINE_BITS-1:0] lbs = {lb1, lb0}; wire [WIDXW-1:0] fw_idx = WIDXW'(f_w0) + WIDXW'(fw_k); wire [31:0] fw = lbs[32'(fw_idx) * 32 +: 32]; @@ -396,18 +431,33 @@ module VX_rtu_oracle import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( end rc_decode = {spec, sval, x[31], m, ex}; endfunction - wire [68:0] rc_dec_a = rc_decode(rda); - wire [68:0] rc_dec_b = rc_decode(rdb); + reg [2:0][31:0] rx; + wire [68:0] rc_dec0 = rc_decode(rx[0]); + wire [68:0] rc_dec1 = rc_decode(rx[1]); + wire [68:0] rc_dec2 = rc_decode(rx[2]); - wire [1:0] rc_w = 2'(k); + // R_WR registers lane k at k < 3 and writes lane k - 5 five cycles later + wire [1:0] rc_w = (k < 5'd3) ? 2'(k) : 2'd0; + wire [1:0] rc_o = (k >= 5'd5) ? 2'(k - 5'd5) : 2'd0; + reg [27:0] rc_in_q; + reg [10:0] rc_in_e; + reg rc_in_st; + always_ff @(posedge clk) begin + rc_in_q <= dv_q[rc_w]; + rc_in_e <= 11'(-11'sd50 - $signed(dv_e[rc_w])); + rc_in_st <= (dv_rem[rc_w] != 25'd0); + end wire [30:0] rc_mag; VX_rtu_f32_round #( - .WB (28), - .EW (11) + .WB (28), + .EW (11), + .LATENCY (4) ) rc_round ( - .mag (dv_q[rc_w]), - .exp (-11'sd50 - $signed(dv_e[rc_w])), - .sticky (dv_rem[rc_w] != 25'd0), + .clk (clk), + .enable (1'b1), + .mag (rc_in_q), + .exp ($signed(rc_in_e)), + .sticky (rc_in_st), .result (rc_mag) ); @@ -441,15 +491,15 @@ module VX_rtu_oracle import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( fsm_wa = RA_WO + 6'(k); fsm_wd = cap[0]; end - S_BOX1: begin + S_BOX2: begin fsm_we = 1'b1; fsm_wa = ((k >= 5'd3) ? RA_BB : RA_BA) + 6'(bx_a); - fsm_wd = f_min(bx_v0, f_min(bx_v1, bx_v2)); + fsm_wd = f_min(bx_v0, tmn); end - S_BOX2: begin + S_BOX3: begin fsm_we = 1'b1; fsm_wa = ((k >= 5'd3) ? RA_BB : RA_BA) + 6'd3 + 6'(bx_a); - fsm_wd = f_max(bx_v0, f_max(bx_v1, bx_v2)); + fsm_wd = f_max(bx_v0, tmx); end B_SUB: begin ra = bt_base + 6'(k); @@ -467,8 +517,8 @@ module VX_rtu_oracle import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( fma_dst = RA_S + 6'(k); end B_RED: begin - ra = RA_S + 6'(k); - rb = RA_S + 6'd3 + 6'(k); + ra = RA_S + ((k < 5'd3) ? 6'(k) : 6'd0); + rb = RA_S + 6'd3 + ((k < 5'd3) ? 6'(k) : 6'd0); end O_4: begin // products: X[i*3+j] = wo[j] * m[i][j], X[9+i*3+j] = wd[j] * m[i][j] @@ -519,9 +569,9 @@ module VX_rtu_oracle import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( ra = RA_RD + 6'd2; end R_WR: begin - fsm_we = 1'b1; - fsm_wa = RA_RI + 6'(k); - fsm_wd = rc_spec[rc_w] ? rc_sval[rc_w] : {dv_s[rc_w], rc_mag}; + fsm_we = (k >= 5'd5); + fsm_wa = RA_RI + 6'(rc_o); + fsm_wd = rc_spec[rc_o] ? rc_sval[rc_o] : {dv_s[rc_o], rc_mag}; end M_MV: begin fw_k = mv_k0 + k; @@ -535,9 +585,7 @@ module VX_rtu_oracle import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( fsm_wa = mv_dst + 6'(k); fsm_wd = rda; end - T_D3B: fw_k = 5'd1; - U_2: fw_k = is_tlas ? 5'd12 : 5'd6; - T_D6, T_D10: fw_k = 5'd13; + F_CAP: fw_k = f_k0; default:; endcase end @@ -551,8 +599,12 @@ module VX_rtu_oracle import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( if (reset) begin state <= S_IDLE; fma_pend <= '0; + rk_valid <= 1'b0; end else begin fma_pend <= fma_pend + 6'(fma_issue) - 6'(wb_v); + if (flush) begin + rk_valid <= 1'b0; + end if ((state == B_SUB) || (state == B_MUL)) begin bt_wbc <= bt_wbc + 3'(wb_v); end else begin @@ -570,6 +622,8 @@ module VX_rtu_oracle import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( psa <= req_a_ps; psb <= req_b_ps; taba <= req_a_tab; tabb <= req_b_tab; cap <= {req_b_v, req_a_v, req_wd, req_wo}; + cw_o <= req_wo; + cw_d <= req_wd; k <= '0; climbs <= '0; state <= S_CAP; @@ -585,9 +639,14 @@ module VX_rtu_oracle import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( end end S_BOX1: begin + tmn <= f_min(bx_v1, bx_v2); + tmx <= f_max(bx_v1, bx_v2); state <= S_BOX2; end S_BOX2: begin + state <= S_BOX3; + end + S_BOX3: begin k <= k + 5'd1; state <= S_BOX1; if (k == 5'd5) begin @@ -596,7 +655,8 @@ module VX_rtu_oracle import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( end end S_DISP: begin - state <= (ia == ib) ? T_S1 : T_D0; + ray_same <= rk_valid && (rk_o == cw_o) && (rk_d == cw_d) && (rk_tlas == tlas_r); + state <= (ia == ib) ? T_S1 : T_D0; end // ── same instance: climb its BLAS table ────────────────── @@ -616,27 +676,30 @@ module VX_rtu_oracle import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( f_off <= taba + 32'd8; f_n <= 5'd1; fs_start <= 1'b1; + f_k0 <= '0; f_ret <= T_S3; state <= F_WAIT; end T_S3: begin - noff_r <= fw; + noff_r <= fw_q; ps0 <= psa; - f_off <= node_off(tab_r, fw, 1'b0, psa) + 32'd28; + f_off <= node_off(tab_r, fw_q, 1'b0, psa) + 32'd28; fs_start <= 1'b1; + f_k0 <= '0; f_ret <= T_S4; state <= F_WAIT; end T_S4: begin - dep0 <= fw + 32'd1; + dep0 <= fw_q + 32'd1; ps1 <= psb; f_off <= node_off(tab_r, noff_r, 1'b0, psb) + 32'd28; fs_start <= 1'b1; + f_k0 <= '0; f_ret <= T_S5; state <= F_WAIT; end T_S5: begin - dep1 <= fw + 32'd1; + dep1 <= fw_q + 32'd1; cul0 <= 1'b0; cul1 <= 1'b0; mv_dst <= RA_CB0; @@ -658,11 +721,22 @@ module VX_rtu_oracle import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( // ── different instances: climb the TLAS table ──────────── T_D0: begin - mv_dst <= RA_RO; // world ray o, d -> current ray - cp_src <= RA_WO; - mv_n <= 5'd6; - c_ret <= T_D1; - state <= C_CP; + if (ray_same && rk_world) begin + // the world ray is set up already + is_tlas <= 1'b1; + tab_r <= tlas_r; + f_off <= tlas_r + 32'd8; + f_n <= 5'd2; + fs_start <= 1'b1; + f_k0 <= '0; + state <= T_D2; + end else begin + mv_dst <= RA_RO; // world ray o, d -> current ray + cp_src <= RA_WO; + mv_n <= 5'd6; + c_ret <= T_D1; + state <= C_CP; + end end T_D1: begin // the TLAS header fetch runs beside the reciprocals @@ -671,6 +745,13 @@ module VX_rtu_oracle import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( f_off <= tlas_r + 32'd8; f_n <= 5'd2; fs_start <= 1'b1; + f_k0 <= '0; + rk_valid <= 1'b1; + rk_world <= 1'b1; + rk_o <= cw_o; + rk_d <= cw_d; + rk_tlas <= tlas_r; + ray_same <= 1'b1; r_ret <= T_D2; state <= R_LD0; end @@ -679,13 +760,15 @@ module VX_rtu_oracle import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( state <= F_WAIT; end T_D3: begin - noff_r <= fw; - state <= T_D3B; + noff_r <= fw_q; + f_k0 <= 5'd1; + f_ret <= T_D3B; + state <= F_CAP; end T_D3B: begin - stride_r <= fw; + stride_r <= fw_q; mul_a <= ia; - mul_b <= fw; + mul_b <= fw_q; mul_p <= '0; x_ret <= T_D4; state <= X_MUL; @@ -694,24 +777,26 @@ module VX_rtu_oracle import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( f_off <= tlas_r + 32'd16 + mul_p; f_n <= 5'd1; fs_start <= 1'b1; + f_k0 <= '0; f_ret <= T_D5; state <= F_WAIT; end T_D5: begin - ps0 <= fw; - if (fw == ROOT) begin + ps0 <= fw_q; + if (fw_q == ROOT) begin dep0 <= '0; state <= T_D7; end else begin - f_off <= node_off(tab_r, noff_r, 1'b1, fw); + f_off <= node_off(tab_r, noff_r, 1'b1, fw_q); f_n <= 5'd14; fs_start <= 1'b1; + f_k0 <= 5'd13; f_ret <= T_D6; state <= F_WAIT; end end T_D6: begin - dep0 <= fw + 32'd1; + dep0 <= fw_q + 32'd1; mv_dst <= RA_CB0; mv_k0 <= ps0[0] ? 5'd6 : 5'd0; mv_n <= 5'd6; @@ -729,24 +814,26 @@ module VX_rtu_oracle import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( f_off <= tlas_r + 32'd16 + mul_p; f_n <= 5'd1; fs_start <= 1'b1; + f_k0 <= '0; f_ret <= T_D9; state <= F_WAIT; end T_D9: begin - ps1 <= fw; - if (fw == ROOT) begin + ps1 <= fw_q; + if (fw_q == ROOT) begin dep1 <= '0; state <= T_D11; end else begin - f_off <= node_off(tab_r, noff_r, 1'b1, fw); + f_off <= node_off(tab_r, noff_r, 1'b1, fw_q); f_n <= 5'd14; fs_start <= 1'b1; + f_k0 <= 5'd13; f_ret <= T_D10; state <= F_WAIT; end end T_D10: begin - dep1 <= fw + 32'd1; + dep1 <= fw_q + 32'd1; mv_dst <= RA_CB1; mv_k0 <= ps1[0] ? 5'd6 : 5'd0; mv_n <= 5'd6; @@ -796,11 +883,12 @@ module VX_rtu_oracle import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( f_off <= (ph ? tabb : taba) + 32'd8; f_n <= 5'd1; fs_start <= 1'b1; + f_k0 <= '0; f_ret <= P_2; state <= F_WAIT; end P_2: begin - noff_r <= fw; + noff_r <= fw_q; ps0 <= ph ? psb : psa; cul0 <= 1'b0; cur <= 1'b0; @@ -867,6 +955,7 @@ module VX_rtu_oracle import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( f_off <= node_off(tab_r, noff_r, is_tlas, cur ? ps1 : ps0); f_n <= is_tlas ? 5'd14 : 5'd8; fs_start <= (climbs != CLIMBW'(MAX_CLIMBS)); + f_k0 <= is_tlas ? 5'd12 : 5'd6; b_ret <= U_1; state <= (climbs == CLIMBW'(MAX_CLIMBS)) ? S_DONE : B_SUB; end @@ -879,10 +968,10 @@ module VX_rtu_oracle import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( end U_2: begin if (cur) begin - ps1 <= fw; + ps1 <= fw_q; dep1 <= dep1 - 32'd1; end else begin - ps0 <= fw; + ps0 <= fw_q; dep0 <= dep0 - 32'd1; end mv_dst <= cur ? RA_CB1 : RA_CB0; @@ -892,13 +981,14 @@ module VX_rtu_oracle import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( // a BLAS node holds its own box, read with its parent link mv_k0 <= 5'd0; state <= M_MV; - end else if (fw == ROOT) begin + end else if (fw_q == ROOT) begin state <= u_ret; end else begin - mv_k0 <= fw[0] ? 5'd6 : 5'd0; - f_off <= node_off(tab_r, noff_r, 1'b1, fw); + mv_k0 <= fw_q[0] ? 5'd6 : 5'd0; + f_off <= node_off(tab_r, noff_r, 1'b1, fw_q); f_n <= 5'd12; fs_start <= 1'b1; + f_k0 <= '0; f_ret <= U_3; state <= F_WAIT; end @@ -929,20 +1019,32 @@ module VX_rtu_oracle import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( end end B_RED: begin - bt_lo <= (k == 5'd0) ? f_min(rda, rdb) : f_max(bt_lo, f_min(rda, rdb)); - bt_hi <= (k == 5'd0) ? f_max(rda, rdb) : f_min(bt_hi, f_max(rda, rdb)); - k <= k + 5'd1; + // read axis k, min/max it at k + 1, fold it at k + 2 + red_a <= rda; + red_b <= rdb; + red_mn <= f_min(red_a, red_b); + red_mx <= f_max(red_a, red_b); if (k == 5'd2) begin + bt_lo <= red_mn; + bt_hi <= red_mx; + end else if (k > 5'd2) begin + bt_lo <= f_max(bt_lo, red_mn); + bt_hi <= f_min(bt_hi, red_mx); + end + k <= k + 5'd1; + if (k == 5'd4) begin k <= '0; - state <= B_FIN; + state <= B_FIN0; end end + B_FIN0: begin + bt_lo0 <= f_max(32'd0, bt_lo); + state <= B_FIN; + end B_FIN: begin // hi >= fmax(0, lo); an empty (NaN) box is never entered logic hit; - logic [31:0] lo0; - lo0 = f_max(32'd0, bt_lo); - hit = !bt_nan && (f_lt(lo0, bt_hi) || f_eq(lo0, bt_hi)); + hit = !bt_nan && (f_lt(bt_lo0, bt_hi) || f_eq(bt_lo0, bt_hi)); bt_key <= hit ? bt_lo : F_INF; bt_pass <= hit && f_lt(bt_lo, bt_tmax); state <= b_ret; @@ -950,16 +1052,22 @@ module VX_rtu_oracle import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( // ── the reference's object-space ray for TLAS leaf o_inst ─ O_0: begin - f_off <= tlas_r + 32'd12; - f_n <= 5'd1; - fs_start <= 1'b1; - f_ret <= O_1; - state <= F_WAIT; + if (ray_same && !rk_world && (rk_inst == o_inst)) begin + state <= o_ret; // this object ray is set up already + end else begin + f_off <= tlas_r + 32'd12; + f_n <= 5'd1; + fs_start <= 1'b1; + f_k0 <= '0; + rk_valid <= 1'b0; + f_ret <= O_1; + state <= F_WAIT; + end end O_1: begin - stride_r <= fw; + stride_r <= fw_q; mul_a <= o_inst; - mul_b <= fw; + mul_b <= fw_q; mul_p <= '0; x_ret <= O_2; state <= X_MUL; @@ -968,6 +1076,7 @@ module VX_rtu_oracle import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( f_off <= tlas_r + 32'd32 + mul_p; f_n <= 5'd12; fs_start <= 1'b1; + f_k0 <= '0; f_ret <= O_3; state <= F_WAIT; end @@ -1011,18 +1120,31 @@ module VX_rtu_oracle import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( end end O_8: begin - r_ret <= o_ret; - state <= R_LD0; + rk_valid <= 1'b1; + rk_world <= 1'b0; + rk_inst <= o_inst; + rk_o <= cw_o; + rk_d <= cw_d; + rk_tlas <= tlas_r; + ray_same <= 1'b1; + r_ret <= o_ret; + state <= R_LD0; end // ── 1/d per axis, correctly rounded ────────────────────── R_LD0: begin - {rc_spec[0], rc_sval[0], dv_s[0], dv_m[0], dv_e[0]} <= rc_dec_a; - {rc_spec[1], rc_sval[1], dv_s[1], dv_m[1], dv_e[1]} <= rc_dec_b; + rx[0] <= rda; + rx[1] <= rdb; state <= R_LD1; end R_LD1: begin - {rc_spec[2], rc_sval[2], dv_s[2], dv_m[2], dv_e[2]} <= rc_dec_a; + rx[2] <= rda; + state <= R_LD2; + end + R_LD2: begin + {rc_spec[0], rc_sval[0], dv_s[0], dv_m[0], dv_e[0]} <= rc_dec0; + {rc_spec[1], rc_sval[1], dv_s[1], dv_m[1], dv_e[1]} <= rc_dec1; + {rc_spec[2], rc_sval[2], dv_s[2], dv_m[2], dv_e[2]} <= rc_dec2; for (integer i = 0; i < 3; ++i) begin dv_rem[i] <= 25'h400000; // 2^22: the dividend's leading bits dv_q[i] <= '0; @@ -1048,7 +1170,7 @@ module VX_rtu_oracle import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( end R_WR: begin k <= k + 5'd1; - if (k == 5'd2) begin + if (k == 5'd7) begin k <= '0; state <= r_ret; end @@ -1057,9 +1179,13 @@ module VX_rtu_oracle import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( // ── leaf routines ──────────────────────────────────────── F_WAIT: begin if (fs_done) begin - state <= f_ret; + state <= F_CAP; end end + F_CAP: begin + fw_q <= fw; + state <= f_ret; + end M_MV, C_CP: begin k <= k + 5'd1; if ((k + 5'd1) == mv_n) begin diff --git a/hw/rtl/rtu/VX_rtu_scheduler.sv b/hw/rtl/rtu/VX_rtu_scheduler.sv index 2dba1c5815..cd4a070294 100644 --- a/hw/rtl/rtu/VX_rtu_scheduler.sv +++ b/hw/rtl/rtu/VX_rtu_scheduler.sv @@ -423,12 +423,33 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( wire tri_valid_out, tri_hit, tri_back; wire [CTX_TAG_W-1:0] tri_tag_out; wire [31:0] tri_t, tri_u, tri_v; - // the near window's bound rides with the result, off the EXEC path + // the near window's bound rides with the result: it is pipelined after + // the tri PE, and the result event (RAM write and wake) is delayed to + // match, so neither lands on the EXEC path + localparam NEAR_LAT = 6; wire [31:0] tri_tn; - VX_rtu_near_t tri_near ( + VX_rtu_near_t #( + .LATENCY (NEAR_LAT) + ) tri_near ( + .clk (clk), + .enable (1'b1), .t (tri_t), .result (tri_tn) ); + wire trd_valid, trd_hit, trd_back; + wire [CTX_TAG_W-1:0] trd_tag; + wire [31:0] trd_t, trd_u, trd_v; + VX_shift_register #( + .DATAW (1 + CTX_TAG_W + 2 + 3 * 32), + .RESETW (1), + .DEPTH (NEAR_LAT) + ) tri_delay ( + .clk (clk), + .reset (reset), + .enable (1'b1), + .data_in ({tri_valid_out, tri_tag_out, tri_hit, tri_back, tri_t, tri_u, tri_v}), + .data_out ({trd_valid, trd_tag, trd_hit, trd_back, trd_t, trd_u, trd_v}) + ); wire [129:0] trires_rdata; VX_dp_ram #( .DATAW (130), @@ -439,10 +460,10 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( .clk (clk), .reset (reset), .read (g1_valid), - .write (tri_valid_out), + .write (trd_valid), .wren (1'b1), - .waddr (tri_tag_out), - .wdata ({tri_hit, tri_back, tri_t, tri_u, tri_v, tri_tn}), + .waddr (trd_tag), + .wdata ({trd_hit, trd_back, trd_t, trd_u, trd_v, tri_tn}), .raddr (g1_idx), .rdata (trires_rdata) ); @@ -904,6 +925,7 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( ) oracle ( .clk (clk), .reset (reset), + .flush (running == '0), .req_valid (orc_start), .req_ready (orc_req_ready), .req_ctx (sel_q), @@ -2026,7 +2048,7 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( (ray_wr_valid ? NUM_CTX'(1) << ray_wr_ctx : NUM_CTX'(0)) | ((mem_rsp_valid && !orc_q[mem_rsp_tag]) ? NUM_CTX'(1) << mem_rsp_tag : NUM_CTX'(0)) | (orc_done_valid ? NUM_CTX'(1) << orc_done_ctx : NUM_CTX'(0)) - | (tri_valid_out ? NUM_CTX'(1) << tri_tag_out : NUM_CTX'(0)) + | (trd_valid ? NUM_CTX'(1) << trd_tag : NUM_CTX'(0)) | (xform_valid_out ? NUM_CTX'(1) << xform_tag_out : NUM_CTX'(0)) | ((recip_valid_out && recip_last_out) ? NUM_CTX'(1) << recip_tag_out : NUM_CTX'(0)) | (box_wake_r ? NUM_CTX'(1) << box_wake_ctx_r : NUM_CTX'(0)) @@ -2382,9 +2404,9 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( `TRACE(2, ("%t: %s rtu-node: ctx=%0d, addr=0x%0h, kind=%0d\n", $time, INSTANCE_ID, sel_q, structaddr_q, node_kind)) end - if (tri_valid_out) begin + if (trd_valid) begin `TRACE(2, ("%t: %s rtu-tri: ctx=%0d, hit=%0d, t=0x%0h\n", - $time, INSTANCE_ID, tri_tag_out, tri_hit, tri_t)) + $time, INSTANCE_ID, trd_tag, trd_hit, trd_t)) end if (| done_r) begin `TRACE(1, ("%t: %s rtu-done: slots=%b\n", $time, INSTANCE_ID, done_r)) diff --git a/sim/simx/rtu/rtu_isect.cpp b/sim/simx/rtu/rtu_isect.cpp index dc3309a297..9716bacc0d 100644 --- a/sim/simx/rtu/rtu_isect.cpp +++ b/sim/simx/rtu/rtu_isect.cpp @@ -172,9 +172,10 @@ uint32_t BoxPe::pipe_depth() { uint32_t TriPe::pipe_depth() { // input select + 1/dir + 3 F32 stages + 5 F64 stages + F64 divide + narrow - // + verdict (VX_rtu_tri_pe). + // + verdict (VX_rtu_tri_pe), then the 6 stages that form the result's near- + // tie window bound before the scheduler sees it (VX_rtu_near_t). return 3 + kRtuFdivLat + 3 * kRtuLatencyFma + 5 * kRtuLatencyFma64 - + kRtuFdiv64Lat; + + kRtuFdiv64Lat + 6; } }} // namespace vortex::rtu diff --git a/tests/raytracing/rt_smoke_tie/golden.h b/tests/raytracing/rt_smoke_tie/golden.h index 60fdac6ed0..3ce413f215 100644 --- a/tests/raytracing/rt_smoke_tie/golden.h +++ b/tests/raytracing/rt_smoke_tie/golden.h @@ -4,30 +4,30 @@ static const uint32_t kGoldenCount = 256; static const uint32_t kGolden[2][256][8] = { { - { 0x0, 0x3ffffffc, 0x80000000, 0x3e800000, 96, 0x80000003, 0, 0x100 }, - { 0x0, 0x403ffffb, 0x80000000, 0x3e800000, 67, 0x00000002, 5, 0x105 }, - { 0x0, 0x3ffffffc, 0x3e800000, 0x3f000000, 113, 0x80000003, 2, 0x102 }, - { 0x0, 0x403ffffc, 0x80000000, 0x3e800000, 101, 0x00000003, 5, 0x105 }, - { 0x0, 0x3ffffffc, 0x3e800000, 0x3f000000, 105, 0x80000003, 2, 0x102 }, - { 0x0, 0x403ffffb, 0x80000000, 0x3e800000, 71, 0x00000002, 5, 0x105 }, - { 0x0, 0x3ffffffc, 0x3e800000, 0x3f000000, 97, 0x80000003, 2, 0x102 }, - { 0x0, 0x403fffff, 0x3e800000, 0x3e800000, 102, 0x00000003, 4, 0x104 }, + { 0x0, 0x40000000, 0x3e800000, 0x3f000000, 25, 0x80000000, 2, 0x102 }, + { 0x0, 0x40400000, 0x3e800000, 0x3e800000, 64, 0x00000002, 4, 0x104 }, + { 0x0, 0x40000000, 0x3f000000, 0x3e800000, 0, 0x80000000, 4, 0x104 }, + { 0x0, 0x403ffffc, 0x80000000, 0x3e800000, 69, 0x00000002, 5, 0x105 }, + { 0x0, 0x40000000, 0x3f000000, 0x3e800000, 2, 0x80000000, 4, 0x104 }, + { 0x0, 0x40400000, 0x3e800000, 0x3e800000, 68, 0x00000002, 4, 0x104 }, + { 0x0, 0x40000000, 0x3f000000, 0x3e800000, 4, 0x80000000, 4, 0x104 }, + { 0x0, 0x403fffff, 0x3e800000, 0x3e800000, 70, 0x00000002, 4, 0x104 }, { 0x0, 0x3fe00000, 0x80000000, 0x80000000, 0, 0x80000000, 3, 0x103 }, { 0x0, 0x403ffffc, 0x00000000, 0x3f400000, 64, 0x00000002, 5, 0x105 }, { 0x0, 0x3ff55555, 0x3eaaaaab, 0x80000000, 2, 0x80000000, 3, 0x103 }, { 0x0, 0x403ffffb, 0x00000000, 0x3f400000, 66, 0x00000002, 5, 0x105 }, - { 0x0, 0x3ffffffe, 0x80000000, 0x3f000000, 107, 0x80000003, 2, 0x102 }, - { 0x0, 0x403ffffb, 0x00000000, 0x3f400000, 100, 0x00000003, 5, 0x105 }, - { 0x0, 0x3ffffffe, 0x80000000, 0x3f000000, 99, 0x80000003, 2, 0x102 }, - { 0x0, 0x403ffffa, 0x00000000, 0x3f400000, 102, 0x00000003, 5, 0x105 }, - { 0x0, 0x3ffffffc, 0x80000000, 0x3e800000, 104, 0x80000003, 0, 0x100 }, + { 0x0, 0x40000000, 0x80000000, 0x80000000, 12, 0x80000000, 4, 0x104 }, + { 0x0, 0x403ffffc, 0x00000000, 0x3f400000, 68, 0x00000002, 5, 0x105 }, + { 0x0, 0x40000000, 0x80000000, 0x3f800000, 39, 0x80000001, 4, 0x104 }, + { 0x0, 0x403ffffb, 0x00000000, 0x3f400000, 70, 0x00000002, 5, 0x105 }, + { 0x0, 0x40000000, 0x3e800000, 0x3f000000, 27, 0x80000000, 2, 0x102 }, { 0x0, 0x403aaaab, 0x3f0aaaab, 0x3e000000, 0, 0x00000000, 3, 0x103 }, { 0x0, 0x3fe00000, 0x80000000, 0x3e000000, 3, 0x80000000, 3, 0x103 }, { 0x0, 0x40300000, 0x3f600000, 0x3e000000, 2, 0x00000000, 3, 0x103 }, { 0x0, 0x3ff55555, 0x3e555555, 0x3e000000, 4, 0x80000000, 3, 0x103 }, - { 0x0, 0x403ffffb, 0x80000000, 0x3e800000, 111, 0x00000003, 5, 0x105 }, - { 0x0, 0x3ffffffe, 0x3e800000, 0x3f000000, 99, 0x80000003, 2, 0x102 }, - { 0x0, 0x403ffffe, 0x3e800000, 0x3e800000, 110, 0x00000003, 4, 0x104 }, + { 0x0, 0x403ffffc, 0x80000000, 0x3e800000, 15, 0x00000000, 5, 0x105 }, + { 0x0, 0x40000000, 0x3f000000, 0x3e800000, 12, 0x80000000, 4, 0x104 }, + { 0x0, 0x40400000, 0x3e800000, 0x3e800000, 14, 0x00000000, 4, 0x104 }, { 0x0, 0x3fe00000, 0x80000000, 0x3f000000, 1, 0x80000000, 3, 0x103 }, { 0x0, 0x403ffffb, 0x00000000, 0x3f400000, 72, 0x00000002, 5, 0x105 }, { 0x0, 0x3ff55555, 0x3eaaaaab, 0x3e2aaaab, 3, 0x80000000, 3, 0x103 }, @@ -35,105 +35,105 @@ static const uint32_t kGolden[2][256][8] = { { 0x0, 0x40000000, 0x80000000, 0x80000000, 20, 0x80000000, 4, 0x104 }, { 0x0, 0x40300000, 0x3f200000, 0x3ec00000, 4, 0x00000000, 3, 0x103 }, { 0x0, 0x40000000, 0x80000000, 0x80000000, 22, 0x80000000, 4, 0x104 }, - { 0x0, 0x403ffffa, 0x00000000, 0x3f400000, 110, 0x00000003, 5, 0x105 }, - { 0x0, 0x3ffffffc, 0x80000000, 0x3e800000, 112, 0x80000003, 0, 0x100 }, + { 0x0, 0x403ffffc, 0x3f400000, 0x00000000, 15, 0x00000000, 5, 0x105 }, + { 0x0, 0x40000000, 0x3e800000, 0x3f000000, 29, 0x80000000, 2, 0x102 }, { 0x0, 0x403aaaab, 0x3d2aaaab, 0x3f200000, 0, 0x00000000, 3, 0x103 }, { 0x0, 0x3fe00000, 0x80000000, 0x3f200000, 3, 0x80000000, 3, 0x103 }, { 0x0, 0x40300000, 0x3ec00000, 0x3f200000, 2, 0x00000000, 3, 0x103 }, { 0x0, 0x3ff55555, 0x3eaaaaab, 0x3e955555, 5, 0x80000000, 3, 0x103 }, - { 0x0, 0x403ffffb, 0x80000000, 0x3e800000, 119, 0x00000003, 5, 0x105 }, + { 0x0, 0x40400000, 0x3e800000, 0x3e800000, 84, 0x00000002, 4, 0x104 }, { 0x0, 0x40000000, 0x3f000000, 0x3e800000, 20, 0x80000000, 4, 0x104 }, - { 0x0, 0x403ffffe, 0x3e800000, 0x3e800000, 118, 0x00000003, 4, 0x104 }, + { 0x0, 0x403fffff, 0x3e800000, 0x3e800000, 86, 0x00000002, 4, 0x104 }, { 0x0, 0x3fe00000, 0x80000000, 0x3f800000, 1, 0x80000000, 3, 0x103 }, { 0x0, 0x403ffffc, 0x00000000, 0x3f400000, 80, 0x00000002, 5, 0x105 }, { 0x0, 0x3ff55555, 0x3eaaaaab, 0x3f2aaaab, 3, 0x80000000, 3, 0x103 }, { 0x0, 0x403aaaab, 0x3f2aaaab, 0x3e555555, 3, 0x00000000, 3, 0x103 }, { 0x0, 0x40000000, 0x80000000, 0x80000000, 28, 0x80000000, 4, 0x104 }, { 0x0, 0x40300000, 0x3e000000, 0x3f600000, 4, 0x00000000, 3, 0x103 }, - { 0x0, 0x40000000, 0x80000000, 0x80000000, 30, 0x80000000, 4, 0x104 }, - { 0x0, 0x403ffffa, 0x00000000, 0x3f400000, 118, 0x00000003, 5, 0x105 }, - { 0x0, 0x3ffffffd, 0x80000000, 0x3e800000, 120, 0x80000003, 0, 0x100 }, + { 0x0, 0x40000000, 0x80000000, 0x3f800000, 55, 0x80000001, 4, 0x104 }, + { 0x0, 0x403ffffb, 0x00000000, 0x3f400000, 86, 0x00000002, 5, 0x105 }, + { 0x0, 0x40000000, 0x3e800000, 0x3f000000, 31, 0x80000000, 2, 0x102 }, { 0x0, 0x403ffffc, 0x80000000, 0x3e800000, 27, 0x00000000, 5, 0x105 }, - { 0x0, 0x3ffffffe, 0x3f000000, 0x3e800000, 120, 0x80000003, 4, 0x104 }, + { 0x0, 0x40000000, 0x3f000000, 0x3e800000, 24, 0x80000000, 4, 0x104 }, { 0x0, 0x403ffffb, 0x80000000, 0x3e800000, 93, 0x00000002, 5, 0x105 }, - { 0x0, 0x40000000, 0x3f000000, 0x3e800000, 122, 0x80000003, 4, 0x104 }, - { 0x0, 0x403ffffb, 0x80000000, 0x3e800000, 127, 0x00000003, 5, 0x105 }, + { 0x0, 0x40000000, 0x3f000000, 0x3e800000, 26, 0x80000000, 4, 0x104 }, + { 0x0, 0x403ffffc, 0x80000000, 0x3e800000, 31, 0x00000000, 5, 0x105 }, { 0x0, 0x40000000, 0x3f000000, 0x3e800000, 28, 0x80000000, 4, 0x104 }, - { 0x0, 0x403ffffe, 0x3e800000, 0x3e800000, 126, 0x00000003, 4, 0x104 }, - { 0x0, 0x3ffffffd, 0x80000000, 0x3f800000, 121, 0x80000003, 4, 0x104 }, + { 0x0, 0x40400000, 0x3e800000, 0x3e800000, 30, 0x00000000, 4, 0x104 }, + { 0x0, 0x40000000, 0x80000000, 0x3f800000, 25, 0x80000000, 4, 0x104 }, { 0x0, 0x403ffffb, 0x00000000, 0x3f400000, 88, 0x00000002, 5, 0x105 }, - { 0x0, 0x3fffffff, 0x80000000, 0x3f800000, 120, 0x80000003, 4, 0x104 }, - { 0x0, 0x403ffffc, 0x00000000, 0x3f400000, 26, 0x00000000, 5, 0x105 }, - { 0x0, 0x40000000, 0x80000000, 0x3f800000, 122, 0x80000003, 4, 0x104 }, - { 0x0, 0x403ffffa, 0x00000000, 0x3f400000, 124, 0x00000003, 5, 0x105 }, + { 0x0, 0x40000000, 0x80000000, 0x3f800000, 24, 0x80000000, 4, 0x104 }, + { 0x0, 0x403ffffc, 0x3f400000, 0x00000000, 27, 0x00000000, 5, 0x105 }, + { 0x0, 0x40000000, 0x80000000, 0x3f800000, 26, 0x80000000, 4, 0x104 }, + { 0x0, 0x403ffffb, 0x00000000, 0x3f400000, 92, 0x00000002, 5, 0x105 }, { 0x0, 0x40000000, 0x80000000, 0x3f800000, 28, 0x80000000, 4, 0x104 }, - { 0x0, 0x403ffffa, 0x00000000, 0x3f400000, 126, 0x00000003, 5, 0x105 }, - { 0x0, 0x3f7ffffd, 0x33800000, 0x3d878f51, 97, 0x80000003, 0, 0x100 }, + { 0x0, 0x403ffffc, 0x3f400000, 0x00000000, 31, 0x00000000, 5, 0x105 }, + { 0x0, 0x3f800000, 0x3d878f40, 0x3f6f0e18, 25, 0x80000000, 2, 0x102 }, { 0x0, 0x3f7ffffa, 0x3f000003, 0x3ef794dc, 73, 0x00000002, 5, 0x105 }, { 0x0, 0x3f7ffff9, 0x359f34db, 0x3f79a6dc, 81, 0x00000002, 5, 0x105 }, { 0x0, 0x3f6ba787, 0x3e658a10, 0x3f3c1362, 2, 0x80000000, 6, 0x106 }, { 0x0, 0x3f800000, 0x3f000000, 0x3e1e8470, 23, 0x80000000, 4, 0x104 }, { 0x0, 0x3f7ffffa, 0x3effffe2, 0x3e4e1b00, 83, 0x00000002, 5, 0x105 }, { 0x0, 0x3f7ffff8, 0x3f78c284, 0x3ce7af42, 64, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7ffffe, 0x3ee053bc, 0x3d7d625e, 102, 0x00000003, 4, 0x104 }, + { 0x0, 0x3f7fffff, 0x3ee053b8, 0x3d7d626e, 70, 0x00000002, 4, 0x104 }, { 0x0, 0x3f800000, 0x3ea018d0, 0x3e3fce60, 22, 0x80000000, 4, 0x104 }, { 0x0, 0x3f7ffff9, 0x3f000008, 0x3e5d7235, 89, 0x00000002, 5, 0x105 }, { 0x0, 0x3f760638, 0x3f30bc73, 0x3e87ee1f, 1, 0x00000000, 3, 0x103 }, { 0x0, 0x3f7ffff9, 0x3f000007, 0x3ea7d5e3, 73, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f800000, 0x331848a2, 0x3f33dbae, 123, 0x80000003, 4, 0x104 }, + { 0x0, 0x3f800000, 0x3e9848a0, 0x3f33dbb0, 24, 0x80000000, 4, 0x104 }, { 0x0, 0x3f7ffff9, 0x3d8967f0, 0x3edda620, 10, 0x00000000, 5, 0x105 }, { 0x0, 0x3f7ffff9, 0x3f000010, 0x3dee19ee, 73, 0x00000002, 5, 0x105 }, { 0x0, 0x3f6052c3, 0x3f3318d4, 0x3dc04463, 1, 0x80000000, 6, 0x106 }, { 0x0, 0x3f800000, 0x3e6c8d40, 0x3f44dcb0, 20, 0x80000000, 4, 0x104 }, - { 0x0, 0x3f7ffffd, 0x3f000004, 0x3d11e150, 111, 0x00000003, 4, 0x104 }, + { 0x0, 0x3f800000, 0x3f000000, 0x3d11e140, 15, 0x00000000, 4, 0x104 }, { 0x0, 0x3f64fc82, 0x3bc646d0, 0x3f4c7ef1, 5, 0x00000000, 6, 0x106 }, { 0x0, 0x3f7ffff8, 0x35593b39, 0x3dd898f2, 73, 0x00000002, 5, 0x105 }, { 0x0, 0x3f711e1a, 0x3e072700, 0x3d39ae65, 0, 0x80000000, 3, 0x103 }, - { 0x0, 0x3f7ffff8, 0x35400002, 0x3eb95dc0, 103, 0x00000003, 5, 0x105 }, + { 0x0, 0x3f7ffff9, 0x3528d448, 0x3eb95dc0, 71, 0x00000002, 5, 0x105 }, { 0x0, 0x3f7ffff9, 0x3efffff8, 0x3de88e01, 89, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7ffff4, 0x3effffb8, 0x3effd986, 127, 0x00000003, 5, 0x105 }, - { 0x0, 0x3f7ffffe, 0x32480a29, 0x3edfd75d, 99, 0x80000003, 4, 0x104 }, + { 0x0, 0x3f7ffff7, 0x3effffc8, 0x3effd970, 95, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f800000, 0x3f101458, 0x3edfd750, 0, 0x80000000, 4, 0x104 }, { 0x0, 0x3f5cae6d, 0x3bd39700, 0x3f4d2acb, 0, 0x80000000, 6, 0x106 }, { 0x0, 0x3f7ffffa, 0x3f17030b, 0x3ed1f9e6, 84, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7ffff7, 0x3ec09e8d, 0x3f1fb0b7, 116, 0x00000003, 5, 0x105 }, + { 0x0, 0x3f7ffff8, 0x3ec09e8e, 0x3f1fb0b7, 84, 0x00000002, 5, 0x105 }, { 0x0, 0x3f7a839a, 0x3e549290, 0x3e4e945d, 2, 0x80000000, 3, 0x103 }, { 0x0, 0x3f76e818, 0x3f26eea0, 0x3e667111, 5, 0x00000000, 3, 0x103 }, { 0x0, 0x3f7ffff9, 0x3db5e363, 0x3ed2872d, 80, 0x00000002, 5, 0x105 }, { 0x0, 0x3f681e3d, 0x3e549f8d, 0x3f3039b5, 2, 0x00000000, 3, 0x103 }, - { 0x0, 0x3f7ffffc, 0x3f000002, 0x3e71bbd9, 97, 0x80000003, 4, 0x104 }, - { 0x0, 0x3f7ffff6, 0x3f00001a, 0x3efda1b0, 103, 0x00000003, 5, 0x105 }, + { 0x0, 0x3f800000, 0x3f000000, 0x3e71bc10, 1, 0x80000000, 4, 0x104 }, + { 0x0, 0x3f7ffff8, 0x3f000016, 0x3efda1c0, 71, 0x00000002, 5, 0x105 }, { 0x0, 0x3f7ffffa, 0x3f792cae, 0x3cda68fc, 66, 0x00000002, 5, 0x105 }, { 0x0, 0x3f7ffff8, 0x3ee56754, 0x3d54c4c7, 64, 0x00000002, 5, 0x105 }, { 0x0, 0x3f6c4aea, 0x3dc95bac, 0x3f42ada5, 7, 0x80000000, 3, 0x103 }, { 0x0, 0x3f7ffff9, 0x3effffe0, 0x3d8aba81, 73, 0x00000002, 5, 0x105 }, { 0x0, 0x3f79d359, 0x3daefdf9, 0x3f0be1b6, 0, 0x00000000, 3, 0x103 }, - { 0x0, 0x3f7ffff6, 0x35100001, 0x3f0311eb, 103, 0x00000003, 5, 0x105 }, + { 0x0, 0x3f7ffff7, 0x34ff3b86, 0x3f0311e8, 71, 0x00000002, 5, 0x105 }, { 0x0, 0x3f52c715, 0x3e0ee67a, 0x3be60824, 9, 0x80000001, 6, 0x106 }, { 0x0, 0x3f7a6dcc, 0x3e9dc06c, 0x3e9eb79b, 4, 0x00000000, 3, 0x103 }, - { 0x0, 0x3f7ffff8, 0x3f141d4a, 0x3ed7c558, 124, 0x00000003, 5, 0x105 }, - { 0x0, 0x3f7ffffe, 0x3f000000, 0x3dc5f281, 103, 0x00000003, 4, 0x104 }, - { 0x0, 0x3f7ffffd, 0x3e4fea82, 0x3e980ac5, 104, 0x80000003, 4, 0x104 }, + { 0x0, 0x3f7ffff9, 0x3f141d4b, 0x3ed7c559, 92, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f7ffffe, 0x3f000000, 0x3dc5f27d, 71, 0x00000002, 4, 0x104 }, + { 0x0, 0x3f7fffff, 0x3e4fea70, 0x3e980ac8, 8, 0x80000000, 4, 0x104 }, { 0x0, 0x3f7ffffa, 0x34b55665, 0x3eaa99c0, 93, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7ffffd, 0x3f000006, 0x3eb315b8, 111, 0x00000003, 4, 0x104 }, - { 0x0, 0x3f800000, 0x80000000, 0x3eebaa80, 25, 0x00000000, 0, 0x100 }, - { 0x0, 0x3f7fffff, 0x3e6973a4, 0x3effffff, 96, 0x80000003, 2, 0x102 }, + { 0x0, 0x3f7fffff, 0x3f000001, 0x3eb315ca, 79, 0x00000002, 4, 0x104 }, + { 0x0, 0x3f800000, 0x3eebaa80, 0x3f0a2ac0, 31, 0x00000000, 2, 0x102 }, + { 0x0, 0x3f800000, 0x3e8b4633, 0x3f3a5ce5, 4, 0x80000000, 4, 0x104 }, { 0x0, 0x3f7ffff9, 0x3f000003, 0x3e611d40, 65, 0x00000002, 5, 0x105 }, { 0x0, 0x3f70d836, 0x3ed96353, 0x3e57f246, 5, 0x80000000, 6, 0x106 }, - { 0x0, 0x3f7ffff8, 0x3e374285, 0x3ea45ea0, 122, 0x00000003, 5, 0x105 }, - { 0x0, 0x3f7fffff, 0x3c80fb20, 0x3ef7f04c, 106, 0x80000003, 4, 0x104 }, + { 0x0, 0x3f7ffff8, 0x3e374280, 0x3ea45ea0, 26, 0x00000000, 5, 0x105 }, + { 0x0, 0x3f800000, 0x3c80fb00, 0x3ef7f050, 10, 0x80000000, 4, 0x104 }, { 0x0, 0x3f7ffffa, 0x3efffff1, 0x3e944807, 89, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7ffff8, 0x3d35a890, 0x3ee94adc, 126, 0x00000003, 5, 0x105 }, - { 0x0, 0x3f7ffff7, 0x3effffd4, 0x3e582798, 125, 0x00000003, 5, 0x105 }, + { 0x0, 0x3f7ffffa, 0x3d35a8a0, 0x3ee94ae0, 30, 0x00000000, 5, 0x105 }, + { 0x0, 0x3f7ffff7, 0x3effffda, 0x3e582790, 93, 0x00000002, 5, 0x105 }, { 0x0, 0x3f719475, 0x3e7218da, 0x3eaa6b9e, 3, 0x80000000, 3, 0x103 }, { 0x0, 0x3f6494a2, 0x3f1fb7ed, 0x3ea1fde7, 6, 0x00000000, 3, 0x103 }, { 0x0, 0x3f6d25ea, 0x3f445833, 0x3e11348c, 2, 0x00000000, 3, 0x103 }, - { 0x0, 0x3f7ffff8, 0x3efffffd, 0x3e5c5ea7, 103, 0x00000003, 5, 0x105 }, - { 0x0, 0x3f7ffffe, 0x343807f7, 0x3f0ff00f, 121, 0x80000003, 4, 0x104 }, + { 0x0, 0x3f7ffff8, 0x3efffffd, 0x3e5c5ea8, 71, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f800000, 0x80000000, 0x3f0ff018, 25, 0x80000000, 4, 0x104 }, { 0x0, 0x3f7ffff9, 0x344380da, 0x3f63f931, 73, 0x00000002, 5, 0x105 }, { 0x0, 0x3f7ffffa, 0x3f4152a4, 0x3e7ab55f, 90, 0x00000002, 5, 0x105 }, { 0x0, 0x3f7ffff7, 0x3effffe3, 0x3e852084, 93, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7fffff, 0x3f0da5a0, 0x3ee4b4c0, 122, 0x80000003, 4, 0x104 }, - { 0x0, 0x3f7ffffa, 0x34a00000, 0x3f1de4e1, 9, 0x00000000, 5, 0x105 }, + { 0x0, 0x3f7fffff, 0x3f0da5a0, 0x3ee4b4c0, 26, 0x80000000, 4, 0x104 }, + { 0x0, 0x3f7ffffa, 0x34ac4364, 0x3f1de4e0, 73, 0x00000002, 5, 0x105 }, { 0x0, 0x3f5c2c0e, 0x3e2ef1fa, 0x3f3d53a4, 1, 0x80000000, 6, 0x106 }, { 0x0, 0x3f7ffff8, 0x3ef77fe1, 0x3f04400c, 90, 0x00000002, 5, 0x105 }, { 0x0, 0x3f800000, 0x3f5f3c5c, 0x3e030e90, 28, 0x80000000, 4, 0x104 }, @@ -144,249 +144,249 @@ static const uint32_t kGolden[2][256][8] = { { 0x0, 0x3f7ffff9, 0x3e3f4086, 0x3ea05f96, 88, 0x00000002, 5, 0x105 }, { 0x0, 0x3f79948f, 0x3f23bab4, 0x3e656df9, 1, 0x00000000, 3, 0x103 }, { 0x0, 0x3f7ffff7, 0x3e60fbae, 0x3f47c0fa, 66, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f800000, 0x3efffffd, 0x3ea2b2aa, 79, 0x80000002, 4, 0x104 }, + { 0x0, 0x3f800000, 0x3f000000, 0x3ea2b2a8, 15, 0x80000000, 4, 0x104 }, { 0x0, 0x3f7ffff8, 0x3ed3c057, 0x3db0feed, 72, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7ffffe, 0x3edc8264, 0x3d8df67a, 102, 0x00000003, 4, 0x104 }, + { 0x0, 0x3f7fffff, 0x3edc8262, 0x3d8df67f, 70, 0x00000002, 4, 0x104 }, { 0x0, 0x3f6fc2c9, 0x3f587690, 0x3e075479, 7, 0x00000000, 3, 0x103 }, - { 0x0, 0x3f7ffffe, 0x3e9236bf, 0x3f000004, 104, 0x80000003, 2, 0x102 }, + { 0x0, 0x3f800000, 0x3e5b9270, 0x3f491b64, 2, 0x80000000, 4, 0x104 }, { 0x0, 0x3f7b01d8, 0x3ecf4c13, 0x3e5732ce, 0, 0x00000000, 3, 0x103 }, { 0x0, 0x3f7ffffa, 0x3ed963a9, 0x3d9a71ca, 68, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7ffff9, 0x3f4cfd30, 0x3e4c0af0, 24, 0x00000000, 5, 0x105 }, - { 0x0, 0x3f7ffffd, 0x3dc53d51, 0x34b3ac29, 112, 0x80000003, 2, 0x102 }, + { 0x0, 0x3f7ffff9, 0x3f4cfd2e, 0x3e4c0af3, 88, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f800000, 0x3eceb0a8, 0x3dc53d60, 2, 0x80000000, 4, 0x104 }, { 0x0, 0x3f7ffff8, 0x3c88b1f1, 0x3f7bba6f, 88, 0x00000002, 5, 0x105 }, { 0x0, 0x3f7ffff8, 0x345d935e, 0x3eec9aed, 69, 0x00000002, 5, 0x105 }, { 0x0, 0x3f763bbb, 0x3eb37160, 0x3eab065a, 0, 0x00000000, 3, 0x103 }, { 0x0, 0x3f7fffff, 0x3e397208, 0x3f51a37e, 28, 0x80000000, 4, 0x104 }, { 0x0, 0x3f7b5ec8, 0x3f058734, 0x3da57d7b, 0, 0x00000000, 3, 0x103 }, - { 0x0, 0x3f7ffff8, 0x33debb08, 0x3f75d83c, 127, 0x00000003, 5, 0x105 }, + { 0x0, 0x3f7ffff9, 0x33bebb08, 0x3f75d83c, 95, 0x00000002, 5, 0x105 }, { 0x0, 0x3f78d7b2, 0x3e9dcb1f, 0x3eaf3a81, 4, 0x00000000, 3, 0x103 }, - { 0x0, 0x3f7ffffd, 0x34fffffc, 0x3e976a28, 97, 0x80000003, 0, 0x100 }, + { 0x0, 0x3f800000, 0x3e976a48, 0x3f344adc, 25, 0x80000000, 2, 0x102 }, { 0x0, 0x3f7ffff9, 0x3eb63614, 0x3e1393eb, 66, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7ffffd, 0x3e965965, 0x3e534d4c, 126, 0x00000003, 4, 0x104 }, + { 0x0, 0x3f800000, 0x3e965948, 0x3e534d70, 30, 0x00000000, 4, 0x104 }, { 0x0, 0x3f7ffff8, 0x3e06ad8a, 0x3ebca923, 88, 0x00000002, 5, 0x105 }, { 0x0, 0x3f800000, 0x3ed7580c, 0x3da29fd0, 14, 0x80000000, 4, 0x104 }, { 0x0, 0x3f66a751, 0x3e0696c8, 0x3f5b5f23, 0, 0x00000000, 3, 0x103 }, - { 0x0, 0x3f7ffffe, 0x3f000002, 0x3eb07957, 111, 0x00000003, 4, 0x104 }, + { 0x0, 0x3f7fffff, 0x3f000000, 0x3eb0795d, 79, 0x00000002, 4, 0x104 }, { 0x0, 0x3f7ffff9, 0x3eba09a4, 0x3e0bec7a, 72, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7fffff, 0x3f000000, 0x3ecf308c, 29, 0x80000000, 4, 0x104 }, + { 0x0, 0x3f7fffff, 0x3f679846, 0x3dc33dd0, 39, 0x80000001, 2, 0x102 }, { 0x0, 0x3f7ffffa, 0x350ed981, 0x3eed9810, 65, 0x00000002, 5, 0x105 }, { 0x0, 0x3f7ffff7, 0x3e9b0a14, 0x3f327af5, 88, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7ffff7, 0x35a00003, 0x3f188ac0, 119, 0x00000003, 5, 0x105 }, + { 0x0, 0x3f7ffff8, 0x358ceea7, 0x3f188ac5, 87, 0x00000002, 5, 0x105 }, { 0x0, 0x3f77a9cc, 0x3c9c7f58, 0x3ed2f161, 0, 0x80000000, 6, 0x106 }, - { 0x0, 0x3f7ffff7, 0x35819393, 0x3dc9c931, 127, 0x00000003, 5, 0x105 }, + { 0x0, 0x3f7ffff9, 0x35400000, 0x3dc9c940, 31, 0x00000000, 5, 0x105 }, { 0x0, 0x3f7ffff8, 0x3599f5c4, 0x3f4fae1d, 81, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7ffff8, 0x35600002, 0x3f1d128e, 111, 0x00000003, 5, 0x105 }, + { 0x0, 0x3f7ffff9, 0x354744a4, 0x3f1d1291, 79, 0x00000002, 5, 0x105 }, { 0x0, 0x3f800000, 0x3f648878, 0x3ddbbc40, 28, 0x80000000, 4, 0x104 }, { 0x0, 0x3f7ffffa, 0x35200000, 0x3ccd882c, 27, 0x00000000, 5, 0x105 }, - { 0x0, 0x3f7ffffe, 0x3e9a0220, 0x3e4bfbd0, 102, 0x00000003, 4, 0x104 }, - { 0x0, 0x3f7ffffd, 0x3f000003, 0x3da72fcc, 111, 0x00000003, 4, 0x104 }, + { 0x0, 0x3f7fffff, 0x3e9a021b, 0x3e4bfbd5, 70, 0x00000002, 4, 0x104 }, + { 0x0, 0x3f800000, 0x3f000000, 0x3da72fe0, 15, 0x00000000, 4, 0x104 }, { 0x0, 0x3f800000, 0x3e151f80, 0x3eb57040, 22, 0x80000000, 4, 0x104 }, - { 0x0, 0x3f7ffffe, 0x3f00000a, 0x3edd2322, 111, 0x00000003, 4, 0x104 }, + { 0x0, 0x3f800000, 0x3f000000, 0x3edd2330, 15, 0x00000000, 4, 0x104 }, { 0x0, 0x3f5a6702, 0x3c4d5104, 0x3f6cfda3, 1, 0x80000000, 6, 0x106 }, { 0x0, 0x3f64eaef, 0x3d8ea9a7, 0x3f640db7, 6, 0x00000000, 3, 0x103 }, - { 0x0, 0x3f7ffffd, 0x3efffff6, 0x3d944b98, 113, 0x80000003, 0, 0x100 }, - { 0x0, 0x3f7ffffa, 0x3cadbb04, 0x3ef52438, 124, 0x00000003, 5, 0x105 }, + { 0x0, 0x3f800000, 0x32944b5a, 0x3f12896c, 81, 0x80000002, 4, 0x104 }, + { 0x0, 0x3f7ffffa, 0x3cadbb04, 0x3ef52438, 92, 0x00000002, 5, 0x105 }, { 0x0, 0x3f7ffff9, 0x3d556639, 0x3ee55350, 88, 0x00000002, 5, 0x105 }, { 0x0, 0x3f7ffff8, 0x3effffda, 0x3e9cec8b, 83, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7ffffd, 0x3dbac67d, 0x3ed14e55, 97, 0x80000003, 2, 0x102 }, + { 0x0, 0x3f800000, 0x3f68a73c, 0x3dbac620, 4, 0x80000000, 4, 0x104 }, { 0x0, 0x3f7ffff9, 0x34de28f9, 0x3ef147c6, 85, 0x00000002, 5, 0x105 }, { 0x0, 0x3f738af7, 0x3e740404, 0x3f1a12dd, 2, 0x80000000, 6, 0x106 }, { 0x0, 0x3f7ffffa, 0x3f1b3c9e, 0x3ec986bc, 84, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f800000, 0x3bf5c441, 0x3efc28f1, 122, 0x80000003, 4, 0x104 }, + { 0x0, 0x3f800000, 0x3bf5c480, 0x3efc28ee, 26, 0x80000000, 4, 0x104 }, { 0x0, 0x3f695785, 0x3e11de66, 0x3f3e2607, 2, 0x00000000, 3, 0x103 }, { 0x0, 0x3f6bfc68, 0x3de2b043, 0x3f15b4a2, 5, 0x00000000, 6, 0x106 }, - { 0x0, 0x3f7ffff9, 0x3f4163b8, 0x3e7a70ce, 108, 0x00000003, 5, 0x105 }, + { 0x0, 0x3f7ffffa, 0x3f4163bc, 0x3e7a70cb, 76, 0x00000002, 5, 0x105 }, { 0x0, 0x3f4ff44b, 0x3e293406, 0x3ec7959a, 1, 0x00000000, 6, 0x106 }, - { 0x0, 0x3f532d8b, 0x3f196c52, 0x3eb804a3, 1, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f532d8b, 0x3f196c52, 0x3eb804a3, 1, 0x00000000, 1, 0x101 }, { 0x0, 0x3f7cdcd6, 0x3e64bb34, 0x3f33fe34, 66, 0x00000002, 5, 0x105 }, { 0x0, 0x3f7e2de1, 0x3ef9c2c5, 0x3ef063c4, 90, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f531696, 0x3f1291a6, 0x3cc460ca, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f531696, 0x3f1291a6, 0x3cc460ca, 0, 0x00000000, 1, 0x101 }, { 0x0, 0x3f7eac1d, 0x3f53136e, 0x3e13d4fd, 90, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f535e35, 0x3ea5d7ee, 0x3e900b53, 0, 0x00000000, 3, 0x103 }, - { 0x0, 0x3f5373f3, 0x3ea61344, 0x3e912be7, 0, 0x00000000, 3, 0x103 }, - { 0x0, 0x3f53b04a, 0x3e4a6ad8, 0x3ed5cf27, 0, 0x00000000, 3, 0x103 }, - { 0x0, 0x3f537b21, 0x3f1bd906, 0x3ebe90d2, 1, 0x00000000, 3, 0x103 }, - { 0x0, 0x3f537ef1, 0x3f1bf784, 0x3e298ebf, 1, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f535e35, 0x3ea5d7ee, 0x3e900b53, 0, 0x00000000, 1, 0x101 }, + { 0x0, 0x3f5373f3, 0x3ea61344, 0x3e912be7, 0, 0x00000000, 1, 0x101 }, + { 0x0, 0x3f53b04a, 0x3e4a6ad8, 0x3ed5cf27, 0, 0x00000000, 1, 0x101 }, + { 0x0, 0x3f537b21, 0x3f1bd906, 0x3ebe90d2, 1, 0x00000000, 1, 0x101 }, + { 0x0, 0x3f537ef1, 0x3f1bf784, 0x3e298ebf, 1, 0x00000000, 1, 0x101 }, { 0x0, 0x3f8086e6, 0x3cca58cb, 0x3f02df2f, 93, 0x00000002, 5, 0x105 }, { 0x0, 0x3f7d37fe, 0x3ed975f9, 0x3f0294f7, 90, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f532e5c, 0x3e4ad678, 0x3ecd7a7d, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f532e5c, 0x3e4ad678, 0x3ecd7a7d, 0, 0x00000000, 1, 0x101 }, { 0x0, 0x3f817b05, 0x3d8e21db, 0x3eae59d9, 93, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f53644a, 0x3f1b2249, 0x3e6c1301, 1, 0x00000000, 3, 0x103 }, - { 0x0, 0x3f536ada, 0x3e6d9970, 0x3ebfe0e0, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f53644a, 0x3f1b2249, 0x3e6c1301, 1, 0x00000000, 1, 0x101 }, + { 0x0, 0x3f536ada, 0x3e6d9970, 0x3ebfe0e0, 0, 0x00000000, 1, 0x101 }, { 0x0, 0x3f7d1615, 0x3f643c3c, 0x3d24840f, 90, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f80641f, 0x3c962e49, 0x3f36282b, 125, 0x00000003, 5, 0x105 }, + { 0x0, 0x3f806420, 0x3c962fe7, 0x3f36281e, 93, 0x00000002, 5, 0x105 }, { 0x0, 0x3f7cfc7f, 0x3f34250b, 0x3e6717bc, 90, 0x00000002, 5, 0x105 }, { 0x0, 0x3f7e0010, 0x3ee8bc6a, 0x3eff4456, 66, 0x00000002, 5, 0x105 }, { 0x0, 0x3f811cfb, 0x3d55bc6f, 0x3dbe83d4, 93, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f53b277, 0x3f089ff1, 0x3da79e02, 0, 0x00000000, 3, 0x103 }, - { 0x0, 0x3f534ae7, 0x3eeb3474, 0x3e12f3ee, 0, 0x00000000, 3, 0x103 }, - { 0x0, 0x3f537348, 0x3cf986e6, 0x3f13ce02, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f53b277, 0x3f089ff1, 0x3da79e02, 0, 0x00000000, 1, 0x101 }, + { 0x0, 0x3f534ae7, 0x3eeb3474, 0x3e12f3ee, 0, 0x00000000, 1, 0x101 }, + { 0x0, 0x3f537348, 0x3cf986e6, 0x3f13ce02, 0, 0x00000000, 1, 0x101 }, { 0x0, 0x3f4b6fe1, 0x3cc535ac, 0x3f081940, 1, 0x00000000, 6, 0x106 }, { 0x0, 0x3f80084c, 0x3ac71c69, 0x3db9cd41, 69, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7edddd, 0x3e77aab9, 0x3f3b487b, 122, 0x00000003, 5, 0x105 }, - { 0x0, 0x3f536fc6, 0x3c0b33db, 0x3f195160, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f7edddf, 0x3e77aaee, 0x3f3b487a, 90, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f536fc6, 0x3c0b33db, 0x3f195160, 0, 0x00000000, 1, 0x101 }, { 0x0, 0x3f80bcc7, 0x3d0d94d9, 0x3e01d49c, 69, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f5334eb, 0x3f099f0e, 0x3d80423a, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f5334eb, 0x3f099f0e, 0x3d80423a, 0, 0x00000000, 1, 0x101 }, { 0x0, 0x3f7cabe8, 0x3e41b8e7, 0x3f3b9934, 66, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f532ef9, 0x3f1977c6, 0x3e85db3e, 1, 0x00000000, 3, 0x103 }, - { 0x0, 0x3f535503, 0x3cc2bcc1, 0x3f14922d, 0, 0x00000000, 3, 0x103 }, - { 0x0, 0x3f538608, 0x3ecabf0b, 0x3e5b42e7, 0, 0x00000000, 3, 0x103 }, - { 0x0, 0x3f80a973, 0x3cfe2caa, 0x3ef96bc0, 125, 0x00000003, 5, 0x105 }, - { 0x0, 0x3f535d6c, 0x3f03219a, 0x3dbe4dfd, 0, 0x00000000, 3, 0x103 }, - { 0x0, 0x3f5343a5, 0x3ea800d4, 0x3e8c3979, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f532ef9, 0x3f1977c6, 0x3e85db3e, 1, 0x00000000, 1, 0x101 }, + { 0x0, 0x3f535503, 0x3cc2bcc1, 0x3f14922d, 0, 0x00000000, 1, 0x101 }, + { 0x0, 0x3f538608, 0x3ecabf0b, 0x3e5b42e7, 0, 0x00000000, 1, 0x101 }, + { 0x0, 0x3f80a973, 0x3cfe2ba5, 0x3ef96bd2, 93, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f535d6c, 0x3f03219a, 0x3dbe4dfd, 0, 0x00000000, 1, 0x101 }, + { 0x0, 0x3f5343a5, 0x3ea800d4, 0x3e8c3979, 0, 0x00000000, 1, 0x101 }, { 0x0, 0x3f81d8a8, 0x3d8fb50c, 0x3c8627a9, 92, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f5369f1, 0x3e00b76b, 0x3ef64354, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f5369f1, 0x3e00b76b, 0x3ef64354, 0, 0x00000000, 1, 0x101 }, { 0x0, 0x3f80688e, 0x3c9cd3f8, 0x3f75bcda, 69, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f5355c1, 0x3ed87cdc, 0x3e39be66, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f5355c1, 0x3ed87cdc, 0x3e39be66, 0, 0x00000000, 1, 0x101 }, { 0x0, 0x3f7e3aa2, 0x3f735972, 0x3c019519, 90, 0x00000002, 5, 0x105 }, { 0x0, 0x3f7f9177, 0x3f1a20d9, 0x3ec68fde, 90, 0x00000002, 5, 0x105 }, { 0x0, 0x3f4facc6, 0x3e20435d, 0x3ecd6d43, 1, 0x00000000, 6, 0x106 }, - { 0x0, 0x3f532fe7, 0x3de1b188, 0x3efa920c, 0, 0x00000000, 3, 0x103 }, - { 0x0, 0x3f539dbd, 0x3f1cede8, 0x3e1d79f8, 1, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f532fe7, 0x3de1b188, 0x3efa920c, 0, 0x00000000, 1, 0x101 }, + { 0x0, 0x3f539dbd, 0x3f1cede8, 0x3e1d79f8, 1, 0x00000000, 1, 0x101 }, { 0x0, 0x3f7f6244, 0x3f48709f, 0x3e4f73d8, 90, 0x00000002, 5, 0x105 }, { 0x0, 0x3f809a25, 0x3ce73741, 0x3f00af59, 69, 0x00000002, 5, 0x105 }, { 0x0, 0x3f7d9e68, 0x3f05869f, 0x3ed85fa1, 90, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f53ae22, 0x3e2a8867, 0x3ee59de4, 0, 0x00000000, 3, 0x103 }, - { 0x0, 0x3f538038, 0x3e9d4a69, 0x3e9ab90a, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f53ae22, 0x3e2a8867, 0x3ee59de4, 0, 0x00000000, 1, 0x101 }, + { 0x0, 0x3f538038, 0x3e9d4a69, 0x3e9ab90a, 0, 0x00000000, 1, 0x101 }, { 0x0, 0x3f80e2ce, 0x3d2a1a42, 0x3ef7c7d2, 93, 0x00000002, 5, 0x105 }, { 0x1, 0x00000000, 0x00000000, 0x00000000, 0, 0x00000000, 0, 0x0 }, - { 0x0, 0x3f53901d, 0x3ee09e4c, 0x3e30c6eb, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f53901d, 0x3ee09e4c, 0x3e30c6eb, 0, 0x00000000, 1, 0x101 }, { 0x0, 0x3f80e782, 0x3d2da155, 0x3e649458, 69, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f53777d, 0x3eb4a8ba, 0x3e82cf14, 0, 0x00000000, 3, 0x103 }, - { 0x0, 0x3f805d9e, 0x3c8c6d1c, 0x3f03f171, 125, 0x00000003, 5, 0x105 }, + { 0x0, 0x3f53777d, 0x3eb4a8ba, 0x3e82cf14, 0, 0x00000000, 1, 0x101 }, + { 0x0, 0x3f805d9e, 0x3c8c6c45, 0x3f03f177, 93, 0x00000002, 5, 0x105 }, { 0x0, 0x3f80f545, 0x3d37f3cd, 0x3e503679, 69, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f538794, 0x3ef8afd2, 0x3dff25a1, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f538794, 0x3ef8afd2, 0x3dff25a1, 0, 0x00000000, 1, 0x101 }, { 0x0, 0x3f7f6e29, 0x3c89b8c4, 0x3f78472b, 66, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f5355c8, 0x3f0d7a08, 0x3d53432d, 0, 0x00000000, 3, 0x103 }, - { 0x0, 0x3f531799, 0x3dc676f8, 0x3effdbcb, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f5355c8, 0x3f0d7a08, 0x3d53432d, 0, 0x00000000, 1, 0x101 }, + { 0x0, 0x3f531799, 0x3dc676f8, 0x3effdbcb, 0, 0x00000000, 1, 0x101 }, { 0x0, 0x3f7df3f6, 0x3eb0e7aa, 0x3f1b43ee, 66, 0x00000002, 5, 0x105 }, }, { - { 0x0, 0x3ffffffc, 0x80000000, 0x3e800000, 96, 0x80000003, 0, 0x100 }, + { 0x0, 0x40000000, 0x80000000, 0x3e800000, 0, 0x80000000, 0, 0x100 }, { 0x0, 0x403ffffb, 0x3f400000, 0x3e800000, 64, 0x00000002, 5, 0x105 }, - { 0x0, 0x3ffffffc, 0x3e800000, 0x3f000000, 113, 0x80000003, 2, 0x102 }, + { 0x0, 0x40000000, 0x80000000, 0x3e800000, 2, 0x80000000, 0, 0x100 }, { 0x0, 0x403ffffc, 0x3f400000, 0x3e800000, 2, 0x00000000, 5, 0x105 }, - { 0x0, 0x3ffffffc, 0x3e800000, 0x3f000000, 105, 0x80000003, 2, 0x102 }, + { 0x0, 0x40000000, 0x80000000, 0x3e800000, 4, 0x80000000, 0, 0x100 }, { 0x0, 0x403ffffb, 0x3f400000, 0x3e800000, 68, 0x00000002, 5, 0x105 }, - { 0x0, 0x3ffffffc, 0x3e800000, 0x3f000000, 97, 0x80000003, 2, 0x102 }, - { 0x0, 0x403ffffa, 0x3f400000, 0x3e800000, 102, 0x00000003, 5, 0x105 }, - { 0x0, 0x3fe00000, 0x80000000, 0x80000000, 0, 0x80000000, 3, 0x103 }, + { 0x0, 0x40000000, 0x80000000, 0x3e800000, 6, 0x80000000, 0, 0x100 }, + { 0x0, 0x403ffffc, 0x3f400000, 0x3e800000, 6, 0x00000000, 5, 0x105 }, + { 0x0, 0x3fe00000, 0x80000000, 0x80000000, 0, 0x80000000, 1, 0x101 }, { 0x0, 0x403ffffc, 0x00000000, 0x3f400000, 0, 0x00000000, 5, 0x105 }, - { 0x0, 0x3ff55555, 0x3eaaaaab, 0x80000000, 2, 0x80000000, 3, 0x103 }, + { 0x0, 0x3ff55555, 0x3eaaaaab, 0x80000000, 2, 0x80000000, 1, 0x101 }, { 0x0, 0x403ffffb, 0x00000000, 0x3f400000, 66, 0x00000002, 5, 0x105 }, - { 0x0, 0x3ffffffe, 0x3f000000, 0x3f000000, 104, 0x80000003, 2, 0x102 }, - { 0x0, 0x403ffffb, 0x00000000, 0x3f400000, 100, 0x00000003, 5, 0x105 }, - { 0x0, 0x3ffffffe, 0x3f000000, 0x3f000000, 96, 0x80000003, 2, 0x102 }, - { 0x0, 0x403ffffa, 0x00000000, 0x3f400000, 102, 0x00000003, 5, 0x105 }, - { 0x0, 0x3ffffffc, 0x80000000, 0x3e800000, 104, 0x80000003, 0, 0x100 }, - { 0x0, 0x403aaaab, 0x3f0aaaab, 0x3e000000, 0, 0x00000000, 3, 0x103 }, - { 0x0, 0x3fe00000, 0x80000000, 0x3e000000, 3, 0x80000000, 3, 0x103 }, - { 0x0, 0x40300000, 0x3f600000, 0x3e000000, 2, 0x00000000, 3, 0x103 }, - { 0x0, 0x3ff55555, 0x3e555555, 0x3e000000, 4, 0x80000000, 3, 0x103 }, - { 0x0, 0x403ffffb, 0x3f400000, 0x3e800000, 108, 0x00000003, 5, 0x105 }, - { 0x0, 0x3ffffffe, 0x3e800000, 0x3f000000, 99, 0x80000003, 2, 0x102 }, - { 0x0, 0x403ffffa, 0x3f400000, 0x3e800000, 110, 0x00000003, 5, 0x105 }, - { 0x0, 0x3fe00000, 0x80000000, 0x3f000000, 1, 0x80000000, 3, 0x103 }, + { 0x0, 0x40000000, 0x3f000000, 0x3f000000, 5, 0x80000000, 0, 0x100 }, + { 0x0, 0x403ffffc, 0x00000000, 0x3f400000, 4, 0x00000000, 5, 0x105 }, + { 0x0, 0x40000000, 0x3f000000, 0x3f000000, 7, 0x80000000, 0, 0x100 }, + { 0x0, 0x403ffffb, 0x00000000, 0x3f400000, 70, 0x00000002, 5, 0x105 }, + { 0x0, 0x40000000, 0x80000000, 0x3e800000, 8, 0x80000000, 0, 0x100 }, + { 0x0, 0x403aaaab, 0x3f0aaaab, 0x3e000000, 0, 0x00000000, 1, 0x101 }, + { 0x0, 0x3fe00000, 0x80000000, 0x3e000000, 3, 0x80000000, 1, 0x101 }, + { 0x0, 0x40300000, 0x3f600000, 0x3e000000, 2, 0x00000000, 1, 0x101 }, + { 0x0, 0x3ff55555, 0x3e555555, 0x3e000000, 4, 0x80000000, 1, 0x101 }, + { 0x0, 0x403ffffc, 0x3f400000, 0x3e800000, 12, 0x00000000, 5, 0x105 }, + { 0x0, 0x40000000, 0x80000000, 0x3e800000, 14, 0x80000000, 0, 0x100 }, + { 0x0, 0x403ffffb, 0x3f400000, 0x3e800000, 78, 0x00000002, 5, 0x105 }, + { 0x0, 0x3fe00000, 0x80000000, 0x3f000000, 1, 0x80000000, 1, 0x101 }, { 0x0, 0x403ffffb, 0x00000000, 0x3f400000, 72, 0x00000002, 5, 0x105 }, - { 0x0, 0x3ff55555, 0x3eaaaaab, 0x3e2aaaab, 3, 0x80000000, 3, 0x103 }, - { 0x0, 0x403aaaab, 0x3e955555, 0x3ec00000, 2, 0x00000000, 3, 0x103 }, + { 0x0, 0x3ff55555, 0x3eaaaaab, 0x3e2aaaab, 3, 0x80000000, 1, 0x101 }, + { 0x0, 0x403aaaab, 0x3e955555, 0x3ec00000, 2, 0x00000000, 1, 0x101 }, { 0x0, 0x40000000, 0x3f000000, 0x3f000000, 13, 0x80000000, 0, 0x100 }, - { 0x0, 0x40300000, 0x3f200000, 0x3ec00000, 4, 0x00000000, 3, 0x103 }, + { 0x0, 0x40300000, 0x3f200000, 0x3ec00000, 4, 0x00000000, 1, 0x101 }, { 0x0, 0x40000000, 0x3f000000, 0x3f000000, 15, 0x80000000, 0, 0x100 }, - { 0x0, 0x403ffffa, 0x00000000, 0x3f400000, 110, 0x00000003, 5, 0x105 }, - { 0x0, 0x3ffffffc, 0x80000000, 0x3e800000, 112, 0x80000003, 0, 0x100 }, - { 0x0, 0x403aaaab, 0x3d2aaaab, 0x3f200000, 0, 0x00000000, 3, 0x103 }, - { 0x0, 0x3fe00000, 0x80000000, 0x3f200000, 3, 0x80000000, 3, 0x103 }, - { 0x0, 0x40300000, 0x3ec00000, 0x3f200000, 2, 0x00000000, 3, 0x103 }, - { 0x0, 0x3ff55555, 0x3eaaaaab, 0x3e955555, 5, 0x80000000, 3, 0x103 }, + { 0x0, 0x403ffffc, 0x00000000, 0x3f400000, 14, 0x00000000, 5, 0x105 }, + { 0x0, 0x40000000, 0x80000000, 0x3e800000, 16, 0x80000000, 0, 0x100 }, + { 0x0, 0x403aaaab, 0x3d2aaaab, 0x3f200000, 0, 0x00000000, 1, 0x101 }, + { 0x0, 0x3fe00000, 0x80000000, 0x3f200000, 3, 0x80000000, 1, 0x101 }, + { 0x0, 0x40300000, 0x3ec00000, 0x3f200000, 2, 0x00000000, 1, 0x101 }, + { 0x0, 0x3ff55555, 0x3eaaaaab, 0x3e955555, 5, 0x80000000, 1, 0x101 }, { 0x0, 0x403ffffb, 0x3f400000, 0x3e800000, 84, 0x00000002, 5, 0x105 }, { 0x0, 0x40000000, 0x80000000, 0x3e800000, 22, 0x80000000, 0, 0x100 }, - { 0x0, 0x403ffffa, 0x3f400000, 0x3e800000, 118, 0x00000003, 5, 0x105 }, - { 0x0, 0x3fe00000, 0x80000000, 0x3f800000, 1, 0x80000000, 3, 0x103 }, + { 0x0, 0x403ffffc, 0x3f400000, 0x3e800000, 22, 0x00000000, 5, 0x105 }, + { 0x0, 0x3fe00000, 0x80000000, 0x3f800000, 1, 0x80000000, 1, 0x101 }, { 0x0, 0x403ffffc, 0x00000000, 0x3f400000, 16, 0x00000000, 5, 0x105 }, - { 0x0, 0x3ff55555, 0x3eaaaaab, 0x3f2aaaab, 3, 0x80000000, 3, 0x103 }, - { 0x0, 0x403aaaab, 0x3f2aaaab, 0x3e555555, 3, 0x00000000, 3, 0x103 }, + { 0x0, 0x3ff55555, 0x3eaaaaab, 0x3f2aaaab, 3, 0x80000000, 1, 0x101 }, + { 0x0, 0x403aaaab, 0x3f2aaaab, 0x3e555555, 3, 0x00000000, 1, 0x101 }, { 0x0, 0x40000000, 0x3f000000, 0x3f000000, 21, 0x80000000, 0, 0x100 }, - { 0x0, 0x40300000, 0x3e000000, 0x3f600000, 4, 0x00000000, 3, 0x103 }, + { 0x0, 0x40300000, 0x3e000000, 0x3f600000, 4, 0x00000000, 1, 0x101 }, { 0x0, 0x40000000, 0x3f000000, 0x3f000000, 23, 0x80000000, 0, 0x100 }, - { 0x0, 0x403ffffa, 0x00000000, 0x3f400000, 118, 0x00000003, 5, 0x105 }, - { 0x0, 0x3ffffffd, 0x80000000, 0x3e800000, 120, 0x80000003, 0, 0x100 }, + { 0x0, 0x403ffffb, 0x00000000, 0x3f400000, 86, 0x00000002, 5, 0x105 }, + { 0x0, 0x40000000, 0x80000000, 0x3e800000, 24, 0x80000000, 0, 0x100 }, { 0x0, 0x403ffffc, 0x3f400000, 0x3e800000, 24, 0x00000000, 5, 0x105 }, - { 0x0, 0x3ffffffe, 0x80000000, 0x3e800000, 122, 0x80000003, 0, 0x100 }, + { 0x0, 0x40000000, 0x80000000, 0x3e800000, 26, 0x80000000, 0, 0x100 }, { 0x0, 0x403ffffb, 0x3f400000, 0x3e800000, 90, 0x00000002, 5, 0x105 }, { 0x0, 0x40000000, 0x80000000, 0x3e800000, 28, 0x80000000, 0, 0x100 }, - { 0x0, 0x403ffffb, 0x3f400000, 0x3e800000, 124, 0x00000003, 5, 0x105 }, + { 0x0, 0x403ffffc, 0x3f400000, 0x3e800000, 28, 0x00000000, 5, 0x105 }, { 0x0, 0x40000000, 0x80000000, 0x3e800000, 30, 0x80000000, 0, 0x100 }, - { 0x0, 0x403ffffa, 0x3f400000, 0x3e800000, 126, 0x00000003, 5, 0x105 }, - { 0x0, 0x3ffffffd, 0x80000000, 0x3f800000, 121, 0x80000003, 4, 0x104 }, + { 0x0, 0x403ffffb, 0x3f400000, 0x3e800000, 94, 0x00000002, 5, 0x105 }, + { 0x0, 0x40000000, 0x3f000000, 0x3f000000, 25, 0x80000000, 0, 0x100 }, { 0x0, 0x403ffffb, 0x00000000, 0x3f400000, 88, 0x00000002, 5, 0x105 }, - { 0x0, 0x3fffffff, 0x80000000, 0x3f800000, 120, 0x80000003, 4, 0x104 }, + { 0x0, 0x40000000, 0x3f000000, 0x3f000000, 27, 0x80000000, 0, 0x100 }, { 0x0, 0x403ffffc, 0x00000000, 0x3f400000, 26, 0x00000000, 5, 0x105 }, { 0x0, 0x40000000, 0x3f000000, 0x3f000000, 29, 0x80000000, 0, 0x100 }, - { 0x0, 0x403ffffa, 0x00000000, 0x3f400000, 124, 0x00000003, 5, 0x105 }, + { 0x0, 0x403ffffb, 0x00000000, 0x3f400000, 92, 0x00000002, 5, 0x105 }, { 0x0, 0x40000000, 0x3f000000, 0x3f000000, 31, 0x80000000, 0, 0x100 }, - { 0x0, 0x403ffffa, 0x00000000, 0x3f400000, 126, 0x00000003, 5, 0x105 }, - { 0x0, 0x3f7ffffd, 0x33800000, 0x3d878f51, 97, 0x80000003, 0, 0x100 }, + { 0x0, 0x403ffffc, 0x00000000, 0x3f400000, 30, 0x00000000, 5, 0x105 }, + { 0x0, 0x3f800000, 0x80000000, 0x3d878f40, 1, 0x80000000, 0, 0x100 }, { 0x0, 0x3f7ffffa, 0x3f000003, 0x3ef794dc, 73, 0x00000002, 5, 0x105 }, { 0x0, 0x3f7ffff9, 0x359f34db, 0x3f79a6dc, 81, 0x00000002, 5, 0x105 }, { 0x0, 0x3f6ba787, 0x3e658a10, 0x3f3c1362, 2, 0x80000000, 6, 0x106 }, { 0x0, 0x3f800000, 0x3eb0bdc8, 0x3f27a11c, 22, 0x80000000, 0, 0x100 }, { 0x0, 0x3f7ffffa, 0x3effffe2, 0x3e4e1b00, 83, 0x00000002, 5, 0x105 }, { 0x0, 0x3f7ffff8, 0x3f78c284, 0x3ce7af42, 64, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7ffffe, 0x3ee053bc, 0x3d7d625e, 102, 0x00000003, 4, 0x104 }, + { 0x0, 0x3f7fffff, 0x3ee053b8, 0x3d7d626e, 70, 0x00000002, 4, 0x104 }, { 0x0, 0x3f800000, 0x3f500c68, 0x3e3fce60, 22, 0x80000000, 0, 0x100 }, { 0x0, 0x3f7ffff9, 0x3f000008, 0x3e5d7235, 89, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f760638, 0x3f30bc73, 0x3e87ee1f, 1, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f760638, 0x3f30bc73, 0x3e87ee1f, 1, 0x00000000, 1, 0x101 }, { 0x0, 0x3f7ffff9, 0x3f000007, 0x3ea7d5e3, 73, 0x00000002, 5, 0x105 }, { 0x0, 0x3f800000, 0x3f000000, 0x3e4f6ec0, 27, 0x80000000, 0, 0x100 }, { 0x0, 0x3f7ffff9, 0x3d8967f0, 0x3edda620, 10, 0x00000000, 5, 0x105 }, { 0x0, 0x3f7ffff9, 0x3f000010, 0x3dee19ee, 73, 0x00000002, 5, 0x105 }, { 0x0, 0x3f6052c3, 0x3f3318d4, 0x3dc04463, 1, 0x80000000, 6, 0x106 }, { 0x0, 0x3f800000, 0x3f000000, 0x3e89b960, 23, 0x80000000, 0, 0x100 }, - { 0x0, 0x3f7ffffd, 0x3f000004, 0x3d11e150, 111, 0x00000003, 4, 0x104 }, + { 0x0, 0x3f800000, 0x3eedc3d8, 0x3f091e14, 14, 0x00000000, 0, 0x100 }, { 0x0, 0x3f64fc82, 0x3bc646d0, 0x3f4c7ef1, 5, 0x00000000, 6, 0x106 }, { 0x0, 0x3f7ffff8, 0x35593b39, 0x3dd898f2, 73, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f711e1a, 0x3e072700, 0x3d39ae65, 0, 0x80000000, 3, 0x103 }, - { 0x0, 0x3f7ffff8, 0x35400002, 0x3eb95dc0, 103, 0x00000003, 5, 0x105 }, + { 0x0, 0x3f711e1a, 0x3e072700, 0x3d39ae65, 0, 0x80000000, 1, 0x101 }, + { 0x0, 0x3f7ffff9, 0x3528d448, 0x3eb95dc0, 71, 0x00000002, 5, 0x105 }, { 0x0, 0x3f7ffff9, 0x3efffff8, 0x3de88e01, 89, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7ffff4, 0x3effffb8, 0x3effd986, 127, 0x00000003, 5, 0x105 }, - { 0x0, 0x3f7ffffe, 0x3d80a2a0, 0x3edfd758, 98, 0x80000003, 0, 0x100 }, + { 0x0, 0x3f7ffff7, 0x3effffc8, 0x3effd970, 95, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f800000, 0x3d80a2c0, 0x3edfd750, 2, 0x80000000, 0, 0x100 }, { 0x0, 0x3f5cae6d, 0x3bd39700, 0x3f4d2acb, 0, 0x80000000, 6, 0x106 }, { 0x0, 0x3f7ffffa, 0x3f17030b, 0x3ed1f9e6, 84, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7ffff7, 0x3ec09e8d, 0x3f1fb0b7, 116, 0x00000003, 5, 0x105 }, - { 0x0, 0x3f7a839a, 0x3e549290, 0x3e4e945d, 2, 0x80000000, 3, 0x103 }, - { 0x0, 0x3f76e818, 0x3f26eea0, 0x3e667111, 5, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f7ffff8, 0x3ec09e8e, 0x3f1fb0b7, 84, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f7a839a, 0x3e549290, 0x3e4e945d, 2, 0x80000000, 1, 0x101 }, + { 0x0, 0x3f76e818, 0x3f26eea0, 0x3e667111, 5, 0x00000000, 1, 0x101 }, { 0x0, 0x3f7ffff9, 0x3db5e363, 0x3ed2872d, 80, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f681e3d, 0x3e549f8d, 0x3f3039b5, 2, 0x00000000, 3, 0x103 }, - { 0x0, 0x3f7ffffc, 0x3f000002, 0x3e71bbd9, 97, 0x80000003, 4, 0x104 }, - { 0x0, 0x3f7ffff6, 0x3f00001a, 0x3efda1b0, 103, 0x00000003, 5, 0x105 }, + { 0x0, 0x3f681e3d, 0x3e549f8d, 0x3f3039b5, 2, 0x00000000, 1, 0x101 }, + { 0x0, 0x3f800000, 0x3e8721f8, 0x3f3c6f04, 0, 0x80000000, 0, 0x100 }, + { 0x0, 0x3f7ffff8, 0x3f000016, 0x3efda1c0, 71, 0x00000002, 5, 0x105 }, { 0x0, 0x3f7ffffa, 0x3f792cae, 0x3cda6900, 2, 0x00000000, 5, 0x105 }, { 0x0, 0x3f7ffff8, 0x3ee56754, 0x3d54c4c7, 64, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f6c4aea, 0x3dc95bac, 0x3f42ada5, 7, 0x80000000, 3, 0x103 }, + { 0x0, 0x3f6c4aea, 0x3dc95bac, 0x3f42ada5, 7, 0x80000000, 1, 0x101 }, { 0x0, 0x3f7ffff9, 0x3effffe0, 0x3d8aba81, 73, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f79d359, 0x3daefdf9, 0x3f0be1b6, 0, 0x00000000, 3, 0x103 }, - { 0x0, 0x3f7ffff6, 0x35100001, 0x3f0311eb, 103, 0x00000003, 5, 0x105 }, + { 0x0, 0x3f79d359, 0x3daefdf9, 0x3f0be1b6, 0, 0x00000000, 1, 0x101 }, + { 0x0, 0x3f7ffff7, 0x34ff3b86, 0x3f0311e8, 71, 0x00000002, 5, 0x105 }, { 0x0, 0x3f52c715, 0x3e0ee67a, 0x3be60824, 9, 0x80000001, 6, 0x106 }, - { 0x0, 0x3f7a6dcc, 0x3e9dc06c, 0x3e9eb79b, 4, 0x00000000, 3, 0x103 }, - { 0x0, 0x3f7ffff8, 0x3f141d4a, 0x3ed7c558, 124, 0x00000003, 5, 0x105 }, + { 0x0, 0x3f7a6dcc, 0x3e9dc06c, 0x3e9eb79b, 4, 0x00000000, 1, 0x101 }, + { 0x0, 0x3f7ffff9, 0x3f141d4b, 0x3ed7c55a, 28, 0x00000000, 5, 0x105 }, { 0x0, 0x3f7ffffe, 0x3f000000, 0x3dc5f27d, 71, 0x00000002, 4, 0x104 }, - { 0x0, 0x3f7ffffd, 0x3e4fea82, 0x3e980ac5, 104, 0x80000003, 4, 0x104 }, + { 0x0, 0x3f7fffff, 0x3f33fa9c, 0x3e980ac8, 8, 0x80000000, 0, 0x100 }, { 0x0, 0x3f7ffffa, 0x34b55665, 0x3eaa99c0, 93, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7ffffd, 0x3f000006, 0x3eb315b8, 111, 0x00000003, 4, 0x104 }, + { 0x0, 0x3f7fffff, 0x3f000001, 0x3eb315ca, 79, 0x00000002, 4, 0x104 }, { 0x0, 0x3f800000, 0x80000000, 0x3eebaa80, 25, 0x00000000, 0, 0x100 }, - { 0x0, 0x3f7fffff, 0x3e6973a4, 0x3effffff, 96, 0x80000003, 2, 0x102 }, + { 0x0, 0x3f800000, 0x3efffffe, 0x3e69739a, 7, 0x80000000, 0, 0x100 }, { 0x0, 0x3f7ffff9, 0x3f000003, 0x3e611d40, 65, 0x00000002, 5, 0x105 }, { 0x0, 0x3f70d836, 0x3ed96353, 0x3e57f246, 5, 0x80000000, 6, 0x106 }, { 0x0, 0x3f7ffff8, 0x3e374280, 0x3ea45ea0, 26, 0x00000000, 5, 0x105 }, - { 0x0, 0x3f7fffff, 0x3ef7f04a, 0x338407da, 114, 0x80000003, 2, 0x102 }, + { 0x0, 0x3f800000, 0x3f0407d8, 0x3ef7f050, 10, 0x80000000, 0, 0x100 }, { 0x0, 0x3f7ffffa, 0x3efffff1, 0x3e944807, 89, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7ffff8, 0x3d35a890, 0x3ee94adc, 126, 0x00000003, 5, 0x105 }, + { 0x0, 0x3f7ffffa, 0x3d35a8a0, 0x3ee94ae0, 30, 0x00000000, 5, 0x105 }, { 0x0, 0x3f7ffff7, 0x3effffda, 0x3e582790, 93, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f719475, 0x3e7218da, 0x3eaa6b9e, 3, 0x80000000, 3, 0x103 }, - { 0x0, 0x3f6494a2, 0x3f1fb7ed, 0x3ea1fde7, 6, 0x00000000, 3, 0x103 }, - { 0x0, 0x3f6d25ea, 0x3f445833, 0x3e11348c, 2, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f719475, 0x3e7218da, 0x3eaa6b9e, 3, 0x80000000, 1, 0x101 }, + { 0x0, 0x3f6494a2, 0x3f1fb7ed, 0x3ea1fde7, 6, 0x00000000, 1, 0x101 }, + { 0x0, 0x3f6d25ea, 0x3f445833, 0x3e11348c, 2, 0x00000000, 1, 0x101 }, { 0x0, 0x3f7ffff8, 0x3efffffd, 0x3e5c5ea8, 71, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7ffffe, 0x3f000002, 0x3d7f0106, 121, 0x80000003, 0, 0x100 }, + { 0x0, 0x3f800000, 0x3efffffe, 0x3d7f0190, 25, 0x80000000, 0, 0x100 }, { 0x0, 0x3f7ffff9, 0x344380da, 0x3f63f931, 73, 0x00000002, 5, 0x105 }, { 0x0, 0x3f7ffffa, 0x3f4152a4, 0x3e7ab55f, 90, 0x00000002, 5, 0x105 }, { 0x0, 0x3f7ffff7, 0x3effffe3, 0x3e852084, 93, 0x00000002, 5, 0x105 }, @@ -395,128 +395,128 @@ static const uint32_t kGolden[2][256][8] = { { 0x0, 0x3f5c2c0e, 0x3e2ef1fa, 0x3f3d53a4, 1, 0x80000000, 6, 0x106 }, { 0x0, 0x3f7ffff8, 0x3ef77fe1, 0x3f04400c, 90, 0x00000002, 5, 0x105 }, { 0x0, 0x3f800000, 0x3ebe78b8, 0x3e030e90, 30, 0x80000000, 0, 0x100 }, - { 0x0, 0x3f6d1af5, 0x3e33a054, 0x3f33789d, 4, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f6d1af5, 0x3e33a054, 0x3f33789d, 4, 0x00000000, 1, 0x101 }, { 0x0, 0x3f7ffffa, 0x34c00000, 0x3e1a338f, 69, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f66463e, 0x3f1cf9a6, 0x3ea4a508, 0, 0x00000000, 3, 0x103 }, - { 0x0, 0x3f6d250b, 0x3db836a6, 0x3f67868d, 5, 0x80000000, 3, 0x103 }, + { 0x0, 0x3f66463e, 0x3f1cf9a6, 0x3ea4a508, 0, 0x00000000, 1, 0x101 }, + { 0x0, 0x3f6d250b, 0x3db836a6, 0x3f67868d, 5, 0x80000000, 1, 0x101 }, { 0x0, 0x3f7ffff9, 0x3e3f4086, 0x3ea05f96, 88, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f79948f, 0x3f23bab4, 0x3e656df9, 1, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f79948f, 0x3f23bab4, 0x3e656df9, 1, 0x00000000, 1, 0x101 }, { 0x0, 0x3f7ffff7, 0x3e60fbae, 0x3f47c0fa, 66, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7fffff, 0x3f515954, 0x80000000, 98, 0x80000003, 2, 0x102 }, + { 0x0, 0x3f800000, 0x3e3a9ab2, 0x3f515954, 14, 0x80000000, 0, 0x100 }, { 0x0, 0x3f7ffff8, 0x3ed3c057, 0x3db0feed, 72, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7ffffe, 0x3edc8264, 0x3d8df67a, 102, 0x00000003, 4, 0x104 }, - { 0x0, 0x3f6fc2c9, 0x3f587690, 0x3e075479, 7, 0x00000000, 3, 0x103 }, - { 0x0, 0x3f7ffffe, 0x3e9236bf, 0x3f000004, 104, 0x80000003, 2, 0x102 }, - { 0x0, 0x3f7b01d8, 0x3ecf4c13, 0x3e5732ce, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f7fffff, 0x3edc8262, 0x3d8df67f, 70, 0x00000002, 4, 0x104 }, + { 0x0, 0x3f6fc2c9, 0x3f587690, 0x3e075479, 7, 0x00000000, 1, 0x101 }, + { 0x0, 0x3f800000, 0x3efffffe, 0x3e9236ca, 5, 0x80000000, 0, 0x100 }, + { 0x0, 0x3f7b01d8, 0x3ecf4c13, 0x3e5732ce, 0, 0x00000000, 1, 0x101 }, { 0x0, 0x3f7ffffa, 0x3ed963a9, 0x3d9a71ca, 68, 0x00000002, 5, 0x105 }, { 0x0, 0x3f7ffff9, 0x3f4cfd30, 0x3e4c0af0, 24, 0x00000000, 5, 0x105 }, - { 0x0, 0x3f7ffffd, 0x3dc53d51, 0x34b3ac29, 112, 0x80000003, 2, 0x102 }, + { 0x0, 0x3f800000, 0x3f675854, 0x3dc53d60, 2, 0x80000000, 0, 0x100 }, { 0x0, 0x3f7ffff8, 0x3c88b1f1, 0x3f7bba6f, 88, 0x00000002, 5, 0x105 }, { 0x0, 0x3f7ffff8, 0x345d935e, 0x3eec9aed, 69, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f763bbb, 0x3eb37160, 0x3eab065a, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f763bbb, 0x3eb37160, 0x3eab065a, 0, 0x00000000, 1, 0x101 }, { 0x0, 0x3f7fffff, 0x3f000000, 0x3ea346fc, 31, 0x80000000, 0, 0x100 }, - { 0x0, 0x3f7b5ec8, 0x3f058734, 0x3da57d7b, 0, 0x00000000, 3, 0x103 }, - { 0x0, 0x3f7ffff8, 0x33debb08, 0x3f75d83c, 127, 0x00000003, 5, 0x105 }, - { 0x0, 0x3f78d7b2, 0x3e9dcb1f, 0x3eaf3a81, 4, 0x00000000, 3, 0x103 }, - { 0x0, 0x3f7ffffd, 0x34fffffc, 0x3e976a28, 97, 0x80000003, 0, 0x100 }, + { 0x0, 0x3f7b5ec8, 0x3f058734, 0x3da57d7b, 0, 0x00000000, 1, 0x101 }, + { 0x0, 0x3f7ffff9, 0x33bebb08, 0x3f75d83c, 95, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f78d7b2, 0x3e9dcb1f, 0x3eaf3a81, 4, 0x00000000, 1, 0x101 }, + { 0x0, 0x3f800000, 0x80000000, 0x3e976a48, 1, 0x80000000, 0, 0x100 }, { 0x0, 0x3f7ffff9, 0x3eb63614, 0x3e1393eb, 66, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7ffffd, 0x3e965965, 0x3e534d4c, 126, 0x00000003, 4, 0x104 }, + { 0x0, 0x3f800000, 0x3f4b2ca4, 0x3e534d70, 30, 0x00000000, 0, 0x100 }, { 0x0, 0x3f7ffff8, 0x3e06ad8a, 0x3ebca923, 88, 0x00000002, 5, 0x105 }, { 0x0, 0x3f800000, 0x3f6bac06, 0x3da29fd0, 14, 0x80000000, 0, 0x100 }, - { 0x0, 0x3f66a751, 0x3e0696c8, 0x3f5b5f23, 0, 0x00000000, 3, 0x103 }, - { 0x0, 0x3f7ffffe, 0x3f000002, 0x3eb07957, 111, 0x00000003, 4, 0x104 }, + { 0x0, 0x3f66a751, 0x3e0696c8, 0x3f5b5f23, 0, 0x00000000, 1, 0x101 }, + { 0x0, 0x3f7fffff, 0x3f000000, 0x3eb0795d, 79, 0x00000002, 4, 0x104 }, { 0x0, 0x3f7ffff9, 0x3eba09a4, 0x3e0bec7a, 72, 0x00000002, 5, 0x105 }, { 0x0, 0x3f7fffff, 0x3dc33dd0, 0x3f679846, 28, 0x80000000, 0, 0x100 }, { 0x0, 0x3f7ffffa, 0x35000000, 0x3eed9814, 1, 0x00000000, 5, 0x105 }, { 0x0, 0x3f7ffff7, 0x3e9b0a14, 0x3f327af5, 88, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7ffff7, 0x35a00003, 0x3f188ac0, 119, 0x00000003, 5, 0x105 }, + { 0x0, 0x3f7ffff8, 0x35800000, 0x3f188ac8, 23, 0x00000000, 5, 0x105 }, { 0x0, 0x3f77a9cc, 0x3c9c7f58, 0x3ed2f161, 0, 0x80000000, 6, 0x106 }, - { 0x0, 0x3f7ffff7, 0x35819393, 0x3dc9c931, 127, 0x00000003, 5, 0x105 }, + { 0x0, 0x3f7ffff9, 0x35400000, 0x3dc9c940, 31, 0x00000000, 5, 0x105 }, { 0x0, 0x3f7ffff8, 0x3599f5c4, 0x3f4fae1d, 81, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7ffff8, 0x35600002, 0x3f1d128e, 111, 0x00000003, 5, 0x105 }, + { 0x0, 0x3f7ffff9, 0x354744a4, 0x3f1d1291, 79, 0x00000002, 5, 0x105 }, { 0x0, 0x3f800000, 0x3ec910f0, 0x3ddbbc40, 30, 0x80000000, 0, 0x100 }, { 0x0, 0x3f7ffffa, 0x35200000, 0x3ccd882c, 27, 0x00000000, 5, 0x105 }, - { 0x0, 0x3f7ffffe, 0x3e9a0220, 0x3e4bfbd0, 102, 0x00000003, 4, 0x104 }, - { 0x0, 0x3f7ffffd, 0x3f000003, 0x3da72fcc, 111, 0x00000003, 4, 0x104 }, + { 0x0, 0x3f7fffff, 0x3e9a021b, 0x3e4bfbd5, 70, 0x00000002, 4, 0x104 }, + { 0x0, 0x3f800000, 0x3ed63408, 0x3f14e5fc, 14, 0x00000000, 0, 0x100 }, { 0x0, 0x3f800000, 0x3f2547e0, 0x3eb57040, 22, 0x80000000, 0, 0x100 }, - { 0x0, 0x3f7ffffe, 0x3f00000a, 0x3edd2322, 111, 0x00000003, 4, 0x104 }, + { 0x0, 0x3f800000, 0x3d8b7340, 0x3f6e9198, 14, 0x00000000, 0, 0x100 }, { 0x0, 0x3f5a6702, 0x3c4d5104, 0x3f6cfda3, 1, 0x80000000, 6, 0x106 }, - { 0x0, 0x3f64eaef, 0x3d8ea9a7, 0x3f640db7, 6, 0x00000000, 3, 0x103 }, - { 0x0, 0x3f7ffffd, 0x3efffff6, 0x3d944b98, 113, 0x80000003, 0, 0x100 }, + { 0x0, 0x3f64eaef, 0x3d8ea9a7, 0x3f640db7, 6, 0x00000000, 1, 0x101 }, + { 0x0, 0x3f800000, 0x3efffffc, 0x3d944b70, 17, 0x80000000, 0, 0x100 }, { 0x0, 0x3f7ffffa, 0x3cadbb04, 0x3ef52438, 92, 0x00000002, 5, 0x105 }, { 0x0, 0x3f7ffff9, 0x3d556639, 0x3ee55350, 88, 0x00000002, 5, 0x105 }, { 0x0, 0x3f7ffff8, 0x3effffda, 0x3e9cec8b, 83, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7ffffd, 0x3dbac67d, 0x3ed14e55, 97, 0x80000003, 2, 0x102 }, + { 0x0, 0x3f800000, 0x3ed14e78, 0x3dbac620, 6, 0x80000000, 0, 0x100 }, { 0x0, 0x3f7ffff9, 0x34de28f9, 0x3ef147c6, 85, 0x00000002, 5, 0x105 }, { 0x0, 0x3f738af7, 0x3e740404, 0x3f1a12dd, 2, 0x80000000, 6, 0x106 }, { 0x0, 0x3f7ffffa, 0x3f1b3c9e, 0x3ec986bc, 84, 0x00000002, 5, 0x105 }, { 0x0, 0x3f800000, 0x3f01eb89, 0x3efc28ee, 26, 0x80000000, 0, 0x100 }, - { 0x0, 0x3f695785, 0x3e11de66, 0x3f3e2607, 2, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f695785, 0x3e11de66, 0x3f3e2607, 2, 0x00000000, 1, 0x101 }, { 0x0, 0x3f6bfc68, 0x3de2b043, 0x3f15b4a2, 5, 0x00000000, 6, 0x106 }, - { 0x0, 0x3f7ffff9, 0x3f4163b8, 0x3e7a70ce, 108, 0x00000003, 5, 0x105 }, + { 0x0, 0x3f7ffffa, 0x3f4163bc, 0x3e7a70cb, 76, 0x00000002, 5, 0x105 }, { 0x0, 0x3f4ff44b, 0x3e293406, 0x3ec7959a, 1, 0x00000000, 6, 0x106 }, - { 0x0, 0x3f532d8b, 0x3f196c52, 0x3eb804a3, 1, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f532d8b, 0x3f196c52, 0x3eb804a3, 1, 0x00000000, 1, 0x101 }, { 0x0, 0x3f7cdcd6, 0x3e64bb34, 0x3f33fe34, 66, 0x00000002, 5, 0x105 }, { 0x0, 0x3f7e2de1, 0x3ef9c2c5, 0x3ef063c4, 90, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f531696, 0x3f1291a6, 0x3cc460ca, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f531696, 0x3f1291a6, 0x3cc460ca, 0, 0x00000000, 1, 0x101 }, { 0x0, 0x3f7eac1d, 0x3f53136e, 0x3e13d4fd, 90, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f535e35, 0x3ea5d7ee, 0x3e900b53, 0, 0x00000000, 3, 0x103 }, - { 0x0, 0x3f5373f3, 0x3ea61344, 0x3e912be7, 0, 0x00000000, 3, 0x103 }, - { 0x0, 0x3f53b04a, 0x3e4a6ad8, 0x3ed5cf27, 0, 0x00000000, 3, 0x103 }, - { 0x0, 0x3f537b21, 0x3f1bd906, 0x3ebe90d2, 1, 0x00000000, 3, 0x103 }, - { 0x0, 0x3f537ef1, 0x3f1bf784, 0x3e298ebf, 1, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f535e35, 0x3ea5d7ee, 0x3e900b53, 0, 0x00000000, 1, 0x101 }, + { 0x0, 0x3f5373f3, 0x3ea61344, 0x3e912be7, 0, 0x00000000, 1, 0x101 }, + { 0x0, 0x3f53b04a, 0x3e4a6ad8, 0x3ed5cf27, 0, 0x00000000, 1, 0x101 }, + { 0x0, 0x3f537b21, 0x3f1bd906, 0x3ebe90d2, 1, 0x00000000, 1, 0x101 }, + { 0x0, 0x3f537ef1, 0x3f1bf784, 0x3e298ebf, 1, 0x00000000, 1, 0x101 }, { 0x0, 0x3f8086e6, 0x3cca58cb, 0x3f02df2f, 93, 0x00000002, 5, 0x105 }, { 0x0, 0x3f7d37fe, 0x3ed975f9, 0x3f0294f7, 90, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f532e5c, 0x3e4ad678, 0x3ecd7a7d, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f532e5c, 0x3e4ad678, 0x3ecd7a7d, 0, 0x00000000, 1, 0x101 }, { 0x0, 0x3f817b05, 0x3d8e21db, 0x3eae59d9, 93, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f53644a, 0x3f1b2249, 0x3e6c1301, 1, 0x00000000, 3, 0x103 }, - { 0x0, 0x3f536ada, 0x3e6d9970, 0x3ebfe0e0, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f53644a, 0x3f1b2249, 0x3e6c1301, 1, 0x00000000, 1, 0x101 }, + { 0x0, 0x3f536ada, 0x3e6d9970, 0x3ebfe0e0, 0, 0x00000000, 1, 0x101 }, { 0x0, 0x3f7d1615, 0x3f643c3c, 0x3d24840f, 90, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f80641f, 0x3c962e49, 0x3f36282b, 125, 0x00000003, 5, 0x105 }, + { 0x0, 0x3f806420, 0x3c962fe7, 0x3f36281e, 93, 0x00000002, 5, 0x105 }, { 0x0, 0x3f7cfc7f, 0x3f34250b, 0x3e6717bc, 90, 0x00000002, 5, 0x105 }, { 0x0, 0x3f7e0010, 0x3ee8bc6a, 0x3eff4456, 66, 0x00000002, 5, 0x105 }, { 0x0, 0x3f811cfb, 0x3d55bc6f, 0x3dbe83d4, 93, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f53b277, 0x3f089ff1, 0x3da79e02, 0, 0x00000000, 3, 0x103 }, - { 0x0, 0x3f534ae7, 0x3eeb3474, 0x3e12f3ee, 0, 0x00000000, 3, 0x103 }, - { 0x0, 0x3f537348, 0x3cf986e6, 0x3f13ce02, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f53b277, 0x3f089ff1, 0x3da79e02, 0, 0x00000000, 1, 0x101 }, + { 0x0, 0x3f534ae7, 0x3eeb3474, 0x3e12f3ee, 0, 0x00000000, 1, 0x101 }, + { 0x0, 0x3f537348, 0x3cf986e6, 0x3f13ce02, 0, 0x00000000, 1, 0x101 }, { 0x0, 0x3f4b6fe1, 0x3cc535ac, 0x3f081940, 1, 0x00000000, 6, 0x106 }, { 0x0, 0x3f80084c, 0x3ac71c69, 0x3db9cd41, 69, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7edddd, 0x3e77aab9, 0x3f3b487b, 122, 0x00000003, 5, 0x105 }, - { 0x0, 0x3f536fc6, 0x3c0b33db, 0x3f195160, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f7edddf, 0x3e77aaee, 0x3f3b487a, 90, 0x00000002, 5, 0x105 }, + { 0x0, 0x3f536fc6, 0x3c0b33db, 0x3f195160, 0, 0x00000000, 1, 0x101 }, { 0x0, 0x3f80bcc7, 0x3d0d94d9, 0x3e01d49c, 69, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f5334eb, 0x3f099f0e, 0x3d80423a, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f5334eb, 0x3f099f0e, 0x3d80423a, 0, 0x00000000, 1, 0x101 }, { 0x0, 0x3f7cabe8, 0x3e41b8e7, 0x3f3b9934, 66, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f532ef9, 0x3f1977c6, 0x3e85db3e, 1, 0x00000000, 3, 0x103 }, - { 0x0, 0x3f535503, 0x3cc2bcc1, 0x3f14922d, 0, 0x00000000, 3, 0x103 }, - { 0x0, 0x3f538608, 0x3ecabf0b, 0x3e5b42e7, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f532ef9, 0x3f1977c6, 0x3e85db3e, 1, 0x00000000, 1, 0x101 }, + { 0x0, 0x3f535503, 0x3cc2bcc1, 0x3f14922d, 0, 0x00000000, 1, 0x101 }, + { 0x0, 0x3f538608, 0x3ecabf0b, 0x3e5b42e7, 0, 0x00000000, 1, 0x101 }, { 0x0, 0x3f80a973, 0x3cfe2ba5, 0x3ef96bd2, 93, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f535d6c, 0x3f03219a, 0x3dbe4dfd, 0, 0x00000000, 3, 0x103 }, - { 0x0, 0x3f5343a5, 0x3ea800d4, 0x3e8c3979, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f535d6c, 0x3f03219a, 0x3dbe4dfd, 0, 0x00000000, 1, 0x101 }, + { 0x0, 0x3f5343a5, 0x3ea800d4, 0x3e8c3979, 0, 0x00000000, 1, 0x101 }, { 0x0, 0x3f81d8a8, 0x3d8fb50c, 0x3c8627a9, 92, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f5369f1, 0x3e00b76b, 0x3ef64354, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f5369f1, 0x3e00b76b, 0x3ef64354, 0, 0x00000000, 1, 0x101 }, { 0x0, 0x3f80688e, 0x3c9cd3f8, 0x3f75bcda, 69, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f5355c1, 0x3ed87cdc, 0x3e39be66, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f5355c1, 0x3ed87cdc, 0x3e39be66, 0, 0x00000000, 1, 0x101 }, { 0x0, 0x3f7e3aa2, 0x3f735972, 0x3c019519, 90, 0x00000002, 5, 0x105 }, { 0x0, 0x3f7f9177, 0x3f1a20d9, 0x3ec68fde, 90, 0x00000002, 5, 0x105 }, { 0x0, 0x3f4facc6, 0x3e20435d, 0x3ecd6d43, 1, 0x00000000, 6, 0x106 }, - { 0x0, 0x3f532fe7, 0x3de1b188, 0x3efa920c, 0, 0x00000000, 3, 0x103 }, - { 0x0, 0x3f539dbd, 0x3f1cede8, 0x3e1d79f8, 1, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f532fe7, 0x3de1b188, 0x3efa920c, 0, 0x00000000, 1, 0x101 }, + { 0x0, 0x3f539dbd, 0x3f1cede8, 0x3e1d79f8, 1, 0x00000000, 1, 0x101 }, { 0x0, 0x3f7f6244, 0x3f48709f, 0x3e4f73d8, 90, 0x00000002, 5, 0x105 }, { 0x0, 0x3f809a25, 0x3ce73741, 0x3f00af59, 69, 0x00000002, 5, 0x105 }, { 0x0, 0x3f7d9e68, 0x3f05869f, 0x3ed85fa1, 90, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f53ae22, 0x3e2a8867, 0x3ee59de4, 0, 0x00000000, 3, 0x103 }, - { 0x0, 0x3f538038, 0x3e9d4a69, 0x3e9ab90a, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f53ae22, 0x3e2a8867, 0x3ee59de4, 0, 0x00000000, 1, 0x101 }, + { 0x0, 0x3f538038, 0x3e9d4a69, 0x3e9ab90a, 0, 0x00000000, 1, 0x101 }, { 0x0, 0x3f80e2ce, 0x3d2a1a42, 0x3ef7c7d2, 93, 0x00000002, 5, 0x105 }, { 0x1, 0x00000000, 0x00000000, 0x00000000, 0, 0x00000000, 0, 0x0 }, - { 0x0, 0x3f53901d, 0x3ee09e4c, 0x3e30c6eb, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f53901d, 0x3ee09e4c, 0x3e30c6eb, 0, 0x00000000, 1, 0x101 }, { 0x0, 0x3f80e782, 0x3d2da155, 0x3e649458, 69, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f53777d, 0x3eb4a8ba, 0x3e82cf14, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f53777d, 0x3eb4a8ba, 0x3e82cf14, 0, 0x00000000, 1, 0x101 }, { 0x0, 0x3f805d9e, 0x3c8c6c45, 0x3f03f177, 93, 0x00000002, 5, 0x105 }, { 0x0, 0x3f80f545, 0x3d37f3cd, 0x3e503679, 69, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f538794, 0x3ef8afd2, 0x3dff25a1, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f538794, 0x3ef8afd2, 0x3dff25a1, 0, 0x00000000, 1, 0x101 }, { 0x0, 0x3f7f6e29, 0x3c89b8c4, 0x3f78472b, 66, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f5355c8, 0x3f0d7a08, 0x3d53432d, 0, 0x00000000, 3, 0x103 }, - { 0x0, 0x3f531799, 0x3dc676f8, 0x3effdbcb, 0, 0x00000000, 3, 0x103 }, + { 0x0, 0x3f5355c8, 0x3f0d7a08, 0x3d53432d, 0, 0x00000000, 1, 0x101 }, + { 0x0, 0x3f531799, 0x3dc676f8, 0x3effdbcb, 0, 0x00000000, 1, 0x101 }, { 0x0, 0x3f7df3f6, 0x3eb0e7aa, 0x3f1b43ee, 66, 0x00000002, 5, 0x105 }, }, }; diff --git a/tests/raytracing/rt_smoke_tie/main.cpp b/tests/raytracing/rt_smoke_tie/main.cpp index 4cf4e6e5cd..6c889a1c98 100644 --- a/tests/raytracing/rt_smoke_tie/main.cpp +++ b/tests/raytracing/rt_smoke_tie/main.cpp @@ -15,9 +15,10 @@ // hits the RTU commits. // // The scene stacks coincident and near-coincident geometry, within one BLAS -// (exact twin triangles, a copy a few ulps off the plane, a copy tilted by a -// hair) and across instances (the same BLAS instanced twice in place, rotated -// a quarter turn onto itself, shifted by half a cell, lifted by 2^-20). The +// (exact twin triangles, a copy an ulp-scale step off the plane) and across +// instances (a BLAS instanced twice in place, rotated a quarter turn onto +// itself, shifted by half a cell, lifted by 2^-20, and a second BLAS holding +// copies of two of the first one's triangles). The // RTU walks its own CW-BVH4; which of those hits it keeps is settled by the // source BVH's visit order, carried as the visit-order tables the Vulkan // driver appends (vortexpipe vp_launch.c): a TLAS table inside the scene and @@ -299,9 +300,8 @@ std::vector insts; void make_geometry() { // BLAS 0, the "sail": a 4x4 grid of quads on z = 0 over [-1, 1]^2 (geometry - // 0); an exact twin of it (geometry 1); a copy a few ulps above the plane on - // some vertices (geometry 2); a copy tilted by a hair, crossing the plane - // along x = 0 (geometry 3). + // 0); an exact twin of it (geometry 1); a copy an ulp-scale step above the + // plane on some vertices (geometry 2). std::vector sail; auto quad = [&](float x0, float y0, float x1, float y1, uint32_t geom, auto zf) { Tri a = { { x0, y0, zf(x0, y0), x1, y0, zf(x1, y0), x1, y1, zf(x1, y1) }, geom }; @@ -309,7 +309,7 @@ void make_geometry() { sail.push_back(a); sail.push_back(b); }; - for (uint32_t g = 0; g < 4; ++g) { + for (uint32_t g = 0; g < 3; ++g) { for (int j = 0; j < 4; ++j) { for (int i = 0; i < 4; ++i) { const float x0 = -1.f + 0.5f * i, y0 = -1.f + 0.5f * j; @@ -318,7 +318,6 @@ void make_geometry() { switch (g) { // every other grid vertex: about an ulp of t for these rays case 2: return (int((x + 1.f) * 4.f + (y + 1.f) * 4.f) & 2) ? 0x1p-22f : 0.f; - case 3: return x * 0x1p-21f + y * 0x1p-23f; default: return 0.f; } }; @@ -366,7 +365,7 @@ void make_geometry() { const float P[12] = { 1, 0, 0, 0.125f, 0, 0, -1, 0, 0, 1, 0, 0 }; const float Pi[12] = { 1, 0, 0, -0.125f, 0, 0, 1, 0, 0, -1, 0, 0 }; inst(0, 0, I, I); - inst(0, 1, I, I); + inst(1, 1, I, I); inst(0, 2, R, Ri); inst(1, 3, I, I); inst(0, 4, T, Ti); From 85c24f6cf53cf3becaa3bb6e205097e3479ca36e Mon Sep 17 00:00:00 2001 From: Blaise Tine Date: Sat, 3 Oct 2026 09:45:35 -0700 Subject: [PATCH 16/31] rtu: RTL short-stack overflow never loses a hit The RTL dropped the children that did not fit the short stack and, once the stack emptied, re-descended the BLAS or the TLAS pruned by best_t, at most 8 times. A re-descent repeats the same nearest-first walk, so it overflows and drops the same children again unless a hit happened to cull them: past the cap the walk ended with those subtrees unvisited. On PARK_PT (deep BLASes, 16-entry stack) a ray from a surface missed its hit at t=0.005 and reported one at t=0.85 (replay ray 657); a 64-entry stack passed, which pinned it. The walk visits nodes in rank-path order: each level's rank in the t-sorted child list, an instance's index in its leaf. A child that does not fit is dropped and recorded (level, rank); the walk then only descends (the stack stays full) and, when it would pop, restarts from the root following the dropped child's rank path, read back from a per-context path RAM that the walk writes on every descent and pop. Every node before that child has been visited; a tightened best_t only culls a suffix of a node's sorted children, so recorded ranks stay valid; each restart starts strictly further along, so the walk is exact and terminates. Stack entries carry their level and rank; the restart cap and the BLAS-only re-descent are gone. rt_smoke_deep_stack gains a decoy scene (boxes walked first whose triangles miss; the one hit sits in dropped subtrees): the old RTL returns a miss, the new one the hit. Co-Authored-By: Claude Opus 5.5 --- docs/designs/ray_tracing_architecture.md | 12 +- hw/rtl/rtu/VX_rtu_scheduler.sv | 215 +++++++++++++----- tests/raytracing/rt_smoke_deep_stack/main.cpp | 163 ++++++------- 3 files changed, 254 insertions(+), 136 deletions(-) diff --git a/docs/designs/ray_tracing_architecture.md b/docs/designs/ray_tracing_architecture.md index e1590aae68..ec5bd26638 100644 --- a/docs/designs/ray_tracing_architecture.md +++ b/docs/designs/ray_tracing_architecture.md @@ -299,9 +299,15 @@ Two work products leave the scheduler: returned `hitAttribute`). Robustness details worth naming: a short-stack of depth `RTU_STACK_DEPTH` bounds -per-context node stack RAM; on overflow the walker sets an `ovf` flag and, at -pop-time, **re-descends** the subtree pruned by the tightened `best_t` (bounded by -`RTU_RESTART_CAP = 8` restarts) — a full traversal on a finite stack. A 16-entry +per-context node stack RAM, and an overflow never loses a hit. The walk visits +nodes in rank-path order (each level's child rank in the t-sorted list, an +instance's index in its leaf); a child that does not fit is dropped and the walk +records it, then only descends until it would pop, and instead **restarts** from +the root along the dropped child's rank path, held in a small per-context path +RAM. Everything before that child has been visited, a tightened `best_t` only +culls a suffix of a node's sorted children (so ranks stay valid), and each +restart starts strictly further along, so the walk is exact and terminates on +any tree up to 63 levels deep. A deep tree costs restarts, not hits. A 16-entry box collector insertion-sorts a node's child hits t-ascending so descent is nearest-first. The insertion slot is decoded from the **admit thermometer**: the collected list is sorted and its count mask is a prefix, so the "entries at or diff --git a/hw/rtl/rtu/VX_rtu_scheduler.sv b/hw/rtl/rtu/VX_rtu_scheduler.sv index cd4a070294..451356f48b 100644 --- a/hw/rtl/rtu/VX_rtu_scheduler.sv +++ b/hw/rtl/rtu/VX_rtu_scheduler.sv @@ -134,10 +134,23 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( localparam RECIP_LAT = (`VX_CFG_RTU_RECIP_DSP_SEED != 0) ? 5 : RTU_FDIV_LAT; - localparam RTU_RESTART_CAP = 8; - localparam RST_CNTW = `CLOG2(RTU_RESTART_CAP + 1); localparam STK_IDXW = `CLOG2(RTU_STACK_DEPTH); + // Short-stack overflow restart. A walk visits the tree depth first, nearer + // child first, so every node has a rank path (its child rank at each level, + // an instance's index within its leaf) and the walk proceeds in ascending + // rank-path order. A child that does not fit on the stack is dropped; the + // walk then only descends (the stack stays full), so when it would next pop + // (a node past the drop) it restarts from the root instead, following the + // smallest dropped rank path: every node before it has been visited. That + // path is the per-level ranks the walk recorded on its way down (unchanged, + // as nothing was popped) plus the dropped child's own rank. A tightened + // best_t only culls a suffix of a node's ordered children, so the ranks stay + // valid across restarts, and each restart starts strictly further along. + localparam LVLW = 6; // tree levels a walk can record + localparam PATHW = 8; // rank at a level (instance index <= 255) + localparam STK_ENTW = 32 + LVLW + RTU_CHILD_BITS; + // box collections in flight; sized so the collector never caps the node // rate the pipelined front end can sustain over the box-PE latency. localparam COLL_SIZE = 16; @@ -235,10 +248,18 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( logic [31:0] inst_cust; logic [7:0] inst_flags; logic [31:0] root_off; - logic ovf_w; - logic ovf_o; - logic [RST_CNTW-1:0] rst_w; - logic [RST_CNTW-1:0] rst_o; + // overflow restart: the current node's level, the instance level of + // the current TLAS leaf, the smallest dropped child (level, rank), and + // the rank path a restarted walk follows down to it + logic [LVLW-1:0] lvl; + logic [LVLW-1:0] ilvl; + logic ovf; + logic [LVLW-1:0] nd_lvl; + logic [RTU_CHILD_BITS-1:0] nd_rank; + logic follow; + logic [LVLW-1:0] trl_lvl; + logic [RTU_CHILD_BITS-1:0] trl_rank; + logic [RTU_CHILD_BITS-1:0] dsc; // the child CS_PUSH descends into logic [2:0][31:0] obj_o; logic [2:0][31:0] obj_d; logic [2:0][31:0] obj_inv_d; @@ -271,6 +292,7 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( reg [NUM_CTX-1:0] objv_q; // candidate came from inside a BLAS reg [NUM_CTX-1:0] attr_q; reg [NUM_CTX-1:0][RTU_STACK_BITS-1:0] sp_q_arr; + reg [NUM_CTX-1:0][LVLW-1:0] lvl_q_arr; reg [NUM_CTX-1:0][LB-1:0] f_slot_q; reg [NUM_CTX-1:0][RTU_CB_ACTION_BITS-1:0] act_q; reg [NUM_CTX-1:0] orc_q; // the oracle holds the context's memory tag @@ -333,7 +355,8 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( ctx_state_t word_q; lane_ray_t ray_q; reg [BUF_BITS-1:0] fbuf_q; - reg [31:0] stacktop_q; + reg [STK_ENTW-1:0] stacktop_q; + reg [PATHW-1:0] pathv_q; // rank path entry below the current node reg [ADDRW-1:0] structaddr_q; reg [RTU_STACK_BITS-1:0] sp_q; reg [15:0] flags_q; @@ -499,12 +522,16 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( wire [95:0] recip_rdata; // ── short stack (BVH only) ──────────────────────────────────────── - wire stk_wr; - wire [31:0] stk_wdata; - wire [31:0] stk_rdata; + wire stk_wr; + wire [STK_ENTW-1:0] stk_wdata; + wire [STK_ENTW-1:0] stk_rdata; + wire path_wr; + wire [LVLW-1:0] path_wlvl; + wire [PATHW-1:0] path_wdata; + wire [PATHW-1:0] path_rdata; if (!FLAT) begin : g_stack VX_dp_ram #( - .DATAW (32), + .DATAW (STK_ENTW), .SIZE (NUM_CTX << STK_IDXW), .OUT_REG (1), .RDW_MODE ("W") @@ -519,9 +546,27 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( .raddr ({g1_idx, STK_IDXW'(sp_q_arr[g1_idx] - RTU_STACK_BITS'(1))}), .rdata (stk_rdata) ); + // the rank recorded at each level of the current path + VX_dp_ram #( + .DATAW (PATHW), + .SIZE (NUM_CTX << LVLW), + .OUT_REG (1), + .RDW_MODE ("W") + ) path_ram ( + .clk (clk), + .reset (reset), + .read (g1_valid), + .write (path_wr), + .wren (1'b1), + .waddr ({sel_q, path_wlvl}), + .wdata (path_wdata), + .raddr ({g1_idx, LVLW'(lvl_q_arr[g1_idx] + LVLW'(1))}), + .rdata (path_rdata) + ); end else begin : g_no_stack - assign stk_rdata = '0; - `UNUSED_VAR ({stk_wr, stk_wdata, sp_q_arr}) + assign stk_rdata = '0; + assign path_rdata = '0; + `UNUSED_VAR ({stk_wr, stk_wdata, sp_q_arr, path_wr, path_wlvl, path_wdata, lvl_q_arr}) end reg [COLL_SIZE-1:0] coll_busy; @@ -600,6 +645,7 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( ray_q <= lane_ray_t'(ray_rdata); fbuf_q <= fbuf; stacktop_q <= stk_rdata; + pathv_q <= path_rdata; // a fresh context's store row is stale and a re-walk restarts: // either walk starts at the scene base (the template's cur_off is 0) structaddr_q <= slot_scene[s1_slot] @@ -801,7 +847,7 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( assign box_tag_pre = '0; `UNUSED_VAR ({box_feed, box_feed_raw, feed_ci, walk_inv_d}) `UNUSED_VAR ({leaf_v0, leaf_v1, leaf_geom, leaf_prim, leaf_flags, leaf_count}) - `UNUSED_VAR ({node, node_kind, node_lines, leaf_lines, stacktop_q}) + `UNUSED_VAR ({node, node_kind, node_lines, leaf_lines, stacktop_q, pathv_q}) end // Row select for the insertion read. It depends only on the tag, which the @@ -1405,8 +1451,12 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( reg orc_start_r; commit_t cf_din_r; reg sp_inc, sp_dec; + reg sp_clr; reg stk_wr_r; - reg [31:0] stk_wdata_r; + reg [STK_ENTW-1:0] stk_wdata_r; + reg path_wr_r; + reg [LVLW-1:0] path_wlvl_r; + reg [PATHW-1:0] path_wdata_r; // the oracle's table fetches go first; a context's fetch retries wire mem_fire = x_valid && mem_issue && mem_req_ready && !orc_mreq_valid; @@ -1419,6 +1469,16 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( wire [RTU_CHILD_BITS-1:0] last_child = node.n_children - RTU_CHILD_BITS'(1); + // overflow restart: the rank a restarted walk takes at the level below + // the current node (the dropped child's own at its level), the instance a + // TLAS leaf resumes at, and the popped stack entry's level and rank + wire [RTU_CHILD_BITS-1:0] follow_rank = + ((word_x.lvl + LVLW'(1)) == word_x.trl_lvl) ? word_x.trl_rank + : RTU_CHILD_BITS'(pathv_q); + wire [31:0] inst_start = word_x.follow ? 32'(pathv_q) : 32'd0; + wire [LVLW-1:0] stk_top_lvl = stacktop_q[STK_ENTW-1 -: LVLW]; + wire [RTU_CHILD_BITS-1:0] stk_top_rank = stacktop_q[32 +: RTU_CHILD_BITS]; + always @(*) begin word_n = word_x; wake_self = 1'b0; @@ -1443,8 +1503,12 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( cf_din_r = '0; sp_inc = 1'b0; sp_dec = 1'b0; + sp_clr = 1'b0; stk_wr_r = 1'b0; stk_wdata_r = '0; + path_wr_r = 1'b0; + path_wlvl_r = '0; + path_wdata_r = '0; cf_din_r.ctx = sel_q; @@ -1598,14 +1662,17 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( end else if (node_kind == RTU_KIND_LEAF_INST && leaf_count != 8'd0) begin // header: the BLAS table, the first instance's TLAS rank, // the TLAS table + // a restart following its path resumes at the recorded instance word_n.blas_tab = leaf_geom; - word_n.iord = leaf_flags; + word_n.iord = leaf_flags + inst_start; word_n.tlas_tab = leaf_prim; word_n.inst_cnt = {24'd0, leaf_count}; - word_n.inst_idx = '0; + word_n.inst_idx = inst_start; word_n.inst_base = word_x.cur_off + 32'(RTU_LEAF_HDR_BYTES); word_n.blas_floor = sp_q; - word_n.cur_off = word_x.cur_off + 32'(RTU_LEAF_HDR_BYTES); + word_n.ilvl = word_x.lvl + LVLW'(1); + word_n.cur_off = word_x.cur_off + 32'(RTU_LEAF_HDR_BYTES) + + (inst_start * 32'(RTU_INST_STRIDE)); word_n.cstate = CS_INST_REQ; wake_self = 1'b1; end else begin @@ -1623,29 +1690,50 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( end end CS_WAIT: begin - // woken by the collector: this node's ordering is complete - word_n.push_ptr = (coll_cnt_q == RTU_CHILD_BITS'(0)) - ? RTU_CHILD_BITS'(0) - : (coll_cnt_q - RTU_CHILD_BITS'(1)); - word_n.cstate = CS_PUSH; - wake_self = 1'b1; + // woken by the collector: this node's ordering is complete. The + // walk descends into the nearest child, or, restarting, into the + // child its path names; nearer children were visited before. + if (word_x.follow && (coll_cnt_q <= follow_rank)) begin + // that child (and every farther one) is culled by now + word_n.follow = 1'b0; + coll_free_r = 1'b1; + word_n.cstate = CS_POP; + end else begin + word_n.dsc = word_x.follow ? follow_rank : RTU_CHILD_BITS'(0); + word_n.push_ptr = (coll_cnt_q == RTU_CHILD_BITS'(0)) + ? RTU_CHILD_BITS'(0) + : (coll_cnt_q - RTU_CHILD_BITS'(1)); + word_n.cstate = CS_PUSH; + end + wake_self = 1'b1; end CS_PUSH: begin - if (word_x.push_ptr != RTU_CHILD_BITS'(0)) begin + if (word_x.push_ptr != word_x.dsc) begin if (sp_q != RTU_STACK_BITS'(RTU_STACK_DEPTH)) begin stk_wr_r = 1'b1; - stk_wdata_r = coll_ordoff[word_x.coll_id][word_x.push_ptr[IDXW-1:0]] - & RTU_CHILD_OFF_MASK; + stk_wdata_r = {word_x.lvl + LVLW'(1), word_x.push_ptr, + coll_ordoff[word_x.coll_id][word_x.push_ptr[IDXW-1:0]] + & RTU_CHILD_OFF_MASK}; sp_inc = 1'b1; - end else if (word_x.in_blas) begin - word_n.ovf_o = 1'b1; end else begin - word_n.ovf_w = 1'b1; + // dropped: pushes run farthest first, so the last one + // dropped is the nearest + word_n.ovf = 1'b1; + word_n.nd_lvl = word_x.lvl + LVLW'(1); + word_n.nd_rank = word_x.push_ptr; end word_n.push_ptr = word_x.push_ptr - RTU_CHILD_BITS'(1); wake_self = 1'b1; end else if (coll_cnt_q != RTU_CHILD_BITS'(0)) begin - word_n.cur_off = coll_ordoff[word_x.coll_id][0] & RTU_CHILD_OFF_MASK; + word_n.cur_off = coll_ordoff[word_x.coll_id][word_x.dsc[IDXW-1:0]] + & RTU_CHILD_OFF_MASK; + word_n.lvl = word_x.lvl + LVLW'(1); + path_wr_r = 1'b1; + path_wlvl_r = word_x.lvl + LVLW'(1); + path_wdata_r = PATHW'(word_x.dsc); + if (word_x.follow && ((word_x.lvl + LVLW'(1)) == word_x.trl_lvl)) begin + word_n.follow = 1'b0; // at the dropped child: walk on normally + end coll_free_r = 1'b1; word_n.cstate = CS_REQ0; wake_self = 1'b1; @@ -1842,31 +1930,33 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( word_n.cstate = CS_DONE; exec_done = 1'b1; end + end else if (word_x.ovf) begin + // every node before the dropped child has been visited: start + // over from the root along its rank path + word_n.ovf = 1'b0; + word_n.follow = 1'b1; + word_n.trl_lvl = word_x.nd_lvl; + word_n.trl_rank = word_x.nd_rank; + word_n.in_blas = 1'b0; + word_n.lvl = '0; + word_n.cur_off = word_x.root_off; + sp_clr = 1'b1; + word_n.cstate = CS_REQ0; + wake_self = 1'b1; end else if (word_x.in_blas && (sp_q == word_x.blas_floor)) begin - if (word_x.ovf_o && (word_x.rst_o != RST_CNTW'(RTU_RESTART_CAP))) begin - // an object-level subtree was dropped: re-descend the BLAS - // root pruning by the tightened best_t - word_n.ovf_o = 1'b0; - word_n.rst_o = word_x.rst_o + RST_CNTW'(1); - word_n.cur_off = word_x.blas_root; - word_n.cstate = CS_REQ0; - end else begin - word_n.cstate = CS_INST_NEXT; - end - wake_self = 1'b1; + word_n.follow = 1'b0; + word_n.cstate = CS_INST_NEXT; + wake_self = 1'b1; end else if (sp_q == '0) begin - if (word_x.ovf_w && (word_x.rst_w != RST_CNTW'(RTU_RESTART_CAP))) begin - word_n.ovf_w = 1'b0; - word_n.rst_w = word_x.rst_w + RST_CNTW'(1); - word_n.cur_off = word_x.root_off; - word_n.cstate = CS_REQ0; - wake_self = 1'b1; - end else begin - word_n.cstate = CS_DONE; - exec_done = 1'b1; - end + word_n.cstate = CS_DONE; + exec_done = 1'b1; end else begin - word_n.cur_off = stacktop_q; + word_n.follow = 1'b0; // popped off a restart's path + word_n.cur_off = stacktop_q[31:0]; + word_n.lvl = stk_top_lvl; + path_wr_r = 1'b1; + path_wlvl_r = stk_top_lvl; + path_wdata_r = PATHW'(stk_top_rank); sp_dec = 1'b1; word_n.cstate = CS_REQ0; wake_self = 1'b1; @@ -1946,15 +2036,20 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( word_n.obj_inv_d = recip_q; word_n.in_blas = 1'b1; word_n.cur_off = word_x.blas_root; - word_n.rst_o = '0; - word_n.ovf_o = 1'b0; + // the BLAS root sits at the instance level, ranked by its index + word_n.lvl = word_x.ilvl; + path_wr_r = 1'b1; + path_wlvl_r = word_x.ilvl; + path_wdata_r = PATHW'(word_x.inst_idx); word_n.cstate = CS_REQ0; wake_self = 1'b1; end CS_INST_NEXT: begin // every instance is scanned: a candidate staged in one instance - // does not hide a nearer one in a later instance + // does not hide a nearer one in a later instance. A restart's + // path ends at the instance it resumed. word_n.in_blas = 1'b0; + word_n.follow = 1'b0; if ((word_x.inst_idx + 32'd1) == word_x.inst_cnt) begin if (FLAT) begin word_n.cstate = CS_DONE; @@ -2012,6 +2107,11 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( assign cs_wdata = word_n; assign stk_wr = x_valid && stk_wr_r; assign stk_wdata = stk_wdata_r; + assign path_wr = x_valid && path_wr_r; + `RUNTIME_ASSERT(~(x_valid && path_wr_r && (path_wlvl_r == '0) && (word_x.cstate == CS_PUSH)), + ("%t: rtu walk deeper than %0d levels", $time, (1 << LVLW) - 1)) + assign path_wlvl = path_wlvl_r; + assign path_wdata = path_wdata_r; assign orc_start = x_valid && orc_start_r; assign mem_req_valid = orc_mreq_valid || (x_valid && mem_issue); assign mem_req_addr = orc_mreq_valid ? orc_mreq_addr @@ -2126,7 +2226,8 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( if (exec_yld_clr) begin yld_q[sel_q] <= 1'b0; end - if (fresh_q || rewalk_q) begin + lvl_q_arr[sel_q] <= word_n.lvl; + if (fresh_q || rewalk_q || sp_clr) begin sp_q_arr[sel_q] <= '0; end else if (sp_inc) begin sp_q_arr[sel_q] <= sp_q + RTU_STACK_BITS'(1); diff --git a/tests/raytracing/rt_smoke_deep_stack/main.cpp b/tests/raytracing/rt_smoke_deep_stack/main.cpp index 71ca750bac..77e900cbe9 100644 --- a/tests/raytracing/rt_smoke_deep_stack/main.cpp +++ b/tests/raytracing/rt_smoke_deep_stack/main.cpp @@ -15,9 +15,10 @@ // // Builds a CW-BVH4 over N triangles stacked in depth along the ray, so the // tree is several levels deep — deeper than the modest short stack the Makefile -// configures (VX_CFG_RTU_STACK_DEPTH). A +z ray hits every triangle; the walker -// must overflow, drop far subtrees, and re-descend (restart) to still -// return the CLOSEST hit (nearest triangle, prim 0, t=5). +// configures (VX_CFG_RTU_STACK_DEPTH), so the walker overflows its stack, drops +// subtrees and restarts. Two scenes: every triangle hit (the closest is on the +// first path walked), and decoys whose boxes the ray enters first while only a +// far triangle is hit (reachable only through restarts). #include #include @@ -68,58 +69,46 @@ int main(int /*argc*/, char* /*argv*/[]) { vx_queue_info_t qi = { sizeof(qi), nullptr, VX_QUEUE_PRIORITY_NORMAL, 0 }; RT_CHECK(vx_queue_create(device, &qi, &queue)); - // N opaque triangles all covering the ray's (x,y) footprint, stacked at - // z = 5, 6, ... The SAH builder splits them into a deep tree. The ray hits - // all of them; triangle 0 (z=5) is the closest. - constexpr uint32_t N = 64; - std::vector tris(N); - for (uint32_t i = 0; i < N; ++i) { - float z = 5.0f + (float)i; - tris[i].v0[0] = 0.f; tris[i].v0[1] = 0.f; tris[i].v0[2] = z; - tris[i].v1[0] = 1.f; tris[i].v1[1] = 0.f; tris[i].v1[2] = z; - tris[i].v2[0] = 0.f; tris[i].v2[1] = 1.f; tris[i].v2[2] = z; - tris[i].flags = RTU_BVH_FLAG_OPAQUE; - } + int errors = 0; - host_bvh_t src = { tris.data(), N, /*geometry_index*/ 0 }; - std::vector scene; - uint64_t root_offset = 0; - if (!build_bvh_scene<4>(src, scene, root_offset)) { - std::cout << "build_bvh_scene failed" << std::endl; - cleanup(); - return 1; - } - std::cout << "scene: " << scene.size() << " B, " << N - << " tris (deep CW-BVH4)" << std::endl; - - RT_CHECK(vx_buffer_create(device, (uint32_t)scene.size(), VX_MEM_READ, &scene_buffer)); - RT_CHECK(vx_buffer_address(scene_buffer, &kernel_arg.scene_addr)); - - uint32_t res_size = sizeof(rtu_result_t); - RT_CHECK(vx_buffer_create(device, res_size, VX_MEM_WRITE, &res_buffer)); - RT_CHECK(vx_buffer_address(res_buffer, &kernel_arg.results_addr)); - - kernel_arg.ray_origin[0] = 0.25f; - kernel_arg.ray_origin[1] = 0.25f; - kernel_arg.ray_origin[2] = 0.0f; - kernel_arg.ray_direction[0] = 0.0f; - kernel_arg.ray_direction[1] = 0.0f; - kernel_arg.ray_direction[2] = 1.0f; - kernel_arg.tmin = 0.001f; - kernel_arg.tmax = 1e30f; - - std::cout << "scene_addr=0x" << std::hex << kernel_arg.scene_addr << std::dec - << " deep CW-BVH4 (closest hit must survive short-stack overflow)" - << std::endl; - - RT_CHECK(vx_enqueue_write(queue, scene_buffer, 0, scene.data(), - (uint32_t)scene.size(), 0, nullptr, nullptr)); - RT_CHECK(vx_module_load_file(device, kernel_file, &module_)); - RT_CHECK(vx_module_get_kernel(module_, "main", &kernel)); - - std::cout << "launch kernel" << std::endl; - vx_event_h launch_ev = nullptr, read_ev = nullptr; - { + // One trace of the +z ray at (0.25, 0.25) against a CW-BVH4 over `tris`; + // the walk must return the closest hit (t, prim). + auto run_case = [&](const char* name, const std::vector& tris, + float exp_t, uint32_t exp_prim) { + host_bvh_t src = { tris.data(), (uint32_t)tris.size(), /*geometry_index*/ 0 }; + std::vector scene; + uint64_t root_offset = 0; + if (!build_bvh_scene<4>(src, scene, root_offset)) { + std::cout << name << ": build_bvh_scene failed" << std::endl; + ++errors; + return; + } + std::cout << name << ": scene " << scene.size() << " B, " << tris.size() + << " tris (deep CW-BVH4)" << std::endl; + + RT_CHECK(vx_buffer_create(device, (uint32_t)scene.size(), VX_MEM_READ, &scene_buffer)); + RT_CHECK(vx_buffer_address(scene_buffer, &kernel_arg.scene_addr)); + uint32_t res_size = sizeof(rtu_result_t); + RT_CHECK(vx_buffer_create(device, res_size, VX_MEM_WRITE, &res_buffer)); + RT_CHECK(vx_buffer_address(res_buffer, &kernel_arg.results_addr)); + + kernel_arg.ray_origin[0] = 0.25f; + kernel_arg.ray_origin[1] = 0.25f; + kernel_arg.ray_origin[2] = 0.0f; + kernel_arg.ray_direction[0] = 0.0f; + kernel_arg.ray_direction[1] = 0.0f; + kernel_arg.ray_direction[2] = 1.0f; + kernel_arg.tmin = 0.001f; + kernel_arg.tmax = 1e30f; + + RT_CHECK(vx_enqueue_write(queue, scene_buffer, 0, scene.data(), + (uint32_t)scene.size(), 0, nullptr, nullptr)); + if (!kernel) { + RT_CHECK(vx_module_load_file(device, kernel_file, &module_)); + RT_CHECK(vx_module_get_kernel(module_, "main", &kernel)); + } + + vx_event_h launch_ev = nullptr, read_ev = nullptr; vx_launch_info_t li = {}; li.struct_size = sizeof(li); li.kernel = kernel; @@ -129,30 +118,52 @@ int main(int /*argc*/, char* /*argv*/[]) { li.grid_dim[0] = 1; li.block_dim[0] = 1; RT_CHECK(vx_enqueue_launch(queue, &li, 0, nullptr, &launch_ev)); - } - - rtu_result_t result = {}; - RT_CHECK(vx_enqueue_read(queue, &result, res_buffer, 0, res_size, - 1, &launch_ev, &read_ev)); - RT_CHECK(vx_event_wait_value(read_ev, 1, VX_TIMEOUT_INFINITE)); - vx_event_release(read_ev); - vx_event_release(launch_ev); - const uint32_t exp_status = VX_RT_STS_DONE_HIT; - const float exp_t = 5.f; // nearest triangle - const uint32_t exp_prim = 0; // source index of the z=5 triangle - std::cout << "oracle: HIT t=" << exp_t << " prim=" << exp_prim << std::endl; + rtu_result_t result = {}; + RT_CHECK(vx_enqueue_read(queue, &result, res_buffer, 0, res_size, + 1, &launch_ev, &read_ev)); + RT_CHECK(vx_event_wait_value(read_ev, 1, VX_TIMEOUT_INFINITE)); + vx_event_release(read_ev); + vx_event_release(launch_ev); + vx_buffer_release(scene_buffer); scene_buffer = nullptr; + vx_buffer_release(res_buffer); res_buffer = nullptr; + + std::cout << name << ": oracle HIT t=" << exp_t << " prim=" << exp_prim << std::endl; + if (result.status != VX_RT_STS_DONE_HIT || std::fabs(result.hit_t - exp_t) >= 1e-4f + || result.primitive_id != exp_prim) { + std::cout << name << ": result status=" << result.status + << " hit_t=" << result.hit_t + << " prim=" << result.primitive_id << std::endl; + ++errors; + } + }; + + auto tri_at = [](float z, float x0) { + host_tri_t t = {}; + t.v0[0] = x0; t.v0[1] = 0.f; t.v0[2] = z; + t.v1[0] = 1.f; t.v1[1] = 0.f; t.v1[2] = z; + t.v2[0] = 0.f; t.v2[1] = 1.f; t.v2[2] = z; + t.flags = RTU_BVH_FLAG_OPAQUE; + return t; + }; - int errors = 0; - bool sts_ok = (result.status == exp_status); - bool t_ok = std::fabs(result.hit_t - exp_t) < 1e-4f; - bool prim_ok = (result.primitive_id == exp_prim); - if (!sts_ok || !t_ok || !prim_ok) { - std::cout << "result: status=" << result.status - << " hit_t=" << result.hit_t - << " prim=" << result.primitive_id << std::endl; - ++errors; - } + // N opaque triangles all covering the ray's (x,y) footprint, stacked at + // z = 5, 6, ... The SAH builder splits them into a deep tree. The ray hits + // all of them; triangle 0 (z=5) is the closest, on the walk's first path. + constexpr uint32_t N = 64; + std::vector tris; + for (uint32_t i = 0; i < N; ++i) tris.push_back(tri_at(5.0f + (float)i, 0.f)); + run_case("stacked", tris, 5.f, 0); + + // The same stack, but only one triangle, deep in the far half, covers the + // ray: the others' boxes do (so they are walked first) while the triangles + // miss it. The hit sits in subtrees the short stack drops, so the walk has + // to restart, repeatedly, to reach it. + constexpr uint32_t kHit = 41; + tris.clear(); + for (uint32_t i = 0; i < N; ++i) + tris.push_back(tri_at(5.0f + (float)i, (i == kHit) ? 0.f : 0.6f)); + run_case("decoys", tris, 5.f + (float)kHit, kHit); cleanup(); From a2fa139c077761a7a0f13200507d9919135f8d9f Mon Sep 17 00:00:00 2001 From: Blaise Tine Date: Sat, 3 Oct 2026 09:45:51 -0700 Subject: [PATCH 17/31] rtu: instances carry lavapipe's world->object matrix; no inverse anywhere The instance record held the object->world transform and each model inverted it on its own: SimX by cofactor inverse, the RTL as R^T (exact only for an orthonormal R). Neither reproduces the object ray the Vulkan reference traces, which applies its own world->object matrix as rounded products added in the order translation + x + y + z (lvp_mul_vec3_mat; llvmpipe lowers ffma, so nothing is fused). The record now holds that world->object matrix (the driver copies lavapipe's wto, the host builder inverts in double), and SimX (world_to_object_ray, also used by the visit-order oracle) and VX_rtu_xform transform the ray in that order, so the object ray matches lavapipe bit for bit for any affine instance, scale and shear included. The RTL keeps its 4*FMA latency: 18 products, then three dependent adds, every stage registered; IEEE specials handled. The unused VX_rtu_fdot3/fcross3/fmac3 go. hw/unittest/rtu_xform: 1M random affine (rotation, non-uniform scale, shear) + 38K directed special cases vs SimX, 0 mismatches. Direct record writers in tests/raytracing store the inverse translation. Co-Authored-By: Claude Opus 5.5 --- docs/designs/ray_tracing_architecture.md | 23 +- hw/rtl/rtu/VX_rtu_fcross3.sv | 127 ----------- hw/rtl/rtu/VX_rtu_fdot3.sv | 101 --------- hw/rtl/rtu/VX_rtu_fmac3.sv | 208 ------------------ hw/rtl/rtu/VX_rtu_pkg.sv | 7 +- hw/rtl/rtu/VX_rtu_xform.sv | 208 ++++++++++-------- hw/unittest/rtu_xform/Makefile | 31 +++ hw/unittest/rtu_xform/main.cpp | 183 +++++++++++++++ sim/simx/rtu/rtu_bvh.h | 5 +- sim/simx/rtu/rtu_isect.cpp | 52 ++--- sim/simx/rtu/rtu_isect.h | 24 +- sim/simx/rtu/rtu_types.h | 8 +- sim/simx/rtu/rtu_walker.cpp | 26 +-- sw/common/rtu_cfg.h | 3 +- sw/runtime/include/raytrace.h | 25 ++- .../rt_smoke_bvh_instanced/main.cpp | 8 +- tests/raytracing/rt_smoke_cull_mask/main.cpp | 2 +- tests/raytracing/rt_smoke_inst_flags/main.cpp | 2 +- tests/raytracing/rt_smoke_proc_inst/main.cpp | 5 +- tests/raytracing/rt_smoke_tie/main.cpp | 2 +- tests/raytracing/rt_smoke_tlas/main.cpp | 5 +- 21 files changed, 423 insertions(+), 632 deletions(-) delete mode 100644 hw/rtl/rtu/VX_rtu_fcross3.sv delete mode 100644 hw/rtl/rtu/VX_rtu_fdot3.sv delete mode 100644 hw/rtl/rtu/VX_rtu_fmac3.sv create mode 100644 hw/unittest/rtu_xform/Makefile create mode 100644 hw/unittest/rtu_xform/main.cpp diff --git a/docs/designs/ray_tracing_architecture.md b/docs/designs/ray_tracing_architecture.md index ec5bd26638..45933526af 100644 --- a/docs/designs/ray_tracing_architecture.md +++ b/docs/designs/ray_tracing_architecture.md @@ -327,20 +327,15 @@ subnormals flushed either way), and `VX_CFG_FMA_LATENCY` / whatever depth results: - **`VX_rtu_box_pe`** — pipelined ray/AABB slab test, one child box per cycle, - emitting `{hit, t_near}`. Dequantizes the node's int8 child corners - (`origin + q·2^exp`), does the slab test with `VX_fma_unit` + `VX_fncp_unit`, - and subtracts the ray origin *before* multiplying by `inv_d` so axis-aligned - rays (`inv_d = ±inf`) stay NaN-free. Also handles raw/procedural boxes. -- **`VX_rtu_tri_pe`** — pipelined Möller–Trumbore triangle test, one triangle per - cycle, emitting `{hit, t, u, v, back_facing}`; reuses `VX_fma_unit`, - `VX_fdiv_unit` (1/det), `VX_fncp_unit`, and `VX_rtu_fdot3`/`fcross3`. The - dot/cross helpers pipeline their 24×24 mantissa products into DSP multipliers - (`LATENCY_IMUL` deep) fed the **raw** mantissas: a flushed (subnormal/zero) - term is discarded downstream in the `VX_rtu_fmac3` accumulator by its zero - product-exponent, so no subnormal-flush select sits in front of the multiplier - inputs and the DSPs launch straight from the source flops. -- **`VX_rtu_xform`** — TLAS world→object transform, `obj = Rᵀ·(ro−t)` — FMA-only - (an orthonormal TLAS rotation needs no determinant or divide). Always built: + emitting `{hit, t_near}`. Mirrors SimX `ray_aabb_intersect` bit for bit: + corners `origin + q·2^exp`, slabs `(corner − ro)·inv_d`, culled against + `[0, t_max]`. Also handles raw/procedural boxes. +- **`VX_rtu_tri_pe`** — pipelined watertight triangle test (F32 shear, F64 edge + functions and t, in the Vulkan reference's op order), one triangle per cycle, + emitting `{hit, t, u, v, back_facing}`; bit-exact against SimX `ray_triangle`. +- **`VX_rtu_xform`** — TLAS world→object transform. The instance record holds + the world→object matrix and the ray is transformed in the reference's op + order (rounded products, then `t + x + y + z`), so no inverse is taken. Always built: the CW-BVH walker descends `LEAF_INST` natively; only the flat walker's (`WIDTH = 0`) instancing loop is gated by `VX_CFG_RTU_TLAS_ENABLE`. - **`VX_rtu_recip`** — F32 reciprocal for `inv_d`, either a portable LUT+Newton diff --git a/hw/rtl/rtu/VX_rtu_fcross3.sv b/hw/rtl/rtu/VX_rtu_fcross3.sv deleted file mode 100644 index e68924eab0..0000000000 --- a/hw/rtl/rtu/VX_rtu_fcross3.sv +++ /dev/null @@ -1,127 +0,0 @@ -// Copyright © 2019-2023 -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -// VX_rtu_fcross3 — fused fp32 3-vector cross product result = a × b. -// result[i] = a[(i+1)%3]*b[(i+2)%3] - a[(i+2)%3]*b[(i+1)%3] -// Each axis forms its two lane products and sums them (the second negated) in a -// shared VX_rtu_fmac3, normalizing+rounding ONCE rather than per FMA. Inputs in -// the RTU geometry path are finite; subnormals flushed to zero. Latency padded -// to 2*LATENCY_FMA so the consuming PE keeps its side-band alignment. - -`include "VX_define.vh" - -module VX_rtu_fcross3 import VX_gpu_pkg::*, VX_fpu_pkg::*; #( - parameter LATENCY_FMA = `VX_CFG_FMA_LATENCY, - parameter LATENCY = 2 * LATENCY_FMA -) ( - input wire clk, - input wire reset, - input wire enable, - input wire [2:0][31:0] a, - input wire [2:0][31:0] b, - output wire [2:0][31:0] result -); - // output pad must cover the pipelined multiply + fmac3 depth - `STATIC_ASSERT((LATENCY >= (`LATENCY_IMUL + 7)), ("VX_rtu_fcross3: LATENCY too small for the pipelined multiply")) - - for (genvar i = 0; i < 3; ++i) begin : g_axis - localparam I1 = (i + 1) % 3; - localparam I2 = (i + 2) % 3; - - // term0 = +a[I1]*b[I2], term1 = -a[I2]*b[I1] - wire [7:0] e0a = a[I1][30:23], e0b = b[I2][30:23]; - wire z0a = (e0a == 8'd0), z0b = (e0b == 8'd0); - wire [23:0] m0a = {1'b1, a[I1][22:0]}; - wire [23:0] m0b = {1'b1, b[I2][22:0]}; - - wire [7:0] e1a = a[I2][30:23], e1b = b[I1][30:23]; - wire z1a = (e1a == 8'd0), z1b = (e1b == 8'd0); - wire [23:0] m1a = {1'b1, a[I2][22:0]}; - wire [23:0] m1b = {1'b1, b[I1][22:0]}; - - wire [2:0] m_sign = {1'b0, - ~(a[I2][31] ^ b[I1][31]), // negated (subtraction) - (a[I1][31] ^ b[I2][31])}; - wire [2:0][8:0] m_pe = {9'd0, - (z1a | z1b) ? 9'd0 : ({1'b0, e1a} + {1'b0, e1b}), - (z0a | z0b) ? 9'd0 : ({1'b0, e0a} + {1'b0, e0b})}; - // 24x24 mantissa products pipelined into the DSP48 (LATENCY_IMUL deep) - // so the multiply is registered rather than combinational. The operands - // are the raw mantissas: a flushed term is discarded downstream by its - // pe=0, so nothing selects in front of the multiplier inputs and the - // DSP is driven straight from the source flops. - wire [47:0] pp0, pp1; - VX_multiplier #( - .A_WIDTH (24), - .B_WIDTH (24), - .SIGNED (0), - .LATENCY (`LATENCY_IMUL) - ) mul0 ( - .clk (clk), - .enable (enable), - .dataa (m0a), - .datab (m0b), - .result (pp0) - ); - VX_multiplier #( - .A_WIDTH (24), - .B_WIDTH (24), - .SIGNED (0), - .LATENCY (`LATENCY_IMUL) - ) mul1 ( - .clk (clk), - .enable (enable), - .dataa (m1a), - .datab (m1b), - .result (pp1) - ); - wire [2:0][47:0] q_prod = {48'd0, pp1, pp0}; - - // sign/exponent side-band delayed to align with the multiply latency - wire [2:0] q_sign; - wire [2:0][8:0] q_pe; - VX_pipe_register #( - .DATAW (3 + 3*9), - .DEPTH (`LATENCY_IMUL) - ) p0 ( - .clk (clk), - .reset (reset), - .enable (enable), - .data_in ({m_sign, m_pe}), - .data_out ({q_sign, q_pe}) - ); - - wire [31:0] crs; - VX_rtu_fmac3 mac ( - .clk (clk), - .reset (reset), - .enable (enable), - .sign (q_sign), - .pe (q_pe), - .prod (q_prod), - .result (crs) - ); - - VX_shift_register #( - .DATAW (32), - .DEPTH (LATENCY - (`LATENCY_IMUL + 7)) // 7 = VX_rtu_fmac3 depth - ) sr_pad ( - .clk (clk), - .reset (reset), - .enable (enable), - .data_in (crs), - .data_out (result[i]) - ); - end - -endmodule diff --git a/hw/rtl/rtu/VX_rtu_fdot3.sv b/hw/rtl/rtu/VX_rtu_fdot3.sv deleted file mode 100644 index 79400df589..0000000000 --- a/hw/rtl/rtu/VX_rtu_fdot3.sv +++ /dev/null @@ -1,101 +0,0 @@ -// Copyright © 2019-2023 -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -// VX_rtu_fdot3 — fused fp32 3-vector dot product result = a·b. The three lane -// products feed VX_rtu_fmac3, which aligns and sums them, then normalizes and -// rounds ONCE — versus a chain of three FMAs that each normalize+round. Inputs -// in the RTU geometry path are finite; subnormals are flushed to zero. Latency -// padded to 3*LATENCY_FMA so the consuming PEs keep their side-band alignment. - -`include "VX_define.vh" - -module VX_rtu_fdot3 import VX_gpu_pkg::*, VX_fpu_pkg::*; #( - parameter LATENCY_FMA = `VX_CFG_FMA_LATENCY, - parameter LATENCY = 3 * LATENCY_FMA -) ( - input wire clk, - input wire reset, - input wire enable, - input wire [2:0][31:0] a, - input wire [2:0][31:0] b, - output wire [31:0] result -); - // output pad must cover the pipelined multiply + fmac3 depth - `STATIC_ASSERT((LATENCY >= (`LATENCY_IMUL + 7)), ("VX_rtu_fdot3: LATENCY too small for the pipelined multiply")) - - wire [2:0] m_sign; - wire [2:0][8:0] m_pe; - wire [2:0][47:0] m_prod; - for (genvar i = 0; i < 3; ++i) begin : g_mul - wire [7:0] ea = a[i][30:23], eb = b[i][30:23]; - wire az = (ea == 8'd0), bz = (eb == 8'd0); - wire [23:0] ma = {1'b1, a[i][22:0]}; - wire [23:0] mb = {1'b1, b[i][22:0]}; - assign m_sign[i] = a[i][31] ^ b[i][31]; - assign m_pe[i] = (az | bz) ? 9'd0 : ({1'b0, ea} + {1'b0, eb}); - // 24x24 mantissa product pipelined into the DSP48 (LATENCY_IMUL deep). - // The operands are the raw mantissas: a flushed term is discarded - // downstream by its pe=0, so nothing selects in front of the multiplier - // inputs and the DSP is driven straight from the source flops. - VX_multiplier #( - .A_WIDTH (24), - .B_WIDTH (24), - .SIGNED (0), - .LATENCY (`LATENCY_IMUL) - ) mul ( - .clk (clk), - .enable (enable), - .dataa (ma), - .datab (mb), - .result (m_prod[i]) - ); - end - - // sign/exponent side-band delayed to align with the multiply latency - wire [2:0] q_sign; - wire [2:0][8:0] q_pe; - wire [2:0][47:0] q_prod = m_prod; // products already registered by the DSPs - VX_pipe_register #( - .DATAW (3 + 3*9), - .DEPTH (`LATENCY_IMUL) - ) p0 ( - .clk (clk), - .reset (reset), - .enable (enable), - .data_in ({m_sign, m_pe}), - .data_out ({q_sign, q_pe}) - ); - - wire [31:0] dot; - VX_rtu_fmac3 mac ( - .clk (clk), - .reset (reset), - .enable (enable), - .sign (q_sign), - .pe (q_pe), - .prod (q_prod), - .result (dot) - ); - - VX_shift_register #( - .DATAW (32), - .DEPTH (LATENCY - (`LATENCY_IMUL + 7)) // 7 = VX_rtu_fmac3 depth - ) sr_pad ( - .clk (clk), - .reset (reset), - .enable (enable), - .data_in (dot), - .data_out (result) - ); - -endmodule diff --git a/hw/rtl/rtu/VX_rtu_fmac3.sv b/hw/rtl/rtu/VX_rtu_fmac3.sv deleted file mode 100644 index 155cd7c366..0000000000 --- a/hw/rtl/rtu/VX_rtu_fmac3.sv +++ /dev/null @@ -1,208 +0,0 @@ -// Copyright © 2019-2023 -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -// VX_rtu_fmac3 — fused sum of up to three signed fp32 products. Each term is a -// pre-formed 48-bit mantissa product with a 9-bit product exponent (ea+eb) and -// a sign; an unused term passes pe=0/prod=0. The terms are aligned to the common -// (max) exponent, summed in extended precision, then normalized and rounded -// ONCE. Finite inputs only; subnormals flushed to zero. Deeply pipelined — the -// carry-heavy ops (negate, abs, the wide add and the LZC) each get their own -// cycle — to hold 300 MHz; latency = 7. Shared by VX_rtu_fdot3 (3 terms) and -// VX_rtu_fcross3 (2 terms per axis). - -`include "VX_define.vh" - -module VX_rtu_fmac3 #( - parameter PW = 48 -) ( - input wire clk, - input wire reset, - input wire enable, - input wire [2:0] sign, - input wire [2:0][8:0] pe, // product exponent ea+eb (0 => unused term) - input wire [2:0][PW-1:0] prod, // 48-bit mantissa product - output wire [31:0] result -); - localparam GW = 32; // guard bits below the product - localparam FW = PW + GW; // aligned field width - localparam SW = FW + 2; // signed-sum magnitude width - localparam LZW = `LOG2UP(SW); - - // ── stage 1: max exponent + per-term right-shift (no negate yet) ───── - wire [8:0] pe01 = (pe[0] > pe[1]) ? pe[0] : pe[1]; - wire [8:0] max_pe = (pe01 > pe[2]) ? pe01 : pe[2]; - - // An unused term is identified by pe=0 alone: a term whose operands are both - // normal has pe >= 2, so pe=0 is unambiguous. Discarding its product here - // means the callers do not have to force their mantissas to zero, which - // would put a select in front of the multiplier inputs. The mask sits on the - // product rather than on the shift amount, which controls a full-width - // barrel shifter and is far more sensitive to an extra level. - wire [FW-1:0] field [3]; - for (genvar i = 0; i < 3; ++i) begin : g_shift - wire [8:0] sh = max_pe - pe[i]; - wire [PW-1:0] prod_z = (pe[i] != 9'd0) ? prod[i] : '0; - assign field[i] = ({{(FW-PW){1'b0}}, prod_z} << GW) >> sh; - end - - wire [FW-1:0] s1_f0, s1_f1, s1_f2; - wire [2:0] s1_sign; - wire [8:0] s1_max_pe; - VX_pipe_register #( - .DATAW (3*FW + 3 + 9), - .DEPTH (1) - ) p1 ( - .clk (clk), - .reset (reset), - .enable (enable), - .data_in ({field[0], field[1], field[2], sign, max_pe}), - .data_out ({s1_f0, s1_f1, s1_f2, s1_sign, s1_max_pe}) - ); - - // ── stage 2: apply sign (two's-complement negate) ──────────────────── - wire [FW-1:0] s1_field [3]; - assign s1_field[0] = s1_f0; - assign s1_field[1] = s1_f1; - assign s1_field[2] = s1_f2; - wire signed [SW:0] term [3]; - for (genvar i = 0; i < 3; ++i) begin : g_neg - wire signed [SW:0] fext = $signed({{(SW+1-FW){1'b0}}, s1_field[i]}); - assign term[i] = s1_sign[i] ? -fext : fext; - end - - wire signed [SW:0] s2_t0, s2_t1, s2_t2; - wire [8:0] s2_max_pe; - VX_pipe_register #( - .DATAW (3*(SW+1) + 9), - .DEPTH (1) - ) p2 ( - .clk (clk), - .reset (reset), - .enable (enable), - .data_in ({term[0], term[1], term[2], s1_max_pe}), - .data_out ({s2_t0, s2_t1, s2_t2, s2_max_pe}) - ); - - // ── stage 3: signed sum ────────────────────────────────────────────── - wire signed [SW:0] sum = s2_t0 + s2_t1 + s2_t2; - - wire signed [SW:0] s3_sum; - wire [8:0] s3_max_pe; - VX_pipe_register #( - .DATAW (SW+1 + 9), - .DEPTH (1) - ) p3 ( - .clk (clk), - .reset (reset), - .enable (enable), - .data_in ({sum, s2_max_pe}), - .data_out ({s3_sum, s3_max_pe}) - ); - - // ── stage 4: sign + magnitude (abs) ────────────────────────────────── - wire neg = s3_sum[SW]; - wire [SW-1:0] absS = neg ? (~s3_sum[SW-1:0] + 1'b1) : s3_sum[SW-1:0]; - - wire [SW-1:0] s4_abs; - wire s4_neg; - wire [8:0] s4_max_pe; - VX_pipe_register #( - .DATAW (SW + 1 + 9), - .DEPTH (1) - ) p4 ( - .clk (clk), - .reset (reset), - .enable (enable), - .data_in ({absS, neg, s3_max_pe}), - .data_out ({s4_abs, s4_neg, s4_max_pe}) - ); - - // ── stage 5: leading-zero count ────────────────────────────────────── - wire [LZW-1:0] lz; - wire lz_valid; - VX_lzc #( - .N (SW) - ) lzc_i ( - .data_in (s4_abs), - .data_out (lz), - .valid_out (lz_valid) - ); - - wire [SW-1:0] s5_abs; - wire [LZW-1:0] s5_lz; - wire s5_lzv, s5_neg; - wire [8:0] s5_max_pe; - VX_pipe_register #( - .DATAW (SW + LZW + 1 + 1 + 9), - .DEPTH (1) - ) p5 ( - .clk (clk), - .reset (reset), - .enable (enable), - .data_in ({s4_abs, lz, lz_valid, s4_neg, s4_max_pe}), - .data_out ({s5_abs, s5_lz, s5_lzv, s5_neg, s5_max_pe}) - ); - - // ── stage 6: normalize, extract mantissa + GRS, base exponent ──────── - wire [SW-1:0] norm = s5_abs << s5_lz; // leading 1 at bit SW-1 - wire [23:0] mant = norm[SW-1 -: 24]; - wire g_bit = norm[SW-1-24]; - wire r_bit = norm[SW-1-25]; - wire stky = |norm[SW-1-26 : 0]; - wire zero = (s5_abs == '0); - wire signed [10:0] rexp_b = $signed({2'b0, s5_max_pe}) - 11'sd124 - - $signed({{(11-LZW){1'b0}}, s5_lz}); - - wire [23:0] s6_mant; - wire s6_g, s6_r, s6_s, s6_zero, s6_lzv, s6_neg; - wire signed [10:0] s6_rexp; - VX_pipe_register #( - .DATAW (24 + 6 + 11), - .DEPTH (1) - ) p6 ( - .clk (clk), - .reset (reset), - .enable (enable), - .data_in ({mant, g_bit, r_bit, stky, zero, s5_lzv, s5_neg, rexp_b}), - .data_out ({s6_mant, s6_g, s6_r, s6_s, s6_zero, s6_lzv, s6_neg, s6_rexp}) - ); - - // ── stage 7: round (RNE) + pack ────────────────────────────────────── - wire round_up = s6_g & (s6_r | s6_s | s6_mant[0]); - wire [24:0] mant_r = s6_mant + round_up; - wire carry = mant_r[24]; - wire [22:0] frac = carry ? mant_r[23:1] : mant_r[22:0]; - wire signed [10:0] rexp = s6_rexp + $signed({10'd0, carry}); - - reg [31:0] res; - always @(*) begin - if (s6_zero || !s6_lzv || rexp <= 11'sd0) - res = {s6_neg, 31'd0}; - else if (rexp >= 11'sd255) - res = {s6_neg, 8'hFF, 23'd0}; - else - res = {s6_neg, rexp[7:0], frac}; - end - - VX_pipe_register #( - .DATAW (32), - .DEPTH (1) - ) p7 ( - .clk (clk), - .reset (reset), - .enable (enable), - .data_in (res), - .data_out (result) - ); - -endmodule diff --git a/hw/rtl/rtu/VX_rtu_pkg.sv b/hw/rtl/rtu/VX_rtu_pkg.sv index f94316d891..e8481b8d03 100644 --- a/hw/rtl/rtu/VX_rtu_pkg.sv +++ b/hw/rtl/rtu/VX_rtu_pkg.sv @@ -192,10 +192,9 @@ package VX_rtu_pkg; localparam RTU_FLAT_LINES_BITS = `CLOG2(RTU_FLAT_LINES + 1); // ───────────────────────────────────────────────────────────────── - // TLAS instance record (64 B). The 3x4 row-major affine transform - // (object→world) occupies floats 0..11; - // the walker applies its inverse (VX_rtu_xform) to bring the world ray into - // object space. The two TLAS variants share xform/blas/custom but differ in + // TLAS instance record (64 B). The 3x4 row-major world→object affine + // occupies floats 0..11; the walker applies it (VX_rtu_xform) to bring the + // world ray into object space. The two TLAS variants share xform/blas/custom but differ in // where instance_id and cull_mask sit: // flat TLAS : blas_off@48, custom_id@52, cull_mask@56; instance_id = loop idx // BVH inst : blas_root@48, custom_id@52, instance_id@56, cull_mask@60 diff --git a/hw/rtl/rtu/VX_rtu_xform.sv b/hw/rtl/rtu/VX_rtu_xform.sv index 5e175b03d9..3ac10a7823 100644 --- a/hw/rtl/rtu/VX_rtu_xform.sv +++ b/hw/rtl/rtu/VX_rtu_xform.sv @@ -12,30 +12,19 @@ // limitations under the License. // VX_rtu_xform — world→object ray transform for a TLAS instance. Streams one -// instance's 3x4 affine transform + world ray and emits the object-space ray +// instance's world→object 3x4 matrix + world ray and emits the object-space ray // after a fixed latency. // -// obj_ro = R^T * (ro - t) obj_rd = R^T * rd +// The instance record carries the world→object matrix the source driver +// (lavapipe) builds, and the ray is transformed in the order it does: F32, +// every product rounded, then // -// The instance transform is object→world; its inverse brings the world ray into -// object space. For the orthonormal rotation+translation transforms a TLAS -// carries (every instance in a valid scene), R is orthonormal so R^(-1) = R^T, -// which needs no determinant or division — a pure FMA pipeline. This is bit- -// equivalent to the SimX oracle's explicit cofactor inverse for any orthonormal -// R (the only kind the tests and a valid Vulkan TLAS produce); SimX's singular- -// matrix passthrough is moot here as there is no divide to guard. +// obj_ro[i] = ((m[i][3] + ro.x*m[i][0]) + ro.y*m[i][1]) + ro.z*m[i][2] +// obj_rd[i] = (rd.x*m[i][0] + rd.y*m[i][1]) + rd.z*m[i][2] // -// Layout of the 3x4 row-major transform (matches the shared host/SimX format): -// xform[0..2] = R row 0 xform[3] = t.x -// xform[4..6] = R row 1 xform[7] = t.y -// xform[8..10] = R row 2 xform[11] = t.z -// obj_ro[i] = (column i of R) . (ro - t); column i of R = row i of R^T: -// col0 = {xform[0], xform[4], xform[8]}, etc. -// -// The (ro - t) subtract reuses VX_fma_unit (a*1 - c); the matrix-vector products -// reuse VX_rtu_fdot3. Side-band operands are delayed through shift registers so -// every stage consumes time-aligned inputs at a fixed latency the scheduler -// tracks via valid_out — same structure as VX_rtu_tri_pe / VX_rtu_box_pe. +// so the object ray matches it bit for bit for any affine instance (scale, +// shear included), with no inverse taken anywhere. Layout: m[i][j] = xform[4*i +// + j], row-major, translation in column 3. `include "VX_define.vh" @@ -49,7 +38,7 @@ module VX_rtu_xform import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( input wire valid_in, input wire [TAG_WIDTH-1:0] tag_in, // caller side-band (e.g. context id) - input wire [11:0][31:0] xform, // 3x4 row-major affine (object→world) + input wire [11:0][31:0] xform, // 3x4 row-major affine (world→object) input wire [2:0][31:0] ro, // world ray origin input wire [2:0][31:0] rd, // world ray direction @@ -59,97 +48,136 @@ module VX_rtu_xform import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( output wire [2:0][31:0] obj_rd // object-space ray direction ); localparam F = LATENCY_FMA; - localparam LATENCY = 4 * F; // (ro-t) subtract @F, then dot @3F - - localparam [INST_FMT_BITS-1:0] FMT_SUB = 2'b10; // F32, a*b - c - localparam [31:0] FP_ONE = 32'h3F800000; + localparam LATENCY = 4 * F; // products, then three dependent adds - // translation vector t = {xform[3], xform[7], xform[11]}. - wire [2:0][31:0] tvec; - assign tvec[0] = xform[3]; - assign tvec[1] = xform[7]; - assign tvec[2] = xform[11]; + localparam [INST_FMT_BITS-1:0] FMT_ADD = 2'b00; - // R columns (rows of R^T): col_i[j] = xform[4*j + i]. - wire [2:0][2:0][31:0] col; - for (genvar i = 0; i < 3; ++i) begin : g_col - for (genvar j = 0; j < 3; ++j) begin : g_col_e - assign col[i][j] = xform[4*j + i]; + // ── @0 → @F: every product, rounded ─────────────────────────────── + wire [2:0][2:0][31:0] po, pd; // [row][column] + for (genvar i = 0; i < 3; ++i) begin : g_row + for (genvar j = 0; j < 3; ++j) begin : g_col + VX_fma_unit #( + .USE_DSP (`VX_CFG_RTU_USE_DSP), + .LATENCY (F), + .SUBNORM_ENABLE (0), + .EXCEPT_ENABLE (1) + ) fmul_o ( + .clk (clk), + .reset (reset), + .enable (enable), + .mask (1'b1), + .op_type (INST_FPU_MUL), + .fmt (FMT_ADD), + .frm (INST_FRM_RNE), + .dataa (ro[j]), + .datab (xform[4*i + j]), + .datac ('0), + .result (po[i][j]), + `UNUSED_PIN (fflags) + ); + VX_fma_unit #( + .USE_DSP (`VX_CFG_RTU_USE_DSP), + .LATENCY (F), + .SUBNORM_ENABLE (0), + .EXCEPT_ENABLE (1) + ) fmul_d ( + .clk (clk), + .reset (reset), + .enable (enable), + .mask (1'b1), + .op_type (INST_FPU_MUL), + .fmt (FMT_ADD), + .frm (INST_FRM_RNE), + .dataa (rd[j]), + .datab (xform[4*i + j]), + .datac ('0), + .result (pd[i][j]), + `UNUSED_PIN (fflags) + ); end end - // ── stage 1 (@F): d = ro - t (per axis), reusing the FMA as a*1 - c ── - wire [2:0][31:0] d; - for (genvar a = 0; a < 3; ++a) begin : g_sub - VX_fma_unit #( - .USE_DSP (`VX_CFG_RTU_USE_DSP), // vendor xil_fma on Vivado (soft in sim), like the FPU - .SUBNORM_ENABLE (0), - .LATENCY (F) - ) fma_d ( - .clk (clk), - .reset (reset), - .enable (enable), - .mask (1'b1), - .op_type (INST_FPU_MADD), - .fmt (FMT_SUB), - .frm (INST_FRM_RNE), - .dataa (ro[a]), - .datab (FP_ONE), - .datac (tvec[a]), - .result (d[a]), - `UNUSED_PIN (fflags) - ); - end - - // R columns aligned from @0 to @F to feed the dot products. - wire [2:0][2:0][31:0] col_d; + wire [2:0][31:0] tr_d; // translation column @F VX_shift_register #( - .DATAW (9*32), + .DATAW (3*32), .DEPTH (F) - ) sr_col ( + ) sr_tr ( .clk (clk), .reset (reset), .enable (enable), - .data_in (col), - .data_out (col_d) + .data_in ({xform[11], xform[7], xform[3]}), + .data_out (tr_d) ); - // rd aligned from @0 to @F so the direction dot starts in lock-step with d. - wire [2:0][31:0] rd_d; + + // later addends held until their add issues + wire [2:0][31:0] po1_d, po2_d, pd2_d; // po[.][1] @2F, po[.][2] @3F, pd[.][2] @2F VX_shift_register #( - .DATAW (96), + .DATAW (2*3*32), .DEPTH (F) - ) sr_rd ( + ) sr_p1 ( + .clk (clk), + .reset (reset), + .enable (enable), + .data_in ({po[2][1], po[1][1], po[0][1], pd[2][2], pd[1][2], pd[0][2]}), + .data_out ({po1_d, pd2_d}) + ); + VX_shift_register #( + .DATAW (3*32), + .DEPTH (2 * F) + ) sr_p2 ( .clk (clk), .reset (reset), .enable (enable), - .data_in (rd), - .data_out (rd_d) + .data_in ({po[2][2], po[1][2], po[0][2]}), + .data_out (po2_d) ); - // ── stage 2 (@F+3F = @4F): obj_ro[i] = col_i . d, obj_rd[i] = col_i . rd ── - for (genvar i = 0; i < 3; ++i) begin : g_dot - VX_rtu_fdot3 #( - .LATENCY_FMA (F) - ) dot_ro ( - .clk (clk), - .reset (reset), - .enable (enable), - .a (col_d[i]), - .b (d), - .result (obj_ro[i]) + // ── the dependent adds, one per F ───────────────────────────────── + wire [2:0][31:0] o1, o2, d1, d2; + for (genvar i = 0; i < 3; ++i) begin : g_sum + VX_fma_unit #(.USE_DSP (`VX_CFG_RTU_USE_DSP), .LATENCY (F), .SUBNORM_ENABLE (0), .EXCEPT_ENABLE (1)) fadd_o1 ( + .clk (clk), .reset (reset), .enable (enable), .mask (1'b1), + .op_type (INST_FPU_ADD), .fmt (FMT_ADD), .frm (INST_FRM_RNE), + .dataa (tr_d[i]), .datab (po[i][0]), .datac ('0), + .result (o1[i]), `UNUSED_PIN (fflags) ); - VX_rtu_fdot3 #( - .LATENCY_FMA (F) - ) dot_rd ( - .clk (clk), - .reset (reset), - .enable (enable), - .a (col_d[i]), - .b (rd_d), - .result (obj_rd[i]) + VX_fma_unit #(.USE_DSP (`VX_CFG_RTU_USE_DSP), .LATENCY (F), .SUBNORM_ENABLE (0), .EXCEPT_ENABLE (1)) fadd_o2 ( + .clk (clk), .reset (reset), .enable (enable), .mask (1'b1), + .op_type (INST_FPU_ADD), .fmt (FMT_ADD), .frm (INST_FRM_RNE), + .dataa (o1[i]), .datab (po1_d[i]), .datac ('0), + .result (o2[i]), `UNUSED_PIN (fflags) + ); + VX_fma_unit #(.USE_DSP (`VX_CFG_RTU_USE_DSP), .LATENCY (F), .SUBNORM_ENABLE (0), .EXCEPT_ENABLE (1)) fadd_o3 ( + .clk (clk), .reset (reset), .enable (enable), .mask (1'b1), + .op_type (INST_FPU_ADD), .fmt (FMT_ADD), .frm (INST_FRM_RNE), + .dataa (o2[i]), .datab (po2_d[i]), .datac ('0), + .result (obj_ro[i]), `UNUSED_PIN (fflags) + ); + VX_fma_unit #(.USE_DSP (`VX_CFG_RTU_USE_DSP), .LATENCY (F), .SUBNORM_ENABLE (0), .EXCEPT_ENABLE (1)) fadd_d1 ( + .clk (clk), .reset (reset), .enable (enable), .mask (1'b1), + .op_type (INST_FPU_ADD), .fmt (FMT_ADD), .frm (INST_FRM_RNE), + .dataa (pd[i][0]), .datab (pd[i][1]), .datac ('0), + .result (d1[i]), `UNUSED_PIN (fflags) + ); + VX_fma_unit #(.USE_DSP (`VX_CFG_RTU_USE_DSP), .LATENCY (F), .SUBNORM_ENABLE (0), .EXCEPT_ENABLE (1)) fadd_d2 ( + .clk (clk), .reset (reset), .enable (enable), .mask (1'b1), + .op_type (INST_FPU_ADD), .fmt (FMT_ADD), .frm (INST_FRM_RNE), + .dataa (d1[i]), .datab (pd2_d[i]), .datac ('0), + .result (d2[i]), `UNUSED_PIN (fflags) ); end + VX_shift_register #( + .DATAW (3*32), + .DEPTH (F) + ) sr_d ( + .clk (clk), + .reset (reset), + .enable (enable), + .data_in (d2), + .data_out (obj_rd) + ); + // ── valid + tag pipe, sized to the whole datapath latency ───────── reg [LATENCY-1:0] valid_pipe_r; always_ff @(posedge clk) begin diff --git a/hw/unittest/rtu_xform/Makefile b/hw/unittest/rtu_xform/Makefile new file mode 100644 index 0000000000..12eccb6bad --- /dev/null +++ b/hw/unittest/rtu_xform/Makefile @@ -0,0 +1,31 @@ +ROOT_DIR := $(realpath ../../..) +include $(ROOT_DIR)/config.mk + +PROJECT := rtu_xform + +RTL_DIR := $(VORTEX_HOME)/hw/rtl +SRC_DIR := $(VORTEX_HOME)/hw/unittest/$(PROJECT) +SIMX_DIR := $(VORTEX_HOME)/sim/simx + +# The reference is the SimX model itself, so the RTU config must match both sides. +CONFIGS += -DSIMULATION -DVX_CFG_EXT_RTU_ENABLE + +CXXFLAGS := -I$(SRC_DIR) -I$(VORTEX_HOME)/hw/unittest/common -I$(SW_COMMON_DIR) +CXXFLAGS += -I$(SIMX_DIR) -I$(SIMX_DIR)/rtu -I$(SIM_COMMON_DIR) +CXXFLAGS += -I$(ROOT_DIR)/sw -I$(ROOT_DIR)/hw +CXXFLAGS += -I$(THIRD_PARTY_DIR)/softfloat/source/include + +SRCS += $(SIMX_DIR)/rtu/rtu_isect.cpp +SRCS += $(SRC_DIR)/main.cpp + +PARAMS := -GTAG_WIDTH=32 + +RTL_PKGS += $(RTL_DIR)/VX_gpu_pkg.sv $(RTL_DIR)/fpu/VX_fpu_pkg.sv $(RTL_DIR)/rtu/VX_rtu_pkg.sv +RTL_INCLUDE := -I$(ROOT_DIR)/sw -I$(RTL_DIR) -I$(RTL_DIR)/libs -I$(RTL_DIR)/interfaces +RTL_INCLUDE += -I$(RTL_DIR)/fpu -I$(RTL_DIR)/rtu -I$(SRC_DIR) + +VL_FLAGS += -I$(ROOT_DIR)/hw + +TOP := VX_rtu_xform + +include ../common.mk diff --git a/hw/unittest/rtu_xform/main.cpp b/hw/unittest/rtu_xform/main.cpp new file mode 100644 index 0000000000..1dd8f2fb9f --- /dev/null +++ b/hw/unittest/rtu_xform/main.cpp @@ -0,0 +1,183 @@ +// VX_rtu_xform against SimX's rtu::world_to_object_ray: the object-space origin +// and direction must match bit for bit (any NaN matches any NaN) for random +// affine instances -- rotation, non-uniform scale, shear, translation -- and +// special operands. +// +// The xform's FP units flush subnormals (FTZ/DAZ) while SimX runs IEEE. A case +// whose RTL result differs from SimX but equals SimX evaluated under the host's +// FTZ/DAZ mode, and only there, is counted as a subnormal flush, not a mismatch. + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include "VVX_rtu_xform.h" +#include "verilated.h" +#include "rtu_isect.h" + +namespace { + +struct Case { + float m[12], o[3], d[3]; + int cat = 0; +}; + +const char* const kCatNames[] = { + "random affine", "rigid", "identity", "special matrix", "special ray", +}; +constexpr int kNumCats = int(sizeof(kCatNames) / sizeof(kCatNames[0])); + +struct Result { float o[3], d[3]; }; + +uint32_t bits(float f) { uint32_t b; std::memcpy(&b, &f, 4); return b; } + +bool same(uint32_t rtl, float ref) { + const bool rtl_nan = ((rtl >> 23) & 0xff) == 0xff && (rtl & 0x7fffff) != 0; + return std::isnan(ref) ? rtl_nan : rtl == bits(ref); +} + +Result reference(const Case& c, bool ftz) { + const unsigned csr = _mm_getcsr(); + if (ftz) _mm_setcsr(csr | 0x8040); // FTZ | DAZ + Result r; + vortex::rtu::world_to_object_ray(c.m, c.o, c.d, r.o, r.d); + _mm_setcsr(csr); + return r; +} + +template +bool matches(const Dut& dut, const Result& r) { + for (int a = 0; a < 3; ++a) + if (!same(dut.obj_ro[a], r.o[a]) || !same(dut.obj_rd[a], r.d[a])) return false; + return true; +} + +bool same_result(const Result& a, const Result& b) { + for (int k = 0; k < 3; ++k) + if (!same(bits(a.o[k]), b.o[k]) || !same(bits(a.d[k]), b.d[k])) return false; + return true; +} + +std::mt19937 rng(5); +float uni(float lo, float hi) { return std::uniform_real_distribution(lo, hi)(rng); } + +// A random world->object matrix: rotation x scale x shear (or rigid), plus a +// translation, at assorted magnitudes. +Case make_case(uint32_t i) { + Case c; + const float ax = uni(-1, 1), ay = uni(-1, 1), az = uni(-1, 1); + const float n = std::sqrt(ax * ax + ay * ay + az * az) + 1e-6f; + const float x = ax / n, y = ay / n, z = az / n, th = uni(-3.2f, 3.2f); + const float cs = std::cos(th), sn = std::sin(th), t = 1 - cs; + float R[9] = { t * x * x + cs, t * x * y - sn * z, t * x * z + sn * y, + t * x * y + sn * z, t * y * y + cs, t * y * z - sn * x, + t * x * z - sn * y, t * y * z + sn * x, t * z * z + cs }; + c.cat = (i % 5 == 0) ? 1 : 0; + float S[3] = { 1, 1, 1 }, H[3] = { 0, 0, 0 }; + if (c.cat == 0) { + for (float& s : S) s = (i % 3 == 0) ? uni(1e-3f, 1e3f) : uni(0.1f, 10.f); + for (float& h : H) h = uni(-2, 2); + } + const float M[9] = { S[0], H[0], H[1], 0, S[1], H[2], 0, 0, S[2] }; + for (int r = 0; r < 3; ++r) + for (int k = 0; k < 3; ++k) + c.m[r * 4 + k] = R[r * 3 + 0] * M[0 * 3 + k] + R[r * 3 + 1] * M[1 * 3 + k] + + R[r * 3 + 2] * M[2 * 3 + k]; + const float span = (i % 7 == 0) ? 1e4f : 50.f; + for (int r = 0; r < 3; ++r) c.m[r * 4 + 3] = uni(-span, span); + for (int a = 0; a < 3; ++a) { c.o[a] = uni(-span, span); c.d[a] = uni(-1, 1); } + if (i % 11 == 0) c.d[i % 3] = (i & 1) ? -0.f : 0.f; + return c; +} + +std::vector directed_cases() { + std::vector out; + const float specials[] = { NAN, INFINITY, -INFINITY, 0.f, -0.f, 1e-40f, -1e-40f, + FLT_MAX, -FLT_MAX }; + for (uint32_t i = 0; i < 2000; ++i) { + Case c = make_case(i * 3 + 1); + Case id = c; + id.cat = 2; + for (int k = 0; k < 12; ++k) id.m[k] = (k % 5 == 0) ? 1.f : 0.f; + out.push_back(id); + for (float sp : specials) { + Case s = c; + s.cat = 3; s.m[i % 12] = sp; out.push_back(s); + s = c; s.cat = 4; + if (i & 1) s.o[i % 3] = sp; else s.d[i % 3] = sp; + out.push_back(s); + } + } + return out; +} + +} // namespace + +int main(int argc, char** argv) { + Verilated::commandArgs(argc, argv); + const uint32_t N = (argc > 1) ? uint32_t(std::atoi(argv[1])) : 1000000; + VVX_rtu_xform dut; + dut.clk = 0; dut.reset = 1; dut.enable = 1; dut.valid_in = 0; + auto tick = [&] { dut.clk = 0; dut.eval(); dut.clk = 1; dut.eval(); }; + for (int i = 0; i < 4; ++i) tick(); + dut.reset = 0; + + const std::vector directed = directed_cases(); + const uint32_t ND = uint32_t(directed.size()); + const uint32_t total = N + ND; + struct Expect { uint32_t id; int cat; Result ieee, ftz; }; + std::deque exp; + std::deque cases; + uint32_t sent = 0, checked = 0, errors = 0, flushed = 0; + uint32_t cat_cases[kNumCats] = {}, cat_err[kNumCats] = {}, cat_ftz[kNumCats] = {}; + while (checked < total) { + if (sent < total) { + const Case c = (sent < ND) ? directed[sent] : make_case(sent - ND); + for (int k = 0; k < 12; ++k) dut.xform[k] = bits(c.m[k]); + for (int a = 0; a < 3; ++a) { dut.ro[a] = bits(c.o[a]); dut.rd[a] = bits(c.d[a]); } + dut.tag_in = sent; + dut.valid_in = 1; + exp.push_back({ sent, c.cat, reference(c, false), reference(c, true) }); + cases.push_back(c); + ++sent; + } else { + dut.valid_in = 0; + } + tick(); + if (dut.valid_out) { + const Expect e = exp.front(); exp.pop_front(); + const Case c = cases.front(); cases.pop_front(); + ++cat_cases[e.cat]; + bool ok = (dut.tag_out == e.id) && matches(dut, e.ieee); + if (!ok && dut.tag_out == e.id && matches(dut, e.ftz) && !same_result(e.ieee, e.ftz)) { + ok = true; ++flushed; ++cat_ftz[e.cat]; + } + if (!ok) { + ++errors; + if (cat_err[e.cat]++ < 3) { + std::printf("MISMATCH #%u [%s] m=(", e.id, kCatNames[e.cat]); + for (float v : c.m) std::printf("%a ", v); + std::printf(") o=(%a %a %a) d=(%a %a %a)\n", c.o[0], c.o[1], c.o[2], c.d[0], c.d[1], c.d[2]); + std::printf(" ref o=(%08x %08x %08x) d=(%08x %08x %08x) | rtl o=(%08x %08x %08x) d=(%08x %08x %08x)\n", + bits(e.ieee.o[0]), bits(e.ieee.o[1]), bits(e.ieee.o[2]), + bits(e.ieee.d[0]), bits(e.ieee.d[1]), bits(e.ieee.d[2]), + dut.obj_ro[0], dut.obj_ro[1], dut.obj_ro[2], + dut.obj_rd[0], dut.obj_rd[1], dut.obj_rd[2]); + } + } + ++checked; + } + } + std::printf("rtu_xform: %u cases (%u directed), %u mismatches, %u subnormal flushes\n", + checked, ND, errors, flushed); + for (int k = 0; k < kNumCats; ++k) + std::printf(" %-16s %8u cases %6u mismatches %6u subnormal flushes\n", + kCatNames[k], cat_cases[k], cat_err[k], cat_ftz[k]); + std::printf(errors ? "FAILED!\n" : "PASSED!\n"); + return errors ? 1 : 0; +} diff --git a/sim/simx/rtu/rtu_bvh.h b/sim/simx/rtu/rtu_bvh.h index d2394c4054..7fcafb55ad 100644 --- a/sim/simx/rtu/rtu_bvh.h +++ b/sim/simx/rtu/rtu_bvh.h @@ -31,7 +31,7 @@ // from one decode point. // - Triangle stride (40 B) and TLAS instance stride (64 B) match the // flat-list constants in rtu_core.cpp so the existing intersection -// helpers (`ray_triangle`, `affine_inverse_transform_ray`) drop in +// helpers (`ray_triangle`, `world_to_object_ray`) drop in // unchanged. #ifndef _VX_RTU_BVH_H_ @@ -263,7 +263,8 @@ constexpr uint32_t kVxBvhTriStride = 40; // from the SCENE root, not from a private BLAS base — gives us a // single base address for the whole TLAS+BLAS bundle). // -// floats 0..11 : 48 B object→world affine (3x4, row-major) +// floats 0..11 : 48 B world→object affine (3x4, row-major), +// the inverse of the instance transform // uint32 blas_root_off : 4 B byte offset to this instance's BLAS // root node from the scene-buffer base // uint32 custom_id : 4 B VK_INSTANCE_CUSTOM_INDEX_KHR diff --git a/sim/simx/rtu/rtu_isect.cpp b/sim/simx/rtu/rtu_isect.cpp index 9716bacc0d..fa1a87348c 100644 --- a/sim/simx/rtu/rtu_isect.cpp +++ b/sim/simx/rtu/rtu_isect.cpp @@ -118,45 +118,21 @@ bool ray_aabb_intersect(const float ro[3], const float rd[3], return true; } -void affine_inverse_transform_ray(const float xform[12], - const float ro[3], const float rd[3], - float ro_out[3], float rd_out[3]) { - const float r00 = xform[0], r01 = xform[1], r02 = xform[2], tx = xform[3]; - const float r10 = xform[4], r11 = xform[5], r12 = xform[6], ty = xform[7]; - const float r20 = xform[8], r21 = xform[9], r22 = xform[10], tz = xform[11]; - - // det(R) by cofactor expansion along row 0. - float det = r00 * (r11 * r22 - r12 * r21) - - r01 * (r10 * r22 - r12 * r20) - + r02 * (r10 * r21 - r11 * r20); - if (det > -1e-9f && det < 1e-9f) { - // Singular — pass through (treat as identity). - for (int i = 0; i < 3; ++i) { ro_out[i] = ro[i]; rd_out[i] = rd[i]; } - return; +void world_to_object_ray(const float wto[12], + const float ro[3], const float rd[3], + float ro_out[3], float rd_out[3]) { + for (int i = 0; i < 3; ++i) { + const float* m = wto + 4 * i; + float o = m[3]; + o = o + ro[0] * m[0]; + o = o + ro[1] * m[1]; + o = o + ro[2] * m[2]; + float d = rd[0] * m[0]; + d = d + rd[1] * m[1]; + d = d + rd[2] * m[2]; + ro_out[i] = o; + rd_out[i] = d; } - float inv_det = 1.f / det; - - // R^(-1) = (1/det) * adj(R). - float i00 = (r11 * r22 - r12 * r21) * inv_det; - float i01 = -(r01 * r22 - r02 * r21) * inv_det; - float i02 = (r01 * r12 - r02 * r11) * inv_det; - float i10 = -(r10 * r22 - r12 * r20) * inv_det; - float i11 = (r00 * r22 - r02 * r20) * inv_det; - float i12 = -(r00 * r12 - r02 * r10) * inv_det; - float i20 = (r10 * r21 - r11 * r20) * inv_det; - float i21 = -(r00 * r21 - r01 * r20) * inv_det; - float i22 = (r00 * r11 - r01 * r10) * inv_det; - - // ro_obj = R^(-1) * (ro - t). - float dx = ro[0] - tx, dy = ro[1] - ty, dz = ro[2] - tz; - ro_out[0] = i00 * dx + i01 * dy + i02 * dz; - ro_out[1] = i10 * dx + i11 * dy + i12 * dz; - ro_out[2] = i20 * dx + i21 * dy + i22 * dz; - - // rd_obj = R^(-1) * rd. - rd_out[0] = i00 * rd[0] + i01 * rd[1] + i02 * rd[2]; - rd_out[1] = i10 * rd[0] + i11 * rd[1] + i12 * rd[2]; - rd_out[2] = i20 * rd[0] + i21 * rd[1] + i22 * rd[2]; } // PE cost model. There is ONE box PE and ONE tri PE per RtuCore, each streaming diff --git a/sim/simx/rtu/rtu_isect.h b/sim/simx/rtu/rtu_isect.h index 8b2f33e871..dadcbb59b7 100644 --- a/sim/simx/rtu/rtu_isect.h +++ b/sim/simx/rtu/rtu_isect.h @@ -57,23 +57,19 @@ bool ray_aabb_intersect(const float ro[3], const float rd[3], float tmin, float tmax, float& t_near); // ──────────────────────────────────────────────────────────────────── -// Apply the inverse of a 3x4 row-major affine to a ray, producing the -// object-space ray. Used by the BVH4 walker on LeafInst descent to -// convert world→object space. Mirrors the hardware XFORM unit -// (latency = 3 cycles). +// Bring a world ray into an instance's object space with the instance +// record's world→object 3x4 row-major matrix m, in the Vulkan reference's +// (lavapipe) op order, every product rounded: // -// xform = [r00 r01 r02 tx | r10 r11 r12 ty | r20 r21 r22 tz] -// ro_obj = R^(-1) * (ro_world - t) -// rd_obj = R^(-1) * rd_world +// ro_obj[i] = ((m[i][3] + ro.x*m[i][0]) + ro.y*m[i][1]) + ro.z*m[i][2] +// rd_obj[i] = (rd.x*m[i][0] + rd.y*m[i][1]) + rd.z*m[i][2] // -// For pure rotation+translation (det(R) == ±1) the t parameter is -// preserved across spaces, so the BLAS-reported hit_t is also the -// world hit_t. Non-uniform scale would require renormalising hit_t; -// out of scope. +// The direction is not renormalised, so t is the same in both spaces. +// Mirrors VX_rtu_xform bit for bit. // ──────────────────────────────────────────────────────────────────── -void affine_inverse_transform_ray(const float xform[12], - const float ro[3], const float rd[3], - float ro_out[3], float rd_out[3]); +void world_to_object_ray(const float wto[12], + const float ro[3], const float rd[3], + float ro_out[3], float rd_out[3]); // ════════════════════════════════════════════════════════════════════ // The intersection coprocessors — pipelined BoxPe / TriPe. diff --git a/sim/simx/rtu/rtu_types.h b/sim/simx/rtu/rtu_types.h index d24c27fbda..45cec20052 100644 --- a/sim/simx/rtu/rtu_types.h +++ b/sim/simx/rtu/rtu_types.h @@ -300,14 +300,14 @@ constexpr uint32_t kRtuFdivLat = 17; // reciprocal pipe depth constexpr uint32_t kRtuLatencyFma = 9; // FMA pipe depth constexpr uint32_t kRtuLatencyFma64 = 12; // F64 FMA pipe depth (tri PE) constexpr uint32_t kRtuFdiv64Lat = 32; // F64 divide pipe depth (tri PE) -// Per-instance transform latency = 4 * FMA pipe depth = 36: an (ro-t) subtract -// at FMA depth, then a 3-deep dot product. Charged per TLAS instance descent -// in the SimX cost model. +// Per-instance transform latency = 4 * FMA pipe depth = 36: the products, then +// three dependent adds (VX_rtu_xform). Charged per TLAS instance descent in the +// SimX cost model. constexpr uint32_t kRtuXformLatency = 36; // 4 * FMA pipe depth // TLAS instance record (64 B). Lives inline after the scene header for // "TLAS + inline BLAS" layout. -// floats 0..11 = 3x4 affine transform (rows r0|r1|r2), object→world +// floats 0..11 = 3x4 affine transform (rows r0|r1|r2), world→object // uint32 [48..52) = blas_byte_offset // uint32 [52..56) = custom_id (Vulkan VK_INSTANCE_CUSTOM_INDEX_KHR) // uint32 [56..60) = cull_mask (low byte = Vulkan instance mask; diff --git a/sim/simx/rtu/rtu_walker.cpp b/sim/simx/rtu/rtu_walker.cpp index ec4dda580f..8ced6f95ed 100644 --- a/sim/simx/rtu/rtu_walker.cpp +++ b/sim/simx/rtu/rtu_walker.cpp @@ -25,7 +25,7 @@ // scene-format constants #include "rtu_bvh.h" // CW-BVH node/leaf/instance layouts #include "rtu_isect.h" // ray_triangle, ray_aabb_intersect, - // affine_inverse_transform_ray + // world_to_object_ray #include "rtu_classifier.h" // classify_tri_hit, finalise_lane namespace vortex { namespace rtu { @@ -297,8 +297,8 @@ bool src_blas_path_culled(const SrcTab& t, uint32_t ps, const float box[6], return c.culled; } -// The reference's object-space ray for TLAS leaf `inst`: its world->object -// matrix applied as translation + x + y + z, in that order, in F32. +// The reference's object-space ray for TLAS leaf `inst`, from its +// world->object matrix in the table. SrcRay src_object_ray(SceneView& sv, uint32_t tlas_tab, uint32_t inst, const float wo[3], const float wd[3]) { const uint32_t leaf_stride = scene_u32(sv, tlas_tab + 12); @@ -306,14 +306,7 @@ SrcRay src_object_ray(SceneView& sv, uint32_t tlas_tab, uint32_t inst, read_scene_bytes(sv, tlas_tab + kSrcTabHdr + inst * leaf_stride + kSrcLeafWto, sizeof(m), reinterpret_cast(m)); float o[3], d[3]; - for (int i = 0; i < 3; ++i) { - float ro = m[i * 4 + 3], rd = 0.f; - for (int j = 0; j < 3; ++j) { - ro = ro + wo[j] * m[i * 4 + j]; - rd = j ? rd + wd[j] * m[i * 4 + j] : wd[j] * m[i * 4 + j]; - } - o[i] = ro; d[i] = rd; - } + world_to_object_ray(m, wo, wd, o, d); return src_ray(o, d); } @@ -592,7 +585,7 @@ void walk_bvh4_subtree(SceneView& sv, uint32_t inst_flags2 = (inst->cull_mask >> kRtuInstanceFlagsShift) & kRtuInstanceFlagsMask; float obj_ro[3], obj_rd[3]; - affine_inverse_transform_ray(inst->xform, ro, rd, obj_ro, obj_rd); + world_to_object_ray(inst->xform, ro, rd, obj_ro, obj_rd); ++perf.bvh_instance_descents; walk_bvh4_subtree(sv, obj_ro, obj_rd, inst->blas_root_byte_offset, @@ -857,7 +850,7 @@ WalkResult FlatWalker::walk_lane(const RtuReq& req, uint32_t t, SceneView& sv, WalkCtx ctx = init_ctx(req, t, ro, rd, l); // TLAS scenes walk one or more instances; each instance points at a BLAS (a - // triangle list) and (optionally) applies an object→world affine transform. + // triangle list) through its world→object affine transform. uint32_t num_instances = 1; uint32_t triangle_count = 0; #ifdef VX_CFG_RTU_TLAS_ENABLE @@ -918,10 +911,9 @@ WalkResult FlatWalker::walk_lane(const RtuReq& req, uint32_t t, SceneView& sv, std::memcpy(&cur_custom, inst_buf + kRtuInstanceCustomIdOff, sizeof(uint32_t)); - // World→object ray transform. For pure rotation + translation the t - // parameter is preserved, so the BLAS-reported hit_t is also the world - // hit_t. - affine_inverse_transform_ray(xform, ro, rd, ray_o, ray_d); + // World→object ray transform; the direction is not renormalised, so + // the BLAS-reported hit_t is also the world hit_t. + world_to_object_ray(xform, ro, rd, ray_o, ray_d); ++perf.bvh_instance_descents; uint8_t blas_hdr[4]; read_scene_bytes(sv, blas_byte_off, sizeof(blas_hdr), blas_hdr); diff --git a/sw/common/rtu_cfg.h b/sw/common/rtu_cfg.h index f0afdb0e58..4e92efa146 100644 --- a/sw/common/rtu_cfg.h +++ b/sw/common/rtu_cfg.h @@ -66,7 +66,8 @@ #define RTU_BVH_COUNT_SHIFT 8 // CW-BVH TLAS instance record (64 B). Emitted under a LEAF_INST leaf. -// float xform[12] @0 3x4 row-major object->world affine +// float xform[12] @0 3x4 row-major world->object affine (the inverse of +// the instance transform; the walker applies it) // uint32 blas_root @48 byte offset (from scene base) of this instance's BLAS root // uint32 custom_id @52 VK_INSTANCE_CUSTOM_INDEX_KHR // uint32 instance_id @56 HW-assigned instance ID diff --git a/sw/runtime/include/raytrace.h b/sw/runtime/include/raytrace.h index 6febe1c687..c01a75e7bd 100644 --- a/sw/runtime/include/raytrace.h +++ b/sw/runtime/include/raytrace.h @@ -325,6 +325,27 @@ class BvhBuilder { bool overflow_ = false; }; +// The world->object matrix an instance record carries: the inverse of the +// host's object->world 3x4 affine, in double, rounded once. +inline void world_to_object(const float otw[12], float wto[12]) { + const double a = otw[0], b = otw[1], c = otw[2]; + const double d = otw[4], e = otw[5], f = otw[6]; + const double g = otw[8], h = otw[9], k = otw[10]; + const double det = a * (e * k - f * h) - b * (d * k - f * g) + c * (d * h - e * g); + const double inv = (det != 0.0) ? 1.0 / det : 0.0; + const double r[9] = { (e * k - f * h) * inv, (c * h - b * k) * inv, (b * f - c * e) * inv, + (f * g - d * k) * inv, (a * k - c * g) * inv, (c * d - a * f) * inv, + (d * h - e * g) * inv, (b * g - a * h) * inv, (a * e - b * d) * inv }; + for (int i = 0; i < 3; ++i) { + double t = 0.0; + for (int j = 0; j < 3; ++j) { + wto[i * 4 + j] = float(r[i * 3 + j]); + t -= r[i * 3 + j] * double(otw[j * 4 + 3]); + } + wto[i * 4 + 3] = float(t); + } +} + } // namespace detail // ── Host-side scene preparation ───────────────────────────────────────── @@ -425,7 +446,9 @@ inline bool build_tlas_scene(const host_tlas_t& src, const host_instance_t& in = src.instances[i]; if (in.blas_index >= src.blas_count) { out_scene.clear(); return false; } uint8_t* rec = out_scene.data() + insts_off + i * RTU_BVH_INSTANCE_STRIDE; - std::memcpy(rec, in.xform, sizeof(in.xform)); + float wto[12]; + detail::world_to_object(in.xform, wto); + std::memcpy(rec, wto, sizeof(wto)); uint32_t broot = blas_root[in.blas_index]; uint32_t cid = in.custom_id; uint32_t iid = in.instance_id; diff --git a/tests/raytracing/rt_smoke_bvh_instanced/main.cpp b/tests/raytracing/rt_smoke_bvh_instanced/main.cpp index e2aac68c7b..73ee543537 100644 --- a/tests/raytracing/rt_smoke_bvh_instanced/main.cpp +++ b/tests/raytracing/rt_smoke_bvh_instanced/main.cpp @@ -86,10 +86,10 @@ static void emit_instance(uint8_t* out, float tx, float ty, float tz, uint32_t blas_off, uint32_t custom_id, uint32_t instance_id) { float* xform = reinterpret_cast(out); - // Row-major 3x4 affine: [R t]. Identity R, translation t. - xform[0] = 1.f; xform[1] = 0.f; xform[2] = 0.f; xform[3] = tx; - xform[4] = 0.f; xform[5] = 1.f; xform[6] = 0.f; xform[7] = ty; - xform[8] = 0.f; xform[9] = 0.f; xform[10] = 1.f; xform[11] = tz; + // Row-major 3x4 world->object affine: the inverse of [I t], i.e. [I -t]. + xform[0] = 1.f; xform[1] = 0.f; xform[2] = 0.f; xform[3] = -tx; + xform[4] = 0.f; xform[5] = 1.f; xform[6] = 0.f; xform[7] = -ty; + xform[8] = 0.f; xform[9] = 0.f; xform[10] = 1.f; xform[11] = -tz; *reinterpret_cast(out + VX_BVH_INSTANCE_BLAS_OFF) = blas_off; *reinterpret_cast(out + VX_BVH_INSTANCE_CUSTOM_ID) = custom_id; *reinterpret_cast(out + VX_BVH_INSTANCE_ID_OFFSET) = instance_id; diff --git a/tests/raytracing/rt_smoke_cull_mask/main.cpp b/tests/raytracing/rt_smoke_cull_mask/main.cpp index 91e186ceeb..0664fdc290 100644 --- a/tests/raytracing/rt_smoke_cull_mask/main.cpp +++ b/tests/raytracing/rt_smoke_cull_mask/main.cpp @@ -88,7 +88,7 @@ int main(int /*argc*/, char* /*argv*/[]) { float* xform = reinterpret_cast(inst); xform[0] = 1.f; xform[1] = 0.f; xform[2] = 0.f; xform[3] = 0.f; xform[4] = 0.f; xform[5] = 1.f; xform[6] = 0.f; xform[7] = 0.f; - xform[8] = 0.f; xform[9] = 0.f; xform[10] = 1.f; xform[11] = tz; + xform[8] = 0.f; xform[9] = 0.f; xform[10] = 1.f; xform[11] = -tz; // world->object uint32_t* inst_tail = reinterpret_cast( inst + RTU_INSTANCE_BLAS_OFF_OFF); inst_tail[0] = kBlasOff; diff --git a/tests/raytracing/rt_smoke_inst_flags/main.cpp b/tests/raytracing/rt_smoke_inst_flags/main.cpp index 285b65e943..5d19e2b97c 100644 --- a/tests/raytracing/rt_smoke_inst_flags/main.cpp +++ b/tests/raytracing/rt_smoke_inst_flags/main.cpp @@ -111,7 +111,7 @@ int main(int argc, char* argv[]) { float* xform = reinterpret_cast(inst); xform[0] = 1.f; xform[1] = 0.f; xform[2] = 0.f; xform[3] = 0.f; xform[4] = 0.f; xform[5] = 1.f; xform[6] = 0.f; xform[7] = 0.f; - xform[8] = 0.f; xform[9] = 0.f; xform[10] = 1.f; xform[11] = 5.f; + xform[8] = 0.f; xform[9] = 0.f; xform[10] = 1.f; xform[11] = -5.f; // world->object uint32_t* blas_off = reinterpret_cast(inst + RTU_INSTANCE_BLAS_OFF_OFF); *blas_off = kBlasOff; uint32_t* custom_id = reinterpret_cast(inst + RTU_INSTANCE_CUSTOM_ID_OFF); diff --git a/tests/raytracing/rt_smoke_proc_inst/main.cpp b/tests/raytracing/rt_smoke_proc_inst/main.cpp index 5cc3fe4e29..539a8d3aa1 100644 --- a/tests/raytracing/rt_smoke_proc_inst/main.cpp +++ b/tests/raytracing/rt_smoke_proc_inst/main.cpp @@ -75,11 +75,12 @@ static const uint32_t kInstId[NUM_RAYS] = { 5, 9 }; static const uint32_t kInstCust[NUM_RAYS] = { 0xa5, 0xa9 }; static const float kInstTx[NUM_RAYS] = { -3.f, 3.f }; -// Identity rotation + translation (tx,0,0). +// Identity rotation + translation (tx,0,0); the record holds its inverse +// (world->object). static void emit_instance(uint8_t* out, float tx, uint32_t blas_off, uint32_t custom_id, uint32_t instance_id) { float* x = reinterpret_cast(out); - x[0] = 1.f; x[1] = 0.f; x[2] = 0.f; x[3] = tx; + x[0] = 1.f; x[1] = 0.f; x[2] = 0.f; x[3] = -tx; x[4] = 0.f; x[5] = 1.f; x[6] = 0.f; x[7] = 0.f; x[8] = 0.f; x[9] = 0.f; x[10] = 1.f; x[11] = 0.f; *reinterpret_cast(out + VX_BVH_INSTANCE_BLAS_OFF) = blas_off; diff --git a/tests/raytracing/rt_smoke_tie/main.cpp b/tests/raytracing/rt_smoke_tie/main.cpp index 6c889a1c98..f0548a5243 100644 --- a/tests/raytracing/rt_smoke_tie/main.cpp +++ b/tests/raytracing/rt_smoke_tie/main.cpp @@ -506,7 +506,7 @@ Built build(uint64_t scene_addr, uint64_t tab_addr, bool with_tables) { img.u32(off + 12, tlas_tab); } const uint32_t rec = off + RTU_BVH_LEAF_HDR_BYTES; - for (int k = 0; k < 12; ++k) img.f32(rec + 4 * k, in.otw[k]); + for (int k = 0; k < 12; ++k) img.f32(rec + 4 * k, in.wto[k]); img.u32(rec + RTU_BVH_INSTANCE_BLAS_OFF, blas_root[in.blas]); img.u32(rec + RTU_BVH_INSTANCE_CUSTOM_OFF, in.custom); img.u32(rec + RTU_BVH_INSTANCE_ID_OFF, in.id); diff --git a/tests/raytracing/rt_smoke_tlas/main.cpp b/tests/raytracing/rt_smoke_tlas/main.cpp index fb8cf1bab8..aa7d38a850 100644 --- a/tests/raytracing/rt_smoke_tlas/main.cpp +++ b/tests/raytracing/rt_smoke_tlas/main.cpp @@ -87,10 +87,11 @@ int main(int /*argc*/, char* /*argv*/[]) { uint8_t* inst = scene_bytes.data() + RTU_SCENE_HDR_BYTES + idx * RTU_INSTANCE_STRIDE; float* xform = reinterpret_cast(inst); - // 3x4 affine row-major; identity R + translation t=(0,0,tz). + // 3x4 world->object affine row-major: the inverse of translation + // (0,0,tz). xform[0] = 1.f; xform[1] = 0.f; xform[2] = 0.f; xform[3] = 0.f; xform[4] = 0.f; xform[5] = 1.f; xform[6] = 0.f; xform[7] = 0.f; - xform[8] = 0.f; xform[9] = 0.f; xform[10] = 1.f; xform[11] = tz; + xform[8] = 0.f; xform[9] = 0.f; xform[10] = 1.f; xform[11] = -tz; uint32_t* inst_tail = reinterpret_cast( inst + RTU_INSTANCE_BLAS_OFF_OFF); inst_tail[0] = kBlasOff; // shared inline BLAS From 92925ba349b8827f41cf91cddd1ca76236b9f4a6 Mon Sep 17 00:00:00 2001 From: Blaise Tine Date: Sat, 3 Oct 2026 09:45:51 -0700 Subject: [PATCH 18/31] rtu: a SimX lane without a hit reports t_max and its world ray, as the RTL The RTL returns the ray's own t_max as the hit t of a lane with no committed hit (what a ray query reports as the committed t with none: lavapipe initialises it to t_max) and the world ray as its object ray; SimX returned zeros. Both now report the same. Co-Authored-By: Claude Opus 5.5 --- sim/simx/rtu/rtu_types.h | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/sim/simx/rtu/rtu_types.h b/sim/simx/rtu/rtu_types.h index 45cec20052..ff48d434e6 100644 --- a/sim/simx/rtu/rtu_types.h +++ b/sim/simx/rtu/rtu_types.h @@ -163,8 +163,14 @@ struct RtuRsp { uint32_t slot_idx = 0; RtuRsp() = default; + // A lane without a hit reads back its own ray: t = t_max (the committed t a + // ray query reports with no hit, as the Vulkan reference does) and the world + // ray as its object ray. RtuRsp(const RtuReq& req) : uuid(req.uuid), tag(req.tag), + hit_t(req.tmax), + obj_o_x(req.origin_x), obj_o_y(req.origin_y), obj_o_z(req.origin_z), + obj_d_x(req.dir_x), obj_d_y(req.dir_y), obj_d_z(req.dir_z), trace(req.trace), block_id(req.block_id), warp_id(req.warp_id), slot_idx(req.slot_idx) {} From e7b6f1f99f6a1c745a0ff384f4fae4ed5b0f889c Mon Sep 17 00:00:00 2001 From: Blaise Tine Date: Sat, 3 Oct 2026 12:52:10 -0700 Subject: [PATCH 19/31] simx: ray log snapshots the whole scene image, not just the lines SimX read A model that culls differently descends into nodes the SimX walk never fetched; replayed against only the lines SimX read, those come back as zeros and show up as false mismatches. The first trace against a scene now also snapshots its image from device RAM: the header's scene_bytes after the root and the instance table packed below it (all-zero lines skipped). Still opt-in via VX_RTU_RAYLOG; the processor hands the logger its RAM at attach time. Co-Authored-By: Claude Opus 5.5 --- sim/simx/processor.cpp | 6 ++++ sim/simx/rtu/rtu_core.cpp | 2 +- sim/simx/rtu/rtu_raylog.cpp | 56 +++++++++++++++++++++++++++++++++++-- sim/simx/rtu/rtu_raylog.h | 16 +++++++++-- 4 files changed, 73 insertions(+), 7 deletions(-) diff --git a/sim/simx/processor.cpp b/sim/simx/processor.cpp index c160d1f829..0aa304656f 100644 --- a/sim/simx/processor.cpp +++ b/sim/simx/processor.cpp @@ -16,6 +16,9 @@ #include "core.h" #include "scheduler.h" #include +#ifdef VX_CFG_EXT_RTU_ENABLE +#include "rtu_raylog.h" +#endif #include #include @@ -199,6 +202,9 @@ ProcessorImpl::~ProcessorImpl() { void ProcessorImpl::attach_ram(RAM* ram) { ram_ = ram; memsim_->attach_ram(ram); +#ifdef VX_CFG_EXT_RTU_ENABLE + rtu::raylog::attach_ram(ram); +#endif } void ProcessorImpl::flush_caches() { diff --git a/sim/simx/rtu/rtu_core.cpp b/sim/simx/rtu/rtu_core.cpp index 4f75f5b1a6..8c2872fdb9 100644 --- a/sim/simx/rtu/rtu_core.cpp +++ b/sim/simx/rtu/rtu_core.cpp @@ -357,7 +357,7 @@ class RtuCore::Impl { if (s.req.dir_z[first_active] < 0.f) sig |= 0x4; s.coh_signature = sig; } - if (raylog::enabled()) raylog::on_accept(this, idx); + if (raylog::enabled()) raylog::on_accept(this, idx, s.req); ch.pop(); ++perf_stats_.rays_issued; DT(3, "rtu-core accept: tag=" << s.req.tag << ", slot=" << idx); diff --git a/sim/simx/rtu/rtu_raylog.cpp b/sim/simx/rtu/rtu_raylog.cpp index e42e23d36f..e0a6b5e0d5 100644 --- a/sim/simx/rtu/rtu_raylog.cpp +++ b/sim/simx/rtu/rtu_raylog.cpp @@ -18,6 +18,8 @@ #include #include #include +#include +#include "mem.h" namespace vortex { namespace rtu { namespace raylog { @@ -64,9 +66,17 @@ class Logger { bool on() const { return fp_ != nullptr; } - void accept(const void* owner, uint32_t slot) { + void set_ram(const RAM* ram) { + std::lock_guard g(mu_); + ram_ = ram; + } + + void accept(const void* owner, uint32_t slot, const RtuReq& req) { std::lock_guard g(mu_); cb_[{owner, slot}].fill(0); + for (uint32_t t = 0; t < VX_CFG_NUM_THREADS; ++t) { + if (req.tmask_bits & (1u << t)) snapshot_scene(req.scene_root[t]); + } } void callback(const void* owner, uint32_t slot, uint32_t lane, uint32_t cb_type) { @@ -143,7 +153,43 @@ class Logger { private: bool full() const { return logged_ >= max_rays_; } + // Instance records sit below the scene root, one stride per instance id. + static constexpr uint64_t kInstTableSpan = 64 * 1024; + static constexpr uint64_t kMaxSceneBytes = 1ull << 30; + + void snapshot_scene(uint32_t root) { + if (ram_ == nullptr || full() || !snapped_.insert(root).second) return; + const RAM& ram = *ram_; + uint32_t scene_bytes = 0; + for (int i = 0; i < 4; ++i) scene_bytes |= uint32_t(ram[root + 8 + i]) << (8 * i); + const uint64_t lo = (root > kInstTableSpan ? root - kInstTableSpan : 0) & kRtuLineMask; + const uint64_t hi = uint64_t(root) + std::min(scene_bytes, kMaxSceneBytes); + uint64_t added = 0; + for (uint64_t a = lo; a < hi; a += VX_CFG_MEM_BLOCK_SIZE) { + if (image_.count(a)) continue; + LineBuf line; + bool nonzero = false; + for (uint32_t i = 0; i < VX_CFG_MEM_BLOCK_SIZE; ++i) { + line[i] = ram[a + i]; + nonzero |= line[i] != 0; + } + if (!nonzero) continue; + image_.emplace(a, line); + RaylogLine r{}; + r.type = REC_LINE; + r.epoch = epoch_; + r.addr = a; + std::memcpy(r.data, line.data(), sizeof(r.data)); + std::fwrite(&r, sizeof(r), 1, fp_); + ++added; + } + std::fprintf(stderr, "[rtu-raylog] scene 0x%x: %u bytes, snapshot %llu lines\n", + root, scene_bytes, (unsigned long long)added); + } + std::mutex mu_; + const RAM* ram_ = nullptr; + std::unordered_set snapped_; FILE* fp_ = nullptr; unsigned long long max_rays_ = 4ull << 20; unsigned long long every_ = 1; @@ -167,8 +213,12 @@ bool enabled() { return on; } -void on_accept(const void* owner, uint32_t slot) { - logger().accept(owner, slot); +void attach_ram(const RAM* ram) { + if (enabled()) logger().set_ram(ram); +} + +void on_accept(const void* owner, uint32_t slot, const RtuReq& req) { + logger().accept(owner, slot, req); } void on_callback(const void* owner, uint32_t slot, uint32_t lane, uint32_t cb_type) { diff --git a/sim/simx/rtu/rtu_raylog.h b/sim/simx/rtu/rtu_raylog.h index c2d0ba2a7f..9f64b62eb1 100644 --- a/sim/simx/rtu/rtu_raylog.h +++ b/sim/simx/rtu/rtu_raylog.h @@ -25,6 +25,11 @@ // Optional knobs: // VX_RTU_RAYLOG_MAX= stop after n RAY records (default 4M) // VX_RTU_RAYLOG_EVERY= keep one terminal ray in n (default 1) +// +// The lines a SimX walk reads are not all another model may read: one that +// culls differently descends into nodes SimX never fetched. So the first trace +// against a scene also snapshots its whole image from device memory — the +// header's scene_bytes after the root, and the instance table packed below it. #ifndef _VX_RTU_RAYLOG_H_ #define _VX_RTU_RAYLOG_H_ @@ -34,7 +39,9 @@ #include #include "rtu_types.h" -namespace vortex { namespace rtu { namespace raylog { +namespace vortex { +class RAM; +namespace rtu { namespace raylog { constexpr uint32_t kMagic = 0x4C525856; // "VXRL" constexpr uint32_t kVersion = 1; @@ -93,9 +100,12 @@ static_assert(sizeof(RaylogRay) == 112, "RaylogRay layout"); bool enabled(); +// The device memory the scene snapshots read from. +void attach_ram(const RAM* ram); + // A trace landed in `slot` of the RTU at `owner`: forget the callbacks the -// slot's previous trace raised. -void on_accept(const void* owner, uint32_t slot); +// slot's previous trace raised, and snapshot any scene it is the first to name. +void on_accept(const void* owner, uint32_t slot, const RtuReq& req); // A lane of `slot` yielded a callback of `cb_type`. void on_callback(const void* owner, uint32_t slot, uint32_t lane, uint32_t cb_type); From dacdb9b0daa20eda9228168c5755f9f5986166fd Mon Sep 17 00:00:00 2001 From: Blaise Tine Date: Sat, 3 Oct 2026 13:01:59 -0700 Subject: [PATCH 20/31] rtu: remove the near-tie visit-order oracle The oracle replayed the Vulkan reference's (lavapipe's) own BVH -- appended by the driver as "visit-order tables" -- to decide which of two near-equal-t opaque hits lavapipe would keep, and culled child boxes against a widened bound near_t(best_t) = best_t + |best_t|*2^-19 so every such hit stayed reachable. Which hit a traversal reports among (near-)coincident ones is implementation-defined (Vulkan: "If t < t_max ... the candidate is set as the current closest hit"); mimicking one implementation's tree order is not a hardware feature, it is parity emulation of the CPU reference. - RTL: VX_rtu_oracle, VX_rtu_near_t and VX_rtu_f32_round are gone, with the scheduler's CS_ORC_REQ/CS_ORC_WAIT states, the oracle's memory-port and line-buffer sharing, the table pointers / ranks / vertices in the context word, the near-window precompute at ALIGN, and the 6-cycle delay of the tri result event that carried near_t. Node children are culled at best_t. - SimX: the LCA climb (src_keeps_a and friends), near_t / box_cull_t and every visit-order-table read are gone from rtu_walker.cpp; the TriPe cost model drops the near_t stages. - Leaf headers: LeafTri/LeafInst flags and LeafInst geometry_index/prim_base no longer carry table words; they are reserved and ignored. - tests/raytracing/rt_smoke_tie only exercised the tables; removed. - Lint: mark the box PE's early tag offset bits unused (pre-existing warning). The exact-t static (instance, geometry, primitive) key is still in place; it goes in the next change. Co-Authored-By: Claude Opus 5.5 --- hw/rtl/rtu/VX_rtu_f32_round.sv | 127 --- hw/rtl/rtu/VX_rtu_near_t.sv | 126 --- hw/rtl/rtu/VX_rtu_oracle.sv | 1224 ---------------------- hw/rtl/rtu/VX_rtu_scheduler.sv | 250 +---- sim/simx/rtu/rtu_bvh.h | 12 +- sim/simx/rtu/rtu_isect.cpp | 5 +- sim/simx/rtu/rtu_walker.cpp | 303 +----- tests/raytracing/Makefile | 2 +- tests/raytracing/rt_smoke_tie/Makefile | 22 - tests/raytracing/rt_smoke_tie/common.h | 44 - tests/raytracing/rt_smoke_tie/golden.h | 522 --------- tests/raytracing/rt_smoke_tie/kernel.cpp | 41 - tests/raytracing/rt_smoke_tie/main.cpp | 671 ------------ 13 files changed, 38 insertions(+), 3311 deletions(-) delete mode 100644 hw/rtl/rtu/VX_rtu_f32_round.sv delete mode 100644 hw/rtl/rtu/VX_rtu_near_t.sv delete mode 100644 hw/rtl/rtu/VX_rtu_oracle.sv delete mode 100644 tests/raytracing/rt_smoke_tie/Makefile delete mode 100644 tests/raytracing/rt_smoke_tie/common.h delete mode 100644 tests/raytracing/rt_smoke_tie/golden.h delete mode 100644 tests/raytracing/rt_smoke_tie/kernel.cpp delete mode 100644 tests/raytracing/rt_smoke_tie/main.cpp diff --git a/hw/rtl/rtu/VX_rtu_f32_round.sv b/hw/rtl/rtu/VX_rtu_f32_round.sv deleted file mode 100644 index 7f5ddae7c3..0000000000 --- a/hw/rtl/rtu/VX_rtu_f32_round.sv +++ /dev/null @@ -1,127 +0,0 @@ -// Copyright © 2019-2023 -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -// VX_rtu_f32_round — rounds the positive value (mag + sticky) * 2^exp to F32, -// nearest even, subnormals and overflow to infinity included. `sticky` stands -// for a nonzero remainder below mag's LSB; callers keep at least two bits of -// mag below the result's LSB whenever it is set. Returns the magnitude bits -// {exponent, fraction}; mag == 0 gives +0. LATENCY 0 is combinational; 4 -// registers the leading one, the alignment, the round increment and the -// result. - -`include "VX_define.vh" - -module VX_rtu_f32_round #( - parameter WB = 44, // mag width - parameter EW = 11, // signed exponent width - parameter LATENCY = 0 // 0 or 4 -) ( - input wire clk, - input wire enable, - input wire [WB-1:0] mag, - input wire signed [EW-1:0] exp, - input wire sticky, - output wire [30:0] result -); - `STATIC_ASSERT(((LATENCY == 0) || (LATENCY == 4)), ("invalid LATENCY")) - localparam IW = `CLOG2(WB); - localparam BW = WB + 12; - localparam EX = EW + 2; - - // ── stage 0: leading one, the LSB exponent and the shift ────────── - reg [IW-1:0] msb; - always @(*) begin - msb = '0; - for (integer i = 0; i < WB; ++i) begin - if (mag[i]) msb = IW'(i); - end - end - - // the result LSB's exponent, clamped at the subnormal LSB 2^-149 - wire signed [EX-1:0] lsb_raw = EX'(exp) + EX'($signed({1'b0, msb})) - EX'(23); - wire signed [EX-1:0] s0_lsb = (lsb_raw < -EX'(149)) ? -EX'(149) : lsb_raw; - wire signed [EX-1:0] s0_sh = s0_lsb - EX'(exp); - - // ── stage 1: alignment, guard/sticky ────────────────────────────── - wire [WB-1:0] s1_mag; - wire s1_sticky; - wire signed [EX-1:0] lsb, sh; - VX_pipe_register #( - .DATAW (WB + 1 + 2 * EX), - .DEPTH ((LATENCY != 0) ? 1 : 0) - ) pipe0 ( - .clk (clk), - .reset (1'b0), - .enable (enable), - .data_in ({mag, sticky, s0_lsb, s0_sh}), - .data_out ({s1_mag, s1_sticky, lsb, sh}) - ); - - // sh <= 0: exact, shifted up; sh > 0: drop sh bits, guard + sticky - wire exact = (sh <= 0); - wire [IW-1:0] ush = exact ? IW'(-sh) : IW'(sh); - wire [IW-1:0] gi = exact ? '0 : IW'(sh - EX'(1)); - wire [WB-1:0] q = s1_mag >> ush; - wire [WB-1:0] low = s1_mag & ((WB'(1) << gi) - WB'(1)); - wire [BW-1:0] s1_sig0 = exact ? (BW'(s1_mag) << ush) : BW'(q); - wire s1_inc = !exact && s1_mag[gi] && ((low != '0) || s1_sticky || q[0]); - wire [BW-1:0] s1_ebits = BW'(lsb + EX'(149)) << 23; - wire s1_zero = (s1_mag == '0); - - // ── stage 2: round increment ────────────────────────────────────── - wire [BW-1:0] s2_sig0, s2_ebits; - wire s2_inc, s2_zero; - VX_pipe_register #( - .DATAW (2 * BW + 2), - .DEPTH ((LATENCY != 0) ? 1 : 0) - ) pipe1 ( - .clk (clk), - .reset (1'b0), - .enable (enable), - .data_in ({s1_sig0, s1_ebits, s1_inc, s1_zero}), - .data_out ({s2_sig0, s2_ebits, s2_inc, s2_zero}) - ); - wire [BW-1:0] s2_sig = s2_sig0 + BW'(s2_inc); - - // ── stage 3: {biased exponent of the LSB position, significand}: a - // carry out of the significand bumps the exponent, a subnormal reaching - // 2^23 becomes normal ───────────────────────────────────────────── - wire [BW-1:0] s3_sig, s3_ebits; - wire s3_zero; - VX_pipe_register #( - .DATAW (2 * BW + 1), - .DEPTH ((LATENCY != 0) ? 1 : 0) - ) pipe2 ( - .clk (clk), - .reset (1'b0), - .enable (enable), - .data_in ({s2_sig, s2_ebits, s2_zero}), - .data_out ({s3_sig, s3_ebits, s3_zero}) - ); - wire [BW-1:0] s3_bits = s3_ebits + s3_sig; - wire [30:0] s3_res = s3_zero ? 31'd0 - : (s3_bits >= BW'(32'h7f800000)) ? 31'h7f800000 - : s3_bits[30:0]; - - VX_pipe_register #( - .DATAW (31), - .DEPTH ((LATENCY != 0) ? 1 : 0) - ) pipe3 ( - .clk (clk), - .reset (1'b0), - .enable (enable), - .data_in (s3_res), - .data_out (result) - ); - -endmodule diff --git a/hw/rtl/rtu/VX_rtu_near_t.sv b/hw/rtl/rtu/VX_rtu_near_t.sv deleted file mode 100644 index 6225e78b23..0000000000 --- a/hw/rtl/rtu/VX_rtu_near_t.sv +++ /dev/null @@ -1,126 +0,0 @@ -// Copyright © 2019-2023 -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -// VX_rtu_near_t — the near-hit window bound t + |t| * 2^-19 in F32, both -// operations rounded to nearest even, subnormals included. Two opaque hits -// within it of each other may be ordered either way by the source BVH's F32 -// box cull, so the walker settles them by its visit order. -// -// |t| * 2^-19 is exact unless it lands in the subnormal range, so the sum is -// formed as one integer W * 2^p (the scaled term pre-rounded where it is not -// exact) and rounded once. LATENCY 0 is combinational; 6 registers the scaled -// term, the sum and the four rounding steps. - -`include "VX_define.vh" - -module VX_rtu_near_t #( - parameter LATENCY = 0 // 0 or 6 -) ( - input wire clk, - input wire enable, - input wire [31:0] t, - output wire [31:0] result -); - `STATIC_ASSERT(((LATENCY == 0) || (LATENCY == 6)), ("invalid LATENCY")) - localparam WB = 44; - localparam PD = (LATENCY != 0) ? 1 : 0; - - // ── stage 1: decode, the scaled term ────────────────────────────── - wire s = t[31]; - wire [7:0] e = t[30:23]; - wire [22:0] f = t[22:0]; - - // the result when t is not a finite nonzero value - wire s1_spec = (e == 8'hff) || ((e == 8'h00) && (f == 23'd0)); - wire [31:0] s1_sval = (e != 8'hff) ? 32'h00000000 - : (f != 23'd0) ? (t | 32'h00400000) - : (s ? 32'hffc00000 : 32'h7f800000); - - // |t| = m * 2^q, q = e - 150 (normal) or -149 (subnormal) - wire [23:0] m = {(e != 8'h00), f}; - // |t| * 2^-19 = (m / 2^d) * 2^p: d > 0 only when it is subnormal, where - // its integer m / 2^d is rounded as the F32 multiply rounds it - wire [4:0] d = (e >= 8'd20) ? 5'd0 : ((e == 8'h00) ? 5'd19 : 5'(8'd20 - e)); - wire signed [10:0] s1_p = (d == 5'd0) ? (11'($signed({1'b0, e})) - 11'sd169) : -11'sd149; - - wire [23:0] q0 = m >> d; - wire [4:0] gi = (d == 5'd0) ? 5'd0 : (d - 5'd1); - wire g = (d != 5'd0) && m[gi]; - wire st = (m & ((24'd1 << gi) - 24'd1)) != 24'd0; - wire [23:0] s1_y = q0 + 24'(g && (st || q0[0])); - wire [WB-1:0] s1_mal = WB'(m) << (5'd19 - d); - - // ── stage 2: the exact sum ──────────────────────────────────────── - wire s2_s, s2_spec; - wire [31:0] s2_sval; - wire signed [10:0] s2_p; - wire [23:0] s2_y; - wire [WB-1:0] s2_mal; - VX_pipe_register #( - .DATAW (1 + 1 + 32 + 11 + 24 + WB), - .DEPTH (PD) - ) pipe1 ( - .clk (clk), - .reset (1'b0), - .enable (enable), - .data_in ({s, s1_spec, s1_sval, s1_p, s1_y, s1_mal}), - .data_out ({s2_s, s2_spec, s2_sval, s2_p, s2_y, s2_mal}) - ); - wire [WB-1:0] s2_w = s2_s ? (s2_mal - WB'(s2_y)) : (s2_mal + WB'(s2_y)); - - wire s3_s, s3_spec; - wire [31:0] s3_sval; - wire signed [10:0] s3_p; - wire [WB-1:0] s3_w; - VX_pipe_register #( - .DATAW (1 + 1 + 32 + 11 + WB), - .DEPTH (PD) - ) pipe2 ( - .clk (clk), - .reset (1'b0), - .enable (enable), - .data_in ({s2_s, s2_spec, s2_sval, s2_p, s2_w}), - .data_out ({s3_s, s3_spec, s3_sval, s3_p, s3_w}) - ); - - // ── stages 3..6: the rounding, with the special result alongside ── - wire [30:0] mag; - VX_rtu_f32_round #( - .WB (WB), - .EW (11), - .LATENCY ((LATENCY != 0) ? 4 : 0) - ) round ( - .clk (clk), - .enable (enable), - .mag (s3_w), - .exp (s3_p), - .sticky (1'b0), - .result (mag) - ); - - wire o_s, o_spec; - wire [31:0] o_sval; - VX_pipe_register #( - .DATAW (1 + 1 + 32), - .DEPTH ((LATENCY != 0) ? 4 : 0) - ) pipe_side ( - .clk (clk), - .reset (1'b0), - .enable (enable), - .data_in ({s3_s, s3_spec, s3_sval}), - .data_out ({o_s, o_spec, o_sval}) - ); - - assign result = o_spec ? o_sval : {o_s, mag}; - -endmodule diff --git a/hw/rtl/rtu/VX_rtu_oracle.sv b/hw/rtl/rtu/VX_rtu_oracle.sv deleted file mode 100644 index cc2afa728b..0000000000 --- a/hw/rtl/rtu/VX_rtu_oracle.sv +++ /dev/null @@ -1,1224 +0,0 @@ -// Copyright © 2019-2023 -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -// VX_rtu_oracle — the visit-order oracle: of two opaque triangle hits within -// the near window of each other (VX_rtu_near_t), which one the source BVH's -// traversal keeps. The source traversal (the Vulkan reference) walks its -// binary tree depth-first, nearer child box first, child 0 on equal entry -// distances, tests each box in F32 against the hit committed when the box's -// parent is visited, and commits only a strictly nearer hit. So of two such -// hits the first-reached wins at equal t, and a nearer hit reached second is -// lost when a box on its own path, below where the two paths part, enters at -// or past the first one's t. The driver lays the source trees out as -// visit-order tables (TLAS in the scene, BLAS tables reached by scene offset): -// -// TLAS { n_leaves, n_nodes, nodes_off, leaf_stride } -// leaf[i] @ +16 + i*stride : { parent << 1 | side, _, _, _, world->object 3x4 } -// node[j] @ +nodes_off + j*64 : { child box 0, child box 1, parent << 1 | side, depth } -// BLAS { n_nodes, _, nodes_off, 32 } -// node[j] @ +nodes_off + j*32 : { own box, parent << 1 | side, depth } -// -// a triangle's box being its vertices' F32 min/max. Both leaves climb to -// their lowest common ancestor, every box on the way tested against the other -// hit's t; the two child boxes under it are ordered by their entry distance. -// Across instances the TLAS is climbed with the world ray, and a lost-by-cull -// verdict also climbs the nearer hit's own BLAS path with the reference's -// object-space ray, rebuilt from the table's matrix in its own op order. -// -// One request at a time; the requesting context parks until `done`. The -// engine is a small sequencer around one F32 add/mul unit (IEEE, subnormals), -// three serial correctly-rounded reciprocals and a 64-word LUTRAM register -// file. Table words are fetched a line at a time on the scheduler's memory -// port, under the requesting context's tag, by a fetch unit that runs beside -// the sequencer, so a climb step's parent fetch overlaps its box test. Cost is -// O(tree depth) table reads and box tests, paid only on near ties. - -`include "VX_define.vh" - -module VX_rtu_oracle import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( - parameter CTX_TAG_W = 1, - parameter ADDRW = `VX_CFG_MEM_ADDR_WIDTH, - parameter LINE_BITS = `VX_CFG_MEM_BLOCK_SIZE * 8, - parameter FMA_LAT = RTU_LATENCY_FMA -) ( - input wire clk, - input wire reset, - // the scene may change: forget the ray set up last (asserted while the - // RTU is idle, which it always is between dependent launches) - input wire flush, - - // request: hit a (the new one) against hit b (the committed one) - input wire req_valid, - output wire req_ready, - input wire [CTX_TAG_W-1:0] req_ctx, - input wire [ADDRW-1:0] req_scene, - input wire [2:0][31:0] req_wo, // world ray - input wire [2:0][31:0] req_wd, - input wire [31:0] req_tlas, // TLAS table (scene offset) - input wire [31:0] req_a_t, - input wire [31:0] req_a_inst, // TLAS leaf rank - input wire [31:0] req_a_ps, // parent << 1 | side in its BLAS table - input wire [31:0] req_a_tab, // its BLAS table (scene offset) - input wire [8:0][31:0] req_a_v, // v0.xyz, v1.xyz, v2.xyz - input wire [31:0] req_b_t, - input wire [31:0] req_b_inst, - input wire [31:0] req_b_ps, - input wire [31:0] req_b_tab, - input wire [8:0][31:0] req_b_v, - - // verdict: true when the source traversal keeps hit a - output wire done_valid, - output wire [CTX_TAG_W-1:0] done_ctx, - output wire done_keep, - - // table fetch, one line in flight - output wire mem_req_valid, - output wire [ADDRW-1:0] mem_req_addr, - input wire mem_req_ready, - input wire mem_rsp_valid, - input wire [LINE_BITS-1:0] mem_rsp_data -); - localparam LINE_BYTES = LINE_BITS / 8; - localparam WPL = LINE_BYTES / 4; // words per line - localparam WIDXW = `CLOG2(2 * WPL); - localparam LSELW = `CLOG2(LINE_BYTES); - - `STATIC_ASSERT((LINE_BYTES >= 64), ("table fetches assume >= 64-byte lines")) - - localparam [31:0] ROOT = 32'hffffffff; - localparam [31:0] F_INF = 32'h7f800000; - localparam [31:0] F_MAX = 32'h7f7fffff; - - // Climb steps one verdict may take. Valid tables reach the root in tree - // depth steps; a malformed table (a parent cycle) must still not hang the - // context, so past the cap the verdict falls to the nearer hit. - localparam MAX_CLIMBS = 4096; - localparam CLIMBW = `CLOG2(MAX_CLIMBS + 1); - - // ── register file map ───────────────────────────────────────────── - localparam [5:0] RA_WO = 6'd0, // world ray o, d - RA_WD = 6'd3, - RA_RO = 6'd6, // current ray o, d, 1/d - RA_RD = 6'd9, - RA_RI = 6'd12, - RA_BA = 6'd15, // hit a / b vertex boxes (min.xyz, max.xyz) - RA_BB = 6'd21, - RA_CB0 = 6'd27, // climb boxes - RA_CB1 = 6'd33, - RA_M = 6'd27, // object-ray matrix (aliases the climb boxes) - RA_S = 6'd39, // slab distances - RA_X = 6'd45; // the object ray's products - - // ── sequencer states ────────────────────────────────────────────── - localparam [6:0] - S_IDLE = 7'd0, S_CAP = 7'd1, S_BOX1 = 7'd2, S_BOX2 = 7'd3, - S_BOX3 = 7'd4, S_DISP = 7'd5, S_DONE = 7'd6, - // same instance - T_S1 = 7'd8, T_S2 = 7'd9, T_S3 = 7'd10, T_S4 = 7'd11, - T_S5 = 7'd12, T_S6 = 7'd13, T_S7 = 7'd14, - // different instances - T_D0 = 7'd16, T_D1 = 7'd17, T_D2 = 7'd18, T_D3 = 7'd19, - T_D3B = 7'd20, T_D4 = 7'd21, T_D5 = 7'd22, T_D6 = 7'd23, - T_D7 = 7'd24, T_D8 = 7'd25, T_D9 = 7'd26, T_D10 = 7'd27, - T_D11 = 7'd28, T_D12 = 7'd29, - // a hit's BLAS path cull - P_0 = 7'd32, P_1 = 7'd33, P_2 = 7'd34, P_3 = 7'd35, - // lowest common ancestor - L_0 = 7'd40, L_1 = 7'd41, L_2 = 7'd42, - // one climb step - U_0 = 7'd44, U_1 = 7'd45, U_2 = 7'd46, U_3 = 7'd47, - // box test - B_SUB = 7'd48, B_MUL = 7'd49, B_RED = 7'd50, B_FIN0 = 7'd51, - B_FIN = 7'd52, - // object ray - O_0 = 7'd56, O_1 = 7'd57, O_2 = 7'd58, O_3 = 7'd59, - O_4 = 7'd60, O_5 = 7'd61, O_6 = 7'd62, O_7 = 7'd63, - O_8 = 7'd64, - // reciprocals - R_LD0 = 7'd68, R_LD1 = 7'd69, R_LD2 = 7'd67, R_IT = 7'd70, - R_WR = 7'd71, - // fetch wait / fetched-word capture / move / copy / multiply / drain - F_WAIT = 7'd72, F_CAP = 7'd73, M_MV = 7'd76, C_CP = 7'd77, - X_MUL = 7'd78, W_DRAIN = 7'd79; - - // ── F32 helpers (C fmin/fmax, IEEE ordered compares) ────────────── - function automatic logic f_nan(input logic [30:0] a); - f_nan = (a[30:23] == 8'hff) && (a[22:0] != 23'd0); - endfunction - function automatic logic f_eq(input logic [31:0] a, input logic [31:0] b); - f_eq = !f_nan(a[30:0]) && !f_nan(b[30:0]) - && ((a == b) || ((a[30:0] == 31'd0) && (b[30:0] == 31'd0))); - endfunction - function automatic logic f_lt(input logic [31:0] a, input logic [31:0] b); - if (f_nan(a[30:0]) || f_nan(b[30:0]) || ((a[30:0] == 31'd0) && (b[30:0] == 31'd0))) begin - f_lt = 1'b0; - end else if (a[31] != b[31]) begin - f_lt = a[31]; - end else if (!a[31]) begin - f_lt = (a[30:0] < b[30:0]); - end else begin - f_lt = (a[30:0] > b[30:0]); - end - endfunction - function automatic logic [31:0] f_min(input logic [31:0] a, input logic [31:0] b); - f_min = f_nan(a[30:0]) ? b : (f_nan(b[30:0]) ? a : (f_lt(b, a) ? b : a)); - endfunction - function automatic logic [31:0] f_max(input logic [31:0] a, input logic [31:0] b); - f_max = f_nan(a[30:0]) ? b : (f_nan(b[30:0]) ? a : (f_lt(a, b) ? b : a)); - endfunction - function automatic logic [5:0] mod3(input logic [4:0] k); - mod3 = (k >= 5'd3) ? 6'(k - 5'd3) : 6'(k); - endfunction - - // the scene offset of a table node - function automatic logic [31:0] node_off(input logic [31:0] tab, input logic [31:0] noff, - input logic tl, input logic [31:0] ps); - node_off = tab + noff + (tl ? ((ps >> 1) << 6) : ((ps >> 1) << 5)); - endfunction - - // ── state ───────────────────────────────────────────────────────── - reg [6:0] state; - reg [6:0] f_ret, b_ret, u_ret, l_ret, o_ret, r_ret, w_ret, m_ret, c_ret, x_ret; - - reg [CTX_TAG_W-1:0] ctx_r; - reg [ADDRW-1:0] scene_r; - reg [23:0][31:0] cap; - reg [31:0] tlas_r, ta, tb, ia, ib, psa, psb, taba, tabb; - reg [31:0] tab_r, noff_r, stride_r; - reg is_tlas; - reg [31:0] ps0, ps1, dep0, dep1; - reg cul0, cul1; - reg cur; // the climb UP moves - reg [31:0] upt; // ... and the t its boxes are culled against - reg a_first, keep, ph; - reg [31:0] key0; - - // The current ray (RO/RD/RI) as last set up: the world ray, or a TLAS - // leaf's object ray. Verdicts for one walk come in runs on the same ray, - // so a request whose ray is the one already set up skips the setup. - reg [2:0][31:0] cw_o, cw_d; // this request's world ray - reg rk_valid, rk_world; - reg [31:0] rk_inst, rk_tlas; - reg [2:0][31:0] rk_o, rk_d; - reg ray_same; // rk was set up for this request's world ray - reg [CLIMBW-1:0] climbs; - - // box test - reg [5:0] bt_base; - reg [31:0] bt_tmax; - reg bt_nan; - reg [31:0] bt_lo, bt_hi, bt_key; - reg bt_pass; - reg [2:0] bt_wbc; // slab differences written back so far - reg [4:0] k; // the running routine counter - reg [31:0] red_a, red_b, red_mn, red_mx, bt_lo0; - reg [31:0] tmn, tmx; // vertex box min/max partials - - // object ray / multiply - reg [31:0] o_inst; - reg [31:0] mul_a, mul_b, mul_p; - - // move / copy - reg [5:0] mv_dst, cp_src; - reg [4:0] mv_k0, mv_n; - - // ── register file: two LUTRAM copies for two read ports ────────── - reg [5:0] ra, rb; - wire [31:0] rda, rdb; - reg fsm_we; - reg [5:0] fsm_wa; - reg [31:0] fsm_wd; - wire wb_v; - wire [5:0] wb_dst; - wire [31:0] fma_res; - wire rf_we = wb_v || fsm_we; - wire [5:0] rf_wa = wb_v ? wb_dst : fsm_wa; - wire [31:0] rf_wd = wb_v ? fma_res : fsm_wd; - - VX_dp_ram #( - .DATAW (32), - .SIZE (64), - .LUTRAM (1), - .OUT_REG (0) - ) rf_a ( - .clk (clk), - .reset (reset), - .read (1'b1), - .write (rf_we), - .wren (1'b1), - .waddr (rf_wa), - .wdata (rf_wd), - .raddr (ra), - .rdata (rda) - ); - VX_dp_ram #( - .DATAW (32), - .SIZE (64), - .LUTRAM (1), - .OUT_REG (0) - ) rf_b ( - .clk (clk), - .reset (reset), - .read (1'b1), - .write (rf_we), - .wren (1'b1), - .waddr (rf_wa), - .wdata (rf_wd), - .raddr (rb), - .rdata (rdb) - ); - - // ── F32 add / sub / mul (IEEE, subnormals) ─────────────────────── - reg fma_issue; - reg [1:0] fma_kind; // 0: add, 1: sub, 2: mul - reg [5:0] fma_dst; - reg [5:0] fma_pend; - - // operands are registered off the register file's asynchronous read - reg fx_issue; - reg [1:0] fx_kind; - reg [5:0] fx_dst; - reg [31:0] fx_a, fx_b; - always_ff @(posedge clk) begin - if (reset) begin - fx_issue <= 1'b0; - end else begin - fx_issue <= fma_issue; - end - fx_kind <= fma_kind; - fx_dst <= fma_dst; - fx_a <= rda; - fx_b <= rdb; - end - - VX_fma_unit #( - .LATENCY (FMA_LAT), - .USE_DSP (`VX_CFG_RTU_USE_DSP), - .SUBNORM_ENABLE (1), - .EXCEPT_ENABLE (1) - ) fma ( - .clk (clk), - .reset (reset), - .enable (1'b1), - .mask (fx_issue), - .op_type ((fx_kind == 2'd2) ? INST_FPU_MUL : INST_FPU_ADD), - .fmt ((fx_kind == 2'd1) ? INST_FMT_BITS'(2'b10) : INST_FMT_BITS'(2'b00)), - .frm (INST_FRM_RNE), - .dataa (fx_a), - .datab (fx_b), - .datac (32'd0), - .result (fma_res), - `UNUSED_PIN (fflags) - ); - - VX_shift_register #( - .DATAW (1 + 6), - .RESETW (1), - .DEPTH (FMA_LAT) - ) fma_tags ( - .clk (clk), - .reset (reset), - .enable (1'b1), - .data_in ({fx_issue, fx_dst}), - .data_out ({wb_v, wb_dst}) - ); - - // ── fetch unit: f_n words at scene offset f_off, beside the sequencer ── - localparam [2:0] FS_IDLE = 3'd0, FS_REQ0 = 3'd1, FS_RSP0 = 3'd2, - FS_REQ1 = 3'd3, FS_RSP1 = 3'd4; - reg [2:0] fs_state; - reg fs_start; - reg [31:0] f_off; - reg [4:0] f_n; - reg [ADDRW-1:0] f_line; - reg [LSELW-3:0] f_w0; - reg f_two; - reg [LINE_BITS-1:0] lb0, lb1; - - // tables are word aligned: the low address bits are always zero - wire [ADDRW-1:0] f_addr = scene_r + ADDRW'(f_off); - `UNUSED_VAR (f_addr[1:0]) - - always_ff @(posedge clk) begin - if (reset) begin - fs_state <= FS_IDLE; - end else begin - case (fs_state) - FS_IDLE: begin - if (fs_start) begin - f_line <= {f_addr[ADDRW-1:LSELW], LSELW'(0)}; - f_w0 <= f_addr[LSELW-1:2]; - f_two <= (32'(f_addr[LSELW-1:2]) + 32'(f_n)) > WPL; - fs_state <= FS_REQ0; - end - end - FS_REQ0: if (mem_req_ready) fs_state <= FS_RSP0; - FS_RSP0: begin - if (mem_rsp_valid) begin - lb0 <= mem_rsp_data; - fs_state <= f_two ? FS_REQ1 : FS_IDLE; - end - end - FS_REQ1: if (mem_req_ready) fs_state <= FS_RSP1; - FS_RSP1: begin - if (mem_rsp_valid) begin - lb1 <= mem_rsp_data; - fs_state <= FS_IDLE; - end - end - default: fs_state <= FS_IDLE; - endcase - end - end - wire fs_done = (fs_state == FS_IDLE) && !fs_start; - - assign mem_req_valid = (fs_state == FS_REQ0) || (fs_state == FS_REQ1); - assign mem_req_addr = (fs_state == FS_REQ0) ? f_line : (f_line + ADDRW'(LINE_BYTES)); - - // fetched words: one read port over the two lines - reg [4:0] fw_k; - reg [4:0] f_k0; // the word F_CAP captures into fw_q - reg [31:0] fw_q; - wire [2*LINE_BITS-1:0] lbs = {lb1, lb0}; - wire [WIDXW-1:0] fw_idx = WIDXW'(f_w0) + WIDXW'(fw_k); - wire [31:0] fw = lbs[32'(fw_idx) * 32 +: 32]; - - // ── reciprocals: 2^50 / significand, 28 quotient bits, three lanes ─ - reg [2:0] rc_spec; // lane result is special (rc_sval) - reg [2:0][31:0] rc_sval; - reg [2:0][23:0] dv_m; - reg [2:0][10:0] dv_e; // signed exponent of the operand's significand LSB - reg [2:0] dv_s; - reg [2:0][24:0] dv_rem; - reg [2:0][27:0] dv_q; - reg [4:0] dv_it; - - // operand decode: 1/0 -> FLT_MAX (the reference's zero-direction - // reciprocal), 1/inf -> 0, 1/NaN -> NaN - function automatic logic [68:0] rc_decode(input logic [31:0] x); - logic [7:0] e; - logic [22:0] f; - logic [4:0] msb; - logic spec; - logic [31:0] sval; - logic [23:0] m; - logic [10:0] ex; - e = x[30:23]; - f = x[22:0]; - msb = '0; - for (integer i = 0; i < 23; ++i) begin - if (f[i]) msb = 5'(i); - end - spec = (x[30:0] == 31'd0) || (e == 8'hff); - sval = (x[30:0] == 31'd0) ? F_MAX - : ((f != 23'd0) ? (x | 32'h00400000) : {x[31], 31'd0}); - if (e == 8'h00) begin - m = 24'(f) << (5'd23 - msb); - ex = 11'({6'd0, msb}) - 11'd172; - end else begin - m = {1'b1, f}; - ex = 11'({3'd0, e}) - 11'd150; - end - rc_decode = {spec, sval, x[31], m, ex}; - endfunction - reg [2:0][31:0] rx; - wire [68:0] rc_dec0 = rc_decode(rx[0]); - wire [68:0] rc_dec1 = rc_decode(rx[1]); - wire [68:0] rc_dec2 = rc_decode(rx[2]); - - // R_WR registers lane k at k < 3 and writes lane k - 5 five cycles later - wire [1:0] rc_w = (k < 5'd3) ? 2'(k) : 2'd0; - wire [1:0] rc_o = (k >= 5'd5) ? 2'(k - 5'd5) : 2'd0; - reg [27:0] rc_in_q; - reg [10:0] rc_in_e; - reg rc_in_st; - always_ff @(posedge clk) begin - rc_in_q <= dv_q[rc_w]; - rc_in_e <= 11'(-11'sd50 - $signed(dv_e[rc_w])); - rc_in_st <= (dv_rem[rc_w] != 25'd0); - end - wire [30:0] rc_mag; - VX_rtu_f32_round #( - .WB (28), - .EW (11), - .LATENCY (4) - ) rc_round ( - .clk (clk), - .enable (1'b1), - .mag (rc_in_q), - .exp ($signed(rc_in_e)), - .sticky (rc_in_st), - .result (rc_mag) - ); - - // ── sequencer ───────────────────────────────────────────────────── - // O_4's product index k = (d ? 9 : 0) + i*3 + j - wire [4:0] o4_kk = (k >= 5'd9) ? (k - 5'd9) : k; - wire [5:0] o4_i = (o4_kk >= 5'd6) ? 6'd2 : ((o4_kk >= 5'd3) ? 6'd1 : 6'd0); - wire [5:0] o4_j = 6'(o4_kk) - o4_i * 6'd3; - - // S_BOX: hit k/3, axis k%3, straight off the captured vertices - wire [4:0] bx_h = (k >= 5'd3) ? 5'd9 : 5'd0; - wire [4:0] bx_a = 5'(mod3(k)); - wire [31:0] bx_v0 = cap[bx_h + bx_a]; - wire [31:0] bx_v1 = cap[bx_h + 5'd3 + bx_a]; - wire [31:0] bx_v2 = cap[bx_h + 5'd6 + bx_a]; - - // combinational controls - always @(*) begin - ra = '0; - rb = '0; - fma_issue = 1'b0; - fma_kind = 2'd0; - fma_dst = '0; - fsm_we = 1'b0; - fsm_wa = '0; - fsm_wd = '0; - fw_k = '0; - case (state) - S_CAP: begin - fsm_we = 1'b1; - fsm_wa = RA_WO + 6'(k); - fsm_wd = cap[0]; - end - S_BOX2: begin - fsm_we = 1'b1; - fsm_wa = ((k >= 5'd3) ? RA_BB : RA_BA) + 6'(bx_a); - fsm_wd = f_min(bx_v0, tmn); - end - S_BOX3: begin - fsm_we = 1'b1; - fsm_wa = ((k >= 5'd3) ? RA_BB : RA_BA) + 6'd3 + 6'(bx_a); - fsm_wd = f_max(bx_v0, tmx); - end - B_SUB: begin - ra = bt_base + 6'(k); - rb = RA_RO + mod3(k); - fma_issue = 1'b1; - fma_kind = 2'd1; - fma_dst = RA_S + 6'(k); - end - B_MUL: begin - // each product issues as soon as its difference is written back - ra = RA_S + 6'(k); - rb = RA_RI + mod3(k); - fma_issue = (5'(bt_wbc) > k); - fma_kind = 2'd2; - fma_dst = RA_S + 6'(k); - end - B_RED: begin - ra = RA_S + ((k < 5'd3) ? 6'(k) : 6'd0); - rb = RA_S + 6'd3 + ((k < 5'd3) ? 6'(k) : 6'd0); - end - O_4: begin - // products: X[i*3+j] = wo[j] * m[i][j], X[9+i*3+j] = wd[j] * m[i][j] - ra = ((k >= 5'd9) ? RA_WD : RA_WO) + o4_j; - rb = RA_M + o4_i * 6'd4 + o4_j; - fma_issue = 1'b1; - fma_kind = 2'd2; - fma_dst = RA_X + 6'(k); - end - O_5: begin - // ro[i] = m[i][3] + P[i][0]; rd[i] = Q[i][0] + Q[i][1] - if (k < 5'd3) begin - ra = RA_M + 6'(k) * 6'd4 + 6'd3; - rb = RA_X + 6'(k) * 6'd3; - fma_dst = RA_RO + 6'(k); - end else begin - ra = RA_X + 6'd9 + 6'(k - 5'd3) * 6'd3; - rb = RA_X + 6'd9 + 6'(k - 5'd3) * 6'd3 + 6'd1; - fma_dst = RA_RD + 6'(k - 5'd3); - end - fma_issue = 1'b1; - end - O_6: begin - // ro[i] += P[i][1]; rd[i] += Q[i][2] - if (k < 5'd3) begin - ra = RA_RO + 6'(k); - rb = RA_X + 6'(k) * 6'd3 + 6'd1; - fma_dst = RA_RO + 6'(k); - end else begin - ra = RA_RD + 6'(k - 5'd3); - rb = RA_X + 6'd9 + 6'(k - 5'd3) * 6'd3 + 6'd2; - fma_dst = RA_RD + 6'(k - 5'd3); - end - fma_issue = 1'b1; - end - O_7: begin - // ro[i] += P[i][2] - ra = RA_RO + 6'(k); - rb = RA_X + 6'(k) * 6'd3 + 6'd2; - fma_dst = RA_RO + 6'(k); - fma_issue = 1'b1; - end - R_LD0: begin - ra = RA_RD; - rb = RA_RD + 6'd1; - end - R_LD1: begin - ra = RA_RD + 6'd2; - end - R_WR: begin - fsm_we = (k >= 5'd5); - fsm_wa = RA_RI + 6'(rc_o); - fsm_wd = rc_spec[rc_o] ? rc_sval[rc_o] : {dv_s[rc_o], rc_mag}; - end - M_MV: begin - fw_k = mv_k0 + k; - fsm_we = 1'b1; - fsm_wa = mv_dst + 6'(k); - fsm_wd = fw; - end - C_CP: begin - ra = cp_src + 6'(k); - fsm_we = 1'b1; - fsm_wa = mv_dst + 6'(k); - fsm_wd = rda; - end - F_CAP: fw_k = f_k0; - default:; - endcase - end - - `RUNTIME_ASSERT(!(wb_v && fsm_we), ("%t: rtu oracle: register-file write conflict", $time)) - - wire [31:0] mul_sum = mul_p + (mul_a[0] ? mul_b : 32'd0); - - always_ff @(posedge clk) begin - fs_start <= 1'b0; - if (reset) begin - state <= S_IDLE; - fma_pend <= '0; - rk_valid <= 1'b0; - end else begin - fma_pend <= fma_pend + 6'(fma_issue) - 6'(wb_v); - if (flush) begin - rk_valid <= 1'b0; - end - if ((state == B_SUB) || (state == B_MUL)) begin - bt_wbc <= bt_wbc + 3'(wb_v); - end else begin - bt_wbc <= '0; - end - - case (state) - S_IDLE: begin - if (req_valid) begin - ctx_r <= req_ctx; - scene_r <= req_scene; - tlas_r <= req_tlas; - ta <= req_a_t; tb <= req_b_t; - ia <= req_a_inst; ib <= req_b_inst; - psa <= req_a_ps; psb <= req_b_ps; - taba <= req_a_tab; tabb <= req_b_tab; - cap <= {req_b_v, req_a_v, req_wd, req_wo}; - cw_o <= req_wo; - cw_d <= req_wd; - k <= '0; - climbs <= '0; - state <= S_CAP; - end - end - S_CAP: begin - // world ray into the register file; the vertices stay captured - cap <= cap >> 32; - k <= k + 5'd1; - if (k == 5'd5) begin - k <= '0; - state <= S_BOX1; - end - end - S_BOX1: begin - tmn <= f_min(bx_v1, bx_v2); - tmx <= f_max(bx_v1, bx_v2); - state <= S_BOX2; - end - S_BOX2: begin - state <= S_BOX3; - end - S_BOX3: begin - k <= k + 5'd1; - state <= S_BOX1; - if (k == 5'd5) begin - k <= '0; - state <= S_DISP; - end - end - S_DISP: begin - ray_same <= rk_valid && (rk_o == cw_o) && (rk_d == cw_d) && (rk_tlas == tlas_r); - state <= (ia == ib) ? T_S1 : T_D0; - end - - // ── same instance: climb its BLAS table ────────────────── - T_S1: begin - if ((psa == ROOT) || (psb == ROOT)) begin - keep <= f_lt(ta, tb) || f_eq(ta, tb); - state <= S_DONE; - end else begin - o_inst <= ia; - o_ret <= T_S2; - state <= O_0; - end - end - T_S2: begin - is_tlas <= 1'b0; - tab_r <= taba; - f_off <= taba + 32'd8; - f_n <= 5'd1; - fs_start <= 1'b1; - f_k0 <= '0; - f_ret <= T_S3; - state <= F_WAIT; - end - T_S3: begin - noff_r <= fw_q; - ps0 <= psa; - f_off <= node_off(tab_r, fw_q, 1'b0, psa) + 32'd28; - fs_start <= 1'b1; - f_k0 <= '0; - f_ret <= T_S4; - state <= F_WAIT; - end - T_S4: begin - dep0 <= fw_q + 32'd1; - ps1 <= psb; - f_off <= node_off(tab_r, noff_r, 1'b0, psb) + 32'd28; - fs_start <= 1'b1; - f_k0 <= '0; - f_ret <= T_S5; - state <= F_WAIT; - end - T_S5: begin - dep1 <= fw_q + 32'd1; - cul0 <= 1'b0; - cul1 <= 1'b0; - mv_dst <= RA_CB0; - cp_src <= RA_BA; - mv_n <= 5'd12; - c_ret <= T_S6; - state <= C_CP; - end - T_S6: begin - l_ret <= T_S7; - state <= L_0; - end - T_S7: begin - keep <= f_eq(ta, tb) ? a_first - : f_lt(ta, tb) ? (a_first || !cul0) - : (a_first && cul1); - state <= S_DONE; - end - - // ── different instances: climb the TLAS table ──────────── - T_D0: begin - if (ray_same && rk_world) begin - // the world ray is set up already - is_tlas <= 1'b1; - tab_r <= tlas_r; - f_off <= tlas_r + 32'd8; - f_n <= 5'd2; - fs_start <= 1'b1; - f_k0 <= '0; - state <= T_D2; - end else begin - mv_dst <= RA_RO; // world ray o, d -> current ray - cp_src <= RA_WO; - mv_n <= 5'd6; - c_ret <= T_D1; - state <= C_CP; - end - end - T_D1: begin - // the TLAS header fetch runs beside the reciprocals - is_tlas <= 1'b1; - tab_r <= tlas_r; - f_off <= tlas_r + 32'd8; - f_n <= 5'd2; - fs_start <= 1'b1; - f_k0 <= '0; - rk_valid <= 1'b1; - rk_world <= 1'b1; - rk_o <= cw_o; - rk_d <= cw_d; - rk_tlas <= tlas_r; - ray_same <= 1'b1; - r_ret <= T_D2; - state <= R_LD0; - end - T_D2: begin - f_ret <= T_D3; - state <= F_WAIT; - end - T_D3: begin - noff_r <= fw_q; - f_k0 <= 5'd1; - f_ret <= T_D3B; - state <= F_CAP; - end - T_D3B: begin - stride_r <= fw_q; - mul_a <= ia; - mul_b <= fw_q; - mul_p <= '0; - x_ret <= T_D4; - state <= X_MUL; - end - T_D4: begin - f_off <= tlas_r + 32'd16 + mul_p; - f_n <= 5'd1; - fs_start <= 1'b1; - f_k0 <= '0; - f_ret <= T_D5; - state <= F_WAIT; - end - T_D5: begin - ps0 <= fw_q; - if (fw_q == ROOT) begin - dep0 <= '0; - state <= T_D7; - end else begin - f_off <= node_off(tab_r, noff_r, 1'b1, fw_q); - f_n <= 5'd14; - fs_start <= 1'b1; - f_k0 <= 5'd13; - f_ret <= T_D6; - state <= F_WAIT; - end - end - T_D6: begin - dep0 <= fw_q + 32'd1; - mv_dst <= RA_CB0; - mv_k0 <= ps0[0] ? 5'd6 : 5'd0; - mv_n <= 5'd6; - m_ret <= T_D7; - state <= M_MV; - end - T_D7: begin - mul_a <= ib; - mul_b <= stride_r; - mul_p <= '0; - x_ret <= T_D8; - state <= X_MUL; - end - T_D8: begin - f_off <= tlas_r + 32'd16 + mul_p; - f_n <= 5'd1; - fs_start <= 1'b1; - f_k0 <= '0; - f_ret <= T_D9; - state <= F_WAIT; - end - T_D9: begin - ps1 <= fw_q; - if (fw_q == ROOT) begin - dep1 <= '0; - state <= T_D11; - end else begin - f_off <= node_off(tab_r, noff_r, 1'b1, fw_q); - f_n <= 5'd14; - fs_start <= 1'b1; - f_k0 <= 5'd13; - f_ret <= T_D10; - state <= F_WAIT; - end - end - T_D10: begin - dep1 <= fw_q + 32'd1; - mv_dst <= RA_CB1; - mv_k0 <= ps1[0] ? 5'd6 : 5'd0; - mv_n <= 5'd6; - m_ret <= T_D11; - state <= M_MV; - end - T_D11: begin - cul0 <= 1'b0; - cul1 <= 1'b0; - l_ret <= T_D12; - state <= L_0; - end - T_D12: begin - state <= S_DONE; - if (f_eq(ta, tb)) begin - keep <= a_first; - end else if (f_lt(ta, tb)) begin - if (a_first) begin - keep <= 1'b1; - end else if (cul0) begin - keep <= 1'b0; - end else begin - ph <= 1'b0; - state <= P_0; - end - end else begin - if (!a_first) begin - keep <= 1'b0; - end else if (cul1) begin - keep <= 1'b1; - end else begin - ph <= 1'b1; - state <= P_0; - end - end - end - - // ── the nearer hit's BLAS path, culled against the other t ── - P_0: begin - o_inst <= ph ? ib : ia; - o_ret <= P_1; - state <= O_0; - end - P_1: begin - is_tlas <= 1'b0; - tab_r <= ph ? tabb : taba; - f_off <= (ph ? tabb : taba) + 32'd8; - f_n <= 5'd1; - fs_start <= 1'b1; - f_k0 <= '0; - f_ret <= P_2; - state <= F_WAIT; - end - P_2: begin - noff_r <= fw_q; - ps0 <= ph ? psb : psa; - cul0 <= 1'b0; - cur <= 1'b0; - upt <= ph ? ta : tb; - mv_dst <= RA_CB0; - cp_src <= ph ? RA_BB : RA_BA; - mv_n <= 5'd6; - c_ret <= P_3; - state <= C_CP; - end - P_3: begin - // the verdict only needs whether some box fails - if ((ps0 == ROOT) || cul0) begin - keep <= ph ? cul0 : !cul0; - state <= S_DONE; - end else begin - u_ret <= P_3; - state <= U_0; - end - end - - // ── climb both leaves to their lowest common ancestor ──── - L_0: begin - u_ret <= L_0; - if (dep0 > dep1) begin - cur <= 1'b0; - upt <= tb; - state <= U_0; - end else if (dep1 > dep0) begin - cur <= 1'b1; - upt <= ta; - state <= U_0; - end else if (ps0[31:1] != ps1[31:1]) begin - cur <= 1'b0; - upt <= tb; - state <= U_0; - end else begin - bt_base <= RA_CB0; - b_ret <= L_1; - state <= B_SUB; - end - end - L_1: begin - key0 <= bt_key; - bt_base <= RA_CB1; - b_ret <= L_2; - state <= B_SUB; - end - L_2: begin - // child 1 first only on a strictly nearer entry - a_first <= (ps0[0] == f_lt(ps0[0] ? key0 : bt_key, ps0[0] ? bt_key : key0)); - state <= l_ret; - end - - // ── one climb step: test the box, move to the parent ───── - U_0: begin - // the parent link's fetch runs beside the box test - climbs <= climbs + CLIMBW'(1); - if (climbs == CLIMBW'(MAX_CLIMBS)) begin - keep <= f_lt(ta, tb); - end - bt_base <= cur ? RA_CB1 : RA_CB0; - bt_tmax <= upt; - f_off <= node_off(tab_r, noff_r, is_tlas, cur ? ps1 : ps0); - f_n <= is_tlas ? 5'd14 : 5'd8; - fs_start <= (climbs != CLIMBW'(MAX_CLIMBS)); - f_k0 <= is_tlas ? 5'd12 : 5'd6; - b_ret <= U_1; - state <= (climbs == CLIMBW'(MAX_CLIMBS)) ? S_DONE : B_SUB; - end - U_1: begin - if (!bt_pass) begin - if (cur) cul1 <= 1'b1; else cul0 <= 1'b1; - end - f_ret <= U_2; - state <= F_WAIT; - end - U_2: begin - if (cur) begin - ps1 <= fw_q; - dep1 <= dep1 - 32'd1; - end else begin - ps0 <= fw_q; - dep0 <= dep0 - 32'd1; - end - mv_dst <= cur ? RA_CB1 : RA_CB0; - mv_n <= 5'd6; - m_ret <= u_ret; - if (!is_tlas) begin - // a BLAS node holds its own box, read with its parent link - mv_k0 <= 5'd0; - state <= M_MV; - end else if (fw_q == ROOT) begin - state <= u_ret; - end else begin - mv_k0 <= fw_q[0] ? 5'd6 : 5'd0; - f_off <= node_off(tab_r, noff_r, 1'b1, fw_q); - f_n <= 5'd12; - fs_start <= 1'b1; - f_k0 <= '0; - f_ret <= U_3; - state <= F_WAIT; - end - end - U_3: begin - state <= M_MV; - end - - // ── box test (the reference's slab test) ───────────────── - B_SUB: begin - if (k == 5'd0) begin - bt_nan <= f_nan(rda[30:0]); - end - k <= k + 5'd1; - if (k == 5'd5) begin - k <= '0; - state <= B_MUL; - end - end - B_MUL: begin - if (fma_issue) begin - k <= k + 5'd1; - if (k == 5'd5) begin - k <= '0; - w_ret <= B_RED; - state <= W_DRAIN; - end - end - end - B_RED: begin - // read axis k, min/max it at k + 1, fold it at k + 2 - red_a <= rda; - red_b <= rdb; - red_mn <= f_min(red_a, red_b); - red_mx <= f_max(red_a, red_b); - if (k == 5'd2) begin - bt_lo <= red_mn; - bt_hi <= red_mx; - end else if (k > 5'd2) begin - bt_lo <= f_max(bt_lo, red_mn); - bt_hi <= f_min(bt_hi, red_mx); - end - k <= k + 5'd1; - if (k == 5'd4) begin - k <= '0; - state <= B_FIN0; - end - end - B_FIN0: begin - bt_lo0 <= f_max(32'd0, bt_lo); - state <= B_FIN; - end - B_FIN: begin - // hi >= fmax(0, lo); an empty (NaN) box is never entered - logic hit; - hit = !bt_nan && (f_lt(bt_lo0, bt_hi) || f_eq(bt_lo0, bt_hi)); - bt_key <= hit ? bt_lo : F_INF; - bt_pass <= hit && f_lt(bt_lo, bt_tmax); - state <= b_ret; - end - - // ── the reference's object-space ray for TLAS leaf o_inst ─ - O_0: begin - if (ray_same && !rk_world && (rk_inst == o_inst)) begin - state <= o_ret; // this object ray is set up already - end else begin - f_off <= tlas_r + 32'd12; - f_n <= 5'd1; - fs_start <= 1'b1; - f_k0 <= '0; - rk_valid <= 1'b0; - f_ret <= O_1; - state <= F_WAIT; - end - end - O_1: begin - stride_r <= fw_q; - mul_a <= o_inst; - mul_b <= fw_q; - mul_p <= '0; - x_ret <= O_2; - state <= X_MUL; - end - O_2: begin - f_off <= tlas_r + 32'd32 + mul_p; - f_n <= 5'd12; - fs_start <= 1'b1; - f_k0 <= '0; - f_ret <= O_3; - state <= F_WAIT; - end - O_3: begin - mv_dst <= RA_M; - mv_k0 <= '0; - mv_n <= 5'd12; - m_ret <= O_4; - state <= M_MV; - end - O_4: begin - k <= k + 5'd1; - if (k == 5'd17) begin - k <= '0; - w_ret <= O_5; - state <= W_DRAIN; - end - end - O_5: begin - k <= k + 5'd1; - if (k == 5'd5) begin - k <= '0; - w_ret <= O_6; - state <= W_DRAIN; - end - end - O_6: begin - k <= k + 5'd1; - if (k == 5'd5) begin - k <= '0; - w_ret <= O_7; - state <= W_DRAIN; - end - end - O_7: begin - k <= k + 5'd1; - if (k == 5'd2) begin - k <= '0; - w_ret <= O_8; - state <= W_DRAIN; - end - end - O_8: begin - rk_valid <= 1'b1; - rk_world <= 1'b0; - rk_inst <= o_inst; - rk_o <= cw_o; - rk_d <= cw_d; - rk_tlas <= tlas_r; - ray_same <= 1'b1; - r_ret <= o_ret; - state <= R_LD0; - end - - // ── 1/d per axis, correctly rounded ────────────────────── - R_LD0: begin - rx[0] <= rda; - rx[1] <= rdb; - state <= R_LD1; - end - R_LD1: begin - rx[2] <= rda; - state <= R_LD2; - end - R_LD2: begin - {rc_spec[0], rc_sval[0], dv_s[0], dv_m[0], dv_e[0]} <= rc_dec0; - {rc_spec[1], rc_sval[1], dv_s[1], dv_m[1], dv_e[1]} <= rc_dec1; - {rc_spec[2], rc_sval[2], dv_s[2], dv_m[2], dv_e[2]} <= rc_dec2; - for (integer i = 0; i < 3; ++i) begin - dv_rem[i] <= 25'h400000; // 2^22: the dividend's leading bits - dv_q[i] <= '0; - end - dv_it <= '0; - state <= R_IT; - end - R_IT: begin - for (integer i = 0; i < 3; ++i) begin - if ({dv_rem[i][23:0], 1'b0} >= 25'(dv_m[i])) begin - dv_rem[i] <= {dv_rem[i][23:0], 1'b0} - 25'(dv_m[i]); - dv_q[i] <= {dv_q[i][26:0], 1'b1}; - end else begin - dv_rem[i] <= {dv_rem[i][23:0], 1'b0}; - dv_q[i] <= {dv_q[i][26:0], 1'b0}; - end - end - dv_it <= dv_it + 5'd1; - if (dv_it == 5'd27) begin - k <= '0; - state <= R_WR; - end - end - R_WR: begin - k <= k + 5'd1; - if (k == 5'd7) begin - k <= '0; - state <= r_ret; - end - end - - // ── leaf routines ──────────────────────────────────────── - F_WAIT: begin - if (fs_done) begin - state <= F_CAP; - end - end - F_CAP: begin - fw_q <= fw; - state <= f_ret; - end - M_MV, C_CP: begin - k <= k + 5'd1; - if ((k + 5'd1) == mv_n) begin - k <= '0; - state <= (state == M_MV) ? m_ret : c_ret; - end - end - X_MUL: begin - mul_p <= mul_sum; - mul_a <= mul_a >> 1; - mul_b <= mul_b << 1; - if (mul_a[31:1] == 31'd0) begin - state <= x_ret; - end - end - W_DRAIN: begin - if (fma_pend == 6'd0) begin - state <= w_ret; - end - end - S_DONE: begin - state <= S_IDLE; - end - default: begin - state <= S_IDLE; - end - endcase - end - end - - assign req_ready = (state == S_IDLE); - assign done_valid = (state == S_DONE); - assign done_ctx = ctx_r; - assign done_keep = keep; - -endmodule diff --git a/hw/rtl/rtu/VX_rtu_scheduler.sv b/hw/rtl/rtu/VX_rtu_scheduler.sv index 451356f48b..a48e27f53f 100644 --- a/hw/rtl/rtu/VX_rtu_scheduler.sv +++ b/hw/rtl/rtu/VX_rtu_scheduler.sv @@ -186,9 +186,7 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( CS_OBJ_SETUP_WT = 5'd26, CS_INST_NEXT = 5'd27, CS_BHDR_REQ = 5'd28, // flat TLAS: BLAS header fetch - CS_BHDR_WAIT = 5'd29, - CS_ORC_REQ = 5'd30, // hand a near tie to the oracle - CS_ORC_WAIT = 5'd31; // park: the oracle's verdict + CS_BHDR_WAIT = 5'd29; // ── the context word: everything only the walker's EXEC touches ─── // One row of the context store. State an async producer writes (fetched @@ -205,20 +203,6 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( logic [31:0] best_ki; logic [27:0] best_kg; logic [31:0] best_kp; - // ... and its place in the source BVH's visit-order tables, which - // settle a near tie with it (VX_rtu_oracle): near_t(best_t), its - // instance rank, parent/side, BLAS table and vertices - logic [31:0] best_tn; - logic [31:0] best_iord; - logic [31:0] best_tord; - logic [31:0] best_btab; - logic [8:0][31:0] best_v; - // the walk's position in those tables: the TLAS table, the current - // instance's BLAS table and rank, the current triangle's parent/side - logic [31:0] tlas_tab; - logic [31:0] blas_tab; - logic [31:0] iord; - logic [31:0] tord; logic [31:0] yld_t; // staged candidate's t (compare copy) logic [31:0] yld_ki; // staged candidate's key: instance id logic [31:0] yld_ko; // ... and record offset @@ -295,8 +279,6 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( reg [NUM_CTX-1:0][LVLW-1:0] lvl_q_arr; reg [NUM_CTX-1:0][LB-1:0] f_slot_q; reg [NUM_CTX-1:0][RTU_CB_ACTION_BITS-1:0] act_q; - reg [NUM_CTX-1:0] orc_q; // the oracle holds the context's memory tag - reg [NUM_CTX-1:0] orc_res_q; // its verdict: the new hit replaces the committed one // ── per-slot state ──────────────────────────────────────────────── reg [NUM_SLOTS-1:0] running; @@ -347,11 +329,6 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( // the staged candidate, so EXEC only adds the t compares reg key_gt_floor_q; reg key_lt_yld_q; - // near-tie classification of the tri result against the committed hit, - // also precomputed at ALIGN: within the near window, and the oracle has - // tables for both hits and they are different triangles - reg near_q; - reg orc_ok_q; ctx_state_t word_q; lane_ray_t ray_q; reg [BUF_BITS-1:0] fbuf_q; @@ -362,7 +339,7 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( reg [15:0] flags_q; reg [15:0] cull_q; reg trihit_q, triback_q; - reg [31:0] trit_q, triu_q, triv_q, trin_q; + reg [31:0] trit_q, triu_q, triv_q; reg [2:0][31:0] xfo_q, xfd_q; reg [2:0][31:0] recip_q; // collector head sampled at ALIGN: EXEC reads these registers instead of @@ -431,7 +408,7 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( .clk (clk), .reset (reset), .read (g1_valid), - .write (mem_rsp_valid && !orc_q[mem_rsp_tag] && (f_slot_q[mem_rsp_tag] == LB'(s))), + .write (mem_rsp_valid && (f_slot_q[mem_rsp_tag] == LB'(s))), .wren (1'b1), .waddr (mem_rsp_tag), .wdata (mem_rsp_data), @@ -446,36 +423,9 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( wire tri_valid_out, tri_hit, tri_back; wire [CTX_TAG_W-1:0] tri_tag_out; wire [31:0] tri_t, tri_u, tri_v; - // the near window's bound rides with the result: it is pipelined after - // the tri PE, and the result event (RAM write and wake) is delayed to - // match, so neither lands on the EXEC path - localparam NEAR_LAT = 6; - wire [31:0] tri_tn; - VX_rtu_near_t #( - .LATENCY (NEAR_LAT) - ) tri_near ( - .clk (clk), - .enable (1'b1), - .t (tri_t), - .result (tri_tn) - ); - wire trd_valid, trd_hit, trd_back; - wire [CTX_TAG_W-1:0] trd_tag; - wire [31:0] trd_t, trd_u, trd_v; - VX_shift_register #( - .DATAW (1 + CTX_TAG_W + 2 + 3 * 32), - .RESETW (1), - .DEPTH (NEAR_LAT) - ) tri_delay ( - .clk (clk), - .reset (reset), - .enable (1'b1), - .data_in ({tri_valid_out, tri_tag_out, tri_hit, tri_back, tri_t, tri_u, tri_v}), - .data_out ({trd_valid, trd_tag, trd_hit, trd_back, trd_t, trd_u, trd_v}) - ); - wire [129:0] trires_rdata; + wire [97:0] trires_rdata; VX_dp_ram #( - .DATAW (130), + .DATAW (98), .SIZE (NUM_CTX), .OUT_REG (1), .RDW_MODE ("W") @@ -483,10 +433,10 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( .clk (clk), .reset (reset), .read (g1_valid), - .write (trd_valid), + .write (tri_valid_out), .wren (1'b1), - .waddr (trd_tag), - .wdata ({trd_hit, trd_back, trd_t, trd_u, trd_v, tri_tn}), + .waddr (tri_tag_out), + .wdata ({tri_hit, tri_back, tri_t, tri_u, tri_v}), .raddr (g1_idx), .rdata (trires_rdata) ); @@ -581,42 +531,6 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( wire [31:0] cand_ki_al = cs_word.in_blas ? cs_word.inst_id : 32'd0; - // IEEE F32 ordered compares (NaN false, +0 == -0) - function automatic logic f_nan(input logic [30:0] a); - f_nan = (a[30:23] == 8'hff) && (a[22:0] != 23'd0); - endfunction - function automatic logic f_eq(input logic [31:0] a, input logic [31:0] b); - f_eq = !f_nan(a[30:0]) && !f_nan(b[30:0]) - && ((a == b) || ((a[30:0] == 31'd0) && (b[30:0] == 31'd0))); - endfunction - function automatic logic f_lt(input logic [31:0] a, input logic [31:0] b); - if (f_nan(a[30:0]) || f_nan(b[30:0]) || ((a[30:0] == 31'd0) && (b[30:0] == 31'd0))) begin - f_lt = 1'b0; - end else if (a[31] != b[31]) begin - f_lt = a[31]; - end else if (!a[31]) begin - f_lt = (a[30:0] < b[30:0]); - end else begin - f_lt = (a[30:0] > b[30:0]); - end - endfunction - - // near window: t == best, or nearer with near_t(t) >= best, or farther - // with t <= near_t(best) - wire [31:0] al_t = trires_rdata[127:96]; - wire [31:0] al_tn = trires_rdata[31:0]; - wire al_near = cs_word.best_kv - && (f_eq(al_t, cs_word.best_t) - || (f_lt(al_t, cs_word.best_t) - && (f_lt(cs_word.best_t, al_tn) || f_eq(cs_word.best_t, al_tn))) - || (f_lt(cs_word.best_t, al_t) - && (f_lt(al_t, cs_word.best_tn) || f_eq(al_t, cs_word.best_tn)))); - wire [31:0] al_iord = cs_word.in_blas ? cs_word.iord : 32'd0; - wire [31:0] al_btab = cs_word.in_blas ? cs_word.blas_tab : 32'd0; - wire al_orc_ok = (FLAT == 0) - && (cs_word.tlas_tab != 32'd0) && (al_btab != 32'd0) && (cs_word.best_btab != 32'd0) - && ({al_iord, cs_word.tord} != {cs_word.best_iord, cs_word.best_tord}); - // ═══════════════════════ stage advance ════════════════════════════ always_ff @(posedge clk) begin if (reset) begin @@ -639,8 +553,6 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( rewalk_q <= s1_rewalk; key_gt_floor_q <= {cand_ki_al, cs_word.cur_off} > {cs_word.floor_ki, cs_word.floor_ko}; key_lt_yld_q <= {cand_ki_al, cs_word.cur_off} < {cs_word.yld_ki, cs_word.yld_ko}; - near_q <= al_near; - orc_ok_q <= al_orc_ok; word_q <= cs_word; ray_q <= lane_ray_t'(ray_rdata); fbuf_q <= fbuf; @@ -653,7 +565,7 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( sp_q <= sp_q_arr[s1_sel]; flags_q <= slot_flags[s1_slot]; cull_q <= slot_cull[s1_slot]; - {trihit_q, triback_q, trit_q, triu_q, triv_q, trin_q} <= trires_rdata; + {trihit_q, triback_q, trit_q, triu_q, triv_q} <= trires_rdata; {xfo_q, xfd_q} <= xfres_rdata; recip_q <= recip_rdata; coll_hit_q <= coll_prochit[cs_word.coll_id]; @@ -828,11 +740,7 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( .ro (walk_ro), .inv_d (walk_inv_d), .t_min (ray_q.t_min), - // a node's children are culled with slack past an opaque hit - // committed in this walk, so every hit the near-tie oracle may - // prefer is still reached; a procedural AABB's entry is a - // candidate t, culled at the committed hit itself - .t_max ((box_feed_raw || !word_q.best_kv) ? word_q.best_t : word_q.best_tn), + .t_max (word_q.best_t), .valid_out (box_valid_out), .tag_out (box_tag_out), .tag_out_pre (box_tag_pre), @@ -855,6 +763,7 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( // here rather than repeated on the result cycle -- where it fans out across // every row and lands on the path that ends at the ordering registers. wire [COLL_IDW-1:0] box_coll_pre = box_tag_pre[COLL_IDW+32-1 : 32]; + `UNUSED_VAR (box_tag_pre[31:0]) reg [COLL_SIZE-1:0] box_coll_hot; always @(posedge clk) begin if (reset) begin @@ -917,8 +826,7 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( .v1 (ltri_v1), .v2 (ltri_v2), .t_min (ray_q.t_min), - // the ray's own interval: a hit at or past the committed t still - // reaches the tie-break and the near-tie oracle + // the ray's own interval: an equal-t hit still reaches the tie-break .t_max (ray_q.t_max), .valid_out (tri_valid_out), .tag_out (tri_tag_out), @@ -948,66 +856,6 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( .obj_rd (xform_obj_d) ); - // ── near-tie oracle (BVH only) ──────────────────────────────────── - // A context whose opaque hit lands within the near window of the one it - // committed hands both to the oracle and parks. The oracle fetches table - // lines under the context's tag; their responses are its own, never the - // context's fetched-line buffer (which still holds the triangle record). - wire orc_start; - wire orc_req_ready; - wire orc_done_valid, orc_done_keep; - wire [CTX_TAG_W-1:0] orc_done_ctx; - wire orc_mreq_valid; - wire [ADDRW-1:0] orc_mreq_addr; - wire [SLOT_W-1:0] x_slot = SLOT_W'(32'(sel_q) / NUM_LANES); - wire [31:0] cur_iord = word_q.in_blas ? word_q.iord : 32'd0; - wire [31:0] cur_btab = word_q.in_blas ? word_q.blas_tab : 32'd0; - wire [8:0][31:0] ltri_v = {ltri_v2, ltri_v1, ltri_v0}; - if (!FLAT) begin : g_oracle - VX_rtu_oracle #( - .CTX_TAG_W (CTX_TAG_W), - .ADDRW (ADDRW), - .LINE_BITS (LINE_BITS) - ) oracle ( - .clk (clk), - .reset (reset), - .flush (running == '0), - .req_valid (orc_start), - .req_ready (orc_req_ready), - .req_ctx (sel_q), - .req_scene (slot_scene[x_slot]), - .req_wo (ray_q.origin), - .req_wd (ray_q.dir), - .req_tlas (word_q.tlas_tab), - .req_a_t (trit_q), - .req_a_inst (cur_iord), - .req_a_ps (word_q.tord), - .req_a_tab (cur_btab), - .req_a_v (ltri_v), - .req_b_t (word_q.best_t), - .req_b_inst (word_q.best_iord), - .req_b_ps (word_q.best_tord), - .req_b_tab (word_q.best_btab), - .req_b_v (word_q.best_v), - .done_valid (orc_done_valid), - .done_ctx (orc_done_ctx), - .done_keep (orc_done_keep), - .mem_req_valid (orc_mreq_valid), - .mem_req_addr (orc_mreq_addr), - .mem_req_ready (mem_req_ready), - .mem_rsp_valid (mem_rsp_valid && orc_q[mem_rsp_tag]), - .mem_rsp_data (mem_rsp_data) - ); - end else begin : g_no_oracle - assign orc_req_ready = 1'b0; - assign orc_done_valid = 1'b0; - assign orc_done_keep = 1'b0; - assign orc_done_ctx = '0; - assign orc_mreq_valid = 1'b0; - assign orc_mreq_addr = '0; - `UNUSED_VAR ({orc_start, x_slot, cur_iord, cur_btab, ltri_v}) - end - // ── reciprocal datapath: pipelined, one axis per issue ──────────── wire [1:0] recip_axis; wire [31:0] recip_din = recip_obj ? word_q.obj_d[recip_axis] : ray_q.dir[recip_axis]; @@ -1448,7 +1296,6 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( reg coll_alloc_r; reg coll_free_r; reg cf_push_r; - reg orc_start_r; commit_t cf_din_r; reg sp_inc, sp_dec; reg sp_clr; @@ -1458,14 +1305,7 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( reg [LVLW-1:0] path_wlvl_r; reg [PATHW-1:0] path_wdata_r; - // the oracle's table fetches go first; a context's fetch retries - wire mem_fire = x_valid && mem_issue && mem_req_ready && !orc_mreq_valid; - - // the TRI_WAIT / ORC_WAIT verdict on an opaque hit - wire in_orc_wait = (word_x.cstate == CS_ORC_WAIT); - wire to_oracle = !in_orc_wait && tri_pass && tri_opaque && near_q && orc_ok_q; - wire opq_take = in_orc_wait ? orc_res_q[sel_q] - : ((tri_committable || tri_tie) && tri_opaque); + wire mem_fire = x_valid && mem_issue && mem_req_ready; wire [RTU_CHILD_BITS-1:0] last_child = node.n_children - RTU_CHILD_BITS'(1); @@ -1499,7 +1339,6 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( coll_alloc_r = 1'b0; coll_free_r = 1'b0; cf_push_r = 1'b0; - orc_start_r = 1'b0; cf_din_r = '0; sp_inc = 1'b0; sp_dec = 1'b0; @@ -1636,7 +1475,6 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( end else if (node_kind == RTU_KIND_LEAF_TRI) begin word_n.prim_base = leaf_prim; word_n.geom_r = leaf_geom; - word_n.tord = leaf_flags; // parent/side in its BLAS table word_n.tri_n = 32'(leaf_count); word_n.tri_i = '0; if (leaf_count == 8'd0) begin @@ -1660,12 +1498,7 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( wake_self = 1'b1; end end else if (node_kind == RTU_KIND_LEAF_INST && leaf_count != 8'd0) begin - // header: the BLAS table, the first instance's TLAS rank, - // the TLAS table // a restart following its path resumes at the recorded instance - word_n.blas_tab = leaf_geom; - word_n.iord = leaf_flags + inst_start; - word_n.tlas_tab = leaf_prim; word_n.inst_cnt = {24'd0, leaf_count}; word_n.inst_idx = inst_start; word_n.inst_base = word_x.cur_off + 32'(RTU_LEAF_HDR_BYTES); @@ -1816,14 +1649,10 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( wake_self = 1'b1; end end - CS_TRI_WAIT, CS_ORC_WAIT: begin + CS_TRI_WAIT: begin // woken by the tri PE result (held in its result RAM, so a retry - // on a full commit queue re-reads the same result), or by the - // oracle's verdict on it - if (to_oracle) begin - word_n.cstate = CS_ORC_REQ; - wake_self = 1'b1; - end else if (opq_take) begin + // on a full commit queue re-reads the same result) + if ((tri_committable || tri_tie) && tri_opaque) begin cf_din_r.kind = CK_HIT; cf_din_r.t = trit_q; cf_din_r.u = triu_q; @@ -1842,11 +1671,6 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( word_n.best_ki = tri_ki; word_n.best_kg = tri_kg; word_n.best_kp = tri_kp; - word_n.best_tn = trin_q; - word_n.best_iord = cur_iord; - word_n.best_tord = word_x.tord; - word_n.best_btab = cur_btab; - word_n.best_v = ltri_v; // a closer opaque hit occludes a farther candidate if (yld_q[sel_q] && (word_x.yld_t >= trit_q)) begin exec_yld_clr = 1'b1; @@ -1856,7 +1680,6 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( exec_done = 1'b1; end else if ((word_x.tri_i + 32'd1) < word_x.tri_n) begin word_n.tri_i = word_x.tri_i + 32'd1; - word_n.tord = word_x.tord + 32'd1; word_n.cur_off = word_x.cur_off + 32'(RTU_TRI_STRIDE); word_n.cstate = CS_LTRI_REQ0; wake_self = 1'b1; @@ -1865,8 +1688,7 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( wake_self = 1'b1; end end - end else if (!in_orc_wait - && tri_committable + end else if (tri_committable && above_floor(trit_q) && before_yld(trit_q)) begin cf_din_r.kind = CK_YLDA; @@ -1892,7 +1714,6 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( word_n.yld_ko = word_x.cur_off; if ((word_x.tri_i + 32'd1) < word_x.tri_n) begin word_n.tri_i = word_x.tri_i + 32'd1; - word_n.tord = word_x.tord + 32'd1; word_n.cur_off = word_x.cur_off + 32'(RTU_TRI_STRIDE); word_n.cstate = CS_LTRI_REQ0; end else begin @@ -1903,7 +1724,6 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( end else begin if ((word_x.tri_i + 32'd1) < word_x.tri_n) begin word_n.tri_i = word_x.tri_i + 32'd1; - word_n.tord = word_x.tord + 32'd1; word_n.cur_off = word_x.cur_off + 32'(RTU_TRI_STRIDE); word_n.cstate = CS_LTRI_REQ0; end else begin @@ -1912,15 +1732,6 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( wake_self = 1'b1; end end - CS_ORC_REQ: begin - // the triangle record stays in the line buffer while parked - if (orc_req_ready) begin - orc_start_r = 1'b1; - word_n.cstate = CS_ORC_WAIT; - end else begin - wake_self = 1'b1; - end - end CS_POP: begin if (FLAT) begin if (word_x.in_blas) begin @@ -2060,7 +1871,6 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( end end else begin word_n.inst_idx = word_x.inst_idx + 32'd1; - word_n.iord = word_x.iord + 32'd1; word_n.cur_off = word_x.inst_base + ((word_x.inst_idx + 32'd1) * 32'(RTU_INST_STRIDE)); word_n.cstate = CS_INST_REQ; @@ -2112,11 +1922,9 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( ("%t: rtu walk deeper than %0d levels", $time, (1 << LVLW) - 1)) assign path_wlvl = path_wlvl_r; assign path_wdata = path_wdata_r; - assign orc_start = x_valid && orc_start_r; - assign mem_req_valid = orc_mreq_valid || (x_valid && mem_issue); - assign mem_req_addr = orc_mreq_valid ? orc_mreq_addr - : (structaddr_q + (ADDRW'(mem_fslot) << RTU_LINE_SEL_BITS)); - assign mem_req_tag = orc_mreq_valid ? orc_done_ctx : sel_q; + assign mem_req_valid = x_valid && mem_issue; + assign mem_req_addr = structaddr_q + (ADDRW'(mem_fslot) << RTU_LINE_SEL_BITS); + assign mem_req_tag = sel_q; // ═══════════════════════ hot-state update ═════════════════════════ // The wake vector's next state, fed to the SELECT arbiter. Wake events @@ -2146,9 +1954,8 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( wire [NUM_CTX-1:0] rdy_wake_mask = (ray_wr_valid ? NUM_CTX'(1) << ray_wr_ctx : NUM_CTX'(0)) - | ((mem_rsp_valid && !orc_q[mem_rsp_tag]) ? NUM_CTX'(1) << mem_rsp_tag : NUM_CTX'(0)) - | (orc_done_valid ? NUM_CTX'(1) << orc_done_ctx : NUM_CTX'(0)) - | (trd_valid ? NUM_CTX'(1) << trd_tag : NUM_CTX'(0)) + | (mem_rsp_valid ? NUM_CTX'(1) << mem_rsp_tag : NUM_CTX'(0)) + | (tri_valid_out ? NUM_CTX'(1) << tri_tag_out : NUM_CTX'(0)) | (xform_valid_out ? NUM_CTX'(1) << xform_tag_out : NUM_CTX'(0)) | ((recip_valid_out && recip_last_out) ? NUM_CTX'(1) << recip_tag_out : NUM_CTX'(0)) | (box_wake_r ? NUM_CTX'(1) << box_wake_ctx_r : NUM_CTX'(0)) @@ -2171,7 +1978,6 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( yld_q <= '0; objv_q <= '0; attr_q <= '0; - orc_q <= '0; running <= '0; finalised <= '0; done_r <= '0; @@ -2237,14 +2043,6 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( if (mem_fire) begin f_slot_q[sel_q] <= mem_fslot; end - if (orc_start_r) begin - orc_q[sel_q] <= 1'b1; - end - end - - if (orc_done_valid) begin - orc_q[orc_done_ctx] <= 1'b0; - orc_res_q[orc_done_ctx] <= orc_done_keep; end // resume: capture the actions, queue the walker job @@ -2505,9 +2303,9 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( `TRACE(2, ("%t: %s rtu-node: ctx=%0d, addr=0x%0h, kind=%0d\n", $time, INSTANCE_ID, sel_q, structaddr_q, node_kind)) end - if (trd_valid) begin + if (tri_valid_out) begin `TRACE(2, ("%t: %s rtu-tri: ctx=%0d, hit=%0d, t=0x%0h\n", - $time, INSTANCE_ID, trd_tag, trd_hit, trd_t)) + $time, INSTANCE_ID, tri_tag_out, tri_hit, tri_t)) end if (| done_r) begin `TRACE(1, ("%t: %s rtu-done: slots=%b\n", $time, INSTANCE_ID, done_r)) diff --git a/sim/simx/rtu/rtu_bvh.h b/sim/simx/rtu/rtu_bvh.h index 7fcafb55ad..e684b98fc8 100644 --- a/sim/simx/rtu/rtu_bvh.h +++ b/sim/simx/rtu/rtu_bvh.h @@ -200,15 +200,9 @@ inline void decode_bvh6_node(const VxBvh6InternalNode* n, uint32_t count, // uint32 geometry_index : Vulkan gl_GeometryIndexEXT for this leaf // uint32 flags : LeafProc: bit 0 = OPAQUE (all prims), bit 1 = // forced non-opaque, bits 8..15 = SBT_IDX. -// LeafTri / LeafInst: the leaf's place in the -// source BVH's visit-order table, which settles -// an exact-t tie between opaque hits the way a -// first-visited-wins traversal of the source BVH -// does (rtu_walker.cpp). LeafTri: parent << 1 | -// side in its BLAS's table. LeafInst: the -// instance's visit rank in the TLAS's table, with -// geometry_index = the BLAS's table and prim_base -// = the TLAS's table (scene offsets; 0 = none). +// LeafTri / LeafInst: reserved, ignored (a +// triangle carries its own flag word). LeafInst +// also ignores geometry_index and prim_base. // uint32 prim_base : gl_PrimitiveID of this leaf's first // primitive; the walker reports // prim_base + within-leaf index so a diff --git a/sim/simx/rtu/rtu_isect.cpp b/sim/simx/rtu/rtu_isect.cpp index fa1a87348c..0981d0d892 100644 --- a/sim/simx/rtu/rtu_isect.cpp +++ b/sim/simx/rtu/rtu_isect.cpp @@ -148,10 +148,9 @@ uint32_t BoxPe::pipe_depth() { uint32_t TriPe::pipe_depth() { // input select + 1/dir + 3 F32 stages + 5 F64 stages + F64 divide + narrow - // + verdict (VX_rtu_tri_pe), then the 6 stages that form the result's near- - // tie window bound before the scheduler sees it (VX_rtu_near_t). + // + verdict (VX_rtu_tri_pe). return 3 + kRtuFdivLat + 3 * kRtuLatencyFma + 5 * kRtuLatencyFma64 - + kRtuFdiv64Lat + 6; + + kRtuFdiv64Lat; } }} // namespace vortex::rtu diff --git a/sim/simx/rtu/rtu_walker.cpp b/sim/simx/rtu/rtu_walker.cpp index 8ced6f95ed..096136e676 100644 --- a/sim/simx/rtu/rtu_walker.cpp +++ b/sim/simx/rtu/rtu_walker.cpp @@ -14,7 +14,6 @@ #include "rtu_walker.h" #include -#include #include #include #include @@ -93,11 +92,6 @@ struct WalkCtx { uint32_t best_custom; // VK_INSTANCE_CUSTOM_INDEX of the committed instance uint32_t best_geom; // gl_GeometryIndexEXT of the committed leaf bool best_kv; // an opaque hit committed in this walk - uint64_t best_order; // (instance order << 32) | triangle order of it - uint32_t best_blas_tab; // its BLAS's visit-order table (0: none) - float best_tri_box[6]; // its triangle's vertex min/max - uint32_t tlas_tab; // the TLAS's visit-order table (0: none) - float world_o[3], world_d[3]; bool any_hit; bool yield_pending; float yield_t, yield_u, yield_v; @@ -143,268 +137,10 @@ inline uint32_t hit_facing_bit(bool back_facing, uint32_t inst_flags) { return back_facing ? VX_RT_HIT_BACK_FACING : 0u; } -inline uint64_t hit_order(uint32_t inst_order, uint32_t tri_order) { - return (uint64_t(inst_order) << 32) | tri_order; -} - -inline uint32_t scene_u32(SceneView& sv, uint32_t off) { - uint32_t v = 0; - read_scene_bytes(sv, off, sizeof(v), reinterpret_cast(&v)); - return v; -} - -// A source-BVH child box tested exactly as the reference traversal tests it -// (F32, (bound - origin) * 1/dir, 1/0 -> FLT_MAX): `lo` is its entry distance -// (the child-order key), and the box is visited iff it passes against the -// committed t at the time. -struct SrcBox { float lo, hi; }; - -SrcBox src_box(const float box[6], const float o[3], const float inv[3]) { - if (std::isnan(box[0])) return { INFINITY, -INFINITY }; - float b0[3], b1[3]; - for (int i = 0; i < 3; ++i) { - b0[i] = (box[i] - o[i]) * inv[i]; - b1[i] = (box[3 + i] - o[i]) * inv[i]; - } - const float lo = std::fmax(std::fmax(std::fmin(b0[0], b1[0]), - std::fmin(b0[1], b1[1])), - std::fmin(b0[2], b1[2])); - const float hi = std::fmin(std::fmin(std::fmax(b0[0], b1[0]), - std::fmax(b0[1], b1[1])), - std::fmax(b0[2], b1[2])); - return { lo, hi }; -} - -inline bool src_box_hit(const SrcBox& b) { return b.hi >= std::fmax(0.f, b.lo); } -inline float src_box_key(const SrcBox& b) { return src_box_hit(b) ? b.lo : INFINITY; } -inline bool src_box_passes(const SrcBox& b, float tmax) { - return src_box_hit(b) && b.lo < tmax; -} - -// The reference's ray in a table's space. -struct SrcRay { float o[3], d[3], inv[3]; }; - -SrcRay src_ray(const float o[3], const float d[3]) { - SrcRay r; - for (int i = 0; i < 3; ++i) { - r.o[i] = o[i]; r.d[i] = d[i]; - r.inv[i] = (d[i] == 0.f) ? FLT_MAX : 1.0f / d[i]; - } - return r; -} - -// Visit-order tables: the source BVH's binary tree, which the reference walks -// depth-first, nearer child box first, child 0 on equal entry distances, each -// box tested against the hit committed when its parent is visited. -// TLAS table at `tab` (leaves = instances, indexed by visit rank): -// { n_leaves, n_nodes, nodes_off, leaf_stride } -// leaf[i] at tab + 16 + i*leaf_stride: { parent << 1 | side, _, _, _, -// world->object 3x4 } -// node[j] at tab + nodes_off + j*64: { child box 0 min/max, child box 1 -// min/max, parent << 1 | side (~0: root), depth } -// BLAS table at `tab` (leaves = triangles, each naming its parent << 1 | side -// in its leaf header; a triangle's box is its vertices' min/max): -// { n_nodes, _, nodes_off, 32 } -// node[j] at tab + nodes_off + j*32: { own box min/max, parent << 1 | side -// (~0: root), depth } -constexpr uint32_t kSrcRoot = 0xffffffffu; -constexpr uint32_t kSrcTabHdr = 16; -constexpr uint32_t kSrcLeafWto = 16; -constexpr uint32_t kSrcTlasNodeBytes = 64; -constexpr uint32_t kSrcTlasNodeParent = 48; -constexpr uint32_t kSrcTlasNodeDepth = 52; -constexpr uint32_t kSrcBlasNodeBytes = 32; -constexpr uint32_t kSrcBlasNodeParent = 24; -constexpr uint32_t kSrcBlasNodeDepth = 28; - -// One leaf's climb towards the root. `box` is the box of the child the climb -// is at (tested when its parent was visited); each box the climb leaves below -// the lowest common ancestor's child was tested after the other leaf's -// subtree, had that come first, so the climb notes whether one of them fails -// against `tmax` (the other hit's t). -struct SrcClimb { - uint32_t ps; // parent << 1 | side of the current child - uint32_t depth; // depth of node(ps) - float box[6]; - bool culled; -}; - -struct SrcTab { - SceneView& sv; - uint32_t tab, nodes_off, node_bytes, parent_off, depth_off; - bool tlas; - uint32_t node(uint32_t ps) const { return nodes_off + (ps >> 1) * node_bytes; } - uint32_t parent(uint32_t ps) const { return scene_u32(sv, node(ps) + parent_off); } - uint32_t depth(uint32_t ps) const { return scene_u32(sv, node(ps) + depth_off); } - // Box of the child at ps: a TLAS node holds its children's boxes, a BLAS - // node its own. - void child_box(uint32_t ps, uint32_t below, float box[6]) const { - if (tlas) - read_scene_bytes(sv, node(ps) + 24 * (ps & 1u), 24, reinterpret_cast(box)); - else - read_scene_bytes(sv, node(below), 24, reinterpret_cast(box)); - } -}; - -SrcTab src_tab(SceneView& sv, uint32_t tab, bool tlas) { - return { sv, tab, tab + scene_u32(sv, tab + 8), - tlas ? kSrcTlasNodeBytes : kSrcBlasNodeBytes, - tlas ? kSrcTlasNodeParent : kSrcBlasNodeParent, - tlas ? kSrcTlasNodeDepth : kSrcBlasNodeDepth, tlas }; -} - -void src_up(const SrcTab& t, SrcClimb& c, const SrcRay& r, float tmax) { - if (!src_box_passes(src_box(c.box, r.o, r.inv), tmax)) c.culled = true; - const uint32_t below = c.ps; - c.ps = t.parent(c.ps); - --c.depth; - t.child_box(c.ps, below, c.box); -} - -// Climb both leaves to their lowest common ancestor; true when a's side is -// visited first there. a.culled / b.culled report the boxes left on the way. -bool src_lca(const SrcTab& t, SrcClimb& a, SrcClimb& b, const SrcRay& r, - float ta, float tb) { - while (!t.sv.miss && a.depth > b.depth) src_up(t, a, r, tb); - while (!t.sv.miss && b.depth > a.depth) src_up(t, b, r, ta); - while (!t.sv.miss && (a.ps >> 1) != (b.ps >> 1)) { - src_up(t, a, r, tb); - src_up(t, b, r, ta); - } - const float ka = src_box_key(src_box(a.box, r.o, r.inv)); - const float kb = src_box_key(src_box(b.box, r.o, r.inv)); - const float d0 = (a.ps & 1u) ? kb : ka; - const float d1 = (a.ps & 1u) ? ka : kb; - const uint32_t first = (d1 < d0) ? 1u : 0u; - return (a.ps & 1u) == first; -} - -SrcClimb src_blas_leaf(const SrcTab& t, uint32_t ps, const float box[6]) { - SrcClimb c; - c.ps = ps; - c.depth = (ps == kSrcRoot) ? 0 : t.depth(ps) + 1; - std::memcpy(c.box, box, sizeof(c.box)); - c.culled = false; - return c; -} - -// Whether a box on the leaf's whole BLAS path fails against tmax: every box -// below the BLAS root is tested after the instance is entered. -bool src_blas_path_culled(const SrcTab& t, uint32_t ps, const float box[6], - const SrcRay& r, float tmax) { - SrcClimb c = src_blas_leaf(t, ps, box); - while (!t.sv.miss && c.ps != kSrcRoot) src_up(t, c, r, tmax); - return c.culled; -} - -// The reference's object-space ray for TLAS leaf `inst`, from its -// world->object matrix in the table. -SrcRay src_object_ray(SceneView& sv, uint32_t tlas_tab, uint32_t inst, - const float wo[3], const float wd[3]) { - const uint32_t leaf_stride = scene_u32(sv, tlas_tab + 12); - float m[12]; - read_scene_bytes(sv, tlas_tab + kSrcTabHdr + inst * leaf_stride + kSrcLeafWto, - sizeof(m), reinterpret_cast(m)); - float o[3], d[3]; - world_to_object_ray(m, wo, wd, o, d); - return src_ray(o, d); -} - -inline void tri_box(const float* v, float box[6]) { - for (int a = 0; a < 3; ++a) { - box[a] = std::fmin(v[a], std::fmin(v[3 + a], v[6 + a])); - box[3 + a] = std::fmax(v[a], std::fmax(v[3 + a], v[6 + a])); - } -} - -// An opaque triangle hit as the source traversal sees it. -struct SrcHit { - float t; - uint32_t inst; // TLAS visit rank - uint32_t ps; // parent << 1 | side in its BLAS table - uint32_t blas_tab; - float box[6]; -}; - -// Which of two opaque hits within a few ulps of each other the source -// traversal keeps: it commits a hit only nearer than the committed one, and -// tests each box against the committed t, so the first-reached of two equal-t -// hits wins, and a nearer hit reached second is lost when a box on its way -// (below where the two paths part) enters at or past the first one's t. -bool src_keeps_a(SceneView& sv, uint32_t tlas_tab, const float wo[3], - const float wd[3], const SrcHit& a, const SrcHit& b) { - bool a_first; - bool a_culled, b_culled; - if (a.inst == b.inst) { - const SrcTab t = src_tab(sv, a.blas_tab, false); - const SrcRay r = src_object_ray(sv, tlas_tab, a.inst, wo, wd); - if (a.ps == kSrcRoot || b.ps == kSrcRoot) return a.t <= b.t; - SrcClimb ca = src_blas_leaf(t, a.ps, a.box); - SrcClimb cb = src_blas_leaf(t, b.ps, b.box); - a_first = src_lca(t, ca, cb, r, a.t, b.t); - a_culled = ca.culled; b_culled = cb.culled; - } else { - const SrcTab tt = src_tab(sv, tlas_tab, true); - const SrcRay rw = src_ray(wo, wd); - const uint32_t stride = scene_u32(sv, tlas_tab + 12); - auto tlas_leaf = [&](uint32_t inst) { - SrcClimb c; - c.ps = scene_u32(sv, tlas_tab + kSrcTabHdr + inst * stride); - c.depth = (c.ps == kSrcRoot) ? 0 : tt.depth(c.ps) + 1; - tt.child_box(c.ps, 0, c.box); - c.culled = false; - return c; - }; - SrcClimb ca = tlas_leaf(a.inst), cb = tlas_leaf(b.inst); - if (sv.miss) return false; - a_first = src_lca(tt, ca, cb, rw, a.t, b.t); - a_culled = ca.culled - || src_blas_path_culled(src_tab(sv, a.blas_tab, false), a.ps, a.box, - src_object_ray(sv, tlas_tab, a.inst, wo, wd), b.t); - b_culled = cb.culled - || src_blas_path_culled(src_tab(sv, b.blas_tab, false), b.ps, b.box, - src_object_ray(sv, tlas_tab, b.inst, wo, wd), a.t); - } - if (a.t == b.t) return a_first; - if (a.t < b.t) return !(!a_first && a_culled); - return a_first && b_culled; -} - -// A small relative bound past the committed t: within it, a box's slab entry -// (a few ulps of rounding a triangle's watertight t does not carry) can land -// on either side of the committed t, so which of two such hits the source -// traversal keeps depends on its visit order. Beyond it the nearer hit wins. -inline float near_t(float t) { return t + std::fabs(t) * 0x1p-19f; } - -// The bound a child box's entry is culled against: kept open by near_t, so -// every hit the source traversal might keep over the committed one is reached. -inline float box_cull_t(const WalkCtx& ctx) { - return ctx.best_kv ? near_t(ctx.best_t) : ctx.best_t; -} - -// Whether an opaque hit at t replaces the committed one. Within near_t of an -// opaque hit committed in this walk, the scene's visit-order tables settle it -// as the source traversal would; without tables, an exact tie goes to the -// lowest (instance, geometry, primitive) and otherwise the nearer hit wins. -bool commit_takes(SceneView& sv, const WalkCtx& ctx, float t, uint64_t order, - uint32_t blas_tab, const float* tri, uint32_t instance_id, +// Whether an opaque hit at t replaces the committed one: the nearer hit wins, +// and an exact tie goes to the lowest (instance, geometry, primitive). +bool commit_takes(const WalkCtx& ctx, float t, uint32_t instance_id, uint32_t geom, uint32_t prim) { - const bool near = ctx.best_kv - && (t == ctx.best_t - || (t < ctx.best_t && near_t(t) >= ctx.best_t) - || (t > ctx.best_t && t <= near_t(ctx.best_t))); - if (near && ctx.tlas_tab && tri && blas_tab && ctx.best_blas_tab - && order != ctx.best_order) { - SrcHit a, b; - a.t = t; a.inst = uint32_t(order >> 32); a.ps = uint32_t(order); - a.blas_tab = blas_tab; tri_box(tri, a.box); - b.t = ctx.best_t; b.inst = uint32_t(ctx.best_order >> 32); - b.ps = uint32_t(ctx.best_order); b.blas_tab = ctx.best_blas_tab; - std::memcpy(b.box, ctx.best_tri_box, sizeof(b.box)); - const bool keeps_new = src_keeps_a(sv, ctx.tlas_tab, ctx.world_o, - ctx.world_d, a, b); - return !sv.miss && keeps_new; - } if (t < ctx.best_t) return true; if (!(ctx.best_kv && t == ctx.best_t)) return false; const uint32_t best_geom = ctx.best_geom & VX_RT_HIT_GEOMETRY_MASK; @@ -428,7 +164,6 @@ void walk_bvh4_subtree(SceneView& sv, const float ro[3], const float rd[3], uint32_t root_off, uint32_t instance_id, uint32_t custom_id, uint32_t inst_flags, - uint32_t inst_order, uint32_t blas_tab, WalkCtx& ctx, PerfStats& perf) { auto visit_leaf_tri = [&](uint32_t leaf_off, uint32_t count) { uint8_t hdr_buf[kVxBvhLeafHeaderBytes]; @@ -438,7 +173,6 @@ void walk_bvh4_subtree(SceneView& sv, reinterpret_cast(hdr_buf); uint32_t leaf_geom = hdr->geometry_index; uint32_t leaf_prim_base = hdr->prim_base; // Vulkan gl_PrimitiveID base - uint32_t leaf_order = hdr->flags; // LeafTri: source parent/side uint32_t tris_off = leaf_off + kVxBvhLeafHeaderBytes; for (uint32_t i = 0; i < count; ++i) { if (ctx.terminated) return; @@ -466,16 +200,9 @@ void walk_bvh4_subtree(SceneView& sv, if (cls.action == TriAction::Ignore) continue; if (cls.action == TriAction::Commit) { - const uint64_t order = hit_order(inst_order, leaf_order + i); - const bool takes = commit_takes(sv, ctx, t_hit, order, blas_tab, tri, - instance_id, leaf_geom, - leaf_prim_base + i); - if (sv.miss) return; - if (takes) { + if (commit_takes(ctx, t_hit, instance_id, leaf_geom, + leaf_prim_base + i)) { ctx.best_t = t_hit; ctx.best_u = u; ctx.best_v = v; - ctx.best_order = order; - ctx.best_blas_tab = blas_tab; - tri_box(tri, ctx.best_tri_box); ctx.best_prim = leaf_prim_base + i; ctx.best_instance = instance_id; ctx.best_custom = custom_id; @@ -563,12 +290,6 @@ void walk_bvh4_subtree(SceneView& sv, uint8_t hdr_buf[kVxBvhLeafHeaderBytes]; read_scene_bytes(sv, leaf_off, sizeof(hdr_buf), hdr_buf); if (sv.miss) return; - // LeafInst header: geometry_index = the BLAS's visit-order table, - // flags = the instance's visit order, prim_base = the TLAS's table. - const VxBvhLeafHeader* ihdr = - reinterpret_cast(hdr_buf); - const uint32_t leaf_order = ihdr->flags; - ctx.tlas_tab = ihdr->prim_base; uint32_t insts_off = leaf_off + kVxBvhLeafHeaderBytes; for (uint32_t i = 0; i < count; ++i) { uint8_t inst_buf[kVxBvhInstanceStride]; @@ -591,7 +312,6 @@ void walk_bvh4_subtree(SceneView& sv, inst->blas_root_byte_offset, inst->instance_id, inst->custom_id, inst_flags2, - leaf_order + i, ihdr->geometry_index, ctx, perf); if (sv.miss) return; } @@ -676,7 +396,7 @@ void walk_bvh4_subtree(SceneView& sv, float t_near = 0.f; ++perf.bvh_box_tests; if (!ray_aabb_intersect(ro, rd, mn, mx, - ctx.tmin, box_cull_t(ctx), t_near)) { + ctx.tmin, ctx.best_t, t_near)) { continue; } hits[hit_count++] = { child_off, t_near }; @@ -799,10 +519,6 @@ WalkCtx init_ctx(const RtuReq& req, uint32_t t, ctx.best_geom = 0; ctx.any_hit = false; ctx.best_kv = false; - ctx.best_order = 0; - ctx.best_blas_tab = 0; - ctx.tlas_tab = 0; - vcopy3(ctx.world_o, ro); vcopy3(ctx.world_d, rd); ctx.yield_pending = false; ctx.yield_t = ctx.tmax; ctx.yield_u = 0.f; ctx.yield_v = 0.f; ctx.yield_prim = 0; ctx.yield_sbt = 0; @@ -961,11 +677,8 @@ WalkResult FlatWalker::walk_lane(const RtuReq& req, uint32_t t, SceneView& sv, if (cls.action == TriAction::Ignore) continue; if (cls.action == TriAction::Commit) { - const bool takes = commit_takes(sv, ctx, t_hit, hit_order(inst_idx, i), - 0, nullptr, inst_idx, 0, i); - if (takes) { + if (commit_takes(ctx, t_hit, inst_idx, 0, i)) { ctx.best_t = t_hit; ctx.best_u = u; ctx.best_v = v; - ctx.best_order = hit_order(inst_idx, i); ctx.best_prim = i; ctx.best_instance = inst_idx; ctx.best_custom = cur_custom; @@ -1027,7 +740,7 @@ WalkResult Bvh4Walker::walk_lane(const RtuReq& req, uint32_t t, SceneView& sv, WalkCtx ctx = init_ctx(req, t, ro, rd, l); // Top-level (non-instanced) triangles carry no instance flags. - walk_bvh4_subtree(sv, ro, rd, root_off, 0, 0, 0, 0, 0, ctx, perf); + walk_bvh4_subtree(sv, ro, rd, root_off, 0, 0, 0, ctx, perf); if (sv.miss) return {true, false}; return {false, emit_lane_result(req, l, t, ctx)}; diff --git a/tests/raytracing/Makefile b/tests/raytracing/Makefile index 2b2857e18c..da11b9fee6 100644 --- a/tests/raytracing/Makefile +++ b/tests/raytracing/Makefile @@ -18,7 +18,7 @@ TESTS := \ rt_smoke_proc rt_smoke_bvh6 rt_bvh_multinode rt_smoke_numctx \ rt_smoke_inst_flags rt_smoke_tlas_builder rt_smoke_deep_stack \ rt_smoke_ahs_custom rt_smoke_fat_leaf rt_smoke_ahs_geom \ - rt_smoke_host_cfg rt_smoke_deep_tlas rt_smoke_tie rt_raycast + rt_smoke_host_cfg rt_smoke_deep_tlas rt_raycast # --- common exclude list --------------------------------------------- EXCLUDE := diff --git a/tests/raytracing/rt_smoke_tie/Makefile b/tests/raytracing/rt_smoke_tie/Makefile deleted file mode 100644 index db41a4f2ea..0000000000 --- a/tests/raytracing/rt_smoke_tie/Makefile +++ /dev/null @@ -1,22 +0,0 @@ -ROOT_DIR := $(realpath ../../..) -include $(ROOT_DIR)/config.mk - -CONFIGS := $(if $(findstring -DVX_CFG_EXT_RTU_ENABLE,$(CONFIGS)),$(CONFIGS),$(CONFIGS) -DVX_CFG_EXT_RTU_ENABLE) -# CW-BVH4 scene -> build the RTU as a CW-BVH4 walker. -CONFIGS += -DVX_CFG_RTU_BVH_WIDTH=4 - -PROJECT := rt_smoke_tie - -SRC_DIR := $(VORTEX_HOME)/tests/raytracing/$(PROJECT) - -SRCS := $(SRC_DIR)/main.cpp -HDRS := $(SRC_DIR)/common.h $(SRC_DIR)/golden.h - -VX_SRCS := $(SRC_DIR)/kernel.cpp -VX_HDRS := $(SRC_DIR)/common.h - -OPTS ?= - -KERNEL_LIB := vortex2 - -include ../common.mk diff --git a/tests/raytracing/rt_smoke_tie/common.h b/tests/raytracing/rt_smoke_tie/common.h deleted file mode 100644 index 6c8dec3582..0000000000 --- a/tests/raytracing/rt_smoke_tie/common.h +++ /dev/null @@ -1,44 +0,0 @@ -// Copyright © 2019-2023 -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#ifndef _RT_SMOKE_TIE_COMMON_H_ -#define _RT_SMOKE_TIE_COMMON_H_ - -#include - -typedef struct { - float origin[3]; - float dir[3]; - float tmin; - float tmax; -} tie_ray_t; - -typedef struct { - uint32_t status; - uint32_t t; // float bits - uint32_t u; - uint32_t v; - uint32_t prim; - uint32_t geom; // geometry index | VX_RT_HIT_BACK_FACING - uint32_t inst_id; - uint32_t inst_custom; -} tie_result_t; - -typedef struct { - uint64_t rays_addr; - uint64_t results_addr; - uint32_t scene; - uint32_t count; -} kernel_arg_t; - -#endif // _RT_SMOKE_TIE_COMMON_H_ diff --git a/tests/raytracing/rt_smoke_tie/golden.h b/tests/raytracing/rt_smoke_tie/golden.h deleted file mode 100644 index 3ce413f215..0000000000 --- a/tests/raytracing/rt_smoke_tie/golden.h +++ /dev/null @@ -1,522 +0,0 @@ -// Generated by rt_smoke_tie -g on the SimX reference. Do not edit. -#pragma once -#include -static const uint32_t kGoldenCount = 256; -static const uint32_t kGolden[2][256][8] = { - { - { 0x0, 0x40000000, 0x3e800000, 0x3f000000, 25, 0x80000000, 2, 0x102 }, - { 0x0, 0x40400000, 0x3e800000, 0x3e800000, 64, 0x00000002, 4, 0x104 }, - { 0x0, 0x40000000, 0x3f000000, 0x3e800000, 0, 0x80000000, 4, 0x104 }, - { 0x0, 0x403ffffc, 0x80000000, 0x3e800000, 69, 0x00000002, 5, 0x105 }, - { 0x0, 0x40000000, 0x3f000000, 0x3e800000, 2, 0x80000000, 4, 0x104 }, - { 0x0, 0x40400000, 0x3e800000, 0x3e800000, 68, 0x00000002, 4, 0x104 }, - { 0x0, 0x40000000, 0x3f000000, 0x3e800000, 4, 0x80000000, 4, 0x104 }, - { 0x0, 0x403fffff, 0x3e800000, 0x3e800000, 70, 0x00000002, 4, 0x104 }, - { 0x0, 0x3fe00000, 0x80000000, 0x80000000, 0, 0x80000000, 3, 0x103 }, - { 0x0, 0x403ffffc, 0x00000000, 0x3f400000, 64, 0x00000002, 5, 0x105 }, - { 0x0, 0x3ff55555, 0x3eaaaaab, 0x80000000, 2, 0x80000000, 3, 0x103 }, - { 0x0, 0x403ffffb, 0x00000000, 0x3f400000, 66, 0x00000002, 5, 0x105 }, - { 0x0, 0x40000000, 0x80000000, 0x80000000, 12, 0x80000000, 4, 0x104 }, - { 0x0, 0x403ffffc, 0x00000000, 0x3f400000, 68, 0x00000002, 5, 0x105 }, - { 0x0, 0x40000000, 0x80000000, 0x3f800000, 39, 0x80000001, 4, 0x104 }, - { 0x0, 0x403ffffb, 0x00000000, 0x3f400000, 70, 0x00000002, 5, 0x105 }, - { 0x0, 0x40000000, 0x3e800000, 0x3f000000, 27, 0x80000000, 2, 0x102 }, - { 0x0, 0x403aaaab, 0x3f0aaaab, 0x3e000000, 0, 0x00000000, 3, 0x103 }, - { 0x0, 0x3fe00000, 0x80000000, 0x3e000000, 3, 0x80000000, 3, 0x103 }, - { 0x0, 0x40300000, 0x3f600000, 0x3e000000, 2, 0x00000000, 3, 0x103 }, - { 0x0, 0x3ff55555, 0x3e555555, 0x3e000000, 4, 0x80000000, 3, 0x103 }, - { 0x0, 0x403ffffc, 0x80000000, 0x3e800000, 15, 0x00000000, 5, 0x105 }, - { 0x0, 0x40000000, 0x3f000000, 0x3e800000, 12, 0x80000000, 4, 0x104 }, - { 0x0, 0x40400000, 0x3e800000, 0x3e800000, 14, 0x00000000, 4, 0x104 }, - { 0x0, 0x3fe00000, 0x80000000, 0x3f000000, 1, 0x80000000, 3, 0x103 }, - { 0x0, 0x403ffffb, 0x00000000, 0x3f400000, 72, 0x00000002, 5, 0x105 }, - { 0x0, 0x3ff55555, 0x3eaaaaab, 0x3e2aaaab, 3, 0x80000000, 3, 0x103 }, - { 0x0, 0x403aaaab, 0x3e955555, 0x3ec00000, 2, 0x00000000, 3, 0x103 }, - { 0x0, 0x40000000, 0x80000000, 0x80000000, 20, 0x80000000, 4, 0x104 }, - { 0x0, 0x40300000, 0x3f200000, 0x3ec00000, 4, 0x00000000, 3, 0x103 }, - { 0x0, 0x40000000, 0x80000000, 0x80000000, 22, 0x80000000, 4, 0x104 }, - { 0x0, 0x403ffffc, 0x3f400000, 0x00000000, 15, 0x00000000, 5, 0x105 }, - { 0x0, 0x40000000, 0x3e800000, 0x3f000000, 29, 0x80000000, 2, 0x102 }, - { 0x0, 0x403aaaab, 0x3d2aaaab, 0x3f200000, 0, 0x00000000, 3, 0x103 }, - { 0x0, 0x3fe00000, 0x80000000, 0x3f200000, 3, 0x80000000, 3, 0x103 }, - { 0x0, 0x40300000, 0x3ec00000, 0x3f200000, 2, 0x00000000, 3, 0x103 }, - { 0x0, 0x3ff55555, 0x3eaaaaab, 0x3e955555, 5, 0x80000000, 3, 0x103 }, - { 0x0, 0x40400000, 0x3e800000, 0x3e800000, 84, 0x00000002, 4, 0x104 }, - { 0x0, 0x40000000, 0x3f000000, 0x3e800000, 20, 0x80000000, 4, 0x104 }, - { 0x0, 0x403fffff, 0x3e800000, 0x3e800000, 86, 0x00000002, 4, 0x104 }, - { 0x0, 0x3fe00000, 0x80000000, 0x3f800000, 1, 0x80000000, 3, 0x103 }, - { 0x0, 0x403ffffc, 0x00000000, 0x3f400000, 80, 0x00000002, 5, 0x105 }, - { 0x0, 0x3ff55555, 0x3eaaaaab, 0x3f2aaaab, 3, 0x80000000, 3, 0x103 }, - { 0x0, 0x403aaaab, 0x3f2aaaab, 0x3e555555, 3, 0x00000000, 3, 0x103 }, - { 0x0, 0x40000000, 0x80000000, 0x80000000, 28, 0x80000000, 4, 0x104 }, - { 0x0, 0x40300000, 0x3e000000, 0x3f600000, 4, 0x00000000, 3, 0x103 }, - { 0x0, 0x40000000, 0x80000000, 0x3f800000, 55, 0x80000001, 4, 0x104 }, - { 0x0, 0x403ffffb, 0x00000000, 0x3f400000, 86, 0x00000002, 5, 0x105 }, - { 0x0, 0x40000000, 0x3e800000, 0x3f000000, 31, 0x80000000, 2, 0x102 }, - { 0x0, 0x403ffffc, 0x80000000, 0x3e800000, 27, 0x00000000, 5, 0x105 }, - { 0x0, 0x40000000, 0x3f000000, 0x3e800000, 24, 0x80000000, 4, 0x104 }, - { 0x0, 0x403ffffb, 0x80000000, 0x3e800000, 93, 0x00000002, 5, 0x105 }, - { 0x0, 0x40000000, 0x3f000000, 0x3e800000, 26, 0x80000000, 4, 0x104 }, - { 0x0, 0x403ffffc, 0x80000000, 0x3e800000, 31, 0x00000000, 5, 0x105 }, - { 0x0, 0x40000000, 0x3f000000, 0x3e800000, 28, 0x80000000, 4, 0x104 }, - { 0x0, 0x40400000, 0x3e800000, 0x3e800000, 30, 0x00000000, 4, 0x104 }, - { 0x0, 0x40000000, 0x80000000, 0x3f800000, 25, 0x80000000, 4, 0x104 }, - { 0x0, 0x403ffffb, 0x00000000, 0x3f400000, 88, 0x00000002, 5, 0x105 }, - { 0x0, 0x40000000, 0x80000000, 0x3f800000, 24, 0x80000000, 4, 0x104 }, - { 0x0, 0x403ffffc, 0x3f400000, 0x00000000, 27, 0x00000000, 5, 0x105 }, - { 0x0, 0x40000000, 0x80000000, 0x3f800000, 26, 0x80000000, 4, 0x104 }, - { 0x0, 0x403ffffb, 0x00000000, 0x3f400000, 92, 0x00000002, 5, 0x105 }, - { 0x0, 0x40000000, 0x80000000, 0x3f800000, 28, 0x80000000, 4, 0x104 }, - { 0x0, 0x403ffffc, 0x3f400000, 0x00000000, 31, 0x00000000, 5, 0x105 }, - { 0x0, 0x3f800000, 0x3d878f40, 0x3f6f0e18, 25, 0x80000000, 2, 0x102 }, - { 0x0, 0x3f7ffffa, 0x3f000003, 0x3ef794dc, 73, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7ffff9, 0x359f34db, 0x3f79a6dc, 81, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f6ba787, 0x3e658a10, 0x3f3c1362, 2, 0x80000000, 6, 0x106 }, - { 0x0, 0x3f800000, 0x3f000000, 0x3e1e8470, 23, 0x80000000, 4, 0x104 }, - { 0x0, 0x3f7ffffa, 0x3effffe2, 0x3e4e1b00, 83, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7ffff8, 0x3f78c284, 0x3ce7af42, 64, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7fffff, 0x3ee053b8, 0x3d7d626e, 70, 0x00000002, 4, 0x104 }, - { 0x0, 0x3f800000, 0x3ea018d0, 0x3e3fce60, 22, 0x80000000, 4, 0x104 }, - { 0x0, 0x3f7ffff9, 0x3f000008, 0x3e5d7235, 89, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f760638, 0x3f30bc73, 0x3e87ee1f, 1, 0x00000000, 3, 0x103 }, - { 0x0, 0x3f7ffff9, 0x3f000007, 0x3ea7d5e3, 73, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f800000, 0x3e9848a0, 0x3f33dbb0, 24, 0x80000000, 4, 0x104 }, - { 0x0, 0x3f7ffff9, 0x3d8967f0, 0x3edda620, 10, 0x00000000, 5, 0x105 }, - { 0x0, 0x3f7ffff9, 0x3f000010, 0x3dee19ee, 73, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f6052c3, 0x3f3318d4, 0x3dc04463, 1, 0x80000000, 6, 0x106 }, - { 0x0, 0x3f800000, 0x3e6c8d40, 0x3f44dcb0, 20, 0x80000000, 4, 0x104 }, - { 0x0, 0x3f800000, 0x3f000000, 0x3d11e140, 15, 0x00000000, 4, 0x104 }, - { 0x0, 0x3f64fc82, 0x3bc646d0, 0x3f4c7ef1, 5, 0x00000000, 6, 0x106 }, - { 0x0, 0x3f7ffff8, 0x35593b39, 0x3dd898f2, 73, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f711e1a, 0x3e072700, 0x3d39ae65, 0, 0x80000000, 3, 0x103 }, - { 0x0, 0x3f7ffff9, 0x3528d448, 0x3eb95dc0, 71, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7ffff9, 0x3efffff8, 0x3de88e01, 89, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7ffff7, 0x3effffc8, 0x3effd970, 95, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f800000, 0x3f101458, 0x3edfd750, 0, 0x80000000, 4, 0x104 }, - { 0x0, 0x3f5cae6d, 0x3bd39700, 0x3f4d2acb, 0, 0x80000000, 6, 0x106 }, - { 0x0, 0x3f7ffffa, 0x3f17030b, 0x3ed1f9e6, 84, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7ffff8, 0x3ec09e8e, 0x3f1fb0b7, 84, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7a839a, 0x3e549290, 0x3e4e945d, 2, 0x80000000, 3, 0x103 }, - { 0x0, 0x3f76e818, 0x3f26eea0, 0x3e667111, 5, 0x00000000, 3, 0x103 }, - { 0x0, 0x3f7ffff9, 0x3db5e363, 0x3ed2872d, 80, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f681e3d, 0x3e549f8d, 0x3f3039b5, 2, 0x00000000, 3, 0x103 }, - { 0x0, 0x3f800000, 0x3f000000, 0x3e71bc10, 1, 0x80000000, 4, 0x104 }, - { 0x0, 0x3f7ffff8, 0x3f000016, 0x3efda1c0, 71, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7ffffa, 0x3f792cae, 0x3cda68fc, 66, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7ffff8, 0x3ee56754, 0x3d54c4c7, 64, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f6c4aea, 0x3dc95bac, 0x3f42ada5, 7, 0x80000000, 3, 0x103 }, - { 0x0, 0x3f7ffff9, 0x3effffe0, 0x3d8aba81, 73, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f79d359, 0x3daefdf9, 0x3f0be1b6, 0, 0x00000000, 3, 0x103 }, - { 0x0, 0x3f7ffff7, 0x34ff3b86, 0x3f0311e8, 71, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f52c715, 0x3e0ee67a, 0x3be60824, 9, 0x80000001, 6, 0x106 }, - { 0x0, 0x3f7a6dcc, 0x3e9dc06c, 0x3e9eb79b, 4, 0x00000000, 3, 0x103 }, - { 0x0, 0x3f7ffff9, 0x3f141d4b, 0x3ed7c559, 92, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7ffffe, 0x3f000000, 0x3dc5f27d, 71, 0x00000002, 4, 0x104 }, - { 0x0, 0x3f7fffff, 0x3e4fea70, 0x3e980ac8, 8, 0x80000000, 4, 0x104 }, - { 0x0, 0x3f7ffffa, 0x34b55665, 0x3eaa99c0, 93, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7fffff, 0x3f000001, 0x3eb315ca, 79, 0x00000002, 4, 0x104 }, - { 0x0, 0x3f800000, 0x3eebaa80, 0x3f0a2ac0, 31, 0x00000000, 2, 0x102 }, - { 0x0, 0x3f800000, 0x3e8b4633, 0x3f3a5ce5, 4, 0x80000000, 4, 0x104 }, - { 0x0, 0x3f7ffff9, 0x3f000003, 0x3e611d40, 65, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f70d836, 0x3ed96353, 0x3e57f246, 5, 0x80000000, 6, 0x106 }, - { 0x0, 0x3f7ffff8, 0x3e374280, 0x3ea45ea0, 26, 0x00000000, 5, 0x105 }, - { 0x0, 0x3f800000, 0x3c80fb00, 0x3ef7f050, 10, 0x80000000, 4, 0x104 }, - { 0x0, 0x3f7ffffa, 0x3efffff1, 0x3e944807, 89, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7ffffa, 0x3d35a8a0, 0x3ee94ae0, 30, 0x00000000, 5, 0x105 }, - { 0x0, 0x3f7ffff7, 0x3effffda, 0x3e582790, 93, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f719475, 0x3e7218da, 0x3eaa6b9e, 3, 0x80000000, 3, 0x103 }, - { 0x0, 0x3f6494a2, 0x3f1fb7ed, 0x3ea1fde7, 6, 0x00000000, 3, 0x103 }, - { 0x0, 0x3f6d25ea, 0x3f445833, 0x3e11348c, 2, 0x00000000, 3, 0x103 }, - { 0x0, 0x3f7ffff8, 0x3efffffd, 0x3e5c5ea8, 71, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f800000, 0x80000000, 0x3f0ff018, 25, 0x80000000, 4, 0x104 }, - { 0x0, 0x3f7ffff9, 0x344380da, 0x3f63f931, 73, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7ffffa, 0x3f4152a4, 0x3e7ab55f, 90, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7ffff7, 0x3effffe3, 0x3e852084, 93, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7fffff, 0x3f0da5a0, 0x3ee4b4c0, 26, 0x80000000, 4, 0x104 }, - { 0x0, 0x3f7ffffa, 0x34ac4364, 0x3f1de4e0, 73, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f5c2c0e, 0x3e2ef1fa, 0x3f3d53a4, 1, 0x80000000, 6, 0x106 }, - { 0x0, 0x3f7ffff8, 0x3ef77fe1, 0x3f04400c, 90, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f800000, 0x3f5f3c5c, 0x3e030e90, 28, 0x80000000, 4, 0x104 }, - { 0x0, 0x3f6d1af5, 0x3e33a054, 0x3f33789d, 4, 0x00000000, 3, 0x103 }, - { 0x0, 0x3f7ffffa, 0x34c00000, 0x3e1a338f, 69, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f66463e, 0x3f1cf9a6, 0x3ea4a508, 0, 0x00000000, 3, 0x103 }, - { 0x0, 0x3f6d250b, 0x3db836a6, 0x3f67868d, 5, 0x80000000, 3, 0x103 }, - { 0x0, 0x3f7ffff9, 0x3e3f4086, 0x3ea05f96, 88, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f79948f, 0x3f23bab4, 0x3e656df9, 1, 0x00000000, 3, 0x103 }, - { 0x0, 0x3f7ffff7, 0x3e60fbae, 0x3f47c0fa, 66, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f800000, 0x3f000000, 0x3ea2b2a8, 15, 0x80000000, 4, 0x104 }, - { 0x0, 0x3f7ffff8, 0x3ed3c057, 0x3db0feed, 72, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7fffff, 0x3edc8262, 0x3d8df67f, 70, 0x00000002, 4, 0x104 }, - { 0x0, 0x3f6fc2c9, 0x3f587690, 0x3e075479, 7, 0x00000000, 3, 0x103 }, - { 0x0, 0x3f800000, 0x3e5b9270, 0x3f491b64, 2, 0x80000000, 4, 0x104 }, - { 0x0, 0x3f7b01d8, 0x3ecf4c13, 0x3e5732ce, 0, 0x00000000, 3, 0x103 }, - { 0x0, 0x3f7ffffa, 0x3ed963a9, 0x3d9a71ca, 68, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7ffff9, 0x3f4cfd2e, 0x3e4c0af3, 88, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f800000, 0x3eceb0a8, 0x3dc53d60, 2, 0x80000000, 4, 0x104 }, - { 0x0, 0x3f7ffff8, 0x3c88b1f1, 0x3f7bba6f, 88, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7ffff8, 0x345d935e, 0x3eec9aed, 69, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f763bbb, 0x3eb37160, 0x3eab065a, 0, 0x00000000, 3, 0x103 }, - { 0x0, 0x3f7fffff, 0x3e397208, 0x3f51a37e, 28, 0x80000000, 4, 0x104 }, - { 0x0, 0x3f7b5ec8, 0x3f058734, 0x3da57d7b, 0, 0x00000000, 3, 0x103 }, - { 0x0, 0x3f7ffff9, 0x33bebb08, 0x3f75d83c, 95, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f78d7b2, 0x3e9dcb1f, 0x3eaf3a81, 4, 0x00000000, 3, 0x103 }, - { 0x0, 0x3f800000, 0x3e976a48, 0x3f344adc, 25, 0x80000000, 2, 0x102 }, - { 0x0, 0x3f7ffff9, 0x3eb63614, 0x3e1393eb, 66, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f800000, 0x3e965948, 0x3e534d70, 30, 0x00000000, 4, 0x104 }, - { 0x0, 0x3f7ffff8, 0x3e06ad8a, 0x3ebca923, 88, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f800000, 0x3ed7580c, 0x3da29fd0, 14, 0x80000000, 4, 0x104 }, - { 0x0, 0x3f66a751, 0x3e0696c8, 0x3f5b5f23, 0, 0x00000000, 3, 0x103 }, - { 0x0, 0x3f7fffff, 0x3f000000, 0x3eb0795d, 79, 0x00000002, 4, 0x104 }, - { 0x0, 0x3f7ffff9, 0x3eba09a4, 0x3e0bec7a, 72, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7fffff, 0x3f679846, 0x3dc33dd0, 39, 0x80000001, 2, 0x102 }, - { 0x0, 0x3f7ffffa, 0x350ed981, 0x3eed9810, 65, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7ffff7, 0x3e9b0a14, 0x3f327af5, 88, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7ffff8, 0x358ceea7, 0x3f188ac5, 87, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f77a9cc, 0x3c9c7f58, 0x3ed2f161, 0, 0x80000000, 6, 0x106 }, - { 0x0, 0x3f7ffff9, 0x35400000, 0x3dc9c940, 31, 0x00000000, 5, 0x105 }, - { 0x0, 0x3f7ffff8, 0x3599f5c4, 0x3f4fae1d, 81, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7ffff9, 0x354744a4, 0x3f1d1291, 79, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f800000, 0x3f648878, 0x3ddbbc40, 28, 0x80000000, 4, 0x104 }, - { 0x0, 0x3f7ffffa, 0x35200000, 0x3ccd882c, 27, 0x00000000, 5, 0x105 }, - { 0x0, 0x3f7fffff, 0x3e9a021b, 0x3e4bfbd5, 70, 0x00000002, 4, 0x104 }, - { 0x0, 0x3f800000, 0x3f000000, 0x3da72fe0, 15, 0x00000000, 4, 0x104 }, - { 0x0, 0x3f800000, 0x3e151f80, 0x3eb57040, 22, 0x80000000, 4, 0x104 }, - { 0x0, 0x3f800000, 0x3f000000, 0x3edd2330, 15, 0x00000000, 4, 0x104 }, - { 0x0, 0x3f5a6702, 0x3c4d5104, 0x3f6cfda3, 1, 0x80000000, 6, 0x106 }, - { 0x0, 0x3f64eaef, 0x3d8ea9a7, 0x3f640db7, 6, 0x00000000, 3, 0x103 }, - { 0x0, 0x3f800000, 0x32944b5a, 0x3f12896c, 81, 0x80000002, 4, 0x104 }, - { 0x0, 0x3f7ffffa, 0x3cadbb04, 0x3ef52438, 92, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7ffff9, 0x3d556639, 0x3ee55350, 88, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7ffff8, 0x3effffda, 0x3e9cec8b, 83, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f800000, 0x3f68a73c, 0x3dbac620, 4, 0x80000000, 4, 0x104 }, - { 0x0, 0x3f7ffff9, 0x34de28f9, 0x3ef147c6, 85, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f738af7, 0x3e740404, 0x3f1a12dd, 2, 0x80000000, 6, 0x106 }, - { 0x0, 0x3f7ffffa, 0x3f1b3c9e, 0x3ec986bc, 84, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f800000, 0x3bf5c480, 0x3efc28ee, 26, 0x80000000, 4, 0x104 }, - { 0x0, 0x3f695785, 0x3e11de66, 0x3f3e2607, 2, 0x00000000, 3, 0x103 }, - { 0x0, 0x3f6bfc68, 0x3de2b043, 0x3f15b4a2, 5, 0x00000000, 6, 0x106 }, - { 0x0, 0x3f7ffffa, 0x3f4163bc, 0x3e7a70cb, 76, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f4ff44b, 0x3e293406, 0x3ec7959a, 1, 0x00000000, 6, 0x106 }, - { 0x0, 0x3f532d8b, 0x3f196c52, 0x3eb804a3, 1, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f7cdcd6, 0x3e64bb34, 0x3f33fe34, 66, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7e2de1, 0x3ef9c2c5, 0x3ef063c4, 90, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f531696, 0x3f1291a6, 0x3cc460ca, 0, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f7eac1d, 0x3f53136e, 0x3e13d4fd, 90, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f535e35, 0x3ea5d7ee, 0x3e900b53, 0, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f5373f3, 0x3ea61344, 0x3e912be7, 0, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f53b04a, 0x3e4a6ad8, 0x3ed5cf27, 0, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f537b21, 0x3f1bd906, 0x3ebe90d2, 1, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f537ef1, 0x3f1bf784, 0x3e298ebf, 1, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f8086e6, 0x3cca58cb, 0x3f02df2f, 93, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7d37fe, 0x3ed975f9, 0x3f0294f7, 90, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f532e5c, 0x3e4ad678, 0x3ecd7a7d, 0, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f817b05, 0x3d8e21db, 0x3eae59d9, 93, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f53644a, 0x3f1b2249, 0x3e6c1301, 1, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f536ada, 0x3e6d9970, 0x3ebfe0e0, 0, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f7d1615, 0x3f643c3c, 0x3d24840f, 90, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f806420, 0x3c962fe7, 0x3f36281e, 93, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7cfc7f, 0x3f34250b, 0x3e6717bc, 90, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7e0010, 0x3ee8bc6a, 0x3eff4456, 66, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f811cfb, 0x3d55bc6f, 0x3dbe83d4, 93, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f53b277, 0x3f089ff1, 0x3da79e02, 0, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f534ae7, 0x3eeb3474, 0x3e12f3ee, 0, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f537348, 0x3cf986e6, 0x3f13ce02, 0, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f4b6fe1, 0x3cc535ac, 0x3f081940, 1, 0x00000000, 6, 0x106 }, - { 0x0, 0x3f80084c, 0x3ac71c69, 0x3db9cd41, 69, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7edddf, 0x3e77aaee, 0x3f3b487a, 90, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f536fc6, 0x3c0b33db, 0x3f195160, 0, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f80bcc7, 0x3d0d94d9, 0x3e01d49c, 69, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f5334eb, 0x3f099f0e, 0x3d80423a, 0, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f7cabe8, 0x3e41b8e7, 0x3f3b9934, 66, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f532ef9, 0x3f1977c6, 0x3e85db3e, 1, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f535503, 0x3cc2bcc1, 0x3f14922d, 0, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f538608, 0x3ecabf0b, 0x3e5b42e7, 0, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f80a973, 0x3cfe2ba5, 0x3ef96bd2, 93, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f535d6c, 0x3f03219a, 0x3dbe4dfd, 0, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f5343a5, 0x3ea800d4, 0x3e8c3979, 0, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f81d8a8, 0x3d8fb50c, 0x3c8627a9, 92, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f5369f1, 0x3e00b76b, 0x3ef64354, 0, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f80688e, 0x3c9cd3f8, 0x3f75bcda, 69, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f5355c1, 0x3ed87cdc, 0x3e39be66, 0, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f7e3aa2, 0x3f735972, 0x3c019519, 90, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7f9177, 0x3f1a20d9, 0x3ec68fde, 90, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f4facc6, 0x3e20435d, 0x3ecd6d43, 1, 0x00000000, 6, 0x106 }, - { 0x0, 0x3f532fe7, 0x3de1b188, 0x3efa920c, 0, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f539dbd, 0x3f1cede8, 0x3e1d79f8, 1, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f7f6244, 0x3f48709f, 0x3e4f73d8, 90, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f809a25, 0x3ce73741, 0x3f00af59, 69, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7d9e68, 0x3f05869f, 0x3ed85fa1, 90, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f53ae22, 0x3e2a8867, 0x3ee59de4, 0, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f538038, 0x3e9d4a69, 0x3e9ab90a, 0, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f80e2ce, 0x3d2a1a42, 0x3ef7c7d2, 93, 0x00000002, 5, 0x105 }, - { 0x1, 0x00000000, 0x00000000, 0x00000000, 0, 0x00000000, 0, 0x0 }, - { 0x0, 0x3f53901d, 0x3ee09e4c, 0x3e30c6eb, 0, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f80e782, 0x3d2da155, 0x3e649458, 69, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f53777d, 0x3eb4a8ba, 0x3e82cf14, 0, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f805d9e, 0x3c8c6c45, 0x3f03f177, 93, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f80f545, 0x3d37f3cd, 0x3e503679, 69, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f538794, 0x3ef8afd2, 0x3dff25a1, 0, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f7f6e29, 0x3c89b8c4, 0x3f78472b, 66, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f5355c8, 0x3f0d7a08, 0x3d53432d, 0, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f531799, 0x3dc676f8, 0x3effdbcb, 0, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f7df3f6, 0x3eb0e7aa, 0x3f1b43ee, 66, 0x00000002, 5, 0x105 }, - }, - { - { 0x0, 0x40000000, 0x80000000, 0x3e800000, 0, 0x80000000, 0, 0x100 }, - { 0x0, 0x403ffffb, 0x3f400000, 0x3e800000, 64, 0x00000002, 5, 0x105 }, - { 0x0, 0x40000000, 0x80000000, 0x3e800000, 2, 0x80000000, 0, 0x100 }, - { 0x0, 0x403ffffc, 0x3f400000, 0x3e800000, 2, 0x00000000, 5, 0x105 }, - { 0x0, 0x40000000, 0x80000000, 0x3e800000, 4, 0x80000000, 0, 0x100 }, - { 0x0, 0x403ffffb, 0x3f400000, 0x3e800000, 68, 0x00000002, 5, 0x105 }, - { 0x0, 0x40000000, 0x80000000, 0x3e800000, 6, 0x80000000, 0, 0x100 }, - { 0x0, 0x403ffffc, 0x3f400000, 0x3e800000, 6, 0x00000000, 5, 0x105 }, - { 0x0, 0x3fe00000, 0x80000000, 0x80000000, 0, 0x80000000, 1, 0x101 }, - { 0x0, 0x403ffffc, 0x00000000, 0x3f400000, 0, 0x00000000, 5, 0x105 }, - { 0x0, 0x3ff55555, 0x3eaaaaab, 0x80000000, 2, 0x80000000, 1, 0x101 }, - { 0x0, 0x403ffffb, 0x00000000, 0x3f400000, 66, 0x00000002, 5, 0x105 }, - { 0x0, 0x40000000, 0x3f000000, 0x3f000000, 5, 0x80000000, 0, 0x100 }, - { 0x0, 0x403ffffc, 0x00000000, 0x3f400000, 4, 0x00000000, 5, 0x105 }, - { 0x0, 0x40000000, 0x3f000000, 0x3f000000, 7, 0x80000000, 0, 0x100 }, - { 0x0, 0x403ffffb, 0x00000000, 0x3f400000, 70, 0x00000002, 5, 0x105 }, - { 0x0, 0x40000000, 0x80000000, 0x3e800000, 8, 0x80000000, 0, 0x100 }, - { 0x0, 0x403aaaab, 0x3f0aaaab, 0x3e000000, 0, 0x00000000, 1, 0x101 }, - { 0x0, 0x3fe00000, 0x80000000, 0x3e000000, 3, 0x80000000, 1, 0x101 }, - { 0x0, 0x40300000, 0x3f600000, 0x3e000000, 2, 0x00000000, 1, 0x101 }, - { 0x0, 0x3ff55555, 0x3e555555, 0x3e000000, 4, 0x80000000, 1, 0x101 }, - { 0x0, 0x403ffffc, 0x3f400000, 0x3e800000, 12, 0x00000000, 5, 0x105 }, - { 0x0, 0x40000000, 0x80000000, 0x3e800000, 14, 0x80000000, 0, 0x100 }, - { 0x0, 0x403ffffb, 0x3f400000, 0x3e800000, 78, 0x00000002, 5, 0x105 }, - { 0x0, 0x3fe00000, 0x80000000, 0x3f000000, 1, 0x80000000, 1, 0x101 }, - { 0x0, 0x403ffffb, 0x00000000, 0x3f400000, 72, 0x00000002, 5, 0x105 }, - { 0x0, 0x3ff55555, 0x3eaaaaab, 0x3e2aaaab, 3, 0x80000000, 1, 0x101 }, - { 0x0, 0x403aaaab, 0x3e955555, 0x3ec00000, 2, 0x00000000, 1, 0x101 }, - { 0x0, 0x40000000, 0x3f000000, 0x3f000000, 13, 0x80000000, 0, 0x100 }, - { 0x0, 0x40300000, 0x3f200000, 0x3ec00000, 4, 0x00000000, 1, 0x101 }, - { 0x0, 0x40000000, 0x3f000000, 0x3f000000, 15, 0x80000000, 0, 0x100 }, - { 0x0, 0x403ffffc, 0x00000000, 0x3f400000, 14, 0x00000000, 5, 0x105 }, - { 0x0, 0x40000000, 0x80000000, 0x3e800000, 16, 0x80000000, 0, 0x100 }, - { 0x0, 0x403aaaab, 0x3d2aaaab, 0x3f200000, 0, 0x00000000, 1, 0x101 }, - { 0x0, 0x3fe00000, 0x80000000, 0x3f200000, 3, 0x80000000, 1, 0x101 }, - { 0x0, 0x40300000, 0x3ec00000, 0x3f200000, 2, 0x00000000, 1, 0x101 }, - { 0x0, 0x3ff55555, 0x3eaaaaab, 0x3e955555, 5, 0x80000000, 1, 0x101 }, - { 0x0, 0x403ffffb, 0x3f400000, 0x3e800000, 84, 0x00000002, 5, 0x105 }, - { 0x0, 0x40000000, 0x80000000, 0x3e800000, 22, 0x80000000, 0, 0x100 }, - { 0x0, 0x403ffffc, 0x3f400000, 0x3e800000, 22, 0x00000000, 5, 0x105 }, - { 0x0, 0x3fe00000, 0x80000000, 0x3f800000, 1, 0x80000000, 1, 0x101 }, - { 0x0, 0x403ffffc, 0x00000000, 0x3f400000, 16, 0x00000000, 5, 0x105 }, - { 0x0, 0x3ff55555, 0x3eaaaaab, 0x3f2aaaab, 3, 0x80000000, 1, 0x101 }, - { 0x0, 0x403aaaab, 0x3f2aaaab, 0x3e555555, 3, 0x00000000, 1, 0x101 }, - { 0x0, 0x40000000, 0x3f000000, 0x3f000000, 21, 0x80000000, 0, 0x100 }, - { 0x0, 0x40300000, 0x3e000000, 0x3f600000, 4, 0x00000000, 1, 0x101 }, - { 0x0, 0x40000000, 0x3f000000, 0x3f000000, 23, 0x80000000, 0, 0x100 }, - { 0x0, 0x403ffffb, 0x00000000, 0x3f400000, 86, 0x00000002, 5, 0x105 }, - { 0x0, 0x40000000, 0x80000000, 0x3e800000, 24, 0x80000000, 0, 0x100 }, - { 0x0, 0x403ffffc, 0x3f400000, 0x3e800000, 24, 0x00000000, 5, 0x105 }, - { 0x0, 0x40000000, 0x80000000, 0x3e800000, 26, 0x80000000, 0, 0x100 }, - { 0x0, 0x403ffffb, 0x3f400000, 0x3e800000, 90, 0x00000002, 5, 0x105 }, - { 0x0, 0x40000000, 0x80000000, 0x3e800000, 28, 0x80000000, 0, 0x100 }, - { 0x0, 0x403ffffc, 0x3f400000, 0x3e800000, 28, 0x00000000, 5, 0x105 }, - { 0x0, 0x40000000, 0x80000000, 0x3e800000, 30, 0x80000000, 0, 0x100 }, - { 0x0, 0x403ffffb, 0x3f400000, 0x3e800000, 94, 0x00000002, 5, 0x105 }, - { 0x0, 0x40000000, 0x3f000000, 0x3f000000, 25, 0x80000000, 0, 0x100 }, - { 0x0, 0x403ffffb, 0x00000000, 0x3f400000, 88, 0x00000002, 5, 0x105 }, - { 0x0, 0x40000000, 0x3f000000, 0x3f000000, 27, 0x80000000, 0, 0x100 }, - { 0x0, 0x403ffffc, 0x00000000, 0x3f400000, 26, 0x00000000, 5, 0x105 }, - { 0x0, 0x40000000, 0x3f000000, 0x3f000000, 29, 0x80000000, 0, 0x100 }, - { 0x0, 0x403ffffb, 0x00000000, 0x3f400000, 92, 0x00000002, 5, 0x105 }, - { 0x0, 0x40000000, 0x3f000000, 0x3f000000, 31, 0x80000000, 0, 0x100 }, - { 0x0, 0x403ffffc, 0x00000000, 0x3f400000, 30, 0x00000000, 5, 0x105 }, - { 0x0, 0x3f800000, 0x80000000, 0x3d878f40, 1, 0x80000000, 0, 0x100 }, - { 0x0, 0x3f7ffffa, 0x3f000003, 0x3ef794dc, 73, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7ffff9, 0x359f34db, 0x3f79a6dc, 81, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f6ba787, 0x3e658a10, 0x3f3c1362, 2, 0x80000000, 6, 0x106 }, - { 0x0, 0x3f800000, 0x3eb0bdc8, 0x3f27a11c, 22, 0x80000000, 0, 0x100 }, - { 0x0, 0x3f7ffffa, 0x3effffe2, 0x3e4e1b00, 83, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7ffff8, 0x3f78c284, 0x3ce7af42, 64, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7fffff, 0x3ee053b8, 0x3d7d626e, 70, 0x00000002, 4, 0x104 }, - { 0x0, 0x3f800000, 0x3f500c68, 0x3e3fce60, 22, 0x80000000, 0, 0x100 }, - { 0x0, 0x3f7ffff9, 0x3f000008, 0x3e5d7235, 89, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f760638, 0x3f30bc73, 0x3e87ee1f, 1, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f7ffff9, 0x3f000007, 0x3ea7d5e3, 73, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f800000, 0x3f000000, 0x3e4f6ec0, 27, 0x80000000, 0, 0x100 }, - { 0x0, 0x3f7ffff9, 0x3d8967f0, 0x3edda620, 10, 0x00000000, 5, 0x105 }, - { 0x0, 0x3f7ffff9, 0x3f000010, 0x3dee19ee, 73, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f6052c3, 0x3f3318d4, 0x3dc04463, 1, 0x80000000, 6, 0x106 }, - { 0x0, 0x3f800000, 0x3f000000, 0x3e89b960, 23, 0x80000000, 0, 0x100 }, - { 0x0, 0x3f800000, 0x3eedc3d8, 0x3f091e14, 14, 0x00000000, 0, 0x100 }, - { 0x0, 0x3f64fc82, 0x3bc646d0, 0x3f4c7ef1, 5, 0x00000000, 6, 0x106 }, - { 0x0, 0x3f7ffff8, 0x35593b39, 0x3dd898f2, 73, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f711e1a, 0x3e072700, 0x3d39ae65, 0, 0x80000000, 1, 0x101 }, - { 0x0, 0x3f7ffff9, 0x3528d448, 0x3eb95dc0, 71, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7ffff9, 0x3efffff8, 0x3de88e01, 89, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7ffff7, 0x3effffc8, 0x3effd970, 95, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f800000, 0x3d80a2c0, 0x3edfd750, 2, 0x80000000, 0, 0x100 }, - { 0x0, 0x3f5cae6d, 0x3bd39700, 0x3f4d2acb, 0, 0x80000000, 6, 0x106 }, - { 0x0, 0x3f7ffffa, 0x3f17030b, 0x3ed1f9e6, 84, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7ffff8, 0x3ec09e8e, 0x3f1fb0b7, 84, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7a839a, 0x3e549290, 0x3e4e945d, 2, 0x80000000, 1, 0x101 }, - { 0x0, 0x3f76e818, 0x3f26eea0, 0x3e667111, 5, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f7ffff9, 0x3db5e363, 0x3ed2872d, 80, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f681e3d, 0x3e549f8d, 0x3f3039b5, 2, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f800000, 0x3e8721f8, 0x3f3c6f04, 0, 0x80000000, 0, 0x100 }, - { 0x0, 0x3f7ffff8, 0x3f000016, 0x3efda1c0, 71, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7ffffa, 0x3f792cae, 0x3cda6900, 2, 0x00000000, 5, 0x105 }, - { 0x0, 0x3f7ffff8, 0x3ee56754, 0x3d54c4c7, 64, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f6c4aea, 0x3dc95bac, 0x3f42ada5, 7, 0x80000000, 1, 0x101 }, - { 0x0, 0x3f7ffff9, 0x3effffe0, 0x3d8aba81, 73, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f79d359, 0x3daefdf9, 0x3f0be1b6, 0, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f7ffff7, 0x34ff3b86, 0x3f0311e8, 71, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f52c715, 0x3e0ee67a, 0x3be60824, 9, 0x80000001, 6, 0x106 }, - { 0x0, 0x3f7a6dcc, 0x3e9dc06c, 0x3e9eb79b, 4, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f7ffff9, 0x3f141d4b, 0x3ed7c55a, 28, 0x00000000, 5, 0x105 }, - { 0x0, 0x3f7ffffe, 0x3f000000, 0x3dc5f27d, 71, 0x00000002, 4, 0x104 }, - { 0x0, 0x3f7fffff, 0x3f33fa9c, 0x3e980ac8, 8, 0x80000000, 0, 0x100 }, - { 0x0, 0x3f7ffffa, 0x34b55665, 0x3eaa99c0, 93, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7fffff, 0x3f000001, 0x3eb315ca, 79, 0x00000002, 4, 0x104 }, - { 0x0, 0x3f800000, 0x80000000, 0x3eebaa80, 25, 0x00000000, 0, 0x100 }, - { 0x0, 0x3f800000, 0x3efffffe, 0x3e69739a, 7, 0x80000000, 0, 0x100 }, - { 0x0, 0x3f7ffff9, 0x3f000003, 0x3e611d40, 65, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f70d836, 0x3ed96353, 0x3e57f246, 5, 0x80000000, 6, 0x106 }, - { 0x0, 0x3f7ffff8, 0x3e374280, 0x3ea45ea0, 26, 0x00000000, 5, 0x105 }, - { 0x0, 0x3f800000, 0x3f0407d8, 0x3ef7f050, 10, 0x80000000, 0, 0x100 }, - { 0x0, 0x3f7ffffa, 0x3efffff1, 0x3e944807, 89, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7ffffa, 0x3d35a8a0, 0x3ee94ae0, 30, 0x00000000, 5, 0x105 }, - { 0x0, 0x3f7ffff7, 0x3effffda, 0x3e582790, 93, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f719475, 0x3e7218da, 0x3eaa6b9e, 3, 0x80000000, 1, 0x101 }, - { 0x0, 0x3f6494a2, 0x3f1fb7ed, 0x3ea1fde7, 6, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f6d25ea, 0x3f445833, 0x3e11348c, 2, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f7ffff8, 0x3efffffd, 0x3e5c5ea8, 71, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f800000, 0x3efffffe, 0x3d7f0190, 25, 0x80000000, 0, 0x100 }, - { 0x0, 0x3f7ffff9, 0x344380da, 0x3f63f931, 73, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7ffffa, 0x3f4152a4, 0x3e7ab55f, 90, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7ffff7, 0x3effffe3, 0x3e852084, 93, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7fffff, 0x3d5a5a00, 0x3ee4b4c0, 28, 0x80000000, 0, 0x100 }, - { 0x0, 0x3f7ffffa, 0x34a00000, 0x3f1de4e1, 9, 0x00000000, 5, 0x105 }, - { 0x0, 0x3f5c2c0e, 0x3e2ef1fa, 0x3f3d53a4, 1, 0x80000000, 6, 0x106 }, - { 0x0, 0x3f7ffff8, 0x3ef77fe1, 0x3f04400c, 90, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f800000, 0x3ebe78b8, 0x3e030e90, 30, 0x80000000, 0, 0x100 }, - { 0x0, 0x3f6d1af5, 0x3e33a054, 0x3f33789d, 4, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f7ffffa, 0x34c00000, 0x3e1a338f, 69, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f66463e, 0x3f1cf9a6, 0x3ea4a508, 0, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f6d250b, 0x3db836a6, 0x3f67868d, 5, 0x80000000, 1, 0x101 }, - { 0x0, 0x3f7ffff9, 0x3e3f4086, 0x3ea05f96, 88, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f79948f, 0x3f23bab4, 0x3e656df9, 1, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f7ffff7, 0x3e60fbae, 0x3f47c0fa, 66, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f800000, 0x3e3a9ab2, 0x3f515954, 14, 0x80000000, 0, 0x100 }, - { 0x0, 0x3f7ffff8, 0x3ed3c057, 0x3db0feed, 72, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7fffff, 0x3edc8262, 0x3d8df67f, 70, 0x00000002, 4, 0x104 }, - { 0x0, 0x3f6fc2c9, 0x3f587690, 0x3e075479, 7, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f800000, 0x3efffffe, 0x3e9236ca, 5, 0x80000000, 0, 0x100 }, - { 0x0, 0x3f7b01d8, 0x3ecf4c13, 0x3e5732ce, 0, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f7ffffa, 0x3ed963a9, 0x3d9a71ca, 68, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7ffff9, 0x3f4cfd30, 0x3e4c0af0, 24, 0x00000000, 5, 0x105 }, - { 0x0, 0x3f800000, 0x3f675854, 0x3dc53d60, 2, 0x80000000, 0, 0x100 }, - { 0x0, 0x3f7ffff8, 0x3c88b1f1, 0x3f7bba6f, 88, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7ffff8, 0x345d935e, 0x3eec9aed, 69, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f763bbb, 0x3eb37160, 0x3eab065a, 0, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f7fffff, 0x3f000000, 0x3ea346fc, 31, 0x80000000, 0, 0x100 }, - { 0x0, 0x3f7b5ec8, 0x3f058734, 0x3da57d7b, 0, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f7ffff9, 0x33bebb08, 0x3f75d83c, 95, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f78d7b2, 0x3e9dcb1f, 0x3eaf3a81, 4, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f800000, 0x80000000, 0x3e976a48, 1, 0x80000000, 0, 0x100 }, - { 0x0, 0x3f7ffff9, 0x3eb63614, 0x3e1393eb, 66, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f800000, 0x3f4b2ca4, 0x3e534d70, 30, 0x00000000, 0, 0x100 }, - { 0x0, 0x3f7ffff8, 0x3e06ad8a, 0x3ebca923, 88, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f800000, 0x3f6bac06, 0x3da29fd0, 14, 0x80000000, 0, 0x100 }, - { 0x0, 0x3f66a751, 0x3e0696c8, 0x3f5b5f23, 0, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f7fffff, 0x3f000000, 0x3eb0795d, 79, 0x00000002, 4, 0x104 }, - { 0x0, 0x3f7ffff9, 0x3eba09a4, 0x3e0bec7a, 72, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7fffff, 0x3dc33dd0, 0x3f679846, 28, 0x80000000, 0, 0x100 }, - { 0x0, 0x3f7ffffa, 0x35000000, 0x3eed9814, 1, 0x00000000, 5, 0x105 }, - { 0x0, 0x3f7ffff7, 0x3e9b0a14, 0x3f327af5, 88, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7ffff8, 0x35800000, 0x3f188ac8, 23, 0x00000000, 5, 0x105 }, - { 0x0, 0x3f77a9cc, 0x3c9c7f58, 0x3ed2f161, 0, 0x80000000, 6, 0x106 }, - { 0x0, 0x3f7ffff9, 0x35400000, 0x3dc9c940, 31, 0x00000000, 5, 0x105 }, - { 0x0, 0x3f7ffff8, 0x3599f5c4, 0x3f4fae1d, 81, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7ffff9, 0x354744a4, 0x3f1d1291, 79, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f800000, 0x3ec910f0, 0x3ddbbc40, 30, 0x80000000, 0, 0x100 }, - { 0x0, 0x3f7ffffa, 0x35200000, 0x3ccd882c, 27, 0x00000000, 5, 0x105 }, - { 0x0, 0x3f7fffff, 0x3e9a021b, 0x3e4bfbd5, 70, 0x00000002, 4, 0x104 }, - { 0x0, 0x3f800000, 0x3ed63408, 0x3f14e5fc, 14, 0x00000000, 0, 0x100 }, - { 0x0, 0x3f800000, 0x3f2547e0, 0x3eb57040, 22, 0x80000000, 0, 0x100 }, - { 0x0, 0x3f800000, 0x3d8b7340, 0x3f6e9198, 14, 0x00000000, 0, 0x100 }, - { 0x0, 0x3f5a6702, 0x3c4d5104, 0x3f6cfda3, 1, 0x80000000, 6, 0x106 }, - { 0x0, 0x3f64eaef, 0x3d8ea9a7, 0x3f640db7, 6, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f800000, 0x3efffffc, 0x3d944b70, 17, 0x80000000, 0, 0x100 }, - { 0x0, 0x3f7ffffa, 0x3cadbb04, 0x3ef52438, 92, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7ffff9, 0x3d556639, 0x3ee55350, 88, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7ffff8, 0x3effffda, 0x3e9cec8b, 83, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f800000, 0x3ed14e78, 0x3dbac620, 6, 0x80000000, 0, 0x100 }, - { 0x0, 0x3f7ffff9, 0x34de28f9, 0x3ef147c6, 85, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f738af7, 0x3e740404, 0x3f1a12dd, 2, 0x80000000, 6, 0x106 }, - { 0x0, 0x3f7ffffa, 0x3f1b3c9e, 0x3ec986bc, 84, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f800000, 0x3f01eb89, 0x3efc28ee, 26, 0x80000000, 0, 0x100 }, - { 0x0, 0x3f695785, 0x3e11de66, 0x3f3e2607, 2, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f6bfc68, 0x3de2b043, 0x3f15b4a2, 5, 0x00000000, 6, 0x106 }, - { 0x0, 0x3f7ffffa, 0x3f4163bc, 0x3e7a70cb, 76, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f4ff44b, 0x3e293406, 0x3ec7959a, 1, 0x00000000, 6, 0x106 }, - { 0x0, 0x3f532d8b, 0x3f196c52, 0x3eb804a3, 1, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f7cdcd6, 0x3e64bb34, 0x3f33fe34, 66, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7e2de1, 0x3ef9c2c5, 0x3ef063c4, 90, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f531696, 0x3f1291a6, 0x3cc460ca, 0, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f7eac1d, 0x3f53136e, 0x3e13d4fd, 90, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f535e35, 0x3ea5d7ee, 0x3e900b53, 0, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f5373f3, 0x3ea61344, 0x3e912be7, 0, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f53b04a, 0x3e4a6ad8, 0x3ed5cf27, 0, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f537b21, 0x3f1bd906, 0x3ebe90d2, 1, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f537ef1, 0x3f1bf784, 0x3e298ebf, 1, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f8086e6, 0x3cca58cb, 0x3f02df2f, 93, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7d37fe, 0x3ed975f9, 0x3f0294f7, 90, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f532e5c, 0x3e4ad678, 0x3ecd7a7d, 0, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f817b05, 0x3d8e21db, 0x3eae59d9, 93, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f53644a, 0x3f1b2249, 0x3e6c1301, 1, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f536ada, 0x3e6d9970, 0x3ebfe0e0, 0, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f7d1615, 0x3f643c3c, 0x3d24840f, 90, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f806420, 0x3c962fe7, 0x3f36281e, 93, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7cfc7f, 0x3f34250b, 0x3e6717bc, 90, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7e0010, 0x3ee8bc6a, 0x3eff4456, 66, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f811cfb, 0x3d55bc6f, 0x3dbe83d4, 93, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f53b277, 0x3f089ff1, 0x3da79e02, 0, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f534ae7, 0x3eeb3474, 0x3e12f3ee, 0, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f537348, 0x3cf986e6, 0x3f13ce02, 0, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f4b6fe1, 0x3cc535ac, 0x3f081940, 1, 0x00000000, 6, 0x106 }, - { 0x0, 0x3f80084c, 0x3ac71c69, 0x3db9cd41, 69, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7edddf, 0x3e77aaee, 0x3f3b487a, 90, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f536fc6, 0x3c0b33db, 0x3f195160, 0, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f80bcc7, 0x3d0d94d9, 0x3e01d49c, 69, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f5334eb, 0x3f099f0e, 0x3d80423a, 0, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f7cabe8, 0x3e41b8e7, 0x3f3b9934, 66, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f532ef9, 0x3f1977c6, 0x3e85db3e, 1, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f535503, 0x3cc2bcc1, 0x3f14922d, 0, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f538608, 0x3ecabf0b, 0x3e5b42e7, 0, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f80a973, 0x3cfe2ba5, 0x3ef96bd2, 93, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f535d6c, 0x3f03219a, 0x3dbe4dfd, 0, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f5343a5, 0x3ea800d4, 0x3e8c3979, 0, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f81d8a8, 0x3d8fb50c, 0x3c8627a9, 92, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f5369f1, 0x3e00b76b, 0x3ef64354, 0, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f80688e, 0x3c9cd3f8, 0x3f75bcda, 69, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f5355c1, 0x3ed87cdc, 0x3e39be66, 0, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f7e3aa2, 0x3f735972, 0x3c019519, 90, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7f9177, 0x3f1a20d9, 0x3ec68fde, 90, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f4facc6, 0x3e20435d, 0x3ecd6d43, 1, 0x00000000, 6, 0x106 }, - { 0x0, 0x3f532fe7, 0x3de1b188, 0x3efa920c, 0, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f539dbd, 0x3f1cede8, 0x3e1d79f8, 1, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f7f6244, 0x3f48709f, 0x3e4f73d8, 90, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f809a25, 0x3ce73741, 0x3f00af59, 69, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f7d9e68, 0x3f05869f, 0x3ed85fa1, 90, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f53ae22, 0x3e2a8867, 0x3ee59de4, 0, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f538038, 0x3e9d4a69, 0x3e9ab90a, 0, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f80e2ce, 0x3d2a1a42, 0x3ef7c7d2, 93, 0x00000002, 5, 0x105 }, - { 0x1, 0x00000000, 0x00000000, 0x00000000, 0, 0x00000000, 0, 0x0 }, - { 0x0, 0x3f53901d, 0x3ee09e4c, 0x3e30c6eb, 0, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f80e782, 0x3d2da155, 0x3e649458, 69, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f53777d, 0x3eb4a8ba, 0x3e82cf14, 0, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f805d9e, 0x3c8c6c45, 0x3f03f177, 93, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f80f545, 0x3d37f3cd, 0x3e503679, 69, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f538794, 0x3ef8afd2, 0x3dff25a1, 0, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f7f6e29, 0x3c89b8c4, 0x3f78472b, 66, 0x00000002, 5, 0x105 }, - { 0x0, 0x3f5355c8, 0x3f0d7a08, 0x3d53432d, 0, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f531799, 0x3dc676f8, 0x3effdbcb, 0, 0x00000000, 1, 0x101 }, - { 0x0, 0x3f7df3f6, 0x3eb0e7aa, 0x3f1b43ee, 66, 0x00000002, 5, 0x105 }, - }, -}; diff --git a/tests/raytracing/rt_smoke_tie/kernel.cpp b/tests/raytracing/rt_smoke_tie/kernel.cpp deleted file mode 100644 index 42aba68f51..0000000000 --- a/tests/raytracing/rt_smoke_tie/kernel.cpp +++ /dev/null @@ -1,41 +0,0 @@ -// Copyright © 2019-2023 -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -#include -#include -#include "common.h" - -// One ray per thread against one scene: every hit is opaque, so the trace -// ends on the committed hit. -__kernel void kernel_main(kernel_arg_t* arg) { - uint32_t i = blockIdx.x * blockDim.x + threadIdx.x; - if (i >= arg->count) return; - - const tie_ray_t& r = ((const tie_ray_t*)((uintptr_t)arg->rays_addr))[i]; - vx_ray_t ray = { {r.origin[0], r.origin[1], r.origin[2]}, - {r.dir[0], r.dir[1], r.dir[2]}, - r.tmin, r.tmax }; - uint32_t h = vx_rt_wtrace(arg->scene, 0u, VX_RT_FLAG_OPAQUE, 0xffu, &ray); - vx_hit_t hit; - uint32_t sts = vx_rt_wait(h, &hit); - - tie_result_t* res = (tie_result_t*)((uintptr_t)arg->results_addr) + i; - res->status = sts; - res->t = __builtin_bit_cast(uint32_t, hit.t); - res->u = __builtin_bit_cast(uint32_t, hit.u); - res->v = __builtin_bit_cast(uint32_t, hit.v); - res->prim = hit.primitive_id; - res->geom = hit.geometry_index | (hit.back_facing ? VX_RT_HIT_BACK_FACING : 0u); - res->inst_id = hit.instance_id; - res->inst_custom = hit.instance_custom; -} diff --git a/tests/raytracing/rt_smoke_tie/main.cpp b/tests/raytracing/rt_smoke_tie/main.cpp deleted file mode 100644 index f0548a5243..0000000000 --- a/tests/raytracing/rt_smoke_tie/main.cpp +++ /dev/null @@ -1,671 +0,0 @@ -// Copyright © 2019-2023 -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. -// -// rt_smoke_tie — host driver: which of several equal / near-equal-t opaque -// hits the RTU commits. -// -// The scene stacks coincident and near-coincident geometry, within one BLAS -// (exact twin triangles, a copy an ulp-scale step off the plane) and across -// instances (a BLAS instanced twice in place, rotated a quarter turn onto -// itself, shifted by half a cell, lifted by 2^-20, and a second BLAS holding -// copies of two of the first one's triangles). The -// RTU walks its own CW-BVH4; which of those hits it keeps is settled by the -// source BVH's visit order, carried as the visit-order tables the Vulkan -// driver appends (vortexpipe vp_launch.c): a TLAS table inside the scene and -// compact BLAS tables in a separate buffer reached by scene offset, triangle -// leaves naming their parent/side, instance leaves naming their BLAS table, -// TLAS rank and the TLAS table. Here the "source" trees are binary -// median-split trees built differently from the CW-BVH4, so their order is -// not the RTU's own. -// -// The same scene is traced twice: with the tables, and with the instance -// leaves' table words zeroed (the static (instance, geometry, primitive) tie -// key). Expected results are the SimX reference's (golden.h, regenerated with -// -g); every field of a hit is compared bit-exactly, and the two runs must disagree on -// some rays, or the tables were never consulted. - -#include -#include -#include -#include -#include -#include -#include - -#include -#include -#include -#include "common.h" -#include "golden.h" - -#define RT_CHECK(_expr) \ - do { \ - int _ret = _expr; \ - if (0 == _ret) break; \ - printf("Error: '%s' returned %d!\n", #_expr, (int)_ret); \ - cleanup(); \ - exit(-1); \ - } while (false) - -namespace { - -const char* kernel_file = "kernel.vxbin"; - -vx_device_h device = nullptr; -vx_queue_h queue = nullptr; -vx_module_h module_ = nullptr; -vx_kernel_h kernel = nullptr; -vx_buffer_h scene_buf[2] = { nullptr, nullptr }; -vx_buffer_h tab_buf = nullptr; -vx_buffer_h rays_buf = nullptr; -vx_buffer_h res_buf = nullptr; - -void cleanup() { - if (!device) return; - for (auto b : scene_buf) if (b) vx_buffer_release(b); - if (tab_buf) vx_buffer_release(tab_buf); - if (rays_buf) vx_buffer_release(rays_buf); - if (res_buf) vx_buffer_release(res_buf); - if (kernel) vx_kernel_release(kernel); - if (module_) vx_module_release(module_); - if (queue) vx_queue_release(queue); - vx_device_release(device); - device = nullptr; -} - -constexpr uint32_t kRoot = 0xffffffffu; - -struct Tri { - float v[9]; - uint32_t geom; -}; - -struct Instance { - float otw[12]; // object -> world - float wto[12]; // world -> object, as the source driver stores it - uint32_t blas; - uint32_t id; - uint32_t custom; -}; - -struct Box { - float mn[3], mx[3]; -}; - -Box tri_box(const Tri& t) { - Box b; - for (int a = 0; a < 3; ++a) { - b.mn[a] = std::fmin(t.v[a], std::fmin(t.v[3 + a], t.v[6 + a])); - b.mx[a] = std::fmax(t.v[a], std::fmax(t.v[3 + a], t.v[6 + a])); - } - return b; -} - -Box box_union(const Box& x, const Box& y) { - Box b; - for (int a = 0; a < 3; ++a) { - b.mn[a] = std::fmin(x.mn[a], y.mn[a]); - b.mx[a] = std::fmax(x.mx[a], y.mx[a]); - } - return b; -} - -// The world box of a BLAS box under an affine map, widened by an ulp. -Box xform_box(const float m[12], const Box& b) { - Box r; - for (int a = 0; a < 3; ++a) { r.mn[a] = INFINITY; r.mx[a] = -INFINITY; } - for (int c = 0; c < 8; ++c) { - const double p[3] = { (c & 1) ? b.mx[0] : b.mn[0], - (c & 2) ? b.mx[1] : b.mn[1], - (c & 4) ? b.mx[2] : b.mn[2] }; - for (int a = 0; a < 3; ++a) { - const double w = m[a * 4] * p[0] + m[a * 4 + 1] * p[1] + m[a * 4 + 2] * p[2] + m[a * 4 + 3]; - r.mn[a] = std::fmin(r.mn[a], std::nextafter(float(w), -INFINITY)); - r.mx[a] = std::fmax(r.mx[a], std::nextafter(float(w), INFINITY)); - } - } - return r; -} - -// ── the source BVH: a binary median-split tree ────────────────────────── -struct SrcNode { - Box box; - int child[2]; // node index, or ~item for a leaf -}; - -struct SrcTree { - std::vector nodes; - int root = 0; - - int build(const std::vector& boxes, std::vector items, int depth) { - if (items.size() == 1) return ~int(items[0]); - // split on the axis that cycles with depth, so the order differs from - // the CW-BVH4's longest-axis split - const int axis = depth % 3; - std::stable_sort(items.begin(), items.end(), [&](uint32_t a, uint32_t b) { - return boxes[a].mn[axis] + boxes[a].mx[axis] < boxes[b].mn[axis] + boxes[b].mx[axis]; - }); - const size_t h = items.size() / 2; - std::vector l(items.begin(), items.begin() + h), r(items.begin() + h, items.end()); - const int me = int(nodes.size()); - nodes.push_back(SrcNode()); - const int c0 = build(boxes, l, depth + 1); - const int c1 = build(boxes, r, depth + 1); - nodes[me].child[0] = c0; - nodes[me].child[1] = c1; - auto cbox = [&](int c) { return c < 0 ? boxes[~c] : nodes[c].box; }; - nodes[me].box = box_union(cbox(c0), cbox(c1)); - return me; - } - - void make(const std::vector& boxes) { - nodes.clear(); - std::vector items(boxes.size()); - for (uint32_t i = 0; i < items.size(); ++i) items[i] = i; - root = build(boxes, items, 0); - } - - Box child_box(const std::vector& boxes, int c) const { - return c < 0 ? boxes[~c] : nodes[c].box; - } -}; - -// The source driver's table order: depth first, child 0 first. Table node -// j records its parent/side and depth; each leaf its parent/side, and leaves -// are ranked in visit order. -struct SrcTable { - struct Node { int src; uint32_t ps, depth; }; - std::vector nodes; - std::vector leaf_ps; // per item - std::vector leaf_rank; // per item - std::vector rank_item; // per rank - - void make(const SrcTree& t, size_t n_items) { - nodes.clear(); - leaf_ps.assign(n_items, kRoot); - leaf_rank.assign(n_items, 0); - rank_item.clear(); - std::vector> st = { { t.root, kRoot } }; - while (!st.empty()) { - auto [c, ps] = st.back(); - st.pop_back(); - if (c < 0) { - leaf_ps[~c] = ps; - leaf_rank[~c] = uint32_t(rank_item.size()); - rank_item.push_back(uint32_t(~c)); - continue; - } - const uint32_t idx = uint32_t(nodes.size()); - nodes.push_back({ c, ps, ps == kRoot ? 0u : nodes[ps >> 1].depth + 1 }); - st.push_back({ t.nodes[c].child[1], idx << 1 | 1u }); - st.push_back({ t.nodes[c].child[0], idx << 1 }); - } - } -}; - -// ── byte image helpers ────────────────────────────────────────────────── -struct Image { - std::vector b; - uint32_t alloc(uint32_t n, uint32_t align = 4) { - const uint32_t off = uint32_t((b.size() + align - 1) & ~size_t(align - 1)); - b.resize(off + n, 0); - return off; - } - void u32(uint32_t off, uint32_t v) { std::memcpy(&b[off], &v, 4); } - void f32(uint32_t off, float v) { std::memcpy(&b[off], &v, 4); } - void box(uint32_t off, const Box& x) { - for (int a = 0; a < 3; ++a) { f32(off + 4 * a, x.mn[a]); f32(off + 12 + 4 * a, x.mx[a]); } - } -}; - -// ── the RTU's CW-BVH4: post-order, children before parents ───────────── -struct RtuRef { uint32_t off; Box box; bool leaf; }; - -RtuRef emit_internal(Image& img, const std::vector& ch) { - Box nb = ch[0].box; - for (size_t i = 1; i < ch.size(); ++i) nb = box_union(nb, ch[i].box); - int8_t ex[3]; - float step[3]; - for (int a = 0; a < 3; ++a) { - const float ext = nb.mx[a] - nb.mn[a]; - int e = -20; - // headroom: 250 steps must cover the extent, so the 255 the quantizer - // may round up to still lies past it - if (ext > 0.f) std::frexp(ext / 250.0f, &e); - ex[a] = int8_t(std::max(-100, std::min(100, e))); - step[a] = std::ldexp(1.0f, ex[a]); - } - const uint32_t off = img.alloc(RTU_BVH_NODE4_BYTES); - img.u32(off, RTU_BVH_KIND_INTERNAL | (uint32_t(ch.size()) << RTU_BVH_COUNT_SHIFT)); - for (int a = 0; a < 3; ++a) img.f32(off + RTU_BVH_NODE_ORIGIN_OFF + 4 * a, nb.mn[a]); - std::memcpy(&img.b[off + RTU_BVH_NODE_EXP_OFF], ex, 3); - const uint32_t qmin_off = RTU_BVH_NODE_CHILD_OFF + 4 * 4; - const uint32_t qmax_off = qmin_off + 3 * 4; - for (size_t i = 0; i < ch.size(); ++i) { - img.u32(off + RTU_BVH_NODE_CHILD_OFF + 4 * uint32_t(i), - ch[i].off | (ch[i].leaf ? RTU_BVH_CHILD_LEAF_FLAG : 0u)); - for (int a = 0; a < 3; ++a) { - // conservative by a step either way - int qmn = int(std::floor((ch[i].box.mn[a] - nb.mn[a]) / step[a])) - 1; - int qmx = int(std::ceil((ch[i].box.mx[a] - nb.mn[a]) / step[a])) + 1; - qmn = std::max(0, std::min(255, qmn)); - qmx = std::max(qmn, std::min(255, qmx)); - img.b[off + qmin_off + 3 * i + a] = uint8_t(qmn); - img.b[off + qmax_off + 3 * i + a] = uint8_t(qmx); - } - } - return { off, nb, false }; -} - -// Groups of up to four along the longest centroid axis, recursively. -template -RtuRef emit_cwbvh(Image& img, const std::vector& boxes, std::vector items, - EmitLeaf&& leaf) { - if (items.size() == 1) return { leaf(items[0]), boxes[items[0]], true }; - float cmn[3] = { INFINITY, INFINITY, INFINITY }, cmx[3] = { -INFINITY, -INFINITY, -INFINITY }; - for (uint32_t i : items) - for (int a = 0; a < 3; ++a) { - const float c = boxes[i].mn[a] + boxes[i].mx[a]; - cmn[a] = std::min(cmn[a], c); - cmx[a] = std::max(cmx[a], c); - } - int axis = 0; - for (int a = 1; a < 3; ++a) - if (cmx[a] - cmn[a] > cmx[axis] - cmn[axis]) axis = a; - std::stable_sort(items.begin(), items.end(), [&](uint32_t a, uint32_t b) { - return boxes[a].mn[axis] + boxes[a].mx[axis] > boxes[b].mn[axis] + boxes[b].mx[axis]; - }); - const size_t g = std::min(4, items.size()); - std::vector ch; - for (size_t k = 0; k < g; ++k) { - const size_t lo = items.size() * k / g, hi = items.size() * (k + 1) / g; - ch.push_back(emit_cwbvh(img, boxes, std::vector(items.begin() + lo, items.begin() + hi), leaf)); - } - return emit_internal(img, ch); -} - -// ── the scene ─────────────────────────────────────────────────────────── -std::vector> blases; -std::vector insts; - -void make_geometry() { - // BLAS 0, the "sail": a 4x4 grid of quads on z = 0 over [-1, 1]^2 (geometry - // 0); an exact twin of it (geometry 1); a copy an ulp-scale step above the - // plane on some vertices (geometry 2). - std::vector sail; - auto quad = [&](float x0, float y0, float x1, float y1, uint32_t geom, auto zf) { - Tri a = { { x0, y0, zf(x0, y0), x1, y0, zf(x1, y0), x1, y1, zf(x1, y1) }, geom }; - Tri b = { { x0, y0, zf(x0, y0), x1, y1, zf(x1, y1), x0, y1, zf(x0, y1) }, geom }; - sail.push_back(a); - sail.push_back(b); - }; - for (uint32_t g = 0; g < 3; ++g) { - for (int j = 0; j < 4; ++j) { - for (int i = 0; i < 4; ++i) { - const float x0 = -1.f + 0.5f * i, y0 = -1.f + 0.5f * j; - auto z = [&](float x, float y) -> float { - (void)i; (void)j; - switch (g) { - // every other grid vertex: about an ulp of t for these rays - case 2: return (int((x + 1.f) * 4.f + (y + 1.f) * 4.f) & 2) ? 0x1p-22f : 0.f; - default: return 0.f; - } - }; - quad(x0, y0, x0 + 0.5f, y0 + 0.5f, g, z); - } - } - } - blases.push_back(sail); - - // BLAS 1: a small plate tilted across the sail, plus an exact copy of two - // sail triangles (so an instance of it coincides with the sail there). - std::vector plate; - for (int i = 0; i < 4; ++i) { - const float x0 = -0.75f + 0.375f * i; - plate.push_back({ { x0, -0.5f, -0.25f, x0 + 0.375f, -0.5f, 0.25f, x0 + 0.375f, 0.5f, 0.25f }, 0 }); - plate.push_back({ { x0, -0.5f, -0.25f, x0 + 0.375f, 0.5f, 0.25f, x0, 0.5f, -0.25f }, 0 }); - } - plate.push_back({ sail[10].v[0], sail[10].v[1], sail[10].v[2], sail[10].v[3], sail[10].v[4], - sail[10].v[5], sail[10].v[6], sail[10].v[7], sail[10].v[8], 1 }); - plate.push_back({ sail[11].v[0], sail[11].v[1], sail[11].v[2], sail[11].v[3], sail[11].v[4], - sail[11].v[5], sail[11].v[6], sail[11].v[7], sail[11].v[8], 1 }); - blases.push_back(plate); - - auto inst = [&](uint32_t blas, uint32_t id, const float otw[12], const float wto[12]) { - Instance in; - std::memcpy(in.otw, otw, sizeof(in.otw)); - std::memcpy(in.wto, wto, sizeof(in.wto)); - in.blas = blas; - in.id = id; - in.custom = 0x100 + id; - insts.push_back(in); - }; - const float I[12] = { 1, 0, 0, 0, 0, 1, 0, 0, 0, 0, 1, 0 }; - // a quarter turn about z maps the grid onto itself - const float R[12] = { 0, -1, 0, 0, 1, 0, 0, 0, 0, 0, 1, 0 }; - const float Ri[12] = { 0, 1, 0, 0, -1, 0, 0, 0, 0, 0, 1, 0 }; - // half a cell along x - const float T[12] = { 1, 0, 0, 0.25f, 0, 1, 0, 0, 0, 0, 1, 0 }; - const float Ti[12] = { 1, 0, 0, -0.25f, 0, 1, 0, 0, 0, 0, 1, 0 }; - // lifted off the plane by 2^-20 - const float U[12] = { 1, 0, 0, 0, 0, 1, 0, 0, 0, 0, 1, 0x1p-20f }; - const float Ui[12] = { 1, 0, 0, 0, 0, 1, 0, 0, 0, 0, 1, -0x1p-20f }; - // the plate: a quarter turn about x, then shifted (the RTU inverts an - // instance transform as a rotation, so instances stay orthonormal) - const float P[12] = { 1, 0, 0, 0.125f, 0, 0, -1, 0, 0, 1, 0, 0 }; - const float Pi[12] = { 1, 0, 0, -0.125f, 0, 0, 1, 0, 0, -1, 0, 0 }; - inst(0, 0, I, I); - inst(1, 1, I, I); - inst(0, 2, R, Ri); - inst(1, 3, I, I); - inst(0, 4, T, Ti); - inst(0, 5, U, Ui); - inst(1, 6, P, Pi); -} - -std::vector make_rays() { - std::vector rays; - auto ray = [&](float ox, float oy, float oz, float dx, float dy, float dz) { - rays.push_back({ { ox, oy, oz }, { dx, dy, dz }, 0.001f, 100.f }); - }; - // straight down and straight up, on cell interiors, edges and vertices - for (int j = 0; j < 8; ++j) - for (int i = 0; i < 8; ++i) { - const float x = -0.875f + 0.25f * i, y = -0.875f + 0.25f * j; - const float ex = ((i + j) & 1) ? 0.125f : 0.f; // half the rays sit on an edge - if ((i + j * 8) & 1) ray(x + ex, y, 3.f, 0.f, 0.f, -1.f); - else ray(x + ex, y + ex, -2.f, 0.f, 0.f, 1.f); - } - // oblique, aimed at grid lines and vertices from scattered origins - uint32_t s = 12345u; - auto rnd = [&]() { s = s * 1664525u + 1013904223u; return float(s >> 8) / float(1u << 24); }; - for (int k = 0; k < 128; ++k) { - const float tx = -1.f + 0.25f * float(int(rnd() * 9.f)); - const float ty = -1.f + 0.5f * rnd() * 4.f; - const float ox = tx + (rnd() - 0.5f) * 3.f, oy = ty + (rnd() - 0.5f) * 3.f; - const float oz = (k & 3) ? 2.f + rnd() : -2.f - rnd(); - ray(ox, oy, oz, tx - ox, ty - oy, -oz); - } - // shallow rays skimming the plate and the sail - for (int k = 0; k < 64; ++k) { - const float ox = -3.f, oy = -0.9f + 1.8f * rnd(); - const float tz = (rnd() - 0.5f) * 0.02f; - ray(ox, oy, 0.3f + tz, 3.f, (rnd() - 0.5f) * 0.4f, -0.3f - tz * 0.5f); - } - return rays; -} - -// Build the scene image for one scene base address. tab_base is the BLAS -// table buffer's address, or 0 to leave the tables out. -struct Built { - Image scene; - Image tabs; - uint32_t root = 0; -}; - -Built build(uint64_t scene_addr, uint64_t tab_addr, bool with_tables) { - Built out; - Image& img = out.scene; - img.alloc(RTU_BVH_SCENE_HDR_BYTES); - - // BLAS: source trees + tables, then the CW-BVH4 with each leaf naming its - // parent/side in the source tree - std::vector blas_root(blases.size()), blas_tab(blases.size(), 0); - std::vector blas_box(blases.size()); - for (size_t b = 0; b < blases.size(); ++b) { - const auto& tris = blases[b]; - std::vector boxes; - for (const auto& t : tris) boxes.push_back(tri_box(t)); - SrcTree st; - st.make(boxes); - SrcTable tab; - tab.make(st, tris.size()); - - // BLAS table: { n_nodes, 0, 64, 32 }, node[j] = { own box, ps, depth } - const uint32_t n = uint32_t(tab.nodes.size()); - const uint32_t toff = out.tabs.alloc(64 + 32 * n, 64); - out.tabs.u32(toff, n); - out.tabs.u32(toff + 8, 64); - out.tabs.u32(toff + 12, 32); - for (uint32_t j = 0; j < n; ++j) { - const auto& nd = tab.nodes[j]; - const uint32_t o = toff + 64 + 32 * j; - if (nd.ps != kRoot) { - const SrcNode& parent = st.nodes[tab.nodes[nd.ps >> 1].src]; - out.tabs.box(o, st.child_box(boxes, parent.child[nd.ps & 1])); - } - out.tabs.u32(o + 24, nd.ps); - out.tabs.u32(o + 28, nd.depth); - } - blas_tab[b] = uint32_t(tab_addr + toff - scene_addr); - - std::vector items(tris.size()); - for (uint32_t i = 0; i < items.size(); ++i) items[i] = i; - RtuRef r = emit_cwbvh(img, boxes, items, [&](uint32_t i) { - const uint32_t off = img.alloc(RTU_BVH_LEAF_HDR_BYTES + RTU_BVH_TRI_STRIDE); - img.u32(off, RTU_BVH_KIND_LEAF_TRI | (1u << RTU_BVH_COUNT_SHIFT)); - img.u32(off + 4, tris[i].geom); - img.u32(off + 8, tab.leaf_ps[i]); - img.u32(off + 12, i); - for (int k = 0; k < 9; ++k) img.f32(off + RTU_BVH_LEAF_HDR_BYTES + 4 * k, tris[i].v[k]); - img.u32(off + RTU_BVH_LEAF_HDR_BYTES + 36, RTU_BVH_FLAG_OPAQUE); - return off; - }); - blas_root[b] = r.off; - blas_box[b] = r.box; - } - - // TLAS: source tree over the instances' world boxes; table in the scene - std::vector wbox; - for (const auto& in : insts) wbox.push_back(xform_box(in.otw, blas_box[in.blas])); - SrcTree tt; - tt.make(wbox); - SrcTable ttab; - ttab.make(tt, insts.size()); - const uint32_t nl = uint32_t(insts.size()), nn = uint32_t(ttab.nodes.size()); - const uint32_t noff = (16 + nl * 64 + 63) & ~63u; - const uint32_t tlas_tab = img.alloc(noff + 64 * nn, 64); - img.u32(tlas_tab, nl); - img.u32(tlas_tab + 4, nn); - img.u32(tlas_tab + 8, noff); - img.u32(tlas_tab + 12, 64); - for (uint32_t r = 0; r < nl; ++r) { - const uint32_t item = ttab.rank_item[r]; - const uint32_t o = tlas_tab + 16 + 64 * r; - img.u32(o, ttab.leaf_ps[item]); - for (int k = 0; k < 12; ++k) img.f32(o + 16 + 4 * k, insts[item].wto[k]); - } - for (uint32_t j = 0; j < nn; ++j) { - const SrcNode& nd = tt.nodes[ttab.nodes[j].src]; - const uint32_t o = tlas_tab + noff + 64 * j; - img.box(o, tt.child_box(wbox, nd.child[0])); - img.box(o + 24, tt.child_box(wbox, nd.child[1])); - img.u32(o + 48, ttab.nodes[j].ps); - img.u32(o + 52, ttab.nodes[j].depth); - } - - std::vector items(insts.size()); - for (uint32_t i = 0; i < items.size(); ++i) items[i] = i; - RtuRef root = emit_cwbvh(img, wbox, items, [&](uint32_t i) { - const Instance& in = insts[i]; - const uint32_t off = img.alloc(RTU_BVH_LEAF_HDR_BYTES + RTU_BVH_INSTANCE_STRIDE); - img.u32(off, RTU_BVH_KIND_LEAF_INST | (1u << RTU_BVH_COUNT_SHIFT)); - if (with_tables) { - img.u32(off + 4, blas_tab[in.blas]); - img.u32(off + 8, ttab.leaf_rank[i]); - img.u32(off + 12, tlas_tab); - } - const uint32_t rec = off + RTU_BVH_LEAF_HDR_BYTES; - for (int k = 0; k < 12; ++k) img.f32(rec + 4 * k, in.wto[k]); - img.u32(rec + RTU_BVH_INSTANCE_BLAS_OFF, blas_root[in.blas]); - img.u32(rec + RTU_BVH_INSTANCE_CUSTOM_OFF, in.custom); - img.u32(rec + RTU_BVH_INSTANCE_ID_OFF, in.id); - img.u32(rec + RTU_BVH_INSTANCE_CULL_OFF, 0xffu); - return off; - }); - out.root = root.off; - img.u32(0, root.off); - img.u32(4, RTU_SCENE_KIND_BVH4); - img.u32(8, uint32_t(img.b.size())); - img.u32(12, uint32_t(insts.size())); - img.alloc(0, 64); - return out; -} - -const char* field_name[8] = { "status", "t", "u", "v", "prim", "geom", "inst", "custom" }; - -} // namespace - -int main(int argc, char* argv[]) { - const char* golden_out = nullptr; - int c; - while ((c = getopt(argc, argv, "g:h")) != -1) { - if (c == 'g') golden_out = optarg; - else { printf("usage: %s [-g golden.h]\n", argv[0]); return 0; } - } - - make_geometry(); - std::vector rays = make_rays(); - const uint32_t NT = VX_CFG_NUM_THREADS; - while (rays.size() % NT) rays.push_back(rays.back()); - const uint32_t count = uint32_t(rays.size()); - - RT_CHECK(vx_device_open(0, &device)); - vx_queue_info_t qi = { sizeof(qi), nullptr, VX_QUEUE_PRIORITY_NORMAL, 0 }; - RT_CHECK(vx_queue_create(device, &qi, &queue)); - RT_CHECK(vx_module_load_file(device, kernel_file, &module_)); - RT_CHECK(vx_module_get_kernel(module_, "main", &kernel)); - - // Sizes do not depend on the addresses: size the buffers from a dry build. - const Built dry = build(0, 0, true); - const uint32_t scene_bytes = uint32_t(dry.scene.b.size()); - const uint32_t tab_bytes = uint32_t(dry.tabs.b.size()); - uint64_t scene_addr[2], tab_addr = 0; - for (int k = 0; k < 2; ++k) { - RT_CHECK(vx_buffer_create(device, scene_bytes, VX_MEM_READ, &scene_buf[k])); - RT_CHECK(vx_buffer_address(scene_buf[k], &scene_addr[k])); - } - RT_CHECK(vx_buffer_create(device, tab_bytes, VX_MEM_READ, &tab_buf)); - RT_CHECK(vx_buffer_address(tab_buf, &tab_addr)); - if (tab_addr <= scene_addr[0] || tab_addr - scene_addr[0] >= (1ull << 31)) { - printf("Error: the BLAS table buffer is not reachable from the scene\n"); - cleanup(); - return 1; - } - - RT_CHECK(vx_buffer_create(device, count * sizeof(tie_ray_t), VX_MEM_READ, &rays_buf)); - RT_CHECK(vx_buffer_create(device, count * sizeof(tie_result_t), VX_MEM_WRITE, &res_buf)); - uint64_t rays_addr = 0, res_addr = 0; - RT_CHECK(vx_buffer_address(rays_buf, &rays_addr)); - RT_CHECK(vx_buffer_address(res_buf, &res_addr)); - RT_CHECK(vx_enqueue_write(queue, rays_buf, 0, rays.data(), count * sizeof(tie_ray_t), - 0, nullptr, nullptr)); - - std::vector got[2]; - for (int k = 0; k < 2; ++k) { - const bool with_tables = (k == 0); - const Built bi = build(scene_addr[k], tab_addr, with_tables); - RT_CHECK(vx_enqueue_write(queue, scene_buf[k], 0, bi.scene.b.data(), scene_bytes, - 0, nullptr, nullptr)); - if (with_tables) - RT_CHECK(vx_enqueue_write(queue, tab_buf, 0, bi.tabs.b.data(), tab_bytes, - 0, nullptr, nullptr)); - - kernel_arg_t arg = {}; - arg.rays_addr = rays_addr; - arg.results_addr = res_addr; - arg.scene = uint32_t(scene_addr[k]); - arg.count = count; - - vx_event_h launch_ev = nullptr, read_ev = nullptr; - vx_launch_info_t li = {}; - li.struct_size = sizeof(li); - li.kernel = kernel; - li.args_host = &arg; - li.args_size = sizeof(arg); - li.ndim = 1; - li.grid_dim[0] = count / NT; - li.block_dim[0] = NT; - RT_CHECK(vx_enqueue_launch(queue, &li, 0, nullptr, &launch_ev)); - got[k].resize(count); - RT_CHECK(vx_enqueue_read(queue, got[k].data(), res_buf, 0, count * sizeof(tie_result_t), - 1, &launch_ev, &read_ev)); - RT_CHECK(vx_event_wait_value(read_ev, 1, VX_TIMEOUT_INFINITE)); - vx_event_release(read_ev); - vx_event_release(launch_ev); - } - - uint32_t hits = 0, differ = 0; - for (uint32_t i = 0; i < count; ++i) { - hits += (got[0][i].status == VX_RT_STS_DONE_HIT); - differ += (std::memcmp(&got[0][i], &got[1][i], sizeof(tie_result_t)) != 0); - } - printf("rays=%u hits=%u, table vs static-key verdicts differ on %u rays\n", count, hits, differ); - - if (golden_out) { - FILE* f = fopen(golden_out, "w"); - if (!f) { printf("Error: cannot write %s\n", golden_out); cleanup(); return 1; } - fprintf(f, "// Generated by rt_smoke_tie -g on the SimX reference. Do not edit.\n"); - fprintf(f, "#pragma once\n#include \n"); - fprintf(f, "static const uint32_t kGoldenCount = %u;\n", count); - fprintf(f, "static const uint32_t kGolden[2][%u][8] = {\n", count); - for (int k = 0; k < 2; ++k) { - fprintf(f, " {\n"); - for (uint32_t i = 0; i < count; ++i) { - const uint32_t* w = reinterpret_cast(&got[k][i]); - fprintf(f, " { 0x%x, 0x%08x, 0x%08x, 0x%08x, %u, 0x%08x, %u, 0x%x },\n", - w[0], w[1], w[2], w[3], w[4], w[5], w[6], w[7]); - } - fprintf(f, " },\n"); - } - fprintf(f, "};\n"); - fclose(f); - printf("wrote %s\n", golden_out); - cleanup(); - return 0; - } - - int errors = 0; - if (count != kGoldenCount) { - printf("Error: %u rays, golden has %u\n", count, kGoldenCount); - ++errors; - } else { - for (int k = 0; k < 2; ++k) { - for (uint32_t i = 0; i < count; ++i) { - const uint32_t* w = reinterpret_cast(&got[k][i]); - // a miss carries no hit attributes - const int nf = (kGolden[k][i][0] == VX_RT_STS_DONE_HIT) ? 8 : 1; - for (int f = 0; f < nf; ++f) { - if (w[f] != kGolden[k][i][f]) { - if (errors < 20) - printf("%s ray %u: %s got 0x%08x expected 0x%08x\n", - k ? "static-key" : "tables", i, field_name[f], w[f], kGolden[k][i][f]); - ++errors; - } - } - } - } - } - if (differ == 0) { - printf("Error: the visit-order tables changed no verdict\n"); - ++errors; - } - - cleanup(); - if (errors != 0) { - printf("FAILED with %d errors\n", errors); - return 1; - } - printf("PASSED!\n"); - return 0; -} From 7f4edff560977a17b187fae594b2916f8ad64953 Mon Sep 17 00:00:00 2001 From: Blaise Tine Date: Sat, 3 Oct 2026 13:07:55 -0700 Subject: [PATCH 21/31] rtu: commit the nearest hit; on equal t the first one found stays Drop the static (instance, geometry, primitive) key that settled exact-t ties between opaque hits. It existed so two traversal orders would agree on which of two coincident triangles is reported, which the Vulkan spec leaves to the implementation: "If t < t_max, t_max is set to t and the candidate is set as the current closest hit. If t > t_max, the candidate is dropped." (Ray Closest Hit Determination). What real RT units do -- and what this change does -- is shrink the ray interval to the committed hit. - The tri PE (RTL) and ray_triangle (SimX) now test against [t_min, best_t) instead of the ray's own t_max, so a hit is only ever reported strictly nearer than the committed one and the first hit found at a given t stays. A context has at most one triangle in flight and nothing else writes its best_t meanwhile, so the RTL's interval is the one SimX's sequential walk uses. - RTL: best_kv/best_ki/best_kg/best_kp, tri_tie and the separate t < best_t compare go; the PE verdict is the commit condition. - SimX: commit_takes goes; the flat walker also tests against best_t (an opaque hit already occludes every farther candidate, so nothing that could be staged is lost). Co-Authored-By: Claude Opus 5.5 --- hw/rtl/rtu/VX_rtu_scheduler.sv | 27 ++------- sim/simx/rtu/rtu_isect.cpp | 3 +- sim/simx/rtu/rtu_isect.h | 4 +- sim/simx/rtu/rtu_walker.cpp | 102 +++++++++++++-------------------- 4 files changed, 48 insertions(+), 88 deletions(-) diff --git a/hw/rtl/rtu/VX_rtu_scheduler.sv b/hw/rtl/rtu/VX_rtu_scheduler.sv index a48e27f53f..8477773ae5 100644 --- a/hw/rtl/rtu/VX_rtu_scheduler.sv +++ b/hw/rtl/rtu/VX_rtu_scheduler.sv @@ -197,12 +197,6 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( logic [1:0] setup_axis; logic [2:0][31:0] inv_d; logic [31:0] best_t; - // identity of an opaque hit committed in this walk: an exact t tie - // goes to the lower (instance, geometry, primitive) - logic best_kv; - logic [31:0] best_ki; - logic [27:0] best_kg; - logic [31:0] best_kp; logic [31:0] yld_t; // staged candidate's t (compare copy) logic [31:0] yld_ki; // staged candidate's key: instance id logic [31:0] yld_ko; // ... and record offset @@ -677,14 +671,6 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( && (inst_culldis || !(eff_back && cull_back)) && (inst_culldis || !(!eff_back && cull_front)) && !cls_cull; - wire tri_committable = tri_pass && (trit_q < word_q.best_t); - wire [31:0] tri_ki = word_q.in_blas ? word_q.inst_id : 32'd0; - wire [27:0] tri_kg = 28'(word_q.geom_r & `VX_RT_HIT_GEOMETRY_MASK); - wire [31:0] tri_kp = word_q.prim_base + word_q.tri_i; - wire tri_t_eq = (trit_q == word_q.best_t) - || ((trit_q[30:0] == 31'd0) && (word_q.best_t[30:0] == 31'd0)); - wire tri_tie = tri_pass && word_q.best_kv && tri_t_eq - && ({tri_ki, tri_kg, tri_kp} < {word_q.best_ki, word_q.best_kg, word_q.best_kp}); // the reported geometry word: the leaf's index plus the hit's facing wire [31:0] tri_geom = (word_q.geom_r & `VX_RT_HIT_GEOMETRY_MASK) | (eff_back ? `VX_RT_HIT_BACK_FACING : 32'd0); @@ -826,8 +812,9 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( .v1 (ltri_v1), .v2 (ltri_v2), .t_min (ray_q.t_min), - // the ray's own interval: an equal-t hit still reaches the tie-break - .t_max (ray_q.t_max), + // the interval shrinks to the committed hit: only a strictly nearer + // hit is reported, so on equal t the first one found stays + .t_max (word_q.best_t), .valid_out (tri_valid_out), .tag_out (tri_tag_out), .hit (tri_hit), @@ -1652,7 +1639,7 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( CS_TRI_WAIT: begin // woken by the tri PE result (held in its result RAM, so a retry // on a full commit queue re-reads the same result) - if ((tri_committable || tri_tie) && tri_opaque) begin + if (tri_pass && tri_opaque) begin cf_din_r.kind = CK_HIT; cf_din_r.t = trit_q; cf_din_r.u = triu_q; @@ -1667,10 +1654,6 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( cf_push_r = 1'b1; exec_hit_set = 1'b1; word_n.best_t = trit_q; - word_n.best_kv = 1'b1; - word_n.best_ki = tri_ki; - word_n.best_kg = tri_kg; - word_n.best_kp = tri_kp; // a closer opaque hit occludes a farther candidate if (yld_q[sel_q] && (word_x.yld_t >= trit_q)) begin exec_yld_clr = 1'b1; @@ -1688,7 +1671,7 @@ module VX_rtu_scheduler import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( wake_self = 1'b1; end end - end else if (tri_committable + end else if (tri_pass && above_floor(trit_q) && before_yld(trit_q)) begin cf_din_r.kind = CK_YLDA; diff --git a/sim/simx/rtu/rtu_isect.cpp b/sim/simx/rtu/rtu_isect.cpp index 0981d0d892..aa10099bbf 100644 --- a/sim/simx/rtu/rtu_isect.cpp +++ b/sim/simx/rtu/rtu_isect.cpp @@ -78,8 +78,7 @@ bool ray_triangle(const float ro[3], const float rd[3], const double tp2 = w[2] * pz[2]; const float t = float(((tp0 + tp1) + tp2) / det); // Open interval, as the reference commits (lvp_build_triangle_case: - // tmin < t and t < tmax). Callers pass the RAY's tmax, never the committed - // t, so an equal-t twin still reaches the walker's tie-break. + // tmin < t and t < tmax). if (!(tmin < t && t < tmax)) return false; const float det32 = float(det); diff --git a/sim/simx/rtu/rtu_isect.h b/sim/simx/rtu/rtu_isect.h index dadcbb59b7..893a5f5b19 100644 --- a/sim/simx/rtu/rtu_isect.h +++ b/sim/simx/rtu/rtu_isect.h @@ -36,8 +36,8 @@ namespace vortex { namespace rtu { // triangle's geometric normal (ray-flag face culling). Convention: // triangle front face is the side from which (v0, v1, v2) appear CCW. // Equivalently, det > 0 ↔ ray hits the front face. -// A hit needs tmin < t < tmax (open, as the Vulkan reference); tmax is -// the ray's, not the committed hit's. +// A hit needs tmin < t < tmax (open, as the Vulkan reference); the walker +// passes the committed hit's t as tmax. // ──────────────────────────────────────────────────────────────────── bool ray_triangle(const float ro[3], const float rd[3], const float v0[3], const float v1[3], const float v2[3], diff --git a/sim/simx/rtu/rtu_walker.cpp b/sim/simx/rtu/rtu_walker.cpp index 096136e676..7b4f03e8db 100644 --- a/sim/simx/rtu/rtu_walker.cpp +++ b/sim/simx/rtu/rtu_walker.cpp @@ -91,7 +91,6 @@ struct WalkCtx { uint32_t best_instance; uint32_t best_custom; // VK_INSTANCE_CUSTOM_INDEX of the committed instance uint32_t best_geom; // gl_GeometryIndexEXT of the committed leaf - bool best_kv; // an opaque hit committed in this walk bool any_hit; bool yield_pending; float yield_t, yield_u, yield_v; @@ -137,19 +136,6 @@ inline uint32_t hit_facing_bit(bool back_facing, uint32_t inst_flags) { return back_facing ? VX_RT_HIT_BACK_FACING : 0u; } -// Whether an opaque hit at t replaces the committed one: the nearer hit wins, -// and an exact tie goes to the lowest (instance, geometry, primitive). -bool commit_takes(const WalkCtx& ctx, float t, uint32_t instance_id, - uint32_t geom, uint32_t prim) { - if (t < ctx.best_t) return true; - if (!(ctx.best_kv && t == ctx.best_t)) return false; - const uint32_t best_geom = ctx.best_geom & VX_RT_HIT_GEOMETRY_MASK; - geom &= VX_RT_HIT_GEOMETRY_MASK; - if (instance_id != ctx.best_instance) return instance_id < ctx.best_instance; - if (geom != best_geom) return geom < best_geom; - return prim < ctx.best_prim; -} - // Depth-first walker for one BVH sub-tree under the supplied (object-space) // ray. Recurses on LeafInst so each instance's BLAS gets walked with its // transformed ray. ctx accumulates hits/yields across the whole call tree. @@ -188,10 +174,13 @@ void walk_bvh4_subtree(SceneView& sv, float t_hit = 0.f, u = 0.f, v = 0.f; bool back_facing = false; ++perf.bvh_tri_tests; - const bool tri_hit = ray_triangle(ro, rd, &tri[0], &tri[3], &tri[6], - ctx.tmin, ctx.tmax, - t_hit, u, v, back_facing); - if (!tri_hit) continue; + // The interval shrinks to the committed hit: a hit is only reported + // strictly nearer than it, so on equal t the first one found stays. + if (!ray_triangle(ro, rd, &tri[0], &tri[3], &tri[6], + ctx.tmin, ctx.best_t, + t_hit, u, v, back_facing)) { + continue; + } TriClassify cls = classify_tri_hit(ctx.ray_flags, tri_flags, inst_flags, back_facing); @@ -200,25 +189,21 @@ void walk_bvh4_subtree(SceneView& sv, if (cls.action == TriAction::Ignore) continue; if (cls.action == TriAction::Commit) { - if (commit_takes(ctx, t_hit, instance_id, leaf_geom, - leaf_prim_base + i)) { - ctx.best_t = t_hit; ctx.best_u = u; ctx.best_v = v; - ctx.best_prim = leaf_prim_base + i; - ctx.best_instance = instance_id; - ctx.best_custom = custom_id; - ctx.best_geom = hit_geom; - ctx.any_hit = true; - ctx.best_kv = true; - vcopy3(ctx.best_obj_o, ro); // object-space ray of this BLAS - vcopy3(ctx.best_obj_d, rd); - if (ctx.yield_pending && ctx.yield_t >= ctx.best_t) { - ctx.yield_pending = false; - ctx.yield_t = ctx.tmax; - } - if (cls.terminate_on_first_hit) { - ctx.terminated = true; - return; - } + ctx.best_t = t_hit; ctx.best_u = u; ctx.best_v = v; + ctx.best_prim = leaf_prim_base + i; + ctx.best_instance = instance_id; + ctx.best_custom = custom_id; + ctx.best_geom = hit_geom; + ctx.any_hit = true; + vcopy3(ctx.best_obj_o, ro); // object-space ray of this BLAS + vcopy3(ctx.best_obj_d, rd); + if (ctx.yield_pending && ctx.yield_t >= ctx.best_t) { + ctx.yield_pending = false; + ctx.yield_t = ctx.tmax; + } + if (cls.terminate_on_first_hit) { + ctx.terminated = true; + return; } } else { // TriAction::Yield uint64_t key = cand_key(instance_id, tris_off + i * kVxBvhTriStride); @@ -518,7 +503,6 @@ WalkCtx init_ctx(const RtuReq& req, uint32_t t, ctx.best_prim = 0; ctx.best_instance = 0; ctx.best_custom = 0; ctx.best_geom = 0; ctx.any_hit = false; - ctx.best_kv = false; ctx.yield_pending = false; ctx.yield_t = ctx.tmax; ctx.yield_u = 0.f; ctx.yield_v = 0.f; ctx.yield_prim = 0; ctx.yield_sbt = 0; @@ -662,11 +646,8 @@ WalkResult FlatWalker::walk_lane(const RtuReq& req, uint32_t t, SceneView& sv, float t_hit = 0.f, u = 0.f, v = 0.f; bool back_facing = false; ++perf.bvh_tri_tests; - // Test against ray.tmax (not best_t) so an opaque hit committed earlier in - // this walk doesn't pre-cull a non-opaque candidate that might survive an - // ACCEPT. if (!ray_triangle(ray_o, ray_d, &tri[0], &tri[3], &tri[6], - ctx.tmin, ctx.tmax, + ctx.tmin, ctx.best_t, t_hit, u, v, back_facing)) { continue; } @@ -677,26 +658,23 @@ WalkResult FlatWalker::walk_lane(const RtuReq& req, uint32_t t, SceneView& sv, if (cls.action == TriAction::Ignore) continue; if (cls.action == TriAction::Commit) { - if (commit_takes(ctx, t_hit, inst_idx, 0, i)) { - ctx.best_t = t_hit; ctx.best_u = u; ctx.best_v = v; - ctx.best_prim = i; - ctx.best_instance = inst_idx; - ctx.best_custom = cur_custom; - ctx.best_geom = hit_geom; - ctx.any_hit = true; - ctx.best_kv = true; - vcopy3(ctx.best_obj_o, ray_o); // this instance's object ray - vcopy3(ctx.best_obj_d, ray_d); - if (ctx.yield_pending && ctx.yield_t >= ctx.best_t) { - ctx.yield_pending = false; - ctx.yield_t = ctx.tmax; - } - if (cls.terminate_on_first_hit) { - // Halt the whole walk: this hit is committed as the result and no - // later triangle or instance may replace it. - ctx.terminated = true; - break; - } + ctx.best_t = t_hit; ctx.best_u = u; ctx.best_v = v; + ctx.best_prim = i; + ctx.best_instance = inst_idx; + ctx.best_custom = cur_custom; + ctx.best_geom = hit_geom; + ctx.any_hit = true; + vcopy3(ctx.best_obj_o, ray_o); // this instance's object ray + vcopy3(ctx.best_obj_d, ray_d); + if (ctx.yield_pending && ctx.yield_t >= ctx.best_t) { + ctx.yield_pending = false; + ctx.yield_t = ctx.tmax; + } + if (cls.terminate_on_first_hit) { + // Halt the whole walk: this hit is committed as the result and no + // later triangle or instance may replace it. + ctx.terminated = true; + break; } } else { // TriAction::Yield uint64_t key = cand_key(inst_idx, blas_tri_off + i * kPhase2TriStride); From d76bee1f9136411abc7ae0b775315994a4a2f5d0 Mon Sep 17 00:00:00 2001 From: Blaise Tine Date: Sat, 3 Oct 2026 13:11:49 -0700 Subject: [PATCH 22/31] rtu: fp32 watertight triangle test Replace the F64 edge-function/t/divide datapath with the standard F32 watertight test (Woop, Benthin, Wald, "Watertight Ray/Triangle Intersection", JCGT 2013). The F64 path existed only so t would round like the Vulkan reference's (lavapipe's) op order; an F32 test is what RT hardware builds. kz = argmax|dir|, kx/ky follow (swapped when dir[kz] < 0) sz = 1/dir[kz], sx = dir[kx]*sz, sy = dir[ky]*sz r = v - o, px = fma(-sx, rz, rx), py = fma(-sy, rz, ry), pz = sz*rz w0 = px2*py1 - py2*px1 (w1, w2 likewise), det = (w0 + w1) + w2 T = fma(w2, pz2, fma(w1, pz1, w0*pz0)), rcp = 1/det t = T*rcp, u = w1*rcp, v = w2*rcp, back_facing = det < 0 The edge functions stay two rounded products and a rounded difference of the per-vertex sheared coordinates, so an edge shared by two triangles yields exactly negated weights in both and no ray leaks between them (the sheared coordinates are per vertex, so fusing the shear is safe). Range check: the Vulkan spec defines the triangle interval as open at both ends -- "For any primitive that has within its bounds a position x_r = 0, y_r = 0, and (for triangles) t_min < -z_r/||d|| < t_max, or (otherwise) t_min <= -z_r/||d|| <= t_max, an intersection candidate exists" (Ray Intersection Candidate Determination) -- so a hit needs t_min < t < t_max, with t_max the committed hit (previous change). - VX_rtu_tri_pe: 9 subs, 2+6+3 shear FMA/MULs, 6 MULs + 3 SUBs for the edges, 2 adds + 3 FMA/MULs for det and T, 2 F32 dividers (1/dir[kz], 1/det), 3 MULs for t/u/v -- no F64 unit and no F64 divider. Latency 2 + 2*FDIV + 7*FMA (was 3 + FDIV + 3*FMA + 5*FMA64 + FDIV64), every stage registered. - RTU_LATENCY_FMA64 / RTU_FDIV64_LAT / kRtuLatencyFma64 / kRtuFdiv64Lat go; TriPe::pipe_depth follows the new pipeline. - hw/unittest/rtu_tri_pe: 1M random + 109K directed cases, 0 mismatches vs SimX (46 subnormal-flush cases, the PEs being FTZ/DAZ); new shared-edge family: rays through the shared diagonal of 20K quads, none passes between the two triangles. Co-Authored-By: Claude Opus 5.5 --- docs/designs/ray_tracing_architecture.md | 10 +- hw/rtl/rtu/VX_rtu_pkg.sv | 6 - hw/rtl/rtu/VX_rtu_tri_pe.sv | 576 ++++++++--------------- hw/unittest/rtu_tri_pe/main.cpp | 65 ++- sim/simx/rtu/rtu_isect.cpp | 66 ++- sim/simx/rtu/rtu_isect.h | 6 +- sim/simx/rtu/rtu_types.h | 2 - 7 files changed, 289 insertions(+), 442 deletions(-) diff --git a/docs/designs/ray_tracing_architecture.md b/docs/designs/ray_tracing_architecture.md index 45933526af..cfc65911e6 100644 --- a/docs/designs/ray_tracing_architecture.md +++ b/docs/designs/ray_tracing_architecture.md @@ -330,9 +330,13 @@ whatever depth results: emitting `{hit, t_near}`. Mirrors SimX `ray_aabb_intersect` bit for bit: corners `origin + q·2^exp`, slabs `(corner − ro)·inv_d`, culled against `[0, t_max]`. Also handles raw/procedural boxes. -- **`VX_rtu_tri_pe`** — pipelined watertight triangle test (F32 shear, F64 edge - functions and t, in the Vulkan reference's op order), one triangle per cycle, - emitting `{hit, t, u, v, back_facing}`; bit-exact against SimX `ray_triangle`. +- **`VX_rtu_tri_pe`** — pipelined watertight triangle test (Woop, Benthin, Wald, + JCGT 2013), all F32: shear, edge functions as rounded products and a rounded + difference (so a shared edge evaluates to exactly negated weights in its two + triangles), `det = (w0 + w1) + w2`, `T` as an FMA chain, one `1/det` scaling + `t`, `u`, `v`. Accepts `t_min < t < t_max` (Vulkan's open triangle interval), + with `t_max` the committed hit, one triangle per cycle, emitting + `{hit, t, u, v, back_facing}`; bit-exact against SimX `ray_triangle`. - **`VX_rtu_xform`** — TLAS world→object transform. The instance record holds the world→object matrix and the ray is transformed in the reference's op order (rounded products, then `t + x + y + z`), so no inverse is taken. Always built: diff --git a/hw/rtl/rtu/VX_rtu_pkg.sv b/hw/rtl/rtu/VX_rtu_pkg.sv index e8481b8d03..4a061f53b5 100644 --- a/hw/rtl/rtu/VX_rtu_pkg.sv +++ b/hw/rtl/rtu/VX_rtu_pkg.sv @@ -98,12 +98,6 @@ package VX_rtu_pkg; // 15). PE delay lines + the scheduler setup window scale with this. localparam RTU_FDIV_LAT = `VX_CFG_RTU_FDIV_LATENCY; - // The tri PE's edge functions and t run in F64 on the soft core, whose - // 53-bit multiply wants a deeper pipe than F32; the F64 divider's depth is - // fixed by its radix-2 recurrence. - localparam RTU_LATENCY_FMA64 = (RTU_LATENCY_FMA > 12) ? RTU_LATENCY_FMA : 12; - localparam RTU_FDIV64_LAT = 32; - // ───────────────────────────────────────────────────────────────── // CW-BVH node-kind tag (low byte of word0) and count field (bits 8..15) // ───────────────────────────────────────────────────────────────── diff --git a/hw/rtl/rtu/VX_rtu_tri_pe.sv b/hw/rtl/rtu/VX_rtu_tri_pe.sv index 8b96f1ae97..2bacf98c10 100644 --- a/hw/rtl/rtu/VX_rtu_tri_pe.sv +++ b/hw/rtl/rtu/VX_rtu_tri_pe.sv @@ -12,22 +12,24 @@ // limitations under the License. // VX_rtu_tri_pe — pipelined watertight ray-triangle intersector (Woop, Benthin, -// Wald, JCGT 2013). Streams one triangle per cycle and emits {hit, t, u, v, -// back_facing} after a fixed latency. Mirrors SimX rtu::ray_triangle op for op: +// Wald, "Watertight Ray/Triangle Intersection", JCGT 2013), all F32. Streams one +// triangle per cycle and emits {hit, t, u, v, back_facing} after a fixed +// latency. Mirrors SimX rtu::ray_triangle op for op: // // kz = argmax|dir|, kx/ky follow (swapped when dir[kz] < 0) -// F32: sz = 1/dir[kz], sx = dir[kx]*sz, sy = dir[ky]*sz -// r = vertex - origin, px = rx - sx*rz, py = ry - sy*rz -// F64: pz = sz*rz (exact), w_i = px_a*py_b - py_a*px_b (one rounding) -// det = w0 + (w1 + w2), T = (w0*pz0 + w1*pz1) + w2*pz2 -// t = f32(T / det), (u, v) = f32(w1, w2) / f32(det) -// hit = !(any w < 0 && any w > 0) && det != 0 && tmin < t < tmax +// sz = 1/dir[kz], sx = dir[kx]*sz, sy = dir[ky]*sz +// r = vertex - origin, px = fma(-sx, rz, rx), py = fma(-sy, rz, ry), pz = sz*rz +// w0 = px2*py1 - py2*px1, w1 = px0*py2 - py0*px2, w2 = px1*py0 - py1*px0 +// det = (w0 + w1) + w2, T = fma(w2, pz2, fma(w1, pz1, w0*pz0)) +// rcp = 1/det, t = T*rcp, (u, v) = (w1*rcp, w2*rcp) +// hit = !(any w < 0 && any w > 0) && det != 0 && t_min < t < t_max // back_facing = det < 0 // -// A shared edge evaluates to exactly negated weights in its two triangles, so -// the test is watertight; the F64 edge functions and t keep t within half an -// ulp of the exact intersection, in the op order the Vulkan reference -// (lavapipe) uses, so coincident triangles resolve the same way. +// Each edge function is two rounded products and a rounded difference of the +// sheared vertices alone, so an edge shared by two triangles evaluates to +// exactly negated weights in both: no ray slips between them. t_min < t < t_max +// is the Vulkan ray interval for triangles (open at both ends). The FP units +// flush subnormals. `include "VX_define.vh" @@ -58,123 +60,33 @@ module VX_rtu_tri_pe import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( output wire [31:0] v, output wire back_facing ); - localparam F = LATENCY_FMA; - localparam V = LATENCY_FDIV; - localparam D = RTU_LATENCY_FMA64; - localparam V64 = RTU_FDIV64_LAT; + localparam F = LATENCY_FMA; + localparam V = LATENCY_FDIV; // stage start times (cycles after valid_in) - localparam T_B = 1; // canonical/axis select registered + localparam T_B = 1; // axis select registered localparam T_C = T_B + V; // sz ready localparam T_D = T_C + F; // sx, sy ready - localparam T_E = T_D + F; // sx*rz, sy*rz ready - localparam T_F = T_E + F; // px, py ready - localparam T_G = T_F + 2 * D; // w ready - localparam T_H = T_G + 3 * D; // T ready (det at T_G + 2D) - localparam T_I = T_H + V64 + 1; // t narrowed and registered - localparam LATENCY = T_I + 1; // verdict registered + localparam T_E = T_D + F; // px, py, pz ready + localparam T_F = T_E + 2 * F; // w ready + localparam T_G = T_F + 2 * F + V; // 1/det ready (T at T_F + 3F) + localparam T_H = T_G + F; // t, u, v ready + localparam LATENCY = T_H + 1; // verdict registered - `STATIC_ASSERT(V >= F, ("tri PE: FDIV latency must cover the r subtract")) - `STATIC_ASSERT(T_G + 2 * D + 1 + V <= T_I, ("tri PE: bary divide must land before t")) + `STATIC_ASSERT(V >= F, ("tri PE: FDIV latency must cover the r subtract and T")) localparam [INST_FMT_BITS-1:0] FMT_ADD = 2'b00; localparam [INST_FMT_BITS-1:0] FMT_SUB = 2'b10; localparam [31:0] F32_ONE = 32'h3F800000; - // ── helpers ─────────────────────────────────────────────────────── - // exact F32 -> F64 widening (subnormals normalized) - function automatic [63:0] f32_to_f64(input [31:0] a); - reg [7:0] e; - reg [22:0] m; - reg [4:0] lz; - reg [22:0] mn; - begin - e = a[30:23]; - m = a[22:0]; - if (e == 8'hff) begin - f32_to_f64 = {a[31], 11'h7ff, m, 29'd0}; - end else if (e == 8'd0) begin - if (m == 23'd0) begin - f32_to_f64 = {a[31], 63'd0}; - end else begin - lz = 5'd0; - for (integer i = 22; i >= 0; --i) begin - if (m[i]) begin - lz = 5'(22 - i); - break; - end - end - mn = m << (lz + 5'd1); - f32_to_f64 = {a[31], 11'(11'd896 - 11'(lz)), mn, 29'd0}; - end - end else begin - f32_to_f64 = {a[31], 11'(e) + 11'd896, m, 29'd0}; - end - end - endfunction - - // F64 -> F32, round to nearest even - function automatic [31:0] f64_to_f32(input [63:0] a); - reg s; - reg [10:0] e; - reg [51:0] m; - reg signed [12:0] ue; - reg [52:0] sig; - reg [6:0] sh; - reg [22:0] keep; - reg guard, sticky; - reg [31:0] base; - begin - s = a[63]; - e = a[62:52]; - m = a[51:0]; - ue = 13'(e) - 13'sd896; - if (e == 11'h7ff) begin - f64_to_f32 = {s, 8'hff, (m != 52'd0) ? {1'b1, m[50:29]} : 23'd0}; - end else if (e == 11'd0) begin - f64_to_f32 = {s, 31'd0}; - end else if (ue >= 13'sd255) begin - f64_to_f32 = {s, 8'hff, 23'd0}; - end else if (ue >= 13'sd1) begin - guard = m[28]; - sticky = (m[27:0] != 28'd0); - base = {s, ue[7:0], m[51:29]}; - f64_to_f32 = base + 32'((guard && (sticky || m[29])) ? 1 : 0); - end else begin - // subnormal: mantissa = sig >> (30 - ue), ue <= 0 - sig = {1'b1, m}; - sh = (ue < -13'sd30) ? 7'd61 : 7'(13'sd30 - ue); - if (sh > 7'd54) begin - keep = 23'd0; - guard = 1'b0; - sticky = 1'b1; - end else begin - keep = 23'(sig >> sh); - guard = sig[6'(sh - 7'd1)]; - sticky = (64'(sig) & ((64'd1 << (sh - 7'd1)) - 64'd1)) != 64'd0; - end - base = {s, 8'd0, keep}; - f64_to_f32 = base + 32'((guard && (sticky || keep[0])) ? 1 : 0); - end - end - endfunction - // IEEE ordering on F32 (+0 == -0); NaN compares false - function automatic f32_le(input [31:0] a, input [31:0] b); - reg a_nan, b_nan; - reg [31:0] ka, kb; - begin - a_nan = (a[30:23] == 8'hff) && (a[22:0] != 23'd0); - b_nan = (b[30:23] == 8'hff) && (b[22:0] != 23'd0); - ka = (a[30:0] == 31'd0) ? 32'h80000000 : (a[31] ? ~a : {1'b1, a[30:0]}); - kb = (b[30:0] == 31'd0) ? 32'h80000000 : (b[31] ? ~b : {1'b1, b[30:0]}); - f32_le = !a_nan && !b_nan && (ka <= kb); - end - endfunction - - // strict IEEE a < b (+0 == -0); NaN compares false - function automatic f32_lt(input [31:0] a, input [31:0] b); - f32_lt = f32_le(a, b) && !f32_le(b, a); + function automatic logic f32_lt(input logic [31:0] a, input logic [31:0] b); + logic [31:0] ka, kb; + ka = (a[30:0] == 31'd0) ? 32'h80000000 : (a[31] ? ~a : {1'b1, a[30:0]}); + kb = (b[30:0] == 31'd0) ? 32'h80000000 : (b[31] ? ~b : {1'b1, b[30:0]}); + f32_lt = !((a[30:23] == 8'hff) && (a[22:0] != 23'd0)) + && !((b[30:23] == 8'hff) && (b[22:0] != 23'd0)) + && (ka < kb); endfunction // ── stage A (@0 -> @T_B): axis select ───────────────────────────── @@ -189,7 +101,7 @@ module VX_rtu_tri_pe import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( wire [1:0] kx_w = dz_neg ? ky0 : kx0; wire [1:0] ky_w = dz_neg ? kx0 : ky0; - // per canonical vertex: (x, y, z) components in the sheared frame's axes + // per vertex: (x, y, z) components in the sheared frame's axes wire [2:0][2:0][31:0] q_w; // [vertex][axis x/y/z] wire [2:0][2:0][31:0] cvs = {v2, v1, v0}; for (genvar i = 0; i < 3; ++i) begin : g_q @@ -238,7 +150,7 @@ module VX_rtu_tri_pe import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( for (genvar i = 0; i < 3; ++i) begin : g_r for (genvar a = 0; a < 3; ++a) begin : g_ax VX_fma_unit #( - .LATENCY (F), + .LATENCY (F), .USE_DSP (`VX_CFG_RTU_USE_DSP), .SUBNORM_ENABLE (0), .EXCEPT_ENABLE (1) @@ -259,7 +171,7 @@ module VX_rtu_tri_pe import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( end end - // r from @T_B+F to @T_D (consumed by the sx*rz stage) + // r from @T_B+F to @T_D (consumed by the shear stage) wire [2:0][2:0][31:0] r_d; VX_shift_register #( .DATAW (9 * 32), @@ -288,10 +200,10 @@ module VX_rtu_tri_pe import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( wire [1:0][31:0] sxy_d; for (genvar a = 0; a < 2; ++a) begin : g_sxy VX_fma_unit #( - .LATENCY (F), - .USE_DSP (`VX_CFG_RTU_USE_DSP), - .SUBNORM_ENABLE (0), - .EXCEPT_ENABLE (1) + .LATENCY (F), + .USE_DSP (`VX_CFG_RTU_USE_DSP), + .SUBNORM_ENABLE (0), + .EXCEPT_ENABLE (1) ) fmul_s ( .clk (clk), .reset (reset), @@ -320,35 +232,33 @@ module VX_rtu_tri_pe import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( .data_out (sz_d) ); - // ── stage D (@T_D): sx*rz, sy*rz (F32); pz = sz*rz (F64, exact) ──── - wire [2:0][1:0][31:0] m_e; - wire [2:0][63:0] pz_x; // @T_D + D + // ── stage D (@T_D): px/py = fma(-s, rz, r), pz = sz*rz ──────────── + wire [2:0][1:0][31:0] p_e; // [vertex][x/y] + wire [2:0][31:0] pz_e; for (genvar i = 0; i < 3; ++i) begin : g_shear for (genvar a = 0; a < 2; ++a) begin : g_ax VX_fma_unit #( - .LATENCY (F), + .LATENCY (F), .USE_DSP (`VX_CFG_RTU_USE_DSP), .SUBNORM_ENABLE (0), .EXCEPT_ENABLE (1) - ) fmul_m ( + ) fma_p ( .clk (clk), .reset (reset), .enable (enable), .mask (1'b1), - .op_type (INST_FPU_MUL), + .op_type (INST_FPU_MADD), .fmt (FMT_ADD), .frm (INST_FRM_RNE), - .dataa (sxy_d[a]), + .dataa ({~sxy_d[a][31], sxy_d[a][30:0]}), .datab (r_d[i][2]), - .datac ('0), - .result (m_e[i][a]), + .datac (r_d[i][a]), + .result (p_e[i][a]), `UNUSED_PIN (fflags) ); end VX_fma_unit #( - .LATENCY (D), - .MAN_BITS (52), - .EXP_BITS (11), + .LATENCY (F), .USE_DSP (`VX_CFG_RTU_USE_DSP), .SUBNORM_ENABLE (0), .EXCEPT_ENABLE (1) @@ -360,381 +270,271 @@ module VX_rtu_tri_pe import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( .op_type (INST_FPU_MUL), .fmt (FMT_ADD), .frm (INST_FRM_RNE), - .dataa (f32_to_f64(sz_d)), - .datab (f32_to_f64(r_d[i][2])), + .dataa (sz_d), + .datab (r_d[i][2]), .datac ('0), - .result (pz_x[i]), + .result (pz_e[i]), `UNUSED_PIN (fflags) ); end - // rx, ry from @T_D to @T_E - wire [2:0][1:0][31:0] rxy_e; - VX_shift_register #( - .DATAW (6 * 32), - .DEPTH (T_E - T_D) - ) sr_rxy ( - .clk (clk), - .reset (reset), - .enable (enable), - .data_in ({r_d[2][1], r_d[2][0], r_d[1][1], r_d[1][0], r_d[0][1], r_d[0][0]}), - .data_out (rxy_e) - ); - - // ── stage E (@T_E): px = rx - sx*rz, py = ry - sy*rz ────────────── - wire [2:0][1:0][31:0] p_f; // [vertex][x/y] - for (genvar i = 0; i < 3; ++i) begin : g_p - for (genvar a = 0; a < 2; ++a) begin : g_ax + // ── stage E (@T_E): w_i = px_a*py_b - py_a*px_b ─────────────────── + // (a, b) per weight: w0 <- (2, 1), w1 <- (0, 2), w2 <- (1, 0) + wire [2:0][1:0][31:0] cp_e; // [weight][px_a*py_b, py_a*px_b] + wire [2:0][31:0] w_f; + for (genvar i = 0; i < 3; ++i) begin : g_w + localparam IA = (i == 0) ? 2 : ((i == 1) ? 0 : 1); + localparam IB = (i == 0) ? 1 : ((i == 1) ? 2 : 0); + for (genvar k = 0; k < 2; ++k) begin : g_prod VX_fma_unit #( - .LATENCY (F), + .LATENCY (F), .USE_DSP (`VX_CFG_RTU_USE_DSP), .SUBNORM_ENABLE (0), .EXCEPT_ENABLE (1) - ) fsub_p ( + ) fmul_c ( .clk (clk), .reset (reset), .enable (enable), .mask (1'b1), - .op_type (INST_FPU_ADD), - .fmt (FMT_SUB), + .op_type (INST_FPU_MUL), + .fmt (FMT_ADD), .frm (INST_FRM_RNE), - .dataa (rxy_e[i][a]), - .datab (m_e[i][a]), + .dataa (p_e[IA][k]), + .datab (p_e[IB][1-k]), .datac ('0), - .result (p_f[i][a]), + .result (cp_e[i][k]), `UNUSED_PIN (fflags) ); end - end - - // ── stage F (@T_F): w_i = px_a*py_b - py_a*px_b in F64 ──────────── - // (a, b) per weight: w0 <- (2, 1), w1 <- (0, 2), w2 <- (1, 0) - wire [2:0][1:0][63:0] p64_f; - for (genvar i = 0; i < 3; ++i) begin : g_p64 - assign p64_f[i][0] = f32_to_f64(p_f[i][0]); - assign p64_f[i][1] = f32_to_f64(p_f[i][1]); - end - - wire [2:0][1:0][63:0] p64_g1; // operands delayed D for the fused stage - VX_shift_register #( - .DATAW (6 * 64), - .DEPTH (D) - ) sr_p64 ( - .clk (clk), - .reset (reset), - .enable (enable), - .data_in (p64_f), - .data_out (p64_g1) - ); - - wire [2:0][63:0] w_g; - for (genvar i = 0; i < 3; ++i) begin : g_w - localparam IA = (i == 0) ? 2 : ((i == 1) ? 0 : 1); - localparam IB = (i == 0) ? 1 : ((i == 1) ? 2 : 0); - wire [63:0] cross_q; // py_a * px_b, exact VX_fma_unit #( - .LATENCY (D), - .MAN_BITS (52), - .EXP_BITS (11), + .LATENCY (F), .USE_DSP (`VX_CFG_RTU_USE_DSP), .SUBNORM_ENABLE (0), .EXCEPT_ENABLE (1) - ) fmul_c ( + ) fsub_w ( .clk (clk), .reset (reset), .enable (enable), .mask (1'b1), - .op_type (INST_FPU_MUL), - .fmt (FMT_ADD), - .frm (INST_FRM_RNE), - .dataa (p64_f[IA][1]), - .datab (p64_f[IB][0]), - .datac ('0), - .result (cross_q), - `UNUSED_PIN (fflags) - ); - VX_fma_unit #( - .LATENCY (D), - .MAN_BITS (52), - .EXP_BITS (11), - .USE_DSP (`VX_CFG_RTU_USE_DSP), - .SUBNORM_ENABLE (0), - .EXCEPT_ENABLE (1) - ) fmsub_w ( - .clk (clk), - .reset (reset), - .enable (enable), - .mask (1'b1), - .op_type (INST_FPU_MADD), + .op_type (INST_FPU_ADD), .fmt (FMT_SUB), .frm (INST_FRM_RNE), - .dataa (p64_g1[IA][0]), - .datab (p64_g1[IB][1]), - .datac (cross_q), - .result (w_g[i]), + .dataa (cp_e[i][0]), + .datab (cp_e[i][1]), + .datac ('0), + .result (w_f[i]), `UNUSED_PIN (fflags) ); end - // pz from @T_D+D to @T_G - wire [2:0][63:0] pz_g; + // pz from @T_E to @T_F + wire [2:0][31:0] pz_f; VX_shift_register #( - .DATAW (3 * 64), - .DEPTH (T_G - (T_D + D)) + .DATAW (3 * 32), + .DEPTH (T_F - T_E) ) sr_pz ( .clk (clk), .reset (reset), .enable (enable), - .data_in (pz_x), - .data_out (pz_g) + .data_in (pz_e), + .data_out (pz_f) ); - // ── stage G (@T_G): det = w0 + (w1 + w2); T = (w0 pz0 + w1 pz1) + w2 pz2 - wire [63:0] det12, det_g2, tp0, tp1, tp2, t01, t_num; - wire [63:0] w0_g1, tp2_g2; + // ── stage F (@T_F): det = (w0 + w1) + w2; T = fma chain over w*pz ── + // w, pz delayed one and two FMA stages for the later chain links + wire [2:0][31:0] w_f1, pz_f1; + wire [31:0] w1_f2, w2_f2, pz2_f2; VX_shift_register #( - .DATAW (64), - .DEPTH (D) - ) sr_w0 ( + .DATAW (6 * 32), + .DEPTH (F) + ) sr_wpz1 ( .clk (clk), .reset (reset), .enable (enable), - .data_in (w_g[0]), - .data_out (w0_g1) + .data_in ({w_f, pz_f}), + .data_out ({w_f1, pz_f1}) ); - VX_fma_unit #(.LATENCY (D), .MAN_BITS (52), .EXP_BITS (11), .USE_DSP (`VX_CFG_RTU_USE_DSP), .SUBNORM_ENABLE (0), .EXCEPT_ENABLE (1)) fadd_det12 ( + VX_shift_register #( + .DATAW (3 * 32), + .DEPTH (F) + ) sr_wpz2 ( + .clk (clk), + .reset (reset), + .enable (enable), + .data_in ({w_f1[2], pz_f1[2], w_f1[1]}), + .data_out ({w2_f2, pz2_f2, w1_f2}) + ); + + wire [31:0] det01, det_g, tp0, tp01, t_num; + VX_fma_unit #(.LATENCY (F), .USE_DSP (`VX_CFG_RTU_USE_DSP), .SUBNORM_ENABLE (0), .EXCEPT_ENABLE (1)) fadd_det01 ( .clk (clk), .reset (reset), .enable (enable), .mask (1'b1), .op_type (INST_FPU_ADD), .fmt (FMT_ADD), .frm (INST_FRM_RNE), - .dataa (w_g[1]), .datab (w_g[2]), .datac ('0), - .result (det12), `UNUSED_PIN (fflags) + .dataa (w_f[0]), .datab (w_f[1]), .datac ('0), + .result (det01), `UNUSED_PIN (fflags) ); - VX_fma_unit #(.LATENCY (D), .MAN_BITS (52), .EXP_BITS (11), .USE_DSP (`VX_CFG_RTU_USE_DSP), .SUBNORM_ENABLE (0), .EXCEPT_ENABLE (1)) fadd_det ( + VX_fma_unit #(.LATENCY (F), .USE_DSP (`VX_CFG_RTU_USE_DSP), .SUBNORM_ENABLE (0), .EXCEPT_ENABLE (1)) fadd_det ( .clk (clk), .reset (reset), .enable (enable), .mask (1'b1), .op_type (INST_FPU_ADD), .fmt (FMT_ADD), .frm (INST_FRM_RNE), - .dataa (w0_g1), .datab (det12), .datac ('0), - .result (det_g2), `UNUSED_PIN (fflags) + .dataa (det01), .datab (w_f1[2]), .datac ('0), + .result (det_g), `UNUSED_PIN (fflags) ); - VX_fma_unit #(.LATENCY (D), .MAN_BITS (52), .EXP_BITS (11), .USE_DSP (`VX_CFG_RTU_USE_DSP), .SUBNORM_ENABLE (0), .EXCEPT_ENABLE (1)) fmul_tp0 ( + VX_fma_unit #(.LATENCY (F), .USE_DSP (`VX_CFG_RTU_USE_DSP), .SUBNORM_ENABLE (0), .EXCEPT_ENABLE (1)) fmul_tp0 ( .clk (clk), .reset (reset), .enable (enable), .mask (1'b1), .op_type (INST_FPU_MUL), .fmt (FMT_ADD), .frm (INST_FRM_RNE), - .dataa (w_g[0]), .datab (pz_g[0]), .datac ('0), + .dataa (w_f[0]), .datab (pz_f[0]), .datac ('0), .result (tp0), `UNUSED_PIN (fflags) ); - VX_fma_unit #(.LATENCY (D), .MAN_BITS (52), .EXP_BITS (11), .USE_DSP (`VX_CFG_RTU_USE_DSP), .SUBNORM_ENABLE (0), .EXCEPT_ENABLE (1)) fmul_tp1 ( + VX_fma_unit #(.LATENCY (F), .USE_DSP (`VX_CFG_RTU_USE_DSP), .SUBNORM_ENABLE (0), .EXCEPT_ENABLE (1)) fma_tp01 ( .clk (clk), .reset (reset), .enable (enable), .mask (1'b1), - .op_type (INST_FPU_MUL), .fmt (FMT_ADD), .frm (INST_FRM_RNE), - .dataa (w_g[1]), .datab (pz_g[1]), .datac ('0), - .result (tp1), `UNUSED_PIN (fflags) + .op_type (INST_FPU_MADD), .fmt (FMT_ADD), .frm (INST_FRM_RNE), + .dataa (w_f1[1]), .datab (pz_f1[1]), .datac (tp0), + .result (tp01), `UNUSED_PIN (fflags) ); - VX_fma_unit #(.LATENCY (D), .MAN_BITS (52), .EXP_BITS (11), .USE_DSP (`VX_CFG_RTU_USE_DSP), .SUBNORM_ENABLE (0), .EXCEPT_ENABLE (1)) fmul_tp2 ( + VX_fma_unit #(.LATENCY (F), .USE_DSP (`VX_CFG_RTU_USE_DSP), .SUBNORM_ENABLE (0), .EXCEPT_ENABLE (1)) fma_t ( .clk (clk), .reset (reset), .enable (enable), .mask (1'b1), - .op_type (INST_FPU_MUL), .fmt (FMT_ADD), .frm (INST_FRM_RNE), - .dataa (w_g[2]), .datab (pz_g[2]), .datac ('0), - .result (tp2), `UNUSED_PIN (fflags) - ); - VX_fma_unit #(.LATENCY (D), .MAN_BITS (52), .EXP_BITS (11), .USE_DSP (`VX_CFG_RTU_USE_DSP), .SUBNORM_ENABLE (0), .EXCEPT_ENABLE (1)) fadd_t01 ( - .clk (clk), .reset (reset), .enable (enable), .mask (1'b1), - .op_type (INST_FPU_ADD), .fmt (FMT_ADD), .frm (INST_FRM_RNE), - .dataa (tp0), .datab (tp1), .datac ('0), - .result (t01), `UNUSED_PIN (fflags) - ); - VX_shift_register #( - .DATAW (64), - .DEPTH (D) - ) sr_tp2 ( - .clk (clk), - .reset (reset), - .enable (enable), - .data_in (tp2), - .data_out (tp2_g2) - ); - VX_fma_unit #(.LATENCY (D), .MAN_BITS (52), .EXP_BITS (11), .USE_DSP (`VX_CFG_RTU_USE_DSP), .SUBNORM_ENABLE (0), .EXCEPT_ENABLE (1)) fadd_t ( - .clk (clk), .reset (reset), .enable (enable), .mask (1'b1), - .op_type (INST_FPU_ADD), .fmt (FMT_ADD), .frm (INST_FRM_RNE), - .dataa (t01), .datab (tp2_g2), .datac ('0), + .op_type (INST_FPU_MADD), .fmt (FMT_ADD), .frm (INST_FRM_RNE), + .dataa (w2_f2), .datab (pz2_f2), .datac (tp01), .result (t_num), `UNUSED_PIN (fflags) ); - // det from @T_G+2D to @T_H - wire [63:0] det_h; - VX_shift_register #( - .DATAW (64), - .DEPTH (T_H - (T_G + 2 * D)) - ) sr_det ( - .clk (clk), - .reset (reset), - .enable (enable), - .data_in (det_g2), - .data_out (det_h) - ); - - // ── stage H (@T_H): t = f32(T / det) ────────────────────────────── - wire [63:0] t64; + // ── 1/det (@T_F+2F -> @T_G) ─────────────────────────────────────── + wire [31:0] rcp_g; VX_fdiv_unit #( - .LATENCY (V64), - .FLEN (64), + .LATENCY (V), + .FLEN (32), + .USE_DSP (`VX_CFG_RTU_USE_DSP), .SUBNORM_ENABLE (0), .EXCEPT_ENABLE (1) - ) fdiv_t ( + ) fdiv_rcp ( .clk (clk), .reset (reset), .enable (enable), .mask (1'b1), - .fmt (2'b01), + .fmt ('0), .frm (INST_FRM_RNE), - .dataa (t_num), - .datab (det_h), - .result (t64), + .dataa (F32_ONE), + .datab (det_g), + .result (rcp_g), `UNUSED_PIN (fflags) ); - reg [31:0] t_i; - always_ff @(posedge clk) begin - if (enable) begin - t_i <= f64_to_f32(t64); - end - end - - // ── barycentrics (@T_G+2D): f32(w1) / f32(det), f32(w2) / f32(det) - wire [1:0][63:0] w_b; // w1, w2 + // T from @T_F+3F, w1/w2 from @T_F+2F, to @T_G + wire [31:0] t_num_g, w1_g, w2_g; VX_shift_register #( - .DATAW (2 * 64), - .DEPTH (2 * D) - ) sr_wb ( + .DATAW (32), + .DEPTH (T_G - (T_F + 3 * F)) + ) sr_tnum ( .clk (clk), .reset (reset), .enable (enable), - .data_in ({w_g[2], w_g[1]}), - .data_out (w_b) + .data_in (t_num), + .data_out (t_num_g) ); - reg [31:0] wu_r, wv_r, det32_r; - always_ff @(posedge clk) begin - if (enable) begin - wu_r <= f64_to_f32(w_b[0]); - wv_r <= f64_to_f32(w_b[1]); - det32_r <= f64_to_f32(det_g2); - end - end - - wire [31:0] u_q, v_q; - VX_fdiv_unit #( - .LATENCY (V), - .FLEN (32), - .USE_DSP (`VX_CFG_RTU_USE_DSP), - .SUBNORM_ENABLE (0), - .EXCEPT_ENABLE (1) - ) fdiv_u ( - .clk (clk), - .reset (reset), - .enable (enable), - .mask (1'b1), - .fmt ('0), - .frm (INST_FRM_RNE), - .dataa (wu_r), - .datab (det32_r), - .result (u_q), - `UNUSED_PIN (fflags) - ); - VX_fdiv_unit #( - .LATENCY (V), - .FLEN (32), - .USE_DSP (`VX_CFG_RTU_USE_DSP), - .SUBNORM_ENABLE (0), - .EXCEPT_ENABLE (1) - ) fdiv_v ( - .clk (clk), - .reset (reset), - .enable (enable), - .mask (1'b1), - .fmt ('0), - .frm (INST_FRM_RNE), - .dataa (wv_r), - .datab (det32_r), - .result (v_q), - `UNUSED_PIN (fflags) - ); - - wire [31:0] u_i, v_i; VX_shift_register #( - .DATAW (64), - .DEPTH (T_I - (T_G + 2 * D + 1 + V)) - ) sr_uv ( + .DATAW (2 * 32), + .DEPTH (V) + ) sr_wuv ( .clk (clk), .reset (reset), .enable (enable), - .data_in ({u_q, v_q}), - .data_out ({u_i, v_i}) + .data_in ({w2_f2, w1_f2}), + .data_out ({w2_g, w1_g}) ); + `UNUSED_VAR ({w_f1[0], pz_f1[0]}) + + // ── stage G (@T_G): t = T*rcp, u = w1*rcp, v = w2*rcp ───────────── + wire [2:0][31:0] tuv_h; + wire [2:0][31:0] tuv_num = {w2_g, w1_g, t_num_g}; + for (genvar k = 0; k < 3; ++k) begin : g_scale + VX_fma_unit #( + .LATENCY (F), + .USE_DSP (`VX_CFG_RTU_USE_DSP), + .SUBNORM_ENABLE (0), + .EXCEPT_ENABLE (1) + ) fmul_tuv ( + .clk (clk), + .reset (reset), + .enable (enable), + .mask (1'b1), + .op_type (INST_FPU_MUL), + .fmt (FMT_ADD), + .frm (INST_FRM_RNE), + .dataa (tuv_num[k]), + .datab (rcp_g), + .datac ('0), + .result (tuv_h[k]), + `UNUSED_PIN (fflags) + ); + end - // ── verdict flags: edge signs (@T_G), det tests (@T_G+2D) ───────── - reg edge_ok_g; + // ── verdict flags: edge signs (@T_F), det tests (@T_F+2F) ───────── + reg edge_ok_f; always @(*) begin - reg any_neg, any_pos; + logic any_neg, any_pos; any_neg = 1'b0; any_pos = 1'b0; for (integer i = 0; i < 3; ++i) begin - if (w_g[i][62:0] != 63'd0 - && !((w_g[i][62:52] == 11'h7ff) && (w_g[i][51:0] != 52'd0))) begin - any_neg = any_neg | w_g[i][63]; - any_pos = any_pos | ~w_g[i][63]; + if (w_f[i][30:0] != 31'd0 + && !((w_f[i][30:23] == 8'hff) && (w_f[i][22:0] != 23'd0))) begin + any_neg = any_neg | w_f[i][31]; + any_pos = any_pos | ~w_f[i][31]; end end - edge_ok_g = !(any_neg && any_pos); + edge_ok_f = !(any_neg && any_pos); end wire edge_ok_d; VX_shift_register #( .DATAW (1), - .DEPTH (2 * D) + .DEPTH (2 * F) ) sr_edge ( .clk (clk), .reset (reset), .enable (enable), - .data_in (edge_ok_g), + .data_in (edge_ok_f), .data_out (edge_ok_d) ); - wire det_nan = (det_g2[62:52] == 11'h7ff) && (det_g2[51:0] != 52'd0); - wire det_ok_d = (det_g2[62:0] != 63'd0) && !det_nan; - wire back_d = det_g2[63]; + wire det_nan = (det_g[30:23] == 8'hff) && (det_g[22:0] != 23'd0); + wire det_ok_d = (det_g[30:0] != 31'd0) && !det_nan; + wire back_d = det_g[31]; - wire [2:0] flags_i; + wire [2:0] flags_h; VX_shift_register #( .DATAW (3), - .DEPTH (T_I - (T_G + 2 * D)) + .DEPTH (T_H - (T_F + 2 * F)) ) sr_flags ( .clk (clk), .reset (reset), .enable (enable), .data_in ({edge_ok_d, det_ok_d, back_d}), - .data_out (flags_i) + .data_out (flags_h) ); - wire [63:0] tmm_i; + wire [63:0] tmm_h; VX_shift_register #( .DATAW (64), - .DEPTH (T_I - T_B) + .DEPTH (T_H - T_B) ) sr_tmm ( .clk (clk), .reset (reset), .enable (enable), .data_in ({tmin_a, tmax_a}), - .data_out (tmm_i) + .data_out (tmm_h) ); - // ── stage I (@T_I): range test and commit ───────────────────────── - // open interval, as the Vulkan reference commits a triangle hit - wire range_ok = f32_lt(tmm_i[63:32], t_i) && f32_lt(t_i, tmm_i[31:0]); + // ── stage H (@T_H): t_min < t < t_max ───────────────────────────── + wire range_ok = f32_lt(tmm_h[63:32], tuv_h[0]) && f32_lt(tuv_h[0], tmm_h[31:0]); reg hit_r, bf_r; reg [31:0] u_r, v_r, t_r; always_ff @(posedge clk) begin if (enable) begin - hit_r <= flags_i[2] && flags_i[1] && range_ok; - bf_r <= flags_i[0]; - u_r <= u_i; - v_r <= v_i; - t_r <= t_i; + hit_r <= flags_h[2] && flags_h[1] && range_ok; + bf_r <= flags_h[0]; + t_r <= tuv_h[0]; + u_r <= tuv_h[1]; + v_r <= tuv_h[2]; end end diff --git a/hw/unittest/rtu_tri_pe/main.cpp b/hw/unittest/rtu_tri_pe/main.cpp index b7a16a0195..e33687ada1 100644 --- a/hw/unittest/rtu_tri_pe/main.cpp +++ b/hw/unittest/rtu_tri_pe/main.cpp @@ -28,6 +28,7 @@ const char* const kCatNames[] = { "random", "t==tmin", "t==tmax", "t==tmin==tmax", "t in (t-,t+)", "tmin=t+", "tmax=t-", "origin on triangle", "zero dir", "special origin", "special dir", "special vertex", "special tmin", "special tmax", + "shared edge", }; constexpr int kNumCats = int(sizeof(kCatNames) / sizeof(kCatNames[0])); @@ -105,6 +106,36 @@ Case make_case(uint32_t i, const Case* twin_of) { return c; } +// Quad (p0, p1, p2, p3) split along p0-p2 into (p0, p1, p2) and (p0, p2, p3), +// both wound alike; the ray aims at a point on the shared edge p0-p2. +float quad_pts[20000][4][3]; + +Case shared_edge_case(uint32_t i) { + std::mt19937 g(1000 + i); + auto u = [&](float lo, float hi) { return std::uniform_real_distribution(lo, hi)(g); }; + float (&p)[4][3] = quad_pts[i]; + const float s = (i % 3 == 0) ? 1e3f : ((i % 3 == 1) ? 1.f : 1e-2f); + for (int k = 0; k < 3; ++k) { p[0][k] = u(-s, s); p[2][k] = u(-s, s); } + float m[3], n[3]; + for (int k = 0; k < 3; ++k) { m[k] = 0.5f * (p[0][k] + p[2][k]); n[k] = u(-s, s); } + for (int k = 0; k < 3; ++k) { p[1][k] = m[k] + n[k]; p[3][k] = m[k] - n[k]; } + Case c; + c.cat = 14; + std::memcpy(c.v[0], p[0], 12); + std::memcpy(c.v[1], p[1], 12); + std::memcpy(c.v[2], p[2], 12); + const float a = u(0.f, 1.f); + float e[3]; + for (int k = 0; k < 3; ++k) e[k] = p[0][k] + a * (p[2][k] - p[0][k]); + for (int k = 0; k < 3; ++k) c.o[k] = u(-3 * s, 3 * s); + for (int k = 0; k < 3; ++k) c.d[k] = e[k] - c.o[k]; + c.tmin = 0.f; + c.tmax = INFINITY; + return c; +} + +const float* shared_edge_far(uint32_t i) { return quad_pts[i][3]; } + // Directed cases: t landing exactly on tmin / tmax, a zero t from an origin on // the triangle, and NaN / inf / -0 / subnormal operands. std::vector directed_cases() { @@ -163,9 +194,36 @@ std::vector directed_cases() { z.d[0] = z.d[1] = z.d[2] = (i & 1) ? -0.f : 0.f; out.push_back(z); } + // rays at the shared edge of two triangles (a quad split along its diagonal), + // both triangles traced + for (uint32_t i = 0; i < 20000; ++i) { + Case c = shared_edge_case(i); + out.push_back(c); + std::memcpy(c.v[1], c.v[2], 12); + std::memcpy(c.v[2], shared_edge_far(i), 12); + out.push_back(c); + } return out; } +// Watertightness: a ray through a shared edge hits at least one of the two +// triangles (SimX alone; the RTL then matches it case by case). +uint32_t shared_edge_leaks() { + uint32_t leaks = 0; + for (uint32_t i = 0; i < 20000; ++i) { + Case a = shared_edge_case(i), b = a; + std::memcpy(b.v[1], b.v[2], 12); + std::memcpy(b.v[2], shared_edge_far(i), 12); + float t, u, v; bool bf; + const bool ha = vortex::rtu::ray_triangle(a.o, a.d, a.v[0], a.v[1], a.v[2], + a.tmin, a.tmax, t, u, v, bf); + const bool hb = vortex::rtu::ray_triangle(b.o, b.d, b.v[0], b.v[1], b.v[2], + b.tmin, b.tmax, t, u, v, bf); + leaks += !(ha || hb); + } + return leaks; +} + } // namespace int main(int argc, char** argv) { @@ -184,6 +242,7 @@ int main(int argc, char** argv) { std::deque sent_cases; Case prev{}; const std::vector directed = directed_cases(); + const uint32_t leaks = shared_edge_leaks(); const uint32_t ND = uint32_t(directed.size()); const uint32_t total = N + ND; while (checked < total) { @@ -247,6 +306,8 @@ int main(int argc, char** argv) { for (int k = 0; k < kNumCats; ++k) std::printf(" %-20s %8u cases %6u mismatches %6u subnormal flushes\n", kCatNames[k], cat_cases[k], cat_errors[k], cat_ftz[k]); - std::printf(errors ? "FAILED!\n" : "PASSED!\n"); - return errors ? 1 : 0; + std::printf(" shared-edge rays through neither triangle: %u of 20000\n", leaks); + const bool fail = errors || leaks; + std::printf(fail ? "FAILED!\n" : "PASSED!\n"); + return fail ? 1 : 0; } diff --git a/sim/simx/rtu/rtu_isect.cpp b/sim/simx/rtu/rtu_isect.cpp index aa10099bbf..f0c219bf58 100644 --- a/sim/simx/rtu/rtu_isect.cpp +++ b/sim/simx/rtu/rtu_isect.cpp @@ -25,13 +25,12 @@ bool ray_triangle(const float ro[3], const float rd[3], bool& out_back_facing) { const float* vin[3] = { v0, v1, v2 }; - // Watertight ray/triangle test (Woop, Benthin, Wald, JCGT 2013): shear the - // triangle into the ray's frame so the ray runs along +z, then test the 2D - // edge functions. A shared edge evaluates to exactly negated values in its - // two triangles, so no ray slips between them; F64 edge functions and t keep - // t within half an ulp of the exact intersection. The op order is the one - // the Vulkan reference (lavapipe) evaluates, so t matches it bit for bit, - // coincident triangles included. + // Watertight ray/triangle test (Woop, Benthin, Wald, JCGT 2013), F32 only: + // shear the triangle into the ray's frame so the ray runs along +z, then + // test the 2D edge functions. Each edge function is two rounded products and + // a rounded difference of the sheared vertices alone, so an edge shared by + // two triangles evaluates to exactly negated weights in both and no ray + // slips between them. Mirrors VX_rtu_tri_pe op for op. const float ad[3] = { std::fabs(rd[0]), std::fabs(rd[1]), std::fabs(rd[2]) }; int kz = (ad[0] >= ad[1]) ? ((ad[0] >= ad[2]) ? 0 : 2) : ((ad[1] >= ad[2]) ? 1 : 2); @@ -43,50 +42,42 @@ bool ray_triangle(const float ro[3], const float rd[3], const float sx = rd[kx] * sz; const float sy = rd[ky] * sz; - // F32 shear, as each op rounds in the pipeline; F64 from here on, where the - // products of two F32 values are exact. - float px[3], py[3]; - double pz[3]; + float px[3], py[3], pz[3]; for (int i = 0; i < 3; ++i) { const float* q = vin[i]; const float rx = q[kx] - ro[kx]; const float ry = q[ky] - ro[ky]; const float rz = q[kz] - ro[kz]; - const float mx = sx * rz; - const float my = sy * rz; - px[i] = rx - mx; - py[i] = ry - my; - pz[i] = double(sz) * double(rz); + px[i] = std::fma(-sx, rz, rx); + py[i] = std::fma(-sy, rz, ry); + pz[i] = sz * rz; } // Edge functions: w[i] is the weight of vertex i. - double w[3]; - w[0] = double(px[2]) * py[1] - double(py[2]) * px[1]; - w[1] = double(px[0]) * py[2] - double(py[0]) * px[2]; - w[2] = double(px[1]) * py[0] - double(py[1]) * px[0]; - if ((w[0] < 0.0 || w[1] < 0.0 || w[2] < 0.0) - && (w[0] > 0.0 || w[1] > 0.0 || w[2] > 0.0)) + const float w0 = px[2] * py[1] - py[2] * px[1]; + const float w1 = px[0] * py[2] - py[0] * px[2]; + const float w2 = px[1] * py[0] - py[1] * px[0]; + if ((w0 < 0.f || w1 < 0.f || w2 < 0.f) && (w0 > 0.f || w1 > 0.f || w2 > 0.f)) return false; - const double det = w[0] + (w[1] + w[2]); + const float det = (w0 + w1) + w2; // Reject only an edge-on or zero-area triangle: |det| scales with the // triangle's area, so any epsilon would drop small triangles. - if (!(det != 0.0)) return false; + if (!(det != 0.f)) return false; - const double tp0 = w[0] * pz[0]; - const double tp1 = w[1] * pz[1]; - const double tp2 = w[2] * pz[2]; - const float t = float(((tp0 + tp1) + tp2) / det); - // Open interval, as the reference commits (lvp_build_triangle_case: - // tmin < t and t < tmax). + const float T = std::fma(w2, pz[2], std::fma(w1, pz[1], w0 * pz[0])); + const float rcp = 1.0f / det; + const float t = T * rcp; + // Vulkan's ray interval for a triangle is open at both ends: an intersection + // candidate needs t_min < t < t_max (Ray Intersection Candidate + // Determination). if (!(tmin < t && t < tmax)) return false; - const float det32 = float(det); out_t = t; - out_u = float(w[1]) / det32; - out_v = float(w[2]) / det32; + out_u = w1 * rcp; + out_v = w2 * rcp; // det > 0: (v0, v1, v2) winds counter-clockwise as seen by the ray. - out_back_facing = (det < 0.0); + out_back_facing = (det < 0.f); return true; } @@ -146,10 +137,9 @@ uint32_t BoxPe::pipe_depth() { } uint32_t TriPe::pipe_depth() { - // input select + 1/dir + 3 F32 stages + 5 F64 stages + F64 divide + narrow - // + verdict (VX_rtu_tri_pe). - return 3 + kRtuFdivLat + 3 * kRtuLatencyFma + 5 * kRtuLatencyFma64 - + kRtuFdiv64Lat; + // input select + 1/dir[kz] + shear scale + shear + 2 edge stages + det + + // 1/det + t/u/v scale + verdict (VX_rtu_tri_pe). + return 2 + 2 * kRtuFdivLat + 7 * kRtuLatencyFma; } }} // namespace vortex::rtu diff --git a/sim/simx/rtu/rtu_isect.h b/sim/simx/rtu/rtu_isect.h index 893a5f5b19..240591409b 100644 --- a/sim/simx/rtu/rtu_isect.h +++ b/sim/simx/rtu/rtu_isect.h @@ -30,13 +30,13 @@ namespace vortex { namespace rtu { // ──────────────────────────────────────────────────────────────────── -// Möller-Trumbore ray-triangle intersection. +// Watertight ray-triangle intersection (Woop, Benthin, Wald, JCGT 2013), F32. // // out_back_facing reports whether the ray hit the back side of the // triangle's geometric normal (ray-flag face culling). Convention: // triangle front face is the side from which (v0, v1, v2) appear CCW. // Equivalently, det > 0 ↔ ray hits the front face. -// A hit needs tmin < t < tmax (open, as the Vulkan reference); the walker +// A hit needs tmin < t < tmax (Vulkan's open triangle interval); the walker // passes the committed hit's t as tmax. // ──────────────────────────────────────────────────────────────────── bool ray_triangle(const float ro[3], const float rd[3], @@ -76,7 +76,7 @@ void world_to_object_ray(const float wto[12], // ════════════════════════════════════════════════════════════════════ // // BoxPe (ray-vs-AABB): ONE PE, 1 box/cycle, 31-cycle pipeline depth. -// TriPe (ray-vs-tri): ONE PE, 1 tri/cycle, 91-cycle pipeline depth. +// TriPe (ray-vs-tri): ONE PE, 1 tri/cycle, 99-cycle pipeline depth. // // Both are shared across the whole context array, so the issue slots are handed // out by the orchestrator one per cycle and the contention is modelled, not diff --git a/sim/simx/rtu/rtu_types.h b/sim/simx/rtu/rtu_types.h index ff48d434e6..0b922dda06 100644 --- a/sim/simx/rtu/rtu_types.h +++ b/sim/simx/rtu/rtu_types.h @@ -304,8 +304,6 @@ constexpr uint32_t kRtuImageStatesPerRay = 1; // scene header constexpr uint32_t kRtuSetupLatency = 17; // reciprocal pipe depth constexpr uint32_t kRtuFdivLat = 17; // reciprocal pipe depth constexpr uint32_t kRtuLatencyFma = 9; // FMA pipe depth -constexpr uint32_t kRtuLatencyFma64 = 12; // F64 FMA pipe depth (tri PE) -constexpr uint32_t kRtuFdiv64Lat = 32; // F64 divide pipe depth (tri PE) // Per-instance transform latency = 4 * FMA pipe depth = 36: the products, then // three dependent adds (VX_rtu_xform). Charged per TLAS instance descent in the // SimX cost model. From 05ac466229198de3aa7ca05132d13c55b1cd1d2b Mon Sep 17 00:00:00 2001 From: Blaise Tine Date: Sat, 3 Oct 2026 13:14:24 -0700 Subject: [PATCH 23/31] rtu: cull boxes against the ray interval; one-add box decode The box test culled against [0, t_max] instead of [t_min, t_max]. That was added only so a box whose slab exit lies just below t_min would still be entered when the Vulkan reference's (lavapipe's) rounding put a triangle's t inside the interval. The standard slab test culls against the ray's own interval, with t_max shrunk to the committed hit: lo = max(t_min, min(t0, t1)[x,y,z]), hi = min(t_max, max(t0, t1)[x,y,z]) hit = lo <= hi, t_near = lo Kept: a zero (or subnormal) direction component has reciprocal FLT_MAX, so no slab is ever 0 * inf = NaN; fmin/fmax drop a NaN slab instead of poisoning the fold. Box decode: the corner relative to the ray origin is now formed as q*2^exp + (origin - ro) -- one subtract per axis shared by both corners, then one add per corner (q*2^exp is exact) -- instead of (origin + q*2^exp) - ro. Same 3-FMA-deep pipeline, 15 F32 units instead of 18. A raw (procedural) box uses origin = +0. - VX_rtu_box_pe: t_min/t_max fold into the min/max reduction tree, so the verdict stage is a single compare. - SimX: ray_recip / quant_corner / box_rel / ray_box replace reconstruct_child_aabb + ray_aabb_intersect and mirror the PE op for op (quant_corner flushes a subnormal corner as the PE does; the walker sets up the reciprocals once per ray). - hw/unittest/rtu_box_pe: 1.1M boxes (103.5K directed), 0 mismatches in inv_d / verdict / t_near and 0 in child order; the 1904 subnormal-flush cases all come from the PEs' FTZ (e.g. 1/FLT_MAX). Co-Authored-By: Claude Opus 5.5 --- docs/designs/ray_tracing_architecture.md | 9 +- hw/rtl/rtu/VX_rtu_box_pe.sv | 131 +++++++++-------------- hw/rtl/rtu/VX_rtu_recip.sv | 6 +- hw/unittest/rtu_box_pe/main.cpp | 25 +++-- sim/simx/rtu/rtu_isect.cpp | 52 +++++---- sim/simx/rtu/rtu_isect.h | 25 +++-- sim/simx/rtu/rtu_walker.cpp | 37 +++---- 7 files changed, 139 insertions(+), 146 deletions(-) diff --git a/docs/designs/ray_tracing_architecture.md b/docs/designs/ray_tracing_architecture.md index cfc65911e6..c6570773a4 100644 --- a/docs/designs/ray_tracing_architecture.md +++ b/docs/designs/ray_tracing_architecture.md @@ -327,9 +327,12 @@ subnormals flushed either way), and `VX_CFG_FMA_LATENCY` / whatever depth results: - **`VX_rtu_box_pe`** — pipelined ray/AABB slab test, one child box per cycle, - emitting `{hit, t_near}`. Mirrors SimX `ray_aabb_intersect` bit for bit: - corners `origin + q·2^exp`, slabs `(corner − ro)·inv_d`, culled against - `[0, t_max]`. Also handles raw/procedural boxes. + emitting `{hit, t_near}`. Mirrors SimX `box_rel` + `ray_box` bit for bit: + corners relative to the ray, `q·2^exp + (origin − ro)` (the product exact), + slabs `rel·inv_d`, culled against the ray interval `[t_min, t_max]` with + `t_max` the committed hit. `inv_d` is `FLT_MAX` for a zero direction + component, so no slab is ever `0·inf`. Also handles raw/procedural boxes + (`origin = +0`). - **`VX_rtu_tri_pe`** — pipelined watertight triangle test (Woop, Benthin, Wald, JCGT 2013), all F32: shear, edge functions as rounded products and a rounded difference (so a shared edge evaluates to exactly negated weights in its two diff --git a/hw/rtl/rtu/VX_rtu_box_pe.sv b/hw/rtl/rtu/VX_rtu_box_pe.sv index 55f34f9efa..446e42a1e6 100644 --- a/hw/rtl/rtu/VX_rtu_box_pe.sv +++ b/hw/rtl/rtu/VX_rtu_box_pe.sv @@ -13,20 +13,19 @@ // VX_rtu_box_pe — pipelined ray-vs-AABB slab intersector for one child box. // Streams one box per cycle; emits {hit, t_near} after a fixed latency. Mirrors -// SimX reconstruct_child_aabb + rtu::ray_aabb_intersect op for op, so every -// accept decision and t_near match it: +// SimX rtu::box_rel + rtu::ray_box op for op: // -// dequant mn[a] = origin[a] + q[a]*2^exp[a] (product exact, one add) -// slab t0[a] = (mn[a] - ro[a]) * inv_d[a] (t1 from mx) -// reduce lo = max(-inf, min(t0,t1)[*]) hi = min(+inf, max(t0,t1)[*]) -// (fmin/fmax drop a NaN operand, so lo and hi are never NaN) -// hit = hi >= max(0, lo) && lo <= t_max -// t_near = max(t_min, lo) (descent order only) +// base c[a] = origin[a] - ro[a] (raw box: +0 - ro[a]) +// corner d[a] = q[a]*2^exp[a] + c[a] (product exact; raw: corner + c) +// slab t0[a] = dmn[a] * inv_d[a] (t1 from dmx) +// reduce lo = max(t_min, min(t0,t1)[*]) hi = min(t_max, max(t0,t1)[*]) +// (fmin/fmax drop a NaN operand) +// hit = lo <= hi, t_near = lo // -// The box is culled against [0, t_max], not [t_min, t_max]: t_min belongs to -// the primitive test alone, whose t carries rounding the slab distances do not. -// inv_d is the ray-setup reciprocal (VX_rtu_recip), FLT_MAX for a zero -// direction component. The FP units flush subnormals. +// The box is culled against the ray interval [t_min, t_max] (t_max is the +// committed hit). inv_d is the ray-setup reciprocal (VX_rtu_recip), FLT_MAX for +// a zero direction component, so no slab is ever 0 * inf. The FP units flush +// subnormals. `include "VX_define.vh" @@ -66,12 +65,11 @@ module VX_rtu_box_pe import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( output wire [31:0] t_near ); localparam F = LATENCY_FMA; - localparam LAT_FMA = 3 * F; // mn, mn - ro, * inv_d + localparam LAT_FMA = 3 * F; // origin - ro, + corner, * inv_d localparam LATENCY = LAT_FMA + 4; // + per-axis, 2 reduce, verdict localparam [INST_FMT_BITS-1:0] FMT_ADD = 2'b00; // F32, a*b + c localparam [INST_FMT_BITS-1:0] FMT_SUB = 2'b10; // F32, a*b - c - localparam [31:0] F32_ONE = 32'h3F800000; localparam [31:0] F32_NEG0 = 32'h80000000; // x + -0 == x, signs kept localparam [31:0] F32_PINF = 32'h7F800000; localparam [31:0] F32_NINF = 32'hFF800000; @@ -141,69 +139,46 @@ module VX_rtu_box_pe import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( end endfunction - // ── stage 1: box corners mn = origin + q*2^exp (raw: the corner itself) ── - wire [2:0][31:0] mn_a, mx_a, mnx_c; - for (genvar a = 0; a < 3; ++a) begin : g_prep - assign mn_a[a] = raw ? raw_min[a] : q_scale(qmin[a], exp[a]); - assign mx_a[a] = raw ? raw_max[a] : q_scale(qmax[a], exp[a]); - assign mnx_c[a] = raw ? F32_NEG0 : origin[a]; - end - - wire [2:0][31:0] mn, mx; - for (genvar a = 0; a < 3; ++a) begin : g_corner - VX_fma_unit #( - .USE_DSP (`VX_CFG_RTU_USE_DSP), - .LATENCY (F), - .SUBNORM_ENABLE (0), - .EXCEPT_ENABLE (1) - ) fma_mn ( - .clk (clk), - .reset (reset), - .enable (enable), - .mask (valid_in), - .op_type (INST_FPU_MADD), - .fmt (FMT_ADD), - .frm (INST_FRM_RNE), - .dataa (mn_a[a]), - .datab (F32_ONE), - .datac (mnx_c[a]), - .result (mn[a]), - `UNUSED_PIN (fflags) - ); + // ── stage 1: the box base relative to the ray origin ───────────── + wire [2:0][31:0] mn_a, mx_a, base_a, c_b; + for (genvar a = 0; a < 3; ++a) begin : g_base + assign mn_a[a] = raw ? raw_min[a] : q_scale(qmin[a], exp[a]); + assign mx_a[a] = raw ? raw_max[a] : q_scale(qmax[a], exp[a]); + assign base_a[a] = raw ? 32'd0 : origin[a]; VX_fma_unit #( .USE_DSP (`VX_CFG_RTU_USE_DSP), .LATENCY (F), .SUBNORM_ENABLE (0), .EXCEPT_ENABLE (1) - ) fma_mx ( + ) fsub_c ( .clk (clk), .reset (reset), .enable (enable), - .mask (valid_in), - .op_type (INST_FPU_MADD), - .fmt (FMT_ADD), + .mask (1'b1), + .op_type (INST_FPU_ADD), + .fmt (FMT_SUB), .frm (INST_FRM_RNE), - .dataa (mx_a[a]), - .datab (F32_ONE), - .datac (mnx_c[a]), - .result (mx[a]), + .dataa (base_a[a]), + .datab (ro[a]), + .datac ('0), + .result (c_b[a]), `UNUSED_PIN (fflags) ); end - wire [2:0][31:0] ro_d; + wire [2:0][31:0] mn_b, mx_b; VX_shift_register #( - .DATAW (3*32), + .DATAW (6*32), .DEPTH (F) - ) sr_ro ( + ) sr_corner ( .clk (clk), .reset (reset), .enable (enable), - .data_in (ro), - .data_out (ro_d) + .data_in ({mn_a, mx_a}), + .data_out ({mn_b, mx_b}) ); - // ── stage 2: corners relative to the ray origin, mn - ro / mx - ro ── + // ── stage 2: corners relative to the ray origin, corner + c ────── wire [2:0][31:0] dmn, dmx; for (genvar a = 0; a < 3; ++a) begin : g_rel VX_fma_unit #( @@ -211,17 +186,17 @@ module VX_rtu_box_pe import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( .LATENCY (F), .SUBNORM_ENABLE (0), .EXCEPT_ENABLE (1) - ) fma_dmn ( + ) fadd_dmn ( .clk (clk), .reset (reset), .enable (enable), .mask (1'b1), - .op_type (INST_FPU_MADD), - .fmt (FMT_SUB), + .op_type (INST_FPU_ADD), + .fmt (FMT_ADD), .frm (INST_FRM_RNE), - .dataa (mn[a]), - .datab (F32_ONE), - .datac (ro_d[a]), + .dataa (mn_b[a]), + .datab (c_b[a]), + .datac ('0), .result (dmn[a]), `UNUSED_PIN (fflags) ); @@ -230,17 +205,17 @@ module VX_rtu_box_pe import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( .LATENCY (F), .SUBNORM_ENABLE (0), .EXCEPT_ENABLE (1) - ) fma_dmx ( + ) fadd_dmx ( .clk (clk), .reset (reset), .enable (enable), .mask (1'b1), - .op_type (INST_FPU_MADD), - .fmt (FMT_SUB), + .op_type (INST_FPU_ADD), + .fmt (FMT_ADD), .frm (INST_FRM_RNE), - .dataa (mx[a]), - .datab (F32_ONE), - .datac (ro_d[a]), + .dataa (mx_b[a]), + .datab (c_b[a]), + .datac ('0), .result (dmx[a]), `UNUSED_PIN (fflags) ); @@ -259,6 +234,7 @@ module VX_rtu_box_pe import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( ); // ── stage 3: slab entry/exit per axis = (corner - ro) * inv_d ───── + // a*b + -0 is the exact product, its zero sign included wire [2:0][31:0] t0, t1; for (genvar a = 0; a < 3; ++a) begin : g_slab VX_fma_unit #( @@ -305,7 +281,7 @@ module VX_rtu_box_pe import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( wire [31:0] tmin_r, tmax_r; VX_shift_register #( .DATAW (64), - .DEPTH (LAT_FMA + 3) + .DEPTH (LAT_FMA + 1) ) sr_t ( .clk (clk), .reset (reset), @@ -325,32 +301,29 @@ module VX_rtu_box_pe import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( end end - // ── stage 5/6: lo = max over axes, hi = min over axes ───────────── + // ── stage 5/6: lo = max(t_min, axes), hi = min(t_max, axes) ────── reg [31:0] near_a_r, near_b_r, far_a_r, far_b_r; reg [31:0] lo_all_r, hi_all_r; always_ff @(posedge clk) begin if (enable) begin near_a_r <= f32_max(lo_r[0], lo_r[1], F32_NINF); - near_b_r <= lo_r[2]; + near_b_r <= f32_max(lo_r[2], tmin_r, F32_NINF); far_a_r <= f32_min(hi_r[0], hi_r[1], F32_PINF); - far_b_r <= hi_r[2]; + far_b_r <= f32_min(hi_r[2], tmax_r, F32_PINF); lo_all_r <= f32_max(near_a_r, near_b_r, F32_NINF); hi_all_r <= f32_min(far_a_r, far_b_r, F32_PINF); end end - // ── stage 7: hit = hi >= max(0, lo) && lo <= t_max; t_near = max(t_min, lo) + // ── stage 7: hit = lo <= hi; t_near = lo ───────────────────────── // lo and hi are never NaN; t_near is non-negative for t_min >= 0 and +0 for // a zero, so the consumer may order it as an unsigned integer. - wire hi_ge0 = !hi_all_r[31] || (hi_all_r[30:0] == 31'd0); - wire hit_w = hi_ge0 && f32_le(lo_all_r, hi_all_r) && f32_le(lo_all_r, tmax_r); - wire [31:0] tnear_w = f32_max(tmin_r, lo_all_r, lo_all_r); reg hit_r; reg [31:0] t_near_r; always_ff @(posedge clk) begin if (enable) begin - hit_r <= hit_w; - t_near_r <= (tnear_w[30:0] == 31'd0) ? 32'd0 : tnear_w; + hit_r <= f32_le(lo_all_r, hi_all_r); + t_near_r <= (lo_all_r[30:0] == 31'd0) ? 32'd0 : lo_all_r; end end diff --git a/hw/rtl/rtu/VX_rtu_recip.sv b/hw/rtl/rtu/VX_rtu_recip.sv index 29440da925..86dec04f5d 100644 --- a/hw/rtl/rtu/VX_rtu_recip.sv +++ b/hw/rtl/rtu/VX_rtu_recip.sv @@ -21,9 +21,9 @@ // map to DSP48. Trades ~2K LUT/unit onto the idle BRAM + DSP blocks. // ~9e-8 max relative error (well inside the RTU's 1e-4 tolerance). // -// A zero (or, flushed, subnormal) operand returns +FLT_MAX rather than inf, as -// the box test's reference does: a slab along a zero direction component then -// stays finite, (b - o) * FLT_MAX, instead of 0 * inf = NaN. +// A zero (or, flushed, subnormal) operand returns +FLT_MAX rather than inf +// (SimX rtu::ray_recip): a slab along a zero direction component then stays +// finite, (b - o) * FLT_MAX, instead of 0 * inf = NaN. // // The input is presented combinationally and held stable for the whole setup // span by the scheduler; the result is a fixed-latency pipeline output, valid diff --git a/hw/unittest/rtu_box_pe/main.cpp b/hw/unittest/rtu_box_pe/main.cpp index 9bcaef590a..73f8916470 100644 --- a/hw/unittest/rtu_box_pe/main.cpp +++ b/hw/unittest/rtu_box_pe/main.cpp @@ -1,7 +1,7 @@ -// VX_rtu_recip + VX_rtu_box_pe against SimX's box test (reconstruct_child_aabb -// + rtu::ray_aabb_intersect): inv_d bit for bit, every accept decision, t_near -// by value, and the order the scheduler's insertion collector gives a node's -// accepted children against SimX's nearest-first sort. +// VX_rtu_recip + VX_rtu_box_pe against SimX's box test (rtu::ray_recip, +// quant_corner, box_rel, ray_box): inv_d bit for bit, every accept decision, +// t_near by value, and the order the scheduler's insertion collector gives a +// node's accepted children against SimX's nearest-first sort. // // The PE's FP units flush subnormals (FTZ/DAZ) while SimX runs IEEE. A case // whose RTL result differs from SimX but equals SimX evaluated under the host's @@ -66,7 +66,7 @@ bool same_value(uint32_t rtl, float ref) { return std::isnan(ref) ? is_nan_bits(rtl) : fbits(rtl) == ref; } -// SimX rtu_walker.cpp reconstruct_child_aabb +// the child box in world space, to aim rays at void reconstruct_child_aabb(const float origin[3], const int8_t exp[3], const uint8_t qmin[3], const uint8_t qmax[3], float out_mn[3], float out_mx[3]) { @@ -90,18 +90,23 @@ struct Ref { Ref reference(const Node& nd, int c, bool ftz) { const unsigned csr = _mm_getcsr(); if (ftz) _mm_setcsr(csr | 0x8040); // FTZ | DAZ + namespace rtu = vortex::rtu; Ref r; for (int i = 0; i < 3; ++i) - r.inv[i] = (nd.ray.d[i] == 0.0f) ? FLT_MAX : 1.0f / nd.ray.d[i]; - float mn[3], mx[3]; + r.inv[i] = rtu::ray_recip(nd.ray.d[i]); + static const float kRawBase[3] = { 0.f, 0.f, 0.f }; + float mn[3], mx[3], rel_mn[3], rel_mx[3]; if (nd.raw) { std::memcpy(mn, nd.ch[c].rmin, sizeof mn); std::memcpy(mx, nd.ch[c].rmax, sizeof mx); } else { - reconstruct_child_aabb(nd.origin, nd.exp, nd.ch[c].qmin, nd.ch[c].qmax, mn, mx); + for (int i = 0; i < 3; ++i) { + mn[i] = rtu::quant_corner(nd.ch[c].qmin[i], nd.exp[i]); + mx[i] = rtu::quant_corner(nd.ch[c].qmax[i], nd.exp[i]); + } } - r.box.hit = vortex::rtu::ray_aabb_intersect(nd.ray.o, nd.ray.d, mn, mx, - nd.ray.tmin, nd.ray.tmax, r.box.t_near); + rtu::box_rel(nd.raw ? kRawBase : nd.origin, mn, mx, nd.ray.o, rel_mn, rel_mx); + r.box.hit = rtu::ray_box(rel_mn, rel_mx, r.inv, nd.ray.tmin, nd.ray.tmax, r.box.t_near); _mm_setcsr(csr); return r; } diff --git a/sim/simx/rtu/rtu_isect.cpp b/sim/simx/rtu/rtu_isect.cpp index f0c219bf58..3dde034c81 100644 --- a/sim/simx/rtu/rtu_isect.cpp +++ b/sim/simx/rtu/rtu_isect.cpp @@ -81,30 +81,42 @@ bool ray_triangle(const float ro[3], const float rd[3], return true; } -bool ray_aabb_intersect(const float ro[3], const float rd[3], - const float mn[3], const float mx[3], - float tmin, float tmax, float& t_near) { - // Slab test in the Vulkan reference's (lavapipe) form: a zero direction - // component uses FLT_MAX as its reciprocal, and the box is culled against - // [0, tmax], NOT [tmin, tmax]. The tmin floor belongs to the primitive test - // alone: a primitive's t carries rounding the slab distances do not (the - // watertight triangle t of a large triangle is off by far more than the - // slabs of its flat box), so a box whose exact exit lies below tmin can - // still hold a hit the primitive test reports past tmin. Culling at tmin - // would drop that hit, which the reference keeps. - float lo = -INFINITY, hi = INFINITY; +float ray_recip(float d) { + // A zero (or subnormal, which the PEs flush) component has no reciprocal; + // FLT_MAX keeps every slab product finite, so no slab is 0 * inf = NaN. + if (std::fabs(d) < FLT_MIN) return FLT_MAX; + return 1.0f / d; +} + +float quant_corner(uint8_t q, int8_t e) { + // Exact product; the PE flushes a subnormal result, inf past the range. + const float c = std::ldexp(float(q), e); + return (c < FLT_MIN) ? 0.f : c; +} + +void box_rel(const float base[3], const float mn[3], const float mx[3], + const float ro[3], float rel_mn[3], float rel_mx[3]) { + for (int i = 0; i < 3; ++i) { + const float c = base[i] - ro[i]; + rel_mn[i] = mn[i] + c; + rel_mx[i] = mx[i] + c; + } +} + +bool ray_box(const float rel_mn[3], const float rel_mx[3], const float inv[3], + float tmin, float tmax, float& t_near) { + // fmin/fmax drop a NaN operand, so a NaN slab (a NaN box or ray) drops out + // of the fold rather than poisoning it. + float lo = std::fmax(-INFINITY, tmin); + float hi = std::fmin(INFINITY, tmax); for (int i = 0; i < 3; ++i) { - const float inv = (rd[i] == 0.0f) ? FLT_MAX : 1.0f / rd[i]; - const float t0 = (mn[i] - ro[i]) * inv; - const float t1 = (mx[i] - ro[i]) * inv; + const float t0 = rel_mn[i] * inv[i]; + const float t1 = rel_mx[i] * inv[i]; lo = std::fmax(lo, std::fmin(t0, t1)); hi = std::fmin(hi, std::fmax(t0, t1)); } - // The upper bound stays inclusive (the reference's is strict): a box - // entered exactly at the committed t can hold an equal-t twin that the - // lowest-(instance, geometry, prim) tie-break must still see. - if (!(hi >= std::fmax(0.0f, lo) && lo <= tmax)) return false; - t_near = std::fmax(tmin, lo); // descent order only + if (!(lo <= hi)) return false; + t_near = (lo == 0.f) ? 0.f : lo; return true; } diff --git a/sim/simx/rtu/rtu_isect.h b/sim/simx/rtu/rtu_isect.h index 240591409b..8dd3005eb2 100644 --- a/sim/simx/rtu/rtu_isect.h +++ b/sim/simx/rtu/rtu_isect.h @@ -46,15 +46,22 @@ bool ray_triangle(const float ro[3], const float rd[3], bool& out_back_facing); // ──────────────────────────────────────────────────────────────────── -// Ray-vs-AABB slab test. Returns true if the ray's [0, tmax] interval -// overlaps the AABB (the reference's box test: tmin is applied by the -// primitive test only, see rtu_isect.cpp); t_near is the entry -// parameter (clamped to tmin) used by the BVH4 walker to order descent. -// A zero direction component uses FLT_MAX as its reciprocal. +// Ray-vs-AABB slab test, as VX_rtu_recip + VX_rtu_box_pe compute it. +// +// ray_recip 1/d per direction component, FLT_MAX for a zero one +// quant_corner a quantized child corner q * 2^e (exact; FTZ) +// box_rel box corners relative to the ray origin: m + (base - ro), +// base = the node origin (+0 for a raw procedural box) +// ray_box slabs rel * inv, culled against [tmin, tmax]: hit iff +// max(tmin, entry) <= min(tmax, exit); t_near = that max, +// the entry distance the walker orders children by // ──────────────────────────────────────────────────────────────────── -bool ray_aabb_intersect(const float ro[3], const float rd[3], - const float mn[3], const float mx[3], - float tmin, float tmax, float& t_near); +float ray_recip(float d); +float quant_corner(uint8_t q, int8_t e); +void box_rel(const float base[3], const float mn[3], const float mx[3], + const float ro[3], float rel_mn[3], float rel_mx[3]); +bool ray_box(const float rel_mn[3], const float rel_mx[3], const float inv[3], + float tmin, float tmax, float& t_near); // ──────────────────────────────────────────────────────────────────── // Bring a world ray into an instance's object space with the instance @@ -81,7 +88,7 @@ void world_to_object_ray(const float wto[12], // Both are shared across the whole context array, so the issue slots are handed // out by the orchestrator one per cycle and the contention is modelled, not // assumed away. The math itself is done synchronously by the scalar -// ray_triangle / ray_aabb_intersect helpers above; these classes contribute only +// ray_triangle / ray_box helpers above; these classes contribute only // the drain behind the last test entered. class BoxPe { public: diff --git a/sim/simx/rtu/rtu_walker.cpp b/sim/simx/rtu/rtu_walker.cpp index 7b4f03e8db..3164e9f9d7 100644 --- a/sim/simx/rtu/rtu_walker.cpp +++ b/sim/simx/rtu/rtu_walker.cpp @@ -23,7 +23,7 @@ #include "rtu_types.h" // RtuReq, SceneView, LaneState, PerfStats, // scene-format constants #include "rtu_bvh.h" // CW-BVH node/leaf/instance layouts -#include "rtu_isect.h" // ray_triangle, ray_aabb_intersect, +#include "rtu_isect.h" // ray_triangle, ray_box, // world_to_object_ray #include "rtu_classifier.h" // classify_tri_hit, finalise_lane @@ -62,18 +62,6 @@ void read_scene_bytes(SceneView& sv, uint32_t off, uint32_t len, uint8_t* out) { } } -// CW-BVH: reconstruct a child AABB from quantized representation. -// real = origin + qaabb * 2^exp (per axis) -inline void reconstruct_child_aabb(const float origin[3], const int8_t exp[3], - const uint8_t qmin[3], const uint8_t qmax[3], - float out_mn[3], float out_mx[3]) { - for (int i = 0; i < 3; ++i) { - float scale = std::ldexp(1.0f, exp[i]); - out_mn[i] = origin[i] + static_cast(qmin[i]) * scale; - out_mx[i] = origin[i] + static_cast(qmax[i]) * scale; - } -} - // Copy a 3-vector (object-space ray capture helper). inline void vcopy3(float dst[3], const float src[3]) { dst[0] = src[0]; dst[1] = src[1]; dst[2] = src[2]; @@ -151,6 +139,9 @@ void walk_bvh4_subtree(SceneView& sv, uint32_t root_off, uint32_t instance_id, uint32_t custom_id, uint32_t inst_flags, WalkCtx& ctx, PerfStats& perf) { + // The ray setup's reciprocals, once per (object-space) ray. + const float inv[3] = { ray_recip(rd[0]), ray_recip(rd[1]), ray_recip(rd[2]) }; + auto visit_leaf_tri = [&](uint32_t leaf_off, uint32_t count) { uint8_t hdr_buf[kVxBvhLeafHeaderBytes]; read_scene_bytes(sv, leaf_off, sizeof(hdr_buf), hdr_buf); @@ -245,10 +236,11 @@ void walk_bvh4_subtree(SceneView& sv, if (sv.miss) return; const VxBvhProcAabb* rec = reinterpret_cast(rec_buf); - float t_near = 0.f; + static const float kRawBase[3] = { 0.f, 0.f, 0.f }; + float rel_mn[3], rel_mx[3], t_near = 0.f; + box_rel(kRawBase, rec->aabb_min, rec->aabb_max, ro, rel_mn, rel_mx); ++perf.bvh_box_tests; - if (!ray_aabb_intersect(ro, rd, rec->aabb_min, rec->aabb_max, - ctx.tmin, ctx.best_t, t_near)) { + if (!ray_box(rel_mn, rel_mx, inv, ctx.tmin, ctx.best_t, t_near)) { continue; } // Procedural primitives are inherently non-opaque (the IS decides the @@ -374,14 +366,15 @@ void walk_bvh4_subtree(SceneView& sv, uint32_t off_word = nv.child_offsets[i]; uint32_t child_off = off_word & kVxBvhChildOffsetMask; if (off_word == kVxBvhChildEmpty) continue; - float mn[3], mx[3]; - reconstruct_child_aabb(nv.origin, nv.exp, - nv.qaabb_min[i], nv.qaabb_max[i], - mn, mx); + float mn[3], mx[3], rel_mn[3], rel_mx[3]; + for (int a = 0; a < 3; ++a) { + mn[a] = quant_corner(nv.qaabb_min[i][a], nv.exp[a]); + mx[a] = quant_corner(nv.qaabb_max[i][a], nv.exp[a]); + } + box_rel(nv.origin, mn, mx, ro, rel_mn, rel_mx); float t_near = 0.f; ++perf.bvh_box_tests; - if (!ray_aabb_intersect(ro, rd, mn, mx, - ctx.tmin, ctx.best_t, t_near)) { + if (!ray_box(rel_mn, rel_mx, inv, ctx.tmin, ctx.best_t, t_near)) { continue; } hits[hit_count++] = { child_off, t_near }; From 6a7d4d866bfd2dd46ef5e0ca430b964ddeffbae1 Mon Sep 17 00:00:00 2001 From: Blaise Tine Date: Sat, 3 Oct 2026 13:16:32 -0700 Subject: [PATCH 24/31] rtu: instance transform as FMA chains The instance record keeps carrying the world->object matrix (as real RT hardware does; it also handles scaled and sheared instances with no inverse anywhere). What goes is the emulation of the Vulkan reference's (lavapipe's) op order -- every product rounded, then t + x + y + z, nothing fused -- which cost 18 multipliers and 15 adders over 4 FMA stages only to round like lavapipe. Each object-ray component is now three dependent FMAs: obj_ro[i] = fma(ro.z, m[i][2], fma(ro.y, m[i][1], fma(ro.x, m[i][0], m[i][3]))) obj_rd[i] = fma(rd.z, m[i][2], fma(rd.y, m[i][1], rd.x * m[i][0])) - VX_rtu_xform: 18 FMA units, 3*FMA latency (was 33 units, 4*FMA), every stage registered; the direction's first step adds -0 so the product keeps its own zero sign. - SimX world_to_object_ray mirrors it with std::fma; kRtuXformLatency 36 -> 27. - hw/unittest/rtu_xform: 1M random affine (rotation, non-uniform scale, shear) + 38K directed special cases, 0 mismatches vs SimX (1 subnormal flush). Co-Authored-By: Claude Opus 5.5 --- docs/designs/ray_tracing_architecture.md | 5 +- hw/rtl/rtu/VX_rtu_xform.sv | 171 +++++++---------------- sim/simx/rtu/rtu_isect.cpp | 11 +- sim/simx/rtu/rtu_isect.h | 7 +- sim/simx/rtu/rtu_types.h | 8 +- 5 files changed, 63 insertions(+), 139 deletions(-) diff --git a/docs/designs/ray_tracing_architecture.md b/docs/designs/ray_tracing_architecture.md index c6570773a4..cc100fc1eb 100644 --- a/docs/designs/ray_tracing_architecture.md +++ b/docs/designs/ray_tracing_architecture.md @@ -341,8 +341,9 @@ whatever depth results: with `t_max` the committed hit, one triangle per cycle, emitting `{hit, t, u, v, back_facing}`; bit-exact against SimX `ray_triangle`. - **`VX_rtu_xform`** — TLAS world→object transform. The instance record holds - the world→object matrix and the ray is transformed in the reference's op - order (rounded products, then `t + x + y + z`), so no inverse is taken. Always built: + the world→object matrix, so no inverse is taken; each object-ray component is + three dependent FMAs (`fma(z, m2, fma(y, m1, fma(x, m0, t)))`, the direction + seeded with `x·m0`), 18 FMA units, `3·FMA` deep. Always built: the CW-BVH walker descends `LEAF_INST` natively; only the flat walker's (`WIDTH = 0`) instancing loop is gated by `VX_CFG_RTU_TLAS_ENABLE`. - **`VX_rtu_recip`** — F32 reciprocal for `inv_d`, either a portable LUT+Newton diff --git a/hw/rtl/rtu/VX_rtu_xform.sv b/hw/rtl/rtu/VX_rtu_xform.sv index 3ac10a7823..fb9b0528bf 100644 --- a/hw/rtl/rtu/VX_rtu_xform.sv +++ b/hw/rtl/rtu/VX_rtu_xform.sv @@ -13,18 +13,15 @@ // VX_rtu_xform — world→object ray transform for a TLAS instance. Streams one // instance's world→object 3x4 matrix + world ray and emits the object-space ray -// after a fixed latency. +// after a fixed latency: three dependent FMAs per component, // -// The instance record carries the world→object matrix the source driver -// (lavapipe) builds, and the ray is transformed in the order it does: F32, -// every product rounded, then +// obj_ro[i] = fma(ro.z, m[i][2], fma(ro.y, m[i][1], fma(ro.x, m[i][0], m[i][3]))) +// obj_rd[i] = fma(rd.z, m[i][2], fma(rd.y, m[i][1], rd.x * m[i][0])) // -// obj_ro[i] = ((m[i][3] + ro.x*m[i][0]) + ro.y*m[i][1]) + ro.z*m[i][2] -// obj_rd[i] = (rd.x*m[i][0] + rd.y*m[i][1]) + rd.z*m[i][2] -// -// so the object ray matches it bit for bit for any affine instance (scale, -// shear included), with no inverse taken anywhere. Layout: m[i][j] = xform[4*i -// + j], row-major, translation in column 3. +// The instance record carries the world→object matrix itself, so no inverse +// is taken anywhere; the direction is not renormalised, so t is the same in +// both spaces. Layout: m[i][j] = xform[4*i + j], row-major, translation in +// column 3. Mirrors SimX rtu::world_to_object_ray. `include "VX_define.vh" @@ -48,136 +45,70 @@ module VX_rtu_xform import VX_gpu_pkg::*, VX_fpu_pkg::*, VX_rtu_pkg::*; #( output wire [2:0][31:0] obj_rd // object-space ray direction ); localparam F = LATENCY_FMA; - localparam LATENCY = 4 * F; // products, then three dependent adds + localparam LATENCY = 3 * F; // three dependent FMAs localparam [INST_FMT_BITS-1:0] FMT_ADD = 2'b00; - // ── @0 → @F: every product, rounded ─────────────────────────────── - wire [2:0][2:0][31:0] po, pd; // [row][column] - for (genvar i = 0; i < 3; ++i) begin : g_row - for (genvar j = 0; j < 3; ++j) begin : g_col - VX_fma_unit #( - .USE_DSP (`VX_CFG_RTU_USE_DSP), - .LATENCY (F), - .SUBNORM_ENABLE (0), - .EXCEPT_ENABLE (1) - ) fmul_o ( - .clk (clk), - .reset (reset), - .enable (enable), - .mask (1'b1), - .op_type (INST_FPU_MUL), - .fmt (FMT_ADD), - .frm (INST_FRM_RNE), - .dataa (ro[j]), - .datab (xform[4*i + j]), - .datac ('0), - .result (po[i][j]), - `UNUSED_PIN (fflags) - ); - VX_fma_unit #( - .USE_DSP (`VX_CFG_RTU_USE_DSP), - .LATENCY (F), - .SUBNORM_ENABLE (0), - .EXCEPT_ENABLE (1) - ) fmul_d ( - .clk (clk), - .reset (reset), - .enable (enable), - .mask (1'b1), - .op_type (INST_FPU_MUL), - .fmt (FMT_ADD), - .frm (INST_FRM_RNE), - .dataa (rd[j]), - .datab (xform[4*i + j]), - .datac ('0), - .result (pd[i][j]), - `UNUSED_PIN (fflags) - ); + // column j of the matrix and the ray's j component, held until step j + wire [2:0][2:0][31:0] col; // [j][row] + for (genvar j = 0; j < 3; ++j) begin : g_col + for (genvar i = 0; i < 3; ++i) begin : g_row + assign col[j][i] = xform[4*i + j]; end end - - wire [2:0][31:0] tr_d; // translation column @F - VX_shift_register #( - .DATAW (3*32), - .DEPTH (F) - ) sr_tr ( - .clk (clk), - .reset (reset), - .enable (enable), - .data_in ({xform[11], xform[7], xform[3]}), - .data_out (tr_d) - ); - - // later addends held until their add issues - wire [2:0][31:0] po1_d, po2_d, pd2_d; // po[.][1] @2F, po[.][2] @3F, pd[.][2] @2F + wire [2:0][31:0] col1_d, col2_d; + wire [1:0][31:0] ray1_d, ray2_d; // {rd, ro} component y, then z VX_shift_register #( - .DATAW (2*3*32), + .DATAW (3*32 + 2*32), .DEPTH (F) - ) sr_p1 ( + ) sr_step1 ( .clk (clk), .reset (reset), .enable (enable), - .data_in ({po[2][1], po[1][1], po[0][1], pd[2][2], pd[1][2], pd[0][2]}), - .data_out ({po1_d, pd2_d}) + .data_in ({col[1], rd[1], ro[1]}), + .data_out ({col1_d, ray1_d}) ); VX_shift_register #( - .DATAW (3*32), + .DATAW (3*32 + 2*32), .DEPTH (2 * F) - ) sr_p2 ( + ) sr_step2 ( .clk (clk), .reset (reset), .enable (enable), - .data_in ({po[2][2], po[1][2], po[0][2]}), - .data_out (po2_d) + .data_in ({col[2], rd[2], ro[2]}), + .data_out ({col2_d, ray2_d}) ); - // ── the dependent adds, one per F ───────────────────────────────── - wire [2:0][31:0] o1, o2, d1, d2; - for (genvar i = 0; i < 3; ++i) begin : g_sum - VX_fma_unit #(.USE_DSP (`VX_CFG_RTU_USE_DSP), .LATENCY (F), .SUBNORM_ENABLE (0), .EXCEPT_ENABLE (1)) fadd_o1 ( - .clk (clk), .reset (reset), .enable (enable), .mask (1'b1), - .op_type (INST_FPU_ADD), .fmt (FMT_ADD), .frm (INST_FRM_RNE), - .dataa (tr_d[i]), .datab (po[i][0]), .datac ('0), - .result (o1[i]), `UNUSED_PIN (fflags) - ); - VX_fma_unit #(.USE_DSP (`VX_CFG_RTU_USE_DSP), .LATENCY (F), .SUBNORM_ENABLE (0), .EXCEPT_ENABLE (1)) fadd_o2 ( - .clk (clk), .reset (reset), .enable (enable), .mask (1'b1), - .op_type (INST_FPU_ADD), .fmt (FMT_ADD), .frm (INST_FRM_RNE), - .dataa (o1[i]), .datab (po1_d[i]), .datac ('0), - .result (o2[i]), `UNUSED_PIN (fflags) - ); - VX_fma_unit #(.USE_DSP (`VX_CFG_RTU_USE_DSP), .LATENCY (F), .SUBNORM_ENABLE (0), .EXCEPT_ENABLE (1)) fadd_o3 ( - .clk (clk), .reset (reset), .enable (enable), .mask (1'b1), - .op_type (INST_FPU_ADD), .fmt (FMT_ADD), .frm (INST_FRM_RNE), - .dataa (o2[i]), .datab (po2_d[i]), .datac ('0), - .result (obj_ro[i]), `UNUSED_PIN (fflags) - ); - VX_fma_unit #(.USE_DSP (`VX_CFG_RTU_USE_DSP), .LATENCY (F), .SUBNORM_ENABLE (0), .EXCEPT_ENABLE (1)) fadd_d1 ( - .clk (clk), .reset (reset), .enable (enable), .mask (1'b1), - .op_type (INST_FPU_ADD), .fmt (FMT_ADD), .frm (INST_FRM_RNE), - .dataa (pd[i][0]), .datab (pd[i][1]), .datac ('0), - .result (d1[i]), `UNUSED_PIN (fflags) - ); - VX_fma_unit #(.USE_DSP (`VX_CFG_RTU_USE_DSP), .LATENCY (F), .SUBNORM_ENABLE (0), .EXCEPT_ENABLE (1)) fadd_d2 ( - .clk (clk), .reset (reset), .enable (enable), .mask (1'b1), - .op_type (INST_FPU_ADD), .fmt (FMT_ADD), .frm (INST_FRM_RNE), - .dataa (d1[i]), .datab (pd2_d[i]), .datac ('0), - .result (d2[i]), `UNUSED_PIN (fflags) - ); + // [row][0: origin, 1: direction] accumulators after each step + wire [2:0][1:0][31:0] acc0, acc1, acc2; + for (genvar i = 0; i < 3; ++i) begin : g_row + for (genvar k = 0; k < 2; ++k) begin : g_ray + // step 0: origin seeds with the translation, direction with -0 + // (so the product keeps its own zero sign) + VX_fma_unit #(.USE_DSP (`VX_CFG_RTU_USE_DSP), .LATENCY (F), .SUBNORM_ENABLE (0), .EXCEPT_ENABLE (1)) fma_s0 ( + .clk (clk), .reset (reset), .enable (enable), .mask (1'b1), + .op_type (INST_FPU_MADD), .fmt (FMT_ADD), .frm (INST_FRM_RNE), + .dataa ((k == 0) ? ro[0] : rd[0]), .datab (col[0][i]), + .datac ((k == 0) ? xform[4*i + 3] : 32'h80000000), + .result (acc0[i][k]), `UNUSED_PIN (fflags) + ); + VX_fma_unit #(.USE_DSP (`VX_CFG_RTU_USE_DSP), .LATENCY (F), .SUBNORM_ENABLE (0), .EXCEPT_ENABLE (1)) fma_s1 ( + .clk (clk), .reset (reset), .enable (enable), .mask (1'b1), + .op_type (INST_FPU_MADD), .fmt (FMT_ADD), .frm (INST_FRM_RNE), + .dataa (ray1_d[k]), .datab (col1_d[i]), .datac (acc0[i][k]), + .result (acc1[i][k]), `UNUSED_PIN (fflags) + ); + VX_fma_unit #(.USE_DSP (`VX_CFG_RTU_USE_DSP), .LATENCY (F), .SUBNORM_ENABLE (0), .EXCEPT_ENABLE (1)) fma_s2 ( + .clk (clk), .reset (reset), .enable (enable), .mask (1'b1), + .op_type (INST_FPU_MADD), .fmt (FMT_ADD), .frm (INST_FRM_RNE), + .dataa (ray2_d[k]), .datab (col2_d[i]), .datac (acc1[i][k]), + .result (acc2[i][k]), `UNUSED_PIN (fflags) + ); + end + assign obj_ro[i] = acc2[i][0]; + assign obj_rd[i] = acc2[i][1]; end - VX_shift_register #( - .DATAW (3*32), - .DEPTH (F) - ) sr_d ( - .clk (clk), - .reset (reset), - .enable (enable), - .data_in (d2), - .data_out (obj_rd) - ); - // ── valid + tag pipe, sized to the whole datapath latency ───────── reg [LATENCY-1:0] valid_pipe_r; always_ff @(posedge clk) begin diff --git a/sim/simx/rtu/rtu_isect.cpp b/sim/simx/rtu/rtu_isect.cpp index 3dde034c81..8f4d475954 100644 --- a/sim/simx/rtu/rtu_isect.cpp +++ b/sim/simx/rtu/rtu_isect.cpp @@ -125,15 +125,8 @@ void world_to_object_ray(const float wto[12], float ro_out[3], float rd_out[3]) { for (int i = 0; i < 3; ++i) { const float* m = wto + 4 * i; - float o = m[3]; - o = o + ro[0] * m[0]; - o = o + ro[1] * m[1]; - o = o + ro[2] * m[2]; - float d = rd[0] * m[0]; - d = d + rd[1] * m[1]; - d = d + rd[2] * m[2]; - ro_out[i] = o; - rd_out[i] = d; + ro_out[i] = std::fma(ro[2], m[2], std::fma(ro[1], m[1], std::fma(ro[0], m[0], m[3]))); + rd_out[i] = std::fma(rd[2], m[2], std::fma(rd[1], m[1], rd[0] * m[0])); } } diff --git a/sim/simx/rtu/rtu_isect.h b/sim/simx/rtu/rtu_isect.h index 8dd3005eb2..d03be7727a 100644 --- a/sim/simx/rtu/rtu_isect.h +++ b/sim/simx/rtu/rtu_isect.h @@ -65,11 +65,10 @@ bool ray_box(const float rel_mn[3], const float rel_mx[3], const float inv[3], // ──────────────────────────────────────────────────────────────────── // Bring a world ray into an instance's object space with the instance -// record's world→object 3x4 row-major matrix m, in the Vulkan reference's -// (lavapipe) op order, every product rounded: +// record's world→object 3x4 row-major matrix m, as FMA chains: // -// ro_obj[i] = ((m[i][3] + ro.x*m[i][0]) + ro.y*m[i][1]) + ro.z*m[i][2] -// rd_obj[i] = (rd.x*m[i][0] + rd.y*m[i][1]) + rd.z*m[i][2] +// ro_obj[i] = fma(ro.z, m[i][2], fma(ro.y, m[i][1], fma(ro.x, m[i][0], m[i][3]))) +// rd_obj[i] = fma(rd.z, m[i][2], fma(rd.y, m[i][1], rd.x * m[i][0])) // // The direction is not renormalised, so t is the same in both spaces. // Mirrors VX_rtu_xform bit for bit. diff --git a/sim/simx/rtu/rtu_types.h b/sim/simx/rtu/rtu_types.h index 0b922dda06..07d0ecb695 100644 --- a/sim/simx/rtu/rtu_types.h +++ b/sim/simx/rtu/rtu_types.h @@ -304,10 +304,10 @@ constexpr uint32_t kRtuImageStatesPerRay = 1; // scene header constexpr uint32_t kRtuSetupLatency = 17; // reciprocal pipe depth constexpr uint32_t kRtuFdivLat = 17; // reciprocal pipe depth constexpr uint32_t kRtuLatencyFma = 9; // FMA pipe depth -// Per-instance transform latency = 4 * FMA pipe depth = 36: the products, then -// three dependent adds (VX_rtu_xform). Charged per TLAS instance descent in the -// SimX cost model. -constexpr uint32_t kRtuXformLatency = 36; // 4 * FMA pipe depth +// Per-instance transform latency = 3 * FMA pipe depth = 27: three dependent +// FMAs (VX_rtu_xform). Charged per TLAS instance descent in the SimX cost +// model. +constexpr uint32_t kRtuXformLatency = 27; // 3 * FMA pipe depth // TLAS instance record (64 B). Lives inline after the scene header for // "TLAS + inline BLAS" layout. From 7c6e037c4dc2ea8b41eafc6b2ca2e8272102916d Mon Sep 17 00:00:00 2001 From: Blaise Tine Date: Sat, 3 Oct 2026 13:17:10 -0700 Subject: [PATCH 25/31] fpu: vendor FMA maps FMUL to a*b + -0, not a*b + +0 xil_fma / acl_fmadd only compute a*b + c, so VX_fma_unit maps a multiply to a*b + c with c = +0. Under round-to-nearest -0 + +0 = +0, so every product that is -0 (a negative operand times a zero, or a negative product flushed to zero) came out +0 -- an IEEE 754 sign-of-zero error in fmul.s on the Xilinx/Altera FP IP path, and in every RTU PE multiply on the V80. -0 is the additive identity (x + -0 == x for every x, signed zeros included), so the remap adds -0 instead. The soft core (rtlsim, ASIC) was already correct. Co-Authored-By: Claude Opus 5.5 --- hw/rtl/fpu/VX_fma_unit.sv | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/hw/rtl/fpu/VX_fma_unit.sv b/hw/rtl/fpu/VX_fma_unit.sv index 9337a46f66..6195be219c 100644 --- a/hw/rtl/fpu/VX_fma_unit.sv +++ b/hw/rtl/fpu/VX_fma_unit.sv @@ -71,7 +71,7 @@ module VX_fma_unit import VX_gpu_pkg::*, VX_fpu_pkg::*; #( if (USE_VENDOR_IP) begin : g_vendor // xil_fma / acl_fmadd compute a*b+c, so the FMA-core opcodes are remapped: - // MUL : a*b + 0 + // MUL : a*b + -0 (-0 is the additive identity: a -0 product stays -0) // ADD/SUB : a*1.0 (+/-) b // MADD/NMADD : (+/-)a*b (+/-) c // The vendor IP rounds round-to-nearest-even only (frm is ignored). @@ -89,7 +89,7 @@ module VX_fma_unit import VX_gpu_pkg::*, VX_fpu_pkg::*; #( if (is_neg) begin // MUL a32 = dataa[31:0]; b32 = datab[31:0]; - c32 = '0; + c32 = 32'h80000000; end else begin // ADD/SUB a32 = dataa[31:0]; b32 = 32'h3f800000; // 1.0f From 98058d3713a323a1653312f3c0fcb71c9797adbd Mon Sep 17 00:00:00 2001 From: Blaise Tine Date: Sat, 3 Oct 2026 16:51:25 -0700 Subject: [PATCH 26/31] aved: clock the kernel at the rate its image was timed for A VRT design write does not touch the user clock, so a freshly programmed image ran at whatever rate the previous image had set: a 200 MHz Vortex image came up at 60 MHz left behind by a FireSim image (silently 3.3x slow), and the reverse case overclocks a design past its timing closure. On open, set the clock to the vbin's timed rate (getMaxFrequency) as XRT does from the xclbin on load, read it back, and refuse to run if it is above that rate. Co-Authored-By: Claude Opus 5.5 --- sw/runtime/aved/vortex.cpp | 15 +++++++++++++++ 1 file changed, 15 insertions(+) diff --git a/sw/runtime/aved/vortex.cpp b/sw/runtime/aved/vortex.cpp index d07c3b0558..08d431c41a 100644 --- a/sw/runtime/aved/vortex.cpp +++ b/sw/runtime/aved/vortex.cpp @@ -267,6 +267,21 @@ class vx_device { // Only the hardware platform has a slave bridge, so anything else needs // the explicit host-memory sync in cp_reg_write/cp_reg_read. sim_mode_ = (vrtDevice_.getPlatform() != vrt::Platform::HARDWARE); + // Like XRT's xclbin load, clock the kernel at the rate the image was + // timed for: a design write leaves the user clock at whatever the + // previous image set, which can be far above or below this one's. + if (!sim_mode_) { + const uint64_t timed_hz = vrtDevice_.getMaxFrequency(); + vrtDevice_.setFrequency(timed_hz); + const uint64_t clk_hz = vrtDevice_.getFrequency(); + if (clk_hz == 0 || clk_hz > timed_hz + timed_hz / 200) { + fprintf(stderr, "[VXDRV] Error: user clock reads %lu Hz, image is timed for %lu Hz\n", + (unsigned long)clk_hz, (unsigned long)timed_hz); + return -1; + } + printf("[VXDRV] kernel clock %lu Hz (image timed for %lu Hz)\n", + (unsigned long)clk_hz, (unsigned long)timed_hz); + } vrtKernel_ = vrt::Kernel(vrtDevice_, KERNEL_NAME); VRT_CATCH(-1) From 642420ee29fffd3031e2351b2a2075b3f82eb263 Mon Sep 17 00:00:00 2001 From: Blaise Tine Date: Sat, 3 Oct 2026 17:17:43 -0700 Subject: [PATCH 27/31] runtime: stage device transfers through a bounded host buffer cp_submit_mem_write/read allocated one CP-visible host staging buffer the size of the whole transfer. On the V80 a large upload (a 1.9 GB acceleration structure) asks VRT for a buffer its allocator rejects ("Invalid argument"), so the scene never reached device memory and every ray missed. Move the data through a 64 MB staging buffer chunk by chunk, as pageable copies do on other GPU runtimes. Co-Authored-By: Claude Opus 5.5 --- sw/runtime/common/device.cpp | 58 +++++++++++++++++++++--------------- 1 file changed, 34 insertions(+), 24 deletions(-) diff --git a/sw/runtime/common/device.cpp b/sw/runtime/common/device.cpp index a145d98ac9..0eee737e92 100644 --- a/sw/runtime/common/device.cpp +++ b/sw/runtime/common/device.cpp @@ -14,6 +14,7 @@ #include "scope.h" // vx_scope_drain — lossless SCOPE tap-ring drainer #endif +#include #include #include #include @@ -25,6 +26,9 @@ namespace vx { +// Upper bound on one CP-visible host staging buffer for a device transfer. +static constexpr uint64_t CP_STAGING_CHUNK = uint64_t(64) << 20; + // Resolve the pinned-region size: compile-time default // VX_CFG_VM_PINNED_REGION_SIZE, optionally overridden by the // VORTEX_VM_PINNED_SIZE env var (decimal bytes). Returns 0 when VM is @@ -974,25 +978,28 @@ vx_result_t Device::cp_submit_mem_write(uint64_t dev_dst, const void* host_src, uint64_t size, bool physical) { if (size == 0) return VX_SUCCESS; if (!host_src) return VX_ERR_INVALID_VALUE; - // Stage the payload into CP-visible host memory (a plain memcpy through - // the host pointer), then have the CP DMA it to device memory. `physical` + // Stage through a bounded CP-visible host buffer, one chunk at a time: + // a backend can cap a single host allocation well below a large upload + // (an acceleration structure runs to GBs). Each chunk is a plain memcpy + // through the host pointer, then a CP DMA to device memory. `physical` // (set for page-table writes) tells the CP DMA to skip VM translation. HostMem staging; - auto r = host_alloc(size, &staging); + auto r = host_alloc(std::min(size, CP_STAGING_CHUNK), &staging); if (r != VX_SUCCESS) return r; - std::memcpy(staging.host_ptr, host_src, size); - // Make the fill visible to the CP before the command that reads it can - // be fetched. On shadowing backends this is the ONLY push of this - // region: the backend's doorbell publish deliberately does not touch - // generic regions (a blanket publish can push a half-filled or stale - // shadow over device bytes another agent owns). - r = platform()->host_mem_push(staging.cp_addr); - if (r != VX_SUCCESS) { - host_free(staging.cp_addr); - return r; + auto src = static_cast(host_src); + for (uint64_t off = 0; off < size && r == VX_SUCCESS; off += CP_STAGING_CHUNK) { + const uint64_t n = std::min(size - off, CP_STAGING_CHUNK); + std::memcpy(staging.host_ptr, src + off, n); + // Make the fill visible to the CP before the command that reads it can + // be fetched. On shadowing backends this is the ONLY push of this + // region: the backend's doorbell publish deliberately does not touch + // generic regions (a blanket publish can push a half-filled or stale + // shadow over device bytes another agent owns). + r = platform()->host_mem_push(staging.cp_addr); + if (r == VX_SUCCESS) + r = cp_submit_mem_(CP_OPCODE_MEM_WRITE, dev_dst + off, + staging.cp_addr, n, physical); } - r = cp_submit_mem_(CP_OPCODE_MEM_WRITE, dev_dst, staging.cp_addr, size, - physical); host_free(staging.cp_addr); return r; } @@ -1001,22 +1008,25 @@ vx_result_t Device::cp_submit_mem_read(void* host_dst, uint64_t dev_src, uint64_t size, bool physical) { if (size == 0) return VX_SUCCESS; if (!host_dst) return VX_ERR_INVALID_VALUE; - // Have the CP DMA device->host into a CP-visible host staging buffer, - // then memcpy it back to the caller's pointer. + // Have the CP DMA device->host into a bounded CP-visible host staging + // buffer chunk by chunk (see cp_submit_mem_write), copying each back. HostMem staging; - auto r = host_alloc(size, &staging); + auto r = host_alloc(std::min(size, CP_STAGING_CHUNK), &staging); if (r != VX_SUCCESS) return r; - r = cp_submit_mem_(CP_OPCODE_MEM_READ, staging.cp_addr, dev_src, size, - physical); - if (r == VX_SUCCESS) { + auto dst = static_cast(host_dst); + for (uint64_t off = 0; off < size && r == VX_SUCCESS; off += CP_STAGING_CHUNK) { + const uint64_t n = std::min(size - off, CP_STAGING_CHUNK); + r = cp_submit_mem_(CP_OPCODE_MEM_READ, staging.cp_addr, dev_src + off, + n, physical); // The CP wrote the staging region; on a backend that shadows CP // memory the host copy is stale until pulled. The submit's Q_SEQNUM // poll has already fenced on the completion line, so the pull reads // settled bytes. - r = platform()->host_mem_pull(staging.cp_addr); + if (r == VX_SUCCESS) + r = platform()->host_mem_pull(staging.cp_addr); + if (r == VX_SUCCESS) + std::memcpy(dst + off, staging.host_ptr, n); } - if (r == VX_SUCCESS) - std::memcpy(host_dst, staging.host_ptr, size); host_free(staging.cp_addr); return r; } From 6d81dad7d509f951830b64c75e08a3247920f3ec Mon Sep 17 00:00:00 2001 From: Blaise Tine Date: Sat, 3 Oct 2026 17:17:43 -0700 Subject: [PATCH 28/31] runtime: a failed queued command fails the commands behind it An error from an asynchronous command only reached that command's own event, so a vx_enqueue_write without an event lost it, vx_queue_finish returned success, and the kernel launched behind it ran on device memory that was never written. As in-order queues do elsewhere (OpenCL's error-for-wait-list, CUDA's error at the next synchronize), the first failure is kept: later commands complete with it without executing, and finish() returns it, then clears it so the queue can be used again. Co-Authored-By: Claude Opus 5.5 --- sw/runtime/common/queue.cpp | 16 ++++++++++++++++ sw/runtime/common/vortex2_internal.h | 3 +++ 2 files changed, 19 insertions(+) diff --git a/sw/runtime/common/queue.cpp b/sw/runtime/common/queue.cpp index fd7f54e994..df63f7a9dc 100644 --- a/sw/runtime/common/queue.cpp +++ b/sw/runtime/common/queue.cpp @@ -90,9 +90,17 @@ void Queue::worker_loop() { uint64_t start_ns = submit_ns; uint64_t end_ns = submit_ns; + { + std::lock_guard g(cmd_mu_); + if (r == VX_SUCCESS && async_error_ != VX_SUCCESS) r = async_error_; + } if (r == VX_SUCCESS && cmd.work) { r = cmd.work(&start_ns, &end_ns); } + if (r != VX_SUCCESS) { + std::lock_guard g(cmd_mu_); + if (async_error_ == VX_SUCCESS) async_error_ = r; + } if (cmd.completion) { if (profiling_enabled()) { @@ -169,6 +177,14 @@ vx_result_t Queue::finish(uint64_t timeout_ns) { if (r != VX_SUCCESS) return r; r = to_event(ev)->wait(timeout_ns); to_event(ev)->release(); + if (r == VX_ERR_TIMEOUT) return r; + // Report the first failure since the last finish, then let the queue run + // again (the commands behind it completed with that error, unexecuted). + std::lock_guard g(cmd_mu_); + if (async_error_ != VX_SUCCESS) { + r = async_error_; + async_error_ = VX_SUCCESS; + } return r; } diff --git a/sw/runtime/common/vortex2_internal.h b/sw/runtime/common/vortex2_internal.h index e6c23a2cd8..ef803eefc7 100644 --- a/sw/runtime/common/vortex2_internal.h +++ b/sw/runtime/common/vortex2_internal.h @@ -765,6 +765,9 @@ class Queue : public RefCounted { std::condition_variable cmd_cv_; std::deque commands_; bool shutdown_ = false; + // First failure of an in-order command; later commands fail with it + // instead of running on incomplete state, until finish() reports it. + vx_result_t async_error_ = VX_SUCCESS; std::thread worker_; }; From 31a38324925b7a09be8d5e6dc2958dadfa09a656 Mon Sep 17 00:00:00 2001 From: Blaise Tine Date: Sat, 3 Oct 2026 18:45:21 -0700 Subject: [PATCH 29/31] aved: keep the CP staging aperture out of device memory HOST_TAG=HBM1 put the CP's m_axi_host on the HBM_AXI port whose 512 MB slice starts at +512 MB from the HBM base -- inside the 4 GB device-memory window that MEM_TAG=MEM exposes from that same base. VRT allocated the CP's DMA staging buffers there while Vortex's own allocator handed out the same addresses, so any workload whose device buffers grew past 512 MB was overwritten by its own uploads: LumiBench CAR/ROBOT rendered wrong pixels and PARK's corrupted BVH never finished traversing (hit the 30 min timeout). With the window kept clear (runtime experiment) all three render bit-identical to SimX and PARK_SH completes in 46 s. Use HBM8, which starts at +4 GB, past the last device byte, and refuse HBM0..HBM7 for HOST_TAG at parse time when MEM_TAG is MEM or HBM0. Co-Authored-By: Claude Opus 5.5 --- hw/syn/xilinx/aved/Makefile | 11 ++++++++++- hw/syn/xilinx/aved/platforms.mk | 9 ++++++++- 2 files changed, 18 insertions(+), 2 deletions(-) diff --git a/hw/syn/xilinx/aved/Makefile b/hw/syn/xilinx/aved/Makefile index c9b7c0cc96..f1471e46d8 100644 --- a/hw/syn/xilinx/aved/Makefile +++ b/hw/syn/xilinx/aved/Makefile @@ -128,7 +128,16 @@ HOST_TAG ?= HOST # is a full synthesis run to discover, and it has already cost one, so refuse # the tag at parse time rather than emitting the config. See platforms.mk. ifeq ($(HOST_TAG),HOST) -$(error HOST_TAG=HOST routes m_axi_host to the QDMA slave bridge, whose reads never complete on the compute shell; the CP would hang on its first ring fetch. Use HOST_TAG=HBM1 (the platforms.mk default)) +$(error HOST_TAG=HOST routes m_axi_host to the QDMA slave bridge, whose reads never complete on the compute shell; the CP would hang on its first ring fetch. Use HOST_TAG=HBM8 (the platforms.mk default)) +endif + +# MEM covers the 32-bit device space from the HBM base, i.e. HBM0..HBM7; a CP +# staging aperture there aliases device memory and large uploads overwrite the +# heap. See platforms.mk. +ifneq ($(filter HBM0 HBM1 HBM2 HBM3 HBM4 HBM5 HBM6 HBM7,$(HOST_TAG)),) +ifneq ($(filter MEM HBM0,$(MEM_TAG)),) +$(error HOST_TAG=$(HOST_TAG) lies inside the device-memory window of MEM_TAG=$(MEM_TAG) (HBM0..HBM7); CP staging would alias the Vortex heap. Use HBM8 or above) +endif endif # platforms.mk states the kernel clock in MHz; the linker config wants Hz. diff --git a/hw/syn/xilinx/aved/platforms.mk b/hw/syn/xilinx/aved/platforms.mk index 1a6e44b2ab..9a6b5b2a90 100644 --- a/hw/syn/xilinx/aved/platforms.mk +++ b/hw/syn/xilinx/aved/platforms.mk @@ -60,7 +60,14 @@ MEM_TAG = MEM # next, which cost a bitstream and several hours of hardware debugging. It # belongs here next to MEM_TAG, for the same reason and by the same mechanism: # stated before the Makefile's `HOST_TAG ?= HOST`, so plain `=` suffices. -HOST_TAG = HBM1 +# The aperture must lie OUTSIDE the device-memory window. Each HBM_AXI port +# addresses its own 512 MB slice of the stack, HBM at 0x40_0000_0000 + +# k*512 MB, and the MEM tag above exposes Vortex's whole 32-bit space from that +# same base -- so HBM0..HBM7 ARE device memory. With HBM1, VRT placed the CP's +# staging buffers at +512 MB, inside the heap: any scene whose buffers grew past +# 512 MB was overwritten by its own upload (wrong images; corrupted BVHs that +# traversal never finished). HBM8 starts at +4 GB, past the last device byte. +HOST_TAG = HBM8 # Kernel clock target (MHz). The linker also accepts a frequency request; # the runtime can retune within the platform's supported range. From f70cff0bc7b835b81dd360317861c5a6133b9753 Mon Sep 17 00:00:00 2001 From: Blaise Tine Date: Sat, 3 Oct 2026 21:53:54 -0700 Subject: [PATCH 30/31] tex: weight bilinear taps by f/256, rounded, per Vulkan texel filtering The bilinear tap fraction is the low 8 bits of the scaled texel coordinate (x0s & 0xff), i.e. frac*256. Vulkan's texel filtering with subTexelPrecisionBits = 8 defines the tap weight as f/256, but the 8-bit blend (Lerp8888 in the shared sampler and VX_tex_lerp's FRAC_SCALE=255 path in the RTL) normalised it as f/255. A full-weight tap was over-weighted and the bilinear result drifted up to 2.4 LSB from the Vulkan formula (>1 LSB in ~2.9% of random cases). The float filter and the trilinear level blend already used /256, so the unit was also internally inconsistent. Both implementations now compute (a*(256-f) + b*f + 128) >> 8: exact weights, round to nearest, within 1/2 LSB per lerp. The packed lane peaks below 2^16, so Lerp8888 stays carry-free. VX_tex_lerp keeps one shift-only datapath (no /255 correction stage, same 3-cycle latency); the level blend instance keeps its existing truncation (ROUND=0), matching TexLodLerp. The OM unorm blends, which are legitimately /255, are untouched. New hw/unittest/tex_lerp checks VX_tex_lerp exhaustively (all 2^24 inputs, random stalls) against Lerp8888 / TexLodLerp and the exact Vulkan blend, wired into hw/unittest and a hw-tex-lerp CI lane. The box/evilskull gfx_draw3d goldens (box_ref_128, evilskull_ref_32, evilskull_ref_128) move by 1-2 LSB on 2-3% of pixels; FF (SimX, RTL) and the SW sampler agree bit-for-bit on the new images. Goldens are left for human regeneration. Co-Authored-By: Claude Opus 5.5 --- ci/testcases/unittest.yaml | 7 ++ docs/designs/texture_sampler_architecture.md | 12 +-- hw/rtl/tex/VX_tex_lerp.sv | 61 +++++------- hw/rtl/tex/VX_tex_sampler.sv | 10 +- hw/unittest/Makefile | 3 + hw/unittest/tex_lerp/Makefile | 19 ++++ hw/unittest/tex_lerp/VX_tex_lerp_top.sv | 53 +++++++++++ hw/unittest/tex_lerp/main.cpp | 99 ++++++++++++++++++++ sw/common/vx_gfx_abi.h | 8 +- 9 files changed, 218 insertions(+), 54 deletions(-) create mode 100644 hw/unittest/tex_lerp/Makefile create mode 100644 hw/unittest/tex_lerp/VX_tex_lerp_top.sv create mode 100644 hw/unittest/tex_lerp/main.cpp diff --git a/ci/testcases/unittest.yaml b/ci/testcases/unittest.yaml index 5a3893e535..89c8ba08e8 100644 --- a/ci/testcases/unittest.yaml +++ b/ci/testcases/unittest.yaml @@ -31,3 +31,10 @@ tests: via: script run: "make -C hw/unittest run-tcu-dsp" touches: [hw/rtl/tcu, hw/unittest/tcu_fedp] + # Texel blend (VX_tex_lerp, both forms the sampler uses) against the C model + # the SW sampler and SimX share, and the exact Vulkan f/256 blend, over every + # input. + - id: hw-tex-lerp + via: script + run: "make -C hw/unittest/tex_lerp run" + touches: [hw/rtl/tex, hw/unittest/tex_lerp, sw/common/vx_gfx_abi.h, sw/common/gfx_frag_tex.h] diff --git a/docs/designs/texture_sampler_architecture.md b/docs/designs/texture_sampler_architecture.md index 46ebcc6f86..ee24885894 100644 --- a/docs/designs/texture_sampler_architecture.md +++ b/docs/designs/texture_sampler_architecture.md @@ -256,15 +256,15 @@ stays a plain 4-byte-word interface. colour here instead (§3.1). 2. **U lerps**: per lane and level, 8 `VX_tex_lerp` instances (4 channels × {low, high} texel pairs), each a 3-cycle fixed-point datapath computing - `(s + (s >> 8)) >> 8` with `s = a·(255−f) + b·f + 0x80` — the exact - divide-by-255 rounding, not a plain shift. + `(a·(256−f) + b·f + 0x80) >> 8` — the weight is `f/256`, the 8-bit + subtexel fraction Vulkan's texel filtering defines + (`subTexelPrecisionBits = 8`), rounded to nearest. 3. **V lerp**: 4 more lerps per level blend the two U results, another 3 cycles. 4. **Level lerp**: 4 final lerps blend the two levels' texels by the - request's lod fraction — a fraction of **256** that truncates, the form - the software sampler blends levels in, where a tap weight is a fraction - of 255 (§2.3). A single-level sample carries weight 0 and passes level 0 - through. + request's lod fraction — also a fraction of 256, but truncated rather + than rounded, the form the software sampler blends levels in (§2.3). A + single-level sample carries weight 0 and passes level 0 through. The whole sampler is ~10 cycles fixed latency, one request per cycle throughput, per-channel 8-bit arithmetic — no floating-point anywhere (the diff --git a/hw/rtl/tex/VX_tex_lerp.sv b/hw/rtl/tex/VX_tex_lerp.sv index 33c4c5ed67..906b26ec4f 100644 --- a/hw/rtl/tex/VX_tex_lerp.sv +++ b/hw/rtl/tex/VX_tex_lerp.sv @@ -17,12 +17,10 @@ module VX_tex_lerp #( parameter LATENCY = 3, - // Denominator the weight is normalized against. A bilinear tap weight is a - // fraction of 255, matching the packed-colour blend the software sampler - // uses for the same taps; a mip-level weight is a fraction of 256. The two - // round differently by up to one count, so a sample filtered on this unit - // only reproduces the software sampler if each weight keeps its own form. - parameter FRAC_SCALE = 255 + // The weight is frac/256 (8 subtexel fraction bits). A bilinear tap blend + // rounds to nearest; the mip-level blend truncates, as the software + // sampler's level blend does, so both paths keep producing the same texel. + parameter ROUND = 1 ) ( input wire clk, input wire reset, @@ -34,45 +32,28 @@ module VX_tex_lerp #( ); `UNUSED_VAR (reset) `STATIC_ASSERT(LATENCY == 3, ("invalid value")) - `STATIC_ASSERT(FRAC_SCALE == 255 || FRAC_SCALE == 256, ("invalid value")) + `STATIC_ASSERT(ROUND == 0 || ROUND == 1, ("invalid value")) - if (FRAC_SCALE == 256) begin : g_scale_256 - reg [15:0] p1, p2; - reg [15:0] sum; - reg [7:0] res; - // The blend is a fraction of 256, so the result is the high byte and the - // low one is the remainder this form truncates rather than rounds. - `UNUSED_VAR (sum[7:0]) + localparam [15:0] BIAS = ROUND ? 16'h80 : 16'h0; - wire [8:0] sub = (9'h100 - 9'(frac)); + reg [15:0] p1, p2; + reg [15:0] sum; + reg [7:0] res; + // The result is the high byte; the low one is the discarded remainder. + `UNUSED_VAR (sum[7:0]) - always @(posedge clk) begin - if (enable) begin - p1 <= 16'(in1 * sub); - p2 <= 16'(in2 * frac); - sum <= p1 + p2; - res <= sum[15:8]; - end - end - - assign out = res; - end else begin : g_scale_255 - reg [15:0] p1, p2; - reg [15:0] sum; - reg [7:0] res; - - wire [7:0] sub = (8'hff - frac); + wire [8:0] sub = (9'h100 - 9'(frac)); - always @(posedge clk) begin - if (enable) begin - p1 <= in1 * sub; - p2 <= in2 * frac; - sum <= p1 + p2 + 16'h80; - res <= 8'((sum + (sum >> 8)) >> 8); - end + // 255*256 + 128 < 2^16: the 16-bit accumulator cannot overflow. + always @(posedge clk) begin + if (enable) begin + p1 <= 16'(in1 * sub); + p2 <= 16'(in2 * frac); + sum <= p1 + p2 + BIAS; + res <= sum[15:8]; end - - assign out = res; end + assign out = res; + endmodule diff --git a/hw/rtl/tex/VX_tex_sampler.sv b/hw/rtl/tex/VX_tex_sampler.sv index 4808363136..9009033135 100644 --- a/hw/rtl/tex/VX_tex_sampler.sv +++ b/hw/rtl/tex/VX_tex_sampler.sv @@ -161,14 +161,14 @@ module VX_tex_sampler import VX_gpu_pkg::*, VX_tex_pkg::*; #( .data_out ({valid_s2, req_tag_s2, lodfrac_s2}) ); - // Blend the two levels. The weight is a fraction of 256 rather than of 255, - // which is the form the software sampler blends levels in; a sample that - // moves between the two paths then keeps the same value. + // Blend the two levels. The level weight truncates rather than rounds, + // the form the software sampler blends levels in; a sample that moves + // between the two paths then keeps the same value. for (genvar i = 0; i < NUM_LANES; ++i) begin : g_tex_lerp_LOD for (genvar j = 0; j < 4; ++j) begin : g_j VX_tex_lerp #( - .LATENCY (3), - .FRAC_SCALE (256) + .LATENCY (3), + .ROUND (0) ) tex_lerp_lod ( .clk (clk), .reset(reset), diff --git a/hw/unittest/Makefile b/hw/unittest/Makefile index bca7effb13..0387f2a267 100644 --- a/hw/unittest/Makefile +++ b/hw/unittest/Makefile @@ -27,6 +27,7 @@ all: $(MAKE) -C fsqrt_unit $(MAKE) -C fcvt_unit $(MAKE) -C fdivsqrt_unit + $(MAKE) -C tex_lerp run: $(MAKE) -C generic_queue run @@ -57,6 +58,7 @@ run: $(MAKE) -C fsqrt_unit run $(MAKE) -C fcvt_unit run $(MAKE) -C fdivsqrt_unit run + $(MAKE) -C tex_lerp run # FPU arithmetic units only — executed (not just built) so CI gates correctness. run-fpu: @@ -151,3 +153,4 @@ clean: $(MAKE) -C fsqrt_unit clean $(MAKE) -C fcvt_unit clean $(MAKE) -C fdivsqrt_unit clean + $(MAKE) -C tex_lerp clean diff --git a/hw/unittest/tex_lerp/Makefile b/hw/unittest/tex_lerp/Makefile new file mode 100644 index 0000000000..6e90f797ae --- /dev/null +++ b/hw/unittest/tex_lerp/Makefile @@ -0,0 +1,19 @@ +ROOT_DIR := $(realpath ../../..) +include $(ROOT_DIR)/config.mk + +PROJECT := tex_lerp + +RTL_DIR := $(VORTEX_HOME)/hw/rtl +SRC_DIR := $(VORTEX_HOME)/hw/unittest/$(PROJECT) + +CXXFLAGS := -I$(SRC_DIR) -I$(VORTEX_HOME)/hw/unittest/common -I$(SW_COMMON_DIR) +CXXFLAGS += -I$(ROOT_DIR)/sw -I$(THIRD_PARTY_DIR) + +SRCS := $(SRC_DIR)/main.cpp + +RTL_INCLUDE := -I$(ROOT_DIR)/sw -I$(RTL_DIR) -I$(RTL_DIR)/libs -I$(RTL_DIR)/tex -I$(SRC_DIR) +VL_FLAGS += -I$(ROOT_DIR)/hw + +TOP := VX_tex_lerp_top + +include ../common.mk diff --git a/hw/unittest/tex_lerp/VX_tex_lerp_top.sv b/hw/unittest/tex_lerp/VX_tex_lerp_top.sv new file mode 100644 index 0000000000..d6570a7173 --- /dev/null +++ b/hw/unittest/tex_lerp/VX_tex_lerp_top.sv @@ -0,0 +1,53 @@ +// Copyright © 2019-2023 +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +`include "VX_platform.vh" + +// Both VX_tex_lerp forms the sampler instantiates, on shared inputs. +module VX_tex_lerp_top ( + input wire clk, + input wire reset, + input wire enable, + input wire [7:0] in1, + input wire [7:0] in2, + input wire [7:0] frac, + output wire [7:0] out_round, + output wire [7:0] out_trunc +); + VX_tex_lerp #( + .LATENCY (3), + .ROUND (1) + ) lerp_round ( + .clk (clk), + .reset (reset), + .enable (enable), + .in1 (in1), + .in2 (in2), + .frac (frac), + .out (out_round) + ); + + VX_tex_lerp #( + .LATENCY (3), + .ROUND (0) + ) lerp_trunc ( + .clk (clk), + .reset (reset), + .enable (enable), + .in1 (in1), + .in2 (in2), + .frac (frac), + .out (out_trunc) + ); + +endmodule diff --git a/hw/unittest/tex_lerp/main.cpp b/hw/unittest/tex_lerp/main.cpp new file mode 100644 index 0000000000..c264cb961e --- /dev/null +++ b/hw/unittest/tex_lerp/main.cpp @@ -0,0 +1,99 @@ +// VX_tex_lerp against the sampler's C model (vx_gfx_abi.h Lerp8888 for the +// bilinear taps, gfx_frag_tex.h TexLodLerp for the level blend) and against the +// exact Vulkan blend a*(1-f/256) + b*f/256, over every (in1, in2, frac). + +#include +#include +#include +#include +#include "VVX_tex_lerp_top.h" +#include "verilated.h" +#include + +static VVX_tex_lerp_top* dut; + +static void clock_cycle() { + dut->clk = 0; dut->eval(); + dut->clk = 1; dut->eval(); +} + +struct Vec { uint32_t a, b, f; }; + +static int errors = 0; + +static void check(const Vec& v, uint32_t rnd, uint32_t trn) { + using vortex::graphics::Lerp8888; + // the packed model, in both lanes it blends + uint32_t m_lo = Lerp8888(v.a, v.b, v.f) & 0xff; + uint32_t m_hi = (Lerp8888(v.a << 16, v.b << 16, v.f) >> 16) & 0xff; + uint32_t m_lod = gfx_tex::TexLodLerp(v.a, v.b, v.f) & 0xff; + // exact blend scaled by 256 + int32_t e256 = (int32_t)(v.a * (256 - v.f) + v.b * v.f); + bool rnd_exact = 2 * std::abs((int32_t)rnd * 256 - e256) <= 256; + bool trn_exact = (int32_t)trn * 256 <= e256 && e256 < ((int32_t)trn + 1) * 256; + if (rnd != m_lo || rnd != m_hi || trn != m_lod || !rnd_exact || !trn_exact) { + if (errors < 16) + std::printf("MISMATCH a=%u b=%u f=%u: rtl round=%u trunc=%u, model lo=%u hi=%u lod=%u, exact=%.4f\n", + v.a, v.b, v.f, rnd, trn, m_lo, m_hi, m_lod, e256 / 256.0); + ++errors; + } +} + +int main(int argc, char** argv) { + Verilated::commandArgs(argc, argv); + dut = new VVX_tex_lerp_top; + dut->reset = 1; + dut->enable = 1; + dut->in1 = dut->in2 = dut->frac = 0; + for (int i = 0; i < 4; ++i) clock_cycle(); + dut->reset = 0; + + const uint32_t LATENCY = 3; + std::deque inflight; + uint64_t checked = 0; + + // A stall cycle (enable low) must hold every stage; the results then still + // pair with their inputs after LATENCY enabled cycles. + srand(1); + auto step = [&](const Vec& v) { + if ((rand() % 16) == 0) { + dut->enable = 0; + clock_cycle(); + } + dut->enable = 1; + dut->in1 = v.a; dut->in2 = v.b; dut->frac = v.f; + inflight.push_back(v); + clock_cycle(); + if (inflight.size() == LATENCY) { + check(inflight.front(), dut->out_round, dut->out_trunc); + inflight.pop_front(); + ++checked; + } + }; + + // directed: pass-through at f=0, full-weight end of the range, the half + // point where round and truncate split + const Vec directed[] = { + {0, 255, 0}, {255, 0, 0}, {0, 255, 255}, {255, 0, 255}, {255, 255, 255}, + {8, 10, 128}, {0, 1, 128}, {1, 0, 128}, {200, 100, 1}, {100, 200, 254}, + }; + for (auto& v : directed) step(v); + + // exhaustive + for (uint32_t f = 0; f < 256; ++f) + for (uint32_t a = 0; a < 256; ++a) + for (uint32_t b = 0; b < 256; ++b) { + step({a, b, f}); + } + + for (uint32_t i = 1; i < LATENCY; ++i) step({0, 0, 0}); // drain + dut->final(); + delete dut; + + if (errors) { + std::printf("FAILED: %d mismatches over %llu vectors\n", errors, (unsigned long long)checked); + return 1; + } + std::printf("PASSED: %llu vectors\n", (unsigned long long)checked); + return 0; +} diff --git a/sw/common/vx_gfx_abi.h b/sw/common/vx_gfx_abi.h index ae7b639844..4c1184a1d6 100644 --- a/sw/common/vx_gfx_abi.h +++ b/sw/common/vx_gfx_abi.h @@ -250,10 +250,12 @@ static inline uint32_t Pack8888(uint32_t lo, uint32_t hi) { return (hi << 8) | lo; } +// Texel blend by an 8-bit subtexel fraction: the weight is f/256 (Vulkan +// subTexelPrecisionBits = 8), rounded to nearest. A lane peaks at +// 255*256 + 128 < 2^16, so the two packed lanes never carry into each other. static inline uint32_t Lerp8888(uint32_t a, uint32_t b, uint32_t f) { - uint32_t p = a * (0xff - f) + b * f + 0x00800080; - uint32_t q = (p >> 8) & 0x00ff00ff; - return ((p + q) >> 8) & 0x00ff00ff; + uint32_t p = a * (0x100 - f) + b * f + 0x00800080; + return (p >> 8) & 0x00ff00ff; } } // namespace graphics From 833a108049b69004f604bd771066f804e0ee74e2 Mon Sep 17 00:00:00 2001 From: Blaise Tine Date: Sun, 4 Oct 2026 01:18:12 -0700 Subject: [PATCH 31/31] vxbin: read kernel entries from readelf columns, not a decimal-size pattern readelf -s -W prints a symbol Size of 100000 or more in hex (0x18794). The symbol regexes required a decimal size, so a kernel_main of 100 KB or more matched nothing: the .vxbin got no VXSYMTAB footer, Module::load_bytes fell back to "main" at min_vma, which is __vx_cta_entry itself, and every warp re-entered its own startup code forever -- a silent hang on SimX. Split each readelf row into its columns instead and take Value and Name, for both the kernel entries and the _edata/_end lookups. Verified: a 100,244-byte probe kernel that hung now runs to completion on SimX; 41 ordinary kernels (regression tests and vortexpipe shaders) produce byte-identical .vxbin files; demo passes. Co-Authored-By: Claude Opus 5.5 --- sw/kernel/scripts/vxbin.py | 47 ++++++++++++++++++++------------------ 1 file changed, 25 insertions(+), 22 deletions(-) diff --git a/sw/kernel/scripts/vxbin.py b/sw/kernel/scripts/vxbin.py index dae8d82bfc..3ca24a0853 100755 --- a/sw/kernel/scripts/vxbin.py +++ b/sw/kernel/scripts/vxbin.py @@ -49,18 +49,26 @@ def get_vma_size(elf_file): print("Failed to calculate vma size due to an error: {}".format(str(e))) sys.exit(-1) +def read_symbols(elf_file): + # (value, name) for each `readelf -s -W` row. Split columns rather than + # pattern-match them: readelf prints a Size of 100000 or more in hex. + cmd = ['readelf', '-s', '-W', elf_file] + output = subprocess.check_output(cmd, universal_newlines=True) + symbols = [] + for line in output.splitlines(): + cols = line.split() + if len(cols) == 8 and cols[0][:-1].isdigit() and cols[0].endswith(':'): + symbols.append((int(cols[1], 16), cols[7])) + return symbols + def get_symbol(elf_file, name): # Read a symbol value from the ELF. We use _edata as the start of BSS and # _end as the end of BSS so runtime_size covers the full RW region (the # linker's DATA_SEGMENT_ALIGN can push _edata/_end past the end of the last # LOAD segment when the kernel has little/no data or BSS). - cmd = ['readelf', '-s', '-W', elf_file] - output = subprocess.check_output(cmd, universal_newlines=True) - regex = re.compile(r'\s*\d+:\s+([0-9a-fA-F]+)\s+\d+\s+\S+\s+\S+\s+\S+\s+\S+\s+' + re.escape(name) + r'$') - for line in output.splitlines(): - match = regex.match(line) - if match: - return int(match.group(1), 16) + for value, sym in read_symbols(elf_file): + if sym == name: + return value print("Error: {} symbol not found in {}".format(name, elf_file)) sys.exit(-1) @@ -69,23 +77,18 @@ def get_kernel_entries(elf_file): # "__vx_kentry_" alias per vortex.kernel function; the runtime's # vx_module_get_kernel() resolves to its address. The conventional # single-kernel entry "kernel_main" is exposed under the public name "main". - cmd = ['readelf', '-s', '-W', elf_file] - output = subprocess.check_output(cmd, universal_newlines=True) - regex = re.compile( - r'\s*\d+:\s+([0-9a-fA-F]+)\s+\d+\s+\S+\s+\S+\s+\S+\s+\S+\s+' - r'__vx_kentry_(\S+)$') entries = [] seen = set() - for line in output.splitlines(): - match = regex.match(line) - if match: - name = match.group(2) - if name == 'kernel_main': - name = 'main' - if name in seen: - continue - seen.add(name) - entries.append((name, int(match.group(1), 16))) + for value, sym in read_symbols(elf_file): + if not sym.startswith('__vx_kentry_'): + continue + name = sym[len('__vx_kentry_'):] + if name == 'kernel_main': + name = 'main' + if name in seen: + continue + seen.add(name) + entries.append((name, value)) return entries def build_symtab_footer(entries):