diff --git a/AGENTS.md b/AGENTS.md index b7df8d9..65cf78c 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -61,3 +61,64 @@ GDDR热参数来阻断封装thermal_only;仍保留physical GDDR身份和系统 科学阈值、不绕安全启动器。P2未通过时P3/P4仅可报告显式工程fixture,不造 MODEL_FREEZE。真实授权见工作区eq3_thermal/plans/p2-p3-p4-campaign-v1。 围绕主线推进;逐字节等价性用于首次链路或相关实现变化,不反复全验未变工件。 + +## EQ3 最小侵入修复规则(用户明确要求) + +接口优先,默认关闭,最小侵入。 +任何影响调度、事件顺序、请求/完成生命周期、资源所有权、 +公共ABI、PTX/TMA/future/cache、既有时序或模块职责的重构, +实施前必须提交最小方案并取得用户明确确认。 +不能用代码少、位于独立目录或为了修bug作为豁免。 +用户未回复不视为同意;待确认分支暂存,其他工作继续。 + +本轮 EQ3-MINIMAL-REPAIR-v1 已授权已复现局部修复、语义透明且默认关闭的 +接口适配和固定 CPU 验证;不是重构、科学方法变更、GPU、push/merge 的授权。 + +## EQ3-DECISION-EXECUTION-v2(当前明确阶段授权) + +用户已采纳任务书D1–D6并授权连续执行:验收v2与legacy_v1并列、原train完整 +1mm参考及现有后端数值验证、统一alpha域内分支、新可选升级优先策略与九点 +CPU矩阵、默认关闭只读phase接口及既有gate真实CPU消费者。见 +eq3_thermal/plans/decision-execution-v2/USER_TASKBOOK.md及user-source.md。 +最新补充:取消固定内存/时限/磁盘硬边界,按每点实际需求、机器余量与对其它 +服务影响合理评估并记录资源计划/停止条件;不再因原16/12GiB、3600/600s、 +4/40GiB阈值机械阻断。仍串行CPU求解/实验、OMP/BLAS1,编译协调不争抢。 +既有安全启动器须消费本阶段实际资源计划;保留系统余量和异常停止机制。 +这不是GPU/云/新物性/新求解器/ROM/ABI/调度/所有权/真实维护生命周期重构授权。 +已定范围不逐点请示;盲测依赖、原400K域、原结果及最小侵入确认规则仍生效。 + +## 基础 CPU 全链路补充授权(2026-09-20) + +用户明确授权 HBM 时序、base SRAM 双缓冲和级联链路仲裁的基础简单实现, +用于快速闭合四拓扑 CPU 链路;先实现正确分配、有限缓冲回压与共享链路互斥, +不追求详细器件优化。默认关闭,复用实际 MQSim 提交/时间推进接口,后端完成 +与封装送达分开,不重复计算原延迟。参数未经标定须标工程假设。该授权不含 +原 MQSim 调度/事务所有权改造或真实 die 维护生命周期。授权原文见外层 +eq3_thermal/plans/decision-execution-v2/basic-system-user-source.md。 + +## EQ3-ISOLATED-MAINTENANCE-CAMPAIGN-v1(最新明确授权) + +用户采纳本阶段任务书并授权连续执行。温度模拟 MQSim 必须与原项目 MQSim 隔离:独立源码副本/补丁、构建、头文件、库、可执行与运行工件;原 third_party/mqsim、默认补丁/后端和生产 ABI/PTX/TMA/future/cache 不变。仅实验副本内批准任务书所列同一 engine/FTL/TSU/NAND 的窄维护身份、真实读写擦、有限目的分配、源版本保护、commit/fail 和逐页年龄计数;不批准完整重写调度/FTL/事件引擎。范围见外层 eq3_thermal/plans/isolated-maintenance-campaign-v1/USER_TASKBOOK.md。 +保留原物理假设采用 CONDITIONAL_ENGINEERING_USE,不再为0.25K细化P2;旧失败与盲测锁定不变。授权v3原raw重评分、默认关闭读取率反馈、完整耦合热闭环、先pilot冻结再四拓扑可行矩阵及有依据消融/敏感性。逐点留元信息,无需重复请示。资源按每次实际需求与主机余量有限分配,旧固定上限不机械恢复。GPU0、云0、不删除raw、不push/merge。超出窄结构范围才另行确认。 + +## 完成通知与低频检查(用户最新要求) + +实验运行优先通过已有完成/失败通知唤醒负责代理;没有通知能力时,根据已测 +wall time 与剩余工作估算 ETA,到预计完成时再检查。不要逐波、逐秒或反复读取 +未变化的状态来唤醒 AI。异常、资源风险及用户询问可提前检查。进程内部用于 +看护自身作业的内存/磁盘/watchdog 检查继续保留;这种自动安全检查不需要 AI +参与,不能为减少轮询而取消。记录实际进程/会话与 DONE/FAILED 路径;不得把 +尚未配置的通知机制描述为已经启用。 + +## 用户最新研究目标:速率驱动热模拟(2026-09-20) + +主目标改为各HBF stack不同读取速率/通道活动/die分布引起的温度变化,不要求逐页真实读写。优先用既有完整耦合热网络的能量输入接口,避免为标称带宽扩写MQSim。原隔离事务矩阵按用户转向暂停,保留已完成57点、失败和未执行记录。 +用户明确采用原80W/stack at1.6TB/s研究包络的条件模拟:array40pJ/B、base10pJ/B;未知idle不填成器件零功耗,只报告增量热。当前通道配置选OCP v0.7.0 Grade2,16channel/stack、96GB/s有效上限/channel、1.536TB/s/stack;能量系数不重标,因此满速76.8W。通道→die为单列工程映射,输入规定速率不得冒充实际后端吞吐。保留每die/base源与完整共享热路径;无新增源的die仍被动受热。热方程、材料、原数值失败、300–400K检查和盲测锁定不变。 + +## 速率热闭环继续执行(最新用户要求) + +用户明确要求继续按模型大小构造读取压力并探索热节流;前两个短开环pilot不代表任务完成。按rate-thermal-control-v1预检连续完成固定测试、两拓扑pilot、范围内持续/等均值突发三策略对照和结果分析;每点留元信息,不逐点等待确认。保持原40/10pJ/B、OCP Grade2与完整耦合热模型;新增流体交付不得冒称真实MQSim/fabric吞吐。 + +## 四拓扑系统热闭环阶段(2026-09-20最新明确授权) + +用户以PLEASE IMPLEMENT THIS PLAN批准EQ3四拓扑热闭环与逐栈读取性能计划。新隔离experiments/eq3_system_thermal复用已有MQSim维护后端和完整热网络,批准四拓扑批量服务、因果workload、可靠性情景代理、60点速率矩阵、指定敏感性和消融。按阶段保存元信息并在固定验证和pilot后连续执行,不重复逐点请求。现有原项目MQSim/默认后端、生产ABI及P2旧失败/400K/盲测边界不变。范围内新增实验模块设计已由本计划授权;超出范围的核心重构仍须确认。资源按实际pilot合理有限分配;GPU0/云0/不删除raw/不push或merge。 diff --git a/CMakeLists.txt b/CMakeLists.txt index c7b861b..7523371 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -1052,7 +1052,7 @@ if(HBFSIM_ENABLE_EVAL_TOOLS) endif() if(HBFSIM_ENABLE_MQSIM) add_executable(hbf_mqsim_service benchmarks/replay/hbf_mqsim_service.cpp) - target_link_libraries(hbf_mqsim_service PRIVATE hbfsim_core) + target_link_libraries(hbf_mqsim_service PRIVATE hbfsim_core mqsim_hbf) target_include_directories(hbf_mqsim_service PRIVATE "${CMAKE_CURRENT_SOURCE_DIR}/third_party/bpftime/third_party") if(BUILD_TESTING) diff --git a/benchmarks/replay/hbf_mqsim_service.cpp b/benchmarks/replay/hbf_mqsim_service.cpp index c487f53..bf9d1ee 100644 --- a/benchmarks/replay/hbf_mqsim_service.cpp +++ b/benchmarks/replay/hbf_mqsim_service.cpp @@ -3,20 +3,107 @@ #include #include #include +#include +#include +#include #include "replay_json.hpp" +#include #include #include #include +#include #include #include +#include #include namespace { using Json = nlohmann::json; using hbfsim::eval::integer; -void send(Json reply, hbfsim::MqsimOnlineEngine& engine) +std::uint64_t integer_value(const Json& value,const char* field) +{ + if(value.is_number_unsigned())return value.get(); + if(!value.is_number_integer()||value.get()<0) + throw std::invalid_argument(std::string(field)+" requires a nonnegative integer"); + return static_cast(value.get()); +} + +hbfsim::eq3_thermal::MqsimStackMapAdapter load_stack_map( + const std::string& path,const hbfsim::Profile& profile) +{ + std::ifstream input(path); + if(!input)throw std::runtime_error("failed to open MQSim stack map"); + const auto config=Json::parse(input); + if(integer(config,"schema_version")!=1||config.at("physical_kind")!="HBF"|| + config.at("route")!="direct"||config.at("address_layout")!="GLOBAL_PAGE_STRIPE_V1"|| + config.at("plane_allocation_scheme")!="CWDP"|| + integer(config,"page_bytes")!=profile.page_bytes|| + integer(config,"channels")!=profile.channels|| + integer(config,"dies_per_channel")!=profile.dies_per_channel) + throw std::invalid_argument("MQSim stack map does not match the HBF profile"); + std::vector groups; + for(const auto& row:config.at("stacks")) { + hbfsim::eq3_thermal::MqsimStackChannelGroup group; + group.stack_id=row.at("id").get(); + const auto declared=integer(row,"declared_dies"); + if(declared>std::numeric_limits::max()) + throw std::invalid_argument("MQSim stack-map die count overflows"); + group.declared_dies=static_cast(declared); + for(const auto& channel:row.at("channels")) { + const auto value=integer_value(channel,"channel"); + if(value>std::numeric_limits::max()) + throw std::invalid_argument("MQSim stack-map channel overflows"); + group.channels.push_back(static_cast(value)); + } + groups.push_back(std::move(group)); + } + return {profile,std::move(groups)}; +} + +using PlacementLedger=std::map; + +std::optional requested_stack(const Json& row,bool mapping_enabled) +{ + if(!mapping_enabled)return std::nullopt; + if(!row.contains("stack")||!row.at("stack").is_string()|| + !row.contains("route")||row.at("route")!="direct") + throw std::invalid_argument("enabled MQSim HBF stack map requires stack and direct route"); + const auto stack=row.at("stack").get(); + if(stack.empty())throw std::invalid_argument("empty requested HBF stack identity"); + return stack; +} + +Json placement_json(const hbfsim::eq3_thermal::MqsimStackPlacement& placement) +{ + return {{"external_logical_page",placement.external_page}, + {"backend_logical_page",placement.backend_page}, + {"stack",placement.stack_id ? Json(*placement.stack_id) : Json(nullptr)}, + {"expected_channel",placement.expected_channel ? + Json(*placement.expected_channel) : Json(nullptr)}}; +} + +hbfsim::eq3_thermal::MqsimStackPlacement place_request( + const Json& row,const hbfsim::HbfRequest& external, + const hbfsim::eq3_thermal::MqsimStackMapAdapter& stack_map) +{ + if(!stack_map.enabled())return stack_map.map(external); + const auto requested=requested_stack(row,true); + if(!row.contains("stack_local_page")) + throw std::invalid_argument("enabled MQSim HBF stack map requires stack_local_page"); + const auto local_page=integer(row,"stack_local_page"); + const auto placement=stack_map.map_stack_page(external,*requested,local_page); + if(row.contains("logical_address")&& + integer(row,"logical_address")!=placement.external_page*external.bytes) + throw std::invalid_argument("logical_address conflicts with stack-local placement"); + return placement; +} + +void send(Json reply, hbfsim::MqsimOnlineEngine& engine, + std::vector* native=nullptr, + const hbfsim::eq3_thermal::MqsimStackMapAdapter* stack_map=nullptr, + const PlacementLedger* placements=nullptr) { Json events = Json::array(); for (const auto& event : engine.take_observations()) { @@ -26,6 +113,44 @@ void send(Json reply, hbfsim::MqsimOnlineEngine& engine) {"bytes", event.bytes}, {"device_outstanding", event.device_outstanding}}); } reply["events"] = std::move(events); + Json native_events=Json::array(); + if (native) { + if (SSD_Components::NVM_PHY_ONFI_NVDDR2::Hbf_command_observation_failed()) + throw std::runtime_error("native MQSim command observation sink failed"); + for (const auto& event : *native) { + Json transactions=Json::array(); + for (const auto& tr : event.transactions) { + std::optional actual_stack; + const hbfsim::eq3_thermal::MqsimStackPlacement* expected=nullptr; + if(stack_map&&stack_map->enabled())actual_stack=stack_map->stack_for_channel(tr.channel); + if(placements&&tr.external_request_id) { + const auto found=placements->find(tr.external_request_id); + if(found!=placements->end())expected=&found->second; + } + if(expected&&(!actual_stack||!expected->stack_id|| + *actual_stack!=*expected->stack_id|| + !expected->expected_channel||tr.channel!=*expected->expected_channel|| + (tr.logical_page_known&&tr.logical_page!=expected->backend_page))) + throw std::runtime_error("native MQSim command violated configured HBF stack placement"); + transactions.push_back({{"transaction_id",tr.transaction_id}, + {"external_request_id",tr.external_request_id ? Json(tr.external_request_id) : Json(nullptr)}, + {"source",tr.source},{"type",tr.type}, + {"logical_page",tr.logical_page_known ? Json(tr.logical_page) : Json(nullptr)}, + {"bytes",tr.bytes}, + {"stack",actual_stack ? Json(*actual_stack) : Json(nullptr)}, + {"expected_stack",expected&&expected->stack_id ? Json(*expected->stack_id) : Json(nullptr)}, + {"external_logical_page",expected ? Json(expected->external_page) : Json(nullptr)}, + {"backend_logical_page",expected ? Json(expected->backend_page) : Json(nullptr)}, + {"channel",tr.channel},{"chip",tr.chip}, + {"die",tr.die},{"plane",tr.plane},{"block",tr.block},{"page",tr.page}}); + } + native_events.push_back({{"command_id",event.command_id}, + {"phase",static_cast(event.phase)},{"time_ns",event.time}, + {"command_code",event.command_code},{"transactions",std::move(transactions)}}); + } + native->clear(); + } + reply["native_command_events"] = std::move(native_events); reply["now_ns"] = engine.current_time_ns(); reply["pending"] = engine.pending(); std::cout << reply.dump() << '\n' << std::flush; @@ -37,7 +162,10 @@ int main(int argc, char** argv) { try { std::string profile_path; + std::string stack_map_path; std::uint64_t requested_units = 0; + std::optional gate_not_before_ns; + std::optional native_command_option; for (int i=1; i(profile.channels)*profile.dies_per_channel*profile.planes_per_die; if (requested_units && requested_units!=units) throw std::invalid_argument("PROJECTED_ANALYTICAL required: no exact configured MQSim topology"); + hbfsim::eq3_thermal::MqsimStackMapAdapter stack_map; + if(!stack_map_path.empty())stack_map=load_stack_map(stack_map_path,profile); + if(stack_map.enabled()&&native_command_option&&!*native_command_option) + throw std::invalid_argument("MQSim stack map requires native command verification"); + const bool native_command_observations=stack_map.enabled()||native_command_option.value_or(false); + std::vector native_events; hbfsim::MqsimOnlineEngine engine(profile); + if (native_command_observations) { + SSD_Components::NVM_PHY_ONFI_NVDDR2::Set_hbf_command_observation_sink( + [&native_events](const auto& event) { native_events.push_back(event); }); + } engine.enable_observations(); + hbfsim::eq3_thermal::MqsimSubmissionGateAdapter gate( + engine, + gate_not_before_ns ? hbfsim::eq3_thermal::MqsimGateMode::Enabled + : hbfsim::eq3_thermal::MqsimGateMode::Off, + [gate_not_before_ns](const hbfsim::HbfRequest&,std::uint64_t now) { + if (gate_not_before_ns&&now<*gate_not_before_ns) + return hbfsim::eq3_thermal::MqsimGateDecision{ + hbfsim::eq3_thermal::MqsimGateDisposition::Defer, + *gate_not_before_ns, + "ENGINEERING_FIXTURE target-time gate"}; + return hbfsim::eq3_thermal::MqsimGateDecision{}; + }); send({{"schema_version", 1}, {"service_source", "MQSIM_SIMULATED"}, {"provenance", "PROJECTED"}, {"scope", "READ_ONLY_MEDIA_SERVICE_NOT_HARDWARE"}, {"queue_depth", profile.queue_depth}, - {"parallel_units", units}}, engine); + {"parallel_units", units}, + {"page_bytes",profile.page_bytes}, + {"aggregate_bandwidth_bytes_per_s",profile.aggregate_bandwidth_bytes_per_s}, + {"stack_mapping",stack_map.enabled() ? + "ACTUAL_MQSIM_CHANNEL_PARTITIONED_HBF_STACKS" : "OFF"}, + {"stack_mapping_evidence",stack_map.enabled() ? + "ENGINEERING_FIXTURE_CONFIGURATION_NOT_RESEARCH_GEOMETRY" : "OFF"}, + {"submission_gate", gate_not_before_ns ? "ENGINEERING_FIXTURE_ENABLED" : "OFF"}, + {"native_command_observations",native_command_observations ? "ON" : "OFF"}}, + engine,native_command_observations ? &native_events : nullptr, + stack_map.enabled() ? &stack_map : nullptr,nullptr); std::map accepted; + PlacementLedger placements; std::set completed; std::uint64_t issued_bytes=0, completed_bytes=0, sequence=0; std::string line; @@ -70,12 +241,14 @@ int main(int argc, char** argv) const auto& rows=input.at("requests"); if (!rows.is_array() || rows.empty()) throw std::invalid_argument("submit needs a nonempty request array"); std::vector batch; + std::vector batch_placements; std::set ids; auto total=issued_bytes; // Validate the whole batch before any request is accepted. for (const auto& row : rows) { const auto id=integer(row, "request_id"), issue=integer(row, "issue_ns"); - const auto bytes=integer(row, "bytes"), address=integer(row, "logical_address"); + const auto bytes=integer(row, "bytes"); + const auto address=stack_map.enabled() ? 0 : integer(row,"logical_address"); if (!id || accepted.contains(id) || !ids.insert(id).second) throw std::invalid_argument("zero or reused request ID"); if (row.at("operation")!="read" || !bytes || bytes>std::numeric_limits::max() @@ -85,18 +258,72 @@ int main(int argc, char** argv) if (total>std::numeric_limits::max()-bytes) throw std::overflow_error("request byte total overflow"); total+=bytes; - batch.push_back({.request_id=id, .sequence=0, .arrival_ns=issue, - .logical_address=address, .bytes=static_cast(bytes), .operation=0}); + const hbfsim::HbfRequest external{ + .request_id=id, .sequence=0, .arrival_ns=issue, + .logical_address=address, .bytes=static_cast(bytes), .operation=0}; + auto placement=place_request(row,external,stack_map); + batch.push_back(placement.backend_request); + batch_placements.push_back(std::move(placement)); } if (batch.size()>std::numeric_limits::max()-sequence) throw std::overflow_error("service sequence exhausted"); - for (auto& request : batch) { + for (std::size_t i=0;istd::numeric_limits::max() || bytes%512 || address%512 || + bytes>profile.capacity_bytes || address>profile.capacity_bytes-bytes || + (!gate_not_before_ns&&issuestd::numeric_limits::max()-bytes || + sequence==std::numeric_limits::max()) + throw std::invalid_argument("invalid gated read extent, identity or arrival"); + const hbfsim::HbfRequest external{.request_id=id,.sequence=sequence+1,.arrival_ns=issue, + .logical_address=address,.bytes=static_cast(bytes),.operation=0}; + auto placement=place_request(row,external,stack_map); + const auto attempt=gate.try_submit(placement.backend_request); + Json gate_result={{"submitted",attempt.submitted}, + {"disposition",static_cast(attempt.decision.disposition)}, + {"reason",attempt.decision.reason}, + {"original_arrival_ns",attempt.original_arrival_ns}, + {"evaluated_ns",attempt.evaluated_ns}}; + gate_result["target_ns"]=attempt.decision.target_time_ns ? + Json(*attempt.decision.target_time_ns) : Json(nullptr); + gate_result["backend_arrival_ns"]=attempt.backend_arrival_ns ? + Json(*attempt.backend_arrival_ns) : Json(nullptr); + gate_result["external_wait_ns"]=attempt.external_wait_ns ? + Json(*attempt.external_wait_ns) : Json(nullptr); + if (attempt.submitted) { + ++sequence;issued_bytes+=bytes; + accepted.emplace(id,static_cast(bytes)); + placement.backend_request.sequence=sequence; + if(stack_map.enabled())placements.emplace(id,placement); + } + send({{"gate",std::move(gate_result)}, + {"placement",stack_map.enabled() ? placement_json(placement) : Json(nullptr)}},engine, + native_command_observations ? &native_events : nullptr, + stack_map.enabled() ? &stack_map : nullptr, + stack_map.enabled() ? &placements : nullptr); } else if (command=="until") { const auto completion=engine.run_next_completion_until(integer(input, "deadline_ns")); Json value=nullptr; @@ -108,12 +335,20 @@ int main(int argc, char** argv) value={{"request_id", completion->request_id}, {"reported_complete", completion->modeled_completion_ns}, {"status", completion->status}}; } - send({{"completion", value}}, engine); + send({{"completion", value}}, engine, + native_command_observations ? &native_events : nullptr, + stack_map.enabled() ? &stack_map : nullptr, + stack_map.enabled() ? &placements : nullptr); } else if (command=="finish") { if (engine.pending() || accepted.size()!=completed.size() || issued_bytes!=completed_bytes) throw std::runtime_error("cannot finish with unreturned requests/bytes"); send({{"status", "FINISHED"}, {"issued", accepted.size()}, {"completed", completed.size()}, - {"issued_bytes", issued_bytes}, {"completed_bytes", completed_bytes}}, engine); + {"issued_bytes", issued_bytes}, {"completed_bytes", completed_bytes}}, engine, + native_command_observations ? &native_events : nullptr, + stack_map.enabled() ? &stack_map : nullptr, + stack_map.enabled() ? &placements : nullptr); + if (native_command_observations) + SSD_Components::NVM_PHY_ONFI_NVDDR2::Clear_hbf_command_observation_sink(); return 0; } else throw std::invalid_argument("unknown service command"); } diff --git a/cmake/MQSimPatchedBuild.cmake b/cmake/MQSimPatchedBuild.cmake index d7d4f4e..0069289 100644 --- a/cmake/MQSimPatchedBuild.cmake +++ b/cmake/MQSimPatchedBuild.cmake @@ -4,6 +4,7 @@ function(hbfsim_add_patched_mqsim target_name) set(mqsim_patches "${CMAKE_CURRENT_SOURCE_DIR}/patches/mqsim/0001-online-hbf-api.patch" "${CMAKE_CURRENT_SOURCE_DIR}/patches/mqsim/0002-qlc-support.patch" + "${CMAKE_CURRENT_SOURCE_DIR}/patches/mqsim/0003-hbf-command-observer.patch" ) file(REMOVE_RECURSE "${mqsim_copy}") diff --git a/configs/eq3_thermal/acceptance_v2/development64s.json b/configs/eq3_thermal/acceptance_v2/development64s.json new file mode 100644 index 0000000..37795b6 --- /dev/null +++ b/configs/eq3_thermal/acceptance_v2/development64s.json @@ -0,0 +1,345 @@ +{ + "schema_version": "eq3-acceptance-v2-method-v1", + "method_id": "eq3-d2-development64s-v1", + "temperature_unit": "K", + "sensor_ids": [ + "component:gpu:hotspot", + "component:gpu:mean", + "component:hbf0.base:hotspot", + "component:hbf0.base:mean", + "component:hbf0.die0:hotspot", + "component:hbf0.die0:mean", + "component:hbf0.die10:hotspot", + "component:hbf0.die10:mean", + "component:hbf0.die11:hotspot", + "component:hbf0.die11:mean", + "component:hbf0.die12:hotspot", + "component:hbf0.die12:mean", + "component:hbf0.die13:hotspot", + "component:hbf0.die13:mean", + "component:hbf0.die14:hotspot", + "component:hbf0.die14:mean", + "component:hbf0.die15:hotspot", + "component:hbf0.die15:mean", + "component:hbf0.die1:hotspot", + "component:hbf0.die1:mean", + "component:hbf0.die2:hotspot", + "component:hbf0.die2:mean", + "component:hbf0.die3:hotspot", + "component:hbf0.die3:mean", + "component:hbf0.die4:hotspot", + "component:hbf0.die4:mean", + "component:hbf0.die5:hotspot", + "component:hbf0.die5:mean", + "component:hbf0.die6:hotspot", + "component:hbf0.die6:mean", + "component:hbf0.die7:hotspot", + "component:hbf0.die7:mean", + "component:hbf0.die8:hotspot", + "component:hbf0.die8:mean", + "component:hbf0.die9:hotspot", + "component:hbf0.die9:mean", + "component:hbf1.base:hotspot", + "component:hbf1.base:mean", + "component:hbf1.die0:hotspot", + "component:hbf1.die0:mean", + "component:hbf1.die10:hotspot", + "component:hbf1.die10:mean", + "component:hbf1.die11:hotspot", + "component:hbf1.die11:mean", + "component:hbf1.die12:hotspot", + "component:hbf1.die12:mean", + "component:hbf1.die13:hotspot", + "component:hbf1.die13:mean", + "component:hbf1.die14:hotspot", + "component:hbf1.die14:mean", + "component:hbf1.die15:hotspot", + "component:hbf1.die15:mean", + "component:hbf1.die1:hotspot", + "component:hbf1.die1:mean", + "component:hbf1.die2:hotspot", + "component:hbf1.die2:mean", + "component:hbf1.die3:hotspot", + "component:hbf1.die3:mean", + "component:hbf1.die4:hotspot", + "component:hbf1.die4:mean", + "component:hbf1.die5:hotspot", + "component:hbf1.die5:mean", + "component:hbf1.die6:hotspot", + "component:hbf1.die6:mean", + "component:hbf1.die7:hotspot", + "component:hbf1.die7:mean", + "component:hbf1.die8:hotspot", + "component:hbf1.die8:mean", + "component:hbf1.die9:hotspot", + "component:hbf1.die9:mean", + "component:hbf2.base:hotspot", + "component:hbf2.base:mean", + "component:hbf2.die0:hotspot", + "component:hbf2.die0:mean", + "component:hbf2.die10:hotspot", + "component:hbf2.die10:mean", + "component:hbf2.die11:hotspot", + "component:hbf2.die11:mean", + "component:hbf2.die12:hotspot", + "component:hbf2.die12:mean", + "component:hbf2.die13:hotspot", + "component:hbf2.die13:mean", + "component:hbf2.die14:hotspot", + "component:hbf2.die14:mean", + "component:hbf2.die15:hotspot", + "component:hbf2.die15:mean", + "component:hbf2.die1:hotspot", + "component:hbf2.die1:mean", + "component:hbf2.die2:hotspot", + "component:hbf2.die2:mean", + "component:hbf2.die3:hotspot", + "component:hbf2.die3:mean", + "component:hbf2.die4:hotspot", + "component:hbf2.die4:mean", + "component:hbf2.die5:hotspot", + "component:hbf2.die5:mean", + "component:hbf2.die6:hotspot", + "component:hbf2.die6:mean", + "component:hbf2.die7:hotspot", + "component:hbf2.die7:mean", + "component:hbf2.die8:hotspot", + "component:hbf2.die8:mean", + "component:hbf2.die9:hotspot", + "component:hbf2.die9:mean", + "component:hbf3.base:hotspot", + "component:hbf3.base:mean", + "component:hbf3.die0:hotspot", + "component:hbf3.die0:mean", + "component:hbf3.die10:hotspot", + "component:hbf3.die10:mean", + "component:hbf3.die11:hotspot", + "component:hbf3.die11:mean", + "component:hbf3.die12:hotspot", + "component:hbf3.die12:mean", + "component:hbf3.die13:hotspot", + "component:hbf3.die13:mean", + "component:hbf3.die14:hotspot", + "component:hbf3.die14:mean", + "component:hbf3.die15:hotspot", + "component:hbf3.die15:mean", + "component:hbf3.die1:hotspot", + "component:hbf3.die1:mean", + "component:hbf3.die2:hotspot", + "component:hbf3.die2:mean", + "component:hbf3.die3:hotspot", + "component:hbf3.die3:mean", + "component:hbf3.die4:hotspot", + "component:hbf3.die4:mean", + "component:hbf3.die5:hotspot", + "component:hbf3.die5:mean", + "component:hbf3.die6:hotspot", + "component:hbf3.die6:mean", + "component:hbf3.die7:hotspot", + "component:hbf3.die7:mean", + "component:hbf3.die8:hotspot", + "component:hbf3.die8:mean", + "component:hbf3.die9:hotspot", + "component:hbf3.die9:mean", + "component:hbm0.base:hotspot", + "component:hbm0.base:mean", + "component:hbm0.die0:hotspot", + "component:hbm0.die0:mean", + "component:hbm0.die10:hotspot", + "component:hbm0.die10:mean", + "component:hbm0.die11:hotspot", + "component:hbm0.die11:mean", + "component:hbm0.die1:hotspot", + "component:hbm0.die1:mean", + "component:hbm0.die2:hotspot", + "component:hbm0.die2:mean", + "component:hbm0.die3:hotspot", + "component:hbm0.die3:mean", + "component:hbm0.die4:hotspot", + "component:hbm0.die4:mean", + "component:hbm0.die5:hotspot", + "component:hbm0.die5:mean", + "component:hbm0.die6:hotspot", + "component:hbm0.die6:mean", + "component:hbm0.die7:hotspot", + "component:hbm0.die7:mean", + "component:hbm0.die8:hotspot", + "component:hbm0.die8:mean", + "component:hbm0.die9:hotspot", + "component:hbm0.die9:mean", + "component:hbm1.base:hotspot", + "component:hbm1.base:mean", + "component:hbm1.die0:hotspot", + "component:hbm1.die0:mean", + "component:hbm1.die10:hotspot", + "component:hbm1.die10:mean", + "component:hbm1.die11:hotspot", + "component:hbm1.die11:mean", + "component:hbm1.die1:hotspot", + "component:hbm1.die1:mean", + "component:hbm1.die2:hotspot", + "component:hbm1.die2:mean", + "component:hbm1.die3:hotspot", + "component:hbm1.die3:mean", + "component:hbm1.die4:hotspot", + "component:hbm1.die4:mean", + "component:hbm1.die5:hotspot", + "component:hbm1.die5:mean", + "component:hbm1.die6:hotspot", + "component:hbm1.die6:mean", + "component:hbm1.die7:hotspot", + "component:hbm1.die7:mean", + "component:hbm1.die8:hotspot", + "component:hbm1.die8:mean", + "component:hbm1.die9:hotspot", + "component:hbm1.die9:mean", + "component:hbm2.base:hotspot", + "component:hbm2.base:mean", + "component:hbm2.die0:hotspot", + "component:hbm2.die0:mean", + "component:hbm2.die10:hotspot", + "component:hbm2.die10:mean", + "component:hbm2.die11:hotspot", + "component:hbm2.die11:mean", + "component:hbm2.die1:hotspot", + "component:hbm2.die1:mean", + "component:hbm2.die2:hotspot", + "component:hbm2.die2:mean", + "component:hbm2.die3:hotspot", + "component:hbm2.die3:mean", + "component:hbm2.die4:hotspot", + "component:hbm2.die4:mean", + "component:hbm2.die5:hotspot", + "component:hbm2.die5:mean", + "component:hbm2.die6:hotspot", + "component:hbm2.die6:mean", + "component:hbm2.die7:hotspot", + "component:hbm2.die7:mean", + "component:hbm2.die8:hotspot", + "component:hbm2.die8:mean", + "component:hbm2.die9:hotspot", + "component:hbm2.die9:mean", + "component:hbm3.base:hotspot", + "component:hbm3.base:mean", + "component:hbm3.die0:hotspot", + "component:hbm3.die0:mean", + "component:hbm3.die10:hotspot", + "component:hbm3.die10:mean", + "component:hbm3.die11:hotspot", + "component:hbm3.die11:mean", + "component:hbm3.die1:hotspot", + "component:hbm3.die1:mean", + "component:hbm3.die2:hotspot", + "component:hbm3.die2:mean", + "component:hbm3.die3:hotspot", + "component:hbm3.die3:mean", + "component:hbm3.die4:hotspot", + "component:hbm3.die4:mean", + "component:hbm3.die5:hotspot", + "component:hbm3.die5:mean", + "component:hbm3.die6:hotspot", + "component:hbm3.die6:mean", + "component:hbm3.die7:hotspot", + "component:hbm3.die7:mean", + "component:hbm3.die8:hotspot", + "component:hbm3.die8:mean", + "component:hbm3.die9:hotspot", + "component:hbm3.die9:mean", + "package:powered_hotspot", + "stack:hbf0:array_hotspot", + "stack:hbf0:array_mean", + "stack:hbf0:hotspot", + "stack:hbf0:mean", + "stack:hbf1:array_hotspot", + "stack:hbf1:array_mean", + "stack:hbf1:hotspot", + "stack:hbf1:mean", + "stack:hbf2:array_hotspot", + "stack:hbf2:array_mean", + "stack:hbf2:hotspot", + "stack:hbf2:mean", + "stack:hbf3:array_hotspot", + "stack:hbf3:array_mean", + "stack:hbf3:hotspot", + "stack:hbf3:mean", + "stack:hbm0:array_hotspot", + "stack:hbm0:array_mean", + "stack:hbm0:hotspot", + "stack:hbm0:mean", + "stack:hbm1:array_hotspot", + "stack:hbm1:array_mean", + "stack:hbm1:hotspot", + "stack:hbm1:mean", + "stack:hbm2:array_hotspot", + "stack:hbm2:array_mean", + "stack:hbm2:hotspot", + "stack:hbm2:mean", + "stack:hbm3:array_hotspot", + "stack:hbm3:array_mean", + "stack:hbm3:hotspot", + "stack:hbm3:mean" + ], + "initial_time_s": 0.0, + "initial_temperature_k": 300.0, + "windows": [ + { + "id": "full", + "kind": "full", + "start_s": 0.0, + "end_s": 64.0 + }, + { + "id": "excitation-01", + "kind": "excitation", + "start_s": 0.0, + "end_s": 32.0 + }, + { + "id": "cooling-01", + "kind": "cooling", + "start_s": 32.0, + "end_s": 64.0 + } + ], + "hotspot_sensor_ids": [ + "package:powered_hotspot" + ], + "control_sensor_ids": [], + "crossing_thresholds_k": [ + 301.0, + 330.0 + ], + "crossing_thresholds_semantics": "legacy_v1 diagnostic probes retained for comparison; not device limits", + "crossing_min_time_s": 0.2, + "crossing_fraction": 0.05, + "window_derivation": { + "basis": "union of source activity [start_s,end_s] before output inspection; cooling is complement in full interval", + "event_count": 7728, + "unique_source_intervals": 64, + "merged_excitation_intervals": 1, + "cooling_intervals": 1 + }, + "sensor_role_derivation": { + "grid_hotspot": "package:powered_hotspot has reduction=max over all powered components", + "control": "no active P2 CSV consumer; stack::hotspot is a readonly semantic candidate matching CpuService max-over-stack only" + }, + "source_contract": { + "generated_case": "campaign-R06-2MM-development", + "events_sha256": "f4fad22658d218a604e574f5369b006a5b99eb5116f6575fd533bde144f38070", + "reference_sensors_sha256": "efd93445e2ef3ab82dcca81f294dd5c20db06138d5cf6c52d616bd91248aa674", + "sensor_count": 275, + "all_hotspot_reductions_are_max": true, + "new_output_inspected_for_window_selection": false + }, + "candidate_control_sensor_ids": [ + "stack:hbf0:hotspot", + "stack:hbf1:hotspot", + "stack:hbf2:hotspot", + "stack:hbf3:hotspot", + "stack:hbm0:hotspot", + "stack:hbm1:hotspot", + "stack:hbm2:hotspot", + "stack:hbm3:hotspot" + ], + "candidate_control_sensor_status": "DECLARED_READONLY_CANDIDATE_NOT_ACTIVE" +} diff --git a/configs/eq3_thermal/acceptance_v2/train100s.json b/configs/eq3_thermal/acceptance_v2/train100s.json new file mode 100644 index 0000000..ab0b15c --- /dev/null +++ b/configs/eq3_thermal/acceptance_v2/train100s.json @@ -0,0 +1,747 @@ +{ + "schema_version": "eq3-acceptance-v2-method-v1", + "method_id": "eq3-d2-train100s-v1", + "temperature_unit": "K", + "sensor_ids": [ + "component:gpu:hotspot", + "component:gpu:mean", + "component:hbf0.base:hotspot", + "component:hbf0.base:mean", + "component:hbf0.die0:hotspot", + "component:hbf0.die0:mean", + "component:hbf0.die10:hotspot", + "component:hbf0.die10:mean", + "component:hbf0.die11:hotspot", + "component:hbf0.die11:mean", + "component:hbf0.die12:hotspot", + "component:hbf0.die12:mean", + "component:hbf0.die13:hotspot", + "component:hbf0.die13:mean", + "component:hbf0.die14:hotspot", + "component:hbf0.die14:mean", + "component:hbf0.die15:hotspot", + "component:hbf0.die15:mean", + "component:hbf0.die1:hotspot", + "component:hbf0.die1:mean", + "component:hbf0.die2:hotspot", + "component:hbf0.die2:mean", + "component:hbf0.die3:hotspot", + "component:hbf0.die3:mean", + "component:hbf0.die4:hotspot", + "component:hbf0.die4:mean", + "component:hbf0.die5:hotspot", + "component:hbf0.die5:mean", + "component:hbf0.die6:hotspot", + "component:hbf0.die6:mean", + "component:hbf0.die7:hotspot", + "component:hbf0.die7:mean", + "component:hbf0.die8:hotspot", + "component:hbf0.die8:mean", + "component:hbf0.die9:hotspot", + "component:hbf0.die9:mean", + "component:hbf1.base:hotspot", + "component:hbf1.base:mean", + "component:hbf1.die0:hotspot", + "component:hbf1.die0:mean", + "component:hbf1.die10:hotspot", + "component:hbf1.die10:mean", + "component:hbf1.die11:hotspot", + "component:hbf1.die11:mean", + "component:hbf1.die12:hotspot", + "component:hbf1.die12:mean", + "component:hbf1.die13:hotspot", + "component:hbf1.die13:mean", + "component:hbf1.die14:hotspot", + "component:hbf1.die14:mean", + "component:hbf1.die15:hotspot", + "component:hbf1.die15:mean", + "component:hbf1.die1:hotspot", + "component:hbf1.die1:mean", + "component:hbf1.die2:hotspot", + "component:hbf1.die2:mean", + "component:hbf1.die3:hotspot", + "component:hbf1.die3:mean", + "component:hbf1.die4:hotspot", + "component:hbf1.die4:mean", + "component:hbf1.die5:hotspot", + "component:hbf1.die5:mean", + "component:hbf1.die6:hotspot", + "component:hbf1.die6:mean", + "component:hbf1.die7:hotspot", + "component:hbf1.die7:mean", + "component:hbf1.die8:hotspot", + "component:hbf1.die8:mean", + "component:hbf1.die9:hotspot", + "component:hbf1.die9:mean", + "component:hbf2.base:hotspot", + "component:hbf2.base:mean", + "component:hbf2.die0:hotspot", + "component:hbf2.die0:mean", + "component:hbf2.die10:hotspot", + "component:hbf2.die10:mean", + "component:hbf2.die11:hotspot", + "component:hbf2.die11:mean", + "component:hbf2.die12:hotspot", + "component:hbf2.die12:mean", + "component:hbf2.die13:hotspot", + "component:hbf2.die13:mean", + "component:hbf2.die14:hotspot", + "component:hbf2.die14:mean", + "component:hbf2.die15:hotspot", + "component:hbf2.die15:mean", + "component:hbf2.die1:hotspot", + "component:hbf2.die1:mean", + "component:hbf2.die2:hotspot", + "component:hbf2.die2:mean", + "component:hbf2.die3:hotspot", + "component:hbf2.die3:mean", + "component:hbf2.die4:hotspot", + "component:hbf2.die4:mean", + "component:hbf2.die5:hotspot", + "component:hbf2.die5:mean", + "component:hbf2.die6:hotspot", + "component:hbf2.die6:mean", + "component:hbf2.die7:hotspot", + "component:hbf2.die7:mean", + "component:hbf2.die8:hotspot", + "component:hbf2.die8:mean", + "component:hbf2.die9:hotspot", + "component:hbf2.die9:mean", + "component:hbf3.base:hotspot", + "component:hbf3.base:mean", + "component:hbf3.die0:hotspot", + "component:hbf3.die0:mean", + "component:hbf3.die10:hotspot", + "component:hbf3.die10:mean", + "component:hbf3.die11:hotspot", + "component:hbf3.die11:mean", + "component:hbf3.die12:hotspot", + "component:hbf3.die12:mean", + "component:hbf3.die13:hotspot", + "component:hbf3.die13:mean", + "component:hbf3.die14:hotspot", + "component:hbf3.die14:mean", + "component:hbf3.die15:hotspot", + "component:hbf3.die15:mean", + "component:hbf3.die1:hotspot", + "component:hbf3.die1:mean", + "component:hbf3.die2:hotspot", + "component:hbf3.die2:mean", + "component:hbf3.die3:hotspot", + "component:hbf3.die3:mean", + "component:hbf3.die4:hotspot", + "component:hbf3.die4:mean", + "component:hbf3.die5:hotspot", + "component:hbf3.die5:mean", + "component:hbf3.die6:hotspot", + "component:hbf3.die6:mean", + "component:hbf3.die7:hotspot", + "component:hbf3.die7:mean", + "component:hbf3.die8:hotspot", + "component:hbf3.die8:mean", + "component:hbf3.die9:hotspot", + "component:hbf3.die9:mean", + "component:hbm0.base:hotspot", + "component:hbm0.base:mean", + "component:hbm0.die0:hotspot", + "component:hbm0.die0:mean", + "component:hbm0.die10:hotspot", + "component:hbm0.die10:mean", + "component:hbm0.die11:hotspot", + "component:hbm0.die11:mean", + "component:hbm0.die1:hotspot", + "component:hbm0.die1:mean", + "component:hbm0.die2:hotspot", + "component:hbm0.die2:mean", + "component:hbm0.die3:hotspot", + "component:hbm0.die3:mean", + "component:hbm0.die4:hotspot", + "component:hbm0.die4:mean", + "component:hbm0.die5:hotspot", + "component:hbm0.die5:mean", + "component:hbm0.die6:hotspot", + "component:hbm0.die6:mean", + "component:hbm0.die7:hotspot", + "component:hbm0.die7:mean", + "component:hbm0.die8:hotspot", + "component:hbm0.die8:mean", + "component:hbm0.die9:hotspot", + "component:hbm0.die9:mean", + "component:hbm1.base:hotspot", + "component:hbm1.base:mean", + "component:hbm1.die0:hotspot", + "component:hbm1.die0:mean", + "component:hbm1.die10:hotspot", + "component:hbm1.die10:mean", + "component:hbm1.die11:hotspot", + "component:hbm1.die11:mean", + "component:hbm1.die1:hotspot", + "component:hbm1.die1:mean", + "component:hbm1.die2:hotspot", + "component:hbm1.die2:mean", + "component:hbm1.die3:hotspot", + "component:hbm1.die3:mean", + "component:hbm1.die4:hotspot", + "component:hbm1.die4:mean", + "component:hbm1.die5:hotspot", + "component:hbm1.die5:mean", + "component:hbm1.die6:hotspot", + "component:hbm1.die6:mean", + "component:hbm1.die7:hotspot", + "component:hbm1.die7:mean", + "component:hbm1.die8:hotspot", + "component:hbm1.die8:mean", + "component:hbm1.die9:hotspot", + "component:hbm1.die9:mean", + "component:hbm2.base:hotspot", + "component:hbm2.base:mean", + "component:hbm2.die0:hotspot", + "component:hbm2.die0:mean", + "component:hbm2.die10:hotspot", + "component:hbm2.die10:mean", + "component:hbm2.die11:hotspot", + "component:hbm2.die11:mean", + "component:hbm2.die1:hotspot", + "component:hbm2.die1:mean", + "component:hbm2.die2:hotspot", + "component:hbm2.die2:mean", + "component:hbm2.die3:hotspot", + "component:hbm2.die3:mean", + "component:hbm2.die4:hotspot", + "component:hbm2.die4:mean", + "component:hbm2.die5:hotspot", + "component:hbm2.die5:mean", + "component:hbm2.die6:hotspot", + "component:hbm2.die6:mean", + "component:hbm2.die7:hotspot", + "component:hbm2.die7:mean", + "component:hbm2.die8:hotspot", + "component:hbm2.die8:mean", + "component:hbm2.die9:hotspot", + "component:hbm2.die9:mean", + "component:hbm3.base:hotspot", + "component:hbm3.base:mean", + "component:hbm3.die0:hotspot", + "component:hbm3.die0:mean", + "component:hbm3.die10:hotspot", + "component:hbm3.die10:mean", + "component:hbm3.die11:hotspot", + "component:hbm3.die11:mean", + "component:hbm3.die1:hotspot", + "component:hbm3.die1:mean", + "component:hbm3.die2:hotspot", + "component:hbm3.die2:mean", + "component:hbm3.die3:hotspot", + "component:hbm3.die3:mean", + "component:hbm3.die4:hotspot", + "component:hbm3.die4:mean", + "component:hbm3.die5:hotspot", + "component:hbm3.die5:mean", + "component:hbm3.die6:hotspot", + "component:hbm3.die6:mean", + "component:hbm3.die7:hotspot", + "component:hbm3.die7:mean", + "component:hbm3.die8:hotspot", + "component:hbm3.die8:mean", + "component:hbm3.die9:hotspot", + "component:hbm3.die9:mean", + "package:powered_hotspot", + "stack:hbf0:array_hotspot", + "stack:hbf0:array_mean", + "stack:hbf0:hotspot", + "stack:hbf0:mean", + "stack:hbf1:array_hotspot", + "stack:hbf1:array_mean", + "stack:hbf1:hotspot", + "stack:hbf1:mean", + "stack:hbf2:array_hotspot", + "stack:hbf2:array_mean", + "stack:hbf2:hotspot", + "stack:hbf2:mean", + "stack:hbf3:array_hotspot", + "stack:hbf3:array_mean", + "stack:hbf3:hotspot", + "stack:hbf3:mean", + "stack:hbm0:array_hotspot", + "stack:hbm0:array_mean", + "stack:hbm0:hotspot", + "stack:hbm0:mean", + "stack:hbm1:array_hotspot", + "stack:hbm1:array_mean", + "stack:hbm1:hotspot", + "stack:hbm1:mean", + "stack:hbm2:array_hotspot", + "stack:hbm2:array_mean", + "stack:hbm2:hotspot", + "stack:hbm2:mean", + "stack:hbm3:array_hotspot", + "stack:hbm3:array_mean", + "stack:hbm3:hotspot", + "stack:hbm3:mean" + ], + "initial_time_s": 0.0, + "initial_temperature_k": 300.0, + "windows": [ + { + "id": "full", + "kind": "full", + "start_s": 0.0, + "end_s": 100.0 + }, + { + "id": "excitation-01", + "kind": "excitation", + "start_s": 0.5, + "end_s": 1.0 + }, + { + "id": "excitation-02", + "kind": "excitation", + "start_s": 1.5, + "end_s": 3.0 + }, + { + "id": "excitation-03", + "kind": "excitation", + "start_s": 4.5, + "end_s": 5.0 + }, + { + "id": "excitation-04", + "kind": "excitation", + "start_s": 5.5, + "end_s": 7.0 + }, + { + "id": "excitation-05", + "kind": "excitation", + "start_s": 8.5, + "end_s": 9.0 + }, + { + "id": "excitation-06", + "kind": "excitation", + "start_s": 9.5, + "end_s": 11.0 + }, + { + "id": "excitation-07", + "kind": "excitation", + "start_s": 12.5, + "end_s": 13.0 + }, + { + "id": "excitation-08", + "kind": "excitation", + "start_s": 13.5, + "end_s": 15.0 + }, + { + "id": "excitation-09", + "kind": "excitation", + "start_s": 16.5, + "end_s": 17.0 + }, + { + "id": "excitation-10", + "kind": "excitation", + "start_s": 17.5, + "end_s": 19.0 + }, + { + "id": "excitation-11", + "kind": "excitation", + "start_s": 20.5, + "end_s": 21.0 + }, + { + "id": "excitation-12", + "kind": "excitation", + "start_s": 21.5, + "end_s": 23.0 + }, + { + "id": "excitation-13", + "kind": "excitation", + "start_s": 24.5, + "end_s": 25.0 + }, + { + "id": "excitation-14", + "kind": "excitation", + "start_s": 25.5, + "end_s": 27.0 + }, + { + "id": "excitation-15", + "kind": "excitation", + "start_s": 28.5, + "end_s": 29.0 + }, + { + "id": "excitation-16", + "kind": "excitation", + "start_s": 29.5, + "end_s": 31.0 + }, + { + "id": "excitation-17", + "kind": "excitation", + "start_s": 32.5, + "end_s": 33.0 + }, + { + "id": "excitation-18", + "kind": "excitation", + "start_s": 33.5, + "end_s": 35.0 + }, + { + "id": "excitation-19", + "kind": "excitation", + "start_s": 36.5, + "end_s": 37.0 + }, + { + "id": "excitation-20", + "kind": "excitation", + "start_s": 37.5, + "end_s": 39.0 + }, + { + "id": "excitation-21", + "kind": "excitation", + "start_s": 40.5, + "end_s": 41.0 + }, + { + "id": "excitation-22", + "kind": "excitation", + "start_s": 41.5, + "end_s": 43.0 + }, + { + "id": "excitation-23", + "kind": "excitation", + "start_s": 44.5, + "end_s": 45.0 + }, + { + "id": "excitation-24", + "kind": "excitation", + "start_s": 45.5, + "end_s": 47.0 + }, + { + "id": "excitation-25", + "kind": "excitation", + "start_s": 48.5, + "end_s": 49.0 + }, + { + "id": "excitation-26", + "kind": "excitation", + "start_s": 49.5, + "end_s": 51.0 + }, + { + "id": "excitation-27", + "kind": "excitation", + "start_s": 52.5, + "end_s": 53.0 + }, + { + "id": "excitation-28", + "kind": "excitation", + "start_s": 53.5, + "end_s": 55.0 + }, + { + "id": "excitation-29", + "kind": "excitation", + "start_s": 56.5, + "end_s": 57.0 + }, + { + "id": "excitation-30", + "kind": "excitation", + "start_s": 57.5, + "end_s": 59.0 + }, + { + "id": "excitation-31", + "kind": "excitation", + "start_s": 60.5, + "end_s": 61.0 + }, + { + "id": "excitation-32", + "kind": "excitation", + "start_s": 61.5, + "end_s": 63.0 + }, + { + "id": "excitation-33", + "kind": "excitation", + "start_s": 64.5, + "end_s": 65.0 + }, + { + "id": "excitation-34", + "kind": "excitation", + "start_s": 65.5, + "end_s": 67.0 + }, + { + "id": "cooling-01", + "kind": "cooling", + "start_s": 0.0, + "end_s": 0.5 + }, + { + "id": "cooling-02", + "kind": "cooling", + "start_s": 1.0, + "end_s": 1.5 + }, + { + "id": "cooling-03", + "kind": "cooling", + "start_s": 3.0, + "end_s": 4.5 + }, + { + "id": "cooling-04", + "kind": "cooling", + "start_s": 5.0, + "end_s": 5.5 + }, + { + "id": "cooling-05", + "kind": "cooling", + "start_s": 7.0, + "end_s": 8.5 + }, + { + "id": "cooling-06", + "kind": "cooling", + "start_s": 9.0, + "end_s": 9.5 + }, + { + "id": "cooling-07", + "kind": "cooling", + "start_s": 11.0, + "end_s": 12.5 + }, + { + "id": "cooling-08", + "kind": "cooling", + "start_s": 13.0, + "end_s": 13.5 + }, + { + "id": "cooling-09", + "kind": "cooling", + "start_s": 15.0, + "end_s": 16.5 + }, + { + "id": "cooling-10", + "kind": "cooling", + "start_s": 17.0, + "end_s": 17.5 + }, + { + "id": "cooling-11", + "kind": "cooling", + "start_s": 19.0, + "end_s": 20.5 + }, + { + "id": "cooling-12", + "kind": "cooling", + "start_s": 21.0, + "end_s": 21.5 + }, + { + "id": "cooling-13", + "kind": "cooling", + "start_s": 23.0, + "end_s": 24.5 + }, + { + "id": "cooling-14", + "kind": "cooling", + "start_s": 25.0, + "end_s": 25.5 + }, + { + "id": "cooling-15", + "kind": "cooling", + "start_s": 27.0, + "end_s": 28.5 + }, + { + "id": "cooling-16", + "kind": "cooling", + "start_s": 29.0, + "end_s": 29.5 + }, + { + "id": "cooling-17", + "kind": "cooling", + "start_s": 31.0, + "end_s": 32.5 + }, + { + "id": "cooling-18", + "kind": "cooling", + "start_s": 33.0, + "end_s": 33.5 + }, + { + "id": "cooling-19", + "kind": "cooling", + "start_s": 35.0, + "end_s": 36.5 + }, + { + "id": "cooling-20", + "kind": "cooling", + "start_s": 37.0, + "end_s": 37.5 + }, + { + "id": "cooling-21", + "kind": "cooling", + "start_s": 39.0, + "end_s": 40.5 + }, + { + "id": "cooling-22", + "kind": "cooling", + "start_s": 41.0, + "end_s": 41.5 + }, + { + "id": "cooling-23", + "kind": "cooling", + "start_s": 43.0, + "end_s": 44.5 + }, + { + "id": "cooling-24", + "kind": "cooling", + "start_s": 45.0, + "end_s": 45.5 + }, + { + "id": "cooling-25", + "kind": "cooling", + "start_s": 47.0, + "end_s": 48.5 + }, + { + "id": "cooling-26", + "kind": "cooling", + "start_s": 49.0, + "end_s": 49.5 + }, + { + "id": "cooling-27", + "kind": "cooling", + "start_s": 51.0, + "end_s": 52.5 + }, + { + "id": "cooling-28", + "kind": "cooling", + "start_s": 53.0, + "end_s": 53.5 + }, + { + "id": "cooling-29", + "kind": "cooling", + "start_s": 55.0, + "end_s": 56.5 + }, + { + "id": "cooling-30", + "kind": "cooling", + "start_s": 57.0, + "end_s": 57.5 + }, + { + "id": "cooling-31", + "kind": "cooling", + "start_s": 59.0, + "end_s": 60.5 + }, + { + "id": "cooling-32", + "kind": "cooling", + "start_s": 61.0, + "end_s": 61.5 + }, + { + "id": "cooling-33", + "kind": "cooling", + "start_s": 63.0, + "end_s": 64.5 + }, + { + "id": "cooling-34", + "kind": "cooling", + "start_s": 65.0, + "end_s": 65.5 + }, + { + "id": "cooling-35", + "kind": "cooling", + "start_s": 67.0, + "end_s": 100.0 + } + ], + "hotspot_sensor_ids": [ + "package:powered_hotspot" + ], + "control_sensor_ids": [], + "crossing_thresholds_k": [ + 301.0, + 330.0 + ], + "crossing_thresholds_semantics": "legacy_v1 diagnostic probes retained for comparison; not device limits", + "crossing_min_time_s": 0.2, + "crossing_fraction": 0.05, + "window_derivation": { + "basis": "union of source activity [start_s,end_s] before output inspection; cooling is complement in full interval", + "event_count": 484, + "unique_source_intervals": 68, + "merged_excitation_intervals": 34, + "cooling_intervals": 35 + }, + "sensor_role_derivation": { + "grid_hotspot": "package:powered_hotspot has reduction=max over all powered components", + "control": "no active P2 CSV consumer; stack::hotspot is a readonly semantic candidate matching CpuService max-over-stack only" + }, + "source_contract": { + "generated_case": "campaign-R03-train-1mm-20ms", + "events_sha256": "b1674c3ef83b7373d98eb55317d8109d3e55b00bbb057eefc1cf7b6614cf980e", + "reference_sensors_sha256": "38b3c02cec9c0f5842e8133fa17c2590e2fbb2c1b55d2a066be11943c67e96b2", + "sensor_count": 275, + "all_hotspot_reductions_are_max": true, + "new_output_inspected_for_window_selection": false + }, + "candidate_control_sensor_ids": [ + "stack:hbf0:hotspot", + "stack:hbf1:hotspot", + "stack:hbf2:hotspot", + "stack:hbf3:hotspot", + "stack:hbm0:hotspot", + "stack:hbm1:hotspot", + "stack:hbm2:hotspot", + "stack:hbm3:hotspot" + ], + "candidate_control_sensor_status": "DECLARED_READONLY_CANDIDATE_NOT_ACTIVE" +} diff --git a/docs/52-main-gpu-thermal-integration/README.md b/docs/52-main-gpu-thermal-integration/README.md new file mode 100644 index 0000000..b80e2df --- /dev/null +++ b/docs/52-main-gpu-thermal-integration/README.md @@ -0,0 +1,101 @@ +# Main GPU verification and thermal integration + +The review starts at upstream `78125a5`, with thermal work at `d471697`. The +consume-wait hybrid sleep is inherited in both. Tests found no evidence requiring +a production timing change: known-target sleeps clamp against the earlier of +ready time and deadline; unknown host completions retain backoff; binding and +liveness are checked again after a nap. Deliberate latency/transfer overlap and +separately labelled analytic sum models remain intact. + +Actual RTX5090 checks pass for pending-to-ready and exact values, deadline before +ready, shutdown during a wait, generation invalidation during a wait, native +bypass, and two 16-lane groups with 32 unique lane completions. Generation +invalidation deliberately cannot release a now-unvalidated owner's counter; +the test does not mislabel this ownership rule as a leak. The test loads the +production helper as PTX, matching its module boundary. This is not proof of +end-to-end application speedup, host-mapped polling traffic, or physical HBF timing. + +The upstream build ran 94 CTest cases successfully. The external historical +calibration CSV is unavailable: its provenance reproduction is explicitly skipped, +while the committed profile contract and generated-fixture software checks run. +Set `HBFSIM_VMEM_SOURCE_CSV` to the original file to perform its checksum and +byte-for-byte reproduction check; an explicit bad path or wrong hash still fails. +No synthetic source replaces the original measurement. + +Review fixes are limited to tests and reproducibility: + +- Raw PTX coverage includes both `ld.param.u64` and the unrewritable global load; + the plugin's filtered coverage is a different quantity. The regression now + asserts both exact opcodes and keeps checks active in Release builds. +- Compiled PTX must retain a sleep cycle that reloads time and control; poll must + remain nonblocking. Removing sleep fails a negative fixture. +- Optional CUDA/glibc header compatibility uses a task-local include copy. Only + two declaration exception specifications change. Neither the installed toolkit + nor drivers are edited. Original failure is documented in NVIDIA's tracker: + https://forums.developer.nvidia.com/t/cuda-headers-in-crt-math-functions-h-still-broken-in-debian-13-repo/362708 +- Use one compatible C++ host compiler for both CXX and NVCC host compilation. + Mixing GCC15-built core objects and GCC13 CUDA host linking failed here. + +The existing cache-frame eviction/retirement race described in `docs/51` remains +an explicit limitation requiring a protocol design. No eviction-safety or full +application qualification claim is made by these tests. + +## Reproduce the bounded live test + +Use a supported CUDA13/sm120 machine with CUDA driver access. Choose the installed +paths locally; these commands do not install or alter system dependencies. + +```sh +export CUDA_ROOT=/path/to/cuda-13 +export HOST_CXX=/path/to/g++-13 +mkdir -p build/live-review +# Only needed on the affected CUDA13.1 + recent glibc combination: +python3 scripts/build/prepare_cuda_glibc_overlay.py \ + --include "$CUDA_ROOT/targets/x86_64-linux/include" \ + --output build/live-review/cuda-include +export NVCC_PREPEND_FLAGS="-I$PWD/build/live-review/cuda-include" +"$CUDA_ROOT/bin/nvcc" -std=c++20 -O2 -arch=compute_120 -ptx \ + -ccbin="$HOST_CXX" -DHBFSIM_LIVE_DEVICE_IMAGE=1 \ + -Iinclude -Isrc/cuda_runtime/device tests/gpu/timing_future_wait_live_test.cu \ + -o build/live-review/wait.ptx +"$HOST_CXX" -x c++ -std=c++20 -O2 -Iinclude -I"$CUDA_ROOT/include" \ + tests/gpu/timing_future_wait_live_test.cu -L"$CUDA_ROOT/lib64" -lcuda \ + -o build/live-review/wait +# Check GPU availability and leave capacity for existing services first. +timeout --signal=TERM --kill-after=5s 30s build/live-review/wait build/live-review/wait.ptx +``` + +Dynamic test storage is below 7MiB; CUDA context/JIT overhead is additional. +Tests use one GPU serially with a process watchdog. Upper timing-performance +thresholds are intentionally absent on a shared GPU. Lower timing bounds, +Pending observations, values, states and accounting are asserted. + +## Integrated thermal validation + +The clean merge retains upstream main and all local thermal commits. No production +CUDA runtime, PTX transformer, or future ABI source differs from the live-tested +main. The original MQSim submodule is unchanged; isolated maintenance has its own +copied source, headers, archives and executable. The existing read-only command +observer remains an optional adapter in the normal patched build. + +After integration, 94 main CTest cases pass again. Thermal C++3, system-thermal +Python111, rate-thermal27, maintenance70, isolated-native backend1, persistent +coupled-thermal5, and native codec/observer12 checks pass. The tools suite runs210 +checks with9 initial skips; the7 native-codec skips are covered by the explicit +rebuilt native test, while2 checks requiring historical generated model inputs +remain not exercised in this fresh worktree. No previous P2 failure is promoted +to PASS by these software checks. + +Process-level A/B passes with12 requests and zero time tolerance: request/raw/ +reported completions, actual native phase order, physical addresses and existing +service statistics are equal with maintenance disabled. Each executable links +exactly one MQSim engine. The audit now accepts both Ninja and Makefile evidence +for the isolated target, and rejects cross-linking either engine into the other. +Actual maintenance JSON protocol checks cover successful commit, failed work +without age reset, and expiry rejection. + +This integration does not complete the outstanding ECC/refresh data-identity +mapping, all research ablations, physical calibration, GPU+GDDR inference, or +application-level throughput validation. Historical thermal raw and failures stay +in the external campaign directory; only source, small fixtures and compact +software-verification receipts are published here. diff --git a/docs/52-main-gpu-thermal-integration/evidence/integration.json b/docs/52-main-gpu-thermal-integration/evidence/integration.json new file mode 100644 index 0000000..20f386c --- /dev/null +++ b/docs/52-main-gpu-thermal-integration/evidence/integration.json @@ -0,0 +1,59 @@ +{ + "integration_commit": "6a9893b9bb58d1cd8311ac073fa8b586ca67aebf", + "upstream_main": "78125a5f30f8db8c958fc0f36de74057f7e84b9e", + "main_ctest": { + "tests": 94, + "failures": 0 + }, + "thermal_cpp": { + "tests": 3, + "failures": 0 + }, + "system_thermal_python": { + "tests": 111, + "failures": 0 + }, + "rate_thermal_python": { + "tests": 27, + "failures": 0 + }, + "maintenance_python": { + "tests": 70, + "failures": 0 + }, + "thermal_tools_python": { + "tests": 210, + "failures": 0, + "initial_skips": 9, + "native_codec_skips_closed_by_separate_test": 7, + "remaining_generated_historical_inputs_not_supplied": 2 + }, + "native_codec_and_observer": { + "tests": 12, + "failures": 0 + }, + "isolated_maintenance_cpp": { + "tests": 1, + "failures": 0 + }, + "persistent_coupled_thermal": { + "tests": 5, + "failures": 0 + }, + "native_maintenance_protocol": "PASS", + "native_default_isolated_ab": { + "status": "PASS", + "time_tolerance_ns": 0, + "request_count": 12, + "compared": [ + "request_and_raw_and_reported_completion", + "native_phase_order", + "native_physical_address", + "existing_service_statistics" + ] + }, + "gpu_helper_source_after_merge": "IDENTICAL_TO_LIVE_TESTED_MAIN; no production CUDA/PTX/future ABI changes", + "upstream_mqsim_submodule": "51f0f2d3fed92d88ef4a0fa61a38024b07bf9d16_UNMODIFIED", + "gpu_after": "RTX5090 0% utilization30C; same vLLM pid16200 and27732MiB allocation; test exited", + "scope": "SOFTWARE_INTEGRATION_ONLY_NOT_P2_MODEL_FREEZE_OR_NEW_THERMAL_RESEARCH_RESULTS" +} diff --git a/docs/52-main-gpu-thermal-integration/evidence/live-main.jsonl b/docs/52-main-gpu-thermal-integration/evidence/live-main.jsonl new file mode 100644 index 0000000..54a55eb --- /dev/null +++ b/docs/52-main-gpu-thermal-integration/evidence/live-main.jsonl @@ -0,0 +1,7 @@ +{"case":"ready","issue":0,"poll":0,"status":1,"state":4,"value":2881486848,"elapsed_ns":1996608,"begin_ns":1789914292222228064,"target_ns":1789914292224219648,"issued":1,"pending":0,"ready":1,"consumed":1,"errors":0,"groups_issued":1,"groups_completed":1} +{"case":"timeout","issue":0,"poll":0,"status":5,"state":5,"value":2881486848,"elapsed_ns":1996320,"begin_ns":1789914292224469120,"target_ns":1789914292226461184,"issued":1,"pending":0,"ready":0,"consumed":0,"errors":1,"groups_issued":1,"groups_completed":0} +{"case":"shutdown_midwait","issue":0,"poll":0,"status":7,"state":5,"value":2881486848,"elapsed_ns":2046496,"begin_ns":1789914292226675392,"target_ns":1789914292246667456,"issued":1,"pending":0,"ready":0,"consumed":0,"errors":1,"groups_issued":1,"groups_completed":0} +{"case":"generation_midwait","issue":0,"poll":0,"status":6,"state":5,"value":2881486848,"elapsed_ns":1870944,"begin_ns":1789914292228946464,"target_ns":1789914292248938464,"issued":1,"pending":1,"ready":0,"consumed":0,"errors":0,"groups_issued":1,"groups_completed":0} +{"case":"native","issue":1,"poll":1,"status":1,"state":4,"value":324508639,"elapsed_ns":4576,"begin_ns":1789914292231026752,"target_ns":1789914292231022080,"issued":0,"pending":0,"ready":0,"consumed":0,"errors":0,"groups_issued":0,"groups_completed":0} +{"case":"mixed_groups","issue":0,"poll":0,"status":1,"state":4,"value":2881486848,"elapsed_ns":1992960,"begin_ns":1789914292231236672,"target_ns":1789914292233225664,"issued":32,"pending":0,"ready":32,"consumed":32,"errors":0,"groups_issued":2,"groups_completed":2} +{"status":"PASS","device_storage_upper_mib":7} diff --git a/docs/52-main-gpu-thermal-integration/evidence/review.json b/docs/52-main-gpu-thermal-integration/evidence/review.json new file mode 100644 index 0000000..e4350cc --- /dev/null +++ b/docs/52-main-gpu-thermal-integration/evidence/review.json @@ -0,0 +1,27 @@ +{ + "upstream": "78125a5f30f8db8c958fc0f36de74057f7e84b9e", + "thermal_source": "d471697d4bd23bf8102743a5e598bc5ed34ac14b", + "environment": "CUDA13.1.115, GCC13.4.0, CMake4.2.3, glibc2.43, RTX5090, driver610.57.04", + "main_ctest": { + "passed": 94, + "failed": 0, + "wall_s": 49.5, + "nested_unittest_skip": "External original vmem calibration CSV is unavailable; committed artifact contract and five other software tests run" + }, + "live_gpu": { + "cases": 6, + "status": "PASS", + "receipt": "live-main.jsonl", + "limits": "Real production device helpers; tiny device-memory control fixture, not host PCIe contention or full transformed application throughput" + }, + "preserved_failures": [ + "unmodified CUDA/glibc rsqrt declaration mismatch at compiler identification", + "GCC15 core/GCC13 CUDA host ABI link mismatch", + "standalone direct-host link of PTX-only device symbols", + "stale PTX raw coverage count expectation", + "external calibration source unavailable", + "PATH selected unrelated old CMake" + ], + "runtime_changes": "NONE_REQUIRED_BY_REVIEW", + "known_limitation": "Cache frame retirement ownership protocol remains unchanged; existing eviction race not claimed fixed" +} diff --git a/docs/eq3_thermal/A2_IDEAL_MAINTENANCE_REPLAY_DESIGN.md b/docs/eq3_thermal/A2_IDEAL_MAINTENANCE_REPLAY_DESIGN.md new file mode 100644 index 0000000..dc605ed --- /dev/null +++ b/docs/eq3_thermal/A2_IDEAL_MAINTENANCE_REPLAY_DESIGN.md @@ -0,0 +1,54 @@ +# A2 ideal-independent maintenance replay + +This optional experiment wrapper is a fixed-intent counterfactual supplement +for taskbook section 8.3. It is not actual shared-MQSim maintenance and does not +replace the endogenous-trigger A2 arm. + +`build_ideal_maintenance_bundle.py` binds one completed source point's profile, +stack map, foreground input, maintenance intents, backend state facts, native +read/program phases, and terminal completions. `ideal_maintenance_replay.py` +proxies one real foreground MQSim service, intercepts exact matching maintenance +calls, and releases the frozen facts on their original simulation timestamps. +No maintenance request reaches the proxied engine and replay commits never +modify its mapping. + +The wrapper requires identical input hashes, rejects HBF writes, requires every +maintenance call and submission time to match the frozen source, and fails +closed if the counterfactual control trajectory makes fixed replay incompatible. +Command and transaction identities use `A2R-C` and `A2R-T`; +the source identities remain explicit fields. Counterfactual source and commit +versions are `UNKNOWN_REPLAY`; baseline versions remain provenance only. Age is +reset in the virtual experiment ledger only when the source completion observed +a real mapping commit. + +The proxied service receipt continues to report zero actual backend maintenance. +A separate nested receipt reports replay issue/completion conservation and the +fact that the current MQSim mapping was not changed. + +## Maintenance energy source-label repair proposal + +The service emits `maintenance_request_id`, while +`ActivityEnergyLedger._source` currently checks `maintenance_id`. Consequently, +the PILOT03 maintenance energy is present in the component/window totals but is +labelled `BACKEND_BACKGROUND`. This changes evidence attribution, not energy or +temperature. + +The proposed later patch, after the frozen main run, is limited to `_source`: + +```python +maintenance = tr.get("maintenance_request_id") +if maintenance in (None, 0, "UNKNOWN"): + maintenance = tr.get("maintenance_id") +if maintenance not in (None, 0, "UNKNOWN"): + return "HBF_MAINTENANCE" +``` + +The fixed reproduction is `repro_maintenance_energy_source_label.py`. Historical +raw rows remain unchanged; they can be derived-reclassified by native maintenance +identity without rerunning thermal integration because the total component +energy is unchanged. + +PILOT03 also retained an old derived `cleanup_failed=true` inconsistency for +`COMMITTED_RECLAIM_DEFERRED`. The bundle builder uses the backend completion +object (`mapping_committed=true`, `source_retired=true`, `erase_completed=false`) +and does not copy that obsolete top-level derived flag. diff --git a/docs/eq3_thermal/ACCEPTANCE_V2.md b/docs/eq3_thermal/ACCEPTANCE_V2.md new file mode 100644 index 0000000..84c0bf6 --- /dev/null +++ b/docs/eq3_thermal/ACCEPTANCE_V2.md @@ -0,0 +1,172 @@ +# EQ3 D2 数值验收 v2 冻结说明 + +状态:方法和输入派生规则已冻结,尚未用本方法读取新的求解输出。该分析只比较 +fast candidate 与一份已经选定的 reference;它不使 reference 获得 0.25 K +离散资格,也不构成 `MODEL_FREEZE`。 + +## 主判据 + +对每个预登记传感器和每个预登记窗口,reference 幅值、实际时间加权误差及上限为 + +``` +A = max(T_ref) - min(T_ref) +MAE_tw = integral(abs(T_fast - T_ref), dt) / window_duration +L = min(1 K, 0.25 K + 0.05*A) +``` + +实现对相邻观测的有符号误差作线性插值,并精确积分其绝对值;若误差在区间内变号, +在零点分成两个三角形,不能用端点绝对误差的梯形高估。窗口边界若落在两个观测 +之间,也按相同线性模型插值。reference 和 candidate 必须具有与方法文件完全相同 +的传感器集合,并具有彼此完全相同的时间戳,否则拒绝评分。每个窗口分别要求 +`MAE_tw <= L`。`package:powered_hotspot` 是由实际网格对所有有功组件取 max 的全局 +热点传感器,最大绝对误差还须不超过 2 K。本次离线 P2 sensor CSV 没有被控制器 +实际消费,因此 `control_sensor_ids` 为空;不能把名字/归约语义相似冒充 active +consumer。与 `CpuService::stack_temperature` 的 max-over-stack 语义对应的八个 +`stack::hotspot` 仅列为 `candidate_control_sensor_ids`,状态为 +`DECLARED_READONLY_CANDIDATE_NOT_ACTIVE`。将来只有实际接入后才可移入控制传感器集, +并应用 2 K 最大绝对误差条件。 + +能量条件独立检查 +`abs(Ein - Eboundary - dU) / max(Ein, 1 J) <= 0.001`。分析器不运行求解器, +也不修改任何 raw。 + +## legacy_v1 与阈值歧义 + +`legacy_v1` 是单独结果,不参与 v2 veto。它严格保持历史全轨迹定义:CSV 中的 +实际样本作算术平均,不把合成初态放入 MAE;NMAE 分母为 +`max(max(Tref)-min(Tref), 1 K)`;上限仍为 MAE 1 K、NMAE 5%、名字含 +`hotspot` 的传感器最大误差 2 K,并保留原穿越时序公式。v2 窗口中同时给出的 +算术 MAE 只用于对照,不冒充历史全轨迹评分。能量结果与 legacy_v1 并列报告, +不改写历史比较器的状态。 + +301 K 和 330 K 在这两份历史输入的方法文件中仅为保留的诊断探针,不是器件温限。 +初态明确来自本任务模型的 300 K。若 reference 有两个相邻观测都落在阈值 +`+/-L` 内,结果另记 `THRESHOLD_AMBIGUOUS`。这条“两相邻点”规则是本地工程推定, +不是用户或文献给出的物理标准;它不自动使整个数值模型失败,也不能据此声称阈值 +时序确定正确或系统安全。原穿越时序本身若失败,仍使对应 v2 窗口失败。 + +## 预登记窗口和传感器 + +冻结文件为: + +- `configs/eq3_thermal/acceptance_v2/train100s.json`:完整 0--100 s;从原事件文件 + 在查看新输出前合并得到 34 个 source-on 区间;其在 0--100 s 内的 35 个补集 + 区间分别作为 cooling 窗口。因此长尾冷却不能稀释任一受激窗口。 +- `configs/eq3_thermal/acceptance_v2/development64s.json`:完整 0--64 s;source-on + 为 0--32 s,cooling 为 32--64 s。 + +两份文件都锁定 275 个传感器 ID、事件与传感器定义原件 SHA-256、窗口派生计数、 +初态、热点及控制传感器角色。train 和 development 的传感器定义集合相同。 +这些窗口来自源活动 `[start_s,end_s]` 的并集及其补集,不依赖温度输出。 + +冻结时文件 SHA-256: + +| 文件 | SHA-256 | +|---|---| +| `tools/eq3_acceptance_v2.py` | `51bd993bc1e6c33627ff2c2a162cedc663a332ed12732189963afc68312f97df` | +| `tools/test_eq3_acceptance_v2.py` | `4213ddf68595c11d0fea35b377f99eb5bb28e7ef40b3a5215d728c990746db7f` | +| `train100s.json` | `6bf88c4003bf43c8490a3f4803fbe9071e545d447fb77e54a307202ce9e3469a` | +| `development64s.json` | `1cfcf88c1716d7a3c4e3584191f0fb43ed52c2df6b39ad10b947680eecb98424` | + +固定测试覆盖上下界、低温升、非等间隔时间轴、恒温、受激窗口不被冷却稀释、 +热点/控制 2 K 上限、有符号误差区间内过零、能量独立失败、阈值歧义、丢帧、漏传感器、预登记传感器 +缺失、单位突变,以及 full/excitation/cooling 三类窗口缺失。首次 9/9 通过的收据 +位于 `plans/decision-execution-v2/points/d2-d4-fixed-python/`;加入 exact historical +legacy、预登记集合检查和有符号误差过零积分后的最终复验须由统一串行入口产生新收据, +不能沿用旧哈希。 + +reference 的空间/时间相邻差仍分别以 0.25 K 目标审核。旧结果的任何 v2 重分析 +必须标 `RETROSPECTIVE_V2_REVIEW`,不回填或改写旧 `NUMERICAL_FAIL`。 + +## D3 all-source steady cap 最小实现(待固定验证) + +D3 在既有 `eq3_campaign_rc_runner` 内增加默认关闭的 `--steady-envelope` 模式,复用 +同一个 model/events parser、节点所有权、稀疏矩阵类型和分解后端。它只新增内部 +CLI/输出路径,不改 thermal core 公共 ABI、瞬态积分、checkpoint 或正常运行语义。 +现有 `--equilibrium-diagnostic` 仅在瞬态矩阵 `C/dt+L` 上验证零源等温不动点, +可以复用其验证方式,但它没有组装稳态 `L` 或求 all-source cap,不能直接替代 D3。 + +最小计算为:从模型静态功率形成 `P_fixed`;按每个允许 source/group 在全输入中的 +逐节点最大映射求上限,再跨组求和形成 `P_cap`,避免只取实际同一时刻总功率而低估 +“所有允许组可达上限”的包络。构造只含边和边界散热的稳态 `L`,先验证至少一个 +散热出口、非负源、对称正热网络和可分解性。为同时覆盖非统一边界温度,实际分别解 +`L*T_fixed=P_static+G_boundary*T_boundary`、`L*theta_cap=P_cap`。对 +`{1,.75,.5,.25}` 逐个仅计算 `T_fixed+alpha*theta_cap`,选择不超过 380 K 的最大 alpha;没有候选 +则返回 `DOMAIN_REDESIGN_REQUIRED`。同时逐节点检查初态是否被该包络覆盖。 + +收据需保存每个 source/group 的 cap、逐节点合成 cap、矩阵规模/非零元、分解后端、 +残差、环境/初态假设、最大温度节点、四个候选结果和选中 alpha。输出状态只能是 +`PREDICTED_ENVELOPE`,直到原生 reference/RC 瞬态验证完成。实现只修改 +`tools/eq3_campaign_rc_runner.cpp` 的私有 CLI 和矩阵组装分支。 +`tools/eq3_all_source_cap.py` 从冻结的 `calibration_power.json` 读取 17 个公共 group +cap(合计 840 W),用既有 `eq3_layered_ir.normalize` 将 group 权重映射到 component, +再按既有 RC grid 的真实 cell 体积分配到 node。它不读取 train/development/blind +温度轨迹;生成 receipt 保存 group/component 守恒和原件 SHA-256。当前代码尚未 +编译或运行固定测试,也没有执行实际 cap/steady 计算。 + +建议的最小运行矩阵只有两项,且不读取 blind:先在已生成的 development 事件与 +generator/source ledger 上做 `CAP-EXTRACT`,逐组保存 cap 与合成守恒,预计单核、 +600 s 内、RSS 小于 2 GiB、派生输出小于 100 MiB;再在既有 2 mm/64512 节点 RC +模型上做一次 `STEADY-ENVELOPE-2MM`,单次 factor、两个 RHS 和四个 alpha 的纯代数 +评估。既有同规模瞬态收据显示 factor 13.218 s、`factor_L_nnz=19,798,141`;因此 +保守申请单进程 12 GiB、任务 16 GiB、600 s、点磁盘 1 GiB,并在启动前按实时余量 +复核。这只是资源估计,不从瞬态总耗时外推稳态必然完成。cap 来源必须是冻结 source +ledger 的 17 组允许上限及 generator 的逐节点权重;若 events 无法无歧义恢复组身份, +该点返回 `CAP_SOURCE_IDENTITY_BLOCKED`,不能按 activity ID 或热点位置猜组。 + +固定验证通过后的派生命令分两步。先对已有 2 mm RC grid 以及 1 mm reference grid +分别生成 cap events/receipt,并核对两种网格的 group/component 总量都为 840 W; +这一步不求解,用于排除粗网格源映射丢失。1 mm reference-grid events 不能交给节点 +身份不同的 RC model。随后只先对 2 mm RC model 和匹配的 2 mm cap events 调用: + +``` +eq3_all_source_cap.py --profile candidate_profile.json --power calibration_power.json \ + --rc-grid /rc_grid-or-reference_grid.json --events-output /cap_events.txt \ + --receipt-output /cap_receipt.json + +eq3_campaign_rc_runner --model <2mm>/model.txt --events /cap_events.txt \ + --step-s 0.5 --slot-s 0.5 --end-s 0.5 --sample-s 0.5 \ + --min-k 300 --max-k 400 --envelope-limit-k 380 --steady-envelope +``` + +第二条命令的 step/slot 参数只满足既有输入调度验证;steady 模式组装的是不含 +`C/dt` 的 `L`,不会执行瞬态 workload。reference 未满足 0.25 K 时输出始终保持 +`PREDICTED_ENVELOPE`/`reference_qualified=false`。2 mm 若选出 alpha,仍需评估同源 +1 mm explicit-RC model 生成和 steady solve 的资源并核对结果;不得把 reference-grid +events 错配给 2 mm model,也不得假定粗网格包络不会低估热点。 + +`tools/eq3_domain_v2_input.py` 只读通过固定验证的 steady receipt,从候选顺序中复核 +`selected_alpha` 确为最大合格值,然后一次性派生 train 和 development。它把 17 组 +cap 和两个 trace 的每个非负功率项统一乘同一个 alpha;模型静态源不在 power trace +内,因此保持不变。输出删除其它 trace,receipt 明确 `blind_trajectory_read=false`。 +未来 blind 解封后只能应用 receipt 中同一个 alpha,不允许重选;当前工具主动拒绝 +blind trace 请求。代码及固定测试已准备,但尚未运行。 + +域内输入和两条完整生成命令预登记如下,当前不执行大 grid 生成或 solver: + +``` +python3 -B tools/eq3_domain_v2_input.py \ + --power configs/eq3_thermal/research/calibration_power.json \ + --steady-receipt --trace train --trace development \ + --output /power.json --receipt-output /input_receipt.json + +python3 -B tools/eq3_layered_export.py \ + --profile configs/eq3_thermal/research/candidate_profile.json \ + --power /power.json --trace train --mesh-um 2000 --rc-mesh-um 2000 \ + --step-s 0.02 --sample-s 0.1 --output /train + +python3 -B tools/eq3_layered_export.py \ + --profile configs/eq3_thermal/research/candidate_profile.json \ + --power /power.json --trace development --mesh-um 2000 --rc-mesh-um 2000 \ + --step-s 0.02 --sample-s 0.1 --output /development +``` + +基础闭环先让 native reference 与 RC 都使用相同 2 mm 网格,复用既有材料、边界和 +映射,避免在 reference 已知未满足 0.25 K 时重复昂贵 1 mm 而没有新的机制信息。 +该闭环即使通过也只能记 `REFERENCE_UNQUALIFIED / DISCRETE_EQUIVALENCE_PASS`。 +生成后,RC 的 full train/development 分别使用同目录匹配的 `model.txt/events.txt`, +`step=0.02 s`、`slot=0.5 s`、`sample=0.1 s`、`end=100/64 s` 和原 300--400 K +逐步域检查。native reference 使用生成目录的 `package.stk` 与 floorplans,仍由现有 +reference launcher 执行;不能用 RC 通过代替 native reference。只有前置 reference +空间资格出现足够改善或 2 mm 域内结果提出新的细化问题时,才把 `--mesh-um` 改为 +1000 做可选后继;alpha、输入映射和 v2/0.25 K 判据保持不变。 diff --git a/docs/eq3_thermal/ADDITIONAL_RATE_DIAGNOSTICS_FREEZE.md b/docs/eq3_thermal/ADDITIONAL_RATE_DIAGNOSTICS_FREEZE.md new file mode 100644 index 0000000..8f2f08b --- /dev/null +++ b/docs/eq3_thermal/ADDITIONAL_RATE_DIAGNOSTICS_FREEZE.md @@ -0,0 +1,35 @@ +# Additional rate diagnostics freeze + +Status: `PENDING_DEPENDENCIES_BASE_MATRIX`; no thermal solve has started. + +The prepared stage contains nine new feedback-policy points, each with 20 s of +input and 10 s of recovery at 1.536 TB/s offered per HBF stack: + +- four topologies with the same total offered bytes placed on the first half of + each stack's channels; +- four topologies with the same aggregate offered bytes concentrated on + `hbf0`; +- one mixed-direct point with the same uniform input and the registered + no-cross-domain-lateral thermal ablation. + +The corresponding four uniform feedback points are exact frozen-base +comparators and are not rerun. The index also registers read-only comparisons +between mixed 4x1.536 and all-HBF 8x0.768 TB/s across three policies (same total +offered demand), and between mixed/all-HBF at 1.536 TB/s per stack (different +total demand and package geometry). These byte-pressure scenarios do not +represent token throughput or native MQSim throughput. + +All nine new points use the isolated shared-HBM endpoint adapter. The index +locks the entry point, adapter, base runner, service, energy mapping, thermal +client, binary, configs, and model files. Execution remains blocked on the +base matrix completion receipt. The new output allocation is 4.5 GiB; with +30 GiB base and 27 GiB sensitivity allocations, the 80 GiB parent budget leaves +18.5 GiB for maintenance and causal work. + +Prepared evidence: + +- `plans/four-topology-system-v1/rate-diagnostics-v1/RATE_DIAGNOSTICS_INDEX.json` +- `plans/four-topology-system-v1/sensitivity-v2/SENSITIVITY_INDEX.json` + +The original `sensitivity-v1` inputs remain preserved and unexecuted. Version +2 changes only the future runner binding needed for the endpoint policy fix. diff --git a/docs/eq3_thermal/BASIC_FABRIC_DESIGN.md b/docs/eq3_thermal/BASIC_FABRIC_DESIGN.md new file mode 100644 index 0000000..4cdcd78 --- /dev/null +++ b/docs/eq3_thermal/BASIC_FABRIC_DESIGN.md @@ -0,0 +1,89 @@ +# Basic parameterized transfer fabric + +Status: approved minimal engineering implementation design, 2026-09-20. +The module is default-off and has no existing runtime consumer. Parameters are +`SCENARIO_ASSUMPTION`, not calibrated HBM/HBF or product values. + +## Boundary and interface + +`tools/eq3_basic_fabric.py` begins after a backend says media data are ready. +It does not issue NAND/DRAM commands, predict media readiness, change MQSim or +`CpuService`, call user code, or run a thermal solver. + +```python +fabric = BasicFabric(config) +fabric.enqueue(request_id, source_stack, route, bytes, ready_ns) +if fabric.reserve_source(request_id, source_stack, route, bytes, arrival_ns): + submit_to_backend(request_id) +fabric.mark_source_ready(request_id, backend_completion_ns) +fabric.advance(horizon_ns) +fabric.next_event_ns() # integer timestamp or None +fabric.completions() # immutable copies; polling, no callback/re-entry +fabric.resource_state() # read-only ownership snapshot for leak checks +fabric.immutable_facts() # normalized topology/parameters/limitations +``` + +Accepted transfers are HBF `direct`, HBF `relay`, and HBM `direct`. IDs are +globally unique. Time and byte fields are non-negative integer nanoseconds and +positive integer bytes. Every stage transports the request's original byte +count; completion occurs exactly once. + +`enqueue()` is the synthetic mode: its ready timestamp enters the modeled +shared-fill stage. A real command backend whose completion already includes +NAND data-out to the controller uses the bounded reservation mode instead. +`reserve_source()` occupies the lowest free source-base bank and returns +`True`; only then may the consumer submit the command. HBF commands reserve an +HBF bank. HBM local commands reserve the same HBM base-bank pool later used by +relayed HBF chunks, so local media output and relay receive both have bounded +storage. With both applicable banks occupied it returns +`False` without creating a request. `mark_source_ready()` publishes the backend +completion into that reserved bank and deliberately adds no fill latency or +energy. This prevents both duplicate transfer timing and unbounded buffering +of data already fetched by the backend. + +## State and resources + +An HBF request moves through: + +1. media-ready queue; +2. the stack's single parameterized shared-TSV/SRAM-fill resource; +3. one of two HBF SRAM banks; +4. either its independent GPU-HBF direct link, or its pair-private HBF-HBM + relay link into one of two HBM relay SRAM banks; +5. for relay, the paired HBM GPU link; then completion. + +An HBM local read enters the same HBM GPU-link queue used by relayed HBF data. +It does not consume HBF media or HBF SRAM. HBM local and relay traffic therefore +contend at the required shared link. + +Every HBF has its own fill/direct/relay resources and every HBM has its own GPU +link and relay banks. Pairing is one-to-one. There is no package-global lock. +Two ready HBF banks may drain concurrently through direct and relay links. +Only fill is shared; bank capacity and downstream availability still apply. + +Each link/fill stage has explicit `latency_ns` and `bandwidth_bytes_per_s`. +Duration is `latency_ns + ceil(bytes * 1e9 / bandwidth_bytes_per_s)`. HBF and +HBM bank counts and byte capacities are explicit configuration fields; fixed +tests use two banks, while validation does not silently invent values. + +## Determinism + +At one timestamp the engine performs, in order: + +1. complete every ending stage in request sequence order; +2. release its link/bank and expose the next state; +3. make backend-completion and synthetic source-ready arrivals visible; +4. admit eligible work in original enqueue order, using the lowest free bank. + +New zero-duration events are impossible because all bandwidths and transported +bytes are positive. Enqueue does not execute work; `advance()` is the sole event +progression boundary. Completion is polled after `advance()`, so no callback can +re-enter scheduling. + +## Evidence limits + +Fixed tests cover bytes/IDs, deterministic timestamps, bank backpressure, +same-HBF direct/relay overlap, shared HBM GPU-link serialization, pair +independence, event ordering, and invalid configuration. Passing them means the +composition contract works on CPU. It does not validate real SRAM sizes, +bandwidths, UCIe/HBM timing, arbitration policy, energy, or live MQSim behavior. diff --git a/docs/eq3_thermal/BASIC_HBM_SCENARIO.md b/docs/eq3_thermal/BASIC_HBM_SCENARIO.md new file mode 100644 index 0000000..8e1a111 --- /dev/null +++ b/docs/eq3_thermal/BASIC_HBM_SCENARIO.md @@ -0,0 +1,37 @@ +# Basic parameterized HBM CPU timing + +Status: implementation and fixed tests prepared; not executed while the serial R03 reference is +running. This module is a `PARAMETRIC_HBM_SCENARIO`, not a DRAM timing model or real backend. + +`tools/eq3_basic_hbm.py` provides one independent FIFO media server per configured stack. A read +or write uses + +``` +media_duration_ns = media_latency_ns[op] + + ceil(bytes * 1e9 / media_bandwidth_Bps[op]) +``` + +Both terms are explicit per-stack inputs. The repository contains a 2.048 TB/s raw-interface +design value derived from a 2048-bit interface and an 8 Gbit/s design point. Treating that raw +interface ceiling as media service bandwidth is therefore labeled +`DERIVED_WITH_TRANSFER_ASSUMPTION`, not a calibrated HBM media rate. No supported source supplies +the requested media latency. The opt-in example uses 100 ns for reads and writes and labels it +`SCENARIO_ASSUMPTION`; the class itself has no hidden latency or bandwidth default. + +The event API is `arrival(request)`, `submit(request_id, time_ns)`, `advance(target_ns)`, +`next_event_ns()`, `take_facts()`, and `take_media_completions()`. Arrival, submit, media start, +and media completion remain distinct facts. A media completion has the shared handoff fields +`phase=MEDIA_DONE`, `request_id`, `stack`, `bytes`, and `time_ns`, plus `requires_fabric=true`. +It is not a reported request completion. A separate fabric module owns GPU-link or cascaded-link +arbitration and delay, so link service is not counted here a second time. + +Die and plane remain `UNKNOWN`; supplied die/plane placement is rejected rather than inferred +from an address. With no energy parameter, media energy is JSON null and status is +`UNKNOWN_UNPARAMETERIZED`, never zero. Refresh returns `UNSUPPORTED_CAPABILITY`. The module does +not model channels, banks, row buffers, DRAM timing commands, refresh interference, QoS, or a +physical HBM energy model. + +`tools/test_eq3_basic_hbm.py` prepares fixed coverage for the latency/bandwidth formula, +same-stack FIFO order, cross-stack parallelism, event boundaries, fabric handoff, unknown +energy/die/plane, refresh capability, and invalid lifecycle/placement inputs. These tests still +require execution through the shared serial runner after R03 releases the CPU slot. diff --git a/docs/eq3_thermal/BASIC_SYSTEM_CPU_CHAIN.md b/docs/eq3_thermal/BASIC_SYSTEM_CPU_CHAIN.md new file mode 100644 index 0000000..480b8a1 --- /dev/null +++ b/docs/eq3_thermal/BASIC_SYSTEM_CPU_CHAIN.md @@ -0,0 +1,73 @@ +# Basic four-topology CPU chain + +Status: `USER_AUTHORIZED_ENGINEERING_FIXTURE`, default off. This is a small +composition consumer, not a production daemon, research geometry, calibrated +performance model, or thermal closure. + +`tools/eq3_basic_system.py::BasicSystem` connects the existing +`MqsimService`, `BasicHbm`, and `BasicFabric` APIs. It owns no backend command +or completion. Before submitting either backend it reserves one of the two +source base banks. A failed reservation leaves the request in the external +waiting list. The eventual backend submission timestamp and external wait are +recorded separately from the original arrival. + +HBF requests use persistent `stack` plus `stack_local_page` placement. The +MQSim request always carries backend `route=direct`, because that field selects +native HBF media placement. The requested package route remains in the system +record and is consumed only by the fabric. HBM local work reserves an HBM base +bank before entering the parameterized HBM media server. That bank pool is the +same pool used by the paired relay receive path. + +After reserving its HBF source bank, the coordinator uses the existing +nonblocking `try_submit` gate. A `Defer` result keeps that bounded reservation, +adds the target to the event horizon, and retries the same backend ID once the +target is reached. The gate owns no request and the backend receives it once. +Blocked or unsupported decisions stop this fixture as +`UNSUPPORTED_COMPOSITION`; they are not converted into fabricated service. + +The coordinator advances through `MqsimService.until(horizon)`. Its horizon is +the earliest known external arrival, HBM event, or fabric event. When only an +unknown-time MQSim completion remains, it supplies the maximum horizon and the +existing service returns the earliest actual completion without advancing past +it. At a returned timestamp the coordinator publishes backend completions, +advances the fabric, consumes package completions, and then retries external +waiters. There is no host sleep, background thread, callback re-entry, or +second completion lifecycle. + +For HBF, the observed raw MQSim media callback must equal the unchanged +reported completion. A difference means the generic aggregate-bandwidth bound +is active, so the coordinator returns `UNSUPPORTED_COMPOSITION` rather than +counting an unidentified transfer segment again. On equality, +`mark_source_ready()` starts the package fabric from that timestamp and final +delivery is `max(backend_reported, fabric_completion)`. + +The topology policy is explicit: + +| Mode | Legal paths | +|---|---| +| `all_hbf_direct` | HBF direct only; physical external GDDR identity is retained, while GDDR service and temperature are `UNAVAILABLE` | +| `mixed_direct` | configured HBF direct and HBM direct | +| `relay` | HBF relay through its paired HBM base plus HBM local direct; an HBF direct GPU path is rejected | +| `dash` | each configured HBF may select direct or paired relay; HBM local remains direct | + +The fixed tests use eight HBF stacks for all-HBF, and four HBF/four HBM stacks +for the other modes. Every configured source stack receives at least three +requests (DASH alternates four HBF direct/relay requests), so the two-bank +limit produces real external backpressure. They check per-stack counts, unique final completion, +backend/package route separation, finite-bank waiting, shared HBM resources, +and a final read-only ownership snapshot with no bank or link owner. These are +software fixtures. Capacity and the 12.8–24.5 TB/s target remain unvalidated; +HBM is `PARAMETRIC_HBM_SCENARIO`, missing energy remains `UNKNOWN`, and the new +fabric is not connected to the thermal model. The existing separate thermal +gate fixture remains the only basic CPU thermal-control chain. + +`tools/eq3_basic_system_actual.py` is the explicit actual-service runner. Given +an isolated `hbf_mqsim_service` binary, a base small profile and its explicit +eight-stack map, it creates a new artifact directory. Per case it preserves +byte-for-byte source copies and hashes, then derives a matching profile/map: +eight one-channel/one-die HBF groups for all-HBF and four such groups for the +three 4+4 modes. It runs one fresh MQSim process for each fixed topology, +preserves every service transcript/native event, and writes a case result plus +a four-case summary. It stops on the first failed composition +and retains that failure record; it does not turn the cases into a research +matrix or overwrite an existing artifact directory. diff --git a/docs/eq3_thermal/D3_ACCEPTANCE_V3_RETROSPECTIVE.md b/docs/eq3_thermal/D3_ACCEPTANCE_V3_RETROSPECTIVE.md new file mode 100644 index 0000000..aa7cbc0 --- /dev/null +++ b/docs/eq3_thermal/D3_ACCEPTANCE_V3_RETROSPECTIVE.md @@ -0,0 +1,26 @@ +# D3 acceptance v3 retrospective + +This is a read-only rescore of the immutable canonical D3 development trace. +It did not run a solver, alter raw CSVs, or replace the frozen v1/v2 result. +The reference observation receipt declares `0.001 K` field quantization. The +RC receipt declares no field quantization and emits derived double values, so +v3 uses reference `0.001 K` and candidate `0 K` intervals. + +| Method | Aggregate status | Temperature checks | Crossing result | Max TW-MAE K | Max absolute error K | Energy relative residual | +|---|---|---:|---|---:|---:|---:| +| v1 legacy | PASS | legacy full/sensor semantics | legacy full-trajectory probes pass | 0.000260921 worst legacy MAE | 0.000500001 | not separately changed | +| v2 frozen | PASS_WITH_THRESHOLD_AMBIGUITY | 825/825 pass | 47 PASS, 778 THRESHOLD_AMBIGUOUS, 825 NOT_APPLICABLE across repeated window probes | 0.000214699 | 0.000500001 | 3.070e-11 | +| v3 retrospective | PASS | 825/825 pass | 275 PASS, 275 NOT_APPLICABLE after one full-trajectory extraction per sensor/threshold | 0.000214699 | 0.000500001 | 3.070e-11 | + +V3 matches crossings globally before assigning half-open windows. Development +has no remaining quantization-indeterminate crossing. This does not convert the +unqualified spatial reference into a qualified reference or a MODEL_FREEZE. +The existing train v3 result remains `INDETERMINATE_QUANTIZATION`: one +`hbm1.die8` 301 K up-crossing interval straddles the 39 s phase boundary. + +Evidence is under +`eq3_thermal/plans/isolated-maintenance-campaign-v1/points/D3-DEVELOPMENT-V3/`. +`manifest.json` binds both canonical CSVs, the energy receipt, method and tool; +`v3_score.json` retains all window/crossing records and `result.json` is the +compact comparison above. + diff --git a/docs/eq3_thermal/D3_MINIMAL_EXECUTION.md b/docs/eq3_thermal/D3_MINIMAL_EXECUTION.md new file mode 100644 index 0000000..ed9fd97 --- /dev/null +++ b/docs/eq3_thermal/D3_MINIMAL_EXECUTION.md @@ -0,0 +1,138 @@ +# D3 最小串行执行说明 + +状态:`PREPARED_NOT_EXECUTED`。本说明只冻结已获授权的最小执行顺序;尚未编译 +D3 runner、运行固定测试、求稳态或生成/求解 `DOMAIN_V2_IN_RANGE`。盲测轨迹不读取、 +不导出、不求解。 + +## 证据与复用边界 + +- `USER_CONFIRMED`:统一幅值候选 `{1,.75,.5,.25}`、全 17 组上限、380 K + 筛选上限、train/development 原生 reference 与 RC 完整轨迹,以及 D1 的串行 CPU、 + GPU0/cloud0 和逐点合理资源约束。 +- `DOC_DERIVED`:现存 + `/root/hbfsim-exp/eq3_thermal/generated/campaign-RC2MM-{train,development}` 的 + `model.txt` 与 `rc_grid.json` 分别逐字节相同(SHA-256 + `ecc7d24a...c62559`、`b8e611d0...e5a23`),可作为一次 2 mm steady envelope 的 + 匹配 model/grid。steady 解不使用其旧 workload events,也不依赖 `dt`。 +- 旧 RC2MM 输入的 generation receipt 是 5 ms,且不是新的统一 alpha 输入;旧 + R06 development 原生结果已越过 400 K。因此它们只能复用静态 model/grid 与失败 + 证据,不能冒充 D3 新轨迹。 +- D3 基础闭环重新导出 reference=2 mm、RC=2 mm、step=20 ms、sample=100 ms。 + 这只支持 `REFERENCE_UNQUALIFIED / DISCRETE_EQUIVALENCE`;不授予 0.25 K reference + 资格或 `MODEL_FREEZE`。1 mm 只在 2 mm 给出新细化问题时作为后继。 + +## 串行命令顺序 + +所有命令从 +`/root/hbfsim-exp/eq3_thermal/integration/main-20260920` 执行,由 +`eq3_thermal/plans/decision-execution-v2/run_check.py` 或既有完整 thermal launcher +保存元信息与限额。先等待当前 R03 退出;实际 point ID 不得复用已有目录。 + +1. 建立新的 build 输出目录后,用既有 fresh 静态库编译,不覆盖历史 binary: + +```text +/usr/bin/g++ -std=c++20 -O3 -DNDEBUG -Wall -Wextra -Wpedantic \ + -Iinclude -I/usr/include/eigen3 tools/eq3_campaign_rc_runner.cpp \ + /root/hbfsim-exp/eq3_thermal/plans/minimal-repair-v1/build/libhbfsim_eq3_thermal.a \ + -o /root/hbfsim-exp/eq3_thermal/plans/decision-execution-v2/build/rc_runner_d3 +``` + +2. 在 600 s fixed-test 点中运行: + +```text +env PYTHONPATH=tools \ + EQ3_CAMPAIGN_RC_RUNNER=/root/hbfsim-exp/eq3_thermal/plans/decision-execution-v2/build/rc_runner_d3 \ + EQ3_LAYERED_RC_RUNNER=/root/hbfsim-exp/eq3_thermal/build/converter-core-relocated/eq3_layered_rc_runner \ + EQ3_GENERATED_ROOT=/root/hbfsim-exp/eq3_thermal/generated \ + python3 -B -m unittest tools.test_eq3_all_source_cap \ + tools.test_eq3_domain_v2_input tools.test_eq3_campaign_rc_runner -v +``` + +3. 用公共 source ledger 的 17 个 cap 与现存匹配 2 mm grid 生成一次 cap events。 +输出放入新的 `d3-cap-2mm` point;此步不读任何温度轨迹: + +```text +env PYTHONPATH=tools python3 -B tools/eq3_all_source_cap.py \ + --profile configs/eq3_thermal/research/candidate_profile.json \ + --power configs/eq3_thermal/research/calibration_power.json \ + --rc-grid /root/hbfsim-exp/eq3_thermal/generated/campaign-RC2MM-development/rc_grid.json \ + --events-output /cap_events.txt \ + --receipt-output /cap_receipt.json +``` + +启动 steady 前核对 receipt:17 组均存在,group/component/node 守恒,总 cap 正好 +840 W,`blind_trajectory_read=false`;否则停止 D3 域分支。 + +4. 用同一 2 mm model/cap events 做一次 steady solve。三项 SHA-256 必须在 point +manifest 中先计算并原样传入,stdout 即唯一 steady receipt: + +```text + \ + --model /root/hbfsim-exp/eq3_thermal/generated/campaign-RC2MM-development/model.txt \ + --events /cap_events.txt \ + --step-s 0.5 --slot-s 0.5 --end-s 0.5 --sample-s 0.5 \ + --min-k 300 --max-k 400 --envelope-limit-k 380 \ + --model-sha256 \ + --events-sha256 \ + --runner-source-sha256 \ + --domain-version EQ3-DOMAIN-V2-IN-RANGE-v1 --steady-envelope +``` + +这里只做 `L` 的一次分解、两个 RHS、四个 alpha;0.5 s 参数只满足既有事件调度 +校验,不执行瞬态 workload。结果必须仍为 `reference_qualified=false`。若状态是 +`DOMAIN_REDESIGN_REQUIRED` 或 residual/正网络检查失败,保留 receipt 并停止域内 +轨迹分支,不另选更小 alpha。 + +5. receipt 为 `PREDICTED_ENVELOPE` 时,只派生 train/development: + +```text +python3 -B tools/eq3_domain_v2_input.py \ + --power configs/eq3_thermal/research/calibration_power.json \ + --steady-receipt /stdout.log \ + --trace train --trace development \ + --output /calibration_power.json \ + --receipt-output /input_receipt.json + +python3 -B tools/eq3_layered_export.py \ + --profile configs/eq3_thermal/research/candidate_profile.json \ + --power /calibration_power.json --trace train \ + --mesh-um 2000 --rc-mesh-um 2000 --step-s 0.02 --sample-s 0.1 \ + --output /train + +python3 -B tools/eq3_layered_export.py \ + --profile configs/eq3_thermal/research/candidate_profile.json \ + --power /calibration_power.json --trace development \ + --mesh-um 2000 --rc-mesh-um 2000 --step-s 0.02 --sample-s 0.1 \ + --output /development +``` + +核对两个新 normalized 输入只相对原 trace 统一缩放可变源,持续时间、窗口、源 +映射与静态源不变;receipt 必须明确 `blind_trajectory_read=false`。 + +6. 为这两个新 normalized 输入创建独立的 domain-v2 scope/authorization 派生记录, +绑定 steady receipt、冻结 alpha、scaled `calibration_power.json`、各自 normalized hash +和新能量。不要修改正在使用的 `scope-v1.json`、`authorization-v1.json` 或 R03 +manifest。既有 campaign gate 对 full `reference`/`rc` 可直接核对 scope 中的新 +normalized hash/能量;本分支不需要泛化 gate,也不安排 prefix pilot。 + +7. 串行执行四点:train native reference、train RC、development native reference、 +development RC。native 使用既有 `eq3_campaign_stream.py` 加 3D-ICE launcher:2 mm +网格为 32x32x63,train/development 分别 5000/3200 个 solver-step frame;RC 使用: + +```text + --model /model.txt --events /events.txt \ + --step-s 0.02 --slot-s 0.5 --end-s <100-or-64> --sample-s 0.1 \ + --min-k 300 --max-k 400 \ + --model-sha256 --events-sha256 \ + --runner-source-sha256 \ + --domain-version EQ3-DOMAIN-V2-IN-RANGE-v1 --run +``` + +每点从自己的空 cwd 启动,使 `rc_energy_receipt.json` 或失败诊断不会覆盖别点。 +native 必须由完整 safety launcher 绑定 `package.stk`、63 层 floorplan、stream backend、 +codec、raw 输出和 point manifest;不能用裸 3D-ICE 命令代替。P2 完整 reference/RC +可用 3600 s,单进程 12 GiB、任务 16 GiB、CPU/OMP/BLAS=1、GPU0;启动前按当前磁盘 +重新给 point 输入/raw/临时/派生估算并保留至少 10 GiB,不机械继承旧 R06 的 600 s。 + +四点都保留每步 300--400 K 域检查、唯一 receipt、完整能量和失败证据。任一点越域、 +输出不完整或 launcher gate 不通过就保留失败,不 clamp、不改 alpha、不解封 blind。 diff --git a/docs/eq3_thermal/D4_CONTROL_COST_REPORT.md b/docs/eq3_thermal/D4_CONTROL_COST_REPORT.md new file mode 100644 index 0000000..5a45d4d --- /dev/null +++ b/docs/eq3_thermal/D4_CONTROL_COST_REPORT.md @@ -0,0 +1,139 @@ +# D4 sustained-load control cost report + +## Result and evidence boundary + +The nine D4 runs are a deterministic **ENGINEERING_FIXTURE** comparison of +three policies at 5, 10, and 25 requests/s per stack. Every arm uses the same +`mixed_direct` package, the same four HBM4 plus four HBF stack identities, the +same initial state and maintenance rules, 20 s of arrivals, and 10 s of +recovery. Each rate has one run per arm, so there is no confidence interval. +Temperatures are simulated by the fixture and are neither a calibrated product +prediction nor a P2/P5 acceptance result. + +The result is negative for a simple “control improves the system” claim. Both +control policies reduce the recorded peak temperature, but they also complete +less foreground work. At 25 requests/s per stack they leave 1,500 of 4,000 +requests unfinished and raise completed-request p95 latency from 0.04 s to +10.34 s. The lower energy in controlled arms also accompanies less completed +work, so it is not an efficiency result. + +`none` means no thermal policy. `hyst` is the retained hysteresis policy and +`esc-v2` is the new, default-off `hysteresis_escalation_priority_v2` policy. +All foreground failures were zero and all foreground inflight counts at the +30 s horizon were zero; “unfinished” therefore means queued at the horizon. + +## Foreground, temperature, and latency + +`backend wait` is `start-arrival` for completed foreground requests. `backend +service` is `end-start`. `end-to-end` is `end-arrival`. These are completed-only +statistics; unfinished wait is reported separately rather than censored into +the latency distribution. `external wait` is the fixture's declared value and +is 0 s in every arm; this D4 fixture has no separate bounded external admission +queue. + +| rate/stack (rps) | policy | offered | complete | unfinished | completed bytes | peak K (node) | peak foreground queue | end-to-end p95 s | backend wait p95 s | backend service p95/max s | external wait s | unfinished wait max/sum s | +|---:|---|---:|---:|---:|---:|---|---:|---:|---:|---:|---:|---:| +| 5 | none | 800 | 800 | 0 | 3,276,800 | 300.673743 (`hbf1_die7`) | 4 | 0.020 | 0.000 | 0.020 / 0.020 | 0.000 | 0.000 / 0.000 | +| 5 | hyst | 800 | 672 | 128 | 2,752,512 | 300.635724 (`hbf1_die8`) | 128 | 0.040 | 0.020 | 0.020 / 0.020 | 0.000 | 16.400 / 1,702.400 | +| 5 | esc-v2 | 800 | 672 | 128 | 2,752,512 | 300.635724 (`hbf1_die8`) | 128 | 0.040 | 0.020 | 0.020 / 0.020 | 0.000 | 16.400 / 1,702.400 | +| 10 | none | 1,600 | 1,600 | 0 | 6,553,600 | 300.932365 (`hbf1_die6`) | 4 | 0.040 | 0.020 | 0.020 / 0.020 | 0.000 | 0.000 / 0.000 | +| 10 | hyst | 1,600 | 1,264 | 336 | 5,177,344 | 300.683902 (`hbf1_die7`) | 336 | 0.020 | 0.000 | 0.020 / 0.020 | 0.000 | 18.400 / 4,788.000 | +| 10 | esc-v2 | 1,600 | 1,264 | 336 | 5,177,344 | 300.683902 (`hbf1_die7`) | 336 | 0.020 | 0.000 | 0.020 / 0.020 | 0.000 | 18.400 / 4,788.000 | +| 25 | none | 4,000 | 4,000 | 0 | 16,384,000 | 301.592464 (`hbf2_die7`) | 4 | 0.040 | 0.020 | 0.020 / 0.020 | 0.000 | 0.000 / 0.000 | +| 25 | hyst | 4,000 | 2,500 | 1,500 | 10,240,000 | 300.741831 (`hbf1_die7`) | 2,000 | 10.340 | 10.320 | 0.020 / 0.020 | 0.000 | 23.440 / 24,183.360 | +| 25 | esc-v2 | 4,000 | 2,500 | 1,500 | 10,240,000 | 300.741831 (`hbf1_die7`) | 2,000 | 10.340 | 10.320 | 0.020 / 0.020 | 0.000 | 23.440 / 24,183.360 | + +The apparently lower p95 latency for the controlled 10-rps arms does not mean +better service: p95 is calculated only over the 1,264 completed requests, while +336 requests remain queued. The backlog is the required counterweight to that +completed-only statistic. + +## Maintenance age, backlog, and energy + +Maintenance counts below are operations, while `overdue cohorts` counts die +cohorts. `max overdue` is age beyond the per-kind fixture period. Package energy +is the sum of the recorded package component dynamic energies. External energy +is recorded separately and is zero for this mixed-direct case. + +| rate/stack | policy | maintenance done | failed | queued | inflight | overdue cohorts | max age s | max overdue s | peak maintenance queue | package dynamic J | external J | +|---:|---|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:| +| 5 | none | 3,148 | 0 | 4 | 0 | 4 | 2.020 | 0.020 | 64 | 114.978000 | 0.000000 | +| 5 | hyst | 3,148 | 0 | 0 | 4 | 4 | 2.040 | 0.040 | 64 | 113.703600 | 0.000000 | +| 5 | esc-v2 | 3,148 | 0 | 0 | 4 | 4 | 2.040 | 0.040 | 64 | 113.703600 | 0.000000 | +| 10 | none | 3,128 | 0 | 0 | 0 | 0 | 1.980 | 0.000 | 64 | 123.001200 | 0.000000 | +| 10 | hyst | 3,132 | 0 | 4 | 0 | 4 | 2.020 | 0.020 | 64 | 119.482000 | 0.000000 | +| 10 | esc-v2 | 3,132 | 0 | 4 | 0 | 4 | 2.020 | 0.020 | 64 | 119.482000 | 0.000000 | +| 25 | none | 3,084 | 0 | 4 | 0 | 4 | 2.020 | 0.020 | 64 | 147.240400 | 0.000000 | +| 25 | hyst | 3,112 | 0 | 0 | 0 | 0 | 1.980 | 0.000 | 64 | 132.126800 | 0.000000 | +| 25 | esc-v2 | 3,112 | 0 | 0 | 0 | 0 | 1.980 | 0.000 | 64 | 132.126800 | 0.000000 | + +Maintenance backlog is not monotone with foreground cost. At 10 rps the +controlled arms have four queued maintenance operations and four overdue +cohorts while `none` has neither. At 25 rps the controlled arms clear the +maintenance backlog but leave 1,500 foreground requests queued. Neither case +supports describing the controlled arm as an overall benefit. + +## Same offered distribution, different HBM/HBF service + +Every rate assigns the same number of requests to each of the eight stacks. +The per-stack result below is uniform within each four-stack kind. The different +completion counts therefore come from fixture media/maintenance/control +semantics, rather than an unequal offered distribution. HBF has 16 dies and +program/read/erase maintenance, while HBM4 has 12 dies and 5 ms fixture refresh; +control is applied per stack from its simulated hotspot. These are engineering +choices, not calibrated HBM or NAND timing claims. + +| rate/stack | policy | each HBM4 arrivals / complete / queued / inflight | each HBF arrivals / complete / queued / inflight | +|---:|---|---|---| +| 5 | none | 100 / 100 / 0 / 0 | 100 / 100 / 0 / 0 | +| 5 | hyst | 100 / 100 / 0 / 0 | 100 / 68 / 32 / 0 | +| 5 | esc-v2 | 100 / 100 / 0 / 0 | 100 / 68 / 32 / 0 | +| 10 | none | 200 / 200 / 0 / 0 | 200 / 200 / 0 / 0 | +| 10 | hyst | 200 / 200 / 0 / 0 | 200 / 116 / 84 / 0 | +| 10 | esc-v2 | 200 / 200 / 0 / 0 | 200 / 116 / 84 / 0 | +| 25 | none | 500 / 500 / 0 / 0 | 500 / 500 / 0 / 0 | +| 25 | hyst | 500 / 461 / 39 / 0 | 500 / 164 / 336 / 0 | +| 25 | esc-v2 | 500 / 461 / 39 / 0 | 500 / 164 / 336 / 0 | + +## Why the old and new policies are identical here + +For every rate, `hyst` and `esc-v2` have identical control event times, +completed request IDs, temperatures, service, maintenance, and energy. The +event audit records 16 identical transitions at 5 rps and 12 at both 10 and +25 rps. The v2 policy changes the case where a more severe sampled action would +otherwise wait behind recovery dwell; these trajectories do not expose that +case. In the only recovery followed by another escalation (5 rps), the HBF +stacks change Severe→Light at 12.48 s and Light→Severe at 13.42 s, a 0.94 s +gap versus the configured 0.10 s dwell. The old dwell was already satisfied, +so escalation priority cannot change the outcome. Equality here verifies an +unexercised policy distinction, not proof that the two algorithms are +equivalent. + +## UNKNOWN and unavailable fields + +- Token throughput is `UNAVAILABLE`; this CPU fixture does not model tokens. +- External GDDR temperature is `UNAVAILABLE_OUTSIDE_PACKAGE`; this topology has + no package GDDR node. That does not imply a physical external device has zero + temperature or board-level thermal effect. +- A separate measured external-admission wait distribution is `UNKNOWN`. The + fixture reports only the configured scalar `external_wait_s = 0` and keeps + backlog in its internal foreground queue. +- Real MQSim/NAND command-stage latency, calibrated HBM timing, measured device + energy, Ea/ECC/endurance behavior, and uncertainty across repetitions are + `UNKNOWN` for this matrix. + +## Provenance and retained historical results + +The aggregate values come from +`eq3_thermal/plans/decision-execution-v2/D4_MATRIX_RESULT.json`; per-stack counts +come from `D4_PER_STACK_SERVICE.json`; control equality comes from +`D4_POLICY_EVENT_DIAGNOSTIC.json`; completed-only latency, queue, maintenance, +and energy details come from the nine immutable point `review.json` files, with +service duration and peak node read directly from their corresponding +`stdout.log` raw summaries. This report only reads those completed JSON files; +it launches no pilot or solver. + +The prior six-point engineering results remain unchanged and retained in +`docs/eq3_thermal/P2_P3_P4_CAMPAIGN_RESULT.md`. Their inputs differ from this +nine-point sustained-load matrix, so they are historical evidence rather than +a substitute baseline and are not overwritten or reinterpreted here. diff --git a/docs/eq3_thermal/DECISION_EXECUTION_V2_RESULT.md b/docs/eq3_thermal/DECISION_EXECUTION_V2_RESULT.md new file mode 100644 index 0000000..ba60cfb --- /dev/null +++ b/docs/eq3_thermal/DECISION_EXECUTION_V2_RESULT.md @@ -0,0 +1,215 @@ +# EQ3-DECISION-EXECUTION-v2 实际结果 + +状态:本阶段执行完成,保留数值/模型能力限制;本任务无在途求解或后处理。 + +## 实际交付与边界 + +本轮从 repair/eq3-minimal-v1 的 aef7271 继续,保留原基线 +5eb789d5f1a42f0c040ee6fb5a2cdb5ffa0951d5。未 reset、push、merge,未改全局环境, +无 GPU 或云负载。旧失败、旧六点、原 development 400.911 K 越域结果原样保留。 +有效 AGENTS 已登记 D1–D6、最小侵入和用户后续授权。固定资源数字已被用户改为 +逐实验合理分配,运行收据中的 12/16 GiB 等是当次选择,不是永久不可变上限。 + +| 工作 | 实际结果 | 解释边界 | +|---|---|---| +| D1 原 train 完整 1 mm/20 ms/100 s | 求解及完整观察完成,求解 2763.60 s,观察 833.56 s,求解峰值采样 RSS 7,735,140 KiB | 完成不是参考精度通过 | +| D2 验收 v2 | 10 项固定测试通过,方法/窗口先冻结;v1 并列保存 | 参考未合格时不 MODEL_FREEZE | +| D3 统一功率域 | alpha=0.25;train/development 原生与 RC 四点完整完成 | 两轨迹温度/能量一致;冻结 v2 train 分段穿越计数未通过 | +| D4 可选升级优先策略 | 固定测试通过,九点工程矩阵完成,默认关闭 | 新旧策略在该输入上结果一致;不声称收益 | +| D5 实际 CPU gate/只读命令阶段 | 隔离编译、4 项原生 C++、13 项 service/client Python 验证通过 | 真实命令事实不是已校准能量;生产 host/GPU 未接入 | +| 基础 HBM/SRAM/级联 | 30 项相关固定检查,四拓扑实际 CPU 组合均通过 | HBM 是参数化工程模型,HBF 是真实 MQSim,fabric 是外部基础模型 | +| 真实 die 维护 | 已交精确窄设计,运行时 UNSUPPORTED_CAPABILITY | 未批准的共享仲裁/数据 commit 生命周期没有实施 | + +原始启动命令、输入身份、标准、资源及退出收据集中于外层目录 +`eq3_thermal/plans/decision-execution-v2/points/`。所有 PASS 均限定各自软件/工程范围, +没有把异类 suite 相加成“全项目测试总数”。 + +## 多 stack、拓扑与基础链路 + +显式 stack map 将持久逻辑页映射到互不重叠的 MQSim channel 组,原生命令事件 +反查实际 channel。四拓扑测试都向每个已配置 stack 发出请求;不存在只给一个 +stack 发请求的现象。D4 也每栈均等到达,HBM/HBF 完成量差异来自维护、准入和 +工程服务模型,不是到达分配失衡。 + +| 拓扑 | 实际 CPU 请求/完成 | 分配 | 基础连接 | +|---|---:|---|---| +| 8HBF direct + 外部物理 GDDR | 24/24 | 每个 HBF 3 个 | 每栈独立 direct link,有限双 SRAM bank | +| mixed-direct 4HBM+4HBF | 24/24 | 每个 stack 3 个 | 两类 stack 各自 direct 路径,无虚构 relay | +| 四对 relay | 24/24 | 每个 stack 3 个 | HBF 经配对 HBM base;与该 HBM 本地流量共享 GPU link/bank | +| 四对 DASH | 28/28 | 每 HBF 4 个、每 HBM 3 个 | 三链路、direct/relay 两路;不同 ready bank 可并行排出 | + +每个请求唯一完成,结束时无 bank/link owner 遗留。HBF 媒体完成、后端 reported +completion、外部门控等待和最终 fabric delivery 分开保存;不修改既有 MQSim +完成时间,不无条件再加 NAND/data-out 服务时间。组合当前要求 MQSim raw callback +与原 reported completion 相等,否则返回 UNSUPPORTED_COMPOSITION,避免不明 +aggregate-bandwidth 约束被重复记为 package 传输。 + +配置、热、系统行为三轴详见 FOUR_TOPOLOGY_V2_STATUS.md。基础模型没有实现 pin perimeter/capacity 的产品校准, +不宣称达到 2–4 TiB 或 12.8–24.5 TB/s。新四拓扑组合尚未与完整热模型闭环。 +外部 GDDR 的系统服务与温度为 UNAVAILABLE;不是能量或板级热影响为零。 + +## 控制代价与用户的新目标 + +当前已实现的是 per-stack 热点驱动的 Normal/Light/Severe/Shutdown 门控,含采样、 +动作延迟、滞回、驻留与真实路径联合准入;新策略只让升级不再被恢复驻留延后。 +它没有温度到 ECC/retry 的模型。工程阈值约 26/27/29 °C、固定 20 ms 读取都不能当 +HBF 产品参数。 + +在每栈 25 requests/s、20 s 输入加 10 s 恢复的固定对照中:无控制完成 4000/4000, +两种控制均只完成 2500/4000,积压 1500;最高温度从 301.5925 K 降至 300.7418 K, +已完成请求 p95 从 0.04 s 增至 10.34 s。降温伴随少完成工作,不能称为效率收益。 +完整九点温度、队列、维护年龄/积压、能量与删失口径见 D4_CONTROL_COST_REPORT.md。 + +用户最新目标采纳为:正常调节面向持续 delivered bytes/s、窗口波动、尾延迟和 +积压;温度作为风险/保护约束。热保护仍须优先执行。权威来源足以支持机制与 +代理情景,但 NAND 温度与 retry 不是普适单调关系;磨损、保留时间和温度历史 +也必须区分。HBM2 的温度刷新间隔/占用有公开数值,可作 HBM4 的显式 PROXY, +不能称 HBM4 实测参数。 + +READ_RATE_CONTROL_SOURCE_BASIS.md、HBM_TEMPERATURE_SERVICE_SOURCE_BASIS.md +列出来源、原值、允许转用和缺口。普通速率反馈最小方案尚未实施;没有把新目标 +偷偷写入已完成九点输入。缺数据只限制产品级定量结论,不阻塞基础工程链路。 + +## 可重建、小提交与撤回 + +本轮功能提交:5323629(控制策略/矩阵)、c7b35b3(执行路径/资源适配)、 +ee55722(v2 标准)、e6ce686(D3/空间诊断)、9d2805b(MQSim 接口/观测/映射)、 +afc4609(基础 HBM/fabric/CPU 组合)、1d1a082(温控说明)、1db8d57(测试调用与来源)。 +冻结执行源 decision-p2-frozen=ee55722、decision-d3-frozen=e6ce686 与当前分支分别登记。 +MQSim 补丁通过现有 patches/private build 流程重建,未以 third_party dirty 发布。 +构建及命令入口见每点 manifest;未声称跨主机已验证。 + +关闭可选策略、native observer、stack map 与外部 basic composition 即恢复原选择; +每个局部提交可独立审查/回退,不使用 reset 覆盖成果或删除 raw。 + +## 保留的失败 + +- native-d5-fixed:四项原生检查通过;Python 失败来自错误文案断言和隔离构建路径 + 的测试配置,修复测试消费者后 native-d5-fixed-v2 通过。 +- D3 第一 attempt:840 W 汇总得到 839.999999999993,驱动错误地做精确比较。 + 修复为有量纲合理容差,v2 复用身份验证后的 cap;未改变功率或热方程。 +- 原 development 越域、旧六点负结果和旧 v1 判定全部保留;新域不能覆盖它们。 + +## 温度标准的依据与新问题 + +0.25K来自项目工程决策,不是OCP/Sandisk或数值分析文献的统一要求。需要网格 +误差检查有合理依据,但现有文件没有从读取SLO、测温不确定性、控制保护裕度 +反推这个具体值。TEMPERATURE_TOLERANCE_RATIONALE.md解释两处0.25K的区别, +并提出分开工程闭环、读取稳定性充分性和热保护资格;新合同待明确采用, +旧失败不改写。网格间2.069K是同一输入的数值差,不是现实器件温度波动。 +同一物理范围的粗化细网格场比较已完成,原因与证据见下文。 + +## 问题分类与最小处理 + +| 问题 | 分类 | 实际证据/处理 | +|---|---|---| +| 原relay遗漏受控伙伴端点/Light更新 | CONFIRMED_BUG,已在上一阶段修复 | 本轮继承,不重复改写;固定回归复用/受影响检查通过 | +| D3 cap浮点精确比较 | CONFIRMED_BUG,驱动层 | 840W累计误差约7e-12W;有量纲容差修复,原cap未改 | +| v2原始采样时刻精确比较 | CONFIRMED_BUG,数据接口表示层 | 最大约1.42e-14s;双端Decimal派生规范化,2项固定测试通过;冻结评分未改 | +| 2→1mm热点最大差2.069K | NUMERICAL_ACCURACY_LIMIT,待局部场原因分解 | 旧0.25K FAIL保留;不把网格差称作真实器件变化 | +| 原development400.911K | DOMAIN_FAILURE | 不扩大400K、不裁剪,不重跑同一已知失败 | +| 固定服务时长/功率/整请求占用 | DESIGN_LIMITATION/ENGINEERING_FIXTURE | 不重写CpuService冒充NAND后端;另有真实MQSim阶段事实 | +| 原生成器百万cell防护 | 当前软件能力/资源预检边界 | 不是用户永久内存上限;uniform细化可按已授权范围规划,nonuniform生产链需方案 | +| 真实维护提交/仲裁/commit | DESIGN_LIMITATION | UNSUPPORTED_CAPABILITY;精确结构设计待确认 | +| 升级优先在九点无增益 | NOT_REPRODUCED(该输入未触发策略差别) | 测试覆盖差别,但九点不能宣称新策略更优 | + + +## P2 完整结果与当前采用范围 + +原始train 1mm参考完整100s、5000求解帧、1000观察时刻、275传感器, +输入1365J,温度范围300–369.012K,能量残差1.491e-6。 + +| 相邻网格 | 全窗传感器平均TW-MAE K | 受激段 K | 冷却段 K | 最大差 K | +|---|---:|---:|---:|---:| +| 4→2mm | 0.143305 | 0.233694 | 0.096741 | 3.253 | +| 2→1mm | 0.064942 | 0.108907 | 0.042293 | 2.069 | + +这些是各窗口和传感器上的诊断汇总,正式评分仍逐传感器/窗口检查。 +完整同窗比较不使用4s prefix冒充全部激励,也没有计算全局最大值Richardson阶。 +旧完整RC2mm/10ms对新R03的回顾性评分:v1 NUMERICAL_FAIL(22传感器失败), +v2 NUMERICAL_FAIL_WITH_THRESHOLD_AMBIGUITY(全窗19个失败,共427个传感器窗失败); +最大差2.045948K,最坏全窗MAE0.156614K,能量残差2.15e-11。旧NMAE与1K/2K/5% +结果保存在原评分文件,不因v2或本轮采用决定改成PASS。 + +局部场审计给出原因:最坏位置为hbm3.base@15s,来自邻近HBM2的64W激励。 +2.069K差=1.5065K细单元峰值相对同面积平均+0.5625K平均后仍有的场差; +同base均温差仅0.099943K。没有发现单位、功率重复、界面、层序或输出索引错误。 +新旧native二进制的完整sensor温度线性sanitycheck最大差0.002K,低于量化传播界 +0.0025K,也不能解释2K差异。该分解不是连续解真实误差预算。 + +用户随后明确采纳:只要保留原参数假设且机制物理合理,当前模型用于工程主线。 +因此状态是CONDITIONAL_ENGINEERING_USE;不为0.25K继续无限细化,原reference +资格失败仍保留,MODEL_FREEZE=false,blind仍未开启。物理参数/边界未实测标定的 +限制不妨碍软件闭环,但不能声称产品温度、寿命或安全裕度已验证。 + +0.5mm uniform细化仍在已授权方法内,但当前不启动是主线优先的选择,不是旧资源 +上限造成的永久阻塞。native nonuniform存在但EQ3生产/映射链未接入,若未来采用 +需提交具体结构方案。见P2_NUMERICAL_NEXT_STEP_SCREEN.md。 + +时间离散证据复用原同2mm完整RC10ms对native20ms最大差0.173632K,以及已有RC +步长诊断;这些不能冒称1mm原生参考已完成20→10ms资格。已知空间未达旧门槛, +本轮不新增昂贵1mm时间细化来掩盖空间差。 + +## D3 新域完整原生对照与评分 + +alpha在新输出前冻结为0.25,不按轨迹调参,也未看blind。全源840W稳态包络四候选 +最大温度分别为476.4008/432.3006/388.2004/344.1002K;最大满足380K规划裕度的 +候选为0.25。380K不是产品温限。新train输入341.25J,development2408J,保持全部 +源相对权重、空间、时序与冷却。 + +| 新域完整点 | 温度范围 K | 能量相对残差 | 求解 wall s | 观察 wall s | 采样任务峰值 KiB | +|---|---|---:|---:|---:|---:| +| train native 2mm/20ms/100s | 300–317.333 | 1.708e-5 | 402.259 | 241.760 | 1309892 | +| train RC 2mm/20ms/100s | 300–317.333353 | 3.056e-11 | 293.558 | 178.980 | 442852 | +| development native 2mm/20ms/64s | 300–325.228 | 1.763e-7 | 259.107 | 131.689 | 998384 | +| development RC 2mm/20ms/64s | 300–325.227640 | 3.070e-11 | 191.867 | 105.598 | 435516 | + +RC factor各17.682/15.760s,分解L非零19,798,141。推进与输出没有独立计时, +应为UNKNOWN,不能用总wall减factor冒充实测推进成本。求解wall/sim约2.936/2.998, +每步含setup/输出的摊销约58.7/60.0ms;这个2mm后端不是实时快速模型。运行资源 +与墙钟仅作操作收据,不作不同主机负载下的算法性能比较。 + +| 轨迹 | 原 v1 | 冻结 v2 汇总 | 全窗最坏TW-MAE K | 所有窗最大温差 K | +|---|---|---|---:|---:| +| train | PASS | NUMERICAL_FAIL_WITH_THRESHOLD_AMBIGUITY | 0.000216432 | 0.0005000 | +| development | PASS | PASS_WITH_THRESHOLD_AMBIGUITY | 0.000207961 | 0.0005000 | + +train受激/冷却窗最坏TW-MAE分别0.000462149/0.000460106K;development分别 +0.000214699/0.000211400K。全部温度绝对/幅值/热点条件与能量条件通过,但不能 +把这些分项通过替代上表的冻结汇总。 + +**train v2失败的具体原因:** 只有`component:hbm1.die7:hotspot`在 +excitation-20 [37.5,39.0] 与 cooling-21 [39.0,40.5] 两个窗失败。参考三位小数 +输出在39.000s恰好到301K,RC的上穿为39.008801s;逐窗穿越次数因跨边界不同。 +两窗MAE仅0.0002075/0.0002078K,最大差均低于0.0005K;全轨迹v1穿越通过。 +冻结实现把逐窗legacy穿越失败继续作为v2 veto,即使标THRESHOLD_AMBIGUOUS。 +这是分段判据对量化/边界敏感的限制,不是新热方程失败,不通过改输入或重复求解 +处理。本轮保留此FAIL,另报告温度/能量离散一致;未自行替换正式穿越合同。 + +两次后处理问题均保留原记录:D3 attempt-v2在未规范化时间时评分退出失败, +之后仅对双方派生CSV时间字符串规范化重新评分,四份raw原样复用。最终状态 +以D3_CANONICAL_SCORE_SUMMARY.json及D3_SCORE_CAUSE_DIAGNOSTIC.json为准; +旧D3_FAILED_v2.json继续存在,不删除或改写。 + +## 当前未做与唯一待确认的结构项 + +- 真实MQSim die维护生命周期仍UNSUPPORTED_CAPABILITY。具体文件/符号、 + read/program/erase数据有效性、映射commit/回滚、共享TSU仲裁、唯一完成、 + 测试与撤回见MQSIM_DIE_MAINTENANCE_NARROW_DESIGN.md;需要用户明确批准后实施。 +- 面向稳定读取率的外部反馈策略当前有来源和窄方案,尚未实现。现有默认关闭 + thermal gate和升级优先策略均保持真实状态;没有虚构温度到ECC/retry的曲线。 +- 新基础四拓扑CPU组合尚未与完整热模型耦合。已有单组件Shadow→gate→实际MQSim + fixture验证了接口闭环;它不能冒充多stack物理热闭环或production/GPU。 +- 当前采用是原假设下的工程用途,不是完整MODEL_FREEZE。原高功率域失败、 + 参考网格差、时间资格与真实维护缺口分别报告,blind始终未读/未生成。 + +## 运行索引与复现 + +外层 `eq3_thermal/plans/decision-execution-v2/RUN_INDEX.json` 索引所有原始收据, +`EXECUTION_RESULT.json`保存阶段最终状态。D3继续脚本run_d3.py为create-only, +已有点不应整批重跑;只读评分续跑为run_d3_score_canonical.py。可用当前工具与 +相同明确参数在新输出路径复现诊断,不通过删除原失败目录复用ID。 + +本阶段最终保留约26.75GB十进制(约24.92GiB),主文件系统剩余约651GB十进制; +无新增环境、GPU或云费用。本轮新提交继续包括4198bcf、fe95453、a24165a、36eae00 +以及本汇总提交;源码分支均留本地。 diff --git a/docs/eq3_thermal/ENDPOINT_POLICY_ADAPTER.md b/docs/eq3_thermal/ENDPOINT_POLICY_ADAPTER.md new file mode 100644 index 0000000..d4bea5c --- /dev/null +++ b/docs/eq3_thermal/ENDPOINT_POLICY_ADAPTER.md @@ -0,0 +1,31 @@ +# Shared HBM endpoint policy adapter + +Status: fixed software evidence complete; future sensitivity/diagnostic use is +prepared but not launched. + +The base system runner creates the same read-rate feedback policy for HBF +foreground sources and HBM route endpoints. A route endpoint can have zero +local foreground bytes while still carrying relay traffic. After a Light +window reduces its budget to one half, the legacy feedback branch sees +insufficient local demand and holds that reduced budget after the endpoint +returns to Normal. This is an adapter/caller mismatch, not evidence about the +thermal solver or backend service. + +`endpoint_policy.EndpointAwarePolicy` is default-disconnected and selected +only by `run_endpoint_guard_point.py`. It delegates HBF decisions to the +existing `ReadRatePolicy` without modification. For an HBM shared endpoint it +uses only the observed endpoint thermal guard: Normal restores the configured +baseline, Light applies the configured half cap, and Severe/Shutdown apply the +configured severe/zero cap. It neither fabricates HBM delivered bytes nor +changes the service scheduler, resource ownership, thermal solver, or base +runner. + +The fixed reproducer records the old `INSUFFICIENT_DEMAND` half-cap hold and +the adapter's Normal recovery. It also checks Light, Severe, Shutdown and exact +HBF legacy decision equality. The isolated entry point adds its source hashes +and capability statement to each point manifest. + +The already frozen/running base matrix remains on its original runner. Four +pilots peaked only about 310--321 K for HBM and therefore did not enter Light; +their behavior is unaffected. Completed base raw must still be checked for +any HBM Light transition before it is reused as a comparison. diff --git a/docs/eq3_thermal/FOUR_TOPOLOGY_BASE60_RESULT.md b/docs/eq3_thermal/FOUR_TOPOLOGY_BASE60_RESULT.md new file mode 100644 index 0000000..525f537 --- /dev/null +++ b/docs/eq3_thermal/FOUR_TOPOLOGY_BASE60_RESULT.md @@ -0,0 +1,82 @@ +# 四拓扑基础读取矩阵:已完成结果与范围 + +60/60点完成并通过收据、字节、分源能量、时间轴与逐栈覆盖审计。条件:每栈16通道×96GB/s,新鲜媒体供给1.536TB/s;40+10pJ/物理读取B;20s新增输入+10s继续排空;完整耦合热网络。输入1.920TB/s是需求,不是宣称硬件可交付该速率。 + +下表均为相同每HBF栈压力。交付比例是30s结束累计有效交付/20s累计到达;不能当成稳态每栈速度。四栈与八栈总输入不同;相同总需求比较尚须单独配对,不混入本表。 + +| 拓扑 | 每栈需求TB/s | 策略 | 交付比例 | HBF峰值K | 末尾积压TB | +|---|---:|---|---:|---:|---:| +| all_hbf_direct | 0.384 | guard_only | 100.000% | 334.079 | 0.000 | +| all_hbf_direct | 0.384 | read_rate_feedback_thermal_guard_v1 | 100.000% | 334.079 | 0.000 | +| all_hbf_direct | 0.384 | thermal_hysteresis_guard | 100.000% | 334.079 | 0.000 | +| all_hbf_direct | 0.768 | guard_only | 100.000% | 365.545 | 0.000 | +| all_hbf_direct | 0.768 | read_rate_feedback_thermal_guard_v1 | 100.000% | 363.225 | 0.000 | +| all_hbf_direct | 0.768 | thermal_hysteresis_guard | 100.000% | 363.247 | 0.000 | +| all_hbf_direct | 1.152 | guard_only | 95.600% | 365.540 | 8.110 | +| all_hbf_direct | 1.152 | read_rate_feedback_thermal_guard_v1 | 91.333% | 363.224 | 15.974 | +| all_hbf_direct | 1.152 | thermal_hysteresis_guard | 94.867% | 363.250 | 9.462 | +| all_hbf_direct | 1.536 | guard_only | 72.500% | 365.551 | 67.584 | +| all_hbf_direct | 1.536 | read_rate_feedback_thermal_guard_v1 | 68.860% | 363.222 | 76.530 | +| all_hbf_direct | 1.536 | thermal_hysteresis_guard | 71.700% | 363.247 | 69.550 | +| all_hbf_direct | 1.920 | guard_only | 58.000% | 365.551 | 129.024 | +| all_hbf_direct | 1.920 | read_rate_feedback_thermal_guard_v1 | 55.088% | 363.222 | 137.970 | +| all_hbf_direct | 1.920 | thermal_hysteresis_guard | 57.360% | 363.247 | 130.990 | +| dash | 0.384 | guard_only | 100.000% | 331.938 | 0.000 | +| dash | 0.384 | read_rate_feedback_thermal_guard_v1 | 100.000% | 331.938 | 0.000 | +| dash | 0.384 | thermal_hysteresis_guard | 100.000% | 331.938 | 0.000 | +| dash | 0.768 | guard_only | 100.000% | 365.224 | 0.000 | +| dash | 0.768 | read_rate_feedback_thermal_guard_v1 | 100.000% | 363.158 | 0.000 | +| dash | 0.768 | thermal_hysteresis_guard | 100.000% | 363.164 | 0.000 | +| dash | 1.152 | guard_only | 100.000% | 365.523 | 0.000 | +| dash | 1.152 | read_rate_feedback_thermal_guard_v1 | 99.243% | 363.161 | 0.697 | +| dash | 1.152 | thermal_hysteresis_guard | 100.000% | 363.170 | 0.000 | +| dash | 1.536 | guard_only | 81.900% | 365.522 | 22.241 | +| dash | 1.536 | read_rate_feedback_thermal_guard_v1 | 76.280% | 363.162 | 29.147 | +| dash | 1.536 | thermal_hysteresis_guard | 77.200% | 363.169 | 28.017 | +| dash | 1.920 | guard_only | 65.520% | 365.522 | 52.961 | +| dash | 1.920 | read_rate_feedback_thermal_guard_v1 | 61.024% | 363.162 | 59.867 | +| dash | 1.920 | thermal_hysteresis_guard | 61.760% | 363.169 | 58.737 | +| mixed_direct | 0.384 | guard_only | 100.000% | 331.848 | 0.000 | +| mixed_direct | 0.384 | read_rate_feedback_thermal_guard_v1 | 100.000% | 331.848 | 0.000 | +| mixed_direct | 0.384 | thermal_hysteresis_guard | 100.000% | 331.848 | 0.000 | +| mixed_direct | 0.768 | guard_only | 100.000% | 365.166 | 0.000 | +| mixed_direct | 0.768 | read_rate_feedback_thermal_guard_v1 | 100.000% | 363.155 | 0.000 | +| mixed_direct | 0.768 | thermal_hysteresis_guard | 100.000% | 363.158 | 0.000 | +| mixed_direct | 1.152 | guard_only | 100.000% | 365.519 | 0.000 | +| mixed_direct | 1.152 | read_rate_feedback_thermal_guard_v1 | 99.243% | 363.160 | 0.697 | +| mixed_direct | 1.152 | thermal_hysteresis_guard | 100.000% | 363.166 | 0.000 | +| mixed_direct | 1.536 | guard_only | 82.250% | 365.519 | 21.811 | +| mixed_direct | 1.536 | read_rate_feedback_thermal_guard_v1 | 76.377% | 363.162 | 29.027 | +| mixed_direct | 1.536 | thermal_hysteresis_guard | 77.325% | 363.166 | 27.863 | +| mixed_direct | 1.920 | guard_only | 65.800% | 365.519 | 52.531 | +| mixed_direct | 1.920 | read_rate_feedback_thermal_guard_v1 | 61.102% | 363.162 | 59.747 | +| mixed_direct | 1.920 | thermal_hysteresis_guard | 61.860% | 363.166 | 58.583 | +| relay | 0.384 | guard_only | 100.000% | 332.027 | 0.000 | +| relay | 0.384 | read_rate_feedback_thermal_guard_v1 | 100.000% | 332.027 | 0.000 | +| relay | 0.384 | thermal_hysteresis_guard | 100.000% | 332.027 | 0.000 | +| relay | 0.768 | guard_only | 100.000% | 365.364 | 0.000 | +| relay | 0.768 | read_rate_feedback_thermal_guard_v1 | 100.000% | 363.164 | 0.000 | +| relay | 0.768 | thermal_hysteresis_guard | 100.000% | 363.168 | 0.000 | +| relay | 1.152 | guard_only | 99.983% | 365.524 | 0.015 | +| relay | 1.152 | read_rate_feedback_thermal_guard_v1 | 98.717% | 363.168 | 1.183 | +| relay | 1.152 | thermal_hysteresis_guard | 100.000% | 363.173 | 0.000 | +| relay | 1.536 | guard_only | 81.500% | 365.526 | 22.733 | +| relay | 1.536 | read_rate_feedback_thermal_guard_v1 | 75.867% | 363.168 | 29.654 | +| relay | 1.536 | thermal_hysteresis_guard | 76.975% | 363.174 | 28.293 | +| relay | 1.920 | guard_only | 65.200% | 365.526 | 53.453 | +| relay | 1.920 | read_rate_feedback_thermal_guard_v1 | 60.694% | 363.168 | 60.374 | +| relay | 1.920 | thermal_hysteresis_guard | 61.580% | 363.174 | 59.013 | + +## 解读 + +本组基线未启用温度历史ECC代理,也没有维护请求。它验证路径/资源—能量—热—未来保护闭环,不证明完整可靠性优化。混合直连0.384TB/s保持Normal;0.768及以上已经观察到Light和Severe,不能把0.768直接标成只到Light的Near场景。 + +混合直连1.536TB/s下,guard_only交付约82.3%,滞回77.3%,反馈76.4%;更积极控制的HBF峰值低约2.36K,但完成工作更少。此结果不支持宣称反馈策略已经获益。1.920超载需求下累计交付下降且积压保留;后10s不能统一称纯冷却。 + +HBM在全部60点中保持Normal,GPU最高341.16465K低于该情景363.15K Light;因此这些点未检验HBM限制恢复或GPU先触限收益。无控制18点另表保留16个400K越域,并且全部首次Light/Severe来自HBF并列栈,不能给出compute-first普遍结论。 + +relay/DASH使用自身路径资源与伙伴检查,不回退direct。但本基线没有独立HBM前台/cache流量,不能从其近似结果推导共享链路永远无代价。8HBF只覆盖封装热域,外部GDDR服务/温度UNAVAILABLE。 + +速率工作负载没有因果token完成,所以token/s为UNAVAILABLE。另有四拓扑真实tiny-template结构派生DAG的ECC八点验证;两组证据分别报告,不能用DAG结果回填速率基线。原生MQSim语义对照不等于实际TB/s硬件吞吐,P2原失败和未冻结状态保留。 + +可复现派生来源:`BASE-ANALYSIS01/SYSTEM_THERMAL_CAMPAIGN_ANALYSIS.json`;原始启动索引和逐点收据在同阶段目录;现有16幅时间序列图与逐栈温度图不重复求解。维护争用主19点独立运行中;ECC与refresh数据身份尚未联合接通。 diff --git a/docs/eq3_thermal/FOUR_TOPOLOGY_COVERAGE.md b/docs/eq3_thermal/FOUR_TOPOLOGY_COVERAGE.md index b81c86f..fa29130 100644 --- a/docs/eq3_thermal/FOUR_TOPOLOGY_COVERAGE.md +++ b/docs/eq3_thermal/FOUR_TOPOLOGY_COVERAGE.md @@ -1,17 +1,31 @@ # 四拓扑强制交付与设计冻结修订 v3 +## 2026-09-20 minimal repair update + +See [EQ3-MINIMAL-REPAIR-v1 report](MINIMAL_REPAIR_REPORT.md) for current three-axis +status. Four CpuService CPU functional regressions pass after route-endpoint +Shutdown/Light repair. Actual MQSim request-level optional gate and thermal +cooling consumer pass CPU tests only. Native MQSim command facts and persistent +channel-group placement now pass fixed CPU tests; production host-service active +control remains unconnected and die maintenance remains unsupported. P2 stays +unfrozen and no live GPU result is claimed. External physical GDDR package +temperature is UNAVAILABLE, not zero external energy/board effect. + ## 当前增量状态(保留下面的历史v3设计,不改写冻结清单) -四拓扑实际CPU资源/能量/完成路径现已有工程fixture测试,不再仅JSON生成。 -mixed-direct主线另有六个因果工程pilot;relay伙伴base与HBM直连共享、DASH单一 -上游、8HBF外部物理GDDR服务均已测试。它们不是四种研究封装热/性能全验收。 -P2参考/模型仍未冻结;物理能量、可靠性与live GPU缺口仍保留。 +四拓扑现有一条实际 MQSim CPU 工程链,不再仅靠 JSON 或 `CpuService`:HBF 请求 +经显式 stack/channel map 进入真实 MQSim 命令路径,HBM 使用参数化场景后端, +package direct/relay/DASH 使用有界双 bank 与独立/共享链路。四个小用例均通过, +且保存逐栈计数、唯一完成和零资源泄漏证据。它们不是四种研究封装热/性能全验收。 +P2参考/模型仍未冻结;物理能量、可靠性、生产 host 和 live GPU 缺口仍保留。 三轴完整状态和原始证据入口见[P2/P3/P4实际结果](P2_P3_P4_CAMPAIGN_RESULT.md)。 实施续接修订:用户明确8HBF时GDDR位于封装外,本轮不模拟GDDR温度/板域。 下表原v3板级热模型要求已被此决定覆盖;外部物理身份与系统依赖仍保留。 通用转换器现有固定四拓扑/数量/层数/功率映射测试,首例双后端已静态生成; -热模型仍NOT_VALIDATED、系统行为仍NOT_IMPLEMENTED。具体证据见CONVERTER_VALIDATION.md。 +热模型仍 NOT_VALIDATED;系统行为仅提升为小型 CPU 工程fixture通过,不能写成 +研究拓扑、生产系统或 live GPU 通过。具体证据见 CONVERTER_VALIDATION.md 及 +`decision-execution-v2/points/basic-four-topology-actual`。 USER_CONFIRMED(2026-09-19):4HBM4+4HBF mixed-direct 是首个逐层研究封装和 转换器验证实例;接受已明确分类的厂商规格、代理和研究假设,不代表全部参数 @@ -25,19 +39,19 @@ fixture;几何/功率缺口必须显式失败,不能借用其它拓扑填空 | 必交拓扑 | 器件、几何与连接路径 | 共享资源、功耗归属 | 传感器和模型验证域 | 当前消费者、关键缺口 | 验收条件 | |---|---|---|---|---|---| -| 8HBF direct + 物理 GDDR 快存 | 8 个 HBF 含各自 base/array;GPU↔HBF直连;外部GDDR按用户要求排除本轮封装热域 | 封装内各HBF及GPU能量;外部GDDR服务/能量不伪造,不借HBM base代替 | GPU、每HBF base/die/stack;不输出GDDR/PCB温度或声称板级耦合已验证 | 通用转换器fixture保留外部physical GDDR且无热实体;8HBF研究几何/外存服务仍缺 | 封装内能量守恒与热参考验收;系统请求/共享资源与外存行为独立验收,不要求本轮GDDR板级热模拟 | -| mixed-direct,HBM+HBF 总数 8 | 两类存储各自直连 GPU;混合域 nHBM=1..7,nHBF=8−nHBM;4+4 为首例,端点按单独全同类 profile 验证,不静默套用 | 各 array/base/TSV/链路与 GPU 端资源;是否共享 GPU fabric 及上限须声明;无 relay 能量 | GPU、每 base/die/stack;当前仅 4HBM12H+4HBF16die 条件研究几何,非全部数量/层数域 | P1 默认是 2+6 fixture;新 4+4 JSON 仅静态检查,逐层转换器未实现;真实活动→功率未闭合 | 非 4+4、不同 die 数、重排 ID 的配置测试;逐物理源守恒;参考与 RC 独立通过;直连服务路径验收 | -| 4HBM + 4HBF relay | 显式四配对:GPU↔HBM base↔HBF;需研究型 base 转发能力,不能称常规 HBM4 已支持 | HBM base 转发/PHY、HBM GPU 链路、HBF TSV/array 和仲裁竞争;转发不等于访问 HBM array;接收/发送能量分端归属 | 同上,额外 base 转发热源;热几何相同也不能继承路由/功耗验证 | P1 daisy 图和合成 base 激励;无真实 relay 仲裁;P1 HBF8die,与候选 HBF16die 不同;延迟/带宽/能量缺口 | 显式配对和合法 base 能力;争用/队列/因果完成、无伪 DRAM 访问、base 能量不漏不重;本拓扑热域验收 | -| 四对 DASH | USER_CONFIRMED DSAH→DASH;每对 HBF 同时具有 direct/relay 路径,配对及选择规则显式化 | 两路径共享同一 HBF array/TSV 供给;relay 再占 HBM base/链路;选择、分流、切换、回压及端点 PHY 能量单列;不可双算容量 | 同上;需双路径热点、base/array 非均匀功率域验证 | P1 dual 声明图;当前 HBM16H fixture 不是主线12H;真实选择与共享仲裁未实现 | 两路径独立及并发服务测试、共享瓶颈守恒、不重复完成/计能、四配对正确;本域参考/RC 验收 | +| 8HBF direct + 物理 GDDR 快存 | 8 个 HBF 含各自 base/array;GPU↔HBF直连;外部GDDR按用户要求排除本轮封装热域 | 封装内各HBF及GPU能量;外部GDDR服务/能量不伪造,不借HBM base代替 | GPU、每HBF base/die/stack;不输出GDDR/PCB温度或声称板级耦合已验证 | BasicSystem 实际fixture:8个单channel/单die HBF各3请求,MQSim完成24/24,双bank回压和零泄漏;外部physical GDDR身份保留,但service/temperature均UNAVAILABLE | 封装内能量守恒与热参考仍待验收;GDDR系统服务仍缺,不要求本轮GDDR板级热模拟 | +| mixed-direct,HBM+HBF 总数 8 | 两类存储各自直连 GPU;混合域 nHBM=1..7,nHBF=8−nHBM;4+4 为首例 | 各 array/base/TSV/链路与 GPU 端资源;无 relay 能量 | GPU、每 base/die/stack;当前仅 4HBM12H+4HBF16die 条件研究几何,非全部数量/层数域 | BasicSystem 4+4 actual fixture:每栈3请求,共24个唯一完成;HBF走MQSim,HBM为PARAMETRIC场景,配置中pair/relay为空,最终bank/link全释放 | 非4+4与研究几何/容量/吞吐、真实HBM、逐物理源能量和热参考仍待验收 | +| 4HBM + 4HBF relay | 显式四配对:GPU↔HBM base↔HBF;基础实现不代表常规HBM4产品已有该能力 | HBF source bank→pair-private relay→共享的配对HBM bank/GPU link;不访问HBM array | 同上,额外 base 转发热源仍未接热模型 | BasicSystem actual fixture:4对、每栈3请求、24个唯一完成;HBM local与relay共享有界bank/GPU link,顺序两段fabric完成,零泄漏 | 链路/银行时序仅SCENARIO_ASSUMPTION;真实PHY、能量、热源映射和研究几何仍待验收 | +| 四对 DASH | 每对HBF同时有direct/relay,package route与MQSim native direct placement分账 | 两条drain路径使用独立link;relay占配对HBM bank/GPU link;两bank产生真实外部回压 | 双路径热点、base/array非均匀功率域仍未验证 | BasicSystem actual fixture:每HBF四个交替direct/relay、每HBM三个local,共28个唯一完成;四对映射、共享资源和最终零泄漏通过 | 基础仲裁通过不等于校准带宽/能量、研究封装热验证、生产策略或live GPU通过 | ## 三条独立验收轴 | 拓扑 | 配置覆盖 | 热模型验证 | 系统拓扑行为验证 | |---|---|---|---| -| 8HBF + GDDR | 通用双导出fixture PASS;研究profile INCOMPLETE,GDDR热域排除 | NOT_VALIDATED | NOT_IMPLEMENTED | -| mixed-direct | 非4+4 fixture PASS;4+4研究首例双导出/原生parse-only PASS | NOT_VALIDATED | NOT_IMPLEMENTED | -| relay 4+4 | 通用配对/双导出fixture PASS;研究profile INCOMPLETE | NOT_VALIDATED | NOT_IMPLEMENTED | -| DASH 四对 | 通用配对/双导出fixture PASS;研究profile INCOMPLETE | NOT_VALIDATED | NOT_IMPLEMENTED | +| 8HBF + GDDR | 工程8HBF派生profile/map PASS;研究profile INCOMPLETE,GDDR热域排除 | NOT_VALIDATED | `BASIC_CPU_HBF_PATH_PASS`;GDDR service/temperature UNAVAILABLE,生产host/live GPU未接 | +| mixed-direct | 工程4+4派生profile/map与实际CPU链 PASS;研究首例配置覆盖保留 | NOT_VALIDATED | `BASIC_CPU_FIXTURE_PASS`;真实HBM、生产host/live GPU未接 | +| relay 4+4 | 工程四配对配置与实际CPU链 PASS;研究profile INCOMPLETE | NOT_VALIDATED | `BASIC_CPU_FIXTURE_PASS`;参数化HBM/链路,不是产品能力验证 | +| DASH 四对 | 工程四配对双路径配置与实际CPU链 PASS;研究profile INCOMPLETE | NOT_VALIDATED | `BASIC_CPU_FIXTURE_PASS`;未校准、未热耦合、非生产策略 | 任何一轴通过不提升其它轴;主线标定通过也不提升其它拓扑。旧 RC 仍 FAILED; 旧 40mm 案例只作测试。完成标准是四行各自完成适用的三轴证据,不是四个 JSON 存在。 diff --git a/docs/eq3_thermal/FOUR_TOPOLOGY_V2_STATUS.md b/docs/eq3_thermal/FOUR_TOPOLOGY_V2_STATUS.md new file mode 100644 index 0000000..f126a2e --- /dev/null +++ b/docs/eq3_thermal/FOUR_TOPOLOGY_V2_STATUS.md @@ -0,0 +1,46 @@ +# EQ3 four-topology v2 status + +Status date: 2026-09-20. This table separates configuration, thermal, and system +behavior evidence. A pass on one axis does not promote either other axis. + +## Evidence receipts + +- `native-d5-fixed`: retained `FAILED`. Four C++ tests passed; the service and + client failures were test-only (an incorrect error-string assertion and a + missing explicit isolated artifact root, followed by an unbound test local). + The backend had correctly rejected the overlapping channel map. +- `native-d5-fixed-v2`: `PASS`, 10 service plus 3 client tests, 0 failures. +- `basic-components-fixed`: `PASS`, 30 tests total: 7 parameterized-HBM, + 11 fabric, 6 BasicSystem, and 6 spatial-diagnostic checks. +- `basic-four-topology-actual`: `PASS`, one fresh actual MQSim service process + per topology, with retained source/derived profiles and maps, transcripts, + native command observations, results, and summary. + +All receipts are under +`eq3_thermal/plans/decision-execution-v2/points//` in the outer +workspace. They are fixed CPU engineering evidence, not a research experiment. + +## Three independent axes + +| Topology | Configuration coverage | Thermal-model validation | System-topology behavior | +|---|---|---|---| +| 8HBF direct + external physical GDDR | `ENGINEERING_FIXTURE_PASS`: derived 8-channel profile and eight explicit one-channel/one-die HBF groups; research capacity/geometry incomplete | `NOT_VALIDATED`; BasicSystem fabric has no thermal consumer; GDDR temperature `UNAVAILABLE` by scope | `BASIC_CPU_HBF_PATH_PASS`: actual MQSim 24/24 requests, 3 per HBF, bounded-bank waits and zero final owners. External GDDR identity retained, but GDDR service `UNAVAILABLE`; production host/live GPU not connected | +| 4HBF + 4HBM mixed-direct | `ENGINEERING_FIXTURE_PASS`: derived 4-channel/four-group HBF map; HBF `pair=null`, no relay link; explicit four parameterized HBM stacks | `NOT_VALIDATED`; no HBM/fabric activity-to-power or package thermal coupling | `BASIC_CPU_FIXTURE_PASS`: 24/24 total, 3 per stack; HBF uses actual MQSim, HBM is `PARAMETRIC_HBM_SCENARIO`; all banks/links released. Real DRAM, production host/live GPU not connected | +| 4+4 relay | `ENGINEERING_FIXTURE_PASS`: four explicit one-to-one pairs and topology-matched four-group HBF map; research profile incomplete | `NOT_VALIDATED`; relay/base energy and hotspot mapping are UNKNOWN/unconnected | `BASIC_CPU_FIXTURE_PASS`: 24/24 total, HBF relay is sequential pair-private relay then shared paired-HBM GPU link; HBM-local and relay share bounded HBM banks/link; zero final owners. Timings are scenario assumptions | +| Four-pair DASH | `ENGINEERING_FIXTURE_PASS`: four explicit pairs, direct/relay selection, topology-matched four-group HBF map; research profile incomplete | `NOT_VALIDATED`; dual-path source/base/PHY power and hotspots unvalidated | `BASIC_CPU_FIXTURE_PASS`: 28/28 total; 4 alternating direct/relay requests per HBF plus 3 local requests per HBM; four-pair mapping, backpressure, unique completion and zero final owners pass. No calibrated policy, production host, or live GPU | + +## Shared limits + +- HBF media commands and physical channel/die/plane facts come from actual + MQSim. The package stack identity comes from the explicit channel-group map, + not an address guess. +- HBM is a parameterized FIFO timing scenario and is not a real DRAM backend. + Refresh remains `UNSUPPORTED_CAPABILITY`; die/plane remain `UNKNOWN`. +- Base banks and package links are `SCENARIO_ASSUMPTION`. Unknown operation/link + energy remains `UNKNOWN`, never zero-filled. +- BasicSystem records external arrival/wait, backend completion, fabric + completion, and final completion separately. It rejects composition when raw + MQSim completion differs from its generic bandwidth-bounded reported time. +- The optional chain is default off. The production host service, calibrated + 12.8–24.5 TB/s/capacity claims, thermal solver coupling, and live GPU path are + not validated by these receipts. diff --git a/docs/eq3_thermal/GEOMETRY_AND_BYTE_SCOPE.md b/docs/eq3_thermal/GEOMETRY_AND_BYTE_SCOPE.md new file mode 100644 index 0000000..47f8d8a --- /dev/null +++ b/docs/eq3_thermal/GEOMETRY_AND_BYTE_SCOPE.md @@ -0,0 +1,97 @@ +# Causal geometry and byte scope + +Status: read-only audit, 2026-09-20. No solver, native backend, causal service, or +runner was changed by this audit. + +## Geometry evidence + +| Item | Value used by the OCP-oriented profile | Evidence class and exact limit | +|---|---:|---| +| NAND page | 4,096 B | `SPECIFIED`: OCP HBF v0.7.0 §4.1 specifies 4 KiB pages and that reads do not cross a page. | +| Host channels | 16 per stack | `SPECIFIED`: OCP §4.3. | +| Dies | 16 per cube | `SPECIFIED`: OCP Table 3. | +| Banks | 16 per channel | `SPECIFIED` as banks. OCP does not establish a universal bank-to-NAND-plane identity. | +| Backend mapping | 16 channels/stack × 1 MQSim die/channel × 16 planes/die | `SCENARIO_ASSUMPTION`: `campaign_inputs.configuration(..., geometry="ocp4k16bank")` projects the 16 OCP banks onto MQSim planes (`BANK_AS_MQSIM_PLANE_V1`). It is not a product plane specification. | +| Pages per block | 256 | `SCENARIO_ASSUMPTION`: OCP §5.7 leaves R3/pages per NAND block product-defined. The value has a real MQSim allocator consumer, but is not OCP-specified. | +| Logical capacity | 512 GiB/stack | `SPECIFIED` capacity represented by the full-capacity profile. It does not specify physical spare, bad-block reserve, or overprovisioning. | +| Maintenance aged subset | 4 GiB/stack = 4,096 × 1 MiB blocks | `SCENARIO_ASSUMPTION`: 1/128 = 0.78125% of a 512 GiB logical stack. This is an explicit aged subset, not total capacity or a claim of physical block coverage. | +| Maintenance spare pool | 16 blocks/channel, 256 blocks/stack in the aggregate driver | `SCENARIO_ASSUMPTION`: bounded metadata ownership for the aggregate service. It is not observed product spare capacity. | + +The executed full-capacity native geometry evidence and its limits are recorded +in `ISOLATED_GEOMETRY_AND_THERMAL_EVIDENCE.md`. In particular, CWDP and the +page-level FTL choose physical locations; a logical ordinal is not a physical +block address. + +## Current causal byte semantics + +`causal_workload._tensor_groups` derives logical tensor payload regions from the +registered Qwen architecture metadata. It aligns the next region's logical +address to 1 MiB, but those address gaps are not offered as traffic. +`CausalExecutor._stripe_parts` then distributes exact logical bytes in 4 KiB +round-robin units and assigns the single residual tail to one target. The sum of +all child jobs equals the tensor's logical payload exactly for both 64- and +128-target layouts. + +The current service job has one `bytes` value for media, base, buffer and fabric +phases. Consequently a partial final page currently consumes only its logical +tail bytes at every phase. There is no 4 KiB NAND-media rounding fact in the +causal receipt. Current output must therefore be labelled +`LOGICAL_PAYLOAD_BYTES_MODELLED_AS_MEDIA_BYTES`; it is not an observation that +physical NAND transferred a partial page. + +For a full scan of the registered original Qwen2.5 payloads, rounding each +logical tensor group to a 4 KiB NAND-media page gives: + +| Model | Logical payload | Tensor groups | Page-rounded `physical_payload_bytes` proxy | Padding | Relative padding | 64/128 target effect | +|---|---:|---:|---:|---:|---:|---| +| Qwen2.5-7B-Instruct | 15,231,233,024 B | 59 | 15,231,262,720 B | 29,696 B | 1.949678 ppm | Same padding for 64 and 128 targets; the current algorithm has one residual child per partial tensor group. | +| Qwen2.5-72B-Instruct | 145,412,407,296 B | 163 | 145,412,407,296 B | 0 B | 0 | Same for 64 and 128 targets. | + +This calculation uses the full logical tensor groups in +`qwen2_5_weight_models.json`. Selected-row embeddings and other sub-tensor +accesses require their own request-level rounding calculation. +`physical_payload_bytes` means only the 4 KiB-aligned payload proxy. Actual +NAND page transfer, OOB/ECC bytes, protocol framing, internal movement, and +wire bytes remain `UNKNOWN`; the calculation is not an observed physical NAND +transaction count and does not establish full-capacity physical allocation. + +At 50 pJ/B (the user-confirmed 80 W at 1.6 TB/s read envelope), the 7B tail +padding adds 1.4848 µJ to a 0.7615616512 J full scan. At 500 pJ/B it adds +14.848 µJ to 7.615616512 J. The relative correction remains 1.949678 ppm, so +page-tail rounding is not thermally material for these full scans. A 500 pJ/B +coefficient applied to the entire payload is a 10× energy assumption and is +scientifically material; the page-rounding correction at that coefficient is +not. The 72B full scan has no page-tail correction under this grouping. + +## Minimum correct adapter recommendation + +Do not simply round the existing job's `bytes` to 4 KiB. That would also charge +padding to base/fabric delivery, inflate useful completion bytes, cache volume, +and dependency accounting. It would change more than NAND-media scope. + +The smallest correct default-off extension is: + +1. Keep the existing logical job and its identity, link bytes, useful-byte + completion, cache accounting, and dependency completion unchanged. +2. For each NAND source child with a partial page, generate one parent-linked + `media_page_padding_read` fact/job of `4096 - logical_tail_bytes` on the same + explicit stack/channel. It consumes only the NAND media resource and emits + only `media_read` activity; it consumes no base, buffer, relay, or GPU-link + bytes and produces no useful completion. +3. Aggregate logical completion only after both the original logical job and + its padding media work have completed. Record `logical_bytes`, + `physical_media_bytes`, `padding_bytes`, `page_bytes=4096`, and parent ID in + receipts. Retry padding must be charged once per actual retry attempt, not + once per logical consumer. +4. Reject the option when page size, physical channel, or operation scope is + unknown. HBM jobs remain byte-granular and are not NAND-page rounded. +5. Enable it only in a new explicit profile. Off mode must preserve current + receipts and timing byte-for-byte. + +This requires a narrow operation/resource addition to `CausalTopologyService` +and a parent-completion adapter in `run_causal_point`; the present service +cannot represent media-only padding through an external wrapper because its +single byte count is applied to every phase. It should be implemented only +under a separate reviewed change, with fixed checks for: partial and aligned +pages, 64/128 target conservation, unchanged useful/link bytes, increased media +bytes only, retry deduplication, completion dependency, and off-mode parity. diff --git a/docs/eq3_thermal/GPU_GUARD_SOURCE_AUDIT.md b/docs/eq3_thermal/GPU_GUARD_SOURCE_AUDIT.md new file mode 100644 index 0000000..7160be3 --- /dev/null +++ b/docs/eq3_thermal/GPU_GUARD_SOURCE_AUDIT.md @@ -0,0 +1,15 @@ +# GPU hotspot guard audit and compute–memory feedback boundary + +USER_CONFIRMED 2026-09-20: check comparable current accelerator documentation; keep reasonable limits; consider memory throttling reducing GPU heat and retention effects without delaying the mainline with a full compute/ECC model. + +DOC_DERIVED: AMD-SMI virtualization-host documentation50.2.2, section Command Examples / Static Information, identifies MI300X and lists hotspot slowdown100°C and shutdown110°C. Its MI350X example lists the same hotspot limits. These are vendor-documented example device/firmware readouts, not a promise for every GPU or a future HBF-equipped chip. Source: https://instinct.docs.amd.com/projects/amd-smi-virt/en/8.6.0.k/how_to/amdsmi_cli_usage.html . Viewed2026-09-20. Relevant fields SLOWDOWN_HOTSPOT_TEMPERATURE, SHUTDOWN_HOTSPOT_TEMPERATURE. v9.0.0.k search evidence agrees but direct fetch returned429; v8.6.0.k opened successfully. + +NVIDIA documentation separately defines Max Operating (software clock optimization), Slowdown (hardware thermal clock optimization), Shutdown and relative T.Limit. It does not support transferring one absolute value to every H100/B200. DGX environmental operating temperatures are inlet/system conditions, NOT GPU junction/hotspot limits. Sources: https://docs.nvidia.com/deploy/nvidia-smi/ and https://docs.nvidia.com/dgx/dgxb200-user-guide/introduction-to-dgxb200.html . + +Disposition: keep current GPU hotspot thresholds363.15/373.15/383.15K (90/100/110°C) for this conditional engineering campaign.100/110 are PROXY from comparable MI300X/MI350X hotspot limits;90 is a SCENARIO_ASSUMPTION,10K early-warning margin, not a vendor temperature limit. The modeled GPU geometric hotspot is a spatial maximum, not a calibrated hardware sensor. No claim that any named accelerator will adopt HBF or uses this exact thermal geometry/power/cooling. + +Pilot03 maximum396.0464K=122.8964°C exceeds even the110°C proxy. Numerical success within300–400K therefore is NOT thermal protection success or physically sustainable GPU operation. It demonstrates an insufficient actuator in this experiment: memory gate closes, but independent prescribed GPU200W continues until its declared active end. Report degree/time above limits and backlog cost, not successful safety. + +Feedback distinctions: (1) physical thermal coupling GPU↔HBM/HBF is already in the complete shared network; reducing HBF dissipation can affect GPU temperature through that network; (2) workload coupling read starvation→less GPU compute→less dynamic heat is NOT implemented. The latter is conditional on workload critical path, buffering, other runnable work and idle/static power; it is not a universal proportional rule. No throughput-to-GPU-power coefficient is invented in the mainline. A later bounded proxy can use P_idle+(P_active−P_idle)*u, with a causal delayed utilization u and independently sourced bounds; this would be a new explicit input model, not reinterpretation of current raw. + +Elevated NAND temperature can accelerate retention charge-loss mechanisms. The observed failure probability also depends on elapsed retention time, wear, program/read temperature difference, ECC and retry behavior. Instantaneous error rate is not asserted to rise monotonically under all conditions. Existing OCP85°C/24h condition does not identify an RBER/ECC latency curve. Current age is real simulated seconds, temperature-independent; ECC/RBER/UECC predictions remain UNKNOWN. See existing READ_RATE_CONTROL_SOURCE_BASIS.md. This limitation does not prevent completing the engineering delivery/maintenance/thermal chain. diff --git a/docs/eq3_thermal/HBF_ECC_CONDITIONAL_RESULT.md b/docs/eq3_thermal/HBF_ECC_CONDITIONAL_RESULT.md new file mode 100644 index 0000000..af8c9b6 --- /dev/null +++ b/docs/eq3_thermal/HBF_ECC_CONDITIONAL_RESULT.md @@ -0,0 +1,46 @@ +# HBF NAND 条件纠错成本:实现与实际验证 + +证据身份:CONDITIONAL_SIMULATED / PASS_PAIRED_INTEGRATION。用户已明确器件为 HBF;HBM 不采用 NAND 可靠性模型。原 MQSim、默认后端和生产接口未修改。 + +## 已实现的链路 + +上一观察窗的 HBF array 温度 → 等效保持年龄 → 首次服务时冻结的预期重试工作量 → 同一媒体及逐栈解码资源竞争 → 真实服务活动能量 → 唯一有效交付 → 后续温度与控制。重试消耗内部物理读取,外部有效字节只交付一次;已完成事件不会被重写。新代理默认关闭。 + +| 接口 | 实际生产者 | 实际消费者 | 启用方式及边界 | +|---|---|---|---| +| 温度历史与等效年龄 | 完整热网络的已观察 array 温度 | `ReadCostProxy.observe` | `hbf_read_cost_proxy.mode=conditional_nand_history_v1`;各栈最热 die 保守驱动该栈统一年龄 | +| 读取成本决策 | `ReadCostProxy` 的年龄/P-E/transfer 条件曲线 | `ReliabilityCausalService` 首次资源分配 | 工作量固定点量化;不是实测整数重试或错误率 | +| 物理活动 | 同一服务资源中的媒体与解码阶段 | 现有能量适配器 | array40 + base10 pJ/物理读取B;base 已包含代理纠错能量,不重复加计 | +| 有效完成 | 唯一 fabric delivery | 因果 DAG 的依赖解除及最终计算完成 | 记录模拟 token;计算时间未作 GPU 标定 | +| 刷新提交与年龄 | 维护 extent 的成功版本提交 | 维护年龄模型 | 已独立接通;尚未映射到本 ECC 逐栈年龄,组合显式拒绝,不能刷新一页清零全栈 | + +## 参数依据 + +OCP v0.7.0 的重试/有效性语义提供行为依据,不提供目标 ECC 时延或错误率曲线。旧 48-layer TLC 的 365 天@30°C、2000 P/E、平均19.9次重试为条件锚点,年龄及磨损中间曲线为情景插值;transfer 0、0.1、1 是零/弱/完整转用情景,不是置信区间。参见 [Park 等,ASPLOS 2021](https://arxiv.org/html/2104.09611)。 + +Ea=1.04eV 复用 [HeatWatch,HPCA 2018](https://research.ece.cmu.edu/safari/pubs/heatwatch-3D-nand-errors-and-self-recovery_hpca18.pdf) 的旧 MLC 代理;其20–70°C拟合域外时间单列。不同器件的加速因子并不等价。解码吞吐为新鲜媒体的2倍、固定附加时延为0,均为重叠稳态工程假设。[Sandisk 公开资料](https://documents.sandisk.com/content/dam/asset-library/en_us/assets/public/sandisk/collateral/company/Sandisk-HBF-Fact-Sheet.pdf) 未给出可用于标定的数值 ECC 曲线。 + +没有设置“读取瞬间越热必然越慢”的规律。已有 NAND 研究中,固定保持年龄和磨损后,较冷的读取温度反而可能增加错误;热历史加速保持老化与瞬时读取温度必须分开。 + +## 八点成对实际结果 + +四拓扑分别零/弱代理;共同72B形状再生工作负载,1s新增输入+1s排空,名义每栈1.536TB/s,初始90等效天@30°C、1000P/E,仅严重保护。实际总耗时790.394s;八点完成,原始收据、逐成本决策回放、字节/能量/唯一完成检查通过。 + +| 拓扑 | 零代理有效TB | 弱代理有效TB | 零代理HBF峰值K | 弱代理HBF峰值K | 弱代理重试能量J | +|---|---:|---:|---:|---:|---:| +| mixed_direct | 6.146 | 5.507 | 357.841 | 364.132 | 247.855 | +| relay | 6.146 | 5.507 | 357.841 | 364.136 | 247.855 | +| dash | 6.146 | 5.507 | 357.841 | 364.135 | 247.855 | +| all_hbf_direct | 12.291 | 11.002 | 357.914 | 364.197 | 495.178 | + +TB 为观察窗内全封装累计有效字节,不是每栈TB/s。弱代理 mixed 的完成 DAG token 为37,零代理43;它们不是实测LLM吞吐。此负载没有 HBM 前台/缓存竞争,四种连接的近似结果不能推导为拓扑普遍等价。 + +这证明重试成本已进入服务和热链路,不证明某种控制策略获益。弱代理初始平均0.9次额外尝试,使1.536TB/s新鲜媒体在其他瓶颈前的有效供给上限约0.808TB/s。温控减少未来工作,同时可能减少未来老化;后者不能通过瞬时清年龄伪造。 + +## 可辨识性与剩余边界 + +90天初始年龄下,20s恒定85°C和80°C在当前条件曲线中的预期重试差约0.000091,低于0.001工作量量化;对应1小时约0.0147,6小时约0.0604。这是解析恒温推演,不是额外热运行,也没有将24小时压缩成秒。短窗有可能只能看见节流成本;不能因此人为增大温度惩罚保证正收益。 + +当前代理为批量预期工作量,不预测逐页 RBER、UECC 或纠错本身的随机 p99 波动;初始 P/E 为情景参数。维护逐 extent 年龄与前台数据身份尚未统一,完整刷新—ECC策略收益暂不具备证据。原P2网格差、400.911K越域、v1/v2失败、400K上限和盲测封存状态保持。 + +原始入口:工作区 `eq3_thermal/plans/four-topology-system-v1/ecc-pilot-v1/PILOT_INDEX_V2.json`;派生 `ANALYSIS01/SUMMARY.json` 及逐点六联图/逐栈温度图。旧未执行 V1、原低负载维护结果和所有失败收据均保留。 diff --git a/docs/eq3_thermal/HBF_ECC_REFRESH_IDENTITY_PROPOSAL.md b/docs/eq3_thermal/HBF_ECC_REFRESH_IDENTITY_PROPOSAL.md new file mode 100644 index 0000000..e7c432e --- /dev/null +++ b/docs/eq3_thermal/HBF_ECC_REFRESH_IDENTITY_PROPOSAL.md @@ -0,0 +1,34 @@ +# ECC 与刷新数据身份:最窄剩余结构方案 + +状态:PROPOSED_NOT_IMPLEMENTED。独立 ECC8 与维护4热集成已完成,但不能宣称刷新降低该 ECC 代理成本。该文件不构成用户批准。 + +## 问题与最小复现 + +`run_causal_point.execute` 同时启用 `hbf_read_cost_proxy.mode=conditional_nand_history_v1` 与 shared maintenance 会显式拒绝。前台 `CausalExecutor` 的读取只有 tensor/partition/stripe 身份;`MaintenanceDriver` 有 extent/source_block/version 身份;`ReadCostProxy` 只有逐栈等效年龄。没有可信映射时,成功刷新一页不能重置全栈年龄,也不能确保在途读取使用的源未被擦除。 + +已有 `CausalMaintenanceAgeAdapter.consume_receipt` 能按真实完成时刻推进受影响 extent 年龄并消费版本提交,因此不需要重写其年龄积分或热求解。现有前台迁移的源保护也应优先复用,不能再开第二套版本账本。 + +## 最小接口与必需结构变化 + +1. 隔离 workload 层生成可逆的 tensor 字节范围→extent/版本/stack/channel 映射。整合相同成本的范围,但每次成本加权须保留覆盖关系和有效字节守恒。未知身份拒绝组合,不从任意地址猜 die。 +2. 新只读成本入口按同一维护账本查询 extent 年龄、当前物理块的实际 P/E,并将参考温度统一换算。使用已观察温度外推到首次服务时间,不提前修改账本年龄前沿。 +3. 读取首次取得资源时获取当前源版本的读引用;成功或失败唯一终止时释放。刷新 program 成功后 CAS 提交新版本,只重置提交 extent 的年龄;旧块 erase 必须等待源读引用排空。并发较新前台写导致 CAS 失败时不得覆盖新映射,已发生的能量仍保留。 +4. 刷新和前台仍使用同一个 `CausalTopologyService` 的现有资源;不修改其公平分配算法、不扩展原默认 MQSim、不新增后台线程。物理 NAND 提交仍仅在现有隔离 native 能力边界内对照验证,不声称流体路径实际执行 TB/s NAND。 + +第1和3项影响请求身份和资源生命周期,不能用“只增加接口”作为豁免。有效 AGENTS.md 的最新阶段条款已授权范围内新增隔离实验模块设计;实施前须进一步判断是否能完全复用现有版本/源保护接口。若仅为获批隔离消费者的接入,可在阶段内继续;若需要改变已有调度或所有权核心,才将具体超出部分提交用户确认。不能仅因旧条款提及重构就重复申请已授予的阶段权限。独立维护、ECC和热敏感性工作不受阻。 + +## 拟改位置及行为 + +| 文件/符号 | 拟改行为 | +|---|---| +| `causal_workload.py:CausalExecutor` 的分条提交和完成消费 | 附加真实 extent 覆盖、版本引用和唯一释放;复用已有迁移映射/源保护 | +| 新隔离 extent 成本桥 | 只读同一维护年龄/磨损账本,计算分组预期工作;无自己的资源账本 | +| `ecc_service_adapter.py:ReliabilityCausalService` | 成本输入从逐栈转为已解析范围;关闭时保持现状 | +| `maintenance_driver.py:MaintenanceDriver` 的提交/擦除阶段 | 查询共享源读引用,旧块排空后进入现有 erase 入口 | +| `run_causal_point.py:execute` | 只有完整映射能力通过时才允许组合,保留不支持守卫 | + +## 验证、回滚与资源 + +先固定小数据集:刷新一个 extent 不影响邻居;program失败/CAS冲突不重置年龄;读跨提交使用有效旧版本;旧块读未排空不擦除;重复完成不能重复释放;维护与前台竞争同一资源;无新增输入仍排空/恢复。再做极小 native 元信息语义对照和四拓扑相同输入的新ID paired pilot,记录身份、RSS/wall/磁盘,CPU串行。正式大矩阵复用此前预算方法,但要以该最小 pilot 的实测成本冻结。 + +所有新行为默认关闭,代码和工件独立、小提交可回滚;旧 raw、基线和已接受结果不覆盖。未完成该方案前,报告可给出维护争用、年龄/磨损和独立 ECC 成本,不能给出刷新—ECC联合收益,也不能标完成整个可靠性闭环。 diff --git a/docs/eq3_thermal/HBM_TEMPERATURE_SERVICE_SOURCE_BASIS.md b/docs/eq3_thermal/HBM_TEMPERATURE_SERVICE_SOURCE_BASIS.md new file mode 100644 index 0000000..fbe0417 --- /dev/null +++ b/docs/eq3_thermal/HBM_TEMPERATURE_SERVICE_SOURCE_BASIS.md @@ -0,0 +1,182 @@ +# HBM 温度、刷新、服务占用与功耗:来源基础和最小候选模型 + +日期:2026-09-20 +状态:`READ_ONLY_SOURCE_BASIS`;未启用模型、未运行实验、未读取 blind 数据。 + +## 目的与结论边界 + +本文件为“以稳定读取速率为温控目标”的最小 HBM 服务模型整理证据。用户已明确 +允许在缺失参数时采用有权威资料支撑的合理假设;这一授权不把代理资料变成目标 +Micron HBM4 的标定数据,也不批准把 ECC 错误概率从温度凭空生成。 + +当前可以建立一个默认关闭的 `PARAMETRIC_HBM_SERVICE_PROXY`:用实际 stack 温度 +选择刷新倍率,用刷新占用降低可提供读取带宽,再判断目标读取速率能否稳定维持。 +直接可用的定量时序来自 AMD 集成 HBM2 平台,不是 Micron HBM4 12H。Micron 当前 +公开 HBM4 产品页没有给出 `tREFI`、`tRFC`、刷新能量或温度分档。因此 HBM4 的 +绝对刷新功耗、产品级服务占用和 >95 °C 行为仍为 `UNKNOWN_BLOCKING`。 + +## 证据分层 + +| 来源 | 器件/范围 | 可用事实 | 本项目分类与限制 | +|---|---|---|---| +| AMD AXI HBM Controller PG276,[Raw Throughput Evaluation](https://docs.amd.com/r/en-US/pg276-axi-hbm/Raw-Throughput-Evaluation) | 集成 HBM2,4H/8H | 基础 `tREFI=3.9 us`;4H `tRFC=260 ns`,8H `tRFC=350 ns`;0–85 °C 使用基础间隔,85–95 °C 使用 `1.95 us`;资料给出的峰值效率损失约 7%/9% | `HBM_DIRECT`,但代际、堆叠高度和控制器均不等于目标 HBM4;只能作时序/占用代理 | +| AMD PG276,[Refresh options](https://docs.amd.com/r/en-US/pg276-axi-hbm/Reorder-Refresh-and-Power-Savings-Options-Tab) | 同上 | 支持 single-bank refresh、lookahead、基于温度调整刷新周期、读写 holdoff 和温控 self-refresh | `HBM_DIRECT`;证明可见占用依赖刷新粒度和调度,不能把 `tRFC/tREFI` 当作所有控制器的精确吞吐损失 | +| AMD PG276,[Features](https://docs.amd.com/r/en-US/pg276-axi-hbm/Features) | 同上 | 温控刷新、可选隐藏 single-row refresh;SECDED ECC、scrub、parity/retry 是分别可选的功能 | `HBM_DIRECT`;只证明能力存在,不给 ECC 错误率或温度曲线 | +| AMD PG276,[HBM configuration](https://docs.amd.com/r/en-US/pg276-axi-hbm/HBM-Configuration-Selection-Tab) 与 [register map](https://docs.amd.com/r/en-US/pg276-axi-hbm/Memory-Controller-Register-Map) | 同上 | stack 温度轮询周期可配置;可读刷新命令、self-refresh cycles、带宽和 ECC 计数 | `HBM_DIRECT`;可作为以后实机校准观测量,目前没有相应观测 | +| AMD DS923,[Recommended Operating Conditions](https://docs.amd.com/r/en-US/ds923-virtex-ultrascale-plus/Recommended-Operating-Conditions);AMD XAPP1377,[thermal targets](https://docs.amd.com/r/en-US/xapp1377-heatsinks-thermal/Obtaining-Thermal-and-Power-Targets-to-use-with-Thermal-Simulation) | 特定 AMD 集成 HBM 产品 | 建议连续 HBM 温度上限 95 °C;特定 -2LE 产品允许有限 95–105 °C excursion;高温刷新会影响带宽 | `HBM_PLATFORM_DIRECT`;是平台级运行边界,不是 Micron HBM4 产品温限。DS923 对 >95 °C 的倍率措辞不能无条件移植 | +| Micron,[HBM4 product page](https://www.micron.com/products/memory/hbm/hbm4) | 目标代际,公开产品资料 | 36 GB 12H、2048-bit、超过 11 Gb/s、每 stack 超过 2.8 TB/s;相近速度下相对 HBM3E 能效改善 | `HBM4_DIRECT`;未公开刷新时序、温度分档、绝对功耗或 rail 分账。公开峰值带宽不能直接当持续媒体读取带宽 | +| Micron,[DDR5 New Features](https://www.micron.com/content/dam/micron/global/public/products/white-paper/ddr5-new-features-white-paper.pdf) | DDR5 代理 | all-bank refresh 期间目标 banks 不能读写;示例 16 Gb DDR5 的 `tREFI=3.9 us`、`tRFC=295 ns`;same-bank refresh 可让其他 banks 工作 | `DDR_PROXY`;只支持刷新占用语义和量级检查,不能提供 HBM4 参数 | +| Micron,[ECC Brings Reliability and Power Efficiency to Mobile Devices](https://www.micron.com/content/dam/micron/global/public/products/white-paper/ecc-for-mobile-devices-white-paper.pdf) | LPDDR4 代理 | 85–95 °C 使用 2 倍刷新,95–105 °C 使用 4 倍刷新;资料称 8 Gb LPDDR4 在最高温档 all-bank refresh 占用超过 18% | `LPDDR_PROXY`;不能作为 HBM4 的精确倍率、ECC 能力或错误率 | +| 本地 `MICRONHBM4`、HBM3E/H200 证据账本 | 公开产品与平台代理 | H200/HBM3E 聚合 memory-domain 活跃读取能量约 45.815–46.941 pJ/delivered byte | `PLATFORM_ENERGY_PROXY`;不是 refresh energy,不是 per-stack,不可拆成 HBM4 array/base/PHY | +| OCP HBF 0.7.0,本地原件 SHA-256 `307531eb8053f00cbeccbc907ddff0a9c4fe6f9d0066a077ce33b0ac99312da3` | HBF/NAND | 典型 24–48 h 周期维护;同 die 读/refresh 互斥 | `HBF_ONLY`;它是 NAND 数据维护,不是 DRAM 周期刷新,不参与本 HBM 模型 | + +网页资料访问日期为 2026-09-20。本地注册来源和限制见 +`configs/eq3_thermal/sources.json`、 +`configs/eq3_thermal/research/source_evidence.json`、 +`docs/eq3_thermal/PARAMETER_GAPS.md`。 + +## 可由直接 HBM2 证据复算的服务占用 + +对于 all-bank、刷新期间不提供读取服务、且不隐藏/重叠刷新的简化情景: + +```text +m(T) = 1, T <= 85 degC + 2, 85 degC < T <= 95 degC +tREFI(T) = 3.9 us / m(T) +u_refresh = min(1, tRFC / tREFI(T)) +B_refresh = B_no_refresh * (1 - u_refresh) +``` + +由 PG276 数值直接推导: + +| HBM2 代理 | 温档 | 倍率 | `u_refresh` | 理想剩余服务比例 | +|---|---:|---:|---:|---:| +| 4H, `tRFC=260 ns` | 0–85 °C | 1x | 6.667% | 93.333% | +| 4H, `tRFC=260 ns` | 85–95 °C | 2x | 13.333% | 86.667% | +| 8H, `tRFC=350 ns` | 0–85 °C | 1x | 8.974% | 91.026% | +| 8H, `tRFC=350 ns` | 85–95 °C | 2x | 17.949% | 82.051% | + +这些值与 PG276 报告的常温约 7%/9% 峰值效率损失相符。它们是无重叠 +all-bank 上界式占用;single-bank、hidden refresh、lookahead、访问局部性和排队 +可能降低或改变 host 可见损失。不得把 4H/8H 的 `tRFC` 按 die 数线性外推为 12H。 + +## 建议的最小候选模型 + +### 1. 主域:0–95 °C + +候选名称:`HBM2_REFRESH_SERVICE_PROXY_FOR_HBM4`,默认关闭。 + +- 温度输入必须来自被建模 stack 的传感/热节点,不从地址、请求种类或 ECC 事件猜测。 +- 用上式的 `m(T)` 和两个显式代理档分别评估:`tRFC=260 ns` + (`HBM2_4H_PROXY`) 与 `350 ns` (`HBM2_8H_PROXY`)。它们是两个来源明确的情景, + 不是 HBM4 12H 的置信区间;8H 档仍可能低估 12H 服务占用。 +- 读取速率可行性条件为: + +```text +R_target <= eta_access * B_raw * (1 - u_refresh) +``` + + `eta_access` 单独表示地址映射、bank conflicts、协议和请求尺寸造成的效率;必须由 + 真实后端或明确工程情景给出。不可把 refresh 损失同时包含在 `eta_access` 和 + `u_refresh` 中。`B_raw` 使用目标配置值时仍保留其来源分类;Micron 的 + `>2.8 TB/s` 是产品页下限描述,不等于任意负载的持续读取率。 +- 控制器应以稳定维持 `R_target` 为目标:若温度跨入 2x 档后右侧容量不足,则必须 + 报告 `READ_RATE_INFEASIBLE_AT_REFRESH_GRADE`,而不是通过遗漏刷新、把排队当完成或 + 虚构 ECC 收益维持目标。 +- 阈值附近确定性需要使用同一时间戳的既定温度采样顺序。传感器误差、轮询间隔和 + 温升时延尚无目标平台数值,因此 guard band/hysteresis 大小保持 `UNKNOWN`;它们 + 不能被随意固定后称为产品阈值。 + +### 2. 95–105 °C 只作越域/敏感性诊断 + +AMD 平台将 95 °C 设为连续运行上限,并只对特定产品允许有限 excursion。Micron +LPDDR4 代理在 95–105 °C 使用 4x nominal;DS923 对特定 HBM 平台的文字要求是 +高于 95 °C 时刷新率至少为“95 °C 时刷新率”的 4 倍。与 PG276 的 95 °C 档组合时, +后者可能被读作至少 8x nominal。器件、版本和措辞存在歧义。 + +因此主候选模型在 `T>95 °C` 返回 `OUT_OF_PRIMARY_DOMAIN`,不输出确定的 HBM4 +服务率。若仅为安全敏感性诊断,可显式计算 4x 与 8x nominal 两个端点: + +| HBM2 代理 | 4x nominal 占用 | 8x nominal 占用 | +|---|---:|---:| +| 4H, 260 ns | 26.667% | 53.333% | +| 8H, 350 ns | 35.897% | 71.795% | + +该表只能标为 `EXCURSION_DIAGNOSTIC_ONLY`;不得用于声明 Micron HBM4 95–105 °C +产品行为或安全运行能力。 + +### 3. 调度粒度 + +最小实现可先采用每 stack 单服务域的保守 all-bank 占用,输出: + +```text +stack, temperature_C, refresh_grade, source_profile, +tREFI_ns, tRFC_ns, refresh_occupancy, available_read_Bps, +target_read_Bps, target_feasible, evidence_class +``` + +若以后接入真实 controller counters,应该以 pseudo-channel/bank 可见刷新命令、 +self-refresh cycles 和实际读取带宽校准 host-visible loss。校准后才能替换保守占用; +不能仅因为硬件支持 single-bank refresh 就假设刷新完全隐藏。 + +## 功耗与热反馈 + +当前资料支持“刷新频率升高会增加功耗并减少服务时间”的方向,但没有目标 HBM4 +每次刷新能量。物理形式可以先冻结为: + +```text +P_total = P_idle(T) + P_read(R,T) + P_refresh(T) +P_refresh(T) = N_refresh_domains * E_REF(T) / tREFI(T) +``` + +其中 `E_REF(T)`、实际 refresh domain 数、rail 归属和与读操作的能量重叠均为 +`UNKNOWN_BLOCKING`。因此: + +- 服务占用模型可以在 energy 为 `UNKNOWN` 时独立运行并报告读取率可行性; +- thermal 输入不能把 refresh energy 静默设成 0; +- H200/HBM3E 的 45.815–46.941 pJ/delivered-byte 只可作为聚合活跃读取代理,不能 + 代替 `E_REF`,也不能同时记入 array、base 和 PHY; +- 若为了工程闭环必须提供功耗敏感性,只能另设显式无量纲 + `refresh_busy_power_ratio` 情景,并绑定已测 rail 的 idle/active 差值;没有目标测量 + 前不推荐给出伪精确数值区间,也不得将该情景用于正式温度或能效结论; +- 最小后续实测需要同一平台的 idle、固定读取率和已知刷新档,记录 stack 温度、 + refresh command/self-refresh counters、delivered bytes 与同一 rail 能量。差分要保留 + controller/PHY/DRAM rail 覆盖范围。 + +[Micron DRAM power calculator](https://www.micron.com/sales-support/design-tools/dram-power-calculator) +是官方功耗估算入口,但当前公开页面没有提供本目标 HBM4 的可引用刷新 IDD/能量 +参数,不能据此补数字。 + +## ECC、scrub 与可靠性限制 + +PG276 证明特定 HBM2 controller 可选 SECDED ECC、background scrub 和 parity retry; +Micron LPDDR4 白皮书讨论了 ECC 与高温刷新之间的可能关系。这些事实均不提供目标 +HBM4 的下列数据: + +- 温度到 raw/corrected/uncorrectable bit-error probability 的曲线; +- scrub 的实际周期、服务占用和能量; +- retry 概率、额外延迟和故障分布; +- 使用 ECC 后可安全降低目标产品刷新率的授权。 + +故当前模型不得生成 `P(error|T)`、ECC correction 数、retry 或寿命收益。若硬件真实 +计数可用,只能把观测到的计数作为事实记录;在没有统计暴露量和校准模型时,也不能 +反向解释为温度因果概率。当前 `configs/eq3_thermal/reliability.json` 不消费 ECC/RBER +曲线,这一状态应保持明确。 + +## 能力状态和最小补证顺序 + +| 项目 | 当前状态 | 可支持结论 | +|---|---|---| +| 温度触发 1x/2x 刷新语义 | `AVAILABLE_HBM2_DIRECT` | HBM2 代理情景;不能称 HBM4 产品参数 | +| all-bank 服务占用 | `DERIVED_HBM2_PROXY` | 4H/8H 两个显式情景下的读取容量损失 | +| single-bank/hidden refresh | `CAPABILITY_KNOWN_BEHAVIOR_UNCALIBRATED` | 说明 all-bank 模型可能保守;不能设为零损失 | +| HBM4 12H `tRFC/tREFI` | `UNKNOWN_BLOCKING` | 不支持产品级延迟、占用或温档声明 | +| HBM4 refresh energy/power | `UNKNOWN_BLOCKING` | 不支持绝对热反馈和 refresh 能效结论 | +| stack 温度与 refresh counters | `OBSERVABLE_ON_AMD_HBM2_PLATFORM` | 给出未来校准方案;当前没有目标观测 | +| ECC/error/retry 温度模型 | `UNAVAILABLE` | 不生成概率、错误、retry 或寿命收益 | +| >95 °C | `OUT_OF_PRIMARY_DOMAIN` | 只可做 4x/8x excursion 敏感性,不可作 HBM4 正式结果 | + +补证优先级为:目标 HBM4 part datasheet/JESD270-4 对应时序与温档;目标 controller +刷新粒度和调度;同平台 counters 与 rail 能量;传感器误差和轮询延迟。获得前,最小 +代理应保持默认关闭、参数显式、来源标签随输出保存,并把所有正式 HBM4 温度—服务 +结论标为 `CONDITIONAL_SIMULATED`。 diff --git a/docs/eq3_thermal/INTERFACE_CONTRACT_AND_CONSUMERS.md b/docs/eq3_thermal/INTERFACE_CONTRACT_AND_CONSUMERS.md new file mode 100644 index 0000000..41b90c0 --- /dev/null +++ b/docs/eq3_thermal/INTERFACE_CONTRACT_AND_CONSUMERS.md @@ -0,0 +1,187 @@ +# EQ3 MQSim interface contract and consumers + +Status: CPU interface fixture for `EQ3-MINIMAL-REPAIR-v1`. This document does +not promote P2 to `MODEL_FREEZE`, does not claim live GPU coverage, and does not +turn the engineering occupancy fixture into command-level NAND evidence. + +`EQ3-DECISION-EXECUTION-v2` adds an actual JSON-lines CPU-service consumer of +the existing gate, a default-off native MQSim command observer, persistent HBF +channel partitions, and an opt-in basic four-topology CPU composition. The +fixed native and component tests and the four actual-service fixtures passed. +This remains engineering evidence; the production host service is not wired to +these consumers. + +## Minimal gate design and non-interference argument + +`MqsimSubmissionGateAdapter` is a header-only composition boundary around the +existing `MqsimOnlineEngine::submit()` call. It is `Off` by default. Off mode +passes the original request to the engine without changing its arrival or +completion. Enabled mode asks a caller-provided decision function only when the +request's target arrival has been reached. The callback returns `Allow`, +`Defer`, `Blocked`, or `Unsupported`, with an explicit reason for every result +other than `Allow` and a target time for `Defer`. + +The adapter does not run the simulator, retry, retain completions, call the +observer, reserve backend resources, or own requests after submission. The +decision callback must not re-enter the engine. A caller handles `Defer` by +using the existing `run_next_completion_until(target)` API, consuming every +completion it returns, draining `MqsimObserverAdapter`, and calling +`try_submit()` again at the target. This preserves the existing MQSim event +ordering and lets in-flight work, observation energy, thermal time, cooling, +and an external controller recover through their existing consumers. + +MQSim rejects a newly submitted request whose arrival precedes its current +clock. After an enabled gate delay, the backend copy therefore uses the actual +admission time as its arrival. The result records the original arrival, +backend arrival, and external wait separately. The returned MQSim completion +is never rewritten, and no service delay is appended to it. End-to-end users +must report `external_wait_ns` and MQSim backend latency as distinct terms. +Successful request IDs are remembered only to reject an accidental second +submission through the same wrapper; this does not change backend ownership. + +## Interface and actual-consumer matrix + +| Interface | Actual producer | Actual consumer | Enablement | Observable granularity | Capability | Remaining gap / regression | +|---|---|---|---|---|---|---| +| `MqsimObservation` | Existing `MqsimOnlineEngine` arrival, device handoff callback, and MQSim completion callback | `tests/eq3_thermal/mqsim_observer_tests.cpp::run`, `gated`, and `thermal_gate_closed_loop` through `MqsimObserverAdapter` then `ActivityObserver` | Explicit `enable_observations()`; automatically requested only for observer modes other than Off | Request arrival, admission, media callback end, separately delayed reported completion, logical bytes, device outstanding count | `CPU_TEST_CONNECTED`; production host service `PRODUCTION_NOT_CONNECTED` | No command ID, stack/die/plane, physical bytes, link bytes, or real operation energy. Fixed off/read-only/shadow test preserves these as unknown. | +| Request occupancy to energy | Test-supplied `ObservedEvent` with `ENGINEERING_FIXTURE_REQUEST_OCCUPANCY` | The same three CPU test functions consume the `ActivityObserver` energy ledger; Shadow test functions also consume its thermal model/advice | ReadOnly or Shadow plus explicit binding | Time between actual MQSim admission and media callback, with explicit component power | `CPU_TEST_CONNECTED`, engineering fixture only; `PRODUCTION_NOT_CONNECTED` | It is not NAND start or measured power. Queue wait has no activity energy. Fixed test checks no fabricated command location/bytes. | +| `MqsimSubmissionGateAdapter::try_submit` | Test decision callbacks plus current target clock | `tests/eq3_thermal/mqsim_observer_tests.cpp::{gated,thermal_gate_closed_loop,gate_contract}`, `benchmarks/replay/hbf_mqsim_service.cpp`, and `tools/eq3_basic_system.py::BasicSystem`; accepted requests go to unchanged `MqsimOnlineEngine::submit()` | `MqsimGateMode::Off` by default; service fixture enables a fixed target-time callback only with `--gate-not-before-ns`; BasicSystem is an explicit opt-in caller | Whole demand request before backend submission; decision, reason, retry target, external wait | `CPU_SERVICE_CONNECTED`, fixed tests PASS; no production thermal policy | The service holds no deferred request and performs no retry. BasicSystem retains only its external request and bounded source-bank reservation, advances with `until`, and retries the same backend ID. The CLI policy is an engineering fixture, not active thermal control. | +| `run_next_completion_until` during gate wait | Existing MQSim event queue and reported-completion readiness marker | `tests/eq3_thermal/mqsim_observer_tests.cpp::{gated,thermal_gate_closed_loop,gate_contract}` and `MqsimObserverAdapter::drain_to_current_time()` | Test caller explicitly advances to returned gate target | Target simulation time; one original completion per call | Existing engine API; gate composition is `CPU_TEST_CONNECTED`, production gate consumer `PRODUCTION_NOT_CONNECTED` | Caller must continue until the target because an earlier in-flight completion can be returned first. No host sleep is involved. | +| Demand backend capability | `MqsimSubmissionGateAdapter::capability(Demand)` | Configuration/preflight checks | Query only | Capability identity | `SUPPORTED` for optional pre-submit gate | This says nothing about stack/path-specific physical control unless the caller has real route metadata. | +| Shadow temperature advice to gate | `ActivityObserver` Shadow model after target-time cooling; explicit fixture thresholds | `tests/eq3_thermal/mqsim_observer_tests.cpp::thermal_gate_closed_loop`, then `MqsimSubmissionGateAdapter` | Shadow plus explicit `ENGINEERING_FIXTURE` policy | One declared thermal component and whole demand request | `CPU_TEST_CONNECTED`; production controller `PRODUCTION_NOT_CONNECTED` | Thresholds, power, and one-node cooling model are engineering inputs, not a product control algorithm or calibrated physical claim. | +| Native MQSim command phases | Patched `NVM_PHY_ONFI_NVDDR2` existing command issue, command/data-in completion, chip-ready, and read-data transfer boundaries | `benchmarks/replay/hbf_mqsim_service.cpp`, `tools/eq3_basic_system.py`, and raw assertions in `tests/integration/test_mqsim_service.py` | Off by default; `--native-command-observations on` or an enabled stack map installs an immutable synchronous sink | Unique command and transaction IDs, optional external request parent, MQSim source/type, logical page, physical channel/chip/die/plane/block/page, backend bytes, native time | `NATIVE_PHASE_OBSERVER_CPU`, fixed tests PASS | An unmapped observation has no package stack/route identity. A mapped demand transaction is checked against explicit channel groups. Energy and maintenance parent remain unavailable; background commands preserve a null external parent. Callback only appends facts and never advances/re-enters MQSim. | +| Persistent HBF stack placement | `MqsimStackMapAdapter` address bijection plus native command channel observations | `benchmarks/replay/hbf_mqsim_service.cpp` and `BasicSystem`; raw assertions in `tests/integration/test_mqsim_service.py::test_explicit_eight_hbf_stack_map_reaches_native_channels` and actual four-topology receipts | Off by default; explicit `--stack-map` with HBF/direct/CWDP profile-matched channel groups; enabled requests supply `stack`, `stack_local_page`, and `route=direct` | External/backend page, requested/resolved HBF stack, expected and actual native channel/die/plane | `ACTUAL_MQSIM_CHANNEL_PARTITIONED_HBF_STACKS`, fixed and actual-service fixtures PASS | Same-kind HBF direct, one aligned page only. The 8- or 4-stack/1-die derived profiles are small engineering fixtures, not research geometry. Package direct/relay/DASH is composed outside MQSim. | +| Parameterized HBM media | `tools/eq3_basic_hbm.py::BasicHbm`, one FIFO media server per configured HBM stack | `tools/eq3_basic_system.py::BasicSystem`; component tests and mixed/relay/DASH actual-service fixtures | Only when BasicSystem receives an explicit `PARAMETRIC_HBM_SCENARIO` config | Arrival, submit, media start/end, bytes, stack, read/write, unknown die/plane, optional energy | `CPU_FIXTURE_CONNECTED`; not a real DRAM backend | Timing is a scenario assumption. Refresh is `UNSUPPORTED_CAPABILITY`; unknown energy stays null/UNKNOWN. Production host service and thermal power mapping are not connected. | +| Base banks and package fabric | `tools/eq3_basic_fabric.py::BasicFabric` bounded two-bank ownership and direct/relay/HBM-GPU link events | `BasicSystem`; component tests and all four actual-service fixtures | Default off; explicit `SCENARIO_ASSUMPTION` config | Reserve, backend-ready, link start/end, bytes, bank/link owner, final package delivery | `CPU_FIXTURE_CONNECTED`; fixed and actual-service fixtures PASS | Parameters are uncalibrated. `mark_source_ready` adds no duplicate fill after backend data-out. Energy is UNKNOWN where unparameterized; no thermal consumer or production daemon connection. | +| Four-topology coordination | Actual `MqsimService` for HBF, `BasicHbm` for HBM, and `BasicFabric` for package resources | `tools/eq3_basic_system.py::BasicSystem`, invoked by `tools/eq3_basic_system_actual.py` | Explicit runner only; one fresh MQSim process and topology-matched derived profile/map per case | External arrival/wait, backend media/reported completion, fabric completion, final completion, per-stack counts and final resource snapshot | `CPU_ACTUAL_SERVICE_FIXTURE_PASS` for four small cases | Not research capacity/geometry/throughput, live GPU, or production host integration. Thermal coupling is absent. External GDDR service and temperature remain UNAVAILABLE. | +| Native phase energy | No producer | No consumer | Not available | None | `DECLARED_ONLY` is not claimed; capability is absent | Native timestamps and bytes do not supply operation energy. Occupancy-proxy and native-phase evidence must be selected as mutually exclusive energy inputs. | +| Die-level maintenance submission/completion | No producer in `MqsimOnlineEngine` | Capability/preflight checks only | Query only | None | `UNSUPPORTED_CAPABILITY` | No enqueue/start/end/commit/fail facts, shared arbitration, resource or energy report. A normal host write must not be relabeled as HBF refresh. | +| External GDDR energy | Explicit caller metadata, if available | `ActivityObserver::external_energy_j()` | Explicit binding only | External power integrated over observed activity | Accounting supported | Package temperature is `UNAVAILABLE`; lack of a package node does not mean zero service energy or zero board heat. | + +## Fixed CPU checks + +`tests/eq3_thermal/mqsim_observer_tests.cpp` contains the source-level regression +for this contract. It compares default-off service with direct MQSim, compares +ReadOnly and Shadow event/completion streams, advances a deferred request with +the existing target-time API while delivering an earlier in-flight completion, +then admits the request once and checks that external wait is not folded into +the unchanged MQSim completion. It also verifies no callback is evaluated +before a future request arrives, blocked work never reaches the backend, and +die-level maintenance remains `UNSUPPORTED_CAPABILITY`. A separate one-node +engineering fixture starts above its explicit Light threshold; the caller +advances MQSim's target clock, drains the observer so boundary cooling occurs, +reconsumes real Shadow advice, submits once after recovery, and receives the +original MQSim completion. This closes the CPU chain without adding a runner or +production scheduling policy. + +Final fixed CPU validation is `THERMAL-FINAL2-TEST`: all four thermal suites +passed (13 thermal-core checks plus observer, CPU-service, and actual-MQSim +suites), with the MQSim suite reporting both `CPU_PATH_VERIFIED` and +`MQSIM_GATE` PASS. The immutable evidence is under +`eq3_thermal/plans/minimal-repair-v1/points/THERMAL-FINAL2-TEST/` in the outer +workspace (`result.json`, `stdout.log`, `stderr.log`, manifest and source +snapshot). This was a 0.61 s CTest run; it is a fixed software validation, not +a research experiment or live-GPU result. + +The retained earlier `THERMAL-FINAL-TEST` correctly failed the first test +caller version with `observer did not consume wait/completion time`. That +caller used legacy `run_next_completion()` after gate recovery: MQSim may +return a completion carrying a later bandwidth-bounded reported time while the +engine clock is still at the media callback, so the observer's reported +completion remains pending. The repair changed only the test consumer to keep +using `run_next_completion_until()` and drain after each horizon advance. It +did not alter MQSim, the gate, the completion timestamp, or latency semantics. + +The first D5 aggregate run, `native-d5-fixed`, is also retained as FAILED. Its +four existing C++ MQSim tests passed; the failures were test-only: one assertion +looked for the word `stack` although the backend correctly rejected an +overlapping channel group, and the client test did not pass the isolated build +root into its existing path boundary (then referenced an unbound local after +construction failed). No engine or stack-map behavior was changed. The focused +`native-d5-fixed-v2` receipt passed both service suites: 13 tests total (10 +service and 3 client), 0 failures. Raw result, stdout/stderr, manifest, and +source patch are retained under the correspondingly named decision-execution-v2 +point directories. + +`basic-components-fixed` passed 30 fixed Python checks, including HBM, fabric, +BasicSystem, and spatial diagnostics. `basic-four-topology-actual` then passed +four actual MQSim-service cases: all-HBF direct 24 requests, mixed-direct 24, +relay 24, and DASH 28. Each case retains its source and derived profile/map, +service transcript, native observations, per-stack counts, unique completions, +and a final snapshot with no bank/link owner or unfinished request. These +receipts validate the small CPU composition only. + +This task intentionally does not add a production active-controller policy. +BasicSystem supplies validated engineering stack/path identities to the small +fixtures, but production policy, host integration, calibrated thermal advice, +and live-GPU evidence are still required before claiming topology-aware active +control. + +## D5 CPU-service and physical-address boundary + +The service's `try_submit` command is the actual existing CPU process and uses +the same `run_next_completion_until` path as `until`. It returns the original +arrival, backend admission time, external wait, explicit disposition/reason, +and target time. It does not retain a deferred request. Default-off +`try_submit` is compared with the existing `submit` command for the same +request observation stream and completion. Enabled testing advances to the +returned target, retries once, and checks that the backend completion remains +unchanged rather than receiving a second delay. + +The native command patch is registered as +`patches/mqsim/0003-hbf-command-observer.patch`; the vendored +`third_party/mqsim` tree remains source input rather than the release delta. +When disabled, no sink is installed and no observation IDs are allocated. +When enabled, the engine copies the external request ID into an internal +observation-only field. MQSim assigns transaction IDs lazily when a real +command is observed and assigns one command ID to each native command. A +multi-page request therefore keeps one external parent while exposing multiple +transactions and commands. Multiplane commands may conversely contain multiple +transactions under one command ID. `GC_WL`, mapping, and cache commands retain +their actual MQSim source and a null external request parent unless a real +parent exists; the observer does not infer one. + +`MqsimOnlineEngine` configures one MQSim device with +`Flash_Channel_Count=profile.channels`, one chip per channel, +`Die_No_Per_Chip=profile.dies_per_channel`, and +`Plane_No_Per_Die=profile.planes_per_die`. `Input_Stream_Manager_HBF` segments +logical byte extents into page LPAs. Page-level mapping then assigns LPA to +channel/chip/die/plane according to the selected `plane_allocation_scheme` +(default `CWDP`), while `Flash_Block_Manager` owns the page/block allocation. +The base D5 CPU test submits a two-LPA read and requires real command observations +on at least two physical channels. This is evidence for one MQSim device using +multiple channels. With no `--stack-map`, no configured or consumed mapping +connects a channel to a package stack, so `stack=UNKNOWN` and addresses pass +through unchanged. The optional map partitions those existing channels into +explicit, disjoint same-kind HBF groups and applies a persistent global-page +bijection before the unchanged submit/gate. Native command observations then +hard-check the configured expected channel. This is channel-partitioned HBF +stack evidence, not separate MQSim devices, HBM placement, package topology, +relay routing, or balanced mixed-stack service. + +## Backend completion and optional fabric composition + +MQSim's raw request callback occurs only after its NAND command path, including +native ONFI command and read-data transfer phases. The adapter records that +instant as the `MqsimObservation::Completion.time_ns` media callback. It then +computes the existing serialized aggregate-bandwidth lower bound from +`profile.aggregate_bandwidth_bytes_per_s` and reports +`modeled_completion_ns = max(raw_callback_ns, bandwidth_cursor_ns)`. The +aggregate bound cannot be disabled with zero because profile validation +requires a positive value, and its physical link identity is not encoded in +the profile. + +An optional external base/link fabric must therefore start from the raw media +callback only when its data dependency requires media-ready bytes, and combine +its result as `final = max(existing_modeled_completion, fabric_done)`. It must +not add fabric duration to the already bounded reported completion. The +initial composite consumer must require raw callback time to equal the existing +reported completion for every composed request; otherwise it returns +`UNSUPPORTED_COMPOSITION`. This avoids assigning the unidentified generic +aggregate bound to the new package link. Actual NAND and native ONFI channel +timing remain enabled. No current profile switch disables those native +transfers, and this D5 adapter does not add one. + +The real maintenance design and its approval boundary are recorded in +`docs/eq3_thermal/MQSIM_DIE_MAINTENANCE_NARROW_DESIGN.md`. Runtime maintenance +capability remains `UNSUPPORTED_CAPABILITY`. diff --git a/docs/eq3_thermal/ISOLATED_ABLATION_READINESS.md b/docs/eq3_thermal/ISOLATED_ABLATION_READINESS.md new file mode 100644 index 0000000..8028498 --- /dev/null +++ b/docs/eq3_thermal/ISOLATED_ABLATION_READINESS.md @@ -0,0 +1,179 @@ +# 隔离维护实验:消融与九轴敏感性就绪审查 + +状态:`READINESS_AUDIT_ONLY`,2026-09-20。本审查未启动热求解器,未改变默认 +模型、热方程、边、控制策略或科学阈值。依据为阶段 `USER_TASKBOOK.md` §8.3/8.4、 +当前隔离实验源码,以及已完成的 +`Q1-WEIGHT-MAINT-PILOT03` 原始账本。下文的 `READY` 只表示接口和输入能够形成 +可审计实验点,不表示点已获数值结果、实物参数已校准或 P2 资格已改变。 + +## 2026-09-20 当前状态附录(覆盖下文历史“尚未执行”状态) + +下文保留最初就绪审查和当时的成本/能力判断;本附录是当前执行状态: + +- A1 已完成:九个源域加零源、500 窗、125,500 行;实体均温叠加最大误差 + `8.6061e-11 K`,零源最大偏离 `1.1084e-11 K`。它仍是完整 C/G 上的源消融, + 不是删边或断开封装。 +- A2 已通过固定 ideal replay:前台 4,032/4,032 完成,64/64 回放维护完成;10 个 + 请求提前 620 ns,但完成数、P95、峰温和总能量不变。回放未向当前 MQSim 提交 + 维护、未改 mapping,不能称已实现独立维护调度器。 +- A3 已完成并分类 `FULL_2MM_DISCRETE_EQUIVALENCE`:500×275 传感器最大差 + `4.7180e-12 K`,累计能量最大差 `4.4048e-11 J`。同方程/Eigen 家族,不是独立 + 物理参考或 P2 资格。 +- GPU 外热 OAT 40/100/200 W 已在新 4 KiB v3 活动中冻结并排队,尚无完成收据; + 不将队列状态当数值证据。 +- 当前 OCP 路径的后端单请求是严格一个 **4 KiB** page。下文“16 KiB page”只描述 + 历史 PILOT03 基线;当前仍没有多页物理 coalescer。 + +当前综合证据见 `docs/eq3_thermal/ISOLATED_GEOMETRY_AND_THERMAL_EVIDENCE.md`。 +P2 仍未冻结,blind 仍封存。 + +## 已核对的实际基线 + +`Q1-WEIGHT-MAINT-PILOT03` 是 8 s active 加 2 s recovery observation、20 ms 热窗口的 +`mixed_direct/W1/guard_only` 条件工程 pilot。4032 个请求全部完成,64 次维护提交, +输入能量为 1600.0025647932162 J,峰值 396.0463965549566 K。`energy.csv` 有 +500 个连续 `WINDOW_TOTAL`,实际出现 121 个带能量组件;GPU 为 1600 J,其余 +存储和互连活动合计约 0.002564793216 J。能量系数均标为 +`SCENARIO_ASSUMPTION`。这组比例本身不能外推为产品能耗或一般 workload。 + +manifest 的 `active_ns=8000000000`、`end_ns=10000000000`。500 个原始窗口的 +`phase` 全部为 `OBSERVATION` 是正确合同:`OBSERVATION/DRAIN` 表示是否已越过 +总观察终点,`ACTIVE/RECOVERY` 是另一条正交分段。转换器保留原始 phase;后处理 +按 manifest 的 8 s 边界另行派生 active/recovery segment,不能把 recovery 改名为 +drain。该点没有 `end_ns` 后的额外已提交工作排空,summary 的 drain completion 为 0。 + +工作热服务锁定完整 2 mm 网络:64,512 节点、255 个物理/封装实体、275 个传感器, +初态和上下环境均为 300 K,顶部/底部换热系数分别为 1400/25 W/(m² K),侧壁绝热。 +它在一个进程中只分解一次固定的 `C/dt+G`,逐 20 ms 窗接受组件焦耳。服务和旧 +campaign runner 使用同一离散方程、同一 Eigen 分解族;二者成对一致只能证明 +**离散实现/适配一致性**,不能作为独立物理参考资格。 + +证据分类:任务书约束和已批准阶段为 `USER_CONFIRMED`;上述文件内容与数值为 +`DOC_DERIVED`;尚未运行的成本估计和建议分组为 `INFERRED`。 + +## A1:保留同一网络的源隔离响应 + +### 可做的最小消融 + +当前完整 RC 系统在线性、温度无关材料假设下可按热源叠加。最小方案固定原始 +`C/G`、初态、边界和时间步,对同一个 pilot 能量账本生成以下回放: + +1. 一个零活动基线,验证 300 K 等温状态和零静态源; +2. 九个源域:`gpu`,以及 `hbf0..3`、`hbm0..3`。每个内存域保留该 stack 的 + base、全部 die 及互连端点能量,其他域本窗口能量置零; +3. 对每个 GPU/stack 观察量,采用其所属源域运行的温度;同时保存全耦合运行, + 交叉源贡献定义为 `T_full - T_own_domain`。所有运行使用同一完整网络,因此 + 自热响应、同一边界冷却、共享封装的被动热容和散热路径都保留;被消去的是 + 其他源域对该观察域的温升响应。 + +该构造应命名为 `SOURCE_ISOLATED_DOMAIN_RESPONSE`。它**不是**删边模型,也不是 +“组件之间不存在物理热传导”的全局温度场:其他器件仍作为被动导热/储热路径, +而将九个运行按域拼出的观察集合不满足一个单一全局能量守恒方程。现有 +`coupling off` 只跳过标为 `InterComponent` 的直接边,仍可经共享 substrate、 +interposer、lid 和边界耦合,并改变自热阻抗;不能用它冒充 A1。 + +这个九域版本可直接使用现有服务协议,无核心 ABI 变化。每个独立进程只需一份 +分解和同样 500 次推进。按既有完整模型一次分解约 10.5 s、一次推进约 0.03 s 的 +量级估算,零源加九域约 4–5 CPU 分钟,若保留与 pilot 同粒度 JSON,输出约 +0.7–0.8 GiB;这些是 `INFERRED` 预算,启动前仍应由统一 runner 做实际 preflight。 + +### 严格逐组件版本的边界 + +pilot 有 121 个实际带能量组件。逐组件各跑一次能得到每个传感器对每个源的响应, +但现有服务只输出实体/传感器归约,不输出节点场。stack hotspot 是非线性 `max`, +不能仅用各次 hotspot 标量拼出“每个节点只保留自身组件源响应”的严格逐组件场。 +旧 runner 全场输出可以离线构造该场,但约 122 次完整回放会产生明显更高 I/O 和 +因子开销。故严格逐组件 A1 为 `CAPABILITY_GAP/NOT_READY_FOR_MINIMAL_RUN`;九域版 +是当前可运行的有界消融,结论仅限跨 GPU/stack 源响应。 + +## A3:实际 pilot 能量的完整 2 mm 成对回放 + +新增的实验专用转换器 +`experiments/eq3_maintenance/pilot_energy_replay.py` 只读取 `energy.csv` 的 +`WINDOW_TOTAL` 行,拒绝间隙、重叠、乱序、未知组件、负值和非有限能量。它按 +`rc_grid.json` 中组件 cell 的真实体积权重分配节点焦耳,并写入事件、源/网格/ +事件哈希、逐组件守恒和阶段窗口收据。`ACTIVITY` 行不再次积分,避免重复能量。 +固定测试 4/4 通过;测试未调用 solver。 + +最小执行顺序如下,须放入新的 point 目录并由统一资源 runner 执行: + +```text +python3 experiments/eq3_maintenance/pilot_energy_replay.py \ + --energy-csv /energy.csv \ + --grid /rc_grid.json \ + --events-output /events.txt \ + --receipt-output /conversion_receipt.json + + --run \ + --model /model.txt --events /events.txt \ + --step-s 0.02 --slot-s 0.02 --sample-s 0.02 --end-s 10 \ + --min-k 300 --max-k 400 \ + --model-sha256 --events-sha256 \ + --runner-source-sha256 \ + --domain-version EQ3_MAINTENANCE_PILOT_REPLAY_2MM_V1 +``` + +成对比较必须使用相同的 500 个整数纳秒窗口和相同 275 个 sensor 定义,逐时刻 +比较 temperature、hotspot cell、全局 min/max,以及累计输入、边界损失、储能变化 +和残差。旧 runner 的全节点 stdout 约有 32,320,512 行,应流式归约并受 4 GiB +单点边界约束,不生成无界临时副本。原服务 `thermal.csv` 是另一侧不可修改原始证据。 +温度比较不能只比较峰值。成对回放保留原始 `OBSERVATION` phase,另按 manifest +的已声明 8 s active 边界报告 `[0,8 s]` active 与 `(8,10 s]` recovery;这是从 +已声明边界派生的 segment,不改写原始 phase,也不冒称 drain。 + +此点当前为 `READY_FOR_COORDINATED_RUN`,尚未执行。它不使用 1 mm、不重跑 P2, +也不证明 reference 已合格。若温度一致,只能报告 +`FULL_2MM_DISCRETE_EQUIVALENCE`;策略重新闭环需要可替换热消费者接口及新的成对 +运行,本次开环回放不能代替闭环一致性。 + +## A2 与其余五项核心消融 + +| 消融 | 实际生产者/消费者 | 当前状态 | 最小缺口 | +|---|---|---|---| +| A2 无维护争用 | MQSim 维护命令和 fabric 真实共享资源;年龄仅在成功 commit 后更新 | `NOT_READY` | 只有 maintenance on/off,没有“同一维护意图、能量和年龄提交但使用理想独立资源”的实验模式。关闭维护不等价。需实验 backend/fabric 的显式独立资源策略和固定意图回放;不能在分析层伪造。 | +| issue / consumption | 请求已有 arrival/admission/media/fabric delivery/reported completion | `LIFECYCLE_FACTS_ONLY` | 没有 LLM layer consumption deadline 或 issue dependency 的消费者;`external_wait` 不是 consumption wait。 | +| prefetch | 旧计划登记 `none/on_demand/one_layer_ahead` 名称 | `DECLARED_ONLY_FOR_EQ3` | 当前隔离 coordinator 没有已验收的 prefetch 请求生产者,亦无额外物理流量/热量消费者。 | +| fixed / adaptive placement | W1 全局 page、静态 stripe/local page 真实消费 | `FIXED_BASELINE_ONLY` | 没有 adaptive migration、placement decision、搬运能量或映射更新。不可把改变起始页称 adaptive。 | +| page coalescing | 当前每请求严格一个 16 KiB page,stack map 要求同类、对齐单页 | `NOT_READY` | 无多页合并器及后端/能量/完成语义;逻辑计数合并不构成物理 coalescing。 | +| cache sizes | fabric 有 8 MiB HBF、4 MiB HBM 双 bank;profile 另有 64 MiB `hbm_cache_bytes` | `BUFFER_ONLY/DECLARED_CACHE_FIELD` | 双 bank 是转发缓冲,不是权重/KV cache;`hbm_cache_bytes` 未发现隔离链路 cache 命中/替换消费者。不能扫描该字段冒充 cache 消融。 | + +因此这五项目前没有合法成对实验 arm。固定 placement 可作为基线事实,但缺少 +adaptive 对照。补齐它们会改变 workload 因果或模块职责,超出本就绪审查。 + +## 九个敏感性轴 + +`有消费者` 表示当前 pilot 确实读取并影响运行;`范围` 区分注册来源和本阶段 +已冻结可扫水平。任务书要求无可辩护范围时只阻塞该轴。 + +| 轴 | 当前真实消费者 | 已登记值/依据域 | 就绪结论 | +|---|---|---|---| +| Ea | 无;`configs/eq3_thermal/reliability.json` disabled,当前年龄按秒推进 | HeatWatch 旧 3D NAND 20–70°C、1k–10k P/E:1.01/1.04/1.08 eV;不是目标 HBF uncertainty | `RANGE_AVAILABLE_PROXY, CONSUMER_ABSENT`。可用于将来的条件年龄机制敏感性,不能影响当前运行,更不能生成 ECC/失败概率。 | +| TIM 垂直路径 | 已烘焙进完整网络 C/G:k=10 W/(m K),HBF/HBM/GPU 厚度 125/189/175 µm | 当前值为 MFIT/几何代理;另有 k=5 文献情景提示,但未冻结三水平,HC5000 整体阻抗不适配 | `MODEL_CONSUMER_ACTIVE, SCAN_BLOCKED_RANGE`。改变厚度/材料须重生成一致 C/G;不能逐 edge 乘系数。 | +| 顶部冷却 | model 顶 Robin 边界 | 1400 W/(m² K) 单一 MFIT 代理 | `MODEL_CONSUMER_ACTIVE, SCAN_BLOCKED_RANGE`;没有已冻结低/中/高。 | +| HBF energy/byte | 没有单一 HBF J/B;ledger 分别消费 NAND media 0.05 W、command/data-out 0.01 W 与 fabric 2 pJ/B | 均为工程情景,绝对 HBF read/program/PHY 能量仍缺 | `MULTI_STAGE_CONSUMER_ACTIVE, SCAN_BLOCKED_SEMANTICS_AND_RANGE`。不得把各阶段无条件缩成一个 J/B 旋钮。 | +| ambient | model 初态、top、bottom 均为 300 K | 单一代理值 | `MODEL_CONSUMER_ACTIVE, SCAN_BLOCKED_RANGE`。初态与环境同移是平移诊断;只改环境会引入冷却瞬态,二者不可混称。 | +| GPU 外热 | `ActivityEnergyLedger.flush` 每窗实际加入 GPU 能量 | 当前 pilot 200 W;旧公共 trace 有 40/65/100 W 情景,均非产品校准 | `READY_TO_FREEZE_CONDITIONAL_LEVELS`。接口已经闭合,但三水平组合及共同 workload 尚未在本审查冻结;可不改代码形成 OAT。 | +| HBF 温限 | `ThermalClient` 实际按 353.15/363.15/378.15 K 控制 | 一套 research scenario,产品 limit 仍 UNKNOWN | `CONSUMER_ACTIVE, SCAN_BLOCKED_RANGE`。三个数是 Light/Severe/Shutdown 状态边界,不是一个轴的低/中/高三套配置。 | +| HBM 温限 | 同上,当前同为 353.15/363.15/378.15 K | 一套 research scenario,产品 limit UNKNOWN | `CONSUMER_ACTIVE, SCAN_BLOCKED_RANGE`;不得因数值与 HBF 相同而宣称同一器件限值。 | +| GPU 温限 | `ThermalClient` 实际按 363.15/373.15/383.15 K 控制 | 一套 research scenario,产品 limit UNKNOWN | `CONSUMER_ACTIVE, SCAN_BLOCKED_RANGE`。`compute_die_first` 在没有三套明确限值前不能形成该轴 OAT。 | + +当前没有任何轴同时满足“已冻结三水平、实际消费者、相同输入可成对”的完整条件。 +GPU 外热最接近就绪,只需在已有登记情景中明确三水平和输入等价规则;Ea 即使有 +三点代理,也因运行消费者缺失而不能计入点数。因此本审查的真实可启动敏感性矩阵 +为 **0 点**,而不是先写 38 点。若仅 GPU 外热随后被正式冻结为三水平,则去重 +物理组合为 `1+(3-1)=3`,乘两个策略为 6 点;这只是计数公式,不是本文件启动授权。 + +九轴均不得以 2.069 K 局部网格差作为通用实物误差范围。涉及 TIM、冷却或 ambient +的变化必须重生成完整物理一致网络并做低成本能量/响应检查;其结果不自动继承 +当前模型资格。 + +## 本次新增与验证 + +- `experiments/eq3_maintenance/pilot_energy_replay.py`:A3 只读能量回放转换; +- `experiments/eq3_maintenance/test_pilot_energy_replay.py`:能量/体积分配、阶段保留、 + 窗口连续性、未知组件及非法能量固定测试; +- 执行结果:`4 tests`, `PASS`, 约 0.002 s;无 solver、无数值输出、无原件覆盖。 +- phase 审核:末 2 s 是 observation 内的 recovery,不是 drain;转换器保持该正交 + 语义,本次未修改原始结果或主 coordinator。 + +尚未执行 A1 或 A3 数值点,未生成敏感性点,未打开 blind 数据。 diff --git a/docs/eq3_thermal/ISOLATED_CAMPAIGN_RESULT.md b/docs/eq3_thermal/ISOLATED_CAMPAIGN_RESULT.md new file mode 100644 index 0000000..9f09934 --- /dev/null +++ b/docs/eq3_thermal/ISOLATED_CAMPAIGN_RESULT.md @@ -0,0 +1,86 @@ +# EQ3-ISOLATED-MAINTENANCE-CAMPAIGN-v1 实际交付 + +状态:按用户最新研究目标转向而停止后续事务矩阵。共57个点完成(v3复用48+v4完成9);原3次维护迟交失败保留,另9点未执行。完整身份与行级结果见工件 `campaign-ocp4k-v4/SCOPE_PIVOT_SUMMARY.json`。这不是66点全完成,也不把执行完成视为物理模型通过。 + +本轮建立了与原项目隔离的 MQSim 维护后端,并贯通前台读取、同引擎维护、封装交付、分源能量、完整耦合热网络和未来请求准入。默认 MQSim/HBFSim、生产 ABI/PTX/TMA/future/cache 和原基线未替换。适用状态为 `CONDITIONAL_ENGINEERING_USE`;原 P2 空间差、400.911 K 越域、v1/v2 失败与未开封盲测保留。 + +## 实际能力与证据 + +- 一个实验 MQSim engine 内共用 FTL、有限页分配器、TSU、PHY/NAND。维护执行 read → program destination → 源版本核对 → mapping commit → retire old;只有合法成功提交才清相应页年龄。源块保护和递增版本防止覆盖并发新写。旧块仍含有效页时保留 `COMMITTED_RECLAIM_DEFERRED`,不把它算作擦除或失败。 +- 维护关闭时,与原后端的独立进程 A/B 在 8 stack × 1 die 和 4 stack × 16 die 两组中,完成时间、命令顺序和地址分配保持精确一致。异路径源码重建后再次通过 A/B。实验进程只链接一套 MQSim。 +- 维护验证范围是 `METADATA_VERSION_VALIDITY`,没有实际 payload 字节完整性、ECC/RBER 或器件可靠性测量。主矩阵前台只读,维护产生的 program/erase 独立计数;另有明确隔离的启动写入诊断。 +- CPU 组合的同一时间轴保留原始到达、后端提交、媒体/后端完成、fabric 最终交付。gate 只影响未提交请求;暂停输入时仍推进在途、维护、冷却和恢复。`raw_callback == reported_completion` 守卫保留。 +- 物理热网络包含 GPU、逐 HBM/HBF die、base 与共享封装路径。mixed/relay/DASH 使用 64,512 节点、275 传感器;8HBF 使用独立生成的 40,960 节点、307 传感器封装模型。外部 GDDR 服务与温度未建模。 + +完整接口生产者、消费者、启用方式与缺口见 [接口表](ISOLATED_INTERFACE_STATUS.md);几何和温度依据见 [几何与热证据](ISOLATED_GEOMETRY_AND_THERMAL_EVIDENCE.md)。 + +## Page / block / plane 与负载 + +新 profile 明确使用 4 KiB/page、16 die/stack、512 GiB 逻辑容量/stack。16 host channel 和 bank 层级转为每 stack 16 MQSim channel、每 channel 1 die、每 die 16 plane,是有标记的研究映射。256 page/block 是工程假设,OCP v0.7.0 的产品 R3 未给统一值;不伪称来自规范。MQSim 动态页级 FTL 也不等于完整 OCP zone/direct-block 协议。 + +| 层级 | 本轮实验配置 | 依据/限制 | +|---|---:|---| +| page | 4 KiB | OCP v0.7.0 | +| block | 256 page = 1 MiB | 产品未知,工程假设 | +| plane | 2,048 block = 2 GiB | 与容量联动推导;bank→plane研究映射 | +| die | 16 plane = 32 GiB | 研究映射,不声称目标产品内部布局 | +| HBF stack | 16 die = 512 GiB | OCP逻辑容量;真实spare另计且未知 | +| 4/8 HBF | 2/4 TiB | 实际实例化逻辑容量 | + +全容量探针覆盖 4 × 16 die × 16 plane = 1,024 个实际物理组合。独立边界探针在同一 plane 中观察前 256 页分配到 block 0 的 page 0..255,随后分配到 block 5/page 0;真实分配器保留的写入前沿使“下一个块必为 block 1”的假设不成立。该失败断言与成功复验均保留。64 页指选定的维护样本,而非一个 die 或器件的总页数;它不能证明全容量保持能力。 + +读取需求由官方 Qwen 模型索引中的 BF16 权重字节数产生:7B 为 15,231,233,024 B,72B 为 145,412,407,296 B。使用共同、明确合成的 20,000 s 扫描周期,模型大小改变请求量;W1 为顺序 20 ms 突发,W2 为选定 layer 区域的 100 ms 突发。此有限事件 replay 很稀疏,不能声称真实解码速度、产品 TB/s 或 token/s。所有 stack 均有实际请求。每个点报告实际访问覆盖比例。 + +**读取带宽尚未与完整产品速度档对齐。** fabric采用SanDisk第一代1.6TB/s量级,但MQSim原生通道仍是每栈16×8bit×1600MT/s的NAND工程代理、单页读取10μs,整个engine另有共享512GB/s完成包络。这不是OCP1.536或3.072TB/s每栈的已验证实现;几何对齐不等于性能对齐。现有稀疏主矩阵不能支持目标峰值带宽、饱和利用率或该带宽下的热稳态结论。源头、消费者与已有读取探针见 [读取带宽对齐审计](READ_BANDWIDTH_ALIGNMENT_STATUS.md)。 + +## 控制逻辑及解释边界 + +三臂共享严重保护:P0 guard_only;P1 thermal_hysteresis_guard;P2 read_rate_feedback_thermal_guard_v1。20 ms 采样和动作延迟、100 ms 恢复驻留、2 K 滞回;升级不被恢复驻留额外阻挡。P2 只根据已完成窗口的有效交付、积压、等待、后端忙碌及共享资源事实改变未来 byte 配额。无需求不学习;仅在门控限制且后端可用时增加;没有交付收益而积压/尾延迟恶化时撤回增加。 + +恢复段旧实现仅看新到达,误判既有积压为无需求,已由固定测试复现并修复。旧 16 KiB Stress P2 的 9,074 个未完成请求保留,不能用修复后结果覆盖旧失败。新主矩阵重新冻结并公平重跑。 + +GPU 100°C slowdown/110°C shutdown 参考相似 AMD 加速器官方示例,属于转用代理;90°C 普通预警为 10 K 研究裕度。HBF/HBM 温限仍为研究假设。当前 GPU 外热独立指定,因此即使存储暂停也不会自动关闭 200 W GPU 源;超温说明该执行器不足,不能解释为芯片允许该温度。物理热耦合已经包含 GPU 向存储传热。GPU 读取饥饿导致计算功耗下降、温度加速 retention 和 ECC 重试曲线尚无因果消费者,未虚构加入。 + +维护期限与年龄已观测,严重保护可造成期限违约;当前维护使用真实共享 TSU 的既有优先队列,没有额外实现期限自适应抢占。它是明确的能力限制。 + +## 成对实现与机制验证 + +| 项目 | 实际结果 | 可支持范围 | +|---|---|---| +| 旧 16 KiB 完整 pilot | 4,032/4,032 前台完成,64 实际维护 commit;GPU 峰值 396.046397 K | 旧工程输入完整链;超保护温度并非安全运行 | +| 新 4 KiB 完整 pilot | 7,260/7,260 完成,64 实际维护 commit,0 删失;GPU 峰值 387.619922 K | 满逻辑容量的新几何 CPU 闭环,非产品标定 | +| A1 源域隔离响应 | 实体均温线性叠加误差 ≤8.6061e-11 K;HBF 本源热点约300.0008 K、全源约335.18 K | 原 C/G 和散热路径下的跨源贡献;不是删边独立网络,也不是策略重新闭环 | +| A2 固定维护意图理想独立回放 | 4,032 前台完成;10 请求各提前620 ns,总6,200 ns;P95、峰温、能量不变 | 有限旧输入的争用归因;当前引擎实际维护0,64项是明确的反事实回放,不计作实际维护 | +| A3 全2mm同输入成对回放 | 500×275 sensor 最大差4.718e-12 K,累计能量差4.405e-11 J | 同方程/Eigen求解族的离散适配一致性,非独立物理参考 | + +## 初始状态限制与维护失败 + +新低负载 Safe/Near 的 Q1 原始结果中,每臂38次维护提交、26次 `REJECTED_UNMAPPED`。这些目标页在到期时尚未由首次前台读取建立有效映射;后端立即拒绝,零NAND子事务、零年龄清除,随后才被前台访问。初始年龄元信息不会创建驻留映射,因此这不是实际数据已驻留且老化的全容量权重初始化。 + +统计将无效源拒绝与接受后超截止期分开。它们是初态能力/输入域限制,不能归为维护吞吐改善或后端超时。已有实际映射的 Stress 目标仍可做合法维护对照。完整预装载需要明确的初始化阶段及时间/能量合同;本轮保留原对照,不偷偷插入host写或将额外工作移出统计。详见 [原始证据诊断](V3_Q1_UNMAPPED_MAINTENANCE_DIAGNOSIS.md)。 + +## 启动写入与并发诊断 + +新增独立 caller 复用同一隔离 `MqsimOnlineEngine` 的实际 Write 接口;每次完成立即补充请求,保持有界滚动窗口。没有移除资源锁或更换 TSU 策略。在同一四栈、64个die资源的16,384页固定对照中,窗口1为40.81 MB/s,窗口256为5.182 GB/s模拟program吞吐;滚动窗口整个写入区间平均63.255/64个die活跃(98.837%),包括初始填充与最终排空。该均匀负载下原批次边界恰好没有造成媒体空档;新增滚动补充消除了对这一巧合的依赖。 + +这些是工程后端并发证据,不是HBF产品写带宽。全部16,384次program和同页读取唯一完成,另有64页写→读→维护提交固定检查;真实tensor payload未传输,字节完整性仍UNKNOWN。完整7B/72B/235B预装载尚未运行;当前保留全部事件的路径随页数显著增长,不能把小样本按比例外推为已完成全模型写入。具体阶段与成本见 [启动写入结果](STARTUP_WRITE_FIXED_DIAGNOSTIC_RESULT.md)。 + +实验完成或失败由负责代理通知;没有通知机制的作业按实测ETA检查。进程内资源watchdog继续执行,不让AI反复读取未变化状态。 + +## 数值资格与未实现分支 + +v3 穿越在完整轨迹一次提取、匹配后分窗;旧 train 边界事件为 `INDETERMINATE_QUANTIZATION`,未改为 PASS;development v3 通过所测离散判据。旧 v1/v2 并列保留,未回填。原 0.25 K 参考相邻差要求仍失败,接受工程使用不等于 MODEL_FREEZE。 + +五项额外消融中,issue/consumption 缺 LLM 因果消费者,prefetch/adaptive placement/page coalescing/cache 无已验收的成对实现。fabric 双 bank 是有限转发缓冲,并非权重 cache;未消费的 cache 字段不作为扫描参数。敏感性首批仅 GPU 外热40/100/200 W 的有来源情景轴,其他八轴缺目标范围或实际消费者,均不虚构完整矩阵。详见 [消融就绪及执行记录](ISOLATED_ABLATION_READINESS.md)。 + +源目录:`eq3_thermal/integration/main-20260920`;工件根:`eq3_thermal/plans/isolated-maintenance-campaign-v1`。旧原始结果、失败收据与本轮新增结果分别保存;无 push/merge、GPU 负载、云费用或全局环境修改。 + +## 最新范围转向与维护迟交修复 + +用户明确希望研究各stack不同读取速率引起的温度变化,而非完成真实逐页读写。后续主线转为直接“规定通道速率→分die/base功耗→完整耦合热网络”。MQSim矩阵在当前已启动点正常结束后暂停;无强杀、无删除raw、无继续性能细化。 + +隔离修复82a47f7保留原维护due/deadline,仅将合法事件注入时刻设为max(due,current)。原生过期拒绝策略不变。固定on-time、迟交但未过期、迟交过期零NAND/唯一完成通过。v4已完成9个refresh点每点64个任务均在原截止期后提交,合计576个过期拒绝、0commit,年龄未清零。它们不是维护收益。 + +旧57点仍为条件工程结果:前48点四拓扑均完成且资源清空;P0/P1在16组所选指标相同,P2没有提高完成字节,Stress尾延迟增加;GPU固定200W占活动能量99.9994%以上,说明旧稀疏读取并不适合研究目标高带宽下的HBF自热。更大235B模型元数据与生成器补丁仅作保留资料,未作为已运行正式矩阵报告。 + +新读率热路径使用用户明确确认的50pJ/B增量假设,并按最新要求选OCP Grade2的16通道×96GB/s。该新路径与旧.05W/活跃die工程事件功率不混合;新旧温度不得直接归因为控制改进。 diff --git a/docs/eq3_thermal/ISOLATED_FIX_AND_LIMIT_REGISTER.md b/docs/eq3_thermal/ISOLATED_FIX_AND_LIMIT_REGISTER.md new file mode 100644 index 0000000..b08bc04 --- /dev/null +++ b/docs/eq3_thermal/ISOLATED_FIX_AND_LIMIT_REGISTER.md @@ -0,0 +1,54 @@ +# Isolated MQSim fix and limit register + +This register separates software defects, backend/model limitations, numerical +qualification limits, domain failures, resource blocks, and claims that were +not reproduced after a bounded fix. Evidence paths are relative to +`eq3_thermal/plans/isolated-maintenance-campaign-v1/` unless a path starts with +`../`. Fixed tests and startup checks establish only their named invariant; +they are not counted as a homogeneous experiment pass. + +The machine-readable companion is +`eq3_thermal/plans/isolated-maintenance-campaign-v1/CURRENT_EVIDENCE_INDEX.json`. +It is a receipt index, not a replacement for immutable raw data or the final +campaign aggregation. + +## Repair and limitation register + +| Item | Classification | Required invariant or capability | Reproduction and evidence | Current state and remaining limit | +|---|---|---|---|---| +| Maintenance source version and ownership | `CONFIRMED_BUG` | A maintenance read must commit only if the logical mapping still has the captured version; its source block must remain unavailable to GC until terminal state | `points/backend-generation-cpp-test-v1/result.json`; `points/backend-generation-service-test-v2/result.json`; `points/BACKEND-GENERATION-FROZEN/result.json` | Fixed in the isolated backend with monotonic generation, overflow failure, source pinning, pinned-GC rejection, and an actual scheduled foreground-write race. The guarantee is `METADATA_VERSION_VALIDITY`; no payload equality is claimed. | +| Maintenance energy source label | `CONFIRMED_BUG` | Actual maintenance events must be categorized from their maintenance identity without changing energy totals | `points/energy-fixed-v2/result.json`; actual OCP loop receipt `points/OCP4K-LOOP-PILOT01/DONE.json` | Fixed producer lookup prefers `maintenance_request_id`, with legacy fallback. The immutable PILOT03 raw keeps its older `BACKEND_BACKGROUND` label; stored joules and A3 remain unchanged. | +| Recovery backlog treated as absent demand | `CONFIRMED_BUG` | Carried queued work remains demand after the arrival interval ends | `points/policy-recovery-demand-repro-v1/result.json`; `points/policy-recovery-demand-fixed-v1/DONE.json`; `points/MAIN-v1-partial-analysis/partial-analysis.md` | Fixed demand accounting uses delivered work plus backlog. The old P2 Stress observations remain valid evidence of that implementation, including 9,074 censored requests, but cannot serve as an unqualified comparison of the intended policies; affected pairs need new campaign IDs. | +| GPU key included in memory-stack coverage | `CONFIRMED_BUG` | The separate GPU thermal entity must not be required to appear in the configured memory stack set | Preserved failure `points/Q1-PILOT01/FAILED.json`; fixed wrapper check `points/thermal-wrapper-fixed-v1/result.json` | Fixed. The first pilot is retained as a wrapper failure, not a thermal-domain result. | +| Lowercase native NAND technology token | `CONFIRMED_BUG` | Generated profiles must use a token accepted by the native MQSim parser | `campaign-ocp4k-v2/PAUSED.json`; `campaign-ocp4k-v2/SLC-STARTUP-FIXED02/DONE.json`; `campaign-ocp4k-v3/PREFLIGHT_VALIDATION.json` | Fixed at source head `9507fe2dd6f2ec1162c7063614ad66d870039bc1`. The v2 attempts completed zero points; they are startup failures, not numerical failures. | +| Default product path isolation | `NOT_REPRODUCED_ALREADY_FIXED` | Maintenance-off operation of the isolated executable must preserve the default service result, and an independent rebuild must link only its private MQSim archive | `points/AB-PROCESS-GENERATION-02/result.json`; `points/BACKEND-INDEPENDENT-REBUILD01/result.json`; `points/AB-PROCESS-INDEPENDENT-REBUILD01/result.json` | The bounded A/B cases are exact and the fresh-path rebuild passed. This supports default-off isolation for the tested request sets, not universal equivalence over all MQSim configurations. | +| Full-capacity identity | `DESIGN_LIMITATION` | Logical addressability, physical spare, bad-block allowance, and endurance capacity must remain distinct | `points/GEOMETRY4K-FULLCAP01/result.json`; `points/GEOMETRY4K-PHYSICAL-BLOCK02/result.json`; preserved assumption failure `points/GEOMETRY4K-BOUNDARY01/FAILED.json` | Actual MQSim accepted 4 HBF stacks x 512 GiB logical capacity with 4 KiB pages. Physical spare and bad-block capacity remain unknown. `BANK_AS_MQSIM_PLANE_V1` and 256 pages/block are study projections, and zone-FTL ordinals are not direct physical block/page identities. | +| HBM, external fabric, and GDDR capability | `DESIGN_LIMITATION` | Each backend and link may claim only behavior it actually produces | `points/Q1-WEIGHT-MAINT-PILOT03/DONE.json`; `points/OCP4K-LOOP-PILOT01/DONE.json` | HBF requests and maintenance use the isolated actual MQSim path. HBM remains a parameterized scenario model; fabric is an external basic model; external GDDR has physical identity but unavailable service and package temperature. Unknown energy inputs remain `UNKNOWN`, not zero. | +| A1 source-group response | `DESIGN_LIMITATION` | Source ablation must preserve the same C/G network and state exactly what was removed | `points/A1-DOMAIN-ANALYSIS01/result.json` | Completed with entity-mean superposition error `8.606e-11 K`. It removes other source inputs; it is not a disconnected physical-domain experiment, nodewise hotspot sum, or control-policy reclosure. | +| A2 actual shared maintenance versus ideal replay | `DESIGN_LIMITATION` | Actual backend maintenance must be distinguished from a counterfactual replay that consumes no current-engine maintenance resources | Baseline `points/Q1-WEIGHT-MAINT-PILOT03/DONE.json`; paired audit `points/A2-PILOT03-LOOP01/POSTCHECK.json` | Baseline issued and committed 64 actual jobs. A2 issued zero backend jobs and replayed 64 fixed facts with `UNKNOWN_REPLAY` mapping/version semantics. Ten of 4,032 foreground completions moved 620 ns earlier; aggregate completion count, p95, stored peak temperature, and total energy were unchanged. This localized outcome does not prove independence generally. | +| A3 thermal replay identity | `NOT_REPRODUCED_ALREADY_FIXED` | A full-window replay must preserve all requested sensor frames under the same discrete equation | Preserved harness failure `points/A3-PILOT03-REPLAY01/FAILED.json`; successful receipt `points/A3-PILOT03-REPLAY02/RUN_RECEIPT.json` | Full 2 mm discrete equivalence passed for 500 frames and 275 sensors; maximum sensor difference was `4.718e-12 K`. Shared equation and Eigen family mean this is not an independent physical reference or P2 qualification. | +| P2 adjacent-grid accuracy | `NUMERICAL_ACCURACY_LIMIT` | Retained reference comparisons must meet the original 0.25 K criterion before reference qualification | `../campaign-v1/index.json`; `../minimal-repair-v1/P2-COMMON-FIRST4S-DIAGNOSTIC.json`; source summary `docs/eq3_thermal/DECISION_EXECUTION_V2_RESULT.md` | The old criterion remains failed; later full-window diagnostics report a 2-to-1 mm maximum difference of 2.069 K. Conditional engineering use does not convert this into reference qualification: `MODEL_FREEZE=false`. | +| Development reference above 400 K | `DOMAIN_FAILURE` | A trajectory outside the declared 300--400 K constant-property domain must fail with evidence | `../campaign-v1/index.json`; `../minimal-repair-v1/points/P2-ORIGINAL-DOMAIN-REPRO/result.json` | The independent 2 mm development reference reached 400.911 K and first crossed at 31.14 s. It remains failed; no clamp, threshold widening, or truncated pass was applied. | +| Original full 1 mm, 100 s attempt | `RESOURCE_BLOCKED` (historical, resolved) | The full reference must run as one qualified trajectory under the point's declared budget | `../campaign-v1/index.json`; `../minimal-repair-v1/points/P2-COMMON-FIRST4S/result.json`; completed R03 receipt `../decision-execution-v2/RUN_INDEX.json`; summary `docs/eq3_thermal/DECISION_EXECUTION_V2_RESULT.md` | The original 4 s prefix took 328.012 s and exceeded that campaign's 600 s budget projection; it was not split to bypass the watchdog. The later approved 3,600 s R03 completed all 5,000 solve frames and 275-sensor observation: 2,763.597 s solve, 833.560 s observation, and 7,735,140 KiB solve peak sampled RSS (about 7.38 GiB). The resource block is resolved; spatial qualification still fails and `MODEL_FREEZE=false`. | + +## Evidence inventory by use + +| Evidence set | Kind | What is established | What is not established | +|---|---|---|---| +| Generation/pin, source-label, backlog, GPU-key, parser-case receipts | Fixed tests and minimal reproducers | The named software invariant is reproduced and repaired | Campaign performance, physical calibration, or topology-wide benefit | +| Independent rebuild and process A/B receipts | Build reproducibility and bounded compatibility | The isolated source/patch path rebuilds and the tested maintenance-off request sets are exact | Cross-platform reproduction or equivalence for every native configuration | +| PILOT03 | Legacy 16 KiB engineering pilot | One actual closed CPU loop with real shared MQSim maintenance | OCP 4 KiB geometry, P2 qualification, or product performance | +| OCP4K-LOOP-PILOT01 | 4 KiB full-logical-capacity engineering pilot | One actual closed CPU loop using the new logical geometry and mapping projection | Physical spare, calibrated HBM/fabric, external-GDDR temperature, or a formal matrix result | +| A1 | Source-domain ablation | Bounded linear source-response accounting | Network disconnection or policy reclosure | +| A2 | Paired ideal-resource counterfactual | Difference between actual shared maintenance and fixed replay at the frozen PILOT03 input | A general independent-resource performance claim | +| A3 | Paired full-window discrete replay | Same-equation sensor/energy transport equivalence | Independent physical reference, 1 mm result, or P2 model freeze | +| MAIN v1 partial | Superseded legacy campaign fragment | Nine conditional 16 KiB Q1 results and direct evidence of the old recovery bug | Four-topology completion or a valid affected P2 comparison | +| OCP4K v3 | Live campaign | Validated 66-point queue bound to head `9507fe2` | Any aggregate conclusion before the root task freezes and reviews `campaign-ocp4k-v3/STATUS.json` | + +## Current campaign boundary + +`campaign-ocp4k-v3/` is live and mutable. Its preflight and engineering lock may +be cited as identity evidence, while point completion and aggregate claims must +wait for the root task's final index. Current partial status must not be combined +with the superseded 16 KiB MAIN v1 rows, the standalone pilots, or the A1/A2/A3 +ablations and described as one homogeneous pass count. diff --git a/docs/eq3_thermal/ISOLATED_GEOMETRY_AND_THERMAL_EVIDENCE.md b/docs/eq3_thermal/ISOLATED_GEOMETRY_AND_THERMAL_EVIDENCE.md new file mode 100644 index 0000000..ffac5b2 --- /dev/null +++ b/docs/eq3_thermal/ISOLATED_GEOMETRY_AND_THERMAL_EVIDENCE.md @@ -0,0 +1,189 @@ +# Isolated geometry and thermal evidence + +Status date: 2026-09-20. This is a receipt-backed engineering status report. +It does not create a new thermal model, rerun P2, open blind data, or promote a +conditional pilot into product validation. Evidence labels follow the project +registry: `SPECIFIED`, `DOC_DERIVED`, `SCENARIO_ASSUMPTION`, and +`CONDITIONAL_ENGINEERING` remain distinct. + +## Current conclusion + +The isolated chain now closes a real 4 KiB MQSim request and maintenance path, +full logical HBF capacity, native command/physical-address observation, +activity-energy accounting, and the persistent complete 2 mm thermal service. +Beyond the original mixed-direct pilot, a frozen v3 snapshot contains 48 +completed points: 12 each for mixed-direct, relay, DASH, and 8HBF direct. A1 +source-domain response, A2 ideal-maintenance replay, and A3 full-2 mm replay +have also completed. + +These results establish interface and discrete-engine behavior. They do not +establish target-product NAND organization, calibrated power, ECC/retention +failure probability, token throughput, complete OCP zone/direct addressing, +live GPU behavior, P2 reference qualification, or `MODEL_FREEZE`. + +## Page, plane, block, and capacity + +The registered primary source is OCP *High Bandwidth Flash High-Level Base Die +Specification* v0.7.0, 3 August 2026, +`docs/HBF_OCP/ocp2026-hbf-architecture-specification-v0-7-0.pdf`, SHA-256 +`307531eb8053f00cbeccbc907ddff0a9c4fe6f9d0066a077ce33b0ac99312da3`. + +| Property | Source status | Executed representation and evidence | +|---|---|---| +| Page | OCP §4.1 specifies 4 KiB NAND pages; reads may not cross a 4 KiB page and Core writes use 4 KiB | Native 4096-byte requests are accepted. The stack map rejects a 16 KiB request under this profile and rejects the first page beyond the logical namespace. | +| Capacity | OCP Table 3 specifies 512 GiB and 16 dies/cube; §4.3 specifies 16 host channels | Four-stack probe instantiated 2 TiB logical capacity: 134,217,728 pages/stack, 64 channels total, one MQSim die/channel, and 16 thermal-die identities/stack. | +| Bank/plane | OCP specifies 16 banks/channel but does not define universal bank=plane identity | `BANK_AS_MQSIM_PLANE_V1` maps 16 banks to 16 MQSim planes/die as a `SCENARIO_ASSUMPTION`. It is not a product fact. | +| Block | OCP §5.7 leaves R3/pages per NAND block product-defined | `pages_per_block=256` is a visible `SCENARIO_ASSUMPTION`, producing 2048 simulated blocks/plane at full logical capacity. | +| Physical allocation | MQSim page-level FTL | CWDP selects channel/die/plane; the FTL dynamically selects physical block/page. Logical block-like ordinals must not be reported as physical addresses. | + +`GEOMETRY4K-FULLCAP01` completed 1024/1024 real 4 KiB reads and covered all +`4 stack × 16 channel-derived thermal die × 16 projected plane` tuples. Four +maintenance operations completed native read, destination program, mapping +commit, source retirement, and final completion. Peak RSS was 21,511,976 KiB +and wall time 12.31 s. + +`GEOMETRY4K-BOUNDARY01` retained a failed assertion caused by assuming that a +logical ordinal predetermined physical block/page. It is classified +`NOT_A_BACKEND_BUG_TEST_ASSUMPTION`. Its valid subchecks confirmed completion +of global logical page 536,870,911 and rejection of the first out-of-range +stack page. + +`GEOMETRY4K-PHYSICAL-BLOCK02` then issued 257 serial first-touch pages to one +channel/die/plane. Physical block 0 received pages 0 through 255; allocation +257 entered observed block 5 at page 0. The new block number was not assumed, +because MQSim reserves other work-front blocks. This proves the 256-page block +parameter has a real allocator consumer. It does not implement or validate the +complete OCP zone/direct block-address protocol. + +### Maintenance scale versus capacity + +The completed OCP4K loop committed 64 one-page maintenance jobs. At 4 KiB/page +this is 262,144 B (256 KiB). The four 512 GiB stacks contain 536,870,912 +addressable-equivalent 4 KiB pages, so the maintenance sample covers +`64 / 536,870,912 = 1.1920929e-7` of aggregate logical pages, or +`0.0000119209%`. With the observed equal 16-job distribution, the same fraction +holds per stack. + +This is lifecycle and contention evidence, not full-capacity refresh coverage, +retention qualification, steady maintenance-rate calibration, or lifetime +evidence. Physical spare, bad blocks, overprovisioning, and product R3 remain +unknown. Compared with the old 16 KiB pilot, 64 jobs now cover 256 KiB rather +than 1 MiB; old/new maintenance volume is not byte-equivalent. + +## Connected thermal evidence + +The persistent thermal service uses the complete 2 mm, 64,512-node network, +255 package entities, 275 sensors, 20 ms windows, and one reused sparse +factorization. GPU, every HBF/HBM base and die, shared package materials, and +the common cooling boundary remain in one network. Activity energy is assigned +to components before each causal advance; there is no full-field output in the +service path. + +The completed 4 KiB mixed-direct pilot `OCP4K-LOOP-PILOT01` is +`CONDITIONAL_ENGINEERING_COMPOSITION`: + +- 7,260/7,260 requests completed: 7,100 HBF and 160 parametric HBM; +- 64/64 actual shared-engine maintenance jobs committed, with zero pending or + censored requests; +- 300 windows over 6 s, total activity energy 800.0040984343195 J; +- peak temperature 387.61992164551265 K (114.4699 °C); +- wall time 83.05 s and maximum child RSS 21,510,776 KiB. + +HBM remains a parameterized service, energy coefficients remain +`SCENARIO_ASSUMPTION`, and token/s is `UNAVAILABLE`. This pilot remains a +separate single-loop receipt; the partial four-topology campaign evidence is +reported below and is not backfilled into the pilot. + +The earlier 16 KiB `Q1-WEIGHT-MAINT-PILOT03` remains historical engineering +evidence: 4,032/4,032 requests, 64 maintenance jobs, 500 windows/10 s, +1600.0025647932162 J, and peak 396.0463965549566 K. It is not OCP page-aligned +and is not retroactively relabeled. + +## GPU guard source and observed margin + +AMD's official AMD-SMI example for comparable MI300X/MI350X devices reports +100 °C hotspot slowdown and 110 °C shutdown. Those values are `PROXY`, not a +specification for a future HBF-equipped GPU. The current 90 °C guard is a +`SCENARIO_ASSUMPTION`: a 10 K early-warning margin. The modeled geometric GPU +hotspot is not a calibrated hardware sensor. + +The OCP4K raw thermal ledger places its global peak and GPU hotspot at the same +387.61992164551265 K value at 4 s. Its 114.4699 °C peak is 4.4699 K above the +110 °C proxy. PILOT03's 122.8964 °C peak is 12.8964 K above it. Numerical completion inside the +300–400 K material domain is therefore not thermal-protection success. In +these points the memory gate cannot reduce the independent prescribed GPU heat +source. Read throttling-to-GPU-utilization/power feedback is not implemented, +and no such coefficient is inferred from these results. + +The frozen completed-48 v3 snapshot uses the prescribed 200 W external GPU +source. Other queued/OAT and final-campaign points require their own receipts; +the completed 48 do not stand in for all 66. Power levels must not be confused +with the 100/110 °C guard proxy. The failed +`campaign-ocp4k-v2` startup followed a confirmed SLC serialization/input typo +before a completed scientific point. It is an engineering startup failure, +not a thermal-model failure and not four-topology evidence. The repaired +zero-work native startup receipt is +`campaign-ocp4k-v2/SLC-STARTUP-FIXED02/DONE.json`. + +## A1, A2, and A3 coupling evidence + +| Item | Actual result | Supported interpretation | Limit | +|---|---|---|---| +| A1 source-domain response | Nine source groups plus zero source, 500 windows and 125,500 rows. Entity-mean linear-superposition max error `8.6061e-11 K`; zero-source deviation `1.1084e-11 K`. HBF/HBM full peaks were about 335.18–335.40 K while their own-source peaks were about 300.0001–300.0008 K. | On the unchanged complete C/G network, cross-source package coupling dominates those memory-domain peaks for this prescribed input. | Source ablation retains passive paths; it is not edge deletion, a disconnected package, nodewise hotspot decomposition, or policy reclosure. Coefficients remain assumptions. | +| A2 ideal-maintenance replay | 4,032/4,032 foreground completions and 64/64 fixed replay jobs. Ten requests completed 620 ns earlier; completion count, censoring, P95 latency, peak temperature, and total energy were unchanged at stored precision. | Localized contention exists in the frozen legacy point. | Actual backend issued zero maintenance jobs in the replay; mappings were not mutated and age/commit facts are `UNKNOWN_REPLAY`. This is not an implemented independent-maintenance scheduler and is limited to legacy16K input. | +| A3 full-2 mm replay | 500 × 275 sensor comparisons; max sensor delta `4.7180e-12 K`, global-range delta `2.6148e-12 K`, cumulative-energy delta `4.4048e-11 J`; 32,320,512 node rows streamed. | Persistent service and old sparse runner are discretely equivalent for the immutable PILOT03 energy windows. | Same equation and Eigen family, 2 mm only: not an independent physical reference, P2 rerun, 1 mm qualification, or closed-control comparison. | + +## Four-topology status on three independent axes + +The frozen v3 completed-48 snapshot contains 12 points per topology. All 48 +are `FUNCTIONAL_PASS`, have zero censored requests and conserved fabric +resources, and complete the declared coupled package trajectory. All remain +`CONDITIONAL_NOT_MODEL_PASS`. The remaining 18 points and final 66-point +aggregate are still pending receipts. Small interface fixtures remain separate. + +| Topology | Configuration axis | Thermal axis | System-behavior axis | +|---|---|---|---| +| 8HBF direct + external GDDR | `COMPLETED48_CONFIG_EXECUTED`; 12 points, 8HBF full logical capacity, 2,048 actual native channel/die/plane tuples | `COMPLETED48_COUPLED_THERMAL_EXECUTED_CONDITIONAL`; GPU 343.714--387.600 K, HBF 307.786--328.344 K; GDDR is outside the package thermal domain | `COMPLETED48_FUNCTIONAL_PASS`; balanced eight-stack traffic, zero censoring/leak. External-GDDR service and temperature remain `UNAVAILABLE` | +| Mixed-direct | `COMPLETED48_CONFIG_EXECUTED`; 12 points, 4HBF full logical capacity + 4 parametric HBM, 1,024 actual HBF tuples | `COMPLETED48_COUPLED_THERMAL_EXECUTED_CONDITIONAL`; GPU 343.715--387.620 K, HBF 307.789--328.349 K, HBM 307.883--328.559 K | `COMPLETED48_FUNCTIONAL_PASS`; actual MQSim HBF plus parametric HBM, zero final backlog | +| 4+4 relay | `COMPLETED48_CONFIG_EXECUTED`; 12 paired-route points and 1,024 actual HBF tuples | `COMPLETED48_COUPLED_THERMAL_EXECUTED_CONDITIONAL`; full declared trajectories completed | `COMPLETED48_FUNCTIONAL_PASS`; actual MQSim HBF media plus external relay fabric, zero censoring/leak | +| Four-pair DASH | `COMPLETED48_CONFIG_EXECUTED`; 12 dual-route points and 1,024 actual HBF tuples | `COMPLETED48_COUPLED_THERMAL_EXECUTED_CONDITIONAL`; full declared trajectories completed | `COMPLETED48_FUNCTIONAL_PASS`; Stress representative exercised 3,552 HBF direct and 3,548 HBF relay completions | + +Configuration, thermal, and system-behavior axes are independent. A pass in +one column does not promote either of the others. In these 48 points the fixed +200 W GPU source contributes 99.999473%--99.999887% of recorded activity +energy, while model-payload coverage is only 0.002985%--0.019999%. Scene +duration and fixed GPU heat therefore dominate the thermal result. + +The geometry is OCP-oriented, but the performance model is not aligned to an +OCP target speed grade. Observed HBF active delivery is 0.758--3.454 MB/s; the +native configuration's nominal rate is 25.6 GB/s per stack, its whole-engine +cap is 512 GB/s, and the external fabric uses a 1.6 TB/s scenario target. The +1.536/3.072 TB/s product rates are not implemented. See +[READ_BANDWIDTH_ALIGNMENT_STATUS.md](READ_BANDWIDTH_ALIGNMENT_STATUS.md). + +## Scientific state and evidence paths + +P2 remains unfrozen. The old P2 spatial/reference qualification failures and +400.911 K domain evidence remain intact; `MODEL_FREEZE=false`. Blind inputs +remain sealed and were not read for this report. No new solver or reference run +was started. + +Primary receipts: + +- geometry and allocation: + `eq3_thermal/plans/isolated-maintenance-campaign-v1/points/GEOMETRY4K-FULLCAP01`, + `GEOMETRY4K-BOUNDARY01`, and `GEOMETRY4K-PHYSICAL-BLOCK02`; +- closed 4 KiB pilot: `points/OCP4K-LOOP-PILOT01`; +- historical loop: `points/Q1-WEIGHT-MAINT-PILOT03`; +- A1: `points/A1-DOMAIN-ANALYSIS01/result.json`; +- A2: `points/A2-PILOT03-LOOP01/POSTCHECK.json`; +- A3: `points/A3-PILOT03-REPLAY02/RUN_RECEIPT.json`; +- source audits: `docs/eq3_thermal/OCP_PAGE_CAPACITY_ALIGNMENT.md` and + `docs/eq3_thermal/GPU_GUARD_SOURCE_AUDIT.md`; +- frozen partial v3 evidence: + `campaign-ocp4k-v3/stage/COMPLETED48_INTERPRETATION.md`, + `COMPLETED48_SNAPSHOT.json`, and `COMPLETED48_COMPACT.json`. + +The v3 completed-48 snapshot is evidence only for its frozen IDs. The final 18 +points and campaign-level conclusion wait for their receipts; v2 remains a +startup/input failure rather than a thermal-model failure. diff --git a/docs/eq3_thermal/ISOLATED_INTERFACE_STATUS.md b/docs/eq3_thermal/ISOLATED_INTERFACE_STATUS.md new file mode 100644 index 0000000..7506bae --- /dev/null +++ b/docs/eq3_thermal/ISOLATED_INTERFACE_STATUS.md @@ -0,0 +1,124 @@ +# Isolated EQ3 interface status + +Status date: 2026-09-20. This document distinguishes a connected producer and +consumer from a completed engineering run, and both from physical or scientific +validation. Parameters marked `SCENARIO_ASSUMPTION` are not device calibration. + +## Interface, producer, consumer, and actual evidence + +| Interface | Actual producer | Actual consumer | Enablement | Capability and executed evidence | Remaining gap | +|---|---|---|---|---|---| +| Campaign configuration and arrivals | `campaign_inputs.py` and point-local request/maintenance inputs | `run_point.py` -> `ClosedLoopCoordinator` | Explicit isolated point CLI; default unchanged | Legacy 16 KiB finite-region and OCP 4 KiB/full-logical-capacity profiles are explicit. PILOT03 and OCP4K loops completed | Conditional engineering inputs; no product throughput or token-rate claim | +| Stack/page placement | `MqsimStackMapAdapter`, explicit channel groups and persistent stack-local page map | Maintenance service and foreground `try_submit()` | `--stack-map` plus matching profile | Stable HBF stack identity comes from configured physical channel groups, never address guessing. OCP pilot used 4 KiB single-page requests and 16 channels per HBF stack | HBM placement is parametric; mixed HBM/KV placement, migration, adaptive placement and multi-page requests remain unsupported | +| Foreground MQSim admission/completion | `MaintenanceMqsimService`, inherited `try_submit()` and `until()` | `ClosedLoopCoordinator` | Experimental binary passed explicitly | One engine owns foreground and maintenance. Raw MQSim completion must equal reported completion before fabric composition. PILOT03 completed 3,712 HBF reads; OCP4K completed 7,100 | Arbitrary raw/reported overlap composition and production host/live-GPU consumers are not connected | +| Startup foreground program/read | Standalone startup_write_probe using existing isolated Write/Read API and rolling refill | Same engine/FTL/TSU/PHY, native observer and concurrency analyzer | Explicit separate caller only; main campaign unchanged | 16,384 writes and reads conserved, QD256 mean63.255/64 active die resources;64-page write/read/maintenance lifecycle passed | Engineering finite trace; no full model upload, payload-byte integrity or host-ingress energy claim | +| Native command observation | Experimental command observer | `ActivityEnergyLedger.native()` and native timeline writer | `--native-command-observations on`; default off | Real request -> transaction -> command IDs, phase, bytes and actual channel/die/plane/block/page are retained. Unknown stays `UNKNOWN` | Read/program/transfer power coefficients remain scenario assumptions | +| Native maintenance lifecycle | Isolated unit emits due/queue/read/program/commit/retire/erase/fail and native children | Service client, coordinator timeline and energy ledger | Explicit one-page jobs; isolated binary only | Same FTL, block manager, TSU and PHY as foreground. Generation CAS, source-block pin, GC pin rejection, finite destination allocation and real foreground-write interleaving pass. PILOT03 and OCP4K each issued and committed 64 actual shared-engine jobs | `METADATA_VERSION_VALIDITY` only; payload equality, die-wide/HBM refresh, ECC/RBER and checkpoint restore unavailable | +| Maintenance age | Committed maintenance completion supplies `age_reset_ns` | Coordinator page-age record and output | Only after one-page mapping commit | Age resets only after commit and covers exactly one page; initial age is input metadata in real seconds | Metadata-only bookkeeping; no retention/ECC/failure-probability model | +| Parametric HBM media | `tools/eq3_basic_hbm.py::BasicHbm` | Coordinator, energy ledger and fabric source-ready | Explicit HBM config | PILOT03 completed 320 HBM requests; OCP4K completed 160 on the common horizon | Not a DRAM backend; service die/plane and HBM maintenance unavailable | +| Two-bank package fabric | `tools/eq3_basic_fabric.py::BasicFabric` | Coordinator final delivery, energy ledger and resource probe | Explicit topology config; default off | Direct/relay/link events, bounded banks and release share the event horizon. The frozen v3 completed-48 snapshot exercises mixed-direct, relay, DASH direct+relay, and 8HBF direct without pending fabric ownership | Latency/bandwidth/energy are scenario assumptions; this is not product-speed alignment | +| Activity-to-energy | Native commands, maintenance IDs, HBM/fabric facts and prescribed GPU source | `ActivityEnergyLedger` | Explicit energy profile | Disjoint scopes are written per half-open window. `_source` now prioritizes actual `maintenance_request_id`, with legacy fallback; fixed tests pass. OCP4K raw has 320 `HBF_MAINTENANCE` rows | PILOT03 predates the repair and retains 260 maintenance/background rows as `BACKEND_BACKGROUND`; window totals and joules are unchanged. Coefficients remain assumptions | +| Energy-to-temperature | `ActivityEnergyLedger.flush()` | Persistent full-2 mm `ThermalService.advance()` | Explicit service and locked hashes | PILOT03 ran 500 windows/10 s and OCP4K 300 windows/6 s on the 64,512-node/275-sensor model | Engineering composition, not a new P2 freeze or product calibration | +| Thermal guard/rate policy | Thermal states plus completed-window resource/backlog facts | `ReadRatePolicy`, byte-token ledger and admission gate | Profile-controlled; module default off | In 16 completed three-policy groups, P0/P1 selected workload outcomes are equal. P2 preserves Stress delivery/backlog but adds 299.773--399.350 ms to p95 in this sparse input | The 48-point snapshot is partial and GPU-dominated; final 18 receipts, retry and UECC behavior remain outstanding | +| Resource observation | Native/energy busy intervals plus `BasicFabric.resource_state()` | Coordinator resource timeline and rate-policy window facts | Injected point-local probe; absent facts remain `None` | Both completed loops retained backend busy fractions and instantaneous bank/link ownership without adding scheduler events | Engineering observation adapter, not a public MQSim scheduler API or calibrated utilization model | +| Event horizon | Arrival, backend/fabric, maintenance and thermal-window times | `ClosedLoopCoordinator` via `run_point.py` | Explicit construction | Existing `until(horizon)`, no host sleep. All 48 frozen completed points drain with zero censoring and conserved fabric resources | Three refresh attempts exposed a late-maintenance scheduling bug and are excluded; fixed rerun receipts remain outstanding | +| Output contract | Coordinator and energy ledger | Point-local CSV/JSON and analysis tools | Per point | One time base; effective and physical bytes are separate. A frozen 48-ID snapshot and compact interpretation now exist | Final 66-point aggregation and the remaining 18 point receipts are not yet available | +| A2 ideal-independent replay | Frozen PILOT03 64-job bundle and `ideal_maintenance_replay.py` | Full coordinator loop with real foreground MQSim and wrapper-owned replay facts | `A2-PILOT03-LOOP01`; actual backend maintenance explicitly disabled | COMPLETED counterfactual: real foreground completed 4,032/4,032; wrapper replayed 64/64 fixed jobs, phases, energy and virtual age facts. Ten requests completed 620 ns earlier; P95 latency, completion count, peak temperature and total energy were unchanged | Not actual maintenance execution: backend issued/committed zero jobs, current mapping was not mutated, and replay mapping/age facts remain `UNKNOWN_REPLAY`. Result is limited to the frozen legacy16K point | +| A3 full-2 mm paired replay | PILOT03 `WINDOW_TOTAL` -> replay converter -> old sparse runner | Streaming reducer and immutable service `thermal.csv` | `A3-PILOT03-REPLAY02` only | PASS: 500x275 sensors; max sensor delta `4.7180e-12 K`, global-range delta `2.6148e-12 K`, cumulative-energy delta `4.4048e-11 J`; 32,320,512 node rows streamed | `FULL_2MM_DISCRETE_EQUIVALENCE` only; same equation/Eigen family is not independent physical reference and does not reclose control | +| A1 source-domain response | Nine source-group runs plus zero-source on unchanged C/G | `A1-DOMAIN-ANALYSIS01` | Point-local replay | COMPLETE: 500 windows, 125,500 rows; entity-mean superposition error `8.6061e-11 K`, zero-source deviation `1.1084e-11 K` | Source ablation, not a disconnected network or nodewise hotspot decomposition | +| Process/build isolation | Default, final isolated, and independent-rebuild services | Fixed process A/B harness | Maintenance off for A/B | Final generation/pin binary passed zero-ns A/B for 8x1 and 4x16. Fresh-path build used byte-identical patched sources and one isolated MQSim archive; its A/B also passed | Maintenance-off A/B does not prove maintenance-on performance/product correctness | +| External GDDR/live GPU | No service producer | Identity/limit reports only | Unavailable | 8HBF may retain physical external-GDDR identity outside package thermal scope | GDDR service/board temperature, token rate and live GPU remain `UNAVAILABLE` | + +## Executed geometry profiles remain separate + +PILOT03 is the historical 16 KiB finite-region result: four HBF stacks, one +channel per stack, 16 MQSim dies per channel and one plane per die. It completed +4,032 requests (3,712 HBF and 320 HBM), 64 maintenance jobs and the full 10 s +thermal loop. Its fixed workspace is not evidence for a 512 GiB stack. + +OCP4K is separate. It instantiated four logical 512 GiB HBF stacks, 16 channels +per stack, one MQSim die per channel, 16 planes per die, 256 pages/block and +2,048 blocks/plane. It completed 7,260 requests (7,100 HBF and 160 HBM) and 64 +maintenance jobs in the 6 s loop. Wall time was 83.05 s; child peak RSS was +21,510,776 KiB (about 20.52 GiB). Full *logical* capacity does not establish +target physical spare, bad-block reserve or overprovisioning. + +The 16 OCP banks use `BANK_AS_MQSIM_PLANE_V1`, a one-to-one engineering +projection. CWDP selects channel/die/plane from logical address bits; MQSim's +page-level FTL separately allocates physical block/page. The physical-boundary +probe observed 256 first touches fill block 0 pages 0..255 and the next enter +block 5 page 0. This is not the complete OCP zone/direct block-address protocol, +and logical block/page ordinals cannot be reported as native physical ones. + +## Four-topology status for the new geometry + +The frozen v3 snapshot contains 48 completed points: 12 for each topology. It +covers all W1 Safe/Near/Stress policy triples and all W2 Stress policy triples. +This is a partial campaign result, not the final 66-point aggregation; the +remaining 18 points require their own completion receipts. Earlier +`basic-four-topology-actual` fixtures remain interface tests and are not mixed +into these counts. + +| Topology | Configuration axis | Thermal axis | System-behavior axis | +|---|---|---|---| +| 8HBF direct + external GDDR | `COMPLETED48_CONFIG_EXECUTED`; 12 points, eight full-logical-capacity HBF stacks and 2,048 observed native channel/die/plane tuples | `COMPLETED48_COUPLED_THERMAL_EXECUTED_CONDITIONAL`; HBF package trajectory completed, while external GDDR remains outside the thermal domain | `COMPLETED48_FUNCTIONAL_PASS`; balanced eight-stack requests, zero censoring/leak. External-GDDR service and temperature remain `UNAVAILABLE` | +| Mixed-direct | `COMPLETED48_CONFIG_EXECUTED`; 12 points, 4HBF + 4 parametric HBM and 1,024 observed HBF tuples | `COMPLETED48_COUPLED_THERMAL_EXECUTED_CONDITIONAL`; full declared package trajectories completed | `COMPLETED48_FUNCTIONAL_PASS`; actual MQSim HBF plus parametric HBM, unique completion and zero final backlog | +| 4+4 relay | `COMPLETED48_CONFIG_EXECUTED`; 12 points with paired HBF-to-HBM package routes | `COMPLETED48_COUPLED_THERMAL_EXECUTED_CONDITIONAL` | `COMPLETED48_FUNCTIONAL_PASS`; actual HBF NAND path plus external relay fabric, zero censoring/leak | +| Four-pair DASH | `COMPLETED48_CONFIG_EXECUTED`; 12 points with both HBF direct and relay paths | `COMPLETED48_COUPLED_THERMAL_EXECUTED_CONDITIONAL` | `COMPLETED48_FUNCTIONAL_PASS`; Stress representative completed 3,552 HBF direct and 3,548 HBF relay requests | + +Every completed point remains `CONDITIONAL_NOT_MODEL_PASS`; configuration, +thermal execution and system behavior are still separate axes. The fixed 200 W +GPU source contributes more than 99.9994% of recorded activity energy, and HBF +active delivery is only 0.758--3.454 MB/s. Geometry alignment therefore does +not establish target-speed or saturation behavior. See +[READ_BANDWIDTH_ALIGNMENT_STATUS.md](READ_BANDWIDTH_ALIGNMENT_STATUS.md). + +## Evidence locations + +- Generation/pin/write race: `points/backend-generation-cpp-test-v1` and + `points/backend-generation-service-test-v2`. +- Final and independent A/B: `points/AB-PROCESS-GENERATION-02` and + `points/AB-PROCESS-INDEPENDENT-REBUILD01`. +- 16 KiB loop: `points/Q1-WEIGHT-MAINT-PILOT03`. +- 4 KiB loop: `points/OCP4K-LOOP-PILOT01`. +- A3: `points/A3-PILOT03-REPLAY02/RUN_RECEIPT.json`. +- A1: `points/A1-DOMAIN-ANALYSIS01/result.json`. +- A2 ideal replay: `points/A2-PILOT03-LOOP01/POSTCHECK.json`; its actual + shared-maintenance baseline is `points/Q1-WEIGHT-MAINT-PILOT03`. +- Source-label repair: `points/energy-fixed-v2/result.json`. +- Frozen partial v3 interpretation: + `campaign-ocp4k-v3/stage/COMPLETED48_INTERPRETATION.md` and + `COMPLETED48_COMPACT.json`. + +All point paths are under +`eq3_thermal/plans/isolated-maintenance-campaign-v1/`. The repaired uppercase +`SLC` startup row and its zero-work receipt remain separate startup evidence. +The completed-48 snapshot promotes only the bounded axes shown above; final +campaign conclusions wait for the remaining 18 receipts and frozen aggregate. + +## Backend isolation and capability evidence for the final report + +The maintenance implementation is confined to +`experiments/eq3_maintenance/backend`: its own patch series, copied MQSim +source, headers, CMake graph, archives and executable. The default MQSim source +and build graph are not modified, and each tested process links one engine +archive. Final maintenance-off process A/B tests match default foreground +lifecycle, native order/placement and completion times at zero-ns tolerance; +an independent-path rebuild reproduced the patched source identity and passed +the same A/B cases. + +Within that isolated engine, maintenance shares the real FTL, block manager, +TSU and PHY with foreground work. Fixed tests cover native read -> destination +program -> generation-CAS commit, source pinning, GC exclusion, finite +allocation, terminal cleanup/failure states and a scheduled foreground write +that makes stale maintenance lose exactly once. PILOT03 and OCP4K then show +64/64 actual shared-engine maintenance completions in complete CPU loops. +Capability remains `METADATA_VERSION_VALIDITY`: there is no payload equality, +ECC/RBER, target physical-spare, HBM refresh or checkpoint claim. + +A2 is a separate counterfactual. Its foreground requests run on real MQSim, but +its 64 maintenance jobs are fixed replay facts on ideal independent resources; +the current engine issues no maintenance and mutates no mapping. The observed +620 ns improvement for ten requests is therefore evidence of localized shared +resource contention in the frozen legacy16K input, not evidence that an +independent maintenance scheduler has been implemented. diff --git a/docs/eq3_thermal/MINIMAL_REPAIR_ENDPOINTS.md b/docs/eq3_thermal/MINIMAL_REPAIR_ENDPOINTS.md new file mode 100644 index 0000000..32598f1 --- /dev/null +++ b/docs/eq3_thermal/MINIMAL_REPAIR_ENDPOINTS.md @@ -0,0 +1,107 @@ +# EQ3-MINIMAL-REPAIR-v1: path endpoint admission + +Scope: `CpuService` engineering fixture only. This repair does not change MQSim, +the public ABI, event order, request/completion lifecycle, resource ownership, +checkpoint version, service durations, or maintenance policy. + +## Evidence and classification + +### Relay bypasses a controlled partner endpoint — `CONFIRMED_BUG` + +- Invariant: every stack control domain actually traversed by a route must allow + a job before any resource is reserved or activity energy begins. `Shutdown` + blocks foreground and maintenance. `Severe` blocks foreground but retains the + existing maintenance exemption. `Light` limits foreground admission and does + not consume quota for maintenance. +- Location before repair: `src/eq3_thermal/cpu_service.cpp`, `route()` correctly + included the paired HBM base/link resources, while `admit()` checked and + updated only `job.stack`. +- Minimal reproducer: relay topology, target `hbf0=Normal`, paired + `hbm0=Shutdown`, idle resources, one `hbf0` relay read. +- Actual pre-repair result: test exited 1 with + `relay bypassed paired HBM Shutdown`. +- Preserved evidence: + `eq3_thermal/plans/minimal-repair-v1/points/ENDPOINT-PRE-REPRO/` contains the + source diff, manifest, stdout/stderr, limits, and result. The run used one CPU, + 12 GiB process / 16 GiB task limits, GPU 0, and the 600 s watchdog. + +The earliest verified cause was admission using the target stack rather than +the already resolved route endpoints. This was not a missing resource lock or +thermal-model error. + +## Minimal repair + +`route()` now derives an internal, insertion-ordered, set-deduplicated +`control_endpoints` list alongside its existing resources. A direct request has +its target stack; a relay has the target HBF and paired HBM; external GDDR has +none. `admit()` checks every listed endpoint before resource reservation. A +successful foreground admission updates `next_admit` for every traversed +endpoint currently in `Light`. + +The resource list, energy map, service interval, completion reporting, and +event-loop sequence remain unchanged. A blocked job remains in the existing +queue, so normal simulated-time advancement continues to process in-flight +completion, cooling, control transitions, and maintenance. + +`report()` additionally derives `admission_blocks` from the current queue. It +reports the request, blocked stack, state/reason, exact Light retry time where +known, pending control-transition time where present, recovery condition, +busy resources, and future arrival boundary. This field is read-only and is not +stored in state or checkpoints; Shutdown/Severe recovery time remains unknown +unless a transition is already pending. + +## Fixed regression contract + +`tests/eq3_thermal/cpu_service_tests.cpp::path_endpoint_admission` covers: + +- paired HBM Shutdown blocks relay with zero reservation and zero HBF/base relay + energy; +- DASH relay is blocked while a direct route that does not traverse the partner + remains legal; +- both endpoints in Light, differing endpoint quota times, and quota update on + the paired HBM endpoint; +- DASH direct and relay share the HBF control endpoint without quota bypass; +- recovery applies before admission at the same timestamp, unique completion, + blocked checkpoint/resume equivalence, and in-flight drain; +- Light and Severe maintenance exemptions remain unchanged; paired Shutdown + blocks maintenance without leaking resources; +- blocker reason and unknown recovery time are visible in `admission_blocks`. + +`control_pending_contract` separately confirms existing behavior: a sample whose +desired state equals the applied state cancels stale pending work; a more severe +sample replaces a pending Light action; sampling and action at the same timestamp +produce one deterministic transition. It does not change the control algorithm. + +The post-repair suite before the last two contract additions passed as +`ENDPOINT-POST2-REGRESSION` (exit 0, 1.38 s). The first post-repair attempt is +retained as a test-fixture diagnosis: its recovery case used `policy=none`, so +automatic recovery was not defined. The corrected case explicitly selects the +existing `hysteresis` policy; no production behavior was changed for that test +failure. + +Final frozen-source validation is `THERMAL-FINAL2-BUILD` (exit 0, 9.14 s, +sampled aggregate RSS 503480 KiB) followed by `THERMAL-FINAL2-TEST` (exit 0, +1.53 s): all four thermal/observer/CpuService/actual-MQSim-CPU suites passed. +The CpuService output explicitly reports the endpoint, pending-control, four +topology, maintenance, checkpoint, drain, cooling, and real Dispatcher CPU +fixture checks as passing. The prior `THERMAL-FINAL-TEST` failure is retained: +its binary was built immediately before a one-line test correction from +`advance_to(30 ms)` to `advance_to(30 ms + 1 ns)`. Existing service semantics +process completions at an exact horizon but do not start a new control action +there; the control log still verifies that the action occurred at exactly +30 ms. The incremental rebuild removed this source/build race; no implementation +change followed the frozen build. + +## Capability boundary + +The current relay/DASH constructor requires four unique HBF-to-HBM pairs. +Multiple distinct HBF stacks sharing one HBM partner are therefore +`NOT_SUPPORTED_BY_CURRENT_TOPOLOGY_INVARIANT`; this repair does not relax that +topology rule. Legal shared-endpoint behavior is tested with DASH direct/relay +paths sharing the HBF upstream/control domain. Supporting a many-to-one pairing +would be a separate topology/arbiter design change and is not implied by this +bug fix. + +All observations here remain `ENGINEERING_FIXTURE`. They do not establish real +relay capability, calibrated control temperatures, physical energy, live MQSim +active gating, or GPU validation. diff --git a/docs/eq3_thermal/MINIMAL_REPAIR_P2.md b/docs/eq3_thermal/MINIMAL_REPAIR_P2.md new file mode 100644 index 0000000..388545a --- /dev/null +++ b/docs/eq3_thermal/MINIMAL_REPAIR_P2.md @@ -0,0 +1,89 @@ +# EQ3-MINIMAL-REPAIR-v1:P2 最小诊断修复 + +状态:局部补丁和固定 CPU 验证已完成。本文不改变 +P2 `BLOCKED_WITH_EVIDENCE`、`MODEL_FREEZE` 未成立及盲测封存状态。 + +## 证据与分类 + +| 项目 | 分类 | 证据与不变量 | 本轮处理 | +|---|---|---|---| +| RC 越 300–400 K 域时只报通用错误 | `CONFIRMED_BUG`(诊断证据缺失) | `tools/eq3_campaign_rc_runner.cpp` 原路径在 trial solve 后直接 `require`;不变量是失败仍失败,同时必须区分最后合法状态和失败 trial | 写独立 last-valid/trial CSV 和失败 JSON;仍返回非零,不 clamp、不改步长/输入/阈值 | +| 2 mm development 超过 400 K | `DOMAIN_FAILURE` | 独立参考最高 400.911 K;旧 RC 仅保留最后完整 31.1 s/399.975501 K,失败 trial 数值原为 `UNKNOWN` | 不修改物理;新诊断只供未来受影响的小型/正式运行使用,不回填旧 raw | +| 4→2 mm、2→1 mm 空间差 | `NUMERICAL_ACCURACY_LIMIT` | 完整 4→2 mm 最大差 3.253 K;共同前 4 s 结果见下节 | 保留旧 0.25 K 标准和失败状态;共同窗口仅 `DIAGNOSTIC_ONLY` | +| 1 mm 完整 100 s | `RESOURCE_BLOCKED` | 已有 4 s 运行 328.012 s,完整运行估计 2000–2600 s,超过 600 s | 本轮不重启、不切段;长窗口/后端变更仍待确认 | +| 比较器固定初态 300 K、探针 301/330 K | `CONFIRMED_BUG`(通用分析接口) | 实际代码硬编码,导致其他配置被错误解释;历史数据确为 300/301/330 K | 增加可选 `initial_k`/`probes_k`,CLI 为 `--initial-k`/重复 `--probe-k`;默认值保持历史等价 | + +以上 P2 状态和数值来自已登记 campaign raw/result。新增的共同窗口归约已由 +统一安全入口只读执行,结果为 `DIAGNOSTIC_ONLY`;它不增加物理校准证据。 + +## 失败快照合同 + +`eq3_campaign_rc_runner` 在有限但越域的 trial 上写: + +- `rc_failure_last_valid.csv`:最后成功积分状态; +- `rc_failure_trial.csv`:实际算出的失败 trial 状态; +- `rc_failure_diagnostic.json`:准确 target time、步号、输入 slot 区间、域版本、 + 节点/组/die、越域温度、已完成区间能量以及退出原因。 + +坐标不在当前 model text 公共类型中,明确记录 +`UNKNOWN_NOT_EXPOSED_BY_MODEL_TEXT_API`,不从 ID 推断。模型、events 和 runner +SHA-256 可由现有 launcher 通过新增可选参数传入;未传入时记录 +`UNKNOWN_NOT_SUPPLIED`,不伪造哈希。失败 trial 步的能量积分均为 +`UNKNOWN_NOT_INTEGRATED`,已完成步的输入、储能、边界损失和残差单列。 + +若稀疏 solve 本身失败,trial 向量不可信:只保存最后合法状态,状态为 +`NUMERICAL_FAILURE`,trial 明确 `UNKNOWN_SOLVER_FAILURE`。输入解析阶段的 NaN/Inf +属于 `INPUT_FAILURE` 边界,没有已计算 trial,因此不会伪造状态文件。所有诊断 +文件拒绝覆盖;写盘失败仍向调用者返回失败。 + +成功路径仍先验证 trial,再提交为当前状态及累计能量。该提交顺序只让失败证据 +可见;矩阵、方程、正常步数值、采样点和能量积分未改变。 + +## 方程、单位和缓存静态审核 + +runner 的矩阵仍为:对角 `C/dt + G_boundary + sum(G_edge)`,非对角 +`-G_edge`;theta 右端为 `C/dt*theta + P + G_boundary*(T_boundary-origin)`。 +这与 `C dT/dt = P + sum G(T_neighbor-T) + G_boundary(T_boundary-T)` 的后向欧拉 +离散一致。解析器继续要求 `C>0 J/K`、`G_edge>0 W/K`、边界导热非负、功率有限 +非负;时间单位为秒,温度为 K。 + +runner 对固定 `dt` 只构造/分解一次矩阵。通用 `ThermalModel` 的 factor cache +以 `dt` 为 key;`reset()` 和 `restore()` 均清空 cache,配置在对象生命周期内 +不可变,因此当前没有发现 stale-factor 路径。既有固定测试覆盖单节点解析解、 +步长收敛、内部导热守恒、非 300 K 温度平移、稀疏/稠密同方程及能量一致性; +本轮补充三节点顺序置换等价测试。该审核没有改热方程、物性或公共 ABI。 + +## 既有 raw 的同窗口比较 + +输入均为原 train,时间戳 0.1–4.0 s,共 40 帧、275 个相同 sensor;比较保持 +初态 300 K、探针 301/330 K 和旧 0.25 K reference 相邻网格阈值。 + +| 网格对 | 共同窗口最大绝对差 | 最坏 MAE | 状态 | +|---|---:|---:|---| +| 4 mm R01 → 2 mm R02 | 2.710 K | 1.153575 K | `NUMERICAL_FAIL / DIAGNOSTIC_ONLY` | +| 2 mm R02 → 1 mm R03 prefix | 1.277 K | 0.546275 K(最坏 sensor) | `NUMERICAL_FAIL / DIAGNOSTIC_ONLY` | + +最大差均出现在 HBF 顶层 die/stack hotspot 一组观察量。差值随网格细化减小, +但两级都未达到 0.25 K,且前 4 s 只有 GPU 激励,不覆盖后续 memory/base 激励; +因此不能由此计算或宣称完整 100 s 收敛阶,也不能替代正式参考验收。 + +## 固定验证范围 + +统一串行入口实际执行了:原 v2 binary 的微小越域复现(按预期在原 domain +错误后因不存在 `rc_failure_diagnostic.json` 而失败)、新 runner 编译、 +runner/compare 两组固定测试。最终共 15 项:13 项通过,2 项依赖 checkout 外生成 +candidate 的可选测试跳过;跳过项不计作通过。首次测试曾因 NaN 输入的预期错误 +文本与现有 parser 的实际 `malformed node record` 不同而失败,修正测试契约后重跑 +通过,未修改 parser。runner 测试覆盖有限越域快照、 +非有限输入拒绝、证据拒绝覆盖、稀疏/稠密一致、能量守恒、温度平移、节点重排。 +这些都是软件验证,不是新数值配置、昂贵参考或物理校准。 + +原始证据位于 `eq3_thermal/plans/minimal-repair-v1/points/`: +`P2-ORIGINAL-DOMAIN-REPRO`、`P2-RC-DIAG-BUILD2`、 +`P2-RC-DIAG-TEST`(保留的首轮测试失败)、`P2-RC-DIAG-TEST2`(最终 PASS)和 +`P2-COMMON-FIRST4S`。共同窗口完整派生结果为 +`eq3_thermal/plans/minimal-repair-v1/P2-COMMON-FIRST4S-DIAGNOSTIC.json`。 + +未执行任何完整 1 mm solve、GPU 工作、盲测、阈值变更、物性变更或公共 ABI/ +checkpoint 变更。P2 未冻结原因仍是参考空间精度不合格、1 mm 完整参考资源受阻, +以及 development 轨迹越出已声明常物性域。 diff --git a/docs/eq3_thermal/MINIMAL_REPAIR_REPORT.md b/docs/eq3_thermal/MINIMAL_REPAIR_REPORT.md new file mode 100644 index 0000000..536ffe4 --- /dev/null +++ b/docs/eq3_thermal/MINIMAL_REPAIR_REPORT.md @@ -0,0 +1,146 @@ +# EQ3-MINIMAL-REPAIR-v1 + +Local repair and CPU interface work, based on integrated main `a93c0c1ae2531b4224c00d6917c64c0358878a9a`. +Entry branch `integrate/eq3-main-20260920`, clean. The governance workspace root +is not a Git repository; actual source is `eq3_thermal/integration/main-20260920`. +Immutable baseline `5eb789d5f1a42f0c040ee6fb5a2cdb5ffa0951d5` remains clean. +No reset, push, merge, dependency installation, GPU or research matrix. + +USER_CONFIRMED: current request adopts attached taskbook and explicitly permits +minimal local fixes, optional default-off composition interfaces and CPU checks. +Latest policy added to both effective AGENTS.md. DOC_DERIVED evidence below; +physical inference is not substituted for unavailable measurements. + +## Issue classification and minimal scope + +| Item / source | Invariant and reproduction | Classification / disposition | +|---|---|---| +| `CpuService::Impl::{route,admit}` | HBF Normal + paired HBM Shutdown + free relay resources; original binary starts it, new fixed regression fails with `relay bypassed paired HBM Shutdown` | CONFIRMED_BUG; joint checks on deduplicated actual route domains before any reservation; all successful foreground Light quotas updated | +| `CpuService::Impl::report` | Rejected work needs actionable endpoint/reason without fabricating completion | Read-only `admission_blocks` derived from current queue/control/resources; no checkpoint layout change | +| `control_sample` pending | Existing `desired==applied` already clears pending | NOT_REPRODUCED / ALREADY_FIXED; do not reimplement | +| Escalation dwell | Existing action delay and minimum dwell apply to all transitions; no contrary authoritative emergency contract found | DESIGN_LIMITATION / optional policy decision; retained unchanged | +| CpuService fixed duration/power/base ownership | Fixture explicitly defines whole-request duration and serial base occupancy | DESIGN_LIMITATION; no new NAND simulator, no removed base lock or additive latency rewrite | +| Real MQSim observation | Existing API gives request admission and raw/report completion, not NAND command start/location | DESIGN_LIMITATION; keep unknown die/plane/physical/link bytes, occupancy energy proxy explicit | +| Real MQSim demand gate | Existing horizon API can progress inflight work and cooling before submit | Optional thin nonblocking wrapper; default off, external wait separate, original engine and completion unchanged | +| Real MQSim die maintenance | Backend has no corresponding operation/commit interface | DESIGN_LIMITATION / UNSUPPORTED_CAPABILITY; no fake host writes or duplicate resource ledger | +| Sparse RC failure evidence | Tiny valid-domain initial state leaves domain at first step; frozen v2 reports error but omits last/trial snapshot | CONFIRMED_BUG in diagnostics; local evidence patch, failure remains failure | +| Reference accuracy / low-rise normalized MAE | Historical spatial differences fail0.25K;5% at1K floor is0.05K | NUMERICAL_ACCURACY_LIMIT; old standards retained; common-window diagnostics separate | +| Development400.911K | Independent reference exceeds declared300–400K domain | DOMAIN_FAILURE; preserved, no clamp or changed input | +| Full1mm100s | Pilot-based projected2000–2600s exceeds600s | RESOURCE_BLOCKED; not rerun or split | +| Sparse reuse/order/ownership | Existing sparse solver factors once at immutable config/fixed dt; reference MMD and ownership fixes already present | ALREADY_FIXED / no rewrite | + +Source-level equation review: dense `ThermalModel::Impl::factor` implements +C/dt + boundaryG + edge-Laplacian; RHS uses prior thermal state, input W and +boundary W/K×K. Config is immutable per instance and factor cache is keyed by dt; +new config/reset creates a new instance. Sparse runner factors fixed matrix once, +using theta=T−initial-origin and transforming boundary temperature consistently. +No equation or material changes are authorized or made. + +## Reuse, workload feasibility and old results + +The six previous points are all `mixed_direct` from `tools/eq3_cpu_fixture.py`. +The patch preserves its single controlled endpoint, resource mapping and quotas; +relay-specific additional domains cannot affect that path. Historical Safe/Near/ +Stress remain valid *engineering* records and are reused, not relabeled as reruns. +Near had no cooling benefit; Stress completed448/1000, leaving552 foreground and +56 maintenance queued. No gain is manufactured by changing inputs. + +The same generator supplies a request each40ms/stack (25/s),20ms read occupancy: +foreground nominal base utilization0.5. HBF16die maintenance60ms every2s adds0.48; +HBM12die×5ms/1s adds0.06. Light80ms permits at most12.5 foreground starts/s +before competing maintenance. These DOC_DERIVED screening numbers explain why +lower temperature can come with backlog; they are not new measurements. + +Small fixed2die topology tests do not replace12/16die research geometry. The new +one-node MQSim cooling/input example is ENGINEERING_FIXTURE_REQUEST_OCCUPANCY, +not measured NAND activity, physical calibration or a production scheduler. + +## Evidence and reproduction + +Private immutable point receipts, commands, prelaunch source patch/new-source +snapshots and raw stdout/stderr are at workspace +`eq3_thermal/plans/minimal-repair-v1/points/`. The adapted existing safety runner +is `run_check.py`, with scope and environment records alongside. Existing core +archives from integrated main are bound once in `build_inputs.json`; fresh EQ3 +source is rebuilt in a different build directory. No repeated large raw hashing. +Current source and manifest identities are for this task, not invented historical +observations. Individual point failures are retained even when a test fixture +needs a documented correction. + +Additional detailed results and final commit IDs are appended after validation. +See [interface consumers](INTERFACE_CONTRACT_AND_CONSUMERS.md), +[P2 diagnostics](MINIMAL_REPAIR_P2.md), and +[required decisions](USER_DECISIONS_REQUIRED.md). + +## Four topologies: final acceptance boundaries + +| Topology | Configuration | Thermal research model | System/backend behavior | +|---|---|---|---| +| mixed-direct | Existing generic4+4 and2+6 fixtures;12/16die mainline inputs retained | P2 not frozen; small fixture thermal equations only | CpuService CPU regression PASS; old6points unaffected/reused; actual MQSim request-level gate tested independently, not topology-qualified | +|4+4 relay | Four unique pairs unchanged | Relay/base physical power unqualified | CpuService joint-endpoint gating/Light/resource/energy PASS; actual MQSim relay topology not established | +|four-pair DASH | Direct/relay routes preserved, shared HBF upstream | Research thermal validation incomplete | CpuService legal direct isolation and shared endpoint quota PASS; no new auto-routing; actual MQSim DASH not established | +|8HBF + external physical GDDR | Existing eight-HBF fixture/explicit external GDDR retained | Package excludes GDDR, temperature UNAVAILABLE; research geometry incomplete | CpuService service/energy accounting PASS, nonzero external energy separate; no real GDDR or GPU validation | + +Many-HBF-to-one-HBM pairing is rejected by the existing unique-pair invariant. +Tests use legal DASH shared HBF endpoints and HBM/relay shared resources; no +unapproved topology change is hidden in the repair. CPU_TEST_CONNECTED for the +new MQSim gate does not mean PRODUCTION_CONNECTED or real NAND maintenance. + +## Local validation failures retained + +`ENDPOINT-PRE-REPRO`: required pre-fix failure. `P2-ORIGINAL-DOMAIN-REPRO`: +required missing-snapshot reproduction using frozenv2 binary. +`ENDPOINT-POST-REGRESSION`: new recovery fixture accidentally used policy=none; +corrected fixture to explicit hysteresis, kept production control unchanged. +`THERMAL-FINAL-TEST`: pending-boundary test fix landed after build; incremental +rebuild bound it. MQSim added test initially consumed ordinary next-completion +instead of horizon readiness, so observer had not reached the reported deadline; +corrected caller to existing horizon API, not engine latency. +`THERMAL-FINAL2-TEST`: all4 suites PASS, including actual MQSim off/read_only/ +shadow parity, disabled gate parity, defer/inflight handling, actual thermal +cooling→advice→gate→backend submission→unique completion, and all four CpuService +topologies. No policy or physical input was optimized to make these pass. + +## Final verification / limits / rollback + +- `THERMAL-FINAL2-TEST`:4/4 C++ suites PASS. +- `CORE-RELEVANT-REGRESSION`:10/10 existing MQSim/horizon/service/trace timing + and protocol tests PASS. No full-main-suite or GPU pass is inferred. +- `FOUR-TOPOLOGY-CONFIG-REGRESSION`:59/59 configuration, geometry export, source + mapping and fixed workload checks PASS. No research geometry solve. +- `P2-RC-DIAG-TEST2`:15 total,13 PASS,2 optional generated-candidate checks + skipped (paths absent in integrated checkout); not15 passed plus2 skipped. + Independent2node hand solution, node reorder, non300K equilibrium, dense + comparison, finite domain failure, nonfinite trial, input parse rejection, + write-path refusal and completed-energy1.5J vs unknown failed-step energy pass. +- `P2-COMMON-FIRST4S`: read-only reduction PASS; numerical qualification still + FAIL (2.710K and1.277K against0.25K); window cannot validate full excitation. + +P2-RC-DIAG-TEST retained one test expectation failure: `nan` is rejected by the +existing stream parser as `malformed node record`, before a finite-value check. +Only the assertion was corrected to that actual contract. No parser change. + +All steps serial CPU1,OMP/BLAS1,GPU0/cloud0; build parallel1. Largest measured +child RSS2,203,356KiB (~2.10GiB), sampled aggregate peak <=0.51GiB (sampling +is a lower bound, not instantaneous exact aggregate). All invocations <600s, +new task package <4GiB and cumulative retained ~17.28GBdecimal (~16.09GiB), +below20GiB. No baseline/raw deletion. Existing main two blocked groups +(context_lifecycle12GiB allocation, vmem_tuning externalCSV) remain inherited +limitations; no irrelevant rerun or repaired-status claim. + +Each point has prelaunch command/source status/diff and resource metadata. +Existing host archive identity bound once at setup; final executable fingerprints +are explicitly POST_VALIDATION_HANDOFF in `VALIDATION_SUMMARY.json`, not claimed +as observations captured before the historical runs. Fresh source was built +in the new task directory; cross-host recreation was not performed. + +Local branch `repair/eq3-minimal-v1` starts at integratedmaina93c0c1; commits: +`dfe12d5` rules; `aba746e` endpoint repair; `f2e4ace` optional MQSim gate; +`7153ae6` RC diagnostics and analysis. Documentation-only final commit follows. +Rollback is opt out of the gate (default Off), or revert the relevant local +commit; do not reset baseline/main history. Never reinterpret old relay +fixtures as repaired results; use this task's new evidence. + +The feasible nonstructural repair work is complete. Production route-aware +thermal control, real die maintenance, calibrated energy and P2 qualification +remain open for the reasons and specific options in USER_DECISIONS_REQUIRED.md. diff --git a/docs/eq3_thermal/MQSIM_DIE_MAINTENANCE_NARROW_DESIGN.md b/docs/eq3_thermal/MQSIM_DIE_MAINTENANCE_NARROW_DESIGN.md new file mode 100644 index 0000000..dd977f4 --- /dev/null +++ b/docs/eq3_thermal/MQSIM_DIE_MAINTENANCE_NARROW_DESIGN.md @@ -0,0 +1,173 @@ +# MQSim die-maintenance narrow design + +Current clarification (2026-09-20): the user subsequently approved a narrow +**isolated experimental copy** in EQ3-ISOLATED-MAINTENANCE-CAMPAIGN-v1. +That implementation and actual evidence are in `ISOLATED_INTERFACE_STATUS.md` +and `experiments/eq3_maintenance/backend/docs/IMPLEMENTATION_BOUNDARY.md`. +The default production backend remains unchanged and unsupported. The design +below is preserved as the pre-authorization record; it is not a current block +on the already authorized isolated implementation. + +Historical status: `AWAITING_EXPLICIT_REFACTOR_APPROVAL` + +This is a design record, not an implementation approval. The current +`MqsimOnlineEngine` capability remains `UNSUPPORTED_CAPABILITY` for die-level +maintenance. A host write is not maintenance, and no wrapper-side resource +ledger is proposed. + +## Evidence and required invariants + +The reusable MQSim machinery is inside the backend. `FTL` owns the +`Address_Mapping_Unit`, `Flash_Block_Manager`, `GC_and_WL_Unit`, `TSU`, and +`PHY`. `GC_and_WL_Unit_{Base,Page_Level}` already creates `GC_WL` read, +program, and erase transactions; `Address_Mapping_Unit_Base` defines physical +block and LPA barriers; `Flash_Block_Manager_Base` owns destination allocation +and the `GC_WL_started`/`GC_WL_finished` bookkeeping; `TSU_Base` submits these +transactions to the same arbitration path as foreground traffic. These are +`DOC_DERIVED` facts from the vendored MQSim source. Whether the resulting +operation is a scientifically valid HBF refresh model is still `INFERRED` and +would need a separately sourced operation definition. + +The current page-level GC path is not already a failure-atomic maintenance +API. `Address_Mapping_Unit_Page_Level::allocate_page_in_plane_for_user_write` +invalidates the old page and calls `Update_mapping_info` when the destination +is allocated, before the program command completes. The serviced-write path +then removes the LPA barrier, and MQSim's normal PHY path does not expose a +program/erase failure result. Therefore exact commit/fail semantics cannot be +added only at the online wrapper. They require the mapping and GC/WL lifecycle +changes listed below and explicit approval. + +Any accepted implementation must preserve these invariants: + +1. Exactly one terminal completion is returned for every accepted maintenance + request, after every child transaction reaches a terminal state. +2. The existing mapping remains valid until the replacement program succeeds. + No source page or block is erased before all required relocation commits. +3. Destination pages come only from `Flash_Block_Manager`; foreground and + maintenance commands share `TSU` and `NVM_PHY_ONFI_NVDDR2` arbitration. +4. P/E accounting advances only for program/erase commands that actually + complete. HBF age changes only after the maintenance commit succeeds. +5. Failed and partially completed work keeps already incurred activity and + energy facts, releases or reconciles backend resources through their owner, + and never fabricates an uncomputed command phase. +6. Shutdown/checkpoint handling either drains accepted work or reports a + precise terminal failure once; it never silently drops a cohort. + +## Proposed internal interface + +The smallest structural API is an internal maintenance endpoint beside the +patched HBF demand endpoint. It must not enter `Input_Stream_Manager_HBF` or +construct a `User_Request`. + +```cpp +enum class HbfMaintenanceKind { RelocateAndRefresh }; +enum class HbfMaintenanceStatus { + Committed, RejectedUnsupported, RejectedInvalidTarget, + FailedBeforeCommit, FailedAfterCommitNeedsReconcile +}; + +struct HbfMaintenanceRequest { + uint64_t request_id; + uint64_t cohort_id; + HbfMaintenanceKind kind; + flash_channel_ID_type channel; + flash_chip_ID_type chip; + flash_die_ID_type die; + std::optional plane; + uint64_t policy_version; +}; + +struct HbfMaintenanceCompletion { + uint64_t request_id; + uint64_t cohort_id; + HbfMaintenanceStatus status; + sim_time_type enqueue_time; + sim_time_type start_time; + sim_time_type end_time; + std::vector transaction_ids; +}; +``` + +Proposed entry points are +`MqsimOnlineEngine::submit_die_maintenance(...)` and a new internal +`GC_and_WL_Unit_Base::Submit_hbf_maintenance(...)`. The engine would obtain +the existing `FTL` from `SSD_Device::Firmware`, validate the exact physical +target against configured MQSim bounds, and pass the request to the GC/WL +unit. It would retain only completion correlation, as it does for demand I/O; +the backend would own candidate selection, transactions, barriers, and +resources. The returned transaction and command observations would carry +`request_id`, `cohort_id`, and their existing unique transaction/command IDs. +Background work unrelated to an explicit request would keep a null maintenance +parent rather than borrowing a demand identity. + +This endpoint changes the engine API and backend maintenance state machine. +That is why it is outside the current D5 implementation authorization. + +## Lifecycle and failure model + +On acceptance, the GC/WL unit validates the target die and chooses eligible +source blocks using its existing safety predicates. Each chosen block receives +the existing physical-block barrier and `GC_WL_started` bookkeeping. For each +valid source page, the backend reads the current mapping and content, sets the +existing LPA/MVPN barrier, obtains a destination through +`Allocate_new_page_for_gc`/`Flash_Block_Manager`, and submits the related +read/program transactions through `TSU_Base::{Prepare_for_transaction_submit, +Submit_transaction,Schedule}`. Copyback may be used only when the existing +backend declares it legal for the source and destination. + +The proposed mapping path must switch to the programmed destination only on +successful program completion. Then the old page can be invalidated and its barrier removed. An +erase is submitted only after all valid pages in its source block have +committed. Successful erase completion returns the block to the pool and calls +`GC_WL_finished`; the maintenance request commits only after all selected +blocks finish. Empty targets must return an explicit successful no-op or +rejection according to the approved operation contract, not synthesize NAND +activity. + +Before mapping commit, failure keeps the old mapping valid, removes barriers, +and reconciles any allocated but uncommitted destination through +`Flash_Block_Manager`. After mapping commit, failure must keep the new mapping +and enter a backend-owned reconciliation path; it must not blindly restore a +possibly invalidated source. A failed erase cannot be reported as a committed +refresh. The completion identifies the failed child and preserves prior phase +events. MQSim currently assumes successful flash commands in several paths, so +the exact injection and propagation of command failure is a required design +audit before implementation. + +## Exact proposed change surface + +All MQSim edits would be delivered as a new reproducible patch after the +observer patch, never as a dirty `third_party/mqsim` tree. + +| File/symbol | Proposed change | Behavioral impact | +|---|---|---| +| `include/hbfsim/mqsim_online.hpp`, `src/mqsim_adapter/mqsim_online.cpp` | Add maintenance request/completion/capability types and `submit_die_maintenance`; correlate one terminal callback | Public HBFSim API and request lifecycle change | +| MQSim `src/exec/SSD_Device.{h,cpp}` or a narrowly scoped internal accessor | Give the online adapter typed access to the existing `FTL`; do not create a second FTL/resource owner | Module boundary change | +| MQSim `src/ssd/GC_and_WL_Unit_Base.{h,cpp}` and `GC_and_WL_Unit_Page_Level.{h,cpp}` | Add explicit target validation, cohort state, candidate selection, child accounting, commit/fail callback | Backend maintenance lifecycle and scheduling-visible traffic | +| MQSim `src/ssd/NVM_Transaction.h` | Carry optional maintenance request/cohort identity alongside existing source type | Internal transaction layout change | +| MQSim `src/ssd/Address_Mapping_Unit_{Base,Page_Level}.{h,cpp}` | Expose only the commit/reconcile operations missing from the existing GC path after audit | Mapping ownership and failure semantics change | +| MQSim `src/ssd/Flash_Block_Manager_Base.{h,cpp}` and `Flash_Block_Manager.{h,cpp}` | Expose backend-owned reservation reconciliation if existing GC cleanup is insufficient | Resource ownership change | +| MQSim `src/ssd/NVM_PHY_ONFI_NVDDR2.{h,cpp}` | Extend immutable observations with optional maintenance/cohort parents; do not add events | Observation schema only | +| `cmake/MQSimPatchedBuild.cmake`, `patches/mqsim/0004-...patch` | Register the reproducible patch | Build input change | + +If audit shows the existing GC/WL path cannot keep the old mapping valid until +program completion, the implementation must stop and return with a revised, +larger proposal. It must not imitate completion in `CpuService`. + +## Required validation before acceptance + +The minimum fixed CPU suite must cover an empty target, one-page and multi-page +cohorts, exact die targeting, foreground contention through the same TSU/PHY, +copyback-enabled and disabled paths, failure before and after mapping commit, +program and erase failure, unique completion, no leaked barriers or blocks, +P/E increments only on actual completion, and age update only on commit. It +must also cover shutdown drain, checkpoint/restart identity and cohort state, +observer on/off timing parity, deterministic same-time command ordering, and +demand readback of every relocated LPA. Multi-channel tests cannot be promoted +to multi-stack tests until an explicit channel/chip/die-to-package-stack map is +configured and consumed. + +Rollback is removal of the new patch and engine API, followed by rebuilding +from the unchanged vendored source plus patches 0001--0003. Existing demand +and read-only observation behavior must remain byte-for-byte compatible in +default-off mode. diff --git a/docs/eq3_thermal/MQSIM_HBF_STACK_MAP_MINIMAL_DESIGN.md b/docs/eq3_thermal/MQSIM_HBF_STACK_MAP_MINIMAL_DESIGN.md new file mode 100644 index 0000000..e2ad7cf --- /dev/null +++ b/docs/eq3_thermal/MQSIM_HBF_STACK_MAP_MINIMAL_DESIGN.md @@ -0,0 +1,111 @@ +# Minimal MQSim HBF stack-map design + +Status: `USER_AUTHORIZED_BASIC_CPU_CONSUMER`; implementation is limited to the +additive CPU service described here. It does not authorize production daemon +routing semantics, public ABI changes, an HBM backend, relay/base arbitration, +SRAM modeling, or maintenance. + +## Capability boundary + +`MqsimOnlineEngine` constructs one MQSim SSD device. The profile supplies the +physical channel count, one chip per channel, dies per chip, and planes per +die. With the required `CWDP` page allocation scheme, MQSim's existing read and +write mapping paths both select `channel = LPA % channel_count`; block and page +allocation, GC, scheduling, command timing, and completion remain owned by +MQSim. + +The minimal model treats configured, disjoint groups of those real MQSim +channels as HBF stack partitions. It is reported as +`ACTUAL_MQSIM_CHANNEL_PARTITIONED_HBF_STACKS`. It is not a claim that an MQSim +channel is inherently a package stack, or that MQSim models an HBF base die, +relay link, HBM, GDDR, or package topology. + +## Configuration and bijection + +The optional JSON configuration is disabled when `--stack-map` is absent. It +must declare: + +- schema version 1, physical kind `HBF`, route `direct`, and address layout + `GLOBAL_PAGE_STRIPE_V1`; +- the same page size, channel count, dies per channel, and allocation scheme as + the loaded MQSim profile; +- two or more uniquely named stacks, each with the same nonzero number `K` of + unique channel IDs; +- channel groups that are disjoint and exactly cover all `C` profile channels; +- `declared_dies` for each stack equal to `K * dies_per_channel`. + +For external global page `p`, stack count `S`, and `C = S*K`: + +```text +s = p mod S +q = floor(p / S) +k = q mod K +r = floor(q / K) +c = configured_channel_group[s][k] +backend_page = r*C + c +``` + +This is a full-capacity bijection. The inverse uses the explicit channel table: +find `(s,k)` for actual channel `c`, compute `r=floor(backend_page/C)`, then +`q=r*K+k` and `p=q*S+s`. Channel-to-stack identity therefore comes from the +versioned configuration rather than an address guess. + +The stripe is persistent placement across same-kind HBF stacks. It never +temporarily assigns an existing page to whichever stack is currently idle and +does not place HBM and HBF pages through one round-robin namespace. Mixed +topology HBM/KV placement remains `UNSUPPORTED_CAPABILITY`; it must come from a +real HBM backend and an explicit persistent data-placement contract. + +Enabled service mode accepts only one profile page per request and requires +explicit `stack`, `stack_local_page`, and `route=direct` metadata. It derives +the persistent external global page as `stack_local_page*S + stack_index` and +then applies the bijection above. If a caller also supplies `logical_address`, +it must exactly match that derived page. The requested stack is validation and +does not dynamically override placement. Multi-page requests return unsupported at the consumer boundary +instead of being split and given a new completion lifecycle. All mapped and +unmapped requests must not share one engine instance. When `--stack-map` is +absent, the service neither parses nor requires stack/route metadata and sends +the original address unchanged. + +## Files, symbols, and behavior + +| File / symbol | Change | Behavior | +|---|---|---| +| `include/hbfsim/eq3_thermal/mqsim_stack_map.hpp` / `MqsimStackMapAdapter` | Header-only validation and address transform | Default Off returns the original request; Enabled returns original/backend page, resolved stack, and expected channel without submitting or owning work | +| `benchmarks/replay/hbf_mqsim_service.cpp` | Optional `--stack-map`, request mapping before the existing gate, submission ledger, native channel verification | Uses the existing engine, gate, horizon, and completion path; no retry, buffering, new event, or resource reservation | +| `tests/integration/test_mqsim_service.py` | Small eight-stack engineering profile and raw assertions | Proves all eight configured stacks receive real MQSim commands on their configured channels, with unique completion and default-Off parity retained | +| `docs/eq3_thermal/INTERFACE_CONTRACT_AND_CONSUMERS.md` | Capability/consumer entry after validation | Separates channel-partition evidence from package topology claims | + +The native command consumer records external page, backend page, expected +stack, configured stack recovered from the actual physical channel, and +channel/die/plane. A demand command whose actual channel is outside the +request's expected group is a hard test/service failure. Background commands +with no external request parent retain null request/expected-stack fields and +are annotated only from their actual channel. + +## Non-interference, validation, and rollback + +Off mode does not construct a map and submits the original address. Enabled +mode copies the request and changes only `logical_address`; request ID, +arrival, byte count, operation, generation, gate decision, and backend +completion stay unchanged. MQSim still owns FTL mapping, physical allocation, +GC, TSU/PHY arbitration, and command timing. + +Fixed validation must cover malformed/overlapping/incomplete groups, profile +mismatch, requested-stack mismatch, multi-page rejection, all eight stacks, +native actual-channel verification, unique completions, and existing gate and +observer Off/On regressions. The small eight-stack profile uses one channel and +one die per stack and is labeled `ENGINEERING_FIXTURE`; it is not evidence for +the research 8/16-die package geometry. + +Rollback removes the optional service argument, the header, and the added +integration test. No engine, MQSim patch, FTL, TSU, dispatcher, protocol ABI, +or completion implementation is changed. + +## Explicitly unavailable + +The current `HbmCache` manages VMM frame residency and eviction; it is not an +HBM command, timing, refresh, energy, or per-stack backend. Relay/DASH two-hop +bus occupancy and base SRAM dual-bank arbitration are also absent from MQSim. +Those capabilities remain `UNAVAILABLE` and must not be represented by this +address adapter or by fixed-duration `CpuService` resources. diff --git a/docs/eq3_thermal/NATIVE_PROXY_COMPARISON.md b/docs/eq3_thermal/NATIVE_PROXY_COMPARISON.md new file mode 100644 index 0000000..a604b74 --- /dev/null +++ b/docs/eq3_thermal/NATIVE_PROXY_COMPARISON.md @@ -0,0 +1,125 @@ +# Native MQSim and aggregate topology-service semantic comparison + +Date: 2026-09-20. This is a bounded evidence comparison. It did not start a +thermal solve, rebuild MQSim, or repeat an unchanged native campaign. Existing +native receipts remain immutable. The new service is a default-disconnected +engineering model and is not relabelled as native execution. + +## Result + +The two paths answer different questions. + +- The isolated native backend demonstrates real MQSim command lifecycles, + physical placement, shared FTL/TSU/PHY ownership, out-of-place one-page + maintenance commit, and actual foreground/maintenance interference. It does + not demonstrate target-HBF TB/s, payload equality, die-wide refresh, ECC, or + calibrated product energy. +- `TopologyService` demonstrates deterministic aggregate byte flow through + explicit per-channel media, finite two-bank turnover, direct/relay/DASH + routes, endpoint gates, and shared HBM-link capacity. Its completion and + latency are window-quantized engineering estimates. It has no MQSim command, + FTL mapping, NAND payload, or native program/erase result. +- The 60-point base system/thermal matrix presently sends foreground read-rate + demand only. Every window says + `NO_MAINTENANCE_DEMAND_IN_BASE_RATE_WORKLOAD`; therefore zero maintenance + queues in that matrix mean **not exercised**, not free or infinitely fast + maintenance. + +## Operation-by-operation contract + +| Operation or property | Isolated native backend | Aggregate topology service | Safe comparison | +| --- | --- | --- | --- | +| Foreground read | A request becomes actual MQSim child transactions and emits command-issued, media begin/end, and data-out phases with native transaction IDs and PPA fields. The OCP4K pilot observed 7,100 `USERIO` read children. | A byte cohort consumes configured channel media work and route resources. `media_read` and link activity are model facts; effective delivery completes at an exact integer time under the aggregate pipeline approximation. | Compare conservation, route identity, queue pressure, and qualitative contention. Do not compare either service rate or latency as if both were native. | +| Foreground program | The standalone startup caller submitted 16,384 real 4 KiB writes, drained them, then completed 16,384 same-page native reads. The rolling-QD256 receipt records 81,920 native command events and all loaded pages covered, including the last page. | `program` is an explicit external job whose media work is scaled by a scenario ratio. It emits `media_program`, but has no FTL allocation, mapping generation, data-program success, or readable destination. | The proxy can budget heat/resource pressure from a declared program coefficient. Only native evidence can claim MQSim program and mapping behavior. | +| Page maintenance / “refresh” | `maintain` performs a native source read, destination program, generation-CAS mapping commit, old-page retirement, and optional safe reclaim. Age reset is returned only after commit. This is one-page out-of-place metadata/version maintenance, not die-wide product refresh. | There is no combined refresh transaction or mapping commit. A caller may submit explicit `refresh_read`, `program`, `erase`, or migration jobs with a maintenance ID. Completion means aggregate job bytes finished; it cannot reset native mapping age by itself. | Keep operation phases and energy inputs explicit. Never treat a proxy maintenance completion as a native refresh commit. | +| Erase | The fixed C++ test covers a successful safe erase and an injected post-commit erase failure. The latter preserves the committed destination and reports reconciliation required. The OCP4K loop itself observed zero erase media commands because its 64 committed jobs did not perform a safe reclaim erase. | `erase` consumes its configured aggregate media work and can be blocked by endpoint state. Its payload `bytes` is a work-accounting fixture; no native block is selected or erased. | Native fixed evidence establishes lifecycle/error semantics. Proxy erase is only a resource/energy scenario once explicit coefficients are provided. | +| Failure after activity | Native media activity before a terminal failure remains recorded. Read, program, stale-CAS, and post-commit erase failures preserve their distinct mapping consequences. | The service currently models gating and backlog, not NAND command failure, stale mapping, or reconciliation state. | Failure energy must remain in the native ledger; no native failure probability may be synthesized in the proxy. | + +## Completion identity and conservation + +The native service keeps foreground completion separate from +`maintenance_completions`. `finish` requires zero pending maintenance and one +terminal completion for each accepted maintenance ID. The final backend fixed +tests report `PASS METADATA_VERSION_VALIDITY shared_TSU_PHY maintenance +lifecycle` and `PASS isolated maintain JSON horizon/native-ID protocol`. + +The OCP4K pilot supplied end-to-end runtime evidence: + +- 7,260 offered, submitted, and completed foreground requests; +- zero censored and zero drain-only completions; +- 64 committed maintenance jobs; zero failed, cleanup-failed, or unsupported; +- per-stack foreground equality: each HBF completed 1,775/1,775 and each HBM + completed 40/40; +- 7,228 unique native child transactions: 7,100 `USERIO` plus 128 + `HBF_MAINTENANCE`; the latter are 64 native reads and 64 native programs. + +The aggregate service uses a different identity layer. Explicit jobs retain a +caller `job_id`; `completion_ids` and `maintenance_completion_ids` are emitted +once after the final aggregate batch completes. Automatic foreground cohorts +are retired after completion. Per-stack cumulative offered bytes must equal +cumulative delivered effective bytes plus foreground backlog. These are +model-level identities, not native transaction IDs. + +Fixed proxy tests cover partial completion followed by exactly one terminal +ID, duplicate-ID rejection, completed cohort retirement, and separate +maintenance IDs. The campaign analyzer independently rejects repeated +maintenance completion IDs and per-stack/end-to-end byte imbalance. + +## Shared arbitration: what is and is not the same + +The native backend uses one `MqsimOnlineEngine`. Foreground and maintenance +share the actual address mapper, block manager, TSU, and ONFI PHY. Maintenance +is placed in the existing GC/WL queue class; source pinning and mapping +generation protect ownership without replacing the native scheduler. The fixed +test also schedules a real foreground write during maintenance and observes +that the foreground generation wins exactly once while stale maintenance +discards its destination. + +`TopologyService` shares arithmetic capacities rather than native objects: + +- foreground and maintenance use the same configured channel media capacity; +- HBF relay and HBM-local cohorts share the paired HBM banks and GPU link; +- relay admission jointly checks the HBF and partner-HBM endpoints; +- DASH direct work may proceed when its unrelated relay endpoint is shut down; +- severe state blocks foreground but retains the explicit maintenance policy; + shutdown blocks maintenance until a later legal window; +- the two banks are continuous-turnover buffers, not 20 ms-sized storage. + +Those rules provide a controllable proxy for contention and heat-source +placement. They do not reproduce MQSim TSU ordering, plane command overlap, +FTL allocation, GC, or physical bus timing. Increasing the offered rate to +0.384--1.920 TB/s per stack is deliberate thermal/service pressure; it is not +evidence that the native backend or a product sustains that rate. + +## Evidence identity + +All paths below are under +`eq3_thermal/plans/isolated-maintenance-campaign-v1/points` unless stated +otherwise. + +| Evidence | Result and identity | +| --- | --- | +| `backend-generation-cpp-test-v1/result.json` | PASS; SHA-256 `6eb84a5783265f166b0fad611e7b2d3366cae916d00c15a6cf37ed5c92d4576a` | +| `backend-generation-service-test-v2/result.json` | PASS; same minimal terminal receipt hash; detailed stdout records the JSON horizon/native-ID protocol PASS | +| `OCP4K-LOOP-PILOT01/DONE.json` | COMPLETED in 83.04947019899555 s; SHA-256 `cb977f1c9bb45ec17a660b81ffb3d09cc2d2b3c3530c6c6beb5afd63a868dbcd` | +| `OCP4K-LOOP-PILOT01/summary.json` | Counts above; SHA-256 `4288782e581994510386fcab6be0056348f463fada484e1b3fbc95e961dc13f4` | +| `NATIVE-OPERATION-SUMMARY03/result.json` | PASS; 7,164 native reads, 64 programs, zero erase/unknown commands; SHA-256 `b5a18c1ae8a315bef850358e28e8d076ed962bcadff1cad1ebac87c95e0c94cd` | +| `STARTUP-WRITE-ROLLING-W256-N16384-01/raw/summary.json` | PASS; 16,384 writes + 16,384 read verifies; SHA-256 `332b99a3f1cb7e9549038502e17970edaa32d756e9c8df6c11e953863cdfee30` | +| `experiments/eq3_system_thermal/topology_service.py` | Aggregate proxy source reviewed at SHA-256 `4e40958215f26cd1675ec9280c3674bebb333b943257edbe93f0d54f1fdb55ab` | +| `experiments/eq3_system_thermal/test_topology_service.py` | Proxy fixed contracts reviewed at SHA-256 `ae6344c36b4722f11ca668256b9d084f2105325a61d9863f4afc411436b2636d` | + +The isolated generation/pin binary registered in the backend boundary document +has SHA-256 +`c64610b0f281397975e649b32f44161b00240f12d6a49b9b4d4776b1f554c257`. +The fixed native receipts already cover every requested semantic axis, so this +review did not rerun an unchanged binary merely to obtain a new timestamp. + +## Remaining capability boundary + +Native and proxy evidence together still leave payload equality, target-HBF +RBER/ECC/retry probability, calibrated program/erase energy, die-wide refresh, +physical spare/overprovisioning, product endurance, live GPU, external-GDDR +service/temperature, and token/s unavailable. The new system/thermal matrix +may report conditional aggregate delivery, queueing, resource use, maintenance +facts when actually injected, and complete coupled temperatures. It must keep +native backend capability and aggregate TB/s thermal pressure as separate axes. diff --git a/docs/eq3_thermal/OCP_PAGE_CAPACITY_ALIGNMENT.md b/docs/eq3_thermal/OCP_PAGE_CAPACITY_ALIGNMENT.md new file mode 100644 index 0000000..6857e43 --- /dev/null +++ b/docs/eq3_thermal/OCP_PAGE_CAPACITY_ALIGNMENT.md @@ -0,0 +1,263 @@ +# OCP HBF page, capacity, block, and plane alignment + +Status: **read-only feasibility audit; no campaign input or runtime source was +changed by this audit**. The currently paused campaign remains historical +evidence under its original 16 KiB engineering profile. This document defines +an executable 4 KiB profile for a new experiment version; it does not relabel +old runs. + +## Evidence and limits + +The primary source is the registered local copy of *High Bandwidth Flash +(HBF), High-Level Base Die Specification*, version 0.7.0, 3 August 2026: + +- file: `docs/HBF_OCP/ocp2026-hbf-architecture-specification-v0-7-0.pdf` +- SHA-256: `307531eb8053f00cbeccbc907ddff0a9c4fe6f9d0066a077ce33b0ac99312da3` +- evidence class: `DOC_DERIVED` + +The relevant requirements are: + +| Item | OCP v0.7 evidence | Classification | +|---|---|---| +| NAND page | Section 4.1, p.15 specifies a 4 KiB NAND page. Reads are 64 B through 4 KiB in 64 B multiples and may not cross a 4 KiB NAND-page boundary; writes use a 4 KiB burst at a 4 KiB-aligned address. | `SPECIFIED` | +| Stack/cube | Section 4.1 and Table 3, p.16 specify 16 NAND dies per cube, 16 banks per channel, a 4096 B page, and 512 GiB total cube size. Section 4.3 specifies 16 independent host channels per cube. | `SPECIFIED` | +| Read command granularity | Section 5.3.1, p.56 permits a complete 4 KiB read as 64 separate 64 B commands with different AXI IDs or a supported burst. It also says same-bank sense requests are strictly ordered and that each bank has two page-cache buffers. | `SPECIFIED`, but the present MQSim adapter models a page request rather than the AXI beat protocol | +| Write/block behavior | Section 5.4.1, p.58 says the Core-die write granularity is 4 KiB and the NAND block size is defined by the product specification. | 4 KiB is `SPECIFIED`; pages per block are `UNKNOWN_PRODUCT` | +| Address fields | Section 5.7, p.62 defines R1 as dies, R2 as banks per die, R3 as pages in a bank NAND block, and R4 as 64 B units per NAND page. It does not publish a numeric R3 value for this product. | R3/pages-per-block is `UNKNOWN_PRODUCT` | +| Bank versus plane | Section 11.1, p.112 uses the phrase “banks or planes in die” and describes maximum parallelism in terms of planes, but does not define every OCP bank as exactly one NAND plane. | Direct bank=plane identity is **not specified** | + +Section 4.6's worked address example uses 16 banks per die and four dies. It is +an address-mapping example, and must not override Table 3's 16-die cube or be +treated as a unique physical organization. + +## Existing implementation consumers + +The relevant profile fields are active backend geometry, not descriptive +metadata. + +| Interface/profile field | Actual producer or consumer | Observed behavior | +|---|---|---| +| `page_bytes` | `src/profile/profile.cpp::validate_profile` and `calculate_blocks_per_plane` | Accepts power-of-two values at least 512 B. 4096 B is already legal. Capacity must contain an integral number of complete blocks per plane. | +| one-page stack request | `include/hbfsim/eq3_thermal/mqsim_stack_map.hpp::MqsimStackMapAdapter::map` and `map_stack_page` | When stack mapping is enabled, a request must be exactly one aligned `profile.page_bytes` page. A 16 KiB request under a 4 KiB profile is rejected at this boundary. | +| stack channel/die declaration | `MqsimStackMapAdapter` constructor | Each stack group must contain `channels / stack_count` channels and must declare `channels_per_stack * dies_per_channel` dies. Thus 16 channels per stack with one die per channel maps cleanly to 16 declared thermal dies. | +| `dies_per_channel`, `planes_per_die`, `pages_per_block`, derived blocks/plane, `page_bytes` | `experiments/eq3_maintenance/backend/src/mqsim_online_maintenance.cpp::configure_mqsim` | Writes the fields to MQSim's `Flash_Parameter_Set`. `Chip_No_Per_Channel` remains one. | +| flash geometry | `third_party/mqsim/src/exec/SSD_Device.cpp` | Passes the geometry to the Flash chip/PHY, FTL, TSU, block manager, page-level address mapping, cache manager, and GC/wear-leveling units. | +| blocks and pages | `third_party/mqsim/src/ssd/Flash_Block_Manager_Base.cpp` and page-level address mapping/GC units | Allocates per-plane bookkeeping; pages per block and blocks per plane change free-page pools, mapping, allocation, and GC/erase behavior. | +| native identities | `experiments/eq3_maintenance/backend/service/hbf_mqsim_maintenance_service.cpp` | Exposes native command, transaction, external-request, byte, channel, die, plane, block, and page facts. Maintenance work is also one profile page. | +| thermal die placement | `experiments/eq3_maintenance/energy_ledger.py::ActivityEnergyLedger._placement` | Maps a channel group entry to a thermal-die offset `channel_index * dies_per_channel`, then adds the native die. With 16 channels/stack and one die/channel, all `die0..die15` remain distinct thermal entities. | + +There is therefore no need to rewrite backend scheduling to use a 4 KiB page, +16 channels per stack, or a parameterized plane/block geometry. The geometry +already reaches the real MQSim allocator and arbitration structures. This is +configuration-only at the backend boundary, followed by experiment-local +request generation and metadata adaptation. + +## Recommended executable profile + +For a new, separately identified campaign profile, use: + +| Field | Value | Evidence / meaning | +|---|---:|---| +| `page_bytes` | 4096 | OCP v0.7 normative page size (`SPECIFIED`). | +| channels per HBF stack | 16 | OCP v0.7 host-channel count (`SPECIFIED`). Total `channels = 16 * hbf_stack_count`. | +| `dies_per_channel` | 1 | Engineering projection that makes the existing stack mapper expose all 16 specified dies exactly once. It is not a claim that OCP mandates one die per host channel. | +| `planes_per_die` | 16 | `BANK_AS_MQSIM_PLANE_V1`, an explicit `SCENARIO_ASSUMPTION` that projects the specified 16 banks/channel onto MQSim plane resources. It is an executable first model, not a standards claim or a unique product organization. | +| `pages_per_block` | 256 | Retained `SCENARIO_ASSUMPTION`. OCP explicitly leaves NAND block size/product R3 unspecified. At 4 KiB/page this makes a simulated block 1 MiB. | +| `declared_dies` per stack | 16 | Required by the existing mapper for 16 channels/stack and one die/channel, and agrees with the OCP cube die count. | + +All existing latency, bandwidth, queue, allocation-policy, GC, and energy +parameters should remain separately sourced and unchanged by this alignment. +The OCP page and capacity statements do not calibrate them. + +This profile materially increases modeled arbitration resources. Four HBF +stacks produce 64 MQSim channels and 1024 MQSim planes; eight HBF stacks +produce 128 channels and 2048 planes. No new scheduler code is needed, but a +fixed geometry/identity test and a minimum resource-measured pilot are required +before interpreting performance. In particular, the 16-plane projection can +increase parallelism relative to the prior one-plane fixture. + +## Request alignment + +The simplest OCP-aligned workload emits native 4 KiB requests directly. This +uses the existing one-page stack-map contract without changing default backend +scheduling. + +If an analysis must retain a logical 16 KiB host transaction, the experiment +adapter must create four aligned 4 KiB child requests before submission: + +```text +parent global byte address A, where A % 16384 == 0 +child j address = A + 4096*j, bytes = 4096, j in [0,3] +``` + +Each child is a real MQSim request with its own backend request, transaction, +and NAND command identities. The four children retain the parent's stable ID, +child index, and parent byte count in experiment metadata. The parent completes +at `max(child completion time)` only in derived workload accounting. The +backend does not provide atomic all-four admission or a single native 16 KiB +command, so those properties must not be claimed. + +The current coordinator's request-metadata allowlist does not include parent +and child fields. A thin experiment-local metadata addition is needed if the +16 KiB bundle is retained; otherwise those fields are discarded before the +request ledger is written. Direct 4 KiB generation avoids this compatibility +layer and is the recommended first profile. + +The stack-local ordinal must be expressed in 4 KiB pages. When converting an +old 16 KiB page ordinal `p`, use `4*p+j` for child `j`. The current mapper then +stripes those four page ordinals through the configured channels while keeping +the requested stack identity. Do not apply modulo reduction to a small working +set and describe it as a full-capacity address. + +Minimum fixed assertions for either path are: page alignment; exactly 4096 B +per submitted request; expected stack; physical channel/die/plane in declared +ranges; one unique request and transaction identity per child; distinct native +read-command identities where the backend issues four reads; exactly one +completion per child; and 16384 B total for a four-child logical bundle. + +## Product capacity and finite active namespace + +The OCP product capacity is: + +```text +512 GiB/stack = 549,755,813,888 B + = 134,217,728 addressable-equivalent 4 KiB pages/stack +average attribution over 16 dies = 8,388,608 pages/die +``` + +The per-die value is capacity arithmetic only. It is not an observed physical +page count: factory bad blocks, spare capacity, overprovisioning, and the +physical plane/block organization are unknown. + +Under `BANK_AS_MQSIM_PLANE_V1` and the 256-page/block assumption, a full +512 GiB profile happens to produce 2048 simulated blocks/plane: + +```text +134,217,728 / (16 channels * 1 die/channel * 16 planes/die * 256 pages/block) += 2048 blocks/plane +``` + +This result is conditional on both engineering assumptions; it is not an OCP +block-count specification. + +The paused pilot's finite profile contains 145,492,017,152 B across four stacks +(135.5 GiB total, 33.875 GiB/stack). At 4 KiB/page the same byte extent would be +8,880,128 pages/stack, or 555,008 pages/die by even attribution. It is only +6.6162109375% of the specified 512 GiB per-stack capacity. More importantly, +8,880,128 pages are not divisible by the proposed complete-plane block quantum: + +```text +16 channels * 1 die/channel * 16 planes/die * 256 pages/block += 65,536 pages/stack per blocks/plane increment +``` + +For a valid finite MQSim geometry near the existing active extent, use +8,912,896 pages = 34 GiB per stack, yielding 136 blocks/plane. For four stacks +this is 136 GiB total. This finite namespace is a deliberate scenario and must +be recorded separately from the 512 GiB product identity. It adds 128 MiB per +stack relative to the paused profile and must not be described as byte-identical +to it. + +In general, round the required per-stack active bytes upward to a multiple of +`4096 * 65,536` bytes for this exact geometry. The full 512 GiB capacity can be +kept in the product registry while the smaller, integral namespace is used for +bounded simulation. + +## Maintenance and comparability effects + +Existing maintenance submission consumes one `profile.page_bytes` page. A +change from 16 KiB to 4 KiB therefore changes the byte coverage of an unchanged +maintenance-job count: 64 jobs cover 256 KiB rather than 1 MiB. A new campaign +must define maintenance intent in bytes or explicitly use four times as many +page operations when byte-equivalent coverage is required. Old and new results +are not maintenance-volume comparable without this normalization. + +Changing from 16 KiB pages with 256 pages/block to 4 KiB pages with 256 +pages/block also changes the modeled erase/GC block from 4 MiB to 1 MiB. This is +a real behavior change in MQSim. Since OCP leaves pages per block unknown, any +result sensitive to GC, erase, or block allocation remains +`CONDITIONAL_SIMULATED` under the 256-page assumption. + +## Readiness decision + +Classification: **SUPPORTED_WITH_EXPERIMENT_PROFILE**. + +- A 4096 B page, 16 channels per stack, one die per channel, 16 declared dies, + parameterized planes, and parameterized pages/block are all consumed by the + existing backend. No scheduling or public-ABI refactor is required. +- Native 4 KiB request generation is the minimum path. A logical 16 KiB bundle + needs only an experiment-local four-child adapter and metadata fields; it + does not become one atomic backend command. +- `BANK_AS_MQSIM_PLANE_V1` and `pages_per_block=256` must remain visible + scenario assumptions. Bank/plane equivalence, physical blocks/plane, spare + capacity, overprovisioning, and product bad-block counts remain unknown. +- Results from the new geometry require a new profile/scope identity and cannot + overwrite or retroactively relabel the paused 16 KiB evidence. + +Before a new pilot, validate profile integrality; stack/channel/die coverage; +four-page identities if bundling is used; command/completion uniqueness; energy +placement across all 16 thermal dies; maintenance byte normalization; and peak +RAM/runtime for the enlarged channel/plane count. These are bounded fixed or +minimum-pilot checks, not a request to restart the paused campaign. + +## Bounded backend verification + +`GEOMETRY4K-FULLCAP01` subsequently exercised the existing isolated MQSim +backend with four complete 512 GiB logical stacks and no thermal solve. The +frozen binary hash was +`c64610b0f281397975e649b32f44161b00240f12d6a49b9b4d4776b1f554c257`. + +- The service instantiated 64 channels and reported 1024 parallel units. +- 1024 unique native 4 KiB reads completed once each and covered all + 4 × 16 channel-derived thermal-die identities × 16 projected planes. +- Four one-page maintenance operations retained real source/destination + channel and plane addresses and completed read, destination program, mapping + commit, old-source retirement, and final completion. +- The final receipt reported no pending foreground or maintenance work. +- Execution took 12.31 s and 21,511,976 KiB maximum RSS under a 48 GiB address + limit, with no swap. No thermal, fabric, HBM, GPU-link, or GDDR work ran. + +The point, raw protocol transcript, native facts, resource record, and artifact +hashes are under +`eq3_thermal/plans/isolated-maintenance-campaign-v1/points/GEOMETRY4K-FULLCAP01`. +This verifies actual consumption of the configured geometry. It does not raise +the bank-to-plane or 256-page/block assumptions above `SCENARIO_ASSUMPTION`. + +A bounded follow-up, `GEOMETRY4K-BOUNDARY01`, confirmed that the last legal +stack-local page maps to global/backend logical page 536,870,911, reaches the +expected channel/plane, and completes. A direct fixed test of +`MqsimStackMapAdapter` rejected the next stack-local page as out of range and +also rejected a 16 KiB request under the 4 KiB one-page contract. + +That follow-up deliberately retained a failed physical-block assertion. CWDP +selects channel/chip/die/plane from the LPA, but an unmapped first-touch read +then asks MQSim's allocator for the next free physical block/page in that +plane. Consequently, a logical ordinal described as block-0/page-255 did not +force physical block 0/page 255. The four reads instead occupied page 0 then +page 1 in their selected planes. Logical namespace boundaries and physical +allocator boundaries are distinct. At that point, a sequential physical +page-255→page-0 block crossing remained `NOT_OBSERVED`; no unapproved retry was +started after the failed assumption. The raw failure and diagnosis are under +`eq3_thermal/plans/isolated-maintenance-campaign-v1/points/GEOMETRY4K-BOUNDARY01`. + +After that diagnosis, the separately authorized +`GEOMETRY4K-PHYSICAL-BLOCK02` used 257 serial first-touch logical pages chosen +to remain on one channel/die/plane. The first 256 real MQSim allocations +occupied physical block 0 at pages 0 through 255; allocation 257 entered +physical block 5 at page 0. The new block number was intentionally not assumed +because the FTL reserves other work-front blocks. This confirms that +`pages_per_block=256` has a real allocator consumer and that its physical block +boundary is enforced. + +The two address layers must remain distinct in claims: + +- OCP's 4 KiB host/local addressing and R1/R2/R3/R4 block calculation define + the product-facing logical protocol, with R3 product-defined. +- The current adapter validates the finite/full logical page namespace and + uses CWDP to select MQSim channel/die/plane. MQSim's page-level FTL then + assigns physical block/page locations dynamically. + +Consequently this experiment is an OCP-shaped page/capacity profile with a +documented bank-to-plane projection. It does **not** implement the complete OCP +zone/direct block-addressing protocol and must not be described as fully OCP +compliant. diff --git a/docs/eq3_thermal/P2_4MM_2MM_STATIC_DIAGNOSTIC.md b/docs/eq3_thermal/P2_4MM_2MM_STATIC_DIAGNOSTIC.md new file mode 100644 index 0000000..2294bc4 --- /dev/null +++ b/docs/eq3_thermal/P2_4MM_2MM_STATIC_DIAGNOSTIC.md @@ -0,0 +1,69 @@ +# P2 4 mm to 2 mm static diagnostic + +Status: read-only diagnosis, 2026-09-20. Existing source, generated inputs, +receipts, and retained sensor CSVs were inspected. No solver, build, or large +postprocessing job was run. Findings are `DOC_DERIVED` unless marked otherwise. + +## Exact failing observable + +`eq3_thermal/plans/campaign-v1/R01-R02-comparison.json` compares the full 100 s +R01 4 mm result against the full 100 s R02 2 mm result at the registered 0.1 s +sensor observations. Its global maximum absolute difference is **3.253 K**. +The exact maximum is: + +| Field | R01 4 mm | R02 2 mm | +| --- | ---: | ---: | +| time | 31.0 s | 31.0 s | +| sensor | `component:hbf3.die15:hotspot` | same | +| observable | maximum cell temperature over that physical component | same | +| temperature | 312.007 K | 315.260 K | +| reported hotspot cell | `n60_3_11` | `n60_7_23` | + +The cell identities differ because the lateral grids differ; both identify a +cell in the same component and z slab. The comparison is a maximum over all +registered **sensor/time** pairs, not a cell-by-cell field norm. The comparator +requires identical sensor sets and timestamps, then computes absolute error +(`tools/eq3_campaign_compare.py:28-46`). Both retained CSVs contain the same +275 sensor identities. This remains `NUMERICAL_ACCURACY_LIMIT`; this audit did +not find evidence that the 3.253 K value is caused by a mapping bug. + +## Controlled inputs + +The normalized inputs copied into the R01 and R02 run directories are exactly +equal as JSON. Both use 100 s duration, 20 ms solver step, 0.1 s observation, +1365 J input energy, and 21.1572437888 J/K total reference capacity. The only +intended generated-grid difference is: + +- R01: `16 x 16 x 63`, 16,128 cells, 4 mm lateral pitch. +- R02: `32 x 32 x 63`, 64,512 cells, 2 mm lateral pitch. + +Both retained observations contain all 5,000 solver frames. Their energy +residuals are small but nonzero because output temperature is quantized to +0.001 K and boundary energy is reconstructed with backward-Euler endpoint +flux: R01 `2.47e-6`, R02 `1.91e-6` relative. These receipts support completed +and internally accounted runs; they do not establish spatial convergence. + +## Static consistency audit + +| Contract | Static finding | Classification | +| --- | --- | --- | +| Geometry and material ownership | Uniform x/y planes must align every component edge; no snapping is allowed. Every cell has exactly one component or declared background owner, and every component's cell volumes must sum to its physical volume (`eq3_layered_export.py:23-89`). R01/R02 retain the same 63 z interfaces. | No inconsistency found. | +| Material units | Conductivity is exported from W/(m K) to W/(um K) by `1e-6`; volumetric heat capacity from J/(m3 K) to J/(um3 K) by `1e-18` (`:141-146`). Each z layer uses a material layout, so heterogeneous cells are not inferred from floorplan power labels. | No unit/mapping inconsistency found. | +| Internal interfaces | The independent RC/network audit uses two half-cell resistances in series, `dx/(2 k_left) + dx/(2 k_right)`, including material boundaries (`:92-103`). Stock input preserves every z material slab explicitly. Additional area contact resistance must be exactly zero or export fails (`:210-216`); the shared normalized input declares zero residual contact and represents finite bond/TIM layers explicitly. | Consistent with the declared ideal residual-interface assumption; nonzero contact remains unsupported. | +| Top/bottom Robin boundaries | Stock export converts HTC from W/(m2 K) to W/(um2 K) by `1e-12` and preserves ambient temperatures (`:147-150`). The independent network combines half-cell conduction and `1/h` in series (`:104-112`). Lateral boundaries must be explicitly adiabatic. A single-z-layer model with two HTC faces is rejected (`:135-140`). | No boundary translation inconsistency found. | +| Power/source weights | Each stock floorplan rectangle is exactly one cell. Component watts are split by `cell_volume/component_volume`; geometry volume conservation therefore makes component source weights sum to one across x/y/z (`:153-174`). Power transitions must align to slots, with no time averaging (`:116-132`). Both run receipts retain the same 1365 J source. | No duplication or omitted-source evidence found. | +| Layer and output order | Stack layers are emitted in reverse z declaration order as required by the pinned backend, while one `Tmap` is requested for every z slice at every solver step (`:175-181`). The reader opens fields in ascending z, reads exactly `nx*ny` finite values per frame, and concatenates slices in the same z/y/x order used to assign cell indices (`eq3_layered_observe.py:55-67,145-166`). | No ordering mismatch found. | +| Sensor reduction | Weighted means distribute each declared component weight by cell volume and assert total sensor weight one. Hotspots take the maximum over the exact component-cell index union and retain the winning cell ID (`eq3_layered_observe.py:70-96`). | The failing observable is correctly a component-cell maximum, not a mislabeled mean or request occupancy. | +| Boundary-energy observation | Boundary flux is integrated from every solver-step full field before 0.1 s sensor decimation (`eq3_layered_observe.py:133-180`). | Consistent with receipts; not an independent proof of the stock solver equation. | + +## Interpretation boundary + +The evidence supports a real sensitivity of the component hotspot observable to +4 mm versus 2 mm lateral discretization. Hotspots can change more than +volume-weighted means when a finer grid resolves a localized maximum; that is a +plausible mechanism, but attribution of the full 3.253 K difference remains +`INFERRED` until a same-window finer reference establishes a convergence trend. + +No physical parameter, boundary, source weight, sensor definition, threshold, +or old PASS criterion should be changed to remove this failure. The original +`NUMERICAL_FAIL` remains valid and P2 remains unfrozen. diff --git a/docs/eq3_thermal/P2_GRID_IMPLEMENTATION_CAUSE_AUDIT.md b/docs/eq3_thermal/P2_GRID_IMPLEMENTATION_CAUSE_AUDIT.md new file mode 100644 index 0000000..53744ce --- /dev/null +++ b/docs/eq3_thermal/P2_GRID_IMPLEMENTATION_CAUSE_AUDIT.md @@ -0,0 +1,127 @@ +# P2 2→1 mm 网格温差实现原因审计 + +日期:2026-09-20 +范围:只读审计 `eq3_layered_export.py`、实际 R03 native 3D-ICE 源码、已保存的 R02/R03 输入与传感器结果;未启动求解、构建或大型后处理,也未修改物理参数或实现。 + +## 结论 + +`component:hbm3.base:hotspot` 在 15.0 s 的 2→1 mm 差值为 **2.069 K**:R02 2 mm 是 314.966 K(`n4_7_8`),R03 1 mm 是 317.035 K(`n4_15_16`)。本次逐层源码审计没有发现能直接解释该差值的坐标、单位、半单元界面、各向异性轴、功率重复、z 层倒置、Robin 边界或输出展平错误。因此不能把它分类为 `CONFIRMED_BUG`。 + +现有证据支持把该现象分类为 **NUMERICAL_ACCURACY_LIMIT(局部 cell-max 对网格敏感)**,但这不是“现实器件会相差 2.069 K”的证据。15 s 的热点位于 `hbm3.base` 的北西角附近;同一时段 `hbm3` 全部功率为 0,而与其仅在平面角点相接的 `hbm2` 的 12 个媒体 die 合计输入 64 W。1 mm 网格解析出更窄、更高的角点温度;2 mm 网格把该区域平均到四倍平面面积。与此同时,同一组件的体积均温只从 306.290083 K 变为 306.390026 K,差约 **0.099943 K**。这是“局部最大值支持域改变”而非全组件热量明显不一致的强证据。 + +物理真实性仍为 **UNKNOWN**:硅材料是 `PROXY`,侧面绝热、残余接触热阻为零且几何为情景假设。代码自洽不能把这些假设升级为器件实测事实。 + +## 审计对象与可比性 + +- R02:`plans/campaign-v1/points/R02`,32×32×63、64,512 cells、2 mm、0.02 s step、100 s。 +- R03:`plans/decision-execution-v2/points/R03-V2-FULL1MM`,64×64×63、258,048 cells、1 mm、0.02 s step、100 s。 +- 两者的 `normalized.json` SHA-256 都是 `7a62356f...64ce06`;总输入能量都是 1365 J,总热容分别为 21.1572437888 与 21.157243788800002 J/K。因此比较使用相同几何、材料、边界和功率历史。 +- R02 native 3D-ICE SHA-256 是 `240b598c...7244a7`;R03 storage-ownership v2 native 3D-ICE 是 `ace42d254...ca1d8`。`cc36d620...126e6` 是 `eq3_campaign_stream.py` 的哈希,不是数值引擎哈希。v2 的 `reference/build/3d-ice-storage-owned-u125-v2/source.patch` 只改 factor storage 生命周期和 `sp_ienv(7)` 容量;`DESIGN.md` 明确保留方程、CSC 值、排序和积分。实际 `add_solid_column` 组装式与 stock 对应代码相同。下述 D3 线性恢复又在完整 100 s 传感器输出层排除了这两个 native backend 的可见差异。 + +### 同网格跨 backend 的 DERIVED_LINEAR 排除 + +已有 `D3-V2-TRAIN-REF2MM` 使用 R03 相同的 `ace42d254...ca1d8` native backend、2 mm 网格、完整 100 s,但把全部可变热源统一乘以 0.25。只读前置检查确认: + +- 除功率数值和功率来源状态标签外,几何、材料、边界、传感器和温度域完全相同; +- 200 个功率区间、组件键以及区间端点相同;每个 `power_w`、每组件积分能量和总能量都精确为 R02 的 0.25; +- 初始温度、top ambient、bottom ambient 都是 300 K,没有非零固定源。 + +常物性线性系统因此允许在不新求解的情况下恢复 + +`T_original_pred = 300 K + 4 × (T_D3 - 300 K)`。 + +对 R02 与恢复后的 D3 全部 **275,000 个 sensor/time rows** 对照,最大绝对差为 **0.002000000000294 K**(91.8 s,`component:hbm3.die3:mean`),小于两端三位小数输出传播界 `0.0005 + 4×0.0005 = 0.0025 K`。摘要保存在 `plans/decision-execution-v2/points/backend-linear-sanity-v2/summary.json`,runner 结果为 PASS;最初把功率 provenance/组件积分也当作非功率输入的严格检查被保留为 `backend-linear-sanity` FAILED,随后只排除已证明属于功率缩放的字段。 + +有 31,848 个 hotspot row 的回报 cell id 不同;这些发生在三位小数场输出经 0.25 缩放后产生的并列/量化选择中。正温度缩放下 hotspot 温度本身仍满足上述 0.0025 K 界,但不能用 DERIVED_LINEAR 恢复唯一 argmax cell 身份。因此本检查证明的是完整传感器温度层的 backend 等价,不宣称逐 cell 原始场字节相同,也不是新 native 运行或物理验证。 + +## 1. 网格所有权与源体积守恒 + +`tools/eq3_layered_export.py`: + +- `axes`(23–45)在 x/y 使用统一 pitch 前要求所有实体边界整除,不做坐标 snapping;z 轴始终保留全部几何平面。 +- `discretize`(48–89)以 cell center 判定平面所有权,以完整 z slab 判定活动组件;重叠和未填充都会失败。 +- 同一函数 85–87 行逐组件核对拥有 cell 的总体积等于组件体积。 +- R03 的 `hbm3.base` 是 `[0.016,0,0.001205]` 起、尺寸 `[0.012,0.016,0.00005]` m 的单层硅实体;1 mm 时拥有 192 cells,2 mm 时拥有 48 cells,二者都覆盖相同 9.6e-9 m³。 + +`reference_files`(135–183)为每个 cell 生成一个同尺寸 floorplan rectangle,并把该 cell 的功率设为 `component_power × cell_volume/component_volume`(157–169)。实际 R03 `L4.flp` 中 `C17360` 的位置/尺寸为 `(16000,15000,1000,1000)` µm;`L4.layout` 把同一区域放进 `M3`,而 `package.stk` 中 `M3` 是各向同性 150 W/(m·K) 的硅。 + +native 的 `floorplan_matrix_fill`(`reference/build/3d-ice-stock-e0bb685-gnu17-longint/sources/floorplan_matrix.c:150–223`)按 rectangle 与 cell 的重叠面积除以 rectangle 面积;此处一个 rectangle 恰好覆盖一个 cell,矩阵权重为 1。`power_grid_fill`(同树 `sources/power_grid.c:811–907`)随后只调用一次 `fill_sources_floorplan`。因此 exporter 的体积分配没有被 native 再按组件面积复制。 + +**判定:** 未发现源功率重复或遗漏。生成收据还独立报告 `emitted_energy_j=1365.0000000000107`,与输入 1365 J 的差为浮点舍入量;该收据不能单独证明 native 矩阵正确,但与上述生产者/消费者代码一致。 + +**可证伪假设 S1:** 若某个 floorplan rectangle 因浮点边界被 native 映射到两个 cells,则其 overlap 权重之和仍应为 1,但空间源会跨界。最小检查是只构建 floorplan matrix 并导出 `C17360` 以及 hbm2 角点元素的非零 row/weight;应恰有一个 row、权重 1。当前输入边界均在整 µm 网格线上,源码和已解析结果不支持该假设,但尚无该矩阵的运行时 dump。 + +## 2. 半 cell 界面与各向异性 + +exporter 的独立网络函数 `network`(92–113)对每个正方向邻居使用 + +`G = A / (d1/(2 k1_axis) + d2/(2 k2_axis))`, + +即两个半 cell 热阻串联;该网络用于 RC 导出和观测能量账,不是 native reference 求解器本身。 + +实际 native 路径为: + +- `sources/layer.c:get_thermal_conductivity`(186–226)先按 layout 查当前位置材料,再按 direction 0/1/2 取 x/y/z 导热率。 +- parser `bison/stack_description_parser.y:295–339` 把三个输入值依次写入 `[0],[1],[2]`,没有轴交换。 +- `sources/thermal_grid.c:get_conductance_top/bottom`(963–1288)和 `north/south/east/west`(1292–1642)返回本 cell 到界面的半 cell conductance。 +- 实际 v2 `system_matrix.c:add_solid_column`(732–1045)在 x、y、z 六个方向都用 `PARALLEL(g_side_this,g_opposite_neighbor)` 合成两个半 cell 的串联热阻;对角项加同一 conductance,非对角项写负值。 +- `get_capacity`(stock `sources/thermal_grid.c:433–530`)使用 layout 查得的体积热容乘实际 cell 长、宽、高,和 exporter 的 `volume × Cv` 一致。 + +热点 cell `n4_15_16` 本身是各向同性硅 `[150,150,150]` W/(m·K)。它的西、北邻居均为各向同性 underfill `[1.5,1.5,1.5]`;下邻居 `hbm3.attach`、上邻居 `hbm3.bond0` 均为 bond `[5.5,5.5,113]`。因此,即使把硅的 x/y 轴互换也不会改变该热点;bond 的高 z 导热率由 direction 2 消费,源码路径与输入顺序一致。 + +用 R03 cell 尺寸直接代入,热点到西/北 underfill 的界面导热约为 1.485×10⁻⁴ W/K;到下方 25 µm bond 的界面约 3.606 W/K,到上方 8 µm bond 约 4.947 W/K。native 半 cell `PARALLEL` 与 exporter 公式得到相同量级和表达式。 + +**判定:** 半 cell 和各向异性实现未见不一致;`hbm3.base` 位于内部 z=4,任何 top/bottom 特例不会直接作用于该 cell。 + +**可证伪假设 I1:** 用两种材料、两个 cells 的解析稳态 fixture 分别沿 x/y/z 放置,矩阵界面项应等于上式且对称。仓库已有一般方程证据,但没有绑定本次材料和尺寸的矩阵项收据。若失败才可定为 native interface bug。 + +## 3. z 位置、stack 顺序与 Robin 边界 + +exporter 按几何低 z 到高 z 生成 `L0..L62`,但在 `package.stk` 中以 `S62..S0` 逆序声明(`reference_files` 172–181)。这符合 3D-ICE parser 的约定:`stack_description_parser.y:1198–1248` 明确把列表首元素当 top-most、尾元素当 bottom-most,然后从尾到首赋递增 layer offset。因此 `S4/D4/L4` 的 native offset 是 4,与 `reference_grid.json` 的 z index 4 相同。 + +观测器为每层声明 `Tmap(Sz,"field_z.txt",step)`;native `stack_element_print_thermal_map`(stock `sources/stack_element.c:229–286`)先跳到该 stack element 的 source layer offset,再按 row 外层、column 内层输出。`eq3_layered_observe.py:55–67` 也按行读取并以 x/column 为最快维;123–151 行按 `field_0..field_62` 递增拼接。因此 `index = z·nx·ny + y·nx + x` 与 exporter 的 cell 顺序一致。 + +边界方面,exporter 只允许 top/bottom Robin 和绝热 sides。`package.stk` 把 1400 与 25 W/(m²·K) 分别缩放为 1.4e-9 与 2.5e-11 W/(µm²·K)。native `get_conductance_top` 的 ambient 分支(1012–1027)以及 `get_conductance_bottom` 的 PCB 分支(1202–1217)都实现 `A/(dz/(2k)+1/h)`;`add_solid_column` 仅在最顶/最底层把该项加到对角,`power_grid_fill` 把 `G·Tamb` 加入 RHS。z=4 的 `hbm3.base` 不会误用 Robin 边界。 + +**判定:** 未发现 z 反转、层错配、边界单位或边界施加到内部层的证据。 + +**可证伪假设 Z1:** 读取 native 解析后的 stack dump,应显示 `S0 offset=0`、`S4 offset=4`、`S62 offset=62`,top sink 连到 62、bottom sink 连到 0。源码已强支持此结果,但本轮没有生成运行时 dump。 + +## 4. 15 s 热点的空间和功率因果 + +R03 1 mm 最热点 `n4_15_16`: + +- 范围 x=[16,17] mm、y=[15,16] mm、z=[1.205,1.255] mm;中心 (16.5,15.5,1.230) mm。 +- 西边和北边紧邻 underfill;东边和南边仍为 `hbm3.base`。 +- 它是 `hbm3.base` 的北西角 cell。 + +R02 2 mm 对应最热点 `n4_7_8` 覆盖 x=[16,18] mm、y=[14,16] mm,中心 (17,15,1.230) mm。两个网格的 winner 覆盖同一几何角,但 2 mm cell 的平面面积和热容都是 1 mm winner 的 4 倍。 + +组件布局显示:`hbm3.base` 占 x=[16,28] mm、y=[0,16] mm;`hbm2.base` 占 x=[0,16] mm、y=[16,28] mm。二者只在 (16,16) mm 角点相接。14.5–15.0 s 时 `hbm2.die0..die11` 每层 5.333333 W,合计 64 W;`hbm3.base` 以及 `hbm3` 所有媒体 die 都是 0 W。传感器轨迹在 14.5–15.0 s 从 313.829 K 连续升至 317.035 K,15.1 s 降至 316.682 K,与 15.0 s 激励切换一致。 + +离散矩阵只有面邻接,没有角点直接导热。hbm2 的热量要通过角点周围的 underfill/相邻层传到 hbm3。减小 xy cell 后,这条窄空间路径以及局部温度梯度得到更高分辨率;`max` 取的是一个 cell center 的最高值,不是固定面积传感器平均。因此该位置本来就是最容易出现非单调或慢收敛 cell-max 的位置。 + +**可证伪假设 H1(当前首选):** 2.069 K 主要是角点附近的空间离散误差。若在相同物理面积上比较,例如对 1 mm 的 x=[16,18], y=[14,16] 四个 cells 做体积平均,再与 2 mm winner 比,差应显著小于 2.069 K;同样,沿离角点距离的剖面应显示差值快速衰减。该检查需要场数据任务读取既有 raw;本审计不重复该任务。 + +**可证伪假设 H2:** 若 1 mm 的更高峰是输出索引错位,则 `n4_15_16` 周围的场值不会形成以 (16,16) mm 为中心、向 hbm3 内部衰减的连续梯度,或热点会落到非 `hbm3.base` 材料。现有 mapping 已确认该 cell 为 M3 silicon,但邻域温度剖面仍由独立场数据任务验证。 + +## 5. Hotspot reduction 的含义 + +`eq3_layered_observe.py:sensor_mapping`(70–85)对 `max` 传感器取组件所有 cell index 的集合;`readings`(88–96)直接选温度最高的 cell,并回报其 grid cell id。没有地址推断、插值或额外平滑。native Tmap 使用 `%7.3f` 输出(`stack_element.c:280`),量化上限约 0.0005 K,不能解释 2.069 K。 + +在 15.0 s: + +| 指标 | 2 mm | 1 mm | 绝对差 | +|---|---:|---:|---:| +| `hbm3.base` hotspot | 314.966 K | 317.035 K | 2.069 K | +| `hbm3.base` volume mean | 306.290083 K | 306.390026 K | 0.099943 K | + +所以 2.069 K 是两个不同尺寸 cell 支持域上的离散最大值之差,不能解释成固定位置点温度误差,更不能解释成现实温度波动。 + +## 6. 剩余未知与最小下一步 + +1. **场形状证据(最高信息量,读已有 raw):** 对 14.9、15.0、15.1 s 提取 hbm3/hbm2 角点邻域,不做全局大后处理。验证 H1/H2,报告同面积聚合、剖面和材料边界。 +2. **矩阵单元测试:** 静态构造两材料 x/y/z 界面,读取单个矩阵系数和 floorplan 权重;这是验证实现的固定小测试,不改变物理。当前没有必要先修改求解器。 +3. **物理结论边界:** 即便 0.5 mm 进一步收敛,也只能说明当前常物性、理想接触、绝热侧边模型的离散收敛;不能验证真实 HBM/HBF 封装温度。器件级结论仍需材料、接触和边界标定。 + +当前最窄结论是:**没有确认的实现 bug;2.069 K 有明确的角点局部化和 cell-max 网格敏感机制,旧 0.25 K 标准仍然失败,P2 不冻结。** diff --git a/docs/eq3_thermal/P2_LOCAL_FIELD_DIAGNOSTIC.md b/docs/eq3_thermal/P2_LOCAL_FIELD_DIAGNOSTIC.md new file mode 100644 index 0000000..e1e3b69 --- /dev/null +++ b/docs/eq3_thermal/P2_LOCAL_FIELD_DIAGNOSTIC.md @@ -0,0 +1,114 @@ +# P2 local-field diagnostic: R02 2 mm versus R03 1 mm + +Date: 2026-09-20 +Scope: read-only, `hbm3.base`, 15 s, physical z layer 4 only +Status: `NUMERICAL_ACCURACY_LIMIT`; no `CONFIRMED_BUG` found + +## Answer + +The reported 2.069 K hotspot difference is mostly a spatial-sampling effect, +with a smaller resolved-field difference. It is not a 2.069 K shift of the +whole HBM base and it does not currently indicate an indexing, component-map, +or transport implementation bug. + +After volume-averaging each 2x2 block of the 1 mm result onto the matching +2 mm cell: + +- 1.5065 K (72.81%) is the fine-cell peak above its 2x2 average. This is the + hotspot-sampling part that a 2 mm cell cannot represent. +- 0.5625 K (27.19%) remains between the averaged 1 mm peak cell and the 2 mm + peak cell. This is a spatial-discretization difference in the solved field. +- The `hbm3.base` volume-weighted mean changes by only 0.09994 K, from + 306.29008 K to 306.39003 K. Across all 48 coarse footprints, the mean + absolute coarse-versus-averaged-fine difference is 0.09994 K and the maximum + is 0.5625 K. + +The evidence supports `NUMERICAL_ACCURACY_LIMIT`, rather than +`CONFIRMED_BUG`. It does not prove that the 1 mm absolute temperature is +physically accurate: R03 remains `REFERENCE_UNQUALIFIED`, and neither this +diagnostic nor the existing spatial comparison establishes Model Freeze. + +## Location and source context + +The target is layer 4, z = 1.205--1.255 mm. `hbm3.base` spans x = 16--28 mm +and y = 0--16 mm. + +| Quantity | R02, 2 mm | R03, 1 mm | +|---|---:|---:| +| hotspot | 314.966 K | 317.035 K | +| cell | `n4_7_8` | `n4_15_16` | +| cell center | (17.0, 15.0, 1.230) mm | (16.5, 15.5, 1.230) mm | +| hotspot minus component mean | 8.67592 K | 10.64497 K | + +Both grids therefore select the same physical corner. The four 1 mm cells +inside the hottest 2 mm footprint are 315.473, 314.263, 317.035, and +315.343 K; their volume average is 315.5285 K. The 1 mm peak is concentrated +in the half of that footprint closest to the corner. + +During the completed 14.5--15.0 s source interval, all 12 `hbm2` dies dissipate +64 W total. The `hbm2` footprint x = 0--16 mm, y = 16--28 mm touches the hot +corner of `hbm3.base` at (16,16) mm. In the same interval, `hbm3` dies, +`hbm3.base`, and the GPU have zero applied power. `hbm3.die0` occupies the same +planar footprint as its base, begins 8 um above it, and is unpowered; the GPU +has a 4 mm planar gap and is also unpowered. This is consistent with a steep, +localized temperature gradient carried from the active neighboring HBM2 +stack, although physical validation of that gradient remains outside this +diagnostic. + +![Local field comparison](figures/hbm3_base_15s_local_field.png) + +The two temperature panels use one common color scale. Cyan plus signs show +each grid's hotspot. The difference panel is `2 mm - volume-averaged 1 mm`; +the most negative cell is -0.5625 K at the shared hot footprint. + +## Method and invariants + +Only `field_4` was read. The R02 registered plain raw stream and the R03 +EQ3TMK1 lossless stream were each read through EOF. Their full layer hashes, +byte counts, row counts, and 5000-frame coverage were checked before retaining +frame 750 (15 s). R03's codec footer was therefore also validated. The +selected component has one common z layer and the physical component bounds +match across grids. + +Every 2 mm component cell mapped to exactly four contained 1 mm cells. Each +projection used a volume-weighted average. The peak decomposition obeyed the +identity + +`fine peak - coarse peak = (fine peak - projected fine peak) + (projected fine peak - coarse peak)` + +with a 0 K numerical residual. These percentages decompose the difference +between two computed fields; they are not an error budget against continuum +truth and do not prove that 1 mm is exact. No temperature, timestamp, raw field, solver, +equation, source trace, or acceptance threshold was changed. + +The two runs' `normalized.json` inputs have the same SHA-256, +`7a62356faef94e9d16391c39a1e6ccf81d284843c0d9c3ea0d9471d0d364ce06`. +Both observation receipts declare 0.001 K field-output quantization. That +quantization can contribute only millikelvin-scale rounding here and cannot +explain the 0.5625 K projected-field difference. + +## Evidence + +- Preflight: + `eq3_thermal/plans/decision-execution-v2/P2_LOCAL_FIELD_PREFLIGHT.md` +- Machine-readable result: + `eq3_thermal/plans/decision-execution-v2/points/p2-local-field-hbm3base-15s-v2/local_field.json` +- Safety-run receipt: + `eq3_thermal/plans/decision-execution-v2/points/p2-local-field-hbm3base-15s-v2/result.json` +- Static figure and receipt: + `eq3_thermal/plans/decision-execution-v2/points/p2-local-field-figure/` +- Initial caller setup failure, before any analysis program started: + `eq3_thermal/plans/decision-execution-v2/points/p2-local-field-hbm3base-15s-setup-failed/FAILED.json` + +The successful extraction used 1009 MiB sampled aggregate RSS, 9.79 s wall +time, one CPU, and no GPU. The figure used 108 MiB sampled aggregate RSS and +4.35 s wall time. + +## Limits + +This is a one-frame, one-layer diagnostic at the registered worst hotspot. It +separates subcell sampling from the difference between the two discrete fields, +but cannot identify continuum truth or a convergence order. A moving hotspot +maximum must not be used as a Richardson-like same-observable convergence +estimate. Full-window D2 scores and the existing energy and domain checks +remain separate evidence, and R03 remains unqualified. diff --git a/docs/eq3_thermal/P2_NUMERICAL_NEXT_STEP_SCREEN.md b/docs/eq3_thermal/P2_NUMERICAL_NEXT_STEP_SCREEN.md new file mode 100644 index 0000000..7718e7d --- /dev/null +++ b/docs/eq3_thermal/P2_NUMERICAL_NEXT_STEP_SCREEN.md @@ -0,0 +1,155 @@ +# P2 数值下一步筛选 + +日期:2026-09-20 +状态:**只读筛选;未生成输入、未运行求解、未修改代码** + +## 建议 + +优先闭合当前系统:保留 2→1 mm 未达到 0.25 K 的 +`NUMERICAL_ACCURACY_LIMIT`,P2 reference 继续标为未冻结;不要直接启动 0.5 mm +uniform 求解,也不要为追求 PASS 无限细化。完整 1 mm 已把早先的“前 4 s 不能代表 +完整激励”缺口闭合,但完整 100 s 的 hotspot 最大相邻网格差仍为 **2.069 K**,所以 +0.25 K 标准仍是失败,而不是已经接近到可据此预计下一层必过。 + +若论文结论必须依赖合格 reference,下一项应是一个明确的**数值路径选择**: + +1. 继续 uniform 0.5 mm:科学序列最直接,且属于用户已授权的同方法数值细化;成本 + 和可行性尚未知,可先做显式 guard、输入生成和 symbolic/resource pilot,再决定 + 是否值得求解,无需把它虚构为新的用户审批阻塞。 +2. 使用 native non-uniform/local refinement:可能降低单元数,但会改变网格序列、 + 输入生成、传感器映射和误差解释,属于结构性数值方案,不能作为现有 uniform + 0.5 mm 的便宜替身。 + +从“快速获得完整系统”的目标看,推荐选择是**当前停止细化并保留失败状态**; +只有需要解除 P2 reference 限制的具体结论时,再从上述两条中选择;其中新方法的 +non-uniform/local refinement 仍须明确确认。 + +## 已完成求解的资源事实 + +以下均为同一 100 s train、20 ms step、63 个 z slab 的实际完成运行。内存以 solver +自身 stdout 的 peak 为主;运行收据的 PID-only sampling 明确只是 lower bound。 + +| 运行 | uniform 网格 | cells | wall / emulation | factorization | solver 报告 peak | +|---|---:|---:|---:|---:|---:| +| R02,2 mm | 32×32×63 | 64,512 | 237.649 s / 236.776 s | 6.772 s | 1.17 GB | +| R03-V2-FULL1MM,1 mm | 64×64×63 | 258,048 | 2,759.005 s / 2,753.602 s | 251.894 s | 7.30 GB | + +R03 完整运行是在本轮该实验分配的 3,600 s watchdog、12 GiB 单进程、16 GiB 任务 +配置下完成的。外层 `solve-operational.json` 对 descendants 采样得到 +`7,735,140 KiB`(约 7.38 GiB)任务峰值,与 solver 自报 7.30 GB 相互支持。内层 +`DONE.json` 的约 43 MiB 只是 PID wrapper 范围的 lower bound,不能代表完整进程树。 +此前 R03-OWNED-PILOT4S 的 4 s prefix 也不能代替完整资源或完整激励结论。 + +2→1 mm 时 cells 增加 4 倍,实际 emulation 约增加 11.6 倍、solver peak 约增加 +6.2 倍,factorization 增幅更大。这只表明当前稀疏分解和求解成本呈明显非线性, +属于 **INFERRED 的风险提示**。稀疏 LU 的 fill、排序、矩阵结构和内存峰值都可能随 +网格变化;不能把这些倍率线性或按固定幂次外推到 0.5 mm,也不能据此断言必然 OOM。 + +## 0.5 mm 当前阻点是什么 + +0.5 mm uniform 网格为 `128×128×63 = 1,032,192` cells。当前 +`tools/eq3_layered_export.py::discretize()` 在实际构造 cell list 前使用固定 +`400000` cell safety guard,因此该输入会在静态导出阶段明确失败。 + +这条 `400000` 是导出器自己的显式软件防护,用于要求先审查生成内存;它不是 native +3D-ICE 的已知 cell 上限,也不是用户曾使用的 12 GiB 单进程限制。当前用户规则已经 +改为按实验评估资源,R03 使用 12/16 GiB 是该次实验的具体分配,不能把旧数值当成 +0.5 mm 的永久硬上限。反过来,取消固定上限也不会自动证明 0.5 mm 可运行。 + +若选择 uniform 路径,最小 guard 方案应是把常数改成**默认仍为 400000 的显式参数**, +并在创建百万个 Python cell dict、floorplan map 和 sensor map 之前先计算 shape/cell +count、估计输入与派生文件、核对实时可用 RAM/磁盘,再由绑定的 point scope 提高值。 +这只是解除软件前置拒绝,不构成 0.5 mm 资源可行性证据。参数化 guard、静态生成和 +有停止条件的 resource pilot 属于已经授权的同方法数值细化范围;本报告因主线优先及 +当前信息价值暂不启动它们,而不是等待新的用户批准。 + +## native 与当前生成链的 non-uniform 能力 + +静态源码证据显示,固定的 3D-ICE tree **确实包含 non-uniform native 路径**: + +- scanner 接受 `non-uniform`;自带 `example_transient_nonuniform.stk` 使用 + `non-uniform true` 和 stack-element `discretization x y`; +- `thermal_data.c` 构建 `Cell_list`、`Cell_pointer` 和 non-uniform connections; +- `system_matrix.c`、`layer.c`、`power_grid.c` 与 inspection/output 路径都有 + non-uniform 分支。 + +但当前 EQ3 reference producer **没有接入这项能力**。`eq3_layered_export.py` 对 +reference 使用一个全局 uniform x/y pitch,要求所有实体边缘落在该网格上;生成的 +`dimensions` 没有 `non-uniform true`,stack 中也没有输出逐元素 +`discretization`。`axes(mesh_m=None)` 的 geometry-edge Cartesian 网格当前用于独立 +RC 路径,不是 native reference 输入。因此准确状态是: + +`NATIVE_CAPABILITY_PRESENT / EQ3_REFERENCE_PRODUCER_NOT_CONNECTED`。 + +“nonuniform power”测试只证明每组件/组功率权重可以不均匀,不能证明 reference +使用了局部空间网格。 + +## local refinement 的收益与结构改动边界 + +局部细化可把 0.5 mm 放在高梯度、发热实体边界或已观察 hotspot 周围,其余区域保持 +1/2 mm,因此有机会少于 1,032,192 cells。这里的收益只是 **INFERRED**;尚无由目标 +几何生成的 cell count、fill-in、运行时间、内存或误差数据。 + +这不是修改一个 mesh 数字即可完成。最小完整方案仍需: + +- 给 normalized geometry 生成确定性的逐 stack-element x/y discretization,并禁止 + snapping、重叠和漏填; +- 让 native non-uniform cell identity 回写到 source weight、component ownership、 + mean/hotspot sensor、边界能量和 checkpoint/输出读取; +- 验证总体积、热容、源能量、界面导热和 Robin 面积守恒; +- 固定 refinement rule,不得看完候选输出后追逐 hotspot; +- 重新定义比较:local mesh 不是 uniform 4→2→1→0.5 的同一序列,不能把结果直接 + 填入旧 Richardson/相邻 uniform 收敛阶。 + +这些改动影响数值网格生产者、native 输入、输出映射和误差方法,属于结构性数值方案; +实施前应提交最小设计并取得确认。native 有相关代码只说明方案可调查,不说明当前 +EQ3 链已经可用或正确。 + +## 0.25 K 标准与实现正确性必须分开 + +旧的空间标准是相邻 reference 网格在固定传感器、固定时刻上的最大绝对差不超过 +0.25 K。保留该标准是合理且保守的:温控边界和 hotspot 风险以绝对 K 判断,最大值 +不会被大量低误差 mean 传感器稀释;它也符合预注册要求,不能因失败后再改阈值。 + +同时要承认它的含义有限:不同网格的 hotspot 是同一物理组件内各自最热 cell,cell ID +和中心位置可能变化。这个 maximum-of-component observable 对局部解析度敏感,不是 +同一点场值,也不能用于全局最大值 Richardson 阶估计。mean 是体积加权平均,通常更 +平滑,必须与 hotspot 分组报告,不能用较小 mean 误差覆盖 hotspot 失败。 + +完整 2→1 mm 同窗诊断进一步显示这种差别: + +| 100 s 分组 | 全组平均 time-weighted MAE | 最大 registered absolute error | +|---|---:|---:| +| mean(129 sensors) | 0.0253 K | 0.7976 K | +| hotspot(138 sensors) | 0.1043 K | 2.069 K | + +最坏 hotspot 是 `component:hbm3.base:hotspot`,15.0 s,2 mm 为 314.966 K、1 mm +为 317.035 K;最坏 mean 是 `component:hbm2.base:mean`,同为 15.0 s,差 +0.7976 K。两组都超过 0.25 K 最大差标准,只是 hotspot 更敏感。 + +静态审计已核对 geometry/material ownership、单位、半 cell 界面导热、Robin 边界、 +source weight、层/输出顺序和 sensor reduction,未发现可解释该差异的映射错误。这是 +**实现正确性证据**,不是空间收敛证明。反过来,即便以后 0.5 mm 达到 0.25 K,也只 +缩小离散误差;材料、接触、功率、边界、几何和目标器件参数的不确定性仍属于总体 +物理误差,不能由网格 PASS 消除。 + +## 具体决策点 + +| 选择 | 得到什么 | 代价/边界 | 建议 | +|---|---|---|---| +| A. 当前闭合 | 保留完整链路和诚实的 reference 未冻结状态 | 依赖绝对热点精度的物理结论继续受限 | **推荐**;最符合快速闭合,避免无信息的持续优化 | +| B. uniform 0.5 mm | 保持原相邻 uniform 方法 | 1,032,192 cells;需显式 guard 和生成/resource pilot;成本未知 | 已在同方法细化授权内,但当前信息价值不足,暂不启动 | +| C. native local mesh | 可能以较少 cells 增加热点附近解析度 | 需新的生产、映射、守恒与误差方法;不能回填旧 uniform PASS | 作为单独结构性数值方案,不在本轮直接实施 | + +因此当前报告的可执行结论是:**A,保持 0.25 K FAIL 并闭合其余系统**。B 是已授权 +范围内的可行后续,出现明确论文结论需求时可用一次有停止条件的 resource pilot +启动;C 改变网格方法和映射,仍需明确确认。两者都不应演变为无停止条件的自动细化。 + +## 证据位置 + +- R02:`eq3_thermal/plans/campaign-v1/points/R02/runs/R02/{DONE.json,stdout.log,generation_receipt.json}` +- R03:`eq3_thermal/plans/decision-execution-v2/points/R03-V2-FULL1MM/runs/R03-V2-FULL1MM/{DONE.json,stdout.log,generation_receipt.json}` +- 完整空间诊断:`eq3_thermal/plans/decision-execution-v2/points/r03-spatial-full-v2/spatial_diagnostic.json` +- exporter:`tools/eq3_layered_export.py` +- native:`eq3_thermal/reference/build/3d-ice-stock-e0bb685/{flex,sources,include,bin/example_transient_nonuniform.stk}` +- 先前静态审计:`docs/eq3_thermal/P2_4MM_2MM_STATIC_DIAGNOSTIC.md` diff --git a/docs/eq3_thermal/PARAMETER_GAPS.md b/docs/eq3_thermal/PARAMETER_GAPS.md index 8451af1..7215f4b 100644 --- a/docs/eq3_thermal/PARAMETER_GAPS.md +++ b/docs/eq3_thermal/PARAMETER_GAPS.md @@ -42,3 +42,19 @@ JEDEC官方HBM4页面此次未能取得;Samsung封装网页定向请求40s超 P3需要最小因果活动接口及能量去重,P4维护要有真实模拟资源、失败/在途语义、 完成后年龄更新、耗能与磨损反馈。不得通过改变核心ABI/PTX/TMA/future/cache来 掩盖这些缺口;不能只加一个刷新计数器。非热核心消融逐项NOT_IMPLEMENTED。 + +### 2026-09-20 HBF ECC conditional-proxy update + +USER_CONFIRMED: the device is **HBF**, not HBM; NAND/OCP/SanDisk conditional +proxies are permitted. Target HBF RBER, code strength, UECC probability and decoder +throughput remain unknown. The historical NOT_IMPLEMENTED entry above is retained +as history; an optional actual read-cost consumer now exists under +`experiments/eq3_system_thermal/ecc_proxy/README.md`. + +`run_causal_point` consumes temperature-history age and a source-constrained but +assumed retry-effort interpolant through shared media/decoder resources and +activity energy. Fixed integration tests passed; thermal acceptance is separate. +It predicts conditional successful-read effort, not error probability. Initial +wear is a scenario. Per-stack age is not yet linked to maintained tensor extents, +so combined refresh→ECC improvement remains UNSUPPORTED_COMPOSITION. HBM is excluded. +No instantaneous temperature penalty or compulsory positive throttling benefit. diff --git a/docs/eq3_thermal/Q4_8HBF_THERMAL_MODEL.md b/docs/eq3_thermal/Q4_8HBF_THERMAL_MODEL.md new file mode 100644 index 0000000..513e799 --- /dev/null +++ b/docs/eq3_thermal/Q4_8HBF_THERMAL_MODEL.md @@ -0,0 +1,74 @@ +# Q4 eight-HBF package thermal model + +## Scope and evidence + +The Q4 geometry is an engineering model for the user-confirmed +`EQ3-ISOLATED-MAINTENANCE-CAMPAIGN-v1`. It reuses the candidate package, +GPU, materials, boundaries, background fill, and existing HBF layer template. +Those preserved sections are `DOC_DERIVED`. Replacing the four HBM slots with +translated copies of the corresponding HBF templates is `USER_CONFIRMED`. + +The generated network is **not** a P2 model freeze and does not inherit the P2 +spatial-reference qualification. External GDDR retains a physical system +identity but has no package geometry and no package temperature in this model. + +## Conversion contract + +`tools/eq3_q4_all_hbf_profile.py` takes all paths explicitly and writes new +files only. It rejects an unexpected source inventory, a footprint/orientation +mismatch, an incomplete HBF template, or an existing output. + +| Removed slot | Complete template | New stack | Target XY (um) | Footprint (um) | +|---|---|---|---:|---:| +| `hbm0` | `hbf0` | `hbf4` | 16000,48000 | 12000,16000 | +| `hbm1` | `hbf1` | `hbf5` | 0,36000 | 16000,12000 | +| `hbm2` | `hbf2` | `hbf6` | 0,16000 | 16000,12000 | +| `hbm3` | `hbf3` | `hbf7` | 16000,0 | 12000,16000 | + +Each copied stack contains attach, one base, 16 bond layers, 16 array dies, and +TIM. The component identifiers and power groups change to `hbf4..7`; the +material, size, z coordinate, die index, and sensor contract come from its HBF +template. The original `hbf0..3` stacks remain unchanged. + +## Static validation + +The immutable point is +`eq3_thermal/plans/isolated-maintenance-campaign-v1/points/Q4-THERMAL-STATIC01`. +Its explicit input and output paths are in `manifest.json`, while +`ENGINEERING_MODEL_LOCK.json` binds the profile, power fixture, model, events, +grid, and sensor files by SHA-256. + +The fixed tests pass 5/5. The 2 mm export contains 40,960 nodes, 119,296 edges, +287 entities, 137 powered components, and 307 sensors. The 20 ms mapping fixture +assigns 1 W to every powered component: input energy is 2.74 J and emitted RC +event energy is 2.7400000000002223 J. All preserved profile sections compare +equal by canonical JSON. This fixture is a `SCENARIO_ASSUMPTION` for component +coverage and energy conservation; it is not a workload or calibrated power +trace. + +The static export starts no solver. The separate locked-input response point +`Q4-THERMAL-RESPONSE02` compares the persistent service with the existing +sparse campaign runner for one 20 ms, 0.02 J input to `hbf4.base`. It passes +all 307 sensors with a maximum absolute difference of +2.1032064978498966e-12 K. Campaign and service energy residuals are +2.0384052222373263e-15 J and 2.042923666554895e-15 J. The run used one CPU, +211200 KiB maximum RSS, and 11.24 s wall time; sparse factorization took +5.1775 s. + +`Q4-THERMAL-RESPONSE01` is retained as a pre-solver failure: its manifest +selected an older runner that rejected `--model-sha256`. RESPONSE02 changes +only the runner path to the already validated identity-aware build; model, +grid, sensors, step, and injected energy remain identical. + +## Consumer paths + +- Model: `generated_2mm/model.txt` +- Grid/component mapping: `generated_2mm/rc_grid.json` +- Sensors: `generated_2mm/rc_sensors.json` +- Fixture events: `generated_2mm/events.txt` +- Converted profile: `inputs/candidate_profile_q4_all_hbf.json` +- Fixture power: `inputs/q4_mapping_20ms_power.json` + +Q1, Q2, and Q3 continue to share the original mixed package model because those +topologies change the fabric graph rather than package geometry. Only Q4 uses +this all-HBF model. diff --git a/docs/eq3_thermal/R03_FULL_REFERENCE_DIAGNOSTIC.md b/docs/eq3_thermal/R03_FULL_REFERENCE_DIAGNOSTIC.md new file mode 100644 index 0000000..31c4954 --- /dev/null +++ b/docs/eq3_thermal/R03_FULL_REFERENCE_DIAGNOSTIC.md @@ -0,0 +1,82 @@ +# R03 完整 1 mm reference:只读后处理结果 + +日期:2026-09-20 +状态:`COMPLETE_DIAGNOSTIC_ONLY`;`REFERENCE_UNQUALIFIED`;blind 未读取。 + +## R03 完整性 + +`R03-V2-FULL1MM` 已完成 100 s 求解和观察:5000 个求解帧、0.1 s 观察间隔、 +1000 个时刻、275 个传感器。输入能量为 1365 J,温度范围 300–369.012 K, +相对能量残差为 `1.491468238728679e-6`。solve、observe 和 DONE 收据均成功。 +机器可读复核在 +`plans/decision-execution-v2/points/R03-V2-FULL1MM/read_only_review.json`。 + +R01/R02/R03 的 `normalized.json` 和 `events.txt` 哈希分别完全相同: + +```text +normalized 7a62356faef94e9d16391c39a1e6ccf81d284843c0d9c3ea0d9471d0d364ce06 +events b1674c3ef83b7373d98eb55317d8109d3e55b00bbb057eefc1cf7b6614cf980e +``` + +三者都是 100 s、1365 J、275 sensors、0.02 s solver step;参考网格依次为 +`16x16x63`、`32x32x63`、`64x64x63`。因此完整空间诊断是在同一输入、同一时间窗和 +同一观察量上比较,不再把 1 mm 前 4 s 与完整 100 s 混报。 + +## 完整同窗空间误差 + +下表的 time-weighted MAE 是先对每个 sensor 在各预注册窗口积分,再按窗口时长合并, +最后对 275 sensors 取均值。最大误差来自实际 CSV 观测行,并保留当时 hotspot cell。 +这些汇总用于诊断,不替换既有 0.25 K/1 K/2 K/5%/0.001 标准。 + +| 相邻网格 | 窗口 | sensor 平均 time-weighted MAE (K) | 最大绝对误差 (K) | +|---|---|---:|---:| +| 4→2 mm | 完整 0–100 s | 0.143305 | 3.253000 | +| 4→2 mm | 全部受激窗口,合计 34 s | 0.233694 | 3.253000 | +| 4→2 mm | 全部冷却窗口,合计 66 s | 0.096741 | 3.253000 | +| 2→1 mm | 完整 0–100 s | 0.064942 | 2.069000 | +| 2→1 mm | 全部受激窗口,合计 34 s | 0.108907 | 2.069000 | +| 2→1 mm | 全部冷却窗口,合计 66 s | 0.042293 | 2.069000 | + +4→2 mm 最坏点是 `stack:hbf3:hotspot`、31.0 s:312.007 K 对 315.260 K; +hotspot cell 从 `n60_3_11` 变为 `n60_7_23`。2→1 mm 最坏点是 +`component:hbm3.base:hotspot`、15.0 s:314.966 K 对 317.035 K;cell 从 +`n4_7_8` 变为 `n4_15_16`。最坏 cell 身份随网格改变,因此没有计算 Richardson +收敛阶。完整空间误差仍超过原有精度要求,P2 reference 继续不合格。 + +原始后处理点: +`plans/decision-execution-v2/points/r03-spatial-full-v2`。 + +## 旧 2 mm RC 对 R03 的回顾评分 + +候选是已有完整 100 s `F01-RC2MM-10MS`,没有重跑求解。它同样有 275 sensors、 +1000 个 0.1 s 观察时刻和 1365 J 输入;RC 能量相对残差为 +`2.1497271184372237e-11`,通过 0.001 能量标准。它是旧 2 mm/不同 solver-step 的 +工件,输入导出哈希不与 R03 完全相同,所以结果仅为 retrospective diagnostic。 + +- legacy v1:`NUMERICAL_FAIL`;22 sensors 失败,最大误差 2.045948 K,最坏 MAE + 0.156614 K,最坏 normalized MAE 0.002269。 +- v2:`NUMERICAL_FAIL_WITH_THRESHOLD_AMBIGUITY`;19250 个 sensor-window scores 中 + 427 个失败,其中 full window 有 19 sensors 失败。阈值歧义共 4023 项,单列而不 + 自动决定全模型通过或失败。 +- 最坏点是 `component:hbm3.base:hotspot`、15.0 s:R03 为 317.035 K,RC 为 + 314.989052 K,绝对误差 2.045948 K。 +- v2 的最大受激窗口 time-weighted MAE 为 1.550441 K + (`excitation-08`),对应限值 0.912050 K;最大冷却窗口 MAE 为 0.742257 K + (`cooling-09`),对应限值 0.712350 K。 + +首次 v2 调用保留为输入校验失败:旧 RC 与 R03 的浮点时间戳拼写不同。单边适配仍 +失败后,确认两端分别来自乘法与累加。最终只对派生副本把时间规范到十进制 0.1 s +网格;最大改动 `1.4211e-14 s`,两端各 275000 行的所有非时间字段哈希在适配前后 +保持一致。原始 CSV 未修改。适配器为 `tools/eq3_sensor_time_canonicalize.py`,收据在 +`r03-reference-time-canonical-v2` 与 `r03-rc2mm-time-canonical-v2` 点中。 + +v1、v2 结果分别在 `r03-rc2mm-v1-retro` 和 +`r03-rc2mm-v2-retro-common-grid`。两次输入接口失败也原样保存在 +`r03-rc2mm-v2-retro` 与 `r03-rc2mm-v2-retro-canonical`,没有删除或重写。 + +## 结论限制 + +完整 1 mm 轨迹填补了此前“仅前 4 s”的证据缺口,但没有消除空间误差。R03 仍是 +`REFERENCE_UNQUALIFIED`,不能据此 Model Freeze。旧 2 mm RC 对完整 R03 的 v1/v2 +均未通过;温度量化为 0.001 K,远小于本次最坏空间/RC 差异,但热点位置跨网格变化 +阻止用单个全局最大值推算收敛阶。 diff --git a/docs/eq3_thermal/RATE_THERMAL_CONTROL_ASSUMPTIONS.md b/docs/eq3_thermal/RATE_THERMAL_CONTROL_ASSUMPTIONS.md new file mode 100644 index 0000000..885728a --- /dev/null +++ b/docs/eq3_thermal/RATE_THERMAL_CONTROL_ASSUMPTIONS.md @@ -0,0 +1,134 @@ +# Rate-driven thermal-control assumptions + +Status: engineering interpretation contract for the `eq3_rate_thermal` +fluid-feedback path. This document describes the implementation at the current +source revision; it is not a product thermal limit specification or a model +freeze. + +## What the loop represents + +The new loop accepts synthetic model-weight read demand as integer bytes in +20 ms windows. A persistent per-channel FIFO serves those bytes under a +per-stack budget and the declared 96 GB/s/channel service ceiling. Only served +bytes produce incremental heat: 40 pJ/B is assigned to the mapped HBF array die +and 10 pJ/B to that stack's base. The unchanged full coupled RC network then +advances one window. Temperatures and completed-window fluid facts determine a +budget for the following window. No decision changes service or energy already +recorded in the current window. + +The service and delay observations are `MODELLED_FLUID`. They are not MQSim +commands, NAND completions, fabric completions, or native backend latency. The +capacity-utilization fact passed to the existing feedback policy is +`served_bytes / physical_window_capacity`; its saturation flag means the fluid +channel envelope was filled. The byte-weighted P95 is quantized to window ends. +The policy receives that P95, but the present rate-only profile has no latency +target, so it does not use the P95 as a pass/fail target. + +The workload intentionally permits offered demand above the 1.536 TB/s per +stack service ceiling. This encodes the user-selected sustained-pressure +question. Excess bytes remain in FIFO order instead of being discarded or +reported as achieved bandwidth. The frozen per-stack control target is the +integer value + +``` +min(mean active offered B/s per stack, 0.8 * 1.536e12 B/s). +``` + +The initial and maximum budget is one physical-capacity window +(30,720,000,000 B per 20 ms); the minimum is 0.10 of that value and the control +step is 0.05. A workload's final 10 s has zero new arrivals, but existing +backlog continues to receive service and therefore continues to generate read +energy. “Recovery” means the offered-load source has stopped. It does not +guarantee a pure passive-cooling interval or complete queue drainage. + +## Research thermal guard + +`ThermalService` derives a stack temperature from the hottest thermal entity +owned by that stack. The configured research thresholds are: + +| Entity | Light | Severe | Shutdown | +|---|---:|---:|---:| +| HBF/HBM | 353.15 K (80 °C) | 363.15 K (90 °C) | 378.15 K (105 °C) | +| GPU | 363.15 K (90 °C) | 373.15 K (100 °C) | 383.15 K (110 °C) | + +These thresholds are research guardrails. They do not certify safe operation, +throttling behavior, reliability, or shutdown behavior of a product. A GPU +guard state is propagated to package memory stacks when it is more restrictive +than their own state. State changes take effect after 20 ms. Recovery uses a +100 ms dwell and 2 K hysteresis; escalation is not blocked by recovery dwell. +The thermal domain still fails above 400 K rather than clamping the result. + +The three existing policies retain their original behavior: + +- `guard_only` applies Severe and Shutdown protection but has no ordinary or + Light rate limiter. It returns to the maximum budget after protection clears. +- `thermal_hysteresis_guard` applies the existing Light budget of 0.5 of the + baseline, and applies a zero budget for Severe or Shutdown. +- `read_rate_feedback_thermal_guard_v1` applies the same Light, Severe and + Shutdown protection, then evaluates completed-window rate facts. It holds a + target that is met within the configured 5% tolerance; delivery above the + target plus the 10% smoothing margin can reduce the budget by one step. When + demand and backlog remain, the gate limited service, and the modelled fluid + capacity is not saturated, it can increase by one step. It holds when the + demand is insufficient or the evidence does not support an increase. Its + existing rollback rule reduces a prior increase when delivery fails to + improve while backlog or quantized delay worsens. After an emergency zero + budget, Normal state resumes from the minimum and can step upward using the + explicit modelled fluid spare-capacity fact. + +The words “capacity utilization,” “saturation,” and “delay” in this policy path +refer to the fluid experiment. They must not be restated as native NAND or +fabric measurements. + +## Power and temperature interpretation + +The 50 pJ/B coefficient is a user-confirmed scenario assumption anchored as +40 pJ/B array plus 10 pJ/B base. It gives 76.8 W at the OCP Grade 2 effective +rate of 1.536 TB/s; the coefficient is not renormalized to force 80 W at that +rate. It is applied to served bytes once, with no energy assigned to queued +bytes. + +Idle HBF power, channel activation power, GPU self power, host ingress power, +and external-GDDR thermal input are unknown or omitted in this path. GPU power +is zero *incremental* input, not a claim that a physical GPU consumes zero +power. The initial 300 K is a common simulation reference. Reported values are +conditional incremental-heating results, not predicted absolute product +operating temperatures. Retry, ECC, retention age, maintenance and native +backend latency are also unavailable in this fluid path. The retained P2 +spatial error prevents treating these runs as a thermal model freeze. + +## Four-topology, three-axis coverage + +The axes are configuration identity, thermal behavior, and system behavior. +Evidence from the native CPU campaign and the new fluid campaign remains +separate. + +| Topology | Configuration axis | Thermal axis | System-behavior axis | +|---|---|---|---| +| Q1 mixed-direct | Existing mixed HBM/HBF configuration and channel-to-die scenario map | New fluid path supports the full mixed 2 mm coupled network and per-die/base served-byte sources | New rate/control study supports direct per-stack fluid FIFO service. This is modelled service, not native MQSim. Older native CPU receipts remain separate evidence. | +| Q2 4+4 relay | Existing relay configuration is retained | The mixed package network exists, but the new fluid path has no partner-endpoint energy or shared relay arbitration | Only the older native CPU path provides relay routing/admission evidence. New fluid results must not be reported as relay-system results. | +| Q3 four-pair DASH | Existing DASH configuration is retained | The mixed package network exists, but the new fluid path has no DASH shared-endpoint arbitration or partner-energy accounting | Only the older native CPU path provides DASH routing/admission evidence. New fluid results must not be reported as DASH-system results. | +| Q4 8-HBF direct plus external physical GDDR | Existing all-HBF package configuration is supported; GDDR remains physically external | New fluid path supports the full generated 8-HBF 2 mm coupled package network. External GDDR temperature is unavailable | New rate/control study supports direct per-stack fluid FIFO service. External-GDDR service and temperature are outside this loop; older native CPU evidence remains separate. | + +The older OCP4K CPU campaign snapshot records completed Q1–Q4 native points, +including relay and DASH modes, but is explicitly a partial campaign snapshot +and is not interchangeable with the new model-size fluid matrix. The new study +therefore closes the requested rate-to-temperature loop for Q1 and Q4 while +preserving the Q2/Q3 system-capability gap. The user's current question concerns +read-rate pressure and thermal throttling, so closing it does not require +rewriting MQSim arbitration or inventing relay/DASH fluid behavior. + +## Source anchors + +- `experiments/eq3_maintenance/thermal_client.py`: hotspot guard, thresholds, + action delay, dwell, hysteresis and GPU guard propagation. +- `experiments/eq3_maintenance/read_rate_policy.py`: the three budget policies, + target/tolerance logic, recovery and rollback behavior. +- `experiments/eq3_rate_thermal/fluid_service.py`: persistent FIFO cohorts, + integer capacity, max-min service, delay histograms and conservation. +- `experiments/eq3_rate_thermal/run_controlled.py`: completed-window causality, + modelled-fluid fact adapter, served-byte energy mapping and output labels. +- `experiments/eq3_rate_thermal/model_workloads.py`: official model-size metadata, + continuous/equal-mean burst offered demand and exact integer-byte generation. +- `eq3_thermal/plans/isolated-maintenance-campaign-v1/campaign-ocp4k-v3/stage/COMPLETED48_SNAPSHOT.json`: + retained partial native CPU campaign snapshot across Q1–Q4. diff --git a/docs/eq3_thermal/RATE_THERMAL_CONTROL_V1_RESULT.md b/docs/eq3_thermal/RATE_THERMAL_CONTROL_V1_RESULT.md new file mode 100644 index 0000000..e959f13 --- /dev/null +++ b/docs/eq3_thermal/RATE_THERMAL_CONTROL_V1_RESULT.md @@ -0,0 +1,270 @@ +# EQ3 rate-driven coupled thermal control v1 + +> **Status: COMPLETE_CONDITIONAL_SIMULATION.** +> +> All 41 indexed executions completed: two topology pilots and the strict +> 39-point main whitelist. Execution/accounting and all registered conservation +> checks passed. Results remain conditional fluid-service and incremental-heat +> simulations; they are not native-backend throughput or a thermal model freeze. + +## Scope and evidence boundary + +This stage answers the user-authorized question: how synthetic stored-weight +read pressure, modelled service budgets and thermal guards interact in the full +coupled package RC network. It uses an isolated, default-disconnected fluid +byte service. It does not issue MQSim commands and does not claim native NAND, +fabric or product throughput. + +The frozen stage inputs are: + +- Environment: `eq3-thermal-cpu-v1`, one CPU process, one OMP/BLAS thread, + 4 GiB address-space limit per point and no GPU use. +- Source revision: `cf22dadc0611a1a6d16d59eb8af6f5ec1b541a0a`. +- Thermal executable SHA-256: + `6e84ef0798070a74e468a137d80c68c3b2789c2643b3115493b17afe78febe25`. +- 41 indexed points: two 8 s active + 4 s recovery pilots and 39 main points + with 20 s active + 10 s recovery. +- Main design: two direct topologies, three model extents, continuous and + 200 ms-period/100 ms-on equal-mean burst demand, and three existing control + strategies at 16 full stored-weight scans/s. +- User-confirmed amendment: three additional all-HBF, 235B, continuous, + 32-scan/s points. These match the per-stack offered pressure of four HBFs at + 16 scans/s; they do not replace the original 36 points. +- Incremental read energy: 40 pJ/B array plus 10 pJ/B base. Idle, GPU + background and external-GDDR power remain unknown or omitted. +- Initial thermal reference: 300 K; declared domain 300–400 K; 20 ms thermal + and control windows. + +The source of record for this frozen design is +`eq3_thermal/plans/rate-thermal-control-v1/{PREFLIGHT.json,RUN_INDEX.json}`. +`REPORT_CONTRACT.md` was registered before main results and governs the final +comparisons. Main analysis used the `RUN_INDEX_PHASE_MAIN_WHITELIST` at analysis +commit `6a24d08`, which excluded both pilots and prevented directory discovery +from mixing diagnostic and main results. + +The workload uniformly stripes offered weight bytes across the declared stacks +and channels, using integer carry and a rotating remainder; cumulative channel +counts differ by at most one byte. This is an explicit workload assumption for +the rate-pressure study. It is not evidence that a real LLM stores or accesses +weights with that placement. + +## Actual interface closure + +| Interface | Actual producer | Actual consumer | Enablement | Current capability and gap | +|---|---|---|---|---| +| Model-sized offered-byte windows | `experiments/eq3_rate_thermal/model_workloads.py::build_workload` | `run_controlled.py`, then `FluidService.advance` | Explicit `--workload`; isolated runner only | Exact integer-byte, 20 ms continuous or equal-mean burst demand over 4/8 stacks × 16 channels. Offered demand may exceed capacity and is not achieved throughput. | +| Per-channel FIFO service and queue facts | `experiments/eq3_rate_thermal/fluid_service.py::FluidService` | `run_controlled.py` energy mapping and policy fact adapter | Constructed only by `run_controlled.py`; no default MQSim registration | Persistent backlog, channel capacity, stack budget, byte conservation and window-quantized byte-weighted delay. No command lifecycle, NAND timing, fabric delivery or endpoint-group arbitration. | +| Served bytes to component energy | `run_controlled.py::_served_energy` using the frozen profile channel map and 40/10 pJ/B | Existing `ThermalService.advance`, which sends component `ENERGY` before `ADVANCE` | Explicit `--profile`, `--model-dir` and `--thermal-binary` | Actual mapped die/base sources in the unchanged coupled network. Queued bytes consume no read energy. Idle/channel-activation/host-ingress/GPU/GDDR energy is unavailable. | +| Coupled temperature and guard facts | Existing persistent thermal service plus `experiments/eq3_maintenance/thermal_client.py` | `run_controlled.py` output and existing `ReadRatePolicy.evaluate` | Explicit isolated thermal binary and model paths | Full registered 2 mm package network, 20 ms causal advance, HBF 80/90/105 °C and GPU 90/100/110 °C research guards, 20 ms action delay, 100 ms recovery dwell and 2 K hysteresis. These are research limits, not product guarantees. | +| Future-window byte budget | Existing `experiments/eq3_maintenance/read_rate_policy.py::ReadRatePolicy` | Next `FluidService.advance` call | Explicit `--strategy` among the three frozen policies | Completed-window facts affect only the next window. Fluid utilization and saturation are explicitly modelled facts, not native backend observations. Existing policy code is reused unchanged. | +| Point orchestration and immutable receipts | `prepare_control_stage.py` and `launch_control_stage.py` | `run_controlled.py`, stage index and later derived reporting | Explicit stage invocation; output directory must be new | Input/source/model hashes, STARTED manifest, JSONL facts and DONE/FAILED receipts. It does not enable the path in default MQSim or HBFSim runs. | + +Every interface above has an actual runtime consumer in this isolated stage. +The new fluid path is therefore connected for Q1/Q4 direct thermal-control +experiments, while remaining default-off and separate from native MQSim. + +## Pilot release evidence + +The registered `PILOT_REVIEW.json` reports +`PILOT_REVIEW_PASS_MAIN_RELEASED`. These are engineering release checks, not +main-matrix conclusions: + +| Pilot | Final backlog | Peak coupled temperature | Served-to-thermal energy error | Wall time | Output | +|---|---:|---:|---:|---:|---:| +| mixed-direct, 235B, continuous, 16 scans/s, feedback | 19,685,794,447,360 B | 362.690441 K | 3.5470e-11 J | 35.675 s | 90,688,829 B | +| all-HBF direct, 235B, continuous, 16 scans/s, feedback | 0 B | 363.191949 K | 3.0468e-11 J | 21.518 s | 132,306,653 B | + +Both pilots had zero byte-conservation error. Backlog in the mixed pilot was +registered as expected saturation evidence rather than an execution failure. +The pilot review raised the evidence-based per-point output allowance and +projected the main wall time; it did not change the scientific inputs. The +pilots use shorter time windows and different package geometries, so they are +diagnostic release evidence and are not pooled with the main matrix. + +## Main completion and acceptance + +The stage-level execution completed **41/41 points with no failed point**. The +strict analysis completed **39/39 main points**, 39 policy-pair comparisons and +six registered cross-topology comparisons. All byte, delay-histogram and +50 pJ/B energy checks passed. The steps used for acceptance were: + +1. Freeze the completed main index and enumerate every COMPLETED, FAILED, + DOMAIN_FAILURE or resource-blocked point. Do not silently omit a strategy. +2. Verify source, profile, workload, scenario, model and executable hashes + against `RUN_INDEX.json` for every completed point. +3. Verify per-window and cumulative offered = served + backlog conservation and + served-byte energy = thermal input within the registered tolerance. +4. Preserve declared-domain failures as failures. Do not clamp, truncate or + widen 400 K. +5. Generate every aggregate table and plot solely from immutable point outputs; + no thermal solve may be repeated for reporting. + +Main completion receipt: `rate-thermal-control-v1/DONE.json` +Completed/failed counts: **41 completed / 0 failed**, including two pilots +Frozen RUN_INDEX SHA-256: +`e4cd532b875e3de09e5cba8521d644bd660a90bec026a7d617c5559ec422f0da` +Strict analysis receipt: `MAIN-ANALYSIS01/ANALYSIS_RECEIPT.json` + +## Required final measurements + +The final report must show values separately for active and full observation +windows. A zero-arrival recovery interval may still serve backlog and generate +heat, so it is not automatically a passive-cooling interval. + +The six equal-per-stack-pressure high-load rows are the most direct saturation +comparison. `Active rate` covers the fixed 20 s active interval; delivered, +backlog, delay, final temperature and energy cover the full 30 s observation. +TB uses decimal bytes. `Late CV/zero` covers the preregistered 10–20 s active +second half. + +| Topology / strategy | Active rate TB/s | Active delivered TB | Full delivered TB | Final backlog TB | Fluid P95/P99 s | Peak/final K | Energy J | Late CV / zero fraction | +|---|---:|---:|---:|---:|---:|---:|---:|---:| +| 4-stack guard only | 3.4775 | 69.550 | 101.069 | 49.391 | 15.56 / 17.38 | 365.519 / 361.914 | 5,053.440 | 0.690 / 0.238 | +| 4-stack thermal hysteresis | 3.2456 | 64.911 | 95.017 | 55.443 | 16.42 / 17.28 | 363.166 / 362.159 | 4,750.848 | 0.0866 / 0 | +| 4-stack rate feedback | 3.2052 | 64.103 | 93.853 | 56.607 | 16.48 / 17.56 | 363.162 / 361.507 | 4,692.634 | 0.125 / 0 | +| 8-stack guard only | 6.1686 | 123.372 | 178.176 | 122.744 | 17.14 / 17.96 | 365.551 / 363.307 | 8,908.800 | 1.119 / 0.556 | +| 8-stack thermal hysteresis | 6.0518 | 121.037 | 176.210 | 124.710 | 17.34 / 18.12 | 363.247 / 361.653 | 8,810.496 | 0.352 / 0.110 | +| 8-stack rate feedback | 5.8872 | 117.744 | 169.230 | 131.690 | 17.62 / 18.62 | 363.222 / 357.791 | 8,461.517 | 0.403 / 0.126 | + +“Delay” means byte-weighted, 20 ms-window-quantized fluid delay. Outstanding +bytes are censored and must be reported alongside the quantiles. Backend +latency remains `UNKNOWN`. + +## Frozen comparisons + +### Same offered demand: four versus eight HBF stacks at 16 scans/s + +Compare Q1 mixed-direct and Q4 all-HBF direct only when model, pattern, policy, +weight extent, total offered bytes and observation window match. Report service, +backlog, temperature, energy and delay together. + +At 235B continuous 16 scans/s, the eight-stack configuration halves offered +bytes per stack and completes all 150.460 TB under every policy; the four-stack +configuration remains backlogged by 49.391–56.607 TB. Relative to four stacks, +eight stacks deliver 49.391, 55.443 and 56.607 TB more for guard-only, +hysteresis and feedback, while peak temperature changes by only +0.037, ++0.083 and +0.065 K. This comparison simultaneously changes package geometry, +thermal network, source count and per-stack pressure. It cannot isolate a +causal “stack-count effect.” + +### Same per-stack saturated pressure + +Compare four-HBF 235B continuous 16 scans/s with eight-HBF 235B continuous +32 scans/s for each of the three policies. The amendment equalizes offered +pressure per HBF stack; it does not make the geometries or coupled neighbors +identical. + +All six rows remain saturated at 30 s. Eight stacks receive twice the total +offered bytes and may dissipate roughly twice the aggregate served-byte power; +equal per-stack offered pressure does not mean equal package power. Guard-only +delivers more and leaves less backlog, but its late-window service oscillation +and peak temperature are larger. In four stacks, feedback lowers peak only +0.0048 K relative to hysteresis while delivering 1.164 TB less. In eight +stacks, it lowers peak 0.0252 K while delivering 6.980 TB less and increases +late CV and zero-window fraction. No one policy dominates throughput, +temperature and stability. + +### Continuous versus equal-mean burst + +Paired workloads have exactly equal total offered bytes. Compare transient +peak/final temperature, served bytes, backlog, delay and second-half rate +stability. Do not infer a burst benefit from temperature alone. + +All 7B and 72B points deliver every offered byte and end with zero backlog. +Continuous input is identical across the three policies: aggregate active +rates are 0.243700 TB/s for 7B and 2.326599 TB/s for 72B. At 7B, bursts preserve +delivery and latency but raise peak by 0.2915 K in mixed and 0.1457 K in +all-HBF. At 72B, bursts raise guard-only/hysteresis peaks by 2.7831 K in mixed +and 1.3906 K in all-HBF. Feedback reduces those burst increments to 0.2421 K +and 0.1206 K, with a 100 ms increase in fluid P99. At 235B/16 scans/s, burst +and continuous pairs have identical final delivery/backlog; peak differences +are negligible (below 0.008 K in all-HBF), while P99 rises 20–60 ms. Persistent +saturation and budget control largely mask the original burst shape. + +### Guard and feedback behavior + +For every policy, report guard-state durations, zero-budget fraction, active +and second-half throughput stability, backlog and recovery service. Lower +temperature with less completed work or more censored backlog is a tradeoff, +not an unconditional improvement. + +The high-pressure table reports the preregistered late CV and zero-service +fractions. Guard-only trades higher delivery for sharper stop/restart behavior +and about 2.3 K higher peak than the smoother policies. Temperature reduction +coincides with lower completed work and greater censored backlog, so it is not +scored as an unconditional benefit. + +Uniform input does not guarantee uniform temperature. In mixed continuous 72B, +all stacks deliver exactly 11,632,992,583,680 B with no backlog, yet inner +stacks peak at 348.241 K and outer stacks at 341.801 K, a 6.440 K spread. The +stage-level `STACK_DISTRIBUTION_REVIEW.json` verifies zero cumulative offered- +byte imbalance across stacks for all 39 points. This lower-pressure temperature +spread therefore reflects the registered geometry and thermal coupling rather +than an input stripe imbalance. The analogous 7B +spread is 0.675 K. At mixed continuous 235B, peak spreads are 0.611 K for +guard-only, 8.055 K for hysteresis and 8.152 K for feedback; feedback per-stack +delivery ranges from 22.764 to 24.163 TB. The generated all-HBF geometry is +symmetric: even at 235B/32 scans/s its peak spread is below 1.5e-10 K. + +## Four-topology, three-axis status + +| Topology | Configuration | Thermal behavior | System behavior | +|---|---|---|---| +| Q1 mixed-direct | Registered mixed HBM/HBF configuration and frozen channel map | New fluid stage closes served-byte sources through the full mixed 2 mm coupled network | Direct per-stack FIFO/control behavior completed as modelled fluid. Native MQSim evidence remains a separate earlier result. | +| Q2 4+4 relay | Existing relay configuration remains available | Mixed package network exists, but this new fluid stage does not account for partner-endpoint energy | Shared relay endpoint arbitration is unavailable in fluid v1. Retain prior native CPU routing/admission evidence; do not relabel Q1 fluid results as Q2. | +| Q3 four-pair DASH | Existing DASH configuration remains available | Mixed package network exists, but this new fluid stage does not account for DASH partner sources | Shared DASH endpoint arbitration is unavailable in fluid v1. Retain prior native CPU routing/admission evidence; do not relabel Q1 fluid results as Q3. | +| Q4 8-HBF direct plus external physical GDDR | Registered all-HBF package model; physical GDDR remains external | New fluid stage closes served-byte sources through the full generated 8-HBF 2 mm coupled package network | Direct per-stack FIFO/control behavior completed as modelled fluid. External-GDDR service and temperature are unavailable. | + +This division is deliberate. The current research question is rate-driven +package heating and thermal throttling; it does not require rewriting MQSim or +inventing relay/DASH fluid arbitration. + +## Scientific limits that remain after completion + +- Model-size scans are synthetic full stored-weight scans, not token/s or an + inference execution trace. The 235B all-expert scan is not the 22B active MoE + footprint per token. +- Served rate is produced by the fluid capacity model. It is not measured + MQSim, NAND, link or fabric throughput, and 20 ms bins cannot resolve + command-level or sub-bin peaks. +- The 40/10 pJ/B envelope is a user-confirmed conditional assumption rather + than measured HBF energy. Unknown idle, GPU background, external GDDR and + host power prevent an absolute product-temperature or safety claim. +- Maintenance, ECC, retry, 24-hour retention and native backend latency are not + generated by this path. +- The retained P2 adjacent-grid evidence fails the original 0.25 K spatial + criterion, including a reported 2-to-1 mm maximum difference of 2.069 K. + The historical 400.911 K development-domain failure remains failed. Blind + data remains unopened and `MODEL_FREEZE=false`. + +## Execution identity, resources and derived artifacts + +The matrix ran for 2,305.48 s and retained 10,565,045,495 B. The largest point +used 350,312,325 B of output and 75.71 s wall time, within the evidence-adjusted +413,458,291 B point and 5,400 s stage limits. The maximum recorded peak was +365.563817 K at the all-HBF 235B/16-scan burst guard-only point. All points +remained inside the declared 300–400 K domain. + +Read-only analysis and documentation commits continued during the serial +matrix. Consequently, point manifests contain five Git HEAD values. The +stage-level `RUNTIME_IDENTITY_REVIEW.json` finds exactly one executed consumer +source-hash set, one thermal-binary hash and one environment ID across all 41 +points. Git HEAD differences therefore record analysis/document history and do +not indicate a runtime-algorithm change; `RUN_INDEX.json::source_locks` is the +execution identity authority. + +Small, reviewable commits relevant to this stage include: + +- `277e9d9`: isolated fluid service, model-sized workload and controlled runner. +- `cf22dad`: bounded input preparation and serial campaign launcher. +- `9705c65`, `adbf400`, `6a24d08`, `af53461`: derived analyzer, stability + reporting, strict RUN_INDEX whitelist and offered/temperature figures. +- `c6cecc9`: rate-thermal control assumptions and capability boundary. + +The final derived bundle is +`rate-thermal-control-v1/MAIN-ANALYSIS01/`. Its receipt reports 13 grouped +trajectory figures, 13 three-policy per-stack figures and one campaign summary +figure. `CONTROLLED_CAMPAIGN_ANALYSIS.json` contains all 39 point rows, policy +costs and cross-topology comparisons; `KEY_INTERPRETATION.md` is the concise +review. Analysis started zero thermal solves, so every figure is reproducible +from retained raw JSONL and the whitelisted analysis source. diff --git a/docs/eq3_thermal/RATE_THERMAL_INTERFACE_AUDIT.md b/docs/eq3_thermal/RATE_THERMAL_INTERFACE_AUDIT.md new file mode 100644 index 0000000..45bd357 --- /dev/null +++ b/docs/eq3_thermal/RATE_THERMAL_INTERFACE_AUDIT.md @@ -0,0 +1,251 @@ +# Per-stack read-rate to thermal-input interface audit + +Date: 2026-09-20 +Scope: read-only audit; no solver, MQSim, profile, controller, or campaign run was +started or changed. + +## Answer + +The existing complete package thermal network can already consume a different +time-varying heat input for every HBF stack without issuing NAND requests. The +supported boundary is **component energy per fixed thermal interval**, not +bytes/s. A small, default-disconnected producer can convert an externally +prescribed per-stack rate trace to array/base component energy and feed the +unchanged persistent thermal service. This preserves GPU, every HBM/HBF die, +every base die, and the shared package/cooling paths. + +The unresolved part is the physical conversion from HBF delivered bytes to +watts. Sandisk and OCP specify bandwidth targets but no HBF idle/full power or +read J/B. Therefore rate comparisons are immediately possible as normalized +per-watt responses or as an explicitly named engineering proxy. The user has +now explicitly selected the original 80 W envelope mapping for this conditional +simulation: at 1.6 TB/s, array=64 W and base=16 W, hence 40/10 pJ/B. Idle +increment remains `UNKNOWN_NOT_MODELLED`. This closes the engineering input; +it does not turn it into a product measurement. + +## Existing producer and consumer chain + +| Boundary | Existing symbol/file | Actual behavior | Read-rate use | +|---|---|---|---| +| Power grouping | `tools/eq3_layered_ir.py::_power_groups`, `_normalize_power` | Expands `hbfN.array` equally over its 16 powered die components and keeps `hbfN.base` separate; validates caps and source-to-component energy conservation | Reuse unchanged to expand each stack's array/base watts | +| Offline source generation | `tools/eq3_all_source_cap.py::build` | Converts grouped watts to volume-weighted node energy with a receipt | Reusable if a rate trace is first expressed as grouped power intervals | +| Window replay conversion | `experiments/eq3_maintenance/pilot_energy_replay.py::convert` | Converts contiguous `WINDOW_TOTAL` component joules to native RC events without consuming activity rows twice | Reusable for offline runner comparison | +| Persistent client | `experiments/eq3_maintenance/thermal_client.py::ThermalService.advance` | Accepts `{component_id: energy_j}` for one contiguous 20 ms interval and sends `ENERGY` then `ADVANCE` | Direct consumer for an open-loop rate trace | +| Persistent service | `experiments/eq3_maintenance/thermal/thermal_service.cpp::energy`, `advance_one` | Rejects unknown/duplicate/non-aligned input; computes `power=energy/duration`; distributes it across the component by cell volume; advances the existing sparse RC factorization | No solver or model change is needed | +| Output | persistent service `ADVANCE` response | Returns all entity mean/hotspot temperatures, all registered sensors, and window/cumulative energy conservation | Suitable for per-stack temperature curves and coupling comparisons | + +The service protocol is already sufficient: + +```text +ENERGY START_NS END_NS COMPONENT_ID ENERGY_J +ADVANCE END_NS +``` + +For one rate window of duration `dt`, a source-only adapter would calculate, +for each stack `s`, + +```text +P_array,s = P_array,idle,s + rate_s * e_array +P_base,s = P_base,idle,s + rate_s * e_base +E_die,s,i = P_array,s * dt / 16 (i = 0..15) +E_base,s = P_base,s * dt +``` + +and submit `hbfS.die0..15` plus `hbfS.base`. The 8HBF model uses the same +component convention for `hbf0..hbf7`; the mixed model uses `hbf0..hbf3` while +retaining HBM and GPU as passive coupled components unless they receive their +own explicit sources. Rates must be averaged over the 20 ms thermal interval; +sub-window bursts are integrated energy and cannot produce a resolved +sub-20-ms temperature peak. + +Adding a new `RATE` command to the thermal process is unnecessary. Keeping +rate-to-energy outside the solver makes the uncertain power law versioned and +replaceable while preserving the existing solver protocol and factorization. + +## Available power evidence and its limits + +| Candidate | Values and current consumer | Status for rate-to-heat | +|---|---|---| +| HBF product power | Sandisk Gen1 target is 1.6 TB/s per 16-die stack; OCP v0.7.0 grades are 0.384/1.536/3.072 TB/s. Neither source gives HBF idle watts, full-load watts, or J/B. | `UNKNOWN_BLOCKING` for absolute product temperature; bandwidth alone is not heat | +| Original research excitation envelope | `configs/eq3_thermal/research/calibration_power.json` declares 64 W array plus 16 W base per memory stack. `eq3_layered_ir` consumes these as separate groups; it is synthetic calibration input, not a product value or measured 80/20 partition. | Usable as an explicit `SCENARIO_ASSUMPTION` | +| D3 input domain | `tools/eq3_domain_v2_input.py` applies the frozen common `alpha=0.25`; the resulting per-stack caps are 16 W array plus 4 W base. D3 remains conditional and reference-unqualified. | Inputs within 20 W/stack stay in the executed D3 power domain; this does not confer product calibration | +| HBM3E aggregate proxy | Registered public data derives 45.815--46.941 pJ per delivered byte from aggregate memory-domain `(loaded-idle)/bandwidth`. It has no per-stack samples and no HBF array/base/PHY split. | May be a clearly labelled transfer proxy; must be counted once, not once per stage | +| Later native activity proxy | `ActivityEnergyLedger` uses 0.05 W per active NAND die, 0.01 W during command/data-out intervals, and 2 pJ/B for a separate fabric endpoint. | Cannot convert an arbitrary prescribed bytes/s trace by itself. It requires actual busy intervals and is not a single HBF J/B model | +| HBF idle/full power | No registered HBF value. The HBM3E dataset's idle value is an aggregate platform memory-domain measurement, not HBF idle power. | Keep `UNKNOWN`; zero incremental idle means “not modelled,” not measured zero | + +OCP's 16 banks/channel example does not establish a NAND plane identity or a +power partition. DASH explicitly states that public HBF subarray parallelism +is unavailable and uses its own assumptions (16 dies, 32 planes/die, four +independently accessible subarrays/plane, 3 us read). Those assumptions may +motivate a separate scenario, but they do not turn OCP bank into MQSim plane or +provide HBF J/B. See OCP v0.7.0 Table 3/4 and +[DASH Sections II-B, IV-B and VI-A](https://arxiv.org/html/2608.14333v1). + +## Immediately usable comparison modes + +### 1. Normalized per-watt response + +This requires no HBF energy assumption. Drive `hbfN.array` and `hbfN.base` +independently with a unit source, record temperature rise per applied watt, and +report the response against normalized rate `u_s(t)=rate_s(t)/R_ref`. Results +are transfer responses of the assumed RC package, for example K/W at a sensor +or the temperature trace produced by a declared unit-power waveform. They do +not carry an absolute HBF junction-temperature claim. + +The network is linear for fixed material/boundary parameters, so component +mean temperature increments can be combined from source responses. Hotspot +`max` is a nonlinear reduction and must be recomputed from the combined field +or directly advanced through the existing service; hotspot maxima must not be +added. The completed A1 audit already found full-network entity-mean +superposition consistent for its fixed source replay while retaining all +cross-stack cooling paths. + +### 2. Existing 1.6-TB/s research-envelope scenario + +A minimal complete engineering mapping is + +```text +P_array = 64 W * rate / 1.6e12 B/s +P_base = 16 W * rate / 1.6e12 B/s +``` + +This is equivalent to 40 pJ/B assigned to the array and 10 pJ/B assigned to +the base. These coefficients are derived from the synthetic 80 W envelope, +not measured HBF values. This mapping is `USER_CONFIRMED` for the present +conditional simulation. The resulting prescribed powers are: + +| Per-stack rate | Array W | Base W | Total W | Domain note | +|---:|---:|---:|---:|---| +| 0.384 TB/s | 15.36 | 3.84 | 19.20 | Inside the executed D3 alpha=0.25 caps | +| 1.536 TB/s | 61.44 | 15.36 | 76.80 | Inside original 64/16 envelope, outside D3 alpha=0.25 domain | +| 1.600 TB/s | 64.00 | 16.00 | 80.00 | Original-envelope endpoint, outside D3 alpha=0.25 domain | +| 3.072 TB/s | 122.88 | 30.72 | 153.60 | Outside both original and D3 domains | + +The 3.072-TB/s point must not be clipped to 80 W. It needs either a new +explicit power-domain/scientific approval or normalized-only reporting. +Likewise, D3 accuracy/status cannot be inherited by the 1.536/1.6-TB/s points +merely because the solver accepts their joules. + +### 3. Aggregate HBM3E transfer proxy + +Applying the registered 45.815--46.941 pJ/B range once gives the following +dynamic powers, before any unknown HBF idle power: + +| Per-stack rate | Proxy dynamic W | +|---:|---:| +| 0.384 TB/s | 17.59--18.03 | +| 1.536 TB/s | 70.37--72.10 | +| 1.600 TB/s | 73.30--75.11 | +| 3.072 TB/s | 140.74--144.20 | + +This is useful as a scale comparison with the 80 W research envelope. It is +not an HBF range. To send it into the spatial model, a separate declared +array/base allocation is still required; the same aggregate energy cannot be +added to array, base, PHY, and fabric independently. + +## Retained MQSim feasibility boundary + +The current isolated MQSim profile is not needed for a prescribed-rate thermal +study. Its existing limitation should nevertheless remain visible so a +thermal input is not described as achieved backend throughput: + +- `16 channels * 8 bits * 1600 MT/s = 25.6 GB/s` nominal native bus per stack; +- `256 planes * 4096 B / 10 us = 104.8576 GB/s` is only a loose array + concurrency arithmetic bound. It is not executable independent-plane + throughput because MQSim groups same-die/same-page multiplane commands and + serializes shared channel transfers; +- global QD256 provides only 64 outstanding requests/stack for a balanced + four-stack trace; the retained warm read delivered about 11.96 GB/s/stack; +- the existing 512 GB/s completion envelope is shared by the whole engine, + not a per-stack HBF link. + +Changing only queue depth or channel width cannot establish 0.384, 1.536, 1.6, +or 3.072 TB/s/stack with the current 4-KiB/10-us resource model. A faithful +achieved-throughput model would require separately approved internal bank/ +subarray parallelism, pipeline and per-stack completion-resource semantics. +That structural work is unnecessary for the user's stated rate-to-temperature +question. + +## Minimal implementation and approval boundary + +The smallest implementation is a standalone, default-off trace converter: + +1. input contiguous 20 ms windows with explicit `rate_Bps` for every HBF stack; +2. require a named power model (`NORMALIZED_PER_W`, + `RESEARCH_64_16_AT_1P6TBPS`, or a separately approved proxy), its evidence, + idle status, and array/base allocation; +3. expand group power to real component joules using existing group weights; +4. write a source ledger containing rate, coefficient, source group, energy, + input hashes and conservation totals; +5. feed the unchanged `ThermalService.advance` or existing offline replay path. + +This needs no thermal equation, event ordering, checkpoint, public core ABI, or +backend change. It must fail on missing stack rates, unknown coefficients, +out-of-domain power, non-contiguous windows, or energy mismatch. It must not +silently infer idle power, saturate a high-rate point, or add the same energy to +multiple stages. + +A new absolute HBF energy law, an expanded accepted power domain, a live +rate-feedback controller, or a product-bandwidth backend changes the scientific +input or request/control lifecycle and requires its own versioned preflight and +user confirmation. None is required to produce normalized coupled-package +rate sensitivity with the interfaces already present. + +## Frozen names and model identities for the immediate pilot + +Use the following explicit profile identity in the rate ledger: + +```text +profile_id: EQ3_HBF_RATE_TO_POWER_80W_AT_1P6TBPS_V1 +evidence: USER_CONFIRMED_SCENARIO_ASSUMPTION +rate_semantics: PRESCRIBED_DELIVERED_BYTES_PER_S_NOT_BACKEND_ACHIEVED +array_j_per_byte: 4.0e-11 +base_j_per_byte: 1.0e-11 +idle_power: UNKNOWN_NOT_MODELLED +array_spatial_mapping: EQUAL_OVER_16_THERMAL_DIES_PER_STACK +thermal_window_ns: 20000000 +``` + +For a window `[t0,t1)`, emit `rate_Bps * 4e-11 * dt / 16` joules to every +`hbfN.dieI` and `rate_Bps * 1e-11 * dt` joules to `hbfN.base`. Name the source +groups `hbfN.array` and `hbfN.base`; these match the existing power-group +registry and avoid inventing a PHY component. Record the group total before +expansion and require exact component-sum conservation within floating-point +tolerance. + +For the smallest same-network pilot, use the existing mixed Q1--Q3 model: + +```text +/root/hbfsim-exp/eq3_thermal/generated/campaign-RC2MM-train +``` + +It contains 64,512 cells, 255 entities and 275 sensors. The locked SHA-256 +values are `ecc7d24ad7a107466eaa7e6cbd37add087a57677aa32283e8850fa10c6c62559` +(`model.txt`), `b8e611d0a9c0db325efdade111fb83c89135434197f55a3f3924f9285cce5a23` +(`rc_grid.json`), `fe700161c57d6d66da4eb0dfadf25d72732a24a6fe9b5b0d79c770c934dcd73b` +(`rc_sensors.json`) and +`f61365623ddca5e55ae07fcb2b0599e2302f483810c090c327b6c1b57f394280` +(`normalized.json`). It preserves four HBF stacks, four HBM +stacks, GPU, package and shared cooling, so different `hbf0..hbf3` rates expose +both self-heating and coupling. + +The separate eight-HBF model is at +`eq3_thermal/plans/isolated-maintenance-campaign-v1/points/Q4-THERMAL-STATIC01/generated_2mm` +(40,960 cells, 287 entities, 307 sensors). It should be used only when the +question specifically requires eight HBF stacks; it is a distinct generated +geometry and does not inherit P2 qualification. + +Linearity applies because the locked model holds heat capacities, +conductances, material properties, Robin boundaries and 300 K reference fixed, +and the service solves a constant sparse linear system with one reused +factorization. It has no temperature-dependent material law or radiation. +The 300--400 K domain check, hotspot `max` reduction, and any external thermal +control decisions are not linear superposition operators; the service must +still evaluate the combined input and preserve a domain failure without +clamping. + +## Implemented OCP Grade2 channel envelope + +Latest user requested source-aligned channel counts and rates. The active profile uses16channels/stack,64-bit×16GT/s×75%=96GB/s effective/channel and1.536TB/s total (OCPv0.7.0 Tables2/4). This replaces an unrun equal100GB/s/channel Sandisk aggregate inference. The user-confirmed50pJ/B remains, so1.536TB/s maps to76.8W. channel_map and channel_capacity_Bps are explicit; segment channel_read_Bps drives actual chosen die heat sources. Excess per-channel and aggregate prescribed rates are rejected, not silently capped. No actual UCIe/NAND scheduling is claimed. + +Two bounded3.2s OCPGrade2 thermal points have completed on unchanged mixed2mm full package; artifacts are eq3_thermal/plans/rate-thermal-v1/OCP-G2-UNIFORM01 and OCP-G2-CONCENTRATED01. Each184.32J; latter changes only hbf0's1.4..2.2s .384TB/s from16×24GB/s to4×96GB/s. Old unrun rate plans retained. Derived analysis reports matched-time differences instead of mistaking unchanged earlier global peak for absence of spatial effects. diff --git a/docs/eq3_thermal/RATE_THERMAL_V1_RESULT.md b/docs/eq3_thermal/RATE_THERMAL_V1_RESULT.md new file mode 100644 index 0000000..a75494a --- /dev/null +++ b/docs/eq3_thermal/RATE_THERMAL_V1_RESULT.md @@ -0,0 +1,44 @@ +# 读取速率驱动的逐栈热模拟 v1 + +已实现并实际运行,无需生成NAND读写事务。采用用户确认的50pJ/B增量能量模型,保留完整耦合热网络。当前接口对齐OCP v0.7.0 Grade2:每栈16通道,每通道有效上限96GB/s,每栈合计1.536TB/s;原80W@1.6TB/s能量锚点不重标,因此Grade2满速为76.8W。 + +## 输入、源与温度的粒度 + +| 层次 | 当前实际能力 | 限制 | +|---|---|---| +| 负载 | 每stack、每channel的分段规定有效读取率;整数ns半开时段 | 不是MQSim实测吞吐,无请求排队、重试或协议仲裁 | +| 容量约束 | 每通道96GB/s、每栈合计1.536TB/s;超限报错 | 不是静默截断或重新解释为延迟 | +| 热源 | 按显式channel→die映射累加40pJ/B到实际die;10pJ/B到本栈base一次 | 一通道/一die是独立工程映射;未获得plane/page物理位置 | +| 温度求解 | 既有2mm面内分辨率、逐层几何、64512cell/255entity/275sensor,20ms步长 | 原P2空间精度限制保留;小于20ms峰值不保证解析 | +| 未访问die | 增量读取热源为零,但保留全部热耦合 | 不等于器件真实idle功耗为零 | +| 输出 | 各die/base的均温和热点、每stack热点、GPU/HBM被动受热、能量与冷却曲线 | 原400K模型域失败不截断,不升级MODEL_FREEZE | + +系数array40pJ/B、base10pJ/B属于USER_CONFIRMED_SCENARIO_ASSUMPTION,不是OCP或Sandisk实测能耗。300K是模型参考点,未叠加真实idle/GPU背景时报告增量温升;不能用该参考温度加增量宣称产品实际工作结温。base系数整体覆盖本方案的base活动,不另加一次PHY能量。 + +| 活跃通道数(每通道满额) | 合计有效读率 | 条件读取增量功耗 | +|---|---:|---:| +| 1 | 96GB/s | 4.8W | +| 4 | 384GB/s | 19.2W | +| 8 | 768GB/s | 38.4W | +| 16 | 1536GB/s | 76.8W | + +来源:[OCP v0.7.0,第16页表2/4](https://www.opencompute.org/documents/ocp-hbf-architecture-specification-v0-7-0-final-pdf)。64bit×16GT/s÷8×75%=96GB/s。此前100GB/s/channel是从SanDisk总带宽均分的未运行推算,已由此有明确来源的配置替代;旧预检保留,不当作运行结果。 + +## 已完成的匹配对照 + +两个3.2s输入,160个热窗口。hbf0依次为.384/.768/1.536TB/s,再回到.384;hbf1固定.384、hbf2固定.768,hbf3无读取,最后1s共同冷却。两次仅在1.4..2.2s改变hbf0的空间分布:16通道×24GB/s,或4通道×96GB/s。其余功率、时间、初态、求解器与网络一致。 + +每次输入184.32J,服务输入对账最大误差1.42e-13J,最终热平衡残差约2.80e-9J。耗时16.91/16.74s,子进程峰值约336MiB,输出约14MB/点。没有MQSim、实际I/O、GPU负载或全局环境修改。转换器6项固定测试通过,覆盖分段积分、实体发现、能量分配、通道容量和拒绝非法输入。 + +均匀分配实例hbf0最高增量43.456K;没有读取的hbf3最高增量10.355K,GPU无额外自身功率却增温约12.000K,说明共享热网络确实消费了新分源输入。它们属于本次有限时长/空间分布和能量假设的结果,不能当各读取率的稳态温度表。 + +两臂全程hbf0峰值相同,因为峰值发生在分布改变之前;必须比较1.4..2.2s同一时刻各die和base,而不是只看全局峰值。最终对照显示:1.4s前全部entity/sensor逐帧差为0K;改变分布期间集中减均匀的hbf0热点差为+0.086733..+0.118583K,die/base局部差可正可负(−0.047440..+0.119242K)。这是本离散热网络的成对结果,不可提升为已确认0.1K产品预测精度。最终对照和图见工件`PAIR-ANALYSIS01`。 + +## 交付与复现 + +- 新转换器与runner:`experiments/eq3_rate_thermal/`。默认无消费者自动启用,需要显式profile/schedule/model/binary路径。 +- 接口→生产者→消费者:schedule逐通道速率 → `build_windows`窗口分源能量 → 既有`ThermalService.advance` → 不变的稀疏RC服务 → 逐die/base/stack温度与图。 +- 实际预检、输入、运行索引、原始帧、能量、完成收据:外层`eq3_thermal/plans/rate-thermal-v1/`,包括`PREFLIGHT_OCP_V3.json`、`RUN_INDEX.json`及两个OCP-G2点。 +- 旧MQSim实验按用户目标转向暂停:57点完成、3旧失败保留、9点未执行;`isolated-maintenance-campaign-v1/campaign-ocp4k-v4/SCOPE_PIVOT_SUMMARY.json`。 + +未实现也未宣称:真实HBF绝对功耗标定、通道静态开销、每plane/page热点、真实ECC/读重试、Grade3满速153.6W的扩大功率域。当前Grade2方案已可直接更换各栈速率与die分布输入,保持已确认的物理/能量域和完整热网络。 diff --git a/docs/eq3_thermal/READ_BANDWIDTH_ALIGNMENT_STATUS.md b/docs/eq3_thermal/READ_BANDWIDTH_ALIGNMENT_STATUS.md new file mode 100644 index 0000000..f5f003e --- /dev/null +++ b/docs/eq3_thermal/READ_BANDWIDTH_ALIGNMENT_STATUS.md @@ -0,0 +1,31 @@ +# Current read-bandwidth alignment audit + +Status: DOC_DERIVED from current production consumers of the isolated experiment, 2026-09-20. No running experiment parameters changed. + +The current profile is an engineering composition, not a complete Sandisk or OCP speed-grade implementation. `campaign_inputs.py::configuration` supplies the following values: + +| Layer | Actual parameter | Consumer | Qualification | +|---|---|---|---| +| HBF fabric fill/direct/relay stage | 1,600,000,000,000 B/s, each configured stage;10ns startup | BasicFabric stage duration | Sandisk Gen1 target magnitude; not calibrated sustained NAND delivery | +| Native NAND bus | 16 channels/stack,8bits/channel,1600MT/s | isolated online engine sets Flash_Channel_Width/Transfer_Rate, native NVDDR2 PHY | Engineering NAND proxy, not OCP UCIe host bus | +| Native page read | 10,000ns | profile defaults each Page_Read_Latency field, native NAND | Engineering assumption, not target HBF calibration | +| Whole-engine completion envelope | 512,000,000,000 B/s shared across all HBF stacks | dispatch_to_device callback bandwidth_cursor_ns | Existing engineering aggregate cap; not per stack. Raw/reported-completion equality composition guard remains active | +| Active workload | model payload /20,000s synthetic scan period, finite bursts | actual request generator, future byte gate | Sparse engineering replay, not a saturation test or token trace | + +Sandisk July2025 fact sheet gives Gen1 1.6TB/s per16-die stack,512GB decimal. OCP v0.7.0 Table4 gives speed grades0.384/1.536/3.072TB/s. Table2 derives3072GB/s from16 host channels×8B×32GT/s×75% interface efficiency. Its page15 prose uses TiB/s while these tables use decimal GB/s/TB/s; table calculation is the explicit normalization basis. These are distinct profiles;1.6TB/s is not silently interchangeable with1.536TB/s. + +Nominal internal configured bus capacity is16×1B×1600MT/s=25.6GB/s perstack before command/service/arbitration overhead. This is a calculated engineering interface ceiling, not measured throughput. The 1.6TB/s fabric value cannot create bytes that the native NAND proxy does not supply. Page/bank geometry alignment therefore does not establish bandwidth alignment. + +Existing rolling-write diagnostic includes an actual subsequent read interval:16,384×4096 bytes delivered by native MQSim over1,402,880ns, about47.836GB/s aggregate over4 stacks. This is one finite warm mapped read trace, not a proven maximum, full-capacity OCP qualification, fabric-delivered throughput or a new run. Raw: `points/STARTUP-WRITE-ROLLING-W256-N16384-01/raw/summary.json`. + +The thermal campaign may establish interface, sharing and control behavior under the recorded engineering assumptions. It cannot establish product bandwidth, target bandwidth utilization, or thermal steady-state at advertised sustained throughput. No operation duration, width, queue, power or timing is silently changed to force a product-rate match. Selecting a complete speed-grade proxy requires distinct source/version, native saturation and scaling evidence, causal energy accounting, and a separately registered comparison to these preserved engineering results. + +Sources: +- https://documents.sandisk.com/content/dam/asset-library/en_us/assets/public/sandisk/collateral/company/Sandisk-HBF-Fact-Sheet.pdf +- https://www.opencompute.org/documents/ocp-hbf-architecture-specification-v0-7-0-final-pdf (Table2/Table4,p16) + +## Fixed-parameter feasibility bound + +For4KiB requests and10us media latency, Little's-law minimum simultaneous page service is ceil(B×10us/4096):938/3750/3907/7500 for .384/1.536/1.6/3.072TB/s perstack. This optimistic lower bound excludes command, transfer and arbitration overhead and does not claim queue entries equal independent media slots. Current queue256 is shared across the engine. Even an optimistic256 truly independent plane operations perstack gives104.8576GB/s at10us; the actual MQSim die/multiplane rule may impose tighter limits. Increasing queue or fabric bandwidth alone cannot establish these targets. + +User requested feasibility on2026-09-20. No new speed-grade profile or internal-parallelism mapping is approved or applied by this audit. A target-rate abstraction must expose source assumptions, physical-to-service mapping and energy consequences; numeric rate matching alone is insufficient evidence. diff --git a/docs/eq3_thermal/READ_RATE_CONTROL_SOURCE_BASIS.md b/docs/eq3_thermal/READ_RATE_CONTROL_SOURCE_BASIS.md new file mode 100644 index 0000000..86ab946 --- /dev/null +++ b/docs/eq3_thermal/READ_RATE_CONTROL_SOURCE_BASIS.md @@ -0,0 +1,134 @@ +# 稳定读速控制的资料依据与最小参数边界 + +日期:2026-09-20 +状态:**证据与窄方案;未修改代码、输入或实验状态** + +## 结论 + +现有权威资料足以支持两件事:HBF 应保留 OCP 定义的分档热保护;读速控制应观察 +真实完成率、端到端时延、积压和错误/重试事实,而不能仅凭瞬时温度推算读速。 +资料不足以拟合通用的 `temperature -> read bandwidth` 曲线,也不足以把当前 D4 的 +26/27/29 °C 工程阈值或固定 20 ms 读取解释成产品参数。 + +候选实现是一个**默认关闭、位于提交 gate 外层的离散反馈策略**。它尚未实施,以完成字节率为 +主目标,以温度模式作硬约束;只调尚未提交请求的准入额度,不改后端已经发生的 +完成时间,不追加假想 ECC 延迟。该方案不是 PID/MPC,也不要求改 MQSim 调度。 + +证据标签沿用项目约定:**SPECIFIED** 是目标规范直接事实;**PROXY** 是非目标器件 +实测;**SCENARIO_ASSUMPTION** 是可做敏感性分析的工程选择;**UNKNOWN_BLOCKING** +表示在获得产品资料或实际后端事件前不能形成定量器件结论。 + +## 可采用的字段与证据 + +| 拟用字段 | 原始值、单位与条件 | 证据 | 允许的用途 | 不允许的转用 | +|---|---|---|---|---| +| `junction_temperature_c` | 0–105 °C,HBF junction operating range | **SPECIFIED**,[OCP HBF v0.7.0 §9.1, p106](https://www.opencompute.org/documents/ocp-hbf-architecture-specification-v0-7-0-final-pdf) | 输入域检查、遥测 | 105 °C 不是公开的 LTT/STT,也不能当正常控制目标 | +| `rtt_c/ltt_c/stt_c` | MMIO `0x150`,编码为 `60 °C + value*0.5 °C`,约 60–124 °C;值为 implementation-specific | **SPECIFIED**,OCP §5 register table、§9.2 | 有真实寄存器值时直接使用 | 编码范围不是推荐阈值范围;缺寄存器时保持 UNKNOWN | +| `thermal_mode` | Normal 全性能;Light 可自动降低时钟且响应可能变慢;Severe 反压新命令、先完成在途命令,表注只支持维护;Shutdown 报 link error 并关闭 | **SPECIFIED**,OCP §9.2 pp106–108 | 热保护状态与 gate 上限 | 规范没有给出各档的固定带宽比例或延迟倍率 | +| `cecc/uecc/retry_opcode` | CECC/UECC 可要求 block refresh;host-driven retry 时主机按返回 opcode 重发同一读 | **SPECIFIED**,OCP §11.5.1 pp119–120 | 后端实际报告时记账、触发支持的维护/重试 | 不能从温度猜 CECC/UECC、重试次数或重试耗时 | +| `refresh_due` | 周期维护通常 24–48 h,实际 interval、read-count threshold 均 product-specific;同 die read/refresh 不应并发 | **SPECIFIED**,OCP §11.5 p118 | 产品 profile 有值时作期限和冲突约束 | 24–48 h 不是通用保证,也没有给出维护服务时间 | +| `advertised_read_Bps` | 第一代 Sandisk HBF 目标 1.6 TB/s;16 dies、256 Gb/die、512 GB decimal | 厂商资料,[Sandisk HBF Fact Sheet, July 2025](https://documents.sandisk.com/content/dam/asset-library/en_us/assets/public/sandisk/collateral/company/Sandisk-HBF-Fact-Sheet.pdf) | 拓扑/量级检查及明确标注的产品目标 | 不能当持续完成率、每栈 SLO、热稳态实测或 gate 默认目标;资料脚注包含内部测试/模拟 | +| `retry_count` 与 `retry_latency_ns` | 160 颗 48-layer 3D TLC、超过 1100 万页;`t_retry=N_retry*(t_R+t_DMA+t_ECC)`。fresh page 可无 retry;2K P/E、1 年 retention 平均 19.9 次 retry、平均 read latency 21 倍 | **PROXY**,[Park et al., ASPLOS 2021](https://people.inf.ethz.ch/~omutlu/pub/Reducing-SSD-Read-Latency-by-Optimizing-Read-Retry_asplos21.pdf) | 证明 retry 必须作为独立可观测开销;构造显式标注的 SSD 敏感性点 | 非目标 HBF,不能把次数、倍数、ECC 72 errors/KiB 或论文时序写入 HBF 默认值 | +| `pec/retention_age/operating_temperature` | 同一论文中 0 P/E、6 个月仍有 54.4% 读取至少 7 次 retry;温度测试为 30/55/85 °C。其样品中温度影响小于 P/E 与年龄,且 30/55 °C 的最终 retry 错误反而比 85 °C 多 5/3 bits | **PROXY**,Park et al. Fig.5、Fig.7 | 说明单调“越热 retry 越多”并不成立;要求分开记录历史状态 | 不能据此宣称降低 HBF 温度必然减少 retry 或提高读速 | +| `temperature_history` | 一家厂商 30–40-layer 3D charge-trap MLC;20–70 °C、1k–10k P/E 拟合;`Ea=1.04 eV`,95% CI 1.01–1.08 eV,`R²=0.76`;温度每秒记录并分段累积 acceleration factor | **PROXY**,[Luo et al., HeatWatch, HPCA 2018](https://www.cs.cmu.edu/~yixinluo/index_files/heatwatch_hpca18.pdf) | 历史累积器的数据结构与机制敏感性;可用 1.01/1.04/1.08 eV 作明确标注的代理敏感性 | 不是 Sandisk HBF 标定;不得外推成 85–105 °C 的绝对 RBER、寿命或 retry 分布 | +| `thermal_step` | Samsung PM963 达阈值后逐级降低性能,并在温度未下降时继续下一档 | 厂商 SSD 方法资料,[Samsung PM963 brochure](https://image.semiconductor.samsung.com/content/samsung/p6/semiconductor/newsroom/tech-blog/samsung-ssd-pm963-brochure/Samsung_PM963-1.pdf) | 佐证离散分档和重新测量的方法 | NAND SSD 产品方法不是 HBF 阈值、档位比例或控制周期 | + +Sandisk 的 “no refresh power” 与 OCP 的产品特定周期维护应继续作为不同 profile; +不能在同一结果里同时把二者当成同一器件的已知事实。 + +## 建议的最小配置契约 + +下面只定义控制器需要消费的事实。带数值的工程建议均为 +**SCENARIO_ASSUMPTION**,不能进入产品参数冻结。 + +| 字段 | 最小规则 | +|---|---| +| `enabled` | 默认 `false`;启用必须显式选择 `read_rate_gate_v1` | +| `target_completed_read_Bps` | 必填的 workload/SLO 输入;不得默认取 1.6 TB/s。没有来源时为 UNKNOWN,策略不激活 | +| `target_latency_p95_ns` | 只有要声明时延稳定时才必填;必须含 gate 等待、后端服务、retry 和传输 | +| `window_ns` | 建议 fixture 初值 1 s、敏感性范围 0.5–2 s;这是工程窗口。HeatWatch 的 1 s 是可靠性采样实现,不能当控制周期标定 | +| `throughput_tolerance` | 建议 0.05,敏感性 0.02–0.10 | +| `bad_windows/good_windows` | 建议 2/3 个连续窗口,避免单窗口抖动;均为工程假设 | +| `quota_step_fraction` | 建议每次 0.10,敏感性 0.05–0.20;限制在 `[0,1]` | +| `rtt_c/ltt_c/stt_c` | 优先从实际产品寄存器/配置读取并保留来源;缺失时 UNKNOWN,不采用 D4 fixture 阈值 | +| `hard_max_c` | 产品明确值优先;仅有 OCP 时可把 105 °C 用作工作域越界保护/失败记录,不能冒充 STT | +| `observations` | 每 stack、每窗口记录 offered/admitted/completed bytes 与 requests、外部 queue depth/age、gate wait、backend busy/idle、inflight、资源或 link 利用率、backend service、transfer、retry count/time、CECC/UECC、温度及模式;后端不提供的字段保持 UNKNOWN | +| `history` | 若可靠性模型启用,再记录 program temperature、分段温度驻留、retention age、P/E、read count;缺项保持 UNKNOWN | + +若需要不改产品模型的故障注入,可使用离散 retry **敏感性集合** +`{0, 1, 4, 8, 20}`,其中 20 仅是把论文的 19.9 四舍五入后的上端代理点;它不是概率 +分布。每次 retry 的 HBF 时延和能量仍必须来自实际后端事件或显式 HBF 参数,不能套用 +SSD 数值。 + +## 外部 gate 的简单控制规则 + +每个窗口先固定统计分母:`offered` 是到达 gate 的真实需求,`completed` 是窗口内 +对外完成,跨窗在途请求按真实完成窗口计。控制器不能通过少接收请求来缩小 offered。 + +定义 `target_met` 同时满足: + +1. `offered_Bps >= target_completed_read_Bps`,即窗口确实有足够需求可检验目标; +2. `completed_Bps >= target * (1 - tolerance)`; +3. 若配置了时延 SLO,端到端 p95 不超过它; +4. queue oldest age 和 backlog 没有持续增加; +5. 没有 UECC 或设备失败。 + +若 offered 不足,结果是 `INSUFFICIENT_DEMAND`,不是稳定或失败。若 completed 不足, +结果是 `UNMET_TARGET`,即使温度降低也不能报成收益。 + +策略每个窗口至多移动一个 quota 档,并先区分瓶颈证据: + +- 若未达标,同时 gate 确实在限制准入、外部积压存在,且后端有观测到的空闲能力, + 才可提高一个 quota 档。不能仅因处于 Normal 或出现坏窗口就提高额度。 +- 若后端保持忙碌、内部队列年龄增长、资源利用已饱和或 retry 开销增长,则保持 + `UNMET_TARGET` 并报告瓶颈;提高 quota 会加重拥塞。只有实际观测显示过量在途并发 + 导致服务或尾延迟恶化时,才可降低在途上限作诊断性保护。 +- 若缺少 backend occupancy、内部队列或 retry 事实,瓶颈原因是 UNKNOWN;保持当前 + quota 并报告,既不盲目增加准入,也不凭温度猜测 retry。 +- 达标且存在过量交付、突发波动或无必要的外部排队时,可在任何非紧急温度模式下 + 尝试降低一个 quota 档以平滑交付;不必等待进入 Light。下一窗口若破坏目标则回退。 +- 达标且稳定时保持当前 quota。恢复到 RTT 以下只解除热保护,不要求无条件把 quota + 提高到 1。 +- Light 仍施加产品定义或情景配置的热保护上限,但普通速率反馈在 Normal 同样工作; + 温度是保护与风险信号,不是普通性能反馈的唯一门槛。 +- Severe/CATTRIP:服从 OCP 反压语义;在途完成或返回后端规定错误,维护按规范能力处理。 +- Shutdown/link error:停止新提交并报告终态;不得自动改写成降速成功。 +- CECC、UECC、retry:只消费后端实际事件。CECC 可排入后端支持的 block refresh; + UECC/retry 依照实际能力处理。没有真实能力时返回 `UNSUPPORTED/UNKNOWN`。 + +硬温度保护始终高于性能反馈。性能 gate 只影响尚未提交的工作,等待期间仍由现有 +事件推进处理在途完成、冷却、控制恢复和维护。回调只记录事实,不重入调度。 +来源缺失会阻塞目标器件的温度—ECC—读速量化结论,但不阻塞按上述 UNKNOWN/hold +语义实现和测试这个默认关闭的工程 gate。 + +## HBF 可转用范围与阻塞缺口 + +可以直接转用到 HBF 的只有 OCP 模式、寄存器编码、在途处理、CECC/UECC/retry +协议和维护冲突语义。Sandisk 的 1.6 TB/s 是厂商目标量级;SSD 论文的数据只用于 +证明 retry 是大且依赖历史的延迟项,以及设计敏感性测试。 + +以下仍是 **UNKNOWN_BLOCKING**: + +- 目标 HBF 的实际 RTT/LTT/STT、传感器误差和报告延迟; +- 目标 HBF 的 ECC 强度、corrected-bit 粒度、retry opcode 表、次数分布; +- 每次 retry 的 sensing、ECC、base/PHY/传输时延与能量,以及能否重叠; +- 温度历史、retention、P/E/read-disturb 到 RBER/retry 的目标器件关系; +- block refresh 的服务时间、能量、资源和目标产品 read-count threshold; +- workload 的最低持续完成率与尾延迟 SLO。 + +在这些项关闭前,可声称的结果限于“外部门控在工程 fixture 中是否同时满足既定 +完成率、时延、积压和 OCP 热保护约束”。不能声称已得到 Sandisk HBF 的温度-ECC- +读速曲线、寿命收益或产品最优温度。 + +## 后续验证契约 + +实现前先以固定软件测试验证窗口边界、checkpoint、同时间戳顺序和 quota 单步变化; +再用成对 CPU fixture 报告 offered/admitted/completed、排队、后端 occupancy/service、 +retry、温度与能量。成对对照必须使用相同 offered 到达序列、相同统计窗口和相同初态, +并同时报告未完成工作,不能让 gate 改变需求分母来自己证明收益。验收必须覆盖: +gate 限额且后端空闲时小步提高、后端拥塞时禁止盲增、occupancy UNKNOWN 时 hold、 +Normal 下达标平滑、稳定达标保持,以及高温降低但完成率失败并保持 `UNMET_TARGET` +的负例。只有实际后端提供 retry/ECC 事件后,才可把相应分账从 UNKNOWN 升级为 +OBSERVED。 + +本文件不批准新实验,不修改现有 D4 六点/九点结论,也不把代理参数写入当前输入。 diff --git a/docs/eq3_thermal/SOURCE_AND_GAP_LEDGER.md b/docs/eq3_thermal/SOURCE_AND_GAP_LEDGER.md index 78cbafc..d91bf93 100644 --- a/docs/eq3_thermal/SOURCE_AND_GAP_LEDGER.md +++ b/docs/eq3_thermal/SOURCE_AND_GAP_LEDGER.md @@ -54,3 +54,14 @@ Primary links: [MFIT](https://github.com/AlishKanani/MFIT), [H200](https://www.nvidia.com/en-us/data-center/h200/), [Sandisk fact sheet](https://documents.sandisk.com/content/dam/asset-library/en_us/assets/public/sandisk/collateral/company/Sandisk-HBF-Fact-Sheet.pdf), [Micron HBM4](https://www.micron.com/products/memory/hbm/hbm4). + +### HBF read-cost proxy source addition (2026-09-20) + +Park et al., *Reducing Solid-State Drive Read Latency by Optimizing Read-Retry*, +ASPLOS2021, https://arxiv.org/html/2104.09611 (primary paper reviewed2026-09-20). +Actual old48-layerTLC retry data constrains an explicitly assumed age/wear +interpolant; it is not HBF calibration. HeatWatchEa1.04 retains the original +cross-device proxy, with its temperature fit/extrapolation distinguished. OCP070 +§5.3.2 supports base-managed retry; §9 distinguishes HBF NAND endurance from HBM. +No official numerical ECC latency/J/byte curve was inferred. Consumer, parameters, +source discrepancy and transfer factors are documented in the isolated ECC README. diff --git a/docs/eq3_thermal/STARTUP_WEIGHT_WRITE_INTERFACE_PROPOSAL.md b/docs/eq3_thermal/STARTUP_WEIGHT_WRITE_INTERFACE_PROPOSAL.md new file mode 100644 index 0000000..34a91bc --- /dev/null +++ b/docs/eq3_thermal/STARTUP_WEIGHT_WRITE_INTERFACE_PROPOSAL.md @@ -0,0 +1,240 @@ +# Additive startup weight-write interface proposal + +Date: 2026-09-20 +Status: design only; no implementation or experiment authorization inferred +Scope: isolated `experiments/eq3_maintenance` service only; default and the +active 66-point matrix remain unchanged. + +## Evidence and conclusion + +**USER_CONFIRMED:** a future experiment may consider real startup model-weight +writes and a larger weight model. The current instruction is bounded design +work only. + +**DOC_DERIVED:** the lower-level isolated engine already supports real +foreground writes. `MqsimOnlineEngine::submit` maps +`RequestOperation::Write` to MQSim `UserRequestType::WRITE` in +`experiments/eq3_maintenance/backend/src/mqsim_online_maintenance.cpp`. The +fixed maintenance backend test uses that path for a scheduled foreground write +and observes the mapping-generation race. Thus no new NAND write simulator is +needed. + +The missing capability is above the engine: + +- `hbf_mqsim_maintenance_service.cpp` rejects any `submit` or `try_submit` + operation other than `read` and always passes operation 0; +- `scripts/eval/mqsim_service.py::MqsimService` has no write-stage contract; +- `closed_loop.py::ClosedLoopCoordinator` accepts only HBF reads and hard-codes + `operation="read"` at backend admission; +- the run starts observation at time zero, so there is no persistent + `STARTUP_UPLOAD -> OBSERVATION -> DRAIN` lifecycle. + +This is an experimental service/lifecycle capability gap, not a NAND backend +bug. + +## Minimal interface + +Keep all current behavior byte-for-byte when the new option is off. Add an +explicit service option such as `--startup-writes on`, default `off`. With the +option on, expose two additive JSON-lines commands: + +```json +{"command":"startup_write","requests":[{"request_id":9000001, + "stack":"hbf0","stack_local_page":0,"bytes":4096, + "issue_ns":0,"operation":"write"}]} +{"command":"seal_startup"} +``` + +`startup_write` must reuse the existing stack map, placement ledger, request +identity, `MqsimOnlineEngine::submit`, completion, native-command observation, +and finish conservation. It accepts exactly page-aligned writes and assigns +operation `RequestOperation::Write`. It must not call the maintenance API, +reset age, fabricate payload hashes, or bypass the existing FTL/TSU/PHY. + +`seal_startup` succeeds only after all accepted startup writes have returned. +It permanently closes the write command for that process. Existing `submit` +and `try_submit` remain read-only, so a malformed observation request cannot +become a write. The opt-in header should explicitly report +`startup_weight_write=ACTUAL_MQSIM_FOREGROUND_PROGRAM_METADATA_PAYLOAD_UNAVAILABLE`; +the default header remains the current `READ_ONLY_MEDIA_SERVICE_NOT_HARDWARE` +contract. + +The Python client adds `startup_write(batch)`, uses ordinary `until` calls to +receive every completion, and adds `seal_startup()`. It records phase, request +IDs, bytes, placement, native program commands, first/last completion, and +conservation in a separate startup receipt. A unique numeric ID namespace is +shared across startup, reads, and maintenance. + +## Ownership, time, and energy + +One service process and one MQSim engine must own startup writes, later reads, +and maintenance. A second process would lose the mappings and would not test +resident data. + +The combined absolute timeline is: + +`STARTUP_UPLOAD -> optional STARTUP_COOLDOWN -> OBSERVATION -> DRAIN`. + +Observation arrivals and maintenance due/deadline values are shifted by the +sealed startup boundary. The MQSim clock, thermal clock, energy ledger, and +fabric clock must never be reset. If the scientific question requires an +ambient observation start, cooling is an explicit causal interval; temperature +must not be overwritten with 300 K after upload. + +Startup writes do not use the GPU read-output fabric. The missing host ingress +link remains `UNAVAILABLE` and is not assigned zero energy. Actual native NAND +facts already flow through `ActivityEnergyLedger.native`: operation type 1 uses +`nand_media_w["1"]`, and command/data-in activity is assigned once to the +stack base. Those coefficients remain `SCENARIO_ASSUMPTION`. Program activity +must be labelled `MODEL_WEIGHT_UPLOAD`, either through an optional external-ID +source registry in the ledger or an equivalent phase map; it must not be +labelled refresh or maintenance. + +Files/symbols in the smallest implementation are: + +| File | Local change | +|---|---| +| `experiments/eq3_maintenance/backend/service/hbf_mqsim_maintenance_service.cpp` | default-off option, `startup_write`, `seal_startup`, operation 1, phase conservation | +| `experiments/eq3_maintenance/backend/client/maintenance_service.py` | opt-in client methods and startup receipt validation | +| `experiments/eq3_maintenance/energy_ledger.py` | optional request-ID source classification; no new energy coefficient | +| new `experiments/eq3_maintenance/startup_upload.py` | bounded batch producer/drainer and phase receipt | +| `experiments/eq3_maintenance/run_point.py` | only in a new campaign version: run upload before observation and bind the shifted absolute times | +| `experiments/eq3_maintenance/closed_loop.py` | accept a nonzero causal start boundary; normal HBF observation requests remain reads | +| fixed service/client/coordinator tests | default-off identity, write/read/maintenance ordering, mapping, energy source and conservation | + +No production MQSim public ABI, default binary, thermal solver, fabric +arbitration, or maintenance algorithm needs to change. + +## Required invariants and fixed tests + +1. With the option off, write commands are rejected and existing read-only + transcripts/behavior remain identical. +2. An opted-in 4 KiB startup write emits one foreground program lifecycle with + actual stack/channel/die/plane facts and one completion. +3. A later read of the same logical page uses the same authoritative mapping; + a later maintenance operation can commit exactly once. +4. `seal_startup` rejects pending writes, duplicate IDs, writes after sealing, + wrong operation, out-of-range pages, and partial batches before mutation. +5. Startup NAND energy is derived only from native intervals, is assigned once, + is separated from observation and maintenance, and leaves host-ingress + energy `UNAVAILABLE`. +6. Absolute time is monotonic across all phases. Cooling and thermal windows + continue through the boundary; no state or temperature reset is allowed. +7. Process finish proves request/byte/phase conservation and zero pending work. + +Freshly programmed pages are not retention-aged pages. An immediate +post-upload maintenance check may validate the interface under an explicit +`ENGINEERING_POST_UPLOAD_MAINTENANCE` trigger, but it cannot support a +one-day-retention claim. A retention experiment needs a declared dwell/age +model or an approved aged-state import mechanism. + +## Capacity and bounded-pilot recommendation + +The registered official Safetensors payload sizes imply: + +| model | payload bytes | 4 KiB pages | pages/stack, four-stack stripe | +|---|---:|---:|---:| +| Qwen2.5-7B-Instruct | 15,231,233,024 | 3,718,563 | 929,640 or 929,641 | +| Qwen2.5-72B-Instruct | 145,412,407,296 | 35,501,076 | 8,875,269 | + +These are logical payload extents. MQSim stores no payload buffer (`Data` is +null) and exposes no payload hash, so even a full-page population run can claim +only logical mapping and native program-command coverage, not that actual model +tensor bytes were loaded or preserved. + +Do not start with a full-model upload. Use two bounded checks after the +interface/lifecycle version is explicitly approved: + +1. **Backend interface check:** 64 pages, one maintenance target per current + target identity. Write, drain, seal, read, then submit maintenance. This + verifies the exact cause of the v3 unmapped failures. Expected resources are + dominated by the existing full-capacity engine initialization (the observed + geometry probe used about 21.5 GiB RSS and 12.2 s); request artifacts should + remain below 1 MiB. Use the established 48 GiB address limit and 600 s + watchdog. +2. **Integrated bounded load:** 16,384 pages (64 MiB), striped across all + configured stacks/units in bounded batches, followed by reads and the + maintenance check through the complete thermal loop. At the observed + transcript density this is roughly 50--55 MB of protocol evidence. Stop if + request conservation, mapping, native energy, or thermal-window causality + fails. This is an engineering load pilot, not a full-model-residency claim. + +The ideal media-only lower bounds from 100 microseconds/program and 1,024 +configured units are about 0.36 s for 7B and 3.47 s for 72B, but they exclude +all scheduling, transfer, JSON round trips, native observations, allocation, +and thermal work and must not be reported as expected runtime. The current +protocol returns one completion per `until` call. A full 7B upload therefore +requires at least 3.72 million completion exchanges and, using the existing +full-capacity probe's transcript density only as a storage estimate, roughly +12 GB of transcript; 72B requires 35.5 million exchanges and roughly 116 GB. +Neither is a reasonable 600 s pilot. A future bulk-completion protocol would +be a separate interface change and must preserve individual command identities. + +A larger model does not itself sustain a higher read load. At fixed request +rate it changes address extent and coverage only. Sustained activity requires +an independently declared request-rate/scan schedule and must be analyzed as a +new workload input, not inferred from parameter count. + +## Approval boundary + +The concept of startup writes is user-confirmed, but this concrete implementation +changes the experimental protocol, request lifecycle, phase timing, thermal +initial condition, and result schema. Under the workspace refactor gate it +requires explicit approval of this interface/lifecycle version before edits. +The 64-page and 16,384-page runs also require their own saved preflight/manifest +under the experiment gate. The active frozen 66-point campaign must remain on +its current binary and source hash. + +## Smaller independent diagnostic alternative + +The service/coordinator lifecycle extension above is not required to answer the +first engineering question: whether real foreground programs can populate the +same mappings later read and maintained. An isolated fixed runner can use the +already public experimental engine API without changing any existing service, +request lifecycle, scheduler, ABI, or controller: + +1. construct one `MqsimOnlineEngine` from the selected profile; +2. install the existing native command-observation sink; +3. map a bounded page list with the existing `MqsimStackMapAdapter`; +4. submit real `HbfRequest` writes in bounded batches and drain every + completion before advancing to reads; +5. submit reads of the same logical pages and drain them; +6. optionally submit the existing one-page maintenance requests and drain them; +7. emit an immutable JSON receipt containing request/completion conservation, + native program/read phases, physical placement, and maintenance terminal + facts. + +The existing `run_mqsim_trace` function already accepts historical trace +operation 0=write and 1=read, but it submits the whole trace then returns only +final completions. It does not enable request observations, install the native +command sink, apply the explicit HBF stack map, emit energy facts, or retain a +maintenance follow-on. The existing `hbf_concurrent_trace_timing` executable +does retain request arrival/admission/media-complete facts for both reads and +writes, but likewise lacks native physical command facts, explicit stack-map +verification, and maintenance. It is useful as a zero-change A/B reference, +not as the complete startup-write evidence producer. + +The smallest complete implementation is therefore a new executable under +`experiments/eq3_maintenance/backend/` linked to the existing isolated adapter. +It calls existing methods only and is absent from all default lookup paths. +Its output can be consumed by a standalone Python diagnostic that feeds native +events to the unchanged `ActivityEnergyLedger`, flushes fixed windows, and +advances the unchanged persistent thermal service. This is explicitly +`BACKEND_FIXED_TRACE_PLUS_OPEN_LOOP_THERMAL_REPLAY`; temperature does not feed +back into write admission and it is not a main-controller integration result. + +For the 64-page check, use the frozen OCP4K full-capacity profile and the exact +current maintenance target pages to maximize identity comparability. For the +16,384-page heat/load check, either retain that profile or use a separately +declared finite namespace with the same 4 KiB, channel/die/plane and block +geometry; the latter lowers initialization cost but cannot inherit the +full-capacity identity claim. Both runs retain the earlier 48 GiB/600 s +engineering bounds and require saved manifests. + +This independent caller is a new, default-disconnected test tool rather than a +request/event-lifecycle refactor. Given the explicit user authorization for a +basic startup-write load test, implementing the caller and its fixed tests does +not require an additional refactor decision. Launching either numerical point +still requires its concrete preflight and resource/identity checks. Integrating +startup upload into `MaintenanceMqsimService`, `ClosedLoopCoordinator`, policy +timing, or the main campaign remains the separate approval-gated proposal above. diff --git a/docs/eq3_thermal/STARTUP_WRITE_FIXED_DIAGNOSTIC_RESULT.md b/docs/eq3_thermal/STARTUP_WRITE_FIXED_DIAGNOSTIC_RESULT.md new file mode 100644 index 0000000..2dc174d --- /dev/null +++ b/docs/eq3_thermal/STARTUP_WRITE_FIXED_DIAGNOSTIC_RESULT.md @@ -0,0 +1,207 @@ +# Startup write fixed diagnostic result + +Date: 2026-09-20 +Scope: authorized default-disconnected MQSim caller; no service, coordinator, +scheduler, ABI, thermal solver, or frozen campaign consumer changed. + +## Implemented path + +`experiments/eq3_maintenance/startup/startup_write_probe.cpp` uses the existing +isolated `MqsimOnlineEngine` APIs to submit bounded foreground write windows, +drain them, read the same logical pages, and optionally exercise maintenance. +It installs the existing native command observer and retains exact request, +stack, channel, die, plane, block, page, phase and completion identities. + +The caller emits chronological `native-events.jsonl` for the unchanged +`ActivityEnergyLedger`, equivalent CSV, request observations, maintenance +facts, phase boundaries, conservation summary and create-only `DONE.json`. +Request IDs partition write, verification read and software-lifecycle +maintenance. No host write is renamed refresh. Model payload bytes and host +ingress energy remain `UNAVAILABLE`. + +## Fixed lifecycle check + +`STARTUP-WRITE-FIXED64-01` passed on the 128 MiB four-stack engineering fixture: + +- 64/64 foreground programs uniquely completed; +- 64/64 same-page reads uniquely completed; +- all 64 written physical pages were distinct; +- 64/64 maintenance operations committed after the pages were mapped; +- 384 request observations and 1,024 native command-event records were + conserved; +- peak RSS was 7,588 KiB and internal wall time was 0.0045 s. + +The maintenance trigger is explicitly +`SOFTWARE_LIFECYCLE_TEST_ON_FRESHLY_PROGRAMMED_PAGES_NOT_RETENTION_DUE`. +Fresh age begins at each program completion; this result does not model a +24-hour-aged weight page. + +## Actual concurrency comparison + +The preserved initial paired points use identical 16 GiB finite logical geometry: four stacks, +16 channels/stack, one native MQSim die/channel, 16 planes/die, 16 physical +blocks/plane, 4 KiB pages, unchanged queue depth 256 and unchanged MQSim +TSU/PHY arbitration. Both issue 16,384 programs then 16,384 same-page reads. + +| Result | window 1 | window 256 | +|---|---:|---:| +| Completed write/read operations | 16,384 / 16,384 | 16,384 / 16,384 | +| Peak device outstanding | 1 | 256 | +| Program operations / native commands | 16,384 / 16,384 | 16,384 / 8,192 | +| Maximum overlapping program commands | 1 | 64 | +| Maximum active channel/chip/die resources | 1 | 64 | +| Maximum active channel/chip/die/plane resources | 1 | 192 | +| Distinct program physical resources | 1,024 | 1,024 | +| Simulated write interval | 1,644,314,624 ns | 12,950,656 ns | +| Achieved simulated program rate | 40.81 MB/s | 5.182 GB/s | +| Process wall time | 5.940 s | 5.826 s | +| Peak RSS | 252,168 KiB | 251,100 KiB | +| Retained raw evidence | 59,223,621 B | 53,934,814 B | + +The 127-fold simulated-rate change verifies that the bounded window reaches +real parallel resource arbitration. It is not produced by removing a lock or +changing TSU policy. Program *operations* count unique native transaction IDs; +program *commands* count command IDs, which may group transactions. Media +overlap uses half-open `MEDIA_BEGIN..MEDIA_END` intervals, with endings before +starts at an equal timestamp. + +The initial window-256 producer used a 256-request batch barrier. The final +standalone caller now implements `BOUNDED_ROLLING_REFILL_ON_EACH_COMPLETION` +with an explicit pending-ID to record-index map. A new 16,384-page rolling +point preserved all operation counts and produced the same simulated write +interval, peak 64 program commands, and physical-resource utilization as the +batch result. This uniform workload completes the last command of each batch +at the same timestamp at which the next batch was admitted, so the older batch +barrier happened not to introduce media-idle gaps. The native event ordering +and hashes differ, and the rolling implementation now satisfies the producer +lifecycle independently of this workload-specific equality. + +Across the complete rolling write interval, including initial fill and final +drain, the time-weighted mean was 63.255 active channel/chip/die resources out +of 64 (98.837%). Mean active program commands were also 63.255, peak device +outstanding was 256, and mean active channel/chip/die/plane resources were +126.511. The older batch-window point has the same utilization values. A +64-page rolling-window-8 check separately reached 8 outstanding requests, 8 +overlapping program commands, and a time-weighted mean of 7.971 active die +resources, confirming actual refill and final-drain accounting on a small case. + +The caller also accepts `--verify-pages`. It defaults to all loaded pages; a +smaller value selects evenly stratified page indices including the first and +last loaded page and records loaded versus verified counts. This changes only +read verification coverage, never the number of actual program operations. + +The initial `STARTUP-WRITE-CONCURRENCY-W1-01` is preserved as `INPUT_FAILURE`: +its eight-blocks/plane profile violated the existing MQSim write-fixture safety +contract requiring more than ten blocks/plane. The paired `-02` points use 16 +blocks/plane; no scheduler or solver was changed. The unstarted paired eight- +block point is marked `NOT_STARTED`. + +## Measured cost and model-size limit + +A same-geometry 64-page rolling baseline used 214,160 KiB RSS, 214,881 B +raw evidence and 0.164 s. Comparing it with the 16,384-page rolling result +gives an observed aggregate growth of approximately: + +- 2.224 KiB RSS per loaded-and-verified page; +- 3,293 raw-evidence bytes per page; +- 0.303 ms wall time per page. + +This aggregate includes caller request/placement records, retained native +events, emitted CSV plus JSONL, and backend mapping state. It is not attributed +entirely to MQSim history. Window 1 and 256 have nearly identical RSS at the +same request count, so outstanding queue occupancy is not the main retained- +memory term. + +Straight-line diagnostic projections from these two sizes are only capacity +planning estimates: + +| metadata extent | 4 KiB pages | projected RSS | projected raw | projected wall | idealized simulated write interval | +|---|---:|---:|---:|---:|---:| +| Qwen2.5-7B, 15,231,233,024 B | 3,718,563 | 8.09 GiB | 12.25 GB | 1,126 s | 2.94 s | +| Qwen2.5-72B, 145,412,407,296 B | 35,501,076 | 75.5 GiB | 116.9 GB | 10,749 s | 28.1 s | +| registered 235B extent, 470,187,269,120 B | 114,791,814 | 243.6 GiB | 378.1 GB | 34,757 s | 90.7 s | + +The idealized simulated intervals extrapolate only the observed window-256 +media schedule. Wall/RSS/raw projections assume current all-page verification +and duplicate CSV/JSON evidence and may change at larger allocator occupancy. +They are not experiment results. + +A complete 7B program diagnostic may be technically plausible with a finite +namespace that retains adequate spare blocks and a 48 GiB process budget, but +the current all-page verification/output contract projects beyond 600 s and +about 12 GB raw. It must not be launched without a separate preflight. A +reviewed bounded variant could program all 3,718,563 pages while verifying a +declared stratified sample (for example 1,024 pages across the loaded range), +streaming sufficient native facts instead of retaining duplicate formats. It +could claim all observed native program operations and only the explicit read +sample. It still could not claim real tensor payload transfer or integrity. + +The 72B and 235B full-write cases are not reasonable with the present retained +evidence path. They require a separately reviewed streaming-evidence design; +no core NAND, scheduler, or service lifecycle redesign is indicated by these +results. + +## Native-fact thermal replay + +`STARTUP-WRITE-ROLLING-N16384-THERMAL01` replayed the immutable rolling-16K +native JSONL through the unchanged `ActivityEnergyLedger` and the existing +255-entity mixed 2 mm RC package. It did not invoke MQSim or the live +controller. The model therefore retains all HBF/HBM/GPU components and their +shared cooling paths while applying activity only to the observed hbf0--hbf3 +native placements. GPU incremental power was explicitly zero for this +source-only upload diagnostic; zero does not represent measured idle power. + +The replay completed 11 fixed 20 ms thermal windows: one source window and ten +source-free recovery windows. The observed source energy was 0.04523761664 J: + +- startup programs: 0.04105641984 J; +- verification reads: 0.00418119680 J; +- NAND media: 0.04506419200 J; +- command/address/data-in: 0.00017342464 J. + +All source rows are foreground. No maintenance was requested. In this +frozen trace each `NAND_DATA_OUT` begin and end has the same timestamp, so the +unchanged ledger assigns it zero energy. The replay does not invent a transfer +duration. Host ingress, fabric upload, standby/idle energy and tensor payload +integrity remain unavailable. + +Energy reconciled independently across activity rows, native source, scope, +write/read stage, 20 ms thermal windows, and the thermal service cumulative +input. The largest absolute split discrepancy was 2.32e-14 J; thermal-service +input differed from the ledger by 6.94e-18 J. At 220 ms the service reported +0.04523761664 J input, 0.00358905189 J boundary loss, +0.04164856475 J stored-energy change and 1.54e-13 J residual (maximum relative +residual 3.40e-12). + +The hottest HBF endpoint was 300.023686 K at the first 20 ms boundary. After +200 ms without a source, the hottest HBF was 300.011052 K. Heat continued to +diffuse through the coupled package: the GPU reached 300.003098 K and the +hottest HBM reached 300.000672 K at 220 ms. This is a recovery observation, +not a claim that the package returned to its 300 K initial state. + +The native program interval ends at 12.950656 ms and all native activity ends +at 14.353536 ms. Both lie inside the first 20 ms thermal window. The reported +300.023686 K value is therefore the 20 ms endpoint; the approximately 13 ms +transient peak is unresolved and must not be inferred from this replay. + +The point used one CPU at 99%, 345,132 KiB peak RSS, 12.11 s wall time, no GPU, +no swap, a 4 GiB address limit and a 600 s watchdog. Its result remains +`CONDITIONAL_ENGINEERING_OPEN_LOOP_THERMAL_DIAGNOSTIC`: it is not a full 7B, +72B or 235B upload, controller reclosure, P2 qualification, or ModelFreeze. + +## Evidence locations + +- `eq3_thermal/plans/isolated-maintenance-campaign-v1/points/STARTUP-WRITE-FIXED64-01` +- `.../STARTUP-WRITE-CONCURRENCY-W1-02` +- `.../STARTUP-WRITE-CONCURRENCY-W256-02` +- `.../STARTUP-WRITE-CONCURRENCY-W256-N64-01` +- `.../STARTUP-WRITE-ROLLING-W8-N64-01` +- `.../STARTUP-WRITE-ROLLING-W256-N16384-01` +- `.../STARTUP-WRITE-ROLLING-N16384-THERMAL01` + +The window-256 16,384-page `summary.json` SHA-256 is +`332b99a3f1cb7e9549038502e17970edaa32d756e9c8df6c11e953863cdfee30`; +its chronological `native-events.jsonl` SHA-256 is +`ac6ddd94c6ab83428a8d11d119e5a48089fc4a47dcb0884551b4f60bcd2fa09f`. +The fixed analyzer test distinguishes operations, commands and overlapping +physical resources without invoking MQSim. diff --git a/docs/eq3_thermal/SYSTEM_THERMAL_INTERFACE_COVERAGE.md b/docs/eq3_thermal/SYSTEM_THERMAL_INTERFACE_COVERAGE.md new file mode 100644 index 0000000..fe809b3 --- /dev/null +++ b/docs/eq3_thermal/SYSTEM_THERMAL_INTERFACE_COVERAGE.md @@ -0,0 +1,32 @@ +# Four-topology system thermal interfaces + +This campaign is `CONDITIONAL_SIMULATED`. The original MQSim source, default +backend, production ABI and P2 evidence are unchanged. The TB/s service is an +isolated behavior proxy; it does not claim native MQSim produces those rates. +Native operation evidence and transfer limits remain separately identified in +`NATIVE_PROXY_COMPARISON.md`. + +| Interface | Actual producer | Actual consumer | Enablement | Capability and limit | +|---|---|---|---|---| +| Offered bytes by stack/channel | `RateWorkload` | `TopologyService.advance` | Explicit system point config | Preserves offered overload and distribution; no token inference | +| Topology/resource contract | Existing `BasicFabric` validation plus explicit OCP media limits | `TopologyService` and `CausalTopologyService` | Experimental runner only | Direct/relay/DASH paths, shared HBM link, joint controlled endpoints; continuous buffer turnover approximation | +| Physical activity phases | Shared aggregate service ledger | `EnergyMapper` and causal/maintenance energy adapters | Explicit coefficient profile | Media and transfer facts are separated; read base energy counted once; forwarding alone does not access HBM array | +| Window byte admission | Existing `ReadRatePolicy`, `EndpointAwarePolicy` | Next service window | One of three explicit policies | Completed facts affect future work only; active causal transfers drain; this is not native backend latency | +| Full package temperatures | Existing layered RC service | Policy, reliability age and output analysis | Explicit immutable model/binary | GPU, every array die/base and shared paths; original 300–400 K domain; no qualified fast thermal ROM | +| Exact storage completion | `CausalTopologyService` | `CausalExecutor.complete` | Separate causal runner | Dependency unlocks only after all child stripes and explicit retries finish; rate-only 20 ms completion is never used as a layer timestamp | +| Captured weight/dependency template | Actual tiny NumPy Qwen2-style forward | `build_architecture_trace(tiny_cpu_template)` | Explicit artifact/checksum/context | Observed operation/access order drives target tasks; target model sizes regenerated from official metadata; tiny CPU wall time is not target GPU timing | +| Compute completion | Single shared scenario compute resource in `CausalExecutor` | Token completion and dynamic GPU energy | Explicit compute durations and active power | Validated tiny CPU access/dependency template or separately classified synthetic DAG; target shapes/bytes regenerated; no calibrated token/s, KV or activation traffic claim | +| Cache fill/hit/replacement | Completed source delivery, bounded LRU reservations | Explicit HBM fill/read service jobs | `cache_mode=external_hbm` | Finite cache, actual service costs and pinned in-flight hits; excluded from standalone topology without HBM | +| Placement copy | Access-triggered source-read/destination-program phases | Version commit, later request placement, old-block erase | `migration_mode=basic`, explicit finite destination capacity | Metadata/version semantics and modeled bytes; not native payload integrity or a full FTL | +| Temperature history | Per-die thermal samples | `ReliabilityLedger` | Explicit Ea/age profile | HeatWatch transfer proxy; wall and equivalent age remain distinct; no RBER or lifetime prediction | +| Refresh demand and commit | `MaintenanceDriver`, shared service completions | Per-extent age and physical block operation counters | Explicit maintenance mode and spare pools | Same service resources as foreground; only committed extents reset age; read/program/erase and failure retained | +| ECC/retry scenario | Explicit conditional retry count, rather than inferred error rate | Additional shared read service before logical success | Optional 0/1/4 retries per source read | SSD-inspired cost sensitivity; no claim that temperature monotonically predicts HBF retry probability | +| Native MQSim command/maintenance evidence | Existing isolated backend and immutable receipts | Native/proxy semantic comparison | Separate small native validation path | Preserves native stage/stack mapping; not the producer of TB/s fluid receipts | + +Each runner writes input and source identity before its process starts, then +immutable activity/energy/thermal/control evidence and `DONE` or `FAILED`. +The base rate matrix intentionally has no token DAG and reports token +throughput `UNAVAILABLE`. New causal and maintenance interfaces require their +own completed run receipts before being described as experimentally exercised. +External physical GDDR has no package temperature node and remains +`UNAVAILABLE`; this is not a claim of zero board-level energy or heat. diff --git a/docs/eq3_thermal/SYSTEM_THERMAL_SENSITIVITY_FREEZE_CANDIDATE.md b/docs/eq3_thermal/SYSTEM_THERMAL_SENSITIVITY_FREEZE_CANDIDATE.md new file mode 100644 index 0000000..da5a6eb --- /dev/null +++ b/docs/eq3_thermal/SYSTEM_THERMAL_SENSITIVITY_FREEZE_CANDIDATE.md @@ -0,0 +1,106 @@ +# Four-topology system/thermal sensitivity freeze candidate + +Status: **INPUT METHOD APPROVED; NOT LAUNCHED; DEPENDS ON THE SERIAL BASE MATRIX.** +The user authorized a bounded sensitivity study. This file freezes the +conditional engineering method and prevents an accidental seven-dimensional +grid. It does not claim device calibration or approve a new physical model. + +## Representative scenario and comparison + +The primary OAT scene is mixed-direct Q1, W1 continuous offered read pressure, +1.536 TB/s per HBF stack, 20 s active plus 10 s recovery. The controlled arms +are `guard_only` and `read_rate_feedback_thermal_guard_v1`. A third, +control-disabled execution of every physical combination reports which +temperature constraint appears first without treating the two policies as +equivalent. It preserves the same sources and service resources; it only fixes +future budgets to baseline. + +HBM temperature limits cannot affect Q1 foreground service because this Q1 +workload offers bytes only to HBF and has no relay dependency. The mixed Q1 +HBM-limit rows are therefore `OBSERVATIONAL_ONLY`: they may alter reported +first-constraint classification, not delivery. The two non-nominal HBM levels +are additionally run on relay Q2 with identical offered HBF demand, where the +partner HBM endpoint is an actual joint-admission consumer. This adds six +mechanism points without multiplying every axis by topology. + +## Seven axes and evidence boundaries + +| Axis | Low / nominal / high | Actual consumer | Evidence and claim boundary | +| --- | --- | --- | --- | +| Retention Ea | 1.01 / 1.04 / 1.08 eV | `ReliabilityLedger` equivalent-age integral and maintenance driver, once ready | HeatWatch 95% fit interval for old 30–40-layer charge-trap MLC, 20–70°C and 1k–10k P/E. `PROXY`, not target-HBF uncertainty or an RBER/lifetime model. Deferred now. | +| Ambient | 300 / 310 / 320 K | Derived full RC model: all node initial temperatures plus top/bottom ambient move together | `SCENARIO_ASSUMPTION`. This is a matched temperature translation; it does not introduce an initial-to-ambient cooling transient. | +| External boundary resistance | 0.5 / 1 / 1.5 × nominal `1/h`, on both top and bottom | Derived full RC model recomputes `A/(dz/(2k)+scale/h)` | `CONDITIONAL_ABLATION`. Material `k`, heat capacities and internal edges stay fixed. It is not a TIM axis and not independent per-edge scaling. | +| HBF read energy | 0.5 / 1 / 1.5 × the 40 pJ/B array + 10 pJ/B base scenario | `EnergyMapper` read array and HBF base terms | `SCENARIO_ASSUMPTION`; preserves the 4:1 split. It does not scale HBM, relay endpoint, GPU, program or erase energy. Nominal is user-confirmed engineering input, not product measurement. | +| GPU external heat | 0 / 100 / 200 W | `run_system_point` supplies per-window energy to the actual GPU thermal component | Reuses the registered 0–200 W conditional envelope; independent of throttled memory delivery. It is not token-causal GPU utilization. | +| HBF guard limits | Light/Severe shifted −5 / 0 / +5 K from 353.15/363.15 K; Shutdown fixed 378.15 K | Thermal guard and next-window budgets | Research-policy sensitivity. The OCP 105°C envelope is never raised; none of the triplets is a product control recommendation. | +| HBM guard limits | Same Light/Severe shifts; Shutdown fixed 378.15 K | Observational only in Q1; actual paired endpoint gate in relay extension | Research-policy sensitivity. Values above the 95°C primary HBM evidence domain remain declared excursion scenarios, not HBM4 product limits. | + +GPU guard limits stay fixed at 363.15/373.15/383.15 K. The thermal solver's +400 K domain is unchanged. A trial crossing 400 K is retained as +`DOMAIN_FAILURE`; it is not clamped, retried with a different input, or used to +stop independent points. + +## Point construction + +For the six presently executable axes, OAT contains +`1 + 6 × (3−1) = 13` distinct physical combinations, including the shared +baseline. Three interactions are preregistered because they test coupled +mechanisms that an additive OAT cannot resolve: + +1. 320 K ambient × 1.5 external resistance; +2. 1.5 HBF read-energy scale × 200 W GPU external heat; +3. conservative HBF and HBM Light/Severe thresholds together. + +Each of these 16 combinations has two controlled arms and one uncontrolled +first-constraint arm: 48 mixed-direct points. The two HBM threshold endpoints +have the same three arms on relay: six more points. **Current executable total: +54 deterministic points.** No repetitions or confidence intervals are created +for this deterministic model. + +Ea remains a separate six-point block. Before those points become runnable, +the maintenance consumer must be connected and fixed-tested. The block uses +1.01/1.04/1.08 eV across the two controlled strategies, an exact initial age +near `DAY−2 s`, and the same 4 GiB/stack aged subset. A separate fresh/null +comparator must be retained. These points cannot reuse or be paired as though +they were the fresh read-rate baseline. Outputs remain conditional equivalent +age and maintenance traffic; RBER, ECC, failure probability and lifetime stay +unavailable. + +## Model inputs and preparation contract + +The existing mixed full 2 mm source is reused for 300 K/R1. The registered +single-axis models supply 310 K/R1, 320 K/R1, 300 K/R0.5 and 300 K/R1.5. The +interaction additionally requires the explicitly derived 320 K/R1.5 model; +it cannot be approximated by either single-axis directory. That model contains +64,512 nodes and 188,480 retained edges and is `DERIVED_NOT_SOLVED`. + +`experiments/eq3_system_thermal/prepare_sensitivity.py` writes immutable point +configs, hashes, model paths and `SENSITIVITY_INDEX.json`. It refuses a missing +thermal variant, refuses an output directory that already exists, and never +launches a process. `DEFERRED_EA.json` is machine-readable and has +`runnable=false`. + +The generated index status is `PENDING_DEPENDENCIES_BASE_MATRIX`. Per-point +limits remain one CPU, GPU 0, 8 GiB address space, 600 s and 1 GiB retained +output. The completed pilots bound a 30 s point at 98.34 s and 496,762,063 +bytes; the base projection is 29,805,723,750 bytes. The sensitivity allocation +is 27 GiB inside the existing parent-stage 80 GiB total, with a 6 h shared +stage wall bound, 32 GiB available-RAM reserve and 100 GiB free-disk reserve. + +`launch_sensitivity.py` is separate from the frozen base launcher. It refuses +to start until `BASE_DONE.json` says `COMPLETED`, then rechecks every config, +runtime source, model and thermal binary hash. It persists launch metadata +before each process. An explicit thermal-service `DOMAIN_FAILURE` is retained +and the launcher continues independent points. A numerical/protocol failure, +identity mismatch, resource limit or watchdog still stops the stage. No +failure path relaxes 400 K. + +## Fixed validation + +The preparation fixture generates all 54 configs in a temporary directory and +checks unique identities, 48 mixed plus 6 relay points, 18 uncontrolled arms, +actual relay consumer labels, fixed HBF/HBM shutdown, fixed GPU limits, fixed +400 K domain, source/model/resource locks, and non-runnable six-point Ea +metadata. A second test permits continued execution only for an explicit +thermal `DOMAIN_FAILURE` and rejects a numerical failure. Result: **2 tests, +PASS**. No thermal or native backend process ran. diff --git a/docs/eq3_thermal/TEMPERATURE_TOLERANCE_RATIONALE.md b/docs/eq3_thermal/TEMPERATURE_TOLERANCE_RATIONALE.md new file mode 100644 index 0000000..f152f08 --- /dev/null +++ b/docs/eq3_thermal/TEMPERATURE_TOLERANCE_RATIONALE.md @@ -0,0 +1,71 @@ +# 0.25 K 来源、合理性与可修改边界 + +2026-09-20;回答用户本轮问题。只读审阅/方案,不修改当前冻结标准、输入或运行结果。 + +## 当前来源 + +当前参考空间/时间相邻差最大值目标 0.25 K 来自项目任务书,用户在 +EQ3-DECISION-EXECUTION-v2 D2 明确要求保留。它不是 OCP、Sandisk、Micron 或 +3D-ICE 规定的通用精度。现有记录未提供由读取 SLO、传感器误差、实际热保护 +裕度或可容许性能偏差反推 0.25 K 的定量预算。 + +另一处 0.25 K 是 v2 的低幅值绝对 MAE 底项: +`min(1 K, 0.25 K + 0.05 * reference_range)`。这是不同合同,不能把它等同于 +参考相邻网格最大差。旧 1 K MAE、2 K hotspot/control 最大误差、5% NMAE、 +0.001 能量残差也各有各的被测量与范围。 + +## 合理之处与证据不足 + +相邻网格比较可以发现数值分辨率的影响;固定绝对 K 可以避免用约 300 K +绝对温度作分母掩盖局部温升误差。对热保护控制来说,最大热点误差也不能被 +组件均温平均掉。因此需要这类检查有合理依据。 + +但“必须是 0.25 K”没有得到本应用误差预算的充分支持。NASA 网格收敛教程 +要求考察相关工程量、渐近行为和误差,而不规定统一的开尔文门槛: +https://www.grc.nasa.gov/www/wind/valid/tutorial/spatconv.html 。相邻差也不是 +连续解误差的严格上界,更不等于真实器件预测误差。 + +当前 D4 工程滞回也是 0.25 K,Light/Severe 相隔 1 K;即使相邻差通过0.25K, +也不能自动保证阈值附近控制序列稳定。应把测量、采样/动作延迟、数值误差 +与滞回设计联查。该工程阈值不是产品要求,不应反过来证明参考精度必须如此。 + +完整 2→1 mm 的 mean 最大差0.7976K、hotspot最大差2.069K,说明旧0.25K判据 +确实未满足;mean误差小不能掩盖hotspot。静态实现审计未发现相关转换错误, +不意味着物理模型已标定,也不意味着原数值阈值充分合理。 + +## 建议如何改 + +把两个不同用途分开验收,而不是为本次2.069K结果选择一个刚好能过的阈值: + +1. **工程闭环资格**:请求唯一完成、路由/资源/能量记账、实际消费者、时序和 + 控制接口正确。允许有明确数值不确定性的温度输入,不继承MODEL_FREEZE。 +2. **应用充分性**:先冻结offered workload、最低持续delivered read rate、尾延迟 + 和积压预算,再检查合理温度扰动/更细模型是否改变这些结果及策略排名。 + 需求不足不能算稳定;少完成工作不能算控温收益。性能容许差仍需作为新的 + 工程SLO明确采用,不能冒充文献规定。 +3. **保护与科学资格**:单列保护温度的裕度、不确定性与违反情况。材料有效域 + 400K、研究规划380K均非产品温限。通过性能稳定性不能取代硬保护,也不能 + 把未合格参考升级成真实器件验证。 + +如果有已验证局部灵敏度 S_R=|dR/dT|,可从容许读速偏差epsilon_R推导候选 +温度预算 epsilon_R/S_R;但温度—服务关系在刷新/保护档位处可能不连续,不能 +只用一个导数跨越档位。此时需直接对误差带中的温度轨迹做分档/控制稳健性 +验证。当前资料不足以为目标HBF拟合这个灵敏度,不能先补一条有利的ECC曲线。 + +可先使用现有网格之间的实际观测差做敏感性输入,并注明它不是经证明的总误差 +上界;再独立考虑传感器误差、功率/材料/边界不确定性。由此决定工程用途是否 +足够准确,而不是不断细化直到某个孤立温度数字通过。 + +## 变更记录规则 + +用户可以明确采用新的用途/容差合同。采用后保存新版本、适用域、理由与冻结 +时间;旧0.25K及旧v1/v2结果继续并列。既有数据按新标准重算标回顾性分析。 +本轮问“可以修改吗”本身不等于已选择某个新数值,本文件没有更改阈值。 + +## 用户随后明确的采用决定 + +用户在查看解释后表示:“如果符合原先的参数假设同时在物理上合理就可以采用该方案”。 +据此采用当前模型作原假设下的条件工程验证,不再把旧0.25K达标作为基础系统 +继续推进的前提。局部场与源码证据满足当前模型自洽/机制合理的要求;产品级 +参数精度仍未验证。该决定不指定新的数值阈值、不回填旧PASS、不建立连续解 +误差上界,也不自动满足MODEL_FREEZE/盲测依赖。授权原文独立保存在外层计划。 diff --git a/docs/eq3_thermal/THERMAL_CONTROL_CURRENT_IMPLEMENTATION.md b/docs/eq3_thermal/THERMAL_CONTROL_CURRENT_IMPLEMENTATION.md new file mode 100644 index 0000000..5801e7e --- /dev/null +++ b/docs/eq3_thermal/THERMAL_CONTROL_CURRENT_IMPLEMENTATION.md @@ -0,0 +1,121 @@ +# 当前温控实现与 ECC/高温读速证据说明 + +本文只描述当前 `CpuService` 工程 fixture 的实际代码行为。结论均为 +**DOC_DERIVED**;D4 温度阈值和时序参数是 **SCENARIO_ASSUMPTION / ENGINEERING_FIXTURE**, +不是 SSD、DRAM、OCP 或 Sandisk 产品限值。 + +## 1. 温度如何产生控制状态 + +`control_sample()` 在每个采样时刻读取一个 stack 的全部 `stack_*` 热节点,使用 +其中最高温度。温度依次跨过 Light、Severe、Shutdown 阈值时,建议状态分别为 +Light、Severe、Shutdown;否则建议 Normal。降级时才使用滞回:只要当前温度仍 +不低于“当前已应用状态的阈值减去 hysteresis”,就保持当前状态。若建议重新等于 +已应用状态,已有 pending 动作会被取消。 + +D4 九点矩阵对全部八个 stack 使用相同参数: + +| 参数 | D4 数值 | 含义 | +|---|---:|---| +| 初始温度 | 298.15 K(25 °C) | fixture 初态 | +| 采样周期 | 20 ms | `sample_ns` | +| Light | 299.15 K(26 °C) | 第一档准入限速 | +| Severe | 300.15 K(27 °C) | 停止新的非维护请求 | +| Shutdown | 302.15 K(29 °C) | 停止该端点上的新请求,包括维护 | +| 滞回 | 0.25 K | 仅用于恢复方向 | +| 动作延迟 | 20 ms | 采样建议到动作可应用的延迟 | +| 最小驻留 | 100 ms | 原策略的双向驻留;v2 的恢复驻留 | +| Light 间隔 | 80 ms | 每个受控端点成功准入一次非维护请求后的下次准入时间 | + +这些很低的温度档位用于让短 CPU fixture 覆盖状态机。它们与 OCP 的 0–105 °C +工作结温范围、85 °C/24 h powered-on retention 条件均不是同一概念。 + +原 `hysteresis` 策略把升级和恢复都安排在 +`max(now + 20 ms, last_change + 100 ms)`。默认关闭的 +`hysteresis_escalation_priority_v2` 只让升级跳过恢复驻留,仍等待 20 ms 动作延迟; +恢复继续使用滞回和 100 ms 驻留。`none` 仍记录温度和建议状态,但不生成动作。 + +同一时间戳的顺序是:完成并释放资源 → 温度采样 → 应用到期控制动作 → 生成到期 +维护 → 准入队列 → 推进热模型。该顺序使控制动作在本时间戳的准入前生效。 + +## 2. 控制状态实际改变什么 + +`admit()` 先从 route 得到请求实际经过的全部受控端点和资源,再联合检查: + +- Normal:控制状态不阻塞;仍须等待请求到达且全部资源空闲。 +- Light:维护不受配额限制;非维护请求必须到达各受控端点的 `next_admit`。成功 + 准入后,每个不同端点的 `next_admit` 更新为当前时刻加 80 ms。 +- Severe:新的非维护请求受阻,维护仍可准入。 +- Shutdown:所有新请求受阻,包括维护。 + +direct 路径通常检查一个 stack;relay 路径同时检查目标 HBF 和实际经过的伙伴 +HBM。所有端点和资源检查完成后才整体预留,失败不会留下部分预留或活动能量。 +资源占用还会独立造成等待。 + +控制没有改变已准入请求的完成时间,也没有改变媒体服务时长、带宽、功率、ECC +或错误概率。D4 的 foreground read 服务时间固定为 20 ms;因此 D4 所见的 20 ms +backend service p95 是输入定义产生的稳定值。控制通过延迟或阻止准入降低活动与 +发热,同时可能显著增加排队时间。25 rps/stack 时,受控臂的 backend service +仍为 20 ms,但 backend queue-wait p95 为 10.32 s。不能把“已完成请求的固定服务 +时间”解释成高温下 ECC 保持了读速。 + +## 3. 年龄和维护实际模型 + +年龄是纯模拟墙钟:`now - last_successful_commit`,尚无成功维护时再加固定 +`initial_age_ns`。它没有温度加速、P/E、RBER、read-disturb 或 ECC 项。 + +- HBF:D4 固定周期 2 s,到期后生成 4 KiB maintenance read(20 ms),成功后 + 再生成 program(40 ms);仅最终成功 program 才清 pending、更新 + `last_commit_ns` 并增加 commit。此处没有温度依赖周期,也不是 OCP 的产品级 + 24–48 h 典型刷新间隔。 +- HBM4:D4 固定周期 1 s,生成 4 KiB `dram_refresh`(5 ms);成功完成后更新 + 年龄。刷新周期和时延不随温度变化,也没有 DRAM retention/error 模型。 +- 维护失败只能来自显式 `fail_maintenance_once`/`fail_fraction` 注入;失败后等待 + 一个采样周期重试。注入概率不消费温度。 + +`old_data_valid` 初始化为 true,当前代码不会基于年龄、温度或错误把它变为 +false。program/erase/commit 计数是事件账本,不是寿命、ECC margin 或数据完整性 +推断。 + +## 4. 尚未接通的 ECC/高温读速因果链 + +当前唯一闭合的反馈是: + +`实际活动能量 → 模拟温度 → 控制状态 → 后续请求准入 → 后续活动能量`。 + +下列链路均未实现: + +`温度轨迹 + 数据年龄 + P/E/read-disturb → RBER/retention error → ECC corrected bits +/ margin / uncorrectable error → read retry次数与附加媒体/传输时延及能量 → 完成或失败`。 + +因此目前无法回答“温控是否通过降低 ECC 压力稳定高温读取速度”。能够回答的只有: +准入节流在该工程 fixture 中降低了活动量和峰温,并将代价表现为排队和未完成工作; +一旦请求开始,其固定 20 ms read 时长与温度无关。 + +仓库证据也保持这一边界: + +- `configs/eq3_thermal/reliability.json` 为 `enabled:false`、`NOT_IMPLEMENTED`,没有 + retention reference、目标 HBF ECC/UBER 或 endurance 标定。 +- OCP v0.7.0 的本地登记只给出 85 °C powered-on 24 h retention 条件、产品特定 + P/E/read-disturb,以及典型 24–48 h、产品特定的周期维护;它没有给出可直接 + 生成目标 RBER、ECC strength、retry 分布或控制阈值的参数。 +- HeatWatch 的 3D-NAND 温度/retention 关系只登记为跨器件机制代理,且原拟合温区 + 与磨损范围有限,不能冒充 Sandisk HBF 测量。 +- Sandisk fact sheet 的 “no refresh power” 边界与 OCP 的周期维护语义在仓库中 + 被明确分成不同 profile;不能同时拿来证明当前 2 s 维护 fixture 的物理真实性。 +- HBM4 侧没有目标器件的温度相关 refresh、bank/channel timing、ECC 或 retry + 参数;固定 1 s/5 ms 仅是软件闭环情景。 + +若以后要检验该问题,最小缺口不是再调当前温控阈值,而是增加有来源且可关闭的 +可靠性消费者:按实际温度历史、年龄和磨损生成 raw-error/retry 事实,再让后端的 +实际命令阶段消费 retry 时延与能量,并单列 corrected/uncorrectable 结果。缺少这些 +环节时,只能报告温度、准入、队列、维护事件和固定服务时延,不能报告 ECC 收益。 + +## 代码与数据位置 + +- `src/eq3_thermal/cpu_service.cpp`:`control_sample()`、`apply_controls()`、 + `admit()`、`maintenance_due()`、`complete_due()` 和固定服务时长消费。 +- `eq3_thermal/plans/decision-execution-v2/d4-inputs/rate*-*.json`:D4 阈值、采样、 + 驻留、维护周期和服务时长。 +- `docs/eq3_thermal/D4_CONTROL_COST_REPORT.md`:九点控制成本和排队结果。 +- `configs/eq3_thermal/sources.json`、`reliability.json`,以及 + `docs/eq3_thermal/PARAMETER_SOURCE_AUDIT.md`:OCP、Sandisk、HeatWatch 与缺口边界。 diff --git a/docs/eq3_thermal/THERMAL_CONTROL_MOTIVATION_REVIEW.md b/docs/eq3_thermal/THERMAL_CONTROL_MOTIVATION_REVIEW.md new file mode 100644 index 0000000..353fe5d --- /dev/null +++ b/docs/eq3_thermal/THERMAL_CONTROL_MOTIVATION_REVIEW.md @@ -0,0 +1,44 @@ +# 温控目标与器件依据:2026-09-20 复核 + +USER_CONFIRMED:用户关注持续稳定、高效读取,希望核对 ECC 等高温开销与温控的因果关系。 +这是一项解释与模型缺口审核,不自动授权新增温度错误率曲线或修改 MQSim 完成时间。 +当前实现的精确行为见 THERMAL_CONTROL_CURRENT_IMPLEMENTATION.md。 + +| 原始资料 | 直接支持的事实 | 不能据此推导的结论 | +|---|---|---| +| [OCP HBF v0.7.0 §9.2,pp106–108](https://www.opencompute.org/documents/ocp-hbf-architecture-specification-v0-7-0-final-pdf);本地冻结同版原件 | Normal/Light/Severe/Shutdown;轻度降功耗、严重背压、滞回恢复;表脚注保留维护;在途命令须完成或返回规定错误 | 不能给出某温度下的确定带宽、ECC 迭代数或通用产品阈值 | +| 同版 §5.3,p57 | 高可纠正错误或 UECC 可触发 Base/host 读重试 | 不等于所有高温读取都会多重试,也不是延迟分布标定 | +| 同版 §9,p106 | 运行结温范围0–105°C;85°C、24h是带条件的数据保持指标 | 85°C不能直接充当节流门槛;当前400K数值物性域不等于产品安全上限 | +| [Sandisk HBF Fact Sheet, July 2025](https://documents.sandisk.com/content/dam/asset-library/en_us/assets/public/sandisk/collateral/company/Sandisk-HBF-Fact-Sheet.pdf) | 给出高温稳定性设计目标及封装导热改进 | 没有公开 T→RBER→retry→延迟曲线;不能凭宣传目标证明本模型校准,也不能假设HBF必然高温读速崩落 | +| [Sandisk CloudSpeed Gen II, Rev6, p14](https://downloads.sandisk.com/downloads/ess/cloudspeed2-product-specs.pdf) | 特定SSD以寿命保护为目的,65°C进入写节流,63°C退出 | 写节流阈值不可直接转为HBF读阈值 | +| [OCP Datacenter NVMe SSD v2.6 §9.2](https://www.opencompute.org/documents/datacenter-nvme-ssd-specification-v2-6-2-pdf) | 过温保护、足够频率监测与防传感器误触发 | 不能说明节流必须由ECC开销触发 | +| [Park et al., ASPLOS 2021](https://arxiv.org/abs/2104.09611) | 160颗3D NAND实测支持读重试增加延迟,ECC能力与重试优化相关 | 所测芯片不是当前HBF;其系数不可无条件转用 | +| [Luo et al., HeatWatch, HPCA 2018](https://research.ece.cmu.edu/safari/pubs/heatwatch-3D-nand-errors-and-self-recovery_hpca18.pdf) | 温度、保持时间、磨损和恢复历史共同影响错误与合适的读参考电压 | 不能用仅含当前T的单调错误模型覆盖所有情况 | +| [AMD Family15h BKDG, register D18F2xA4, p414](https://www.amd.com/content/dam/amd/en/documents/archived-tech-docs/programmer-references/55072_AMD_Family_15h_Models_70h-7Fh_BKDG.pdf) | 一个实际DRAM控制器可由温度告警触发双倍刷新及命令总线节流 | 是机制范例,不是HBM4刷新参数或ECC延迟曲线 | + +## 正确区分的因果关系 + +1. 器件温度接近限制 → 保护策略主动降低活动 → 吞吐下降或背压。 +2. 温度/保持年龄/磨损/温度历史 → 错误风险 → 纠错、读重试或维护开销 → 延迟及有效带宽变化。 +3. DRAM温度 → 保持约束及刷新占用 → 可服务时间变化;不能与NAND重读混为同一机制。 + +这三条链可同时存在;当前代码主要具备第一条的工程闭环,不具备后两条的真实器件标定。 +“ECC增加”应具体化为错误数、解码迭代/时长或读重试次数,而不是把固定ECC位数随温度增加。 + +## 最小后续方案(建议,未改变科学输入) + +先保留现有安全状态机作为兜底,采用目标温度带而非单点保证。评价持续交付字节率、 +p95/p99时延、时间窗带宽波动、未完成请求、维护积压及可靠性约束;排队时间必须计入。 +控制阈值需结合器件允许温度、控制延迟和热惯性余量,不能由现有低温fixture阈值迁移。 + +第一步只增加已有事实的报表:逐stack温度、交付率和尾延迟,以及后端实际能提供的 +错误/重试计数;没有计数就标UNKNOWN。第二步若获得对应HBF原始数据,冻结有来源的 +温度/年龄/磨损查表模型和允许域;缺资料时仅可另设明确的SCENARIO_ASSUMPTION。 +不能为展示温控收益,预设一条高温性能下降曲线再把结果称为实测规律。 + +真实重读、纠错和维护若需消耗MQSim媒体/通道资源,必须先给出原调度接口与最小 +生命周期方案并取得确认;禁止在最终完成时间上额外拼接一段“ECC延迟”。当前基础 +HBM/fabric实现继续推进,不因这项科学缺口暂停独立功能验证。 + +D3的380K是已授权的数学研究域包络筛选值,并非产品温控门槛;它甚至略高于上述 +105°C运行上限。因此DOMAIN_V2数值通过不能称为符合HBF产品热规格。 diff --git a/docs/eq3_thermal/TOPOLOGY_CONNECTION_AUDIT_V2.md b/docs/eq3_thermal/TOPOLOGY_CONNECTION_AUDIT_V2.md new file mode 100644 index 0000000..5c845b9 --- /dev/null +++ b/docs/eq3_thermal/TOPOLOGY_CONNECTION_AUDIT_V2.md @@ -0,0 +1,137 @@ +# EQ3 topology connection audit v2 + +Status: updated from the original read-only audit after the authorized basic CPU +implementation, 2026-09-20. The original `CpuService` findings remain below; +the additive `BasicSystem` evidence is reported separately and does not rewrite +old results. Connection semantics are `USER_CONFIRMED`; code statements and +retained receipts are `DOC_DERIVED`. Physical interpretation beyond them is a +gap. + +## Result + +| Requested topology meaning | Current implementation | Verdict | +| --- | --- | --- | +| Eight standalone HBF stacks, each with a base-die SRAM buffer | Legacy `CpuService` still has only a generic base lock. The optional BasicSystem instead configures two bounded banks per HBF, reserves before MQSim submission, holds external work on bank exhaustion, and releases all owners after package delivery. The actual small fixture completed 24/24 HBF requests across eight stacks. | **BASIC CPU FIXTURE MATCH for bounded two-bank ownership and backpressure; capacity/energy are scenario inputs, not validated SRAM hardware. External GDDR service remains UNAVAILABLE.** | +| Side-by-side 4 HBM + 4 HBF, independent pins/buses, full parallel operation, with pin-budget tradeoff | BasicSystem's mixed config explicitly sets every HBF `pair=null` and `relay_link=null`; four HBF stacks use actual MQSim channel partitions and four HBM stacks use independent parameterized media/GPU-link resources. Each stack completed 3 requests (24 total) with all owners released. No pin count, shared package pin budget, or calibrated link power exists. | **BASIC CPU FIXTURE MATCH for independent direct paths; physical pins, real HBM and pin-budget tradeoff remain UNAVAILABLE.** | +| Cascaded HBF behind an HBM base; GPU connects directly only to HBM; HBM direct and HBF relay share the relay-facing path; HBF access is two hops | BasicSystem rejects HBF direct in relay mode. After actual MQSim HBF completion, BasicFabric runs a pair-private HBF→HBM relay stage and then the paired HBM GPU stage. Relay receive and parameterized HBM-local output use the same two HBM banks and GPU-link queue. The actual fixture completed 24/24 requests for four pairs with zero final owners. | **BASIC CPU FIXTURE MATCH for sequential pair-private relay and shared paired-HBM contention; physical timings/energy and any package-global relay bus remain unvalidated.** | +| DASH dual-path read from the same HBF | Legacy `CpuService` retains its whole-base limitation. BasicSystem uses two HBF banks and independent direct/relay drain resources; relay additionally occupies the paired HBM bank/GPU link. The actual fixture alternated four direct/relay requests per HBF plus three local requests per HBM and completed 28/28 with four correct pairs and zero final owners. | **BASIC CPU FIXTURE MATCH for bounded dual-path arbitration; source fill timing, link rates, energy and thermal hotspots remain scenario/unconnected.** | + +## Actual route/resource contract + +The runtime contract is in `src/eq3_thermal/cpu_service.cpp:109-141`. + +| Request | Accepted route | Reserved resources | Controlled endpoints | Energy placement | +| --- | --- | --- | --- | --- | +| HBM | `direct` only | `hbmN:die:D`, `hbmN:base`, `hbmN:gpu-link` | `hbmN` | HBM die + HBM base + GPU | +| HBF direct | `direct` except in `relay` topology | `hbfN:die:D`, `hbfN:base`, `hbfN:upstream`, `hbfN:gpu-link` | `hbfN` | HBF die + HBF base + GPU | +| HBF relay | `relay` only in `relay`/`dash` | HBF die/base/upstream/relay-link + paired HBM base/GPU-link | HBF and paired HBM, deduplicated | HBF die/base + paired HBM relay-base + GPU; no HBM array energy | +| External GDDR | `direct` read/write only in all-HBF | `gddr:service`, `gddr:link` | none | external energy only in `CpuService`; temperature unavailable | + +The constructor enforces exactly eight package stack slots, exactly 8 HBF for +all-HBF, and exactly four unique one-to-one HBF/HBM pairs for relay/DASH +(`cpu_service.cpp:74-102`). Many-HBF-to-one-HBM and a shared relay bus across +pairs are rejected by construction. Resource acquisition is atomic at +admission, but the fixture has no channel count or byte-rate service tail. + +The opt-in BasicSystem contract is separate: + +| Request | Backend | Package resources and completion | +| --- | --- | --- | +| HBF direct | Persistent `stack_local_page` enters a topology-matched MQSim channel group; native request route remains `direct` | Reserve one of two HBF banks before submit; after raw MQSim completion equals reported completion, drain on the HBF direct link | +| HBF relay | Same actual MQSim media path; package route is stored separately and relay topology rejects a package-direct request | Reserve HBF bank, then pair-private relay into a free paired-HBM bank, then serialize on that HBM GPU link | +| HBM local | `PARAMETRIC_HBM_SCENARIO`, not MQSim and not a real DRAM backend | Reserve from the same two HBM banks used by relay receive, then serialize on the same HBM GPU link | + +`BasicSystem` retains external arrival/wait, backend media/reported completion, +package completion, and final completion as separate fields. It advances the +existing MQSim `until(horizon)` interface without host sleep. It rejects +composition if MQSim's generic bandwidth bound makes reported completion differ +from the raw callback, so the package fabric is not appended to an unidentified +transfer term. `mark_source_ready` deliberately adds no second source fill. +This consumer is default off and is not connected to the production host +service or thermal solver. + +For DASH specifically, this shared-resource rule is stricter than the cited +architecture. Sections IV-B and V-B describe independently accessible, +double-buffered SRAM regions that let different ready chunks from one HBF drain +over its direct and relay paths concurrently. The source still shares TSV fill +and requires buffer/path availability; it does not justify removing all HBF +arbitration. See [DASH, arXiv:2608.14333v1, Sections IV-B and +V-B](https://arxiv.org/html/2608.14333v1#S4.SS2). + +The declarative P1 graph is less complete than the runtime contract. +`tools/eq3_thermal_config.py:130-144` declares per-HBM GPU resources and HBF +array/TSV sharing; a relay edge shares the paired HBM GPU resource and marks +`dram_array_access=false`, but the graph reports `arbitration_status` as +`NOT_IMPLEMENTED` (`:158-163`). `CpuService` separately adds paired-HBM base +contention. Drawing the graph is therefore not proof of implemented service. + +## Configuration and request evidence + +- `configs/eq3_thermal/topologies/mixed_direct_8.json` is currently **2 HBM + + 6 HBF**, not the requested side-by-side 4+4 definition. The engineering + fixture changes it in memory to 4+4 at + `tools/eq3_cpu_fixture.py:14-20`; all D4 requests are explicit `direct` + reads (`tools/eq3_cpu_load_matrix.py:21-30`). +- In each retained D4 `none` run (5/10/25 requests/s/stack), the raw log has + eight foreground starts at `t=0`: four HBM requests reserve distinct + `hbmN:{die,base,gpu-link}` resources and four HBF requests reserve distinct + `hbfN:{die,base,upstream,gpu-link}` resources. `D4_PER_STACK_SERVICE.json` + also reports symmetric per-stack counts. This supports only the stated CPU + fixture parallelism; it does not measure pins, buses, or bandwidth. +- D4 covers only `mixed_direct`. It supplies no all-HBF or relay workload + evidence. The retained four-topology regression is a fixed software suite: + 59 Python tests passed, including topology/IR/export checks. Its C++ + `resource_energy_topologies` fixture checks two requests on `hbf0`, paired + HBM contention, relay base energy, and external GDDR accounting. It is not a + throughput or physical topology experiment. +- The cascade contention assertion at + `tests/eq3_thermal/cpu_service_tests.cpp:263-265` observes a paired HBM direct + request starting after the HBF relay releases shared HBM base/link resources. + Endpoint tests also preserve Shutdown/Light admission across both traversed + control domains. +- The newer `basic-four-topology-actual` receipt uses one fresh actual MQSim + service process per topology and topology-matched derived maps: eight + one-channel/one-die HBF groups for all-HBF and four for each 4+4 case. Results + are 24/24 all-HBF, 24/24 mixed, 24/24 relay, and 28/28 DASH. Every configured + source has at least three requests, waits become nonzero under the two-bank + bound, and every final snapshot has null bank/link owners and no unfinished + request. This supersedes `NOT_IMPLEMENTED` only for the small CPU behavior + axis; it does not supersede the research/thermal gaps above. + +## External GDDR scope discrepancy + +The current system fixture and research-layered path preserve the intended +boundary: `CpuService` requires no package `gddr` thermal node and reports +`UNAVAILABLE_OUTSIDE_PACKAGE`; layered IR records an external physical GDDR +with `package_geometry_modeled=false` +(`tools/eq3_layered_ir.py:398-417`). Energy remains separately observable. + +The older standalone P1 generator does something different: +`tools/eq3_thermal_config.py:145-149` creates a `gddr` thermal node and a board +proxy edge, while excluding it from the eight stack slots. Its fixed test +explicitly expects that node. That artifact must not be cited as package-only +all-HBF thermal evidence. This is an existing representation mismatch, not a +request to remove or reinterpret retained evidence. + +## Remaining closure after the basic implementation + +1. Replace the scenario two-bank capacity/timing/energy with sourced parameters + before treating the implemented bounded ownership as a validated SRAM model. +2. Bind per-stack interface width/rate and a package pin-budget accounting + layer before making the side-by-side pin-budget claim. Preserve the current + disjoint-resource fixture as engineering evidence. +3. BasicFabric now models sequential HBF-HBM and HBM-GPU stages with explicit + scenario rates/latencies and pair-private queues. Source/calibrate those + values and decide whether a future research topology instead has a + package-global relay bus. +4. Basic DASH now has two bounded source banks and independent direct/relay + drain resources. Its source-ready point comes from completed MQSim data-out; + a separate calibrated shared-TSV fill/power model is still absent. +5. Choose one external-GDDR thermal boundary for future generated artifacts. + Package-only EQ3 should retain physical identity and external energy while + keeping GDDR temperature unavailable, consistent with the current user + decision. + +These remaining items are research parameterization, production integration, +or further structural changes. The completed basic fixture does not authorize +or validate them. diff --git a/docs/eq3_thermal/USER_DECISIONS_REQUIRED.md b/docs/eq3_thermal/USER_DECISIONS_REQUIRED.md new file mode 100644 index 0000000..0e6b110 --- /dev/null +++ b/docs/eq3_thermal/USER_DECISIONS_REQUIRED.md @@ -0,0 +1,121 @@ +# Current decision status — EQ3-DECISION-EXECUTION-v2 + +The historical requests below are preserved, but the following are superseded: +full1mm reference executed; D2 v2 adopted and implemented; D3 uniform-alpha domain +approved and executed; opt-in escalation priority implemented/tested; default-off +native command facts and CPU service consumer implemented/validated. The user +removed invariant numerical resource caps and permits reasonable per-experiment +allocation. Current model is adopted for conditional engineering use under the +unchanged original assumptions; old .25K FAIL remains a reported result, not a +prerequisite blocking that engineering work. + +The one remaining requested structural decision is true MQSim die maintenance. +A concrete reviewable design now exists in +[MQSIM_DIE_MAINTENANCE_NARROW_DESIGN.md](MQSIM_DIE_MAINTENANCE_NARROW_DESIGN.md): +backend-owned cohort/child identity, shared TSU arbitration, read/program/erase, +allocation/commit/failure reconciliation and unique completion. Implementation is +not authorized. Keep UNSUPPORTED_CAPABILITY until explicitly approved. Optional +nonuniform mesh/ROM remain separate future scopes, not needed to continue the +adopted engineering path. No live GPU/production or calibrated-device claim. + +The sustained read-rate objective and source-backed proxy permission are now +USER_CONFIRMED. Sources and a minimal external feedback proposal are documented +in READ_RATE_CONTROL_SOURCE_BASIS.md; the actual controller still uses the +implemented thermal-state gate. No invented temperature-to-ECC law was added. + +--- + +# EQ3-MINIMAL-REPAIR-v1: decisions required beyond local repair + +All options below are proposals, not approvals. USER_CONFIRMED: local repairs and +CPU checks may continue; structural changes require prior explicit confirmation. +Unchanged P2 failures and old six engineering points remain historical evidence. + +## Real MQSim NAND phases and die maintenance — NEEDS_USER_CONFIRMATION_REFACTOR + +Problem/evidence: `MqsimOnlineEngine` exposes Arrival/Admission/raw Completion; +request admission is not NAND command start. It has no operation expressing +HBF die refresh and no physical die/plane/transfer stage facts at this boundary. +Ordinary host write cannot certify read/program maintenance or commit age. + +Minimal interface option (recommended current boundary): consume existing +request facts, disclose unknown die/plane/physical bytes, use an opt-in external +submission gate with existing horizon advancement, and explicitly return +UNSUPPORTED_CAPABILITY for maintenance. No MQSim timing or lifecycle changes. +This delivers a request-level CPU path, not a full real-backend maintenance loop. + +Minimal structural option requiring approval: first expose immutable stage facts +at existing MQSim command transitions, then add an explicit backend maintenance +operation with identity, enqueue/start/end/commit/fail, ownership and shared +resource arbitration. Proposed affected boundaries: `src/mqsim_adapter/mqsim_online.cpp`, +`include/hbfsim/mqsim_online.hpp`, matching MQSim flash-controller/transaction +interfaces after a narrower source design. Do not retrofit these by extending +CpuService. Exact internal symbols and lifecycle design must be submitted before +this option can be implemented; this is a scope choice, not a ready implementation approval. +Tests: off event/completion parity, command ordering, shared arbitration, +fail-before-commit, unique completion, no callback reentrancy, CPU-only. +Rollback: opt-out wrapper; any backend patch in an independent revertible commit. +Resources stay within existing 16/12GiB, CPU1,600s,4/20GiB; wider work needs a new +concrete preflight. Affected conclusion: actual MQSim active maintenance/energy, +not the already verified engine completion semantics. + +## Full 1mm reference — PENDING_USER_APPROVAL_RESOURCE + +DOC_DERIVED: exact 4s pilot 328.012s, ~4.91GiB, setup/factor dominates; +100s estimate 2000–2600s, not a measurement. Same-window 2→1mm still fails0.25K. +Recommended optional next scope: one complete unchanged train1mm/100s reference, +CPU1,3600s watchdog (new requested limit), process12/task16GiB, point4/task20GiB, +existing reduced/lossless output, no blind opening. Before execution bind exact +input/output forecast and remaining disk to a separate approved preflight. +No automatic downstream/finer-grid authorization and no guarantee of convergence. +Alternative: keep P2 blocked with existing evidence. A different solver/ROM is a +separate scientific/structural proposal and is not a workaround for600s. + +## Development domain — retain failure unless a new scientific version is approved + +DOC_DERIVED: independent2mm reference reaches400.911K, first crossing31.14s; +300–400K domain is unchanged. Recommended: preserve negative result, complete +reference qualification before deciding on physical/input redesign. Alternatives +are source-supported material extension or explicitly redesigned load/cooling, +each with new scope, parameters and comparability statement. Do not clamp or trim. +The present diagnostic patch cannot recover unknown historical trial values. + +## Emergency escalation and numerical budgets — optional policy/science changes + +No existing contract found that exempts more-severe actions from minimum dwell; +`control_sample` uses delay/dwell for every change and already cancels obsolete +pending actions. Recommended: retain this contract. An escalation-priority, +recovery-only dwell policy would require a named opt-in strategy and separately +approved comparisons; it must not replace old parameters or six-point results. + +Retain reference0.25K,MAE1K,hotspot2K,normalized5%,energy0.001 as written. +With a1K normalization floor5% is0.05K, tighter than the0.25K reference budget; +this is a diagnostic mismatch, not permission to relabel old FAIL as PASS. +Any new split reference/RC error budget requires a separate scientific decision. +Physical HBF/base/PHY energy, Ea/ECC/wear and device limits remain in +PARAMETER_GAPS.md. No GPU/live or formal paper matrix is authorized. + +## Verified boundary symbols for the next narrower backend design + +The existing adapter producer is +`MqsimOnlineEngine::Impl::{submit_to_device,dispatch_to_device,observe}` in +`src/mqsim_adapter/mqsim_online.cpp`; the host shim is maintained in +`patches/mqsim/0001-online-hbf-api.patch` (`Host_Interface_HBF::Submit_hbf_request`). +Actual upstream stage boundaries include +`NVM_PHY_ONFI_NVDDR2::Send_command_to_chip` and `transfer_read_data_from_chip` +(`third_party/mqsim/src/ssd/NVM_PHY_ONFI_NVDDR2.{h,cpp}`), plus +`Flash_Chip::Connect_to_chip_ready_signal`/`broadcast_ready_signal` +(`third_party/mqsim/src/nvm_chip/flash_memory/Flash_Chip.{h,cpp}`). +These are verified candidate observation boundaries, not permission to equate +command dispatch with NAND start or change callback ownership. The next design +must trace request→transaction→command identity and all start/end/failure paths +before declaring command energy complete. Cross-module event/type contracts and +maintenance arbitration are the reason this branch awaits a concrete approved +design. Existing CpuService limitations cannot close it. + +Many-HBF-to-one-HBM relay is also unsupported by `CpuService::Impl`'s existing +unique-pair check. Recommended: retain four unique pairs for this task. If this +new topology is wanted, define shared-base policy, geometry and power mapping +first, then separately approve the affected constructor/route/arbitration +changes and contention tests. No need to change pair topology to fix current +relay Shutdown leakage. diff --git a/docs/eq3_thermal/V3_Q1_UNMAPPED_MAINTENANCE_DIAGNOSIS.md b/docs/eq3_thermal/V3_Q1_UNMAPPED_MAINTENANCE_DIAGNOSIS.md new file mode 100644 index 0000000..53a052b --- /dev/null +++ b/docs/eq3_thermal/V3_Q1_UNMAPPED_MAINTENANCE_DIAGNOSIS.md @@ -0,0 +1,120 @@ +# V3 Q1 unmapped-maintenance diagnosis + +Date: 2026-09-20 +Scope: read-only review of completed Q1 points in +`eq3_thermal/plans/isolated-maintenance-campaign-v1/campaign-ocp4k-v3`. +No runtime, input, solver, or raw-result file was changed and no experiment was +started for this diagnosis. + +## Finding and classification + +The 26 failed maintenance jobs in every Q1 Safe and Near arm are +`DOMAIN_FAILURE` at the experiment-input/backend-initialization boundary, not a +confirmed backend implementation bug and not a deadline miss. The maintenance +bundle declares 64 aged pages as eligible sources at 199,980,000 ns, but this +MQSim instance only creates the source mapping when a foreground read first +touches a page. At the due instant, only 38 of the 64 target pages had completed +that first touch. The other 26 correctly returned terminal +`REJECTED_UNMAPPED` without issuing a native transaction or changing a mapping. + +This is also an `INITIAL_STATE_CAPABILITY_LIMITATION`: `initial_age_s` expresses +retention intent but does not initialize a resident MQSim mapping. The current +experiment therefore does not yet represent the intended state “model weights +already resident at time zero” for every scheduled maintenance target. + +## Actual completed-point evidence + +All three arms within a scene use identical request and maintenance inputs and +have identical maintenance outcomes. + +| Q1 scene | request-input SHA-256 prefix | mapped by due | committed | terminal failure | +|---|---:|---:|---:|---:| +| Safe, P0/P1/P2 | `ef1a5307fd60` | 38/64 | 38 | 26 `REJECTED_UNMAPPED` | +| Near, P0/P1/P2 | `272b5dbfcf28` | 38/64 | 38 | 26 `REJECTED_UNMAPPED` | +| Stress-thermal, P0/P1/P2 | `83ea34204e9d` | 64/64 | 64 | 0 | + +All nine points use the same maintenance input +SHA-256 `c508f7404d2e74d843828f3fdc2cd55db63c46413bdd5cbb5233fcfc4612977d`. +It schedules local pages 0 through 15 on each of `hbf0..hbf3`, with due time +199,980,000 ns and deadline 219,980,000 ns. + +For Safe and Near, the 38 mappings present at the due instant are: + +- `hbf0`: local pages 0--9; +- `hbf1`: local pages 0--9; +- `hbf2`: local pages 0--8; +- `hbf3`: local pages 0--8. + +The failed jobs are IDs 1000038--1000063: `hbf2:9`, `hbf3:9`, `hbf0:10`, +then the remaining stripe order through `hbf3:15`. Every failed completion has +the same fail-closed facts: `end_ns == due_ns`, zero native transaction IDs, +`source_version == 0`, `mapping_committed == false`, and no age reset. The +completion field `deadline_met == true` merely records that the immediate +rejection occurred before the deadline; it must not be counted as a successful +deadline-satisfying maintenance operation. + +All 26 rejected sources become mapped later in the same Safe/Near execution, +after a real foreground read. Their first backend completions span +200,010,310--340,010,310 ns. Three (`hbf2:9`, `hbf3:9`, `hbf0:10`) map after +the due time but before the 219,980,000 ns deadline; the other 23 map after the +deadline. The completed foreground records and the isolated backend contract, +which creates metadata for a populated page on first read, support this final +mapping conclusion. No mapping was inferred from an address alone. + +Representative immutable raw evidence is: + +- Safe P0 `result.json`, SHA-256 + `bb59b9ef6b0bf7211436b5289d0a2e1f2c049e4cb7cc481ac2deac4a2ec5e1d2`; +- each point's `maintenance.csv`, `summary.json`, `result.json`, and `DONE.json` + under its `OCP4K-W1-Q1--P` directory. + +## Code-path evidence + +The isolated maintenance unit calls `Begin_hbf_maintenance` before allocating +or submitting a read/program pair. A missing mapping returns +`REJECTED_UNMAPPED` immediately +(`experiments/eq3_maintenance/backend/patches/0004-eq3-maintenance.patch`, +the `HBF_Maintenance_Unit::Submit` hunk). This is the intended fail-closed +behavior: maintenance cannot preserve or relocate an unknown source. + +`ClosedLoopRunner._submit_due_maintenance` submits a due row once. A backend +terminal rejection sets its state to `FAILED`; only `NOT_DUE`, `DEFERRED`, and +`THERMAL_WAIT` rows are considered on later iterations +(`experiments/eq3_maintenance/closed_loop.py`). Consequently, retry after a +later first touch is not currently supported by the campaign coordinator. The +backend could accept a newly identified maintenance request after the page is +mapped, but this campaign neither creates such a request nor preserves a retry +lifecycle for the failed ID. + +## Minimal safe correction choices + +1. **Smallest campaign-only correction:** construct each maintenance bundle + only from pages proven mapped before its due time. This preserves the + backend and its fail-closed invariant. It changes target count/comparability, + so it requires a new input version and affected-point reruns; it cannot be + applied retrospectively to v3. +2. **Closest match to the intended resident-weight initial state:** add an + explicit pre-observation mapping initialization/import capability. It must + establish mappings without pretending that host writes or refresh commands + occurred and without charging fabricated traffic, energy, or wear. This + changes backend initialization semantics and requires a reviewed interface + proposal before implementation. +3. **Retry alternative:** retry a terminal unmapped job only after an observed + foreground completion maps that exact page, retaining the original due, + deadline, and age evidence. The current coordinator has no such lifecycle; + adding it changes maintenance scheduling semantics and requires approval. + In this evidence, 23 of 26 sources only map after the original deadline, so + retry would not turn them into on-time successes. + +Silently issuing host writes to preload the pages is not a valid repair: it +would add program traffic, energy, wear, and timing that the read-only weight +workload did not request. A rejected page must not receive an age reset. + +## Interpretation boundary + +The Q1 Safe/Near foreground and thermal results remain completed observations, +but their maintenance-success counts are conditioned on lazy first-touch +mapping. They are not valid evidence that a resident 7B weight population can +refresh only 38 of these 64 pages. Stress demonstrates that the same backend +path commits all 64 when all targets are mapped before due; it does not by +itself validate the missing resident-at-start initialization model. diff --git a/docs/eq3_thermal/figures/hbm3_base_15s_local_field.png b/docs/eq3_thermal/figures/hbm3_base_15s_local_field.png new file mode 100644 index 0000000..3ba358e Binary files /dev/null and b/docs/eq3_thermal/figures/hbm3_base_15s_local_field.png differ diff --git a/experiments/eq3_maintenance/.gitignore b/experiments/eq3_maintenance/.gitignore new file mode 100644 index 0000000..24315ec --- /dev/null +++ b/experiments/eq3_maintenance/.gitignore @@ -0,0 +1,4 @@ +__pycache__/ +*.pyc +backend/build*/ +thermal/build*/ diff --git a/experiments/eq3_maintenance/WORKLOAD_AND_ENERGY_BASIS.md b/experiments/eq3_maintenance/WORKLOAD_AND_ENERGY_BASIS.md new file mode 100644 index 0000000..54be8c6 --- /dev/null +++ b/experiments/eq3_maintenance/WORKLOAD_AND_ENERGY_BASIS.md @@ -0,0 +1,13 @@ +# Workload and energy basis — pilot, not frozen main matrix + +USER_CONFIRMED: reproducible synthetic intensity/duration workloads are permitted if no usable causal LLM trace exists. Microsoft Vidur (https://github.com/microsoft/vidur, inspected 2026-09-20) consumes request lengths and arrival intervals and produces inference execution traces using profiled execution models. Those inputs do not alone provide physical HBF page addresses, stack placement and token dependencies in this simulator. No LLM token/s, TTFT or TPOT will be inferred from memory throughput. + +W1 proposal: deterministic uniformly striped sequential page reads, steady offered demand with declared bursts; W2: phase-changing skew with all HBF stacks represented. Q4 keeps equal total HBF demand, distributed over8 rather than4 stacks. Background HBM reads remain the same for Q1/Q2/Q3; absent in Q4. Each request has immutable arrival, page, bytes and path. All arms use exactly the same trace and page initial ages. + +Pilot must measure real backend capacity and resource occupancy before choosing offered rates and main duration. Do not inflate media latency or operation power to create thermal pressure. The limited number of actual NAND channels/planes may make target TB/s unreachable; explicitly report this capability limit. If full-source energy at actual capacity cannot approach research thermal limits, Near/Stress-thermal are not ready and a pointless repeated matrix is not authorized by its nominal point count. Continue Safe and real maintenance contention checks, and report the limitation. + +Energy observations are separate from coefficients. Native NAND MEDIA_BEGIN/END gives activity intervals and physical die identity. DATA_OUT gives transfer, fabric events give actual path. Queue residence and command dispatch are not NAND media busy. Failed operations retain occurred energy. HBM backend supplies stack but not die: any distribution across its12 thermal dies is an explicit uniform spatial assumption, not an observed address mapping. Relay never produces HBM array heat. All-HBF excludes external GDDR service and temperature, not claims zero board energy. + +No HBF calibrated per-stage energy has yet been found in the registered profile. A pilot coefficient table must explicitly identify engineering/proxy values and disjoint scope. H200 aggregate HBM pJ/B is not an independently measured HBF array/base/PHY decomposition. Keep its transfer assumption visible; never count an aggregate coefficient independently on every stage. No ECC-temperature curve is invented. + +Research limit selection is declared separately from OCP product limits. Original materials, geometry and cooling stay fixed. The old arbitrary single-node thermal fixture thresholds are not reusable. 24h maintenance uses real86400 seconds; stress uses common initial page ages close to expiry, not a compressed lifetime. Backend working-set capacity may be a finite explicit subset of device capacity, never a claim of full512GB service validation. diff --git a/experiments/eq3_maintenance/aggregate_campaign.py b/experiments/eq3_maintenance/aggregate_campaign.py new file mode 100644 index 0000000..a24643a --- /dev/null +++ b/experiments/eq3_maintenance/aggregate_campaign.py @@ -0,0 +1,500 @@ +#!/usr/bin/env python3 +"""Read-only aggregation for the isolated EQ3 maintenance campaign.""" +from __future__ import annotations + +import argparse +import csv +import hashlib +import json +import math +from pathlib import Path +from typing import Any + +from native_operation_summary import summarize_commands_csv + +from analyze_points import (_boolean, _cell, _integer, _json, _number, _quantile, + _rows, analyze_point) + + +SCHEMA = "eq3-isolated-campaign-aggregate-v1" +POLICY_INPUTS = {"policy-profile.json"} + + +def _canonical(value: Any) -> str: + return json.dumps(value, sort_keys=True, separators=(",", ":"), allow_nan=False) + + +def pair_identity(manifest: dict[str, Any]) -> tuple[str | None, dict[str, Any]]: + hashes = manifest.get("input_sha256") + if not isinstance(hashes, dict): + return None, {} + selected = {key: value for key, value in hashes.items() if key not in POLICY_INPUTS} + identity = { + "source_head": manifest.get("source_head"), + "mode": manifest.get("mode"), "workload": manifest.get("workload"), + "active_ns": manifest.get("active_ns"), "end_ns": manifest.get("end_ns"), + "weight_model": manifest.get("weight_model"), + "thermal_model_dir": manifest.get("thermal_model_dir"), + "executables": manifest.get("executable_sha256"), "inputs": selected, + } + return hashlib.sha256(_canonical(identity).encode()).hexdigest(), identity + + +def scenario_key(spec: dict[str, Any]) -> str: + value = {key: spec.get(key) for key in ("workload", "mode", "scene")} + value["config"] = spec.get("config") + return hashlib.sha256(_canonical(value).encode()).hexdigest()[:16] + + +def _latencies(point: Path) -> list[int]: + values = [] + for row in _rows(point / "requests.csv"): + if str(row.get("phase", "")).upper() == "FINAL_COMPLETE": + value = _integer(row.get("end_to_end_latency_ns")) + if value is not None: + values.append(value) + return values + + +def _energy(point: Path) -> tuple[float | None, float | None, float | None]: + rows = _rows(point / "energy-activity.csv") + values = [(_number(row.get("energy_j")), row.get("component")) for row in rows] + known = [(value, component) for value, component in values if value is not None] + if not known: + return None, None, None + total = sum(value for value, _ in known) + gpu = sum(value for value, component in known if component == "gpu") + return total, gpu, (gpu / total if total else None) + + +def _native_coverage(point: Path) -> dict[str, list[int]] | None: + rows = _rows(point / "native.csv") + if not rows: + return None + # Native die is channel-local. Use the frozen physical channel ownership, + # exactly as the energy producer does; never infer die from logical pages. + mapping = _json(point / "stack-map.json") + per_channel = _integer(mapping.get("dies_per_channel")) + if per_channel is None or per_channel <= 0: + return None + channels = {channel: (group["id"], offset * per_channel) + for group in mapping.get("stacks", []) + for offset, channel in enumerate(group["channels"])} + coverage: dict[str, set[int]] = {} + for row in rows: + transactions = _cell(row.get("transactions")) + if not isinstance(transactions, list): + continue + for tx in transactions: + if not isinstance(tx, dict): + continue + channel = _integer(tx.get("channel")) + die = _integer(tx.get("die")) + chip = _integer(tx.get("chip")) + if channel not in channels or die is None or chip != 0 or not 0 <= die < per_channel: + continue + stack, offset = channels[channel] + if tx.get("stack") not in (None, "UNKNOWN", stack): + raise ValueError("native transaction stack/channel ownership mismatch") + coverage.setdefault(stack, set()).add(offset + die) + return {stack: sorted(dies) for stack, dies in sorted(coverage.items())} + + +def _thermal_transitions(point: Path, active_ns: int) -> dict[str, Any]: + first = {state: None for state in ("light", "severe", "shutdown")} + recovery = {state: None for state in ("severe", "light", "normal")} + peak = {kind: None for kind in ("gpu", "hbf", "hbm")} + representative: list[dict[str, Any]] = [] + for row in _rows(point / "thermal.csv"): + end = _integer(row.get("end_ns")) + states = _cell(row.get("stack_states")); temperatures = _cell(row.get("temperatures")) + if end is None: + continue + if isinstance(states, dict): + observed = set(states.values()) + for state in first: + if first[state] is None and state in observed: + first[state] = end + if end >= active_ns: + for state in recovery: + if recovery[state] is None and state in observed: + recovery[state] = end + grouped = {"gpu": [], "hbf": [], "hbm": []} + if isinstance(temperatures, dict): + for key, value in temperatures.items(): + number = _number(value) + if number is None: + continue + kind = "gpu" if key == "gpu" else "hbf" if str(key).startswith("hbf") else "hbm" + grouped[kind].append(number) + for kind, values in grouped.items(): + if values: + maximum = max(values) + peak[kind] = maximum if peak[kind] is None else max(peak[kind], maximum) + representative.append({"end_ns": end, + "gpu_k": max(grouped["gpu"], default=None), + "hbf_k": max(grouped["hbf"], default=None), + "hbm_k": max(grouped["hbm"], default=None)}) + return {"first": first, "recovery": recovery, "peak": peak, "series": representative} + + +def _resource_conservation(point: Path) -> bool | None: + rows = _rows(point / "resources.csv") + if not rows: + return None + fabric = _cell(rows[-1].get("fabric")) + if not isinstance(fabric, dict) or not isinstance(fabric.get("unfinished"), list): + return None + return len(fabric["unfinished"]) == 0 + + +def _control_stats(point: Path) -> dict[str, Any]: + actions: dict[str, int] = {}; outcomes: dict[str, int] = {} + first_stop = first_increase = None; zero_budget_windows = 0 + for row in _rows(point / "control.csv"): + start = _integer(row.get("applies_to_window_start_ns")) + decisions = _cell(row.get("stack_decisions")) + if not isinstance(decisions, list): + continue + any_zero = False + for decision in decisions: + if not isinstance(decision, dict): continue + action = str(decision.get("action", "UNKNOWN")); outcome = str(decision.get("outcome", "UNKNOWN")) + actions[action] = actions.get(action, 0) + 1 + outcomes[outcome] = outcomes.get(outcome, 0) + 1 + any_zero = any_zero or _integer(decision.get("budget_bytes")) == 0 + if action == "STOP" and first_stop is None: first_stop = start + if action == "INCREASE" and first_increase is None: first_increase = start + zero_budget_windows += int(any_zero) + return {"actions": actions, "outcomes": outcomes, "first_stop_ns": first_stop, + "first_increase_ns": first_increase, "zero_budget_windows": zero_budget_windows} + + +def _cost_metrics(point: Path, end_ns: int) -> dict[str, Any]: + final = [r for r in _rows(point / "requests.csv") if r.get("phase") == "FINAL_COMPLETE"] + metrics = {} + for kind in ("hbf", "hbm"): + selected = [r for r in final if r.get("stack", "").startswith(kind)] + for column in ("external_wait_ns", "backend_latency_ns", "fabric_latency_ns"): + values = [_integer(r.get(column)) for r in selected] + metrics[kind + "_" + column + "_p95"] = _quantile([v for v in values if v is not None], .95) + latest = {} + for row in _rows(point / "maintenance.csv"): + if row.get("request_id"): + latest[row["request_id"]] = row + missed, unknown, rejected, expired, unfulfilled, delay, maximum_age = 0, 0, 0, 0, 0, [], [] + for row in latest.values(): + completion = _cell(row.get("completion")) + completion = completion if isinstance(completion, dict) else {} + reset = _integer(row.get("age_reset_ns")) + if reset is None: reset = _integer(completion.get("age_reset_ns")) + committed = _boolean(row.get("mapping_committed")) + if committed is None: committed = _boolean(completion.get("mapping_committed")) + deadline, due = _integer(row.get("deadline_ns")), _integer(row.get("due_ns")) + if committed is not True: reset = None + status = str(row.get("backend_status") or completion.get("status") or "") + is_rejected = status.startswith("REJECTED_") + rejected += int(is_rejected) + submitted = _integer(row.get("submit_ns")) + # Derived from the recorded actual submit time and original deadline; + # do not relabel every INVALID_TARGET as an expiry. + expired += int(is_rejected and deadline is not None and submitted is not None + and submitted > deadline) + if deadline is None: + unknown += 1 + elif deadline < end_ns and (reset is None or reset > deadline): + unfulfilled += 1 + if not is_rejected: missed += 1 + if reset is not None and due is not None: delay.append(max(0, reset-due)) + initial = _number(row.get("initial_age_s")) + if initial is not None: + before = min(reset, end_ns) if reset is not None else end_ns + maximum_age.append(max(initial + before/1e9, (end_ns-reset)/1e9 if reset is not None else 0)) + metrics.update(maintenance_deadline_missed_by_end=missed, + maintenance_rejected_without_service=rejected, + maintenance_rejected_after_deadline=expired, + maintenance_declared_unfulfilled_by_deadline=unfulfilled, + maintenance_deadline_unknown=unknown, + maintenance_due_to_commit_max_ns=max(delay, default=None), + maintenance_declared_age_peak_s=max(maximum_age, default=None)) + return metrics + + +def derive_completed(point: Path, spec: dict[str, Any]) -> tuple[dict[str, Any], dict[str, Any]]: + derived = analyze_point(point) + manifest = _json(point / "manifest.json"); done = _json(point / "DONE.json") + summary = _json(point / "summary.json") + profile = _json(point / "profile.json") + active_ns = _integer(manifest.get("active_ns")) + if active_ns is None: + raise ValueError(f"{point}: active_ns unavailable") + active = [row for row in derived["time_series"] if row["start_ns"] < active_ns] + recovery = [row for row in derived["time_series"] if row["start_ns"] >= active_ns] + def total(rows, key): + values = [row.get(key) for row in rows] + return None if any(value is None for value in values) else sum(values) + def boundary(rows, key): + return rows[-1].get(key) if rows else None + latencies = _latencies(point) + energy_total, gpu_energy, gpu_fraction = _energy(point) + transitions = _thermal_transitions(point, active_ns) + control = _control_stats(point) + identity_hash, identity = pair_identity(manifest) + rates = [row["delivered_bytes_per_s"] for row in active + if row["delivered_bytes_per_s"] is not None] + mean = sum(rates) / len(rates) if rates else None + stdev = (math.sqrt(sum((value-mean)**2 for value in rates) / len(rates)) + if rates and mean is not None else None) + conservation = _resource_conservation(point) + mismatch = derived["cross_checks"]["rates_vs_request_delivery_mismatch_windows"] + censored = derived["explicit_censored_request_count"] + functional = ("FUNCTIONAL_FAIL" if mismatch or conservation is False else + "FUNCTIONAL_PASS_WITH_CENSOR" if censored else "FUNCTIONAL_PASS") + numerical = done.get("numerical_status") + capability = done.get("capability_status") + scientific = ("SCIENTIFIC_ACCEPTED" if numerical == "ACCEPTED" and capability == "ACCEPTED" + else "CONDITIONAL_NOT_MODEL_PASS") + row = { + "point_id": spec["id"], "scenario_key": scenario_key(spec), + "workload": spec.get("workload"), "mode": spec.get("mode"), + "scene": spec.get("scene"), "policy": spec.get("policy"), + "weight_model": (spec.get("config") or {}).get("weight_model"), + "source_head": manifest.get("source_head"), "page_bytes": profile.get("page_bytes"), + "run_wall_s": done.get("wall_s"), "child_peak_rss_kib": done.get("max_child_rss_kib"), + "model_payload_coverage_fraction": summary.get("weight_coverage", {}).get("model_payload_fraction"), + "per_stack_offered_completed": _canonical(summary.get("per_stack", {})), + "active_s": active_ns / 1e9, + "observation_s": derived["scope"]["observation_end_ns"] / 1e9, + "execution_state": done.get("execution_status", "UNKNOWN"), + "functional_state": functional, "scientific_state": scientific, + "capability_status": capability, "numerical_status": numerical, + "pair_identity_sha256": identity_hash, + "hbf_active_effective_delivered_bytes": total(active, "effective_delivered_hbf_bytes"), + "hbm_active_effective_delivered_bytes": total(active, "effective_delivered_hbm_bytes"), + "hbf_observation_effective_delivered_bytes": total(derived["time_series"], "effective_delivered_hbf_bytes"), + "hbm_observation_effective_delivered_bytes": total(derived["time_series"], "effective_delivered_hbm_bytes"), + "active_rate_p05_bytes_per_s": _quantile(rates, .05), + "active_rate_mean_bytes_per_s": mean, + "active_rate_cv": (stdev / mean if stdev is not None and mean else None), + "latency_p95_ns": _quantile(latencies, .95), "latency_p99_ns": _quantile(latencies, .99), + "latency_sample_count": len(latencies), + "active_end_backlog_effective_bytes": boundary(active, "effective_backlog_bytes"), + "observation_end_backlog_effective_bytes": boundary(derived["time_series"], "effective_backlog_bytes"), + "censored_requests": censored, + "first_light_ns": transitions["first"]["light"], + "first_severe_ns": transitions["first"]["severe"], + "first_shutdown_ns": transitions["first"]["shutdown"], + "recovery_normal_ns": transitions["recovery"]["normal"], + "recovery_severe_ns": transitions["recovery"]["severe"], + "recovery_light_ns": transitions["recovery"]["light"], + "control_first_stop_ns": control["first_stop_ns"], + "control_first_increase_ns": control["first_increase_ns"], + "control_zero_budget_windows": control["zero_budget_windows"], + "control_action_counts": _canonical(control["actions"]), + "control_outcome_counts": _canonical(control["outcomes"]), + "peak_gpu_k": transitions["peak"]["gpu"], "peak_hbf_k": transitions["peak"]["hbf"], + "peak_hbm_k": transitions["peak"]["hbm"], + "maintenance_due": derived["maintenance"].get("due_count"), + "maintenance_committed": derived["maintenance"].get("committed_count"), + "maintenance_failed": derived["maintenance"].get("failed_count"), + "maintenance_cleanup_failed": derived["maintenance"].get("cleanup_failed_after_commit_count"), + "maintenance_declared_page_fraction": derived["maintenance"].get("declared_model_page_fraction"), + "total_activity_energy_j": energy_total, "gpu_external_energy_j": gpu_energy, + "gpu_external_energy_fraction": gpu_fraction, + "energy_relative_residual_max": max((r["cumulative_energy_relative_residual"] + for r in derived["time_series"] + if r["cumulative_energy_relative_residual"] is not None), default=None), + "resource_conservation": conservation, + "hbf_native_die_coverage": _canonical(_native_coverage(point)), + "q4_limit": ("EXTERNAL_GDDR_SERVICE_AND_PACKAGE_TEMPERATURE_UNAVAILABLE" + if spec.get("mode") == "all_hbf_direct" else None), + "pair_status": "PENDING_GROUP_REVIEW", + } + hbf_delivered=row['hbf_observation_effective_delivered_bytes'] + row['package_total_J_per_effective_HBF_GB'] = energy_total/(hbf_delivered/1e9) if energy_total is not None and hbf_delivered else None + row['non_GPU_package_J_per_effective_HBF_GB'] = (energy_total-gpu_energy)/(hbf_delivered/1e9) if energy_total is not None and gpu_energy is not None and hbf_delivered else None + row['energy_per_byte_scope'] = 'PACKAGE numerator includes HBM and fabric; NOT isolated HBF efficiency' + row.update(_cost_metrics(point, derived["scope"]["observation_end_ns"])) + native = summarize_commands_csv(point / "commands.csv") + for operation in ("READ", "PROGRAM", "ERASE"): + for fact in ("media_command_starts", "media_command_ends", "child_transactions"): + row["native_" + operation.lower() + "_" + fact] = native["operations"][operation][fact] + row["native_actual_channel_die_plane_tuple_count"] = native["actual_channel_die_plane_tuple_count"] + row["native_media_end_semantics"] = "NOT_MAPPING_COMMIT_SUCCESS_OR_LIFETIME_PE_CYCLES" + row["maintenance_due_uncommitted_at_end_including_rejections"] = derived["maintenance"].get("backlog_at_observation_end") + row["maintenance_age_end_max_s"] = derived["maintenance"].get("declared_age_at_observation_max_s") + hbf_rates = [r["effective_delivered_hbf_bytes_per_s"] for r in active if r["effective_delivered_hbf_bytes_per_s"] is not None] + row["hbf_active_rate_p05_Bps"] = _quantile(hbf_rates, .05) + hbf_mean = sum(hbf_rates)/len(hbf_rates) if hbf_rates else None + row["hbf_active_rate_mean_Bps"] = hbf_mean + row["hbf_active_rate_cv"] = math.sqrt(sum((v-hbf_mean)**2 for v in hbf_rates)/len(hbf_rates))/hbf_mean if hbf_mean else None + representative = {"point_id": spec["id"], "mode": spec.get("mode"), + "active_ns": active_ns, "rates": derived["time_series"], + "thermal": transitions["series"], "identity": identity} + return row, representative + + +def pending_row(spec: dict[str, Any], state: str, *, reason: Any = None, + launcher: dict[str, Any] | None = None) -> dict[str, Any]: + return {"point_id": spec["id"], "scenario_key": scenario_key(spec), + "workload": spec.get("workload"), "mode": spec.get("mode"), + "scene": spec.get("scene"), "policy": spec.get("policy"), + "weight_model": (spec.get("config") or {}).get("weight_model"), + "execution_state": state, "functional_state": "NOT_EVALUATED", + "scientific_state": "NOT_EVALUATED", "pair_status": "INCOMPLETE", + "not_started_reason": _canonical(reason) if reason is not None else None, + "launcher_status": (launcher or {}).get("status"), + "launcher_exit_code": (launcher or {}).get("exit_code")} + + +def aggregate(queue: dict[str, Any], points_root: Path, + launcher_status: dict[str, Any] | None = None) -> tuple[list[dict[str, Any]], list[dict[str, Any]]]: + rows, representatives = [], [] + launcher_by_id = {row.get("id"): row for row in (launcher_status or {}).get("points", [])} + for spec in queue.get("points", []): + # A retrospective union may reference immutable points from distinct + # execution versions without copying or replacing their raw data. + point = Path(spec.get("artifact_path", points_root / spec["id"])) + if (point / "DONE.json").exists(): + row, representative = derive_completed(point, spec) + launcher = launcher_by_id.get(spec["id"], {}) + row.update(launcher_status=launcher.get("status"), + launcher_exit_code=launcher.get("exit_code"), not_started_reason=None, + artifact_path=str(point.resolve())) + rows.append(row); representatives.append(representative) + elif (point / "NOT_STARTED.json").exists(): + receipt = _json(point / "NOT_STARTED.json") + rows.append(pending_row(spec, receipt.get("execution_status", "NOT_STARTED"), + reason=receipt.get("reason"), + launcher=launcher_by_id.get(spec["id"]))) + elif (point / "FAILED.json").exists(): + rows.append(pending_row(spec, "FAILED", launcher=launcher_by_id.get(spec["id"]))) + else: + launcher = launcher_by_id.get(spec["id"], {}) + rows.append(pending_row(spec, launcher.get("status", "NOT_OBSERVED"), + reason="NO_TERMINAL_POINT_RECEIPT", + launcher=launcher_by_id.get(spec["id"]))) + groups: dict[str, list[dict[str, Any]]] = {} + for row in rows: + groups.setdefault(row["scenario_key"], []).append(row) + for group in groups.values(): + completed = [row for row in group if row["execution_state"] == "COMPLETED"] + identities = {row.get("pair_identity_sha256") for row in completed} + status = ("PAIR_READY" if len(completed) >= 2 and len(completed) == len(group) and len(identities) == 1 and None not in identities + else "PAIR_IDENTITY_MISMATCH" if len(completed) >= 2 and len(identities) > 1 + else "SINGLE_ARM_MATCH_BASELINE_SEPARATELY" if len(group) == len(completed) == 1 + else "INCOMPLETE") + for row in group: + row["pair_status"] = status + return rows, representatives + + +def write_csv(path: Path, rows: list[dict[str, Any]]) -> None: + keys = list(dict.fromkeys(key for row in rows for key in row)) + with path.open("w", newline="", encoding="utf-8") as handle: + writer = csv.DictWriter(handle, fieldnames=keys); writer.writeheader() + for row in rows: + writer.writerow({key: "UNKNOWN" if value is None else value for key, value in row.items()}) + + +def plot_overview(rows: list[dict[str, Any]], path: Path) -> None: + import matplotlib + matplotlib.use("Agg") + import matplotlib.pyplot as plt + completed = [row for row in rows if row["execution_state"] == "COMPLETED"] + fig, axes = plt.subplots(3, 1, figsize=(12, 8), sharex=True, constrained_layout=True) + if not completed: + axes[1].text(.5, .5, "No completed points", ha="center", va="center") + else: + x = list(range(len(completed))) + color = {"guard_only": "#777777", "thermal_hysteresis_guard": "#0072B2", + "read_rate_feedback_thermal_guard_v1": "#D55E00"} + c = [color.get(row["policy"], "black") for row in completed] + axes[0].scatter(x, [(row["hbf_active_effective_delivered_bytes"] or 0)/row["active_s"]/1e9 + for row in completed], c=c, s=18) + axes[0].set_ylabel("HBF active effective GB/s") + axes[1].scatter(x, [row["peak_gpu_k"] for row in completed], c=c, s=18, label="GPU") + axes[1].scatter(x, [row["peak_hbf_k"] for row in completed], c=c, marker="x", s=18, + label="HBF max") + axes[1].set_ylabel("Peak K"); axes[1].legend(frameon=False, ncol=2) + axes[2].scatter(x, [row["censored_requests"] for row in completed], c=c, s=18) + axes[2].set_ylabel("Censored requests"); axes[2].set_xlabel("Completed point index (see main-table.csv)") + for axis in axes: axis.grid(alpha=.25) + fig.suptitle("EQ3 campaign aggregate — execution success is not scientific acceptance") + fig.savefig(path, dpi=170); plt.close(fig) + + +def plot_representatives(representatives: list[dict[str, Any]], path: Path) -> None: + import matplotlib + matplotlib.use("Agg") + import matplotlib.pyplot as plt + selected = [] + for mode in ("mixed_direct", "relay", "dash", "all_hbf_direct"): + candidate = next((row for row in representatives if row["mode"] == mode + and "W1-" in row["point_id"] and "Stress-thermal-P2" in row["point_id"]), None) + if candidate is None: + candidate = next((row for row in representatives if row["mode"] == mode), None) + if candidate: selected.append(candidate) + fig, axes = plt.subplots(max(1, len(selected)), 1, figsize=(10, 2.6*max(1, len(selected))), + squeeze=False, constrained_layout=True) + if not selected: + axes[0][0].text(.5, .5, "No completed representative", ha="center", va="center") + for axis, item in zip((row[0] for row in axes), selected): + x = [(row["start_ns"]+row["end_ns"])/2e9 for row in item["rates"]] + y = [(row["effective_delivered_hbf_bytes_per_s"] or 0)/1e9 for row in item["rates"]] + axis.plot(x, y, color="#0072B2", label="HBF effective GB/s") + temp = axis.twinx(); temp.plot([row["end_ns"]/1e9 for row in item["thermal"]], + [row["gpu_k"] for row in item["thermal"]], + color="#D55E00", alpha=.75, label="GPU K") + axis.axvline(item["active_ns"]/1e9, color="black", linestyle=":", linewidth=.8) + axis.set_title(f"{item['mode']}: {item['point_id']}"); axis.set_ylabel("GB/s"); temp.set_ylabel("K") + axis.grid(alpha=.2) + axes[-1][0].set_xlabel("Simulation time (s)") + fig.savefig(path, dpi=170); plt.close(fig) + + +def main(argv=None) -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--queue", type=Path, required=True) + parser.add_argument("--points-root", type=Path, required=True) + parser.add_argument("--status", type=Path, + help="launcher status; NOT_STARTED receipts still take precedence") + parser.add_argument("--output", type=Path, required=True) + args = parser.parse_args(argv) + if args.output.exists() and any(args.output.iterdir()): + parser.error("output must be absent or empty") + args.output.mkdir(parents=True, exist_ok=True) + queue = _json(args.queue) + launcher_status = _json(args.status) if args.status else None + rows, representatives = aggregate(queue, args.points_root, launcher_status) + lock_ref = queue.get("lock") + lock_path = Path(lock_ref) if isinstance(lock_ref, str) else None + if lock_path is not None and not lock_path.is_absolute(): + lock_path = args.queue.parent / lock_path + lock_sha256 = (hashlib.sha256(lock_path.read_bytes()).hexdigest() + if lock_path is not None and lock_path.is_file() else None) + for row in rows: + row["campaign_lock"] = lock_ref + row["campaign_lock_sha256"] = lock_sha256 + write_csv(args.output / "main-table.csv", rows) + (args.output / "aggregate.json").write_text(json.dumps({ + "schema_version": SCHEMA, "point_count": len(rows), + "campaign_lock": lock_ref, "campaign_lock_sha256": lock_sha256, + "launcher_status_source": str(args.status.resolve()) if args.status else None, + "completed_count": sum(row["execution_state"] == "COMPLETED" for row in rows), + "rows": rows, + "status_semantics": {"execution": "process/output completion", + "functional": "raw consistency/resource conservation; censor retained", + "scientific": "explicit acceptance only; exit zero is insufficient"}, + }, indent=2, sort_keys=True) + "\n") + plot_overview(rows, args.output / "campaign-overview.png") + plot_representatives(representatives, args.output / "representative-timeseries.png") + (args.output / "DONE.json").write_text(json.dumps({ + "execution_status": "COMPLETED", "raw_modified": False, + "completed_points": sum(row["execution_state"] == "COMPLETED" for row in rows), + }, indent=2) + "\n") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/experiments/eq3_maintenance/analyze_points.py b/experiments/eq3_maintenance/analyze_points.py new file mode 100644 index 0000000..e643cf8 --- /dev/null +++ b/experiments/eq3_maintenance/analyze_points.py @@ -0,0 +1,628 @@ +#!/usr/bin/env python3 +"""Derive compact closed-loop diagnostics from immutable point CSV outputs. + +This module is deliberately a postprocessor: it neither imports the simulator +clients nor advances either backend. JSON null is used for unavailable facts; +an observed zero remains zero. +""" + +from __future__ import annotations + +import argparse +import csv +import hashlib +import json +import math +import os +import statistics +import sys +from pathlib import Path +from typing import Any, Iterable + + +SCHEMA_VERSION = "eq3-closed-loop-analysis-v1" + + +def _json(path: Path) -> dict[str, Any]: + value = json.loads(path.read_text(encoding="utf-8")) + if not isinstance(value, dict): + raise ValueError(f"{path}: expected JSON object") + return value + + +def _rows(path: Path) -> list[dict[str, str]]: + if not path.exists() or path.stat().st_size <= 2: + return [] + with path.open(newline="", encoding="utf-8") as handle: + return list(csv.DictReader(handle)) + + +def _cell(value: Any) -> Any: + if value is None or value == "" or value == "UNKNOWN": + return None + if not isinstance(value, str): + return value + try: + return json.loads(value) + except (json.JSONDecodeError, TypeError): + return value + + +def _number(value: Any) -> float | None: + value = _cell(value) + if isinstance(value, bool) or value is None: + return None + try: + result = float(value) + except (TypeError, ValueError): + return None + return result if math.isfinite(result) else None + + +def _integer(value: Any) -> int | None: + number = _number(value) + if number is None or not number.is_integer(): + return None + return int(number) + + +def _boolean(value: Any) -> bool | None: + value = _cell(value) + if isinstance(value, bool): + return value + if value in (1, "1", "true", "True"): + return True + if value in (0, "0", "false", "False"): + return False + return None + + +def _sum_known(values: Iterable[float | int | None]) -> float | None: + materialized = list(values) + if not materialized or any(v is None for v in materialized): + return None + return float(sum(materialized)) + + +def _quantile(values: Iterable[float | int], probability: float) -> float | None: + ordered = sorted(float(v) for v in values if math.isfinite(float(v))) + if not ordered: + return None + # Nearest-rank empirical quantile. The definition is recorded in output. + rank = max(1, math.ceil(probability * len(ordered))) + return ordered[rank - 1] + + +def _sha256(path: Path) -> str: + digest = hashlib.sha256() + with path.open("rb") as handle: + for block in iter(lambda: handle.read(1024 * 1024), b""): + digest.update(block) + return digest.hexdigest() + + +def _window_index(windows: list[dict[str, Any]], timestamp_ns: int) -> int | None: + # Fixed windows use [start,end); a completion exactly at end belongs to the + # next window. The small point counts make this explicit scan sufficient. + for index, window in enumerate(windows): + if window["start_ns"] <= timestamp_ns < window["end_ns"]: + return index + return None + + +def _maintenance_summary(rows: list[dict[str, str]], manifest: dict[str, Any], + observation_end_ns: int, model_page_count: int | None) -> dict[str, Any]: + declared = _integer(manifest.get("maintenance_count")) + if not rows: + observed_zero = declared == 0 + return { + "declared_count": declared, + "due_count": 0 if observed_zero else None, + "committed_count": 0 if observed_zero else None, + "failed_count": 0 if observed_zero else None, + "backlog_at_observation_end": 0 if observed_zero else None, + "initial_age_max_s": None, + "declared_page_count": 0 if observed_zero else None, + "declared_model_page_fraction": 0.0 if observed_zero and model_page_count else None, + "declared_age_at_observation_max_s": None, + "age_reset_count": 0 if observed_zero else None, + "semantics": "OBSERVED_DISABLED" if observed_zero else "UNKNOWN_NO_ROWS", + } + + latest: dict[str, dict[str, str]] = {} + for index, row in enumerate(rows): + request_id = row.get("request_id") or row.get("maintenance_id") or f"row-{index}" + latest[str(request_id)] = row + values = list(latest.values()) + due = [r for r in values if (_integer(r.get("due_ns")) is not None and + _integer(r.get("due_ns")) <= observation_end_ns)] + def completion(row: dict[str, str]) -> dict[str, Any]: + nested = _cell(row.get("completion")) + return nested if isinstance(nested, dict) else {} + def field(row: dict[str, str], name: str) -> Any: + direct = _cell(row.get(name)) + return direct if direct is not None else completion(row).get(name) + committed = [r for r in values if _boolean(field(r, "mapping_committed")) is True] + failed = [r for r in values if _boolean(field(r, "mapping_committed")) is False and + "FAIL" in str(field(r, "status") or r.get("state") or "").upper()] + # erase_completed=false is also the normal safe RECLAIM_DEFERRED state. + # Only the explicit backend failure terminal (or coordinator state derived + # from it) proves a cleanup failure. Ignore legacy cleanup_failed booleans + # that were derived solely from erase_completed. + cleanup_failed = [r for r in committed + if str(field(r, "status") or r.get("backend_status") or + r.get("state") or "").upper() + in {"FAILED_AFTER_COMMIT_NEEDS_RECONCILE", + "COMMITTED_CLEANUP_FAILED"}] + initial_ages = [_number(r.get("initial_age_s")) for r in values] + reset_count = sum(1 for r in committed if _integer(field(r, "age_reset_ns")) is not None) + committed_by_observation = [r for r in committed + if (_integer(field(r, "age_reset_ns")) is not None and + _integer(field(r, "age_reset_ns")) <= observation_end_ns)] + page_keys = {(r.get("stack"), _integer(r.get("stack_local_page"))) for r in values + if r.get("stack") and _integer(r.get("stack_local_page")) is not None} + ages_at_observation = [] + for row in values: + initial_age = _number(row.get("initial_age_s")) + if initial_age is None: + continue + reset = _integer(field(row, "age_reset_ns")) + mapping = _boolean(field(row, "mapping_committed")) is True + if mapping and reset is not None and reset <= observation_end_ns: + ages_at_observation.append((observation_end_ns-reset) / 1e9) + else: + ages_at_observation.append(initial_age + observation_end_ns / 1e9) + return { + "declared_count": declared, + "due_count": len(due), + "committed_count": len(committed), + "failed_count": len(failed), + "cleanup_failed_after_commit_count": len(cleanup_failed), + "backlog_at_observation_end": max(0, len(due) - len(committed_by_observation)), + "initial_age_max_s": max((v for v in initial_ages if v is not None), default=None), + "declared_page_count": len(page_keys), + "declared_model_page_fraction": (len(page_keys) / model_page_count + if model_page_count else None), + "declared_age_at_observation_max_s": max(ages_at_observation, default=None), + "age_reset_count": reset_count, + "semantics": ("RAW_LATEST_ROW_PER_REQUEST; age clears only on mapping_committed=true " + "with observed age_reset_ns; undeclared pages are UNKNOWN"), + } + + +def analyze_point(point: Path) -> dict[str, Any]: + point = point.resolve() + manifest = _json(point / "manifest.json") + done = _json(point / "DONE.json") + rate_rows = _rows(point / "rates.csv") + request_rows = _rows(point / "requests.csv") + thermal_rows = _rows(point / "thermal.csv") + control_rows = _rows(point / "control.csv") + maintenance_rows = _rows(point / "maintenance.csv") + model_page_count = None + extent_path = point / "weight-model-extent.json" + if extent_path.exists(): + model_page_count = _integer(_json(extent_path).get("global_page_count")) + if not rate_rows: + raise ValueError(f"{point}: rates.csv has no fixed windows") + + windows: list[dict[str, Any]] = [] + rate_facts: dict[tuple[int, str], dict[str, Any]] = {} + for row in rate_rows: + start = _integer(row.get("start_ns")); end = _integer(row.get("end_ns")) + stacks = _cell(row.get("stacks")) + if start is None or end is None or end <= start or not isinstance(stacks, list): + raise ValueError(f"{point}/rates.csv: malformed fixed window") + index = len(windows) + windows.append({"start_ns": start, "end_ns": end}) + for fact in stacks: + if not isinstance(fact, dict) or not fact.get("stack_id"): + raise ValueError(f"{point}/rates.csv: malformed stack fact") + rate_facts[(index, str(fact["stack_id"]))] = fact + + observation_end = windows[-1]["end_ns"] + stacks = sorted({stack for _, stack in rate_facts}) + + arrivals: dict[str, dict[str, str]] = {} + completions: dict[str, dict[str, str]] = {} + censored: dict[str, dict[str, str]] = {} + for row in request_rows: + request_id = row.get("request_id") + if not request_id: + continue + phase = str(row.get("phase") or "").upper() + if phase == "ARRIVAL": + arrivals.setdefault(request_id, row) + elif phase == "FINAL_COMPLETE": + completions[request_id] = row + elif phase == "CENSORED": + censored[request_id] = row + + request_counts: dict[str, dict[str, int]] = { + stack: {"arrived": 0, "final_delivered": 0, "censored": 0} for stack in stacks + } + request_bytes: dict[str, dict[str, int]] = { + stack: {"arrived_effective": 0, "arrived_physical": 0, + "final_delivered_effective": 0, "final_delivered_physical": 0, + "censored_effective": 0, "censored_physical": 0} for stack in stacks + } + completion_samples: dict[tuple[int, str], list[float]] = {} + derived_delivered: dict[tuple[int, str], int] = {} + for request_id, row in arrivals.items(): + stack = str(row.get("stack") or "UNKNOWN") + if stack not in request_counts: + request_counts[stack] = {"arrived": 0, "final_delivered": 0, "censored": 0} + request_bytes[stack] = {"arrived_effective": 0, "arrived_physical": 0, + "final_delivered_effective": 0, "final_delivered_physical": 0, + "censored_effective": 0, "censored_physical": 0} + size = _integer(row.get("bytes")); valid = _integer(row.get("valid_weight_bytes")) + valid = size if valid is None else valid + request_counts[stack]["arrived"] += 1 + if size is not None: + request_bytes[stack]["arrived_physical"] += size + if valid is not None: + request_bytes[stack]["arrived_effective"] += valid + for request_id, row in completions.items(): + stack = str(row.get("stack") or arrivals.get(request_id, {}).get("stack") or "UNKNOWN") + size = _integer(row.get("bytes")); valid = _integer(row.get("valid_weight_bytes")) + valid = size if valid is None else valid + request_counts.setdefault(stack, {"arrived": 0, "final_delivered": 0, "censored": 0}) + request_bytes.setdefault(stack, {"arrived_effective": 0, "arrived_physical": 0, + "final_delivered_effective": 0, "final_delivered_physical": 0, + "censored_effective": 0, "censored_physical": 0}) + request_counts[stack]["final_delivered"] += 1 + if size is not None: + request_bytes[stack]["final_delivered_physical"] += size + if valid is not None: + request_bytes[stack]["final_delivered_effective"] += valid + timestamp = _integer(row.get("final_completion_ns")) + latency = _number(row.get("end_to_end_latency_ns")) + if timestamp is not None: + index = _window_index(windows, timestamp) + if index is not None: + if valid is not None: + derived_delivered[(index, stack)] = derived_delivered.get((index, stack), 0) + valid + if latency is not None: + completion_samples.setdefault((index, stack), []).append(latency) + for request_id, row in censored.items(): + stack = str(row.get("stack") or arrivals.get(request_id, {}).get("stack") or "UNKNOWN") + size = _integer(row.get("bytes")); valid = _integer(row.get("valid_weight_bytes")) + valid = size if valid is None else valid + request_counts.setdefault(stack, {"arrived": 0, "final_delivered": 0, "censored": 0}) + request_bytes.setdefault(stack, {"arrived_effective": 0, "arrived_physical": 0, + "final_delivered_effective": 0, "final_delivered_physical": 0, + "censored_effective": 0, "censored_physical": 0}) + request_counts[stack]["censored"] += 1 + if size is not None: + request_bytes[stack]["censored_physical"] += size + if valid is not None: + request_bytes[stack]["censored_effective"] += valid + + control_by_start: dict[int, dict[str, Any]] = {} + for row in control_rows: + start = _integer(row.get("applies_to_window_start_ns")) + if start is not None: + control_by_start[start] = row + thermal_by_end: dict[int, dict[str, Any]] = {} + for row in thermal_rows: + end = _integer(row.get("end_ns")) + if end is not None: + thermal_by_end[end] = row + + time_series: list[dict[str, Any]] = [] + mismatch_windows = 0 + for index, window in enumerate(windows): + facts = [rate_facts[(index, stack)] for stack in stacks if (index, stack) in rate_facts] + duration_s = (window["end_ns"] - window["start_ns"]) / 1e9 + offered = _sum_known(_integer(f.get("offered_bytes")) for f in facts) + delivered = _sum_known(_integer(f.get("delivered_bytes")) for f in facts) + backlog = _sum_known(_integer(f.get("backlog_bytes")) for f in facts) + physical_offered = _sum_known(_integer(f.get("physical_offered_bytes", f.get("offered_bytes"))) + for f in facts) + physical_delivered = _sum_known(_integer(f.get("physical_delivered_bytes", + f.get("delivered_bytes"))) for f in facts) + physical_backlog = _sum_known(_integer(f.get("physical_backlog_bytes", f.get("backlog_bytes"))) + for f in facts) + delivered_hbf = _sum_known(_integer(f.get("delivered_bytes")) for f in facts + if str(f.get("stack_id", "")).startswith("hbf")) + delivered_hbm = _sum_known(_integer(f.get("delivered_bytes")) for f in facts + if str(f.get("stack_id", "")).startswith("hbm")) + incomplete = _sum_known(_integer(f.get("censored_requests")) for f in facts) + raw_sample_count = sum(len(completion_samples.get((index, stack), [])) for stack in stacks) + raw_samples = [v for stack in stacks for v in completion_samples.get((index, stack), [])] + derived_bytes = sum(derived_delivered.get((index, stack), 0) for stack in stacks) + if delivered is not None and int(delivered) != derived_bytes: + mismatch_windows += 1 + + control = control_by_start.get(window["start_ns"]) + decisions = _cell(control.get("stack_decisions")) if control else None + if isinstance(decisions, list): + budgets = [_integer(d.get("budget_bytes")) for d in decisions if isinstance(d, dict)] + permit = _sum_known(budgets) + actions = sorted({str(d.get("action")) for d in decisions if isinstance(d, dict)}) + else: + permit = None; actions = [] + + thermal = thermal_by_end.get(window["end_ns"]) + temperatures = _cell(thermal.get("temperatures")) if thermal else None + groups: dict[str, float | None] = {"gpu": None, "hbf_max": None, "hbm_max": None} + if isinstance(temperatures, dict): + groups["gpu"] = _number(temperatures.get("gpu")) + hbf = [_number(v) for k, v in temperatures.items() if str(k).startswith("hbf")] + hbm = [_number(v) for k, v in temperatures.items() if str(k).startswith("hbm")] + groups["hbf_max"] = max((v for v in hbf if v is not None), default=None) + groups["hbm_max"] = max((v for v in hbm if v is not None), default=None) + relative_residual = None + cumulative_energy = None + energy = _cell(thermal.get("energy_j")) if thermal else None + if isinstance(energy, dict) and isinstance(energy.get("cumulative"), dict): + cumulative = energy["cumulative"] + residual = _number(cumulative.get("energy_residual_j")) + total_input = _number(cumulative.get("total_input_j")) + cumulative_energy = total_input + if residual is not None and total_input not in (None, 0): + relative_residual = abs(residual) / abs(total_input) + + time_series.append({ + **window, + "offered_bytes": None if offered is None else int(offered), + "delivered_bytes": None if delivered is None else int(delivered), + "effective_offered_bytes": None if offered is None else int(offered), + "effective_delivered_bytes": None if delivered is None else int(delivered), + "physical_offered_bytes": None if physical_offered is None else int(physical_offered), + "physical_delivered_bytes": None if physical_delivered is None else int(physical_delivered), + "offered_bytes_per_s": None if offered is None else offered / duration_s, + "delivered_bytes_per_s": None if delivered is None else delivered / duration_s, + "effective_delivered_hbf_bytes": None if delivered_hbf is None else int(delivered_hbf), + "effective_delivered_hbm_bytes": None if delivered_hbm is None else int(delivered_hbm), + "effective_delivered_hbf_bytes_per_s": (None if delivered_hbf is None + else delivered_hbf / duration_s), + "effective_delivered_hbm_bytes_per_s": (None if delivered_hbm is None + else delivered_hbm / duration_s), + "backlog_bytes": None if backlog is None else int(backlog), + "effective_backlog_bytes": None if backlog is None else int(backlog), + "physical_backlog_bytes": None if physical_backlog is None else int(physical_backlog), + "incomplete_or_censored_requests": None if incomplete is None else int(incomplete), + "tail_latency_p95_ns": _quantile(raw_samples, 0.95), + "tail_latency_sample_count": raw_sample_count, + "control_permit_bytes": None if permit is None else int(permit), + "control_actions": actions or None, + "temperatures_k": groups, + "cumulative_energy_input_j": cumulative_energy, + "cumulative_energy_relative_residual": relative_residual, + }) + + active_ns = _integer(manifest.get("active_ns")) + active_rates = [row["delivered_bytes_per_s"] for row in time_series + if row["delivered_bytes_per_s"] is not None and + (active_ns is None or row["start_ns"] < active_ns)] + variability = { + "scope": "FIXED_WINDOWS_INTERSECTING_ACTIVE_INTERVAL", + "window_count": len(active_rates), + "delivered_rate_p05_bytes_per_s": _quantile(active_rates, 0.05), + "delivered_rate_mean_bytes_per_s": statistics.fmean(active_rates) if active_rates else None, + "delivered_rate_stdev_bytes_per_s": statistics.pstdev(active_rates) if active_rates else None, + "quantile_definition": "EMPIRICAL_NEAREST_RANK", + } + if variability["delivered_rate_mean_bytes_per_s"] not in (None, 0): + variability["delivered_rate_cv"] = ( + variability["delivered_rate_stdev_bytes_per_s"] / + variability["delivered_rate_mean_bytes_per_s"]) + else: + variability["delivered_rate_cv"] = None + + required = ["manifest.json", "DONE.json", "requests.csv", "rates.csv", "thermal.csv", + "control.csv", "energy.csv", "maintenance.csv"] + provenance = {name: _sha256(point / name) for name in required if (point / name).exists()} + return { + "schema_version": SCHEMA_VERSION, + "point_id": point.name, + "source_point": str(point), + "execution_status": done.get("execution_status"), + "scope": { + "workload": manifest.get("workload"), + "policy": manifest.get("policy"), + "active_ns": active_ns, + "observation_end_ns": observation_end, + "interpretation": "CONDITIONAL_ENGINEERING_PILOT", + "host_write_workload": "NOHOSTWRITE", + }, + "requests_by_stack": { + stack: {"counts": request_counts.get(stack), "bytes": request_bytes.get(stack)} + for stack in sorted(request_counts) + }, + "fixed_window_incomplete_at_end": time_series[-1]["incomplete_or_censored_requests"], + "explicit_censored_request_count": len(censored), + "rate_variability": variability, + "delivered_totals_by_backend_kind": { + kind: { + "effective_bytes": sum(value["final_delivered_effective"] + for stack, value in request_bytes.items() + if stack.startswith(kind.lower())), + "physical_bytes": sum(value["final_delivered_physical"] + for stack, value in request_bytes.items() + if stack.startswith(kind.lower())), + } for kind in ("HBF", "HBM") + }, + "maintenance": _maintenance_summary(maintenance_rows, manifest, observation_end, + model_page_count), + "cross_checks": { + "rates_vs_request_delivery_mismatch_windows": mismatch_windows, + "window_count": len(windows), + "temperature_row_count": len(thermal_rows), + "control_row_count": len(control_rows), + }, + "time_series": time_series, + "provenance_sha256": provenance, + } + + +def plot_point(result: dict[str, Any], path: Path) -> None: + import matplotlib + matplotlib.use("Agg") + import matplotlib.pyplot as plt + + rows = result["time_series"] + x = [(r["start_ns"] + r["end_ns"]) / 2e9 for r in rows] + active_s = result["scope"]["active_ns"] / 1e9 if result["scope"]["active_ns"] is not None else None + + def series(key: str, scale: float = 1.0) -> list[float]: + values = [] + for row in rows: + value: Any = row + for part in key.split("."): + value = value.get(part) if isinstance(value, dict) else None + values.append(math.nan if value is None else float(value) / scale) + return values + + plt.rcParams.update({"font.size": 8.5, "axes.grid": True, "grid.alpha": 0.22, + "axes.spines.top": False, "axes.spines.right": False}) + fig, axes = plt.subplots(5, 1, figsize=(8.4, 9.4), sharex=True, constrained_layout=True) + fig.suptitle(f"EQ3 closed-loop raw diagnostics — {result['point_id']}", fontsize=12) + + axes[0].plot(x, series("offered_bytes_per_s", 1e9), label="effective offered", color="#777777") + axes[0].plot(x, series("effective_delivered_hbf_bytes_per_s", 1e9), + label="HBF effective delivered", color="#0072B2") + axes[0].plot(x, series("effective_delivered_hbm_bytes_per_s", 1e9), + label="HBM effective delivered", color="#E69F00") + axes[0].set_ylabel("GB/s") + axes[0].legend(ncol=3, frameon=False, loc="upper right") + + axes[1].plot(x, series("backlog_bytes", 2**20), color="#D55E00", label="backlog") + axes[1].set_ylabel("Backlog MiB") + queue2 = axes[1].twinx() + queue2.plot(x, series("incomplete_or_censored_requests"), color="#CC79A7", linestyle="--", + label="incomplete/censored") + queue2.set_ylabel("Requests") + + axes[2].plot(x, series("temperatures_k.gpu"), label="GPU", color="#D55E00") + axes[2].plot(x, series("temperatures_k.hbf_max"), label="HBF max", color="#009E73") + axes[2].plot(x, series("temperatures_k.hbm_max"), label="HBM max", color="#56B4E9") + axes[2].set_ylabel("Temperature K") + axes[2].legend(ncol=3, frameon=False, loc="upper left") + + axes[3].step(x, series("control_permit_bytes", 2**20), where="mid", color="#0072B2") + axes[3].set_ylabel("Permit MiB/window") + maintenance = result["maintenance"] + display = lambda value: "UNKNOWN" if value is None else str(value) + axes[3].text(0.995, 0.08, + f"maint due={display(maintenance['due_count'])} " + f"commit={display(maintenance['committed_count'])} " + f"declared age max={display(maintenance['initial_age_max_s'])}", + ha="right", va="bottom", transform=axes[3].transAxes, fontsize=7.5) + + residual = series("cumulative_energy_relative_residual") + positive = [v if math.isnan(v) or v > 0 else math.nan for v in residual] + axes[4].plot(x, positive, color="#009E73") + axes[4].set_yscale("log") + axes[4].set_ylabel("|energy residual|\n/ input") + axes[4].set_xlabel("Simulation time (s)") + + if active_s is not None: + for axis in axes: + axis.axvline(active_s, color="black", linewidth=0.8, linestyle=":") + axes[0].text(active_s, 1.01, "active end", transform=axes[0].get_xaxis_transform(), + ha="center", va="bottom", fontsize=7) + fig.savefig(path, dpi=180) + plt.close(fig) + + +def _markdown(results: list[dict[str, Any]]) -> str: + lines = [ + "# EQ3 closed-loop point diagnostics", + "", + "This is a raw-derived engineering diagnostic. It does not advance MQSim or the thermal solver.", + "Missing fields remain `UNKNOWN`; observed zero values remain zero.", + "", + "| Point | Workload / policy | active / observed | arrived / delivered / censored | HBF / HBM effective delivered | active rate p05 / mean | tail samples | peak GPU / HBF / HBM K | energy residual max |", + "|---|---|---:|---:|---:|---:|---:|---:|---:|", + ] + for result in results: + rows = result["time_series"] + req = result["requests_by_stack"].values() + arrived = sum(v["counts"]["arrived"] for v in req) + delivered = sum(v["counts"]["final_delivered"] for v in req) + censored = result["explicit_censored_request_count"] + variability = result["rate_variability"] + kinds = result["delivered_totals_by_backend_kind"] + tail_samples = sum(r["tail_latency_sample_count"] for r in rows) + def maximum(key: str) -> float | None: + values = [] + for row in rows: + value: Any = row + for part in key.split("."): + value = value.get(part) if isinstance(value, dict) else None + if value is not None: + values.append(float(value)) + return max(values) if values else None + fmt = lambda v, scale=1.0: "UNKNOWN" if v is None else f"{v/scale:.6g}" + lines.append( + f"| {result['point_id']} | {result['scope']['workload']} / {result['scope']['policy']} | " + f"{fmt(result['scope']['active_ns'],1e9)} / {fmt(result['scope']['observation_end_ns'],1e9)} s | " + f"{arrived} / {delivered} / {censored} | " + f"{kinds['HBF']['effective_bytes']} / {kinds['HBM']['effective_bytes']} B | " + f"{fmt(variability['delivered_rate_p05_bytes_per_s'],1e9)} / " + f"{fmt(variability['delivered_rate_mean_bytes_per_s'],1e9)} GB/s | {tail_samples} | " + f"{fmt(maximum('temperatures_k.gpu'))} / {fmt(maximum('temperatures_k.hbf_max'))} / " + f"{fmt(maximum('temperatures_k.hbm_max'))} | " + f"{fmt(maximum('cumulative_energy_relative_residual'))} |") + intervals = ", ".join( + f"{result['point_id']}: {result['scope']['active_ns']/1e9:g} s active + " + f"{(result['scope']['observation_end_ns']-result['scope']['active_ns'])/1e9:g} s recovery" + for result in results if result["scope"]["active_ns"] is not None) + lines += [ + "", + "## Interpretation boundary", + "", + "- `NOHOSTWRITE`: the current W1 input is read-only. It cannot support a host-write, write-amplification, or write-maintenance claim.", + f"- Window scope: {intervals}. These are conditional engineering points; duration alone does not make a main research result.", + "- Rate p05 uses effective `valid_weight_bytes` (defaulting to physical bytes) and the empirical nearest-rank quantile across fixed windows intersecting the active interval. Tail latency reports its raw completion sample count.", + "- HBF and HBM final-delivery totals remain separate. Their sum is a combined delivered-byte diagnostic and is not labeled HBF goodput.", + "- Temperature traces use GPU plus the maximum reported HBF and HBM stack owner temperatures per window; this intentionally avoids a redundant plot for every sensor.", + "- Maintenance age covers only explicitly declared pages. Age clears only with `mapping_committed=true` and an observed `age_reset_ns`; cleanup failure after commit remains distinct from a pre-commit failure.", + "- Energy relative residual is `abs(cumulative residual) / abs(cumulative total input)` from each thermal window. No missing denominator is replaced with zero.", + "", + ] + return "\n".join(lines) + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--point", action="append", required=True, type=Path, + help="completed point directory; repeat for comparison") + parser.add_argument("--output", required=True, type=Path) + args = parser.parse_args(argv) + output = args.output.resolve() + if output.exists() and any(output.iterdir()): + parser.error(f"output must be absent or empty: {output}") + output.mkdir(parents=True, exist_ok=True) + manifest = { + "schema_version": SCHEMA_VERSION, + "kind": "READ_ONLY_POSTPROCESS", + "source_points": [str(p.resolve()) for p in args.point], + "analyzer_sha256": _sha256(Path(__file__)), + "pid": os.getpid(), + } + (output / "analysis-manifest.json").write_text(json.dumps(manifest, indent=2) + "\n") + try: + results = [analyze_point(point) for point in args.point] + for result in results: + plot_point(result, output / f"{result['point_id']}-closed-loop.png") + payload = {"schema_version": SCHEMA_VERSION, "points": results} + (output / "derived-summary.json").write_text(json.dumps(payload, indent=2, sort_keys=True) + "\n") + (output / "analysis.md").write_text(_markdown(results), encoding="utf-8") + (output / "DONE.json").write_text(json.dumps({ + "execution_status": "COMPLETED", + "point_count": len(results), + "raw_modified": False, + }, indent=2) + "\n") + except Exception as error: + (output / "FAILED.json").write_text(json.dumps({ + "execution_status": "FAILED", "error": f"{type(error).__name__}: {error}" + }, indent=2) + "\n") + raise + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/experiments/eq3_maintenance/analyze_source_domain_replay.py b/experiments/eq3_maintenance/analyze_source_domain_replay.py new file mode 100644 index 0000000..a3d77ec --- /dev/null +++ b/experiments/eq3_maintenance/analyze_source_domain_replay.py @@ -0,0 +1,29 @@ +#!/usr/bin/env python3 +"""Compare full vs own-source responses; sum only linear entity means.""" +import argparse,csv,json,pathlib +p=argparse.ArgumentParser();p.add_argument('--stage',type=pathlib.Path,required=True);p.add_argument('--output',type=pathlib.Path,required=True);a=p.parse_args();a.output.mkdir(parents=True,exist_ok=False) +s=a.stage;full=[] +for row in csv.DictReader((s/'points/Q1-WEIGHT-MAINT-PILOT03/thermal.csv').open()): + full.append(dict(time_ns=int(row['end_ns']),entities=json.loads(row['entity_temperatures_k']))) +domains=['gpu']+[f'{k}{i}' for k in ('hbf','hbm') for i in range(4)];summed=[{k:300.0 for k in row['entities']} for row in full];summary={};count=0 +with (a.output/'domain-observations.csv').open('w') as f: + w=csv.writer(f);w.writerow(['time_ns','domain','component','full_mean_k','own_source_mean_k','cross_source_mean_k','full_hotspot_k','own_source_hotspot_k']) + for d in domains: + point=s/'points'/('A1-SOURCE-'+d+'01');assert (point/'DONE.json').exists() + m=dict(full_peak_k=300.,own_source_peak_k=300.,maximum_cross_source_mean_k=0.,maximum_absolute_full_own_hotspot_difference_k=0.) + rows=0 + for index,line in enumerate((point/'observations.jsonl').open()): + row=json.loads(line);ref=full[index];assert row['end_ns']==ref['time_ns'];rows+=1 + for entity,v in row['entity_temperatures_k'].items(): + summed[index][entity]+=v['mean_k']-300 + if entity.split('.')[0]!=d:continue + rv=ref['entities'][entity];cross=rv['mean_k']-v['mean_k'];count+=1 + w.writerow([ref['time_ns'],d,entity,rv['mean_k'],v['mean_k'],cross,rv['hotspot_k'],v['hotspot_k']]) + m['full_peak_k']=max(m['full_peak_k'],rv['hotspot_k']);m['own_source_peak_k']=max(m['own_source_peak_k'],v['hotspot_k']);m['maximum_cross_source_mean_k']=max(m['maximum_cross_source_mean_k'],cross);m['maximum_absolute_full_own_hotspot_difference_k']=max(m['maximum_absolute_full_own_hotspot_difference_k'],abs(rv['hotspot_k']-v['hotspot_k'])) + assert rows==len(full);summary[d]=m +linear=max(abs(summed[i][k]-v['mean_k']) for i,row in enumerate(full) for k,v in row['entities'].items()) +zero=max(abs(v['mean_k']-300) for line in (s/'points/A1-SOURCE-ZERO01/observations.jsonl').open() for v in json.loads(line)['entity_temperatures_k'].values()) +result=dict(execution_status='COMPLETED',capability_status='SOURCE_DOMAIN_ABLATION_EXECUTED',scientific_scope='Same C/G self response and cooling; removes other source contributions, not a disconnected physical network. No nodewise hotspot summation and no policy reclosure.',windows=len(full),rows=count,domain_results=summary,linear_entity_mean_superposition_max_error_k=linear,zero_source_max_mean_deviation_k=zero) +(a.output/'result.json').write_text(json.dumps(result,indent=2)) +(a.output/'REPORT.md').write_text('# A1 source-domain response ablation\n\nTen actual full-network runs (zero + nine owner-source groups), same original C/G/boundaries/20ms/10s and immutable Pilot03 energy. These are open-loop source-response comparisons, not physically disconnected components or new calibrated thermal models.\n\nLinear entity-mean superposition maximum error: '+str(linear)+' K; zero-source deviation '+str(zero)+' K. Hotspot maxima are compared at the same time but never added as linear observables.\n\n'+ '\n'.join(f'- {d}: full peak {v["full_peak_k"]:.6f} K; own-source peak {v["own_source_peak_k"]:.6f} K; maximum cross-source contribution to entity mean {v["maximum_cross_source_mean_k"]:.6f} K.' for d,v in summary.items())+'\n\nGPU power and operation coefficients remain engineering inputs. This identifies coupling in this model, not a measured product heat budget. The shared passive cooling structure is preserved in every source run.\n') +print(json.dumps(result)) diff --git a/experiments/eq3_maintenance/backend/CMakeLists.txt b/experiments/eq3_maintenance/backend/CMakeLists.txt new file mode 100644 index 0000000..44055e5 --- /dev/null +++ b/experiments/eq3_maintenance/backend/CMakeLists.txt @@ -0,0 +1,88 @@ +cmake_minimum_required(VERSION 3.20) +project(eq3_isolated_mqsim_maintenance LANGUAGES CXX) + +find_package(Git REQUIRED) +set(HBFSIM_SOURCE_ROOT "${CMAKE_CURRENT_LIST_DIR}/../../.." CACHE PATH + "Read-only HBFSim source containing the verified MQSim snapshot and patches 0001-0003") +get_filename_component(HBFSIM_SOURCE_ROOT "${HBFSIM_SOURCE_ROOT}" ABSOLUTE) +set(MQSIM_UPSTREAM "${HBFSIM_SOURCE_ROOT}/third_party/mqsim") +set(MQSIM_EXPECTED_REVISION "51f0f2d3fed92d88ef4a0fa61a38024b07bf9d16") +execute_process(COMMAND "${GIT_EXECUTABLE}" rev-parse HEAD + WORKING_DIRECTORY "${MQSIM_UPSTREAM}" OUTPUT_VARIABLE MQSIM_ACTUAL_REVISION + OUTPUT_STRIP_TRAILING_WHITESPACE RESULT_VARIABLE revision_result) +if(NOT revision_result EQUAL 0 OR NOT MQSIM_ACTUAL_REVISION STREQUAL MQSIM_EXPECTED_REVISION) + message(FATAL_ERROR "isolated maintenance requires MQSim ${MQSIM_EXPECTED_REVISION}; found ${MQSIM_ACTUAL_REVISION}") +endif() +set(MQSIM_EXPERIMENT_SOURCE "${CMAKE_CURRENT_BINARY_DIR}/source/mqsim_eq3_maint") +set(MQSIM_PATCHES + "${HBFSIM_SOURCE_ROOT}/patches/mqsim/0001-online-hbf-api.patch" + "${HBFSIM_SOURCE_ROOT}/patches/mqsim/0002-qlc-support.patch" + "${HBFSIM_SOURCE_ROOT}/patches/mqsim/0003-hbf-command-observer.patch" + "${CMAKE_CURRENT_LIST_DIR}/patches/0004-eq3-maintenance.patch") + +file(REMOVE_RECURSE "${MQSIM_EXPERIMENT_SOURCE}") +file(MAKE_DIRECTORY "${MQSIM_EXPERIMENT_SOURCE}") +file(COPY "${MQSIM_UPSTREAM}/" DESTINATION "${MQSIM_EXPERIMENT_SOURCE}" + PATTERN ".git" EXCLUDE PATTERN "build" EXCLUDE PATTERN "MQSim" EXCLUDE) +foreach(patch IN LISTS MQSIM_PATCHES) + execute_process(COMMAND "${CMAKE_COMMAND}" -E env + "GIT_CEILING_DIRECTORIES=${CMAKE_CURRENT_BINARY_DIR}/source" + "${GIT_EXECUTABLE}" apply --check --unsafe-paths "${patch}" + WORKING_DIRECTORY "${MQSIM_EXPERIMENT_SOURCE}" + RESULT_VARIABLE check_result ERROR_VARIABLE check_error) + if(NOT check_result EQUAL 0) + message(FATAL_ERROR "isolated MQSim patch check failed for ${patch}: ${check_error}") + endif() + execute_process(COMMAND "${CMAKE_COMMAND}" -E env + "GIT_CEILING_DIRECTORIES=${CMAKE_CURRENT_BINARY_DIR}/source" + "${GIT_EXECUTABLE}" apply --unsafe-paths "${patch}" + WORKING_DIRECTORY "${MQSIM_EXPERIMENT_SOURCE}" + RESULT_VARIABLE apply_result ERROR_VARIABLE apply_error) + if(NOT apply_result EQUAL 0) + message(FATAL_ERROR "isolated MQSim patch apply failed for ${patch}: ${apply_error}") + endif() +endforeach() + +file(GLOB_RECURSE MQSIM_SOURCES CONFIGURE_DEPENDS "${MQSIM_EXPERIMENT_SOURCE}/src/*.cpp") +list(FILTER MQSIM_SOURCES EXCLUDE REGEX "/src/main\\.cpp$") +add_library(mqsim_eq3_maint STATIC ${MQSIM_SOURCES}) +target_compile_features(mqsim_eq3_maint PUBLIC cxx_std_17) +target_include_directories(mqsim_eq3_maint PUBLIC + "${MQSIM_EXPERIMENT_SOURCE}/src" "${MQSIM_EXPERIMENT_SOURCE}/src/exec" + "${MQSIM_EXPERIMENT_SOURCE}/src/host" "${MQSIM_EXPERIMENT_SOURCE}/src/nvm_chip" + "${MQSIM_EXPERIMENT_SOURCE}/src/nvm_chip/flash_memory" + "${MQSIM_EXPERIMENT_SOURCE}/src/sim" "${MQSIM_EXPERIMENT_SOURCE}/src/ssd" + "${MQSIM_EXPERIMENT_SOURCE}/src/utils") +set_target_properties(mqsim_eq3_maint PROPERTIES + ARCHIVE_OUTPUT_DIRECTORY "${CMAKE_CURRENT_BINARY_DIR}/lib") + +add_library(hbfsim_eq3_maint_adapter STATIC + "${HBFSIM_SOURCE_ROOT}/src/profile/profile.cpp" + "${CMAKE_CURRENT_LIST_DIR}/src/mqsim_online_maintenance.cpp") +target_compile_features(hbfsim_eq3_maint_adapter PUBLIC cxx_std_20) +target_include_directories(hbfsim_eq3_maint_adapter PUBLIC + "${CMAKE_CURRENT_LIST_DIR}/include" "${HBFSIM_SOURCE_ROOT}/include" + "${HBFSIM_SOURCE_ROOT}/third_party/bpftime/third_party") +target_link_libraries(hbfsim_eq3_maint_adapter PUBLIC mqsim_eq3_maint) +set_target_properties(hbfsim_eq3_maint_adapter PROPERTIES + ARCHIVE_OUTPUT_DIRECTORY "${CMAKE_CURRENT_BINARY_DIR}/lib") + +add_executable(hbf_mqsim_eq3_maint + "${CMAKE_CURRENT_LIST_DIR}/service/hbf_mqsim_maintenance_service.cpp") +target_compile_features(hbf_mqsim_eq3_maint PRIVATE cxx_std_20) +target_include_directories(hbf_mqsim_eq3_maint PRIVATE + "${CMAKE_CURRENT_LIST_DIR}/service" + "${HBFSIM_SOURCE_ROOT}/third_party/bpftime/third_party") +target_link_libraries(hbf_mqsim_eq3_maint PRIVATE hbfsim_eq3_maint_adapter) +set_target_properties(hbf_mqsim_eq3_maint PROPERTIES + RUNTIME_OUTPUT_DIRECTORY "${CMAKE_CURRENT_BINARY_DIR}/bin") + +enable_testing() +add_executable(eq3_maintenance_backend_tests + "${CMAKE_CURRENT_LIST_DIR}/tests/maintenance_backend_tests.cpp") +target_compile_features(eq3_maintenance_backend_tests PRIVATE cxx_std_20) +target_link_libraries(eq3_maintenance_backend_tests PRIVATE hbfsim_eq3_maint_adapter) +set_target_properties(eq3_maintenance_backend_tests PROPERTIES + RUNTIME_OUTPUT_DIRECTORY "${CMAKE_CURRENT_BINARY_DIR}/tests") +add_test(NAME eq3_maintenance_backend COMMAND eq3_maintenance_backend_tests + "${HBFSIM_SOURCE_ROOT}/configs/profiles/nominal.json") diff --git a/experiments/eq3_maintenance/backend/client/maintenance_service.py b/experiments/eq3_maintenance/backend/client/maintenance_service.py new file mode 100644 index 0000000..665e3fa --- /dev/null +++ b/experiments/eq3_maintenance/backend/client/maintenance_service.py @@ -0,0 +1,80 @@ +"""Client for the isolated EQ3 MQSim maintenance experiment service.""" +from __future__ import annotations + +from scripts.eval.mqsim_service import MqsimService + + +class MaintenanceMqsimService(MqsimService): + """Preserves the demand API and adds page-maintenance lifecycle facts. + + The caller still advances MQSim exclusively with ``until(horizon)``. A + maintenance request is accepted for later injection; this method does not + run the event loop or wait for completion. + """ + + def __init__(self, *args, **kwargs): + self.maintenance_requests = {} + self.maintenance_events = [] + self.maintenance_completions = {} + super().__init__(*args, **kwargs) + if (self.header.get("maintenance_backend") != + "EXPERIMENTAL_OUT_OF_PLACE_PAGE_MAINTENANCE" + or self.header.get("maintenance_data_semantics") != + "METADATA_VERSION_VALIDITY" + or self.header.get("maintenance_version_semantics") != + "MONOTONIC_LPA_MAPPING_GENERATION_U64" + or self.header.get("maintenance_payload_validation") != "UNAVAILABLE"): + self.close() + raise ValueError("isolated maintenance service capability mismatch") + + def read(self): + response = super().read() + self.maintenance_events.extend(response.get("maintenance_events", [])) + for completion in response.get("maintenance_completions", []): + request_id = completion["request_id"] + if (request_id not in self.maintenance_requests + or request_id in self.maintenance_completions): + raise ValueError("duplicate or unknown maintenance completion") + self.maintenance_completions[request_id] = completion + return response + + def maintain(self, job=None, **kwargs): + """Accept either one job dict or equivalent keyword arguments.""" + if job is not None: + if not isinstance(job, dict) or kwargs: + raise TypeError("maintain accepts one job dict or keyword arguments") + kwargs = dict(job) + request_id = kwargs.pop("request_id") + stack = kwargs.pop("stack") + stack_local_page = kwargs.pop("stack_local_page") + due_ns = kwargs.pop("due_ns") + deadline_ns = kwargs.pop("deadline_ns", 0) + parent_id = kwargs.pop("parent_id", None) + reclaim_source_block = kwargs.pop("reclaim_source_block", False) + failure_injection = kwargs.pop("failure_injection", "none") + trigger_reason = kwargs.pop("trigger_reason", "UNSPECIFIED") + if kwargs: + raise TypeError(f"unknown maintenance fields: {sorted(kwargs)}") + if request_id in self.maintenance_requests: + raise ValueError("duplicate maintenance request ID") + request = dict(command="maintain", request_id=request_id, + parent_id=request_id if parent_id is None else parent_id, + stack=stack, stack_local_page=stack_local_page, + due_ns=due_ns, deadline_ns=deadline_ns, + reclaim_source_block=reclaim_source_block, + failure_injection=failure_injection, + trigger_reason=trigger_reason) + response = self.command(request) + if response.get("maintenance_accepted") != 1: + raise ValueError("maintenance service did not accept one request") + self.maintenance_requests[request_id] = request + return response + + def finish(self): + receipt = super().finish() + if (set(self.maintenance_requests) != set(self.maintenance_completions) + or receipt.get("maintenance_issued") != len(self.maintenance_requests) + or receipt.get("maintenance_completed") != len(self.maintenance_requests) + or receipt.get("pending_maintenance") != 0): + raise ValueError("maintenance finish conservation failed") + return receipt diff --git a/experiments/eq3_maintenance/backend/docs/IMPLEMENTATION_BOUNDARY.md b/experiments/eq3_maintenance/backend/docs/IMPLEMENTATION_BOUNDARY.md new file mode 100644 index 0000000..e0fd2ce --- /dev/null +++ b/experiments/eq3_maintenance/backend/docs/IMPLEMENTATION_BOUNDARY.md @@ -0,0 +1,235 @@ +# Isolated MQSim maintenance backend + +## Evidence classification and scope + +- **USER_CONFIRMED:** the isolated experimental fork may add the narrow + maintenance source, queue visibility, native read/program/erase path, + destination allocation, version commit, and failure cleanup described by + `EQ3-ISOLATED-MAINTENANCE-CAMPAIGN-v1`. +- **DOC_DERIVED:** the source base is MQSim + `51f0f2d3fed92d88ef4a0fa61a38024b07bf9d16` followed by the registered + HBFSim patches 0001--0003 and local patch 0004. Exact hashes are in + `patches/series.json`. +- **SCENARIO_ASSUMPTION:** this is + `EXPERIMENTAL_OUT_OF_PLACE_PAGE_MAINTENANCE`. It is not asserted to be a + universal HBF product/OCP refresh algorithm. +- **Capability:** `METADATA_VERSION_VALIDITY`. MQSim has no payload buffer or + payload hash here, so byte equality is `UNAVAILABLE` and is never inferred. + +No tracked default file under `third_party/mqsim`, `patches/mqsim`, `include`, +`src`, or the default CMake graph is changed. The executable is not on a +default lookup path and there is no fallback from the production service. + +## Ownership and state transition + +One `MqsimOnlineEngine` owns foreground requests and the maintenance unit. The +unit receives the same FTL address mapper, block manager, TSU, and ONFI PHY as +foreground work. Its explicit source identity is `HBF_MAINTENANCE`; the +existing TSU policies place it in the existing GC/WL queue class without +changing the scheduler or foreground priority algorithm. Every native child +transaction retains both maintenance request ID and parent ID. + +The successful path is: + +`DUE -> QUEUED -> READ -> PROGRAM_DEST -> COMMIT -> RETIRE_OLD -> [RECLAIM_ERASE] -> DONE` + +The old mapping remains authoritative through read and destination program. +Every logical page has a 64-bit monotonic mapping generation in the isolated +page-mapping domain. Every real mapping update increments it and fails closed +before mutation at overflow. Maintenance captures both source PPA and +generation. After the real program callback, commit requires both to match, +then increments the generation, updates the mapping, and invalidates the old +page. A failed read leaves no destination. Failed program or stale compare +invalidates the allocated destination and retains the old source. An erase +failure happens after mapping commit, retains the new mapping, and reports +`FAILED_AFTER_COMMIT_NEEDS_RECONCILE`. + +Destination pages come from the existing finite GC write frontier. The source +block receives an explicit maintenance pin before the native read and holds it +until terminal cleanup or an atomic transition to erase ownership. Existing +GC eligibility rejects pinned blocks. The source block is erased only when all +written pages are invalid, it has no active +read/program/erase or GC reference, and it is not a data/GC/translation write +frontier. A requested erase that is not safe reports +`COMMITTED_RECLAIM_DEFERRED`; it is not treated as free capacity. Existing +MQSim GC remains responsible for general partially valid-block reclamation. + +The isolated patch also closes MQSim online-HBF first-read bookkeeping. That +path creates metadata for an already populated page by using the data allocator +but has no program command that could clear the allocator's synthetic +in-flight-program count. Patch 0004 calls the existing program-serviced +bookkeeping hook immediately after that metadata allocation. It adds no NAND +command, time, or energy and is limited to the isolated fork. Without this +repair safe erase was always deferred; the preserved failure is +`backend-cpp-test-v5` and the passing repair evidence is +`backend-cpp-test-v6`/`v7`. + +## Service contract + +The executable preserves `submit`, `try_submit`, `until`, and `finish`. +`maintain` accepts one page: + +```json +{ + "command": "maintain", + "request_id": 101, + "parent_id": 7001, + "stack": "hbf0", + "stack_local_page": 15, + "due_ns": 120000, + "deadline_ns": 1120000, + "trigger_reason": "FIXED_RETENTION_DUE", + "reclaim_source_block": false, + "failure_injection": "none" +} +``` + +An explicit direct-HBF stack map is mandatory. The persistent stack-local page +mapping supplies the backend LPA; CWDP then supplies exact channel/die/plane. +The service does not guess a stack from an address. `maintain` only schedules +the due event. The caller must advance the same engine with `until(horizon)`; +there is no host sleep, inner retry loop, callback re-entry, or second engine. +`until` continues to return only foreground completions. Maintenance facts are +separate `maintenance_events` and `maintenance_completions` arrays on every +response. + +Completion contains status, enqueue/start/end, source and committed versions, +commit/source-retire/erase flags, native transaction IDs, exact stack/local +page, trigger, one-page coverage, deadline result, and `age_reset_ns`. The last +field is non-null only after mapping commit and covers exactly that page. +`deadline_ns=0` means no deadline. Fault injection is an engineering test +facility (`none/read/program/stale_commit/erase`), never a measured device +failure model. + +The dedicated client is +`client/maintenance_service.py::MaintenanceMqsimService`. It accepts either +`maintain(job_dict)` or equivalent keyword arguments. It retains cumulative +`maintenance_events` and `maintenance_completions[id]`; `until` still returns +only a foreground completion. `finish` enforces zero pending maintenance and +one terminal completion per accepted maintenance ID. + +## Verified and unavailable behavior + +Fixed CPU evidence: + +- `AB-PROCESS-GENERATION-02`: default backend A versus final generation/pin + backend B with maintenance off. Both the 8x1 24-request fixture and Q1 + 4-channel x 16-die 12-request fixture matched request/status, media and + reported completion time, native command order, and placement at zero-ns + tolerance; both stderr logs were empty. +- `backend-generation-cpp-test-v1`: maintenance-off foreground completion + equality; native read/program/generation-CAS/retire/erase; monotonic source + and committed generations; a real future-arrival foreground MQSim write + interleaved with maintenance and won exactly once while stale maintenance + discarded its destination; source/new mapping readability; + read, program, stale-CAS, and post-commit erase failures; source/destination + readability; unique parent completion; real transaction IDs; 1 GiB finite + workspace with one channel and all 16 dies, including die 15. +- `backend-generation-service-test-v2`: final generation backend process protocol, horizon + advancement, explicit + stack placement to channel 0/die 15, parent/native IDs, lifecycle, age commit, + trigger, one-page coverage, deadline result, and finish conservation. +- `backend-generation-build-v1` plus the capability-only incremental + `backend-generation-build-v2`: isolated source recreation and binary build + passed with at most two compile threads. Binary SHA-256 is + `c64610b0f281397975e649b32f44161b00240f12d6a49b9b4d4776b1f554c257`. +- `BACKEND-INDEPENDENT-REBUILD01`: a fresh, separate build directory recreated + all 179 patched source files byte-for-byte, linked exactly one isolated MQSim + archive and no default archive, and completed configuration/build. Its ELF + hash differs only because absolute source/build paths change the build ID. + `AB-PROCESS-INDEPENDENT-REBUILD01` then passed the same maintenance-off 8x1 + and 4x16 process comparisons at zero-ns tolerance. +- `Q1-WEIGHT-MAINT-PILOT03`: the legacy 16 KiB finite-region profile completed + 3,712 HBF foreground reads, 320 parametric-HBM reads and 64 committed + maintenance jobs through the actual 10 s coordinator/thermal loop. +- `OCP4K-LOOP-PILOT01`: the 4 KiB/full-logical-capacity profile completed + 7,100 HBF reads, 160 parametric-HBM reads and 64 committed maintenance jobs + through the actual 6 s loop. All 7,260 foreground requests completed once; + the engine finished with no foreground or maintenance work pending. +- `A2-PILOT03-LOOP01`: real foreground MQSim completed the frozen 4,032 + PILOT03 requests, while the counterfactual wrapper replayed the fixed 64-job + maintenance bundle on ideal independent resources. The actual backend + issued and committed zero maintenance jobs and its current mapping was not + mutated. Ten foreground requests completed 620 ns earlier than the actual + shared-maintenance PILOT03 baseline; completion count, censoring, P95 + latency, peak temperature and energy were unchanged. This is executed + counterfactual evidence, not an additional backend capability. +- The prior v11 PPA-token source and binary remain immutable under + `points/BACKEND-V11-FROZEN`; they are historical evidence only. + +## Geometry and capacity boundary + +The two executed profiles are not interchangeable: + +- The historical 16 KiB profile has one channel per HBF stack, 16 MQSim dies + per channel, one plane per die, and a finite 1 GiB/stack allocator region. +- The OCP study profile has 4 KiB pages and instantiates four logical + 512 GiB stacks. Each stack has 16 channels, one MQSim die per channel, + 16 planes per die, 256 pages/block and 2,048 blocks/plane. The actual + service therefore creates 1,024 parallel channel/plane units across four + stacks. The full-capacity probe covered all 1,024 configured + `(stack, thermal-die, projected-plane)` tuples and completed real + read/program/commit maintenance. + +The OCP 16-bank value is represented as +`BANK_AS_MQSIM_PLANE_V1_SCENARIO_PROJECTION_NOT_PRODUCT_IDENTITY`. A channel's +position within its configured stack supplies thermal `die0..die15`; the +native MQSim die field remains zero because there is one die per channel. +This projection is explicit configuration, not an observation that an OCP bank +is universally a NAND plane. + +CWDP selects channel/die/plane from the backend logical page. MQSim's +page-level FTL then allocates a physical block and page from its own write +frontier. `GEOMETRY4K-PHYSICAL-BLOCK02` observed 256 serial first touches in +one plane occupy physical block 0 pages 0..255, followed by block 5 page 0; +block 5 was observed rather than prescribed because MQSim reserves other +frontiers. This consumes the configured pages/block value, but it is not the +complete OCP zone/direct block-addressing protocol. + +The 512 GiB/stack figure is a full *logical namespace* instantiation. MQSim +has a finite allocator and spare destinations inside that instantiated +geometry, but neither OCP material nor this run establishes target-device +physical spare, bad-block reserve or overprovisioning. The 256 pages/block +value and bank-to-plane projection remain scenario assumptions. + +## Downstream identity and energy boundary + +The service emits `maintenance_request_id` on real maintenance child +transactions. The downstream `ActivityEnergyLedger._source` now consumes +that actual field first and retains `maintenance_id` only as a legacy +fallback. Fixed source-classification tests pass, and the OCP4K loop records +320 maintenance activity rows as `HBF_MAINTENANCE`. + +PILOT03 was produced before that downstream repair, so its 260 maintenance and +other backend-background rows remain immutably labelled +`BACKEND_BACKGROUND`. The repair changes only source classification: +PILOT03's per-window component joules are unchanged, which is why its A3 +full-2 mm paired replay remains comparable. Energy coefficients are still +engineering assumptions. + +Payload validation, die-wide refresh, HBM refresh, ECC/RBER, read-disturb, +wear-life prediction, zone-standard conformance, and cross-process checkpoint +restore are `UNAVAILABLE`. Run artifacts and transcripts are isolated and +restartable, but engine state is not serialized; no checkpoint claim is made. +The JSON experiment service intentionally accepts foreground reads only because +the campaign workload is read-only weights; host write is not a campaign +feature. The lower-level engine does support writes, and the fixed safety test +uses a real scheduled foreground write and real NAND program to validate stale +generation rejection. The maintenance path does not reuse MQSim's GC LPA +barrier, whose upstream release path approximates waiting writes. A source +block pin prevents GC erase/reuse, and the generation prevents PPA ABA from +being accepted as the captured logical version. +Neither successful logical-capacity instantiation nor the completed loops +validate product throughput, calibrated maintenance energy, physical spare, or +payload preservation. + +The actual maintenance path and the A2 replay must remain distinct. In +`Q1-WEIGHT-MAINT-PILOT03` and `OCP4K-LOOP-PILOT01`, the isolated engine +receives maintenance commands, schedules native reads and destination programs, +commits mapping generations, and returns `age_reset_ns` only after commit. +In `A2-PILOT03-LOOP01`, foreground still uses that real engine, but +maintenance is disabled at the service wrapper. Fixed source phases, energy, +terminal facts and virtual age facts are replayed without native commands, +resource ownership or mapping mutation; their mapping/age semantics are +`UNKNOWN_REPLAY`. A2 therefore measures one bounded ideal-resource +counterfactual and does not implement or validate an independent scheduler. diff --git a/experiments/eq3_maintenance/backend/docs/REBUILD_AND_VALIDATE.md b/experiments/eq3_maintenance/backend/docs/REBUILD_AND_VALIDATE.md new file mode 100644 index 0000000..92dc5e4 --- /dev/null +++ b/experiments/eq3_maintenance/backend/docs/REBUILD_AND_VALIDATE.md @@ -0,0 +1,39 @@ +# Rebuild and fixed validation + +Run from the integration checkout. Use the system CMake explicitly because an +obsolete vendor CMake earlier failed before configuration; that preserved +launcher failure is `backend-build-v1`. + +```sh +/usr/bin/cmake -S experiments/eq3_maintenance/backend \ + -B experiments/eq3_maintenance/backend/build-isolated -G Ninja +/usr/bin/cmake --build experiments/eq3_maintenance/backend/build-isolated \ + --parallel 2 +``` + +Configuration verifies the upstream MQSim revision, creates a fresh writable +copy only inside `build-isolated/source/mqsim_eq3_maint`, checks and applies +patches 0001--0004 in order, and uses distinct library and executable names. +The resulting executable is: + +`experiments/eq3_maintenance/backend/build-isolated/bin/hbf_mqsim_eq3_maint` + +Fixed C++ validation: + +```sh +experiments/eq3_maintenance/backend/build-isolated/tests/eq3_maintenance_backend_tests \ + configs/profiles/nominal.json +``` + +Process protocol validation requires a new artifact directory: + +```sh +python3 experiments/eq3_maintenance/backend/tests/maintenance_service_test.py \ + "$PWD" \ + "$PWD/experiments/eq3_maintenance/backend/build-isolated/bin/hbf_mqsim_eq3_maint" \ + /absolute/path/to/new-test-artifact +``` + +These are small engineering checks. They are not research campaign points and +do not validate the thermal model or full-capacity product geometry. + diff --git a/experiments/eq3_maintenance/backend/include/hbfsim/eq3_thermal/mqsim_observer.hpp b/experiments/eq3_maintenance/backend/include/hbfsim/eq3_thermal/mqsim_observer.hpp new file mode 100644 index 0000000..e693174 --- /dev/null +++ b/experiments/eq3_maintenance/backend/include/hbfsim/eq3_thermal/mqsim_observer.hpp @@ -0,0 +1,154 @@ +#pragma once +#include +#include +#include +#include +#include +#include +#include +#include +#include + +namespace hbfsim::eq3_thermal { +// Existing MQSim API exposes request occupancy, NOT NAND command start/die/plane. +// Mapping/power is an explicit caller-supplied proxy or fixture; never inferred +// by hashing addresses or presented as measured command-level activity. +class MqsimObserverAdapter { + public: + MqsimObserverAdapter(MqsimOnlineEngine& engine,ActivityObserver& observer) + :engine_(engine),observer_(observer) { + if(observer.mode()!=RuntimeMode::Off)engine_.enable_observations(); + } + void bind(std::uint64_t request_id,ObservedEvent metadata) { + if(observer_.mode()==RuntimeMode::Off)return; + if(metadata.evidence!="ENGINEERING_FIXTURE_REQUEST_OCCUPANCY") + throw std::invalid_argument("MQSim request occupancy energy is not NAND command measurement"); + metadata.request_id=std::to_string(request_id); + metadata.operation_id="mqsim:"+metadata.request_id; + if(!bindings_.emplace(request_id,std::move(metadata)).second) + throw std::invalid_argument("duplicate MQSim request binding"); + } + void drain_to_current_time() { + if(observer_.mode()!=RuntimeMode::Off)for(const auto& event:engine_.take_observations()) { + const auto it=bindings_.find(event.request_id); + if(it==bindings_.end())throw std::invalid_argument("missing explicit MQSim activity mapping"); + auto value=it->second;value.logical_bytes=event.bytes; + auto emit=[&](Phase phase,std::uint64_t time,const char* tag) { + value.phase=phase;value.time_ns=time;value.event_id=value.operation_id+":"+tag; + observer_.submit(value); + }; + if(event.kind==MqsimEventKind::Arrival)emit(Phase::Arrival,event.time_ns,"arrival"); + else if(event.kind==MqsimEventKind::Admission) { + emit(Phase::Issue,event.time_ns,"admission"); + emit(Phase::Start,event.time_ns,"occupancy-start-not-nand"); + } else { + emit(Phase::End,event.time_ns,"media-end"); + value.outcome="success"; + emit(Phase::Complete,event.modeled_completion_ns,"reported-complete"); + } + } + observer_.advance_to(engine_.current_time_ns()); + } + private: + MqsimOnlineEngine& engine_; + ActivityObserver& observer_; + std::map bindings_; +}; + +// Optional pre-submit composition boundary. The decision callback is pure with +// respect to MQSim: it must not submit requests, advance the engine, or drain +// observations. In particular, this adapter owns no event loop and buffers no +// completions. A caller handles DEFER by advancing through the existing +// run_next_completion_until() API, draining observations, and trying again. +enum class MqsimGateMode { Off, Enabled }; +enum class MqsimGateDisposition { Allow, Defer, Blocked, Unsupported }; + +struct MqsimGateDecision { + MqsimGateDisposition disposition{MqsimGateDisposition::Allow}; + std::optional target_time_ns; + std::string reason; +}; + +struct MqsimGateAttempt { + MqsimGateDecision decision; + bool submitted{}; + std::uint64_t original_arrival_ns{}; + std::uint64_t evaluated_ns{}; + std::optional backend_arrival_ns; + std::optional external_wait_ns; +}; + +enum class MqsimBackendOperation { Demand, DieLevelMaintenance }; +struct MqsimBackendCapability { + bool supported{}; + std::string status; + std::string detail; +}; + +using MqsimAdmissionGate = + std::function; + +class MqsimSubmissionGateAdapter { + public: + MqsimSubmissionGateAdapter(MqsimOnlineEngine& engine, + MqsimGateMode mode=MqsimGateMode::Off, + MqsimAdmissionGate gate={}) + :engine_(engine),mode_(mode),gate_(std::move(gate)) { + if(mode_==MqsimGateMode::Enabled&&!gate_) + throw std::invalid_argument("enabled MQSim submission gate needs a decision callback"); + } + + MqsimGateAttempt try_submit(const HbfRequest& request) { + const auto now=engine_.current_time_ns(); + if(mode_==MqsimGateMode::Off) { + engine_.submit(request); + return {{MqsimGateDisposition::Allow,std::nullopt,"gate disabled"},true, + request.arrival_ns,now,request.arrival_ns,0}; + } + if(submitted_.contains(request.request_id)) + throw std::logic_error("MQSim gate request was already submitted"); + // Do not make a control decision before this request exists in target + // simulation time: the controlled state may change before its arrival. + if(request.arrival_ns>now) + return {{MqsimGateDisposition::Defer,request.arrival_ns, + "request has not reached its target arrival time"},false, + request.arrival_ns,now,std::nullopt,std::nullopt}; + + auto decision=gate_(request,now); + if(decision.disposition==MqsimGateDisposition::Allow) { + if(decision.target_time_ns) + throw std::invalid_argument("ALLOW decision cannot carry a target time"); + auto backend=request; + // MQSim rejects arrivals before its current clock. Preserve that external + // wait in the attempt ledger and leave the backend completion untouched. + backend.arrival_ns=std::max(request.arrival_ns,now); + engine_.submit(backend); + submitted_.insert(request.request_id); + return {std::move(decision),true,request.arrival_ns,now, + backend.arrival_ns,backend.arrival_ns-request.arrival_ns}; + } + if(decision.reason.empty()) + throw std::invalid_argument("non-ALLOW MQSim gate decision needs a reason"); + if(decision.disposition==MqsimGateDisposition::Defer) { + if(!decision.target_time_ns||*decision.target_time_ns<=now) + throw std::invalid_argument("DEFER target must follow current target time"); + } else if(decision.target_time_ns) { + throw std::invalid_argument("BLOCKED/UNSUPPORTED decision cannot carry a target time"); + } + return {std::move(decision),false,request.arrival_ns,now,std::nullopt,std::nullopt}; + } + + static MqsimBackendCapability capability(MqsimBackendOperation operation) { + if(operation==MqsimBackendOperation::Demand) + return {true,"SUPPORTED","optional pre-submit demand gate"}; + return {false,"UNSUPPORTED_CAPABILITY", + "MQSimOnlineEngine exposes no die-level maintenance operation or completion"}; + } + + private: + MqsimOnlineEngine& engine_; + MqsimGateMode mode_; + MqsimAdmissionGate gate_; + std::set submitted_; +}; +} // namespace hbfsim::eq3_thermal diff --git a/experiments/eq3_maintenance/backend/include/hbfsim/eq3_thermal/mqsim_stack_map.hpp b/experiments/eq3_maintenance/backend/include/hbfsim/eq3_thermal/mqsim_stack_map.hpp new file mode 100644 index 0000000..4a1ee5c --- /dev/null +++ b/experiments/eq3_maintenance/backend/include/hbfsim/eq3_thermal/mqsim_stack_map.hpp @@ -0,0 +1,141 @@ +#pragma once + +#include +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include + +namespace hbfsim::eq3_thermal { + +struct MqsimStackChannelGroup { + std::string stack_id; + std::vector channels; + std::uint32_t declared_dies{}; +}; + +struct MqsimStackPlacement { + HbfRequest backend_request{}; + std::uint64_t external_page{}; + std::uint64_t backend_page{}; + std::optional stack_id; + std::optional expected_channel; +}; + +// Optional one-page address-layout composition for same-kind HBF stacks. The +// transform is a persistent bijection, not dynamic polling or rebalancing. It +// does not submit, retry, reserve, or own work. Enabled mappings must be used +// exclusively for the lifetime of their MQSim engine because mapped and +// unmapped LPAs are distinct backend namespaces. +class MqsimStackMapAdapter { + public: + MqsimStackMapAdapter() = default; + + MqsimStackMapAdapter(const Profile& profile, + std::vector groups) + : page_bytes_(profile.page_bytes), capacity_bytes_(profile.capacity_bytes), + channels_(profile.channels), + dies_per_channel_(profile.dies_per_channel), groups_(std::move(groups)) { + if(profile.plane_allocation_scheme!=PlaneAllocationScheme::Cwdp) + throw std::invalid_argument("MQSim stack map requires CWDP allocation"); + if(!page_bytes_||!channels_||groups_.size()<2||channels_%groups_.size()) + throw std::invalid_argument("invalid MQSim stack-map geometry"); + channels_per_stack_=channels_/groups_.size(); + channel_to_stack_.resize(channels_); + for(std::size_t s=0;s=channels_||channel_to_stack_[channel]) + throw std::invalid_argument("overlapping or invalid MQSim channel group"); + channel_to_stack_[channel]=s; + } + } + for(const auto& owner:channel_to_stack_)if(!owner) + throw std::invalid_argument("MQSim stack map must cover every channel"); + enabled_=true; + } + + [[nodiscard]] bool enabled()const noexcept{return enabled_;} + + [[nodiscard]] MqsimStackPlacement map( + const HbfRequest& request, + std::optional requested_stack=std::nullopt)const { + if(!enabled_) + return {request,request.logical_address/page_bytes_, + request.logical_address/page_bytes_,std::nullopt,std::nullopt}; + if(request.bytes!=page_bytes_||request.logical_address%page_bytes_) + throw std::invalid_argument("MQSim stack map supports one aligned profile page"); + const auto page=request.logical_address/page_bytes_; + if(page>=capacity_bytes_/page_bytes_) + throw std::out_of_range("MQSim stack-map request exceeds capacity"); + const auto s=static_cast(page%groups_.size()); + const auto q=page/groups_.size(); + const auto k=static_cast(q%channels_per_stack_); + const auto r=q/channels_per_stack_; + const auto channel=groups_[s].channels[k]; + if(requested_stack&&*requested_stack!=groups_[s].stack_id) + throw std::invalid_argument("requested HBF stack conflicts with configured address stripe"); + if(r>(std::numeric_limits::max()-channel)/channels_) + throw std::overflow_error("MQSim backend page mapping overflows"); + const auto backend_page=r*channels_+channel; + if(backend_page>std::numeric_limits::max()/page_bytes_) + throw std::overflow_error("MQSim backend byte address overflows"); + auto backend=request; + backend.logical_address=backend_page*page_bytes_; + return {backend,page,backend_page,groups_[s].stack_id,channel}; + } + + // Convert a persistent page ordinal inside one explicitly named HBF stack + // to the global striped namespace, then apply the same bijection as map(). + [[nodiscard]] MqsimStackPlacement map_stack_page( + const HbfRequest& request,std::string_view requested_stack, + std::uint64_t stack_local_page)const { + if(!enabled_)throw std::logic_error("MQSim stack-local placement requires enabled mapping"); + std::size_t stack_index=groups_.size(); + for(std::size_t s=0;s(std::numeric_limits::max()-stack_index)/groups_.size()) + throw std::overflow_error("MQSim external page mapping overflows"); + const auto external_page=stack_local_page*groups_.size()+stack_index; + if(external_page>std::numeric_limits::max()/page_bytes_) + throw std::overflow_error("MQSim external byte address overflows"); + auto external=request; + external.logical_address=external_page*page_bytes_; + return map(external,requested_stack); + } + + [[nodiscard]] std::optional stack_for_channel( + std::uint32_t channel)const { + if(!enabled_||channel>=channel_to_stack_.size()||!channel_to_stack_[channel]) + return std::nullopt; + return groups_[*channel_to_stack_[channel]].stack_id; + } + + private: + bool enabled_{}; + std::uint64_t page_bytes_{1}; + std::uint64_t capacity_bytes_{}; + std::size_t channels_{}; + std::size_t dies_per_channel_{}; + std::size_t channels_per_stack_{}; + std::vector groups_; + std::vector> channel_to_stack_; +}; + +} // namespace hbfsim::eq3_thermal diff --git a/experiments/eq3_maintenance/backend/include/hbfsim/eq3_thermal/observer.hpp b/experiments/eq3_maintenance/backend/include/hbfsim/eq3_thermal/observer.hpp new file mode 100644 index 0000000..f0634f1 --- /dev/null +++ b/experiments/eq3_maintenance/backend/include/hbfsim/eq3_thermal/observer.hpp @@ -0,0 +1,68 @@ +#pragma once +#include +#include +#include + +namespace hbfsim::eq3_thermal { +// Target simulation nanoseconds only. Host/wall time never enters this clock. +enum class Phase { Arrival, Issue, Start, End, Fail, Cancel, Complete }; +struct ObservedEvent { + std::string event_id, operation_id, request_id; + Phase phase{Phase::Arrival}; + std::uint64_t time_ns{}; + std::string operation, source, physical_type, stack_id; + std::optional die, plane; + std::uint64_t logical_bytes{}; + std::optional physical_bytes, link_bytes; + std::vector resources; + std::map component_power_w; + double external_power_w{}; // explicitly outside the package thermal domain + std::string evidence, outcome; + bool operator==(const ObservedEvent&) const = default; +}; +struct ObserverCheckpoint { + RuntimeMode mode{}; + std::uint64_t time_ns{}, next_energy_id{}; + std::map seen, active; + std::vector pending, journal, terminals; + std::map energy_j; + double external_energy_j{}; + std::optional thermal; +}; +// A consumer of actual service events, not a service scheduler. Off bypasses +// collection and solver construction; read-only validates/accounts but never solves. +class ActivityObserver { + public: + ActivityObserver(RuntimeMode mode, ThermalModelConfig config); + bool submit(const ObservedEvent& event); + void advance_to(std::uint64_t target_ns); + RuntimeMode mode() const noexcept { return mode_; } + bool solver_constructed() const noexcept { return runtime_.solver_constructed(); } + std::uint64_t time_ns() const noexcept { return time_ns_; } + const auto& journal() const noexcept { return journal_; } + const auto& terminals() const noexcept { return terminals_; } + const auto& energy_j() const noexcept { return energy_j_; } + double external_energy_j() const noexcept { return external_energy_j_; } + const ThermalModel* model() const noexcept { return runtime_.model(); } + SensorSnapshot sensors() const; + // Stateless shadow-only recommendation, never an admission action. Thresholds + // and component selection are explicit caller inputs, not product defaults. + std::optional advise(const std::vector& components, + double light_k,double severe_k,double shutdown_k) const; + ObserverCheckpoint checkpoint() const; + void restore(const ObserverCheckpoint& state); + void reset(); + private: + void integrate_to(std::uint64_t target_ns); + void consume(const ObservedEvent& event); + RuntimeMode mode_; + ThermalModelConfig config_; + ThermalRuntime runtime_; + std::map indices_; + std::map seen_, active_; + std::vector pending_, journal_, terminals_; + std::map energy_j_; + double external_energy_j_{}; + std::uint64_t time_ns_{}, next_energy_id_{1}; +}; +} // namespace hbfsim::eq3_thermal diff --git a/experiments/eq3_maintenance/backend/include/hbfsim/mqsim_online.hpp b/experiments/eq3_maintenance/backend/include/hbfsim/mqsim_online.hpp new file mode 100644 index 0000000..7ae6211 --- /dev/null +++ b/experiments/eq3_maintenance/backend/include/hbfsim/mqsim_online.hpp @@ -0,0 +1,111 @@ +#pragma once + +#include +#include + +#include +#include +#include +#include +#include +#include + +namespace hbfsim { + +enum class MqsimEventKind { Arrival, Admission, Completion }; + +// Optional adapter-boundary observations. Admission is the handoff to MQSim, +// not the start of an individual NAND command. Completion time is the raw +// media callback; modeled_completion_ns also includes the existing bandwidth +// lower bound. The two timestamps must not be conflated. +struct MqsimObservation { + MqsimEventKind kind; + std::uint64_t request_id; + std::uint64_t arrival_ns; + std::uint64_t time_ns; + std::uint64_t modeled_completion_ns; + std::uint32_t bytes; + std::size_t device_outstanding; +}; + +enum class MqsimMaintenanceStatus : std::uint32_t { + Committed, CommittedReclaimDeferred, RejectedUnsupported, + RejectedInvalidTarget, RejectedUnmapped, RejectedNoSpare, + FailedRead, FailedProgram, FailedStaleVersion, + FailedAfterCommitNeedsReconcile, RejectedSourceBusy, + FailedVersionOverflow +}; +enum class MqsimMaintenanceFailurePoint : std::uint32_t { + None, Read, Program, StaleCommit, Erase +}; +enum class MqsimMaintenanceState : std::uint32_t { + Due, Queued, Read, ProgramDest, Commit, RetireOld, + ReclaimErase, Done, Failed +}; +struct MqsimMaintenanceRequest { + std::uint64_t request_id{}; + std::uint64_t parent_id{}; + std::uint64_t due_ns{}; + std::uint64_t deadline_ns{}; + std::uint64_t logical_page{}; + std::uint32_t channel{}, chip{}, die{}, plane{}; + bool plane_is_exact{}; + bool reclaim_invalid_source_block{}; + MqsimMaintenanceFailurePoint failure_point{MqsimMaintenanceFailurePoint::None}; +}; +struct MqsimMaintenanceEvent { + std::uint64_t request_id{}, parent_id{}; + MqsimMaintenanceState state{}; + std::uint64_t time_ns{}, transaction_id{}; + std::uint32_t source_channel{}, source_chip{}, source_die{}, source_plane{}, + source_block{}, source_page{}; + std::optional destination_channel, destination_chip, + destination_die, destination_plane, destination_block, destination_page; +}; +struct MqsimMaintenanceCompletion { + std::uint64_t request_id{}, parent_id{}; + MqsimMaintenanceStatus status{}; + std::uint64_t enqueue_ns{}, start_ns{}, end_ns{}, logical_page{}, + source_version{}, committed_version{}; + bool mapping_committed{}, source_retired{}, erase_completed{}; + std::vector transaction_ids; +}; + +class MqsimOnlineEngine { +public: + explicit MqsimOnlineEngine(const Profile& profile); + ~MqsimOnlineEngine(); + + MqsimOnlineEngine(const MqsimOnlineEngine&) = delete; + MqsimOnlineEngine& operator=(const MqsimOnlineEngine&) = delete; + MqsimOnlineEngine(MqsimOnlineEngine&&) noexcept; + MqsimOnlineEngine& operator=(MqsimOnlineEngine&&) noexcept; + + void submit(const HbfRequest& request); + void submit_maintenance(const MqsimMaintenanceRequest& request); + std::optional run_next_completion(); + // Opt-in coordination with external compute/replay events. Return one + // completion when its reported deadline is reached, or advance exactly to + // deadline_ns and return nullopt. Never advance beyond the horizon. + // Unlike run_next_completion(), nullopt here does not mean lost work. + std::optional run_next_completion_until(std::uint64_t deadline_ns); + [[nodiscard]] std::size_t pending() const noexcept; + [[nodiscard]] std::size_t pending_maintenance() const noexcept; + [[nodiscard]] std::uint64_t current_time_ns() const noexcept; + + // Opt in before the first submission; disabled by default. Callers should + // drain these CPU diagnostics after each returned completion. + void enable_observations(); + [[nodiscard]] std::vector take_observations(); + [[nodiscard]] std::vector take_maintenance_events(); + [[nodiscard]] std::vector take_maintenance_completions(); + +private: + class Impl; + std::unique_ptr impl_; +}; + +std::vector run_mqsim_trace( + const Profile& profile, const std::filesystem::path& trace_path); + +} // namespace hbfsim diff --git a/experiments/eq3_maintenance/backend/patches/0004-eq3-maintenance.patch b/experiments/eq3_maintenance/backend/patches/0004-eq3-maintenance.patch new file mode 100644 index 0000000..360bf7e --- /dev/null +++ b/experiments/eq3_maintenance/backend/patches/0004-eq3-maintenance.patch @@ -0,0 +1,806 @@ +diff --git a/src/ssd/Address_Mapping_Unit_Base.h b/src/ssd/Address_Mapping_Unit_Base.h +index b86812a..1385193 100644 +--- a/src/ssd/Address_Mapping_Unit_Base.h ++++ b/src/ssd/Address_Mapping_Unit_Base.h +@@ -59,6 +59,17 @@ namespace SSD_Components + virtual void Get_data_mapping_info_for_gc(const stream_id_type stream_id, const LPA_type lpa, PPA_type& ppa, page_status_type& page_state) = 0; + virtual void Get_translation_mapping_info_for_gc(const stream_id_type stream_id, const MVPN_type mvpn, MPPN_type& mppa, sim_time_type& timestamp) = 0; + virtual void Allocate_new_page_for_gc(NVM_Transaction_Flash_WR* transaction, bool is_translation_page) = 0; ++ virtual bool Begin_hbf_maintenance(const stream_id_type stream_id, ++ const LPA_type lpa, PPA_type& source_ppa, page_status_type& page_state, ++ std::uint64_t& source_generation) = 0; ++ virtual void Allocate_new_page_for_hbf_maintenance(NVM_Transaction_Flash_WR* transaction) = 0; ++ virtual bool Commit_hbf_maintenance(const stream_id_type stream_id, ++ const LPA_type lpa, const PPA_type expected_source, ++ const PPA_type destination, const page_status_type page_state, ++ const std::uint64_t expected_generation, ++ std::uint64_t& committed_generation, bool& generation_overflow) = 0; ++ virtual void Abort_hbf_maintenance(const stream_id_type stream_id, ++ const LPA_type lpa, const NVM::FlashMemory::Physical_Page_Address* destination) = 0; + unsigned int Get_device_physical_pages_count();//Returns the number of physical pages in the device + CMT_Sharing_Mode Get_CMT_sharing_mode(); + virtual NVM::FlashMemory::Physical_Page_Address Convert_ppa_to_address(const PPA_type ppa) = 0; +diff --git a/src/ssd/Address_Mapping_Unit_Hybrid.h b/src/ssd/Address_Mapping_Unit_Hybrid.h +index d656100..d2c4507 100644 +--- a/src/ssd/Address_Mapping_Unit_Hybrid.h ++++ b/src/ssd/Address_Mapping_Unit_Hybrid.h +@@ -1,6 +1,8 @@ + #ifndef ADDRESS_MAPPING_UNIT_HYBRID_H + #define ADDRESS_MAPPING_UNIT_HYBRID_H + ++#include ++ + #include "Address_Mapping_Unit_Base.h" + + namespace SSD_Components +@@ -26,6 +28,10 @@ namespace SSD_Components + void Get_data_mapping_info_for_gc(const stream_id_type stream_id, const LPA_type lpa, PPA_type& ppa, page_status_type& page_state); + void Get_translation_mapping_info_for_gc(const stream_id_type stream_id, const MVPN_type mvpn, MPPN_type& mppa, sim_time_type& timestamp); + void Allocate_new_page_for_gc(NVM_Transaction_Flash_WR* transaction, bool is_translation_page); ++ bool Begin_hbf_maintenance(const stream_id_type, const LPA_type, PPA_type&, page_status_type&, std::uint64_t&) { return false; } ++ void Allocate_new_page_for_hbf_maintenance(NVM_Transaction_Flash_WR*) { throw std::logic_error("HBF maintenance requires page-level mapping"); } ++ bool Commit_hbf_maintenance(const stream_id_type, const LPA_type, const PPA_type, const PPA_type, const page_status_type, const std::uint64_t, std::uint64_t&, bool&) { return false; } ++ void Abort_hbf_maintenance(const stream_id_type, const LPA_type, const NVM::FlashMemory::Physical_Page_Address*) {} + + void Store_mapping_table_on_flash_at_start(); + LPA_type Get_logical_pages_count(stream_id_type stream_id); +diff --git a/src/ssd/Address_Mapping_Unit_Page_Level.cpp b/src/ssd/Address_Mapping_Unit_Page_Level.cpp +index 1db2d45..6cb51d8 100644 +--- a/src/ssd/Address_Mapping_Unit_Page_Level.cpp ++++ b/src/ssd/Address_Mapping_Unit_Page_Level.cpp +@@ -1,5 +1,6 @@ + #include + #include ++#include + #include + + #include "Address_Mapping_Unit_Page_Level.h" +@@ -198,10 +199,12 @@ namespace SSD_Components + } + + GlobalMappingTable = new GMTEntryType[Total_logical_pages_no]; ++ MappingGenerations = new std::uint64_t[Total_logical_pages_no]; + for (unsigned int i = 0; i < Total_logical_pages_no; i++) { + GlobalMappingTable[i].PPA = NO_PPA; + GlobalMappingTable[i].WrittenStateBitmap = UNWRITTEN_LOGICAL_PAGE; + GlobalMappingTable[i].TimeStamp = 0; ++ MappingGenerations[i] = 0; + } + + //If CMT is NULL, then each address mapping domain should create its own CMT +@@ -225,6 +228,7 @@ namespace SSD_Components + { + delete CMT; + delete[] GlobalMappingTable; ++ delete[] MappingGenerations; + delete[] GlobalTranslationDirectory; + + auto read_entry = Waiting_unmapped_read_transactions.begin(); +@@ -247,6 +251,8 @@ namespace SSD_Components + + inline void AddressMappingDomain::Update_mapping_info(const bool ideal_mapping, const stream_id_type stream_id, const LPA_type lpa, const PPA_type ppa, const page_status_type page_status_bitmap) + { ++ if (MappingGenerations[lpa] == std::numeric_limits::max()) ++ throw std::overflow_error("MQSim mapping generation overflow"); + if (ideal_mapping) { + GlobalMappingTable[lpa].PPA = ppa; + GlobalMappingTable[lpa].WrittenStateBitmap = page_status_bitmap; +@@ -254,6 +260,12 @@ namespace SSD_Components + } else { + CMT->Update_mapping_info(stream_id, lpa, ppa, page_status_bitmap); + } ++ ++MappingGenerations[lpa]; ++ } ++ ++ inline std::uint64_t AddressMappingDomain::Get_mapping_generation(const LPA_type lpa) const ++ { ++ return MappingGenerations[lpa]; + } + + inline page_status_type AddressMappingDomain::Get_page_status(const bool ideal_mapping, const stream_id_type stream_id, const LPA_type lpa) +@@ -1379,6 +1391,12 @@ namespace SSD_Components + } + + block_manager->Allocate_block_and_page_in_plane_for_user_write(stream_id, read_address); ++ // Online HBF creates metadata for a previously populated page on its ++ // first read. No program command will follow this allocation, so close ++ // the synthetic program bookkeeping here; otherwise the source block is ++ // permanently reported as having an in-flight program and can never be ++ // safely reclaimed after experimental maintenance. ++ block_manager->Program_transaction_serviced(read_address); + PPA_type ppa = Convert_address_to_ppa(read_address); + domain->Update_mapping_info(ideal_mapping_table, stream_id, lpa, ppa, read_sectors_bitmap); + +@@ -1747,6 +1765,62 @@ namespace SSD_Components + return domains[stream_id]->Locked_MVPNs.find(mvpn) != domains[stream_id]->Locked_MVPNs.end(); + } + ++ bool Address_Mapping_Unit_Page_Level::Begin_hbf_maintenance( ++ const stream_id_type stream_id, const LPA_type lpa, ++ PPA_type& source_ppa, page_status_type& page_state, ++ std::uint64_t& source_generation) ++ { ++ if (stream_id >= no_of_input_streams || ++ !domains[stream_id]->Mapping_entry_accessible(ideal_mapping_table, stream_id, lpa)) ++ return false; ++ source_ppa = domains[stream_id]->Get_ppa(ideal_mapping_table, stream_id, lpa); ++ if (source_ppa == NO_PPA) return false; ++ page_state = domains[stream_id]->Get_page_status(ideal_mapping_table, stream_id, lpa); ++ source_generation = domains[stream_id]->Get_mapping_generation(lpa); ++ return true; ++ } ++ ++ void Address_Mapping_Unit_Page_Level::Allocate_new_page_for_hbf_maintenance( ++ NVM_Transaction_Flash_WR* transaction) ++ { ++ block_manager->Allocate_block_and_page_in_plane_for_gc_write( ++ transaction->Stream_id, transaction->Address); ++ transaction->PPA = Convert_address_to_ppa(transaction->Address); ++ transaction->Physical_address_determined = true; ++ } ++ ++ bool Address_Mapping_Unit_Page_Level::Commit_hbf_maintenance( ++ const stream_id_type stream_id, const LPA_type lpa, ++ const PPA_type expected_source, const PPA_type destination, ++ const page_status_type page_state, const std::uint64_t expected_generation, ++ std::uint64_t& committed_generation, bool& generation_overflow) ++ { ++ AddressMappingDomain* domain = domains[stream_id]; ++ generation_overflow = false; ++ if (domain->Get_ppa(ideal_mapping_table, stream_id, lpa) != expected_source || ++ domain->Get_mapping_generation(lpa) != expected_generation) ++ return false; ++ if (expected_generation == std::numeric_limits::max()) { ++ generation_overflow = true; ++ return false; ++ } ++ NVM::FlashMemory::Physical_Page_Address old_address; ++ Convert_ppa_to_address(expected_source, old_address); ++ domain->Update_mapping_info(ideal_mapping_table, stream_id, lpa, ++ destination, page_state); ++ committed_generation = domain->Get_mapping_generation(lpa); ++ block_manager->Invalidate_page_in_block(stream_id, old_address); ++ return true; ++ } ++ ++ void Address_Mapping_Unit_Page_Level::Abort_hbf_maintenance( ++ const stream_id_type stream_id, const LPA_type lpa, ++ const NVM::FlashMemory::Physical_Page_Address* destination) ++ { ++ if (destination != NULL) ++ block_manager->Invalidate_page_in_block(stream_id, *destination); ++ } ++ + inline void Address_Mapping_Unit_Page_Level::Set_barrier_for_accessing_lpa(stream_id_type stream_id, LPA_type lpa) + { + auto itr = domains[stream_id]->Locked_LPAs.find(lpa); +diff --git a/src/ssd/Address_Mapping_Unit_Page_Level.h b/src/ssd/Address_Mapping_Unit_Page_Level.h +index de70aff..b66c4b7 100644 +--- a/src/ssd/Address_Mapping_Unit_Page_Level.h ++++ b/src/ssd/Address_Mapping_Unit_Page_Level.h +@@ -95,7 +95,9 @@ namespace SSD_Components + /*The logical to physical address mapping of all data pages that is implemented based on the DFTL (Gupta et al., ASPLOS 2009( + * proposal. It is always stored in non-volatile flash memory.*/ + GMTEntryType* GlobalMappingTable; ++ std::uint64_t* MappingGenerations; + void Update_mapping_info(const bool ideal_mapping, const stream_id_type stream_id, const LPA_type lpa, const PPA_type ppa, const page_status_type page_status_bitmap); ++ std::uint64_t Get_mapping_generation(const LPA_type lpa) const; + page_status_type Get_page_status(const bool ideal_mapping, const stream_id_type stream_id, const LPA_type lpa); + PPA_type Get_ppa(const bool ideal_mapping, const stream_id_type stream_id, const LPA_type lpa); + PPA_type Get_ppa_for_preconditioning(const stream_id_type stream_id, const LPA_type lpa); +@@ -154,6 +156,16 @@ namespace SSD_Components + void Get_data_mapping_info_for_gc(const stream_id_type stream_id, const LPA_type lpa, PPA_type& ppa, page_status_type& page_state); + void Get_translation_mapping_info_for_gc(const stream_id_type stream_id, const MVPN_type mvpn, MPPN_type& mppa, sim_time_type& timestamp); + void Allocate_new_page_for_gc(NVM_Transaction_Flash_WR* transaction, bool is_translation_page); ++ bool Begin_hbf_maintenance(const stream_id_type stream_id, const LPA_type lpa, ++ PPA_type& source_ppa, page_status_type& page_state, ++ std::uint64_t& source_generation); ++ void Allocate_new_page_for_hbf_maintenance(NVM_Transaction_Flash_WR* transaction); ++ bool Commit_hbf_maintenance(const stream_id_type stream_id, const LPA_type lpa, ++ const PPA_type expected_source, const PPA_type destination, ++ const page_status_type page_state, const std::uint64_t expected_generation, ++ std::uint64_t& committed_generation, bool& generation_overflow); ++ void Abort_hbf_maintenance(const stream_id_type stream_id, const LPA_type lpa, ++ const NVM::FlashMemory::Physical_Page_Address* destination); + + void Store_mapping_table_on_flash_at_start(); + LPA_type Get_logical_pages_count(stream_id_type stream_id); +diff --git a/src/ssd/Flash_Block_Manager_Base.cpp b/src/ssd/Flash_Block_Manager_Base.cpp +index 3abe320..56325e7 100644 +--- a/src/ssd/Flash_Block_Manager_Base.cpp ++++ b/src/ssd/Flash_Block_Manager_Base.cpp +@@ -1,4 +1,6 @@ + #include "Flash_Block_Manager.h" ++#include ++#include + + + namespace SSD_Components +@@ -40,6 +42,7 @@ namespace SSD_Components + plane_manager[channelID][chipID][dieID][planeID].Blocks[blockID].Erase_transaction = NULL; + plane_manager[channelID][chipID][dieID][planeID].Blocks[blockID].Ongoing_user_program_count = 0; + plane_manager[channelID][chipID][dieID][planeID].Blocks[blockID].Ongoing_user_read_count = 0; ++ plane_manager[channelID][chipID][dieID][planeID].Blocks[blockID].HBF_maintenance_pin_count = 0; + Block_Pool_Slot_Type::Page_vector_size = pages_no_per_block / (sizeof(uint64_t) * 8) + (pages_no_per_block % (sizeof(uint64_t) * 8) == 0 ? 0 : 1); + plane_manager[channelID][chipID][dieID][planeID].Blocks[blockID].Invalid_page_bitmap = new uint64_t[Block_Pool_Slot_Type::Page_vector_size]; + for (unsigned int i = 0; i < Block_Pool_Slot_Type::Page_vector_size; i++) { +@@ -91,6 +94,8 @@ namespace SSD_Components + + void Block_Pool_Slot_Type::Erase() + { ++ if (HBF_maintenance_pin_count != 0) ++ throw std::logic_error("erasing an HBF-maintenance-pinned block"); + Current_page_write_index = 0; + Invalid_page_count = 0; + Erase_count++; +@@ -189,7 +194,9 @@ namespace SSD_Components + bool Flash_Block_Manager_Base::Can_execute_gc_wl(const NVM::FlashMemory::Physical_Page_Address& block_address) + { + PlaneBookKeepingType *plane_record = &plane_manager[block_address.ChannelID][block_address.ChipID][block_address.DieID][block_address.PlaneID]; +- return (plane_record->Blocks[block_address.BlockID].Ongoing_user_program_count + plane_record->Blocks[block_address.BlockID].Ongoing_user_read_count == 0); ++ return (plane_record->Blocks[block_address.BlockID].Ongoing_user_program_count + ++ plane_record->Blocks[block_address.BlockID].Ongoing_user_read_count == 0 && ++ plane_record->Blocks[block_address.BlockID].HBF_maintenance_pin_count == 0); + } + + void Flash_Block_Manager_Base::GC_WL_started(const NVM::FlashMemory::Physical_Page_Address& block_address) +@@ -241,4 +248,70 @@ namespace SSD_Components + } + return false; + } ++ ++ bool Flash_Block_Manager_Base::Can_allocate_hbf_spare( ++ const stream_id_type stream_id, ++ const NVM::FlashMemory::Physical_Page_Address& plane_address) const ++ { ++ const PlaneBookKeepingType* plane = &plane_manager[plane_address.ChannelID] ++ [plane_address.ChipID][plane_address.DieID][plane_address.PlaneID]; ++ const Block_Pool_Slot_Type* frontier = plane->GC_wf[stream_id]; ++ return frontier->Current_page_write_index < pages_no_per_block || ++ !plane->Free_block_pool.empty(); ++ } ++ ++ bool Flash_Block_Manager_Base::Try_pin_hbf_source( ++ const NVM::FlashMemory::Physical_Page_Address& block_address) ++ { ++ PlaneBookKeepingType* plane = &plane_manager[block_address.ChannelID] ++ [block_address.ChipID][block_address.DieID][block_address.PlaneID]; ++ Block_Pool_Slot_Type& block = plane->Blocks[block_address.BlockID]; ++ if (block.Has_ongoing_gc_wl || ++ plane->Ongoing_erase_operations.count(block_address.BlockID) != 0) ++ return false; ++ if (block.HBF_maintenance_pin_count == std::numeric_limits::max()) ++ throw std::overflow_error("HBF maintenance source pin overflow"); ++ ++block.HBF_maintenance_pin_count; ++ return true; ++ } ++ ++ void Flash_Block_Manager_Base::Unpin_hbf_source( ++ const NVM::FlashMemory::Physical_Page_Address& block_address) ++ { ++ Block_Pool_Slot_Type& block = plane_manager[block_address.ChannelID] ++ [block_address.ChipID][block_address.DieID][block_address.PlaneID] ++ .Blocks[block_address.BlockID]; ++ if (block.HBF_maintenance_pin_count == 0) ++ throw std::logic_error("HBF maintenance source pin underflow"); ++ --block.HBF_maintenance_pin_count; ++ } ++ ++ bool Flash_Block_Manager_Base::Begin_hbf_reclaim( ++ const NVM::FlashMemory::Physical_Page_Address& block_address) ++ { ++ PlaneBookKeepingType* plane = &plane_manager[block_address.ChannelID] ++ [block_address.ChipID][block_address.DieID][block_address.PlaneID]; ++ Block_Pool_Slot_Type* block = &plane->Blocks[block_address.BlockID]; ++ if (block->Current_page_write_index != block->Invalid_page_count || ++ block->Has_ongoing_gc_wl || block->Ongoing_user_read_count != 0 || ++ block->Ongoing_user_program_count != 0 || block->HBF_maintenance_pin_count != 0 || ++ plane->Ongoing_erase_operations.count(block_address.BlockID) != 0) ++ return false; ++ for (unsigned int stream = 0; stream < total_concurrent_streams_no; ++stream) ++ if (plane->Data_wf[stream] == block || plane->GC_wf[stream] == block || ++ plane->Translation_wf[stream] == block) return false; ++ block->Has_ongoing_gc_wl = true; ++ plane->Ongoing_erase_operations.insert(block_address.BlockID); ++ return true; ++ } ++ ++ void Flash_Block_Manager_Base::Finish_hbf_reclaim( ++ const NVM::FlashMemory::Physical_Page_Address& block_address, bool erased) ++ { ++ PlaneBookKeepingType* plane = &plane_manager[block_address.ChannelID] ++ [block_address.ChipID][block_address.DieID][block_address.PlaneID]; ++ plane->Ongoing_erase_operations.erase(block_address.BlockID); ++ if (erased) Add_erased_block_to_pool(block_address); ++ plane->Blocks[block_address.BlockID].Has_ongoing_gc_wl = false; ++ } + } +diff --git a/src/ssd/Flash_Block_Manager_Base.h b/src/ssd/Flash_Block_Manager_Base.h +index 8e09193..d35c73d 100644 +--- a/src/ssd/Flash_Block_Manager_Base.h ++++ b/src/ssd/Flash_Block_Manager_Base.h +@@ -43,6 +43,7 @@ namespace SSD_Components + bool Hot_block = false;//Used for hot/cold separation mentioned in the "On the necessity of hot and cold data identification to reduce the write amplification in flash-based SSDs", Perf. Eval., 2014. + int Ongoing_user_read_count; + int Ongoing_user_program_count; ++ unsigned int HBF_maintenance_pin_count = 0; + void Erase(); + }; + +@@ -96,6 +97,13 @@ namespace SSD_Components + void Program_transaction_serviced(const NVM::FlashMemory::Physical_Page_Address& page_address);//Updates the block bookkeeping record + bool Is_having_ongoing_program(const NVM::FlashMemory::Physical_Page_Address& block_address);//Cheks if block has any ongoing program request + bool Is_page_valid(Block_Pool_Slot_Type* block, flash_page_ID_type page_id);//Make the page invalid in the block bookkeeping record ++ bool Can_allocate_hbf_spare(const stream_id_type stream_id, ++ const NVM::FlashMemory::Physical_Page_Address& plane_address) const; ++ bool Try_pin_hbf_source(const NVM::FlashMemory::Physical_Page_Address& block_address); ++ void Unpin_hbf_source(const NVM::FlashMemory::Physical_Page_Address& block_address); ++ bool Begin_hbf_reclaim(const NVM::FlashMemory::Physical_Page_Address& block_address); ++ void Finish_hbf_reclaim(const NVM::FlashMemory::Physical_Page_Address& block_address, ++ bool erased); + protected: + PlaneBookKeepingType ****plane_manager;//Keeps track of plane block usage information + GC_and_WL_Unit_Base *gc_and_wl_unit; +diff --git a/src/ssd/GC_and_WL_Unit_Base.cpp b/src/ssd/GC_and_WL_Unit_Base.cpp +index fef869d..749697b 100644 +--- a/src/ssd/GC_and_WL_Unit_Base.cpp ++++ b/src/ssd/GC_and_WL_Unit_Base.cpp +@@ -41,6 +41,7 @@ namespace SSD_Components + + void GC_and_WL_Unit_Base::handle_transaction_serviced_signal_from_PHY(NVM_Transaction_Flash* transaction) + { ++ if (transaction->Source == Transaction_Source_Type::HBF_MAINTENANCE) return; + PlaneBookKeepingType* pbke = &(_my_instance->block_manager->plane_manager[transaction->Address.ChannelID][transaction->Address.ChipID][transaction->Address.DieID][transaction->Address.PlaneID]); + + switch (transaction->Source) { +diff --git a/src/ssd/HBF_Maintenance_Unit.cpp b/src/ssd/HBF_Maintenance_Unit.cpp +new file mode 100644 +index 0000000..54588cb +--- /dev/null ++++ b/src/ssd/HBF_Maintenance_Unit.cpp +@@ -0,0 +1,206 @@ ++#include "HBF_Maintenance_Unit.h" ++#include ++#include "NVM_Transaction_Flash_RD.h" ++#include "NVM_Transaction_Flash_WR.h" ++#include "NVM_Transaction_Flash_ER.h" ++ ++namespace SSD_Components { ++struct HBF_Maintenance_Unit::Context { ++ HBF_Maintenance_Request request; ++ HBF_Maintenance_Completion_Callback callback; ++ sim_time_type enqueue_time{0}, start_time{0}; ++ PPA_type source_ppa{NO_PPA}, destination_ppa{NO_PPA}; ++ page_status_type page_state{0}; ++ std::uint64_t source_generation{0}, committed_generation{0}; ++ NVM::FlashMemory::Physical_Page_Address source, destination; ++ bool destination_known{false}, mapping_committed{false}, source_retired{false}, ++ source_pinned{false}; ++ std::vector transaction_ids; ++}; ++HBF_Maintenance_Unit* HBF_Maintenance_Unit::instance = nullptr; ++ ++HBF_Maintenance_Unit::HBF_Maintenance_Unit(const sim_object_id_type& id, ++ Address_Mapping_Unit_Base* mapping, Flash_Block_Manager_Base* blocks, ++ TSU_Base* tsu, NVM_PHY_ONFI* phy, unsigned int page_size_bytes, ++ unsigned int channels, unsigned int chips, unsigned int dies, unsigned int planes) ++ : Sim_Object(id), mapping(mapping), blocks(blocks), tsu(tsu), phy(phy), ++ page_size_bytes(page_size_bytes), channels(channels), chips(chips), dies(dies), planes(planes) ++{ ++ if (instance != nullptr) throw std::logic_error("only one experimental maintenance unit is supported"); ++ instance = this; ++ phy->ConnectToTransactionServicedSignal(serviced); ++} ++HBF_Maintenance_Unit::~HBF_Maintenance_Unit() { if (instance == this) instance = nullptr; } ++ ++bool HBF_Maintenance_Unit::target_matches(const HBF_Maintenance_Request& request, ++ const NVM::FlashMemory::Physical_Page_Address& address) const ++{ ++ return address.ChannelID == request.channel && address.ChipID == request.chip && ++ address.DieID == request.die && (!request.plane_is_exact || address.PlaneID == request.plane); ++} ++void HBF_Maintenance_Unit::emit(const Context& context, HBF_Maintenance_State state, ++ std::uint64_t transaction_id) ++{ ++ if (!event_sink) return; ++ event_sink({context.request.request_id, context.request.parent_id, state, ++ Simulator->Time(), transaction_id, context.source, context.destination, ++ context.destination_known}); ++} ++void HBF_Maintenance_Unit::Submit(const HBF_Maintenance_Request& request, ++ HBF_Maintenance_Completion_Callback callback) ++{ ++ if (!request.request_id || !callback || contexts.count(request.request_id)) ++ throw std::invalid_argument("invalid or duplicate HBF maintenance request"); ++ if (request.channel >= channels || request.chip >= chips || request.die >= dies || ++ (request.plane_is_exact && request.plane >= planes)) ++ throw std::out_of_range("HBF maintenance physical target is out of range"); ++ Context context{request, std::move(callback)}; ++ context.enqueue_time = context.start_time = Simulator->Time(); ++ auto inserted = contexts.emplace(request.request_id, std::move(context)); ++ Context& stored = inserted.first->second; ++ emit(stored, HBF_Maintenance_State::DUE); ++ emit(stored, HBF_Maintenance_State::QUEUED); ++ if (request.deadline && Simulator->Time() > request.deadline) { ++ finish(stored, HBF_Maintenance_Status::REJECTED_INVALID_TARGET); return; ++ } ++ if (!mapping->Begin_hbf_maintenance(request.stream_id, request.logical_page, ++ stored.source_ppa, stored.page_state, stored.source_generation)) { ++ finish(stored, HBF_Maintenance_Status::REJECTED_UNMAPPED); return; ++ } ++ mapping->Convert_ppa_to_address(stored.source_ppa, stored.source); ++ if (!blocks->Try_pin_hbf_source(stored.source)) { ++ finish(stored, HBF_Maintenance_Status::REJECTED_SOURCE_BUSY); return; ++ } ++ stored.source_pinned = true; ++ if (!target_matches(request, stored.source)) { ++ mapping->Abort_hbf_maintenance(request.stream_id, request.logical_page, nullptr); ++ finish(stored, HBF_Maintenance_Status::REJECTED_INVALID_TARGET); return; ++ } ++ auto* write = new NVM_Transaction_Flash_WR(Transaction_Source_Type::HBF_MAINTENANCE, ++ request.stream_id, page_size_bytes, request.logical_page, NO_PPA, stored.source, ++ nullptr, 0, nullptr, stored.page_state, INVALID_TIME_STAMP); ++ auto* read = new NVM_Transaction_Flash_RD(Transaction_Source_Type::HBF_MAINTENANCE, ++ request.stream_id, page_size_bytes, request.logical_page, stored.source_ppa, ++ stored.source, nullptr, 0, write, stored.page_state, INVALID_TIME_STAMP); ++ read->HBF_Maintenance_ID = write->HBF_Maintenance_ID = request.request_id; ++ read->HBF_Maintenance_Parent_ID = write->HBF_Maintenance_Parent_ID = request.parent_id; ++ read->HBF_Force_Failure = request.failure_point == HBF_Maintenance_Failure_Point::READ; ++ write->HBF_Force_Failure = request.failure_point == HBF_Maintenance_Failure_Point::PROGRAM; ++ write->RelatedRead = read; ++ blocks->Read_transaction_issued(stored.source); ++ tsu->Prepare_for_transaction_submit(); tsu->Submit_transaction(read); tsu->Schedule(); ++ emit(stored, HBF_Maintenance_State::READ, read->HBF_Observation_ID); ++} ++void HBF_Maintenance_Unit::serviced(NVM_Transaction_Flash* transaction) ++{ ++ if (instance && transaction->Source == Transaction_Source_Type::HBF_MAINTENANCE) ++ instance->handle(transaction); ++} ++void HBF_Maintenance_Unit::handle(NVM_Transaction_Flash* transaction) ++{ ++ auto found = contexts.find(transaction->HBF_Maintenance_ID); ++ if (found == contexts.end()) throw std::logic_error("orphan HBF maintenance transaction"); ++ Context& context = found->second; ++ context.transaction_ids.push_back(transaction->HBF_Observation_ID); ++ if (transaction->Type == Transaction_Type::READ) { ++ blocks->Read_transaction_serviced(transaction->Address); ++ if (!transaction->HBF_Command_Succeeded) { ++ mapping->Abort_hbf_maintenance(context.request.stream_id, ++ context.request.logical_page, nullptr); ++ finish(context, HBF_Maintenance_Status::FAILED_READ); return; ++ } ++ if (!blocks->Can_allocate_hbf_spare(context.request.stream_id, context.source)) { ++ mapping->Abort_hbf_maintenance(context.request.stream_id, ++ context.request.logical_page, nullptr); ++ finish(context, HBF_Maintenance_Status::REJECTED_NO_SPARE); return; ++ } ++ auto* read = static_cast(transaction); ++ auto* write = read->RelatedWrite; ++ write->RelatedRead = nullptr; ++ write->write_sectors_bitmap = context.page_state; ++ mapping->Allocate_new_page_for_hbf_maintenance(write); ++ context.destination = write->Address; ++ context.destination_ppa = write->PPA; ++ context.destination_known = true; ++ tsu->Prepare_for_transaction_submit(); tsu->Submit_transaction(write); tsu->Schedule(); ++ emit(context, HBF_Maintenance_State::PROGRAM_DEST, write->HBF_Observation_ID); ++ return; ++ } ++ if (transaction->Type == Transaction_Type::WRITE) { ++ if (!transaction->HBF_Command_Succeeded || ++ context.request.failure_point == HBF_Maintenance_Failure_Point::STALE_COMMIT) { ++ mapping->Abort_hbf_maintenance(context.request.stream_id, ++ context.request.logical_page, &context.destination); ++ finish(context, transaction->HBF_Command_Succeeded ? ++ HBF_Maintenance_Status::FAILED_STALE_VERSION : ++ HBF_Maintenance_Status::FAILED_PROGRAM); ++ return; ++ } ++ emit(context, HBF_Maintenance_State::COMMIT, transaction->HBF_Observation_ID); ++ bool generation_overflow = false; ++ if (!mapping->Commit_hbf_maintenance(context.request.stream_id, ++ context.request.logical_page, context.source_ppa, ++ context.destination_ppa, context.page_state, ++ context.source_generation, context.committed_generation, ++ generation_overflow)) { ++ mapping->Abort_hbf_maintenance(context.request.stream_id, ++ context.request.logical_page, &context.destination); ++ finish(context, generation_overflow ? ++ HBF_Maintenance_Status::FAILED_VERSION_OVERFLOW : ++ HBF_Maintenance_Status::FAILED_STALE_VERSION); return; ++ } ++ context.mapping_committed = context.source_retired = true; ++ emit(context, HBF_Maintenance_State::RETIRE_OLD, transaction->HBF_Observation_ID); ++ if (!context.request.reclaim_invalid_source_block) { ++ finish(context, HBF_Maintenance_Status::COMMITTED); return; ++ } ++ blocks->Unpin_hbf_source(context.source); ++ context.source_pinned = false; ++ if (!blocks->Begin_hbf_reclaim(context.source)) { ++ finish(context, context.request.reclaim_invalid_source_block ? ++ HBF_Maintenance_Status::COMMITTED_RECLAIM_DEFERRED : ++ HBF_Maintenance_Status::COMMITTED); ++ return; ++ } ++ auto* erase = new NVM_Transaction_Flash_ER(Transaction_Source_Type::HBF_MAINTENANCE, ++ context.request.stream_id, context.source); ++ erase->HBF_Maintenance_ID = context.request.request_id; ++ erase->HBF_Maintenance_Parent_ID = context.request.parent_id; ++ erase->HBF_Force_Failure = context.request.failure_point == HBF_Maintenance_Failure_Point::ERASE; ++ tsu->Prepare_for_transaction_submit(); tsu->Submit_transaction(erase); tsu->Schedule(); ++ emit(context, HBF_Maintenance_State::RECLAIM_ERASE, erase->HBF_Observation_ID); ++ return; ++ } ++ if (transaction->Type == Transaction_Type::ERASE) { ++ blocks->Finish_hbf_reclaim(context.source, transaction->HBF_Command_Succeeded); ++ finish(context, transaction->HBF_Command_Succeeded ? ++ HBF_Maintenance_Status::COMMITTED : ++ HBF_Maintenance_Status::FAILED_AFTER_COMMIT_NEEDS_RECONCILE); ++ return; ++ } ++ throw std::logic_error("unexpected HBF maintenance transaction type"); ++} ++void HBF_Maintenance_Unit::finish(Context& context, HBF_Maintenance_Status status) ++{ ++ if (context.source_pinned) { ++ blocks->Unpin_hbf_source(context.source); ++ context.source_pinned = false; ++ } ++ HBF_Maintenance_Completion completion{context.request.request_id, ++ context.request.parent_id, status, context.enqueue_time, context.start_time, ++ Simulator->Time(), context.request.logical_page, context.source_generation, ++ context.mapping_committed ? context.committed_generation : context.source_generation, ++ context.source, context.destination, context.destination_known, ++ context.mapping_committed, context.source_retired, ++ status == HBF_Maintenance_Status::COMMITTED && ++ context.request.reclaim_invalid_source_block, ++ context.transaction_ids}; ++ emit(context, status == HBF_Maintenance_Status::COMMITTED || ++ status == HBF_Maintenance_Status::COMMITTED_RECLAIM_DEFERRED ? ++ HBF_Maintenance_State::DONE : HBF_Maintenance_State::FAILED); ++ auto callback = context.callback; ++ const auto id = context.request.request_id; ++ contexts.erase(id); ++ callback(completion); ++} ++} +diff --git a/src/ssd/HBF_Maintenance_Unit.h b/src/ssd/HBF_Maintenance_Unit.h +new file mode 100644 +index 0000000..b41182d +--- /dev/null ++++ b/src/ssd/HBF_Maintenance_Unit.h +@@ -0,0 +1,92 @@ ++#ifndef HBF_MAINTENANCE_UNIT_H ++#define HBF_MAINTENANCE_UNIT_H ++ ++#include ++#include ++#include ++#include ++#include "../sim/Sim_Object.h" ++#include "Address_Mapping_Unit_Base.h" ++#include "Flash_Block_Manager_Base.h" ++#include "TSU_Base.h" ++#include "NVM_PHY_ONFI.h" ++ ++namespace SSD_Components { ++enum class HBF_Maintenance_State { ++ DUE, QUEUED, READ, PROGRAM_DEST, COMMIT, RETIRE_OLD, RECLAIM_ERASE, DONE, FAILED ++}; ++enum class HBF_Maintenance_Status { ++ COMMITTED, COMMITTED_RECLAIM_DEFERRED, REJECTED_UNSUPPORTED, ++ REJECTED_INVALID_TARGET, REJECTED_UNMAPPED, REJECTED_NO_SPARE, ++ FAILED_READ, FAILED_PROGRAM, FAILED_STALE_VERSION, ++ FAILED_AFTER_COMMIT_NEEDS_RECONCILE, REJECTED_SOURCE_BUSY, ++ FAILED_VERSION_OVERFLOW ++}; ++enum class HBF_Maintenance_Failure_Point { NONE, READ, PROGRAM, STALE_COMMIT, ERASE }; ++ ++struct HBF_Maintenance_Request { ++ std::uint64_t request_id{0}, parent_id{0}; ++ stream_id_type stream_id{0}; ++ LPA_type logical_page{0}; ++ flash_channel_ID_type channel{0}; ++ flash_chip_ID_type chip{0}; ++ flash_die_ID_type die{0}; ++ flash_plane_ID_type plane{0}; ++ bool plane_is_exact{false}; ++ sim_time_type due_time{0}, deadline{0}; ++ bool reclaim_invalid_source_block{false}; ++ HBF_Maintenance_Failure_Point failure_point{HBF_Maintenance_Failure_Point::NONE}; ++}; ++struct HBF_Maintenance_Event { ++ std::uint64_t request_id, parent_id; ++ HBF_Maintenance_State state; ++ sim_time_type time; ++ std::uint64_t transaction_observation_id; ++ NVM::FlashMemory::Physical_Page_Address source, destination; ++ bool destination_known; ++}; ++struct HBF_Maintenance_Completion { ++ std::uint64_t request_id, parent_id; ++ HBF_Maintenance_Status status; ++ sim_time_type enqueue_time, start_time, end_time; ++ LPA_type logical_page; ++ PPA_type source_version, committed_version; ++ NVM::FlashMemory::Physical_Page_Address source, destination; ++ bool destination_known, mapping_committed, source_retired, erase_completed; ++ std::vector transaction_observation_ids; ++}; ++using HBF_Maintenance_Completion_Callback = std::function; ++using HBF_Maintenance_Event_Sink = std::function; ++ ++class HBF_Maintenance_Unit : public MQSimEngine::Sim_Object { ++public: ++ HBF_Maintenance_Unit(const sim_object_id_type& id, ++ Address_Mapping_Unit_Base* mapping, Flash_Block_Manager_Base* blocks, ++ TSU_Base* tsu, NVM_PHY_ONFI* phy, unsigned int page_size_bytes, ++ unsigned int channels, unsigned int chips, unsigned int dies, unsigned int planes); ++ ~HBF_Maintenance_Unit(); ++ void Submit(const HBF_Maintenance_Request&, HBF_Maintenance_Completion_Callback); ++ void Set_event_sink(HBF_Maintenance_Event_Sink sink) { event_sink = std::move(sink); } ++ std::size_t Pending() const { return contexts.size(); } ++ void Start_simulation() override {} ++ void Validate_simulation_config() override {} ++ void Execute_simulator_event(MQSimEngine::Sim_Event*) override {} ++private: ++ struct Context; ++ static HBF_Maintenance_Unit* instance; ++ static void serviced(NVM_Transaction_Flash*); ++ void handle(NVM_Transaction_Flash*); ++ void finish(Context&, HBF_Maintenance_Status); ++ void emit(const Context&, HBF_Maintenance_State, std::uint64_t = 0); ++ bool target_matches(const HBF_Maintenance_Request&, ++ const NVM::FlashMemory::Physical_Page_Address&) const; ++ Address_Mapping_Unit_Base* mapping; ++ Flash_Block_Manager_Base* blocks; ++ TSU_Base* tsu; ++ NVM_PHY_ONFI* phy; ++ unsigned int page_size_bytes, channels, chips, dies, planes; ++ std::map contexts; ++ HBF_Maintenance_Event_Sink event_sink; ++}; ++} ++#endif +diff --git a/src/ssd/NVM_PHY_ONFI.cpp b/src/ssd/NVM_PHY_ONFI.cpp +index a225dbe..266021f 100644 +--- a/src/ssd/NVM_PHY_ONFI.cpp ++++ b/src/ssd/NVM_PHY_ONFI.cpp +@@ -15,6 +15,9 @@ namespace SSD_Components { + */ + void NVM_PHY_ONFI::broadcastTransactionServicedSignal(NVM_Transaction_Flash* transaction) + { ++ if (transaction->Source == Transaction_Source_Type::HBF_MAINTENANCE && ++ transaction->HBF_Force_Failure) ++ transaction->HBF_Command_Succeeded = false; + for (std::vector::iterator it = connectedTransactionServicedHandlers.begin(); + it != connectedTransactionServicedHandlers.end(); it++) { + (*it)(transaction); +@@ -47,4 +50,4 @@ namespace SSD_Components { + (*it)(chip); + } + } +-} +\ No newline at end of file ++} +diff --git a/src/ssd/NVM_PHY_ONFI_NVDDR2.cpp b/src/ssd/NVM_PHY_ONFI_NVDDR2.cpp +index 00114de..42a3c1d 100644 +--- a/src/ssd/NVM_PHY_ONFI_NVDDR2.cpp ++++ b/src/ssd/NVM_PHY_ONFI_NVDDR2.cpp +@@ -14,6 +14,7 @@ namespace SSD_Components { + if (tr->HBF_Observation_ID == 0) tr->HBF_Observation_ID = next_hbf_transaction_id++; + return {tr->HBF_Observation_ID, + tr->UserIORequest == NULL ? 0 : tr->UserIORequest->HBF_External_Request_ID, ++ tr->HBF_Maintenance_ID, tr->HBF_Maintenance_Parent_ID, + static_cast(tr->Source), static_cast(tr->Type), + tr->LPA, tr->LPA != NO_LPA, tr->Data_and_metadata_size_in_byte, tr->Address.ChannelID, + tr->Address.ChipID, tr->Address.DieID, tr->Address.PlaneID, +diff --git a/src/ssd/NVM_PHY_ONFI_NVDDR2.h b/src/ssd/NVM_PHY_ONFI_NVDDR2.h +index 1f72064..e20a249 100644 +--- a/src/ssd/NVM_PHY_ONFI_NVDDR2.h ++++ b/src/ssd/NVM_PHY_ONFI_NVDDR2.h +@@ -18,6 +18,7 @@ namespace SSD_Components + enum class HBF_Command_Observation_Phase { COMMAND_ISSUED, MEDIA_BEGIN, MEDIA_END, DATA_OUT_BEGIN, DATA_OUT_END }; + struct HBF_Transaction_Observation { + std::uint64_t transaction_id, external_request_id; ++ std::uint64_t maintenance_request_id, maintenance_parent_id; + unsigned int source, type; + std::uint64_t logical_page; bool logical_page_known; + std::uint32_t bytes; +diff --git a/src/ssd/NVM_Transaction.h b/src/ssd/NVM_Transaction.h +index 4cd5702..84e7a77 100644 +--- a/src/ssd/NVM_Transaction.h ++++ b/src/ssd/NVM_Transaction.h +@@ -12,7 +12,7 @@ namespace SSD_Components + class User_Request; + + enum class Transaction_Type { READ, WRITE, ERASE, UNKOWN }; +- enum class Transaction_Source_Type { USERIO, CACHE, GC_WL, MAPPING }; ++ enum class Transaction_Source_Type { USERIO, CACHE, GC_WL, MAPPING, HBF_MAINTENANCE }; + + class NVM_Transaction + { +diff --git a/src/ssd/NVM_Transaction_Flash.h b/src/ssd/NVM_Transaction_Flash.h +index ccb8fd6..5dcb72c 100644 +--- a/src/ssd/NVM_Transaction_Flash.h ++++ b/src/ssd/NVM_Transaction_Flash.h +@@ -34,6 +34,11 @@ namespace SSD_Components + bool Physical_address_determined; + sim_time_type Estimated_alone_waiting_time;//Used in scheduling methods, such as FLIN, where fairness and QoS is considered in scheduling + bool FLIN_Barrier;//Especially used in queue reordering in FLIN scheduler ++ // Experimental EQ3 maintenance identity. Zero means no maintenance parent. ++ std::uint64_t HBF_Maintenance_ID{0}; ++ std::uint64_t HBF_Maintenance_Parent_ID{0}; ++ bool HBF_Force_Failure{false}; ++ bool HBF_Command_Succeeded{true}; + private: + + }; +diff --git a/src/ssd/TSU_FLIN.cpp b/src/ssd/TSU_FLIN.cpp +index 383bb4f..1790313 100644 +--- a/src/ssd/TSU_FLIN.cpp ++++ b/src/ssd/TSU_FLIN.cpp +@@ -250,6 +250,7 @@ namespace SSD_Components + MappingReadTRQueue[channel_id][chip_id].push_back(*it); + break; + case Transaction_Source_Type::GC_WL: ++ case Transaction_Source_Type::HBF_MAINTENANCE: + GCReadTRQueue[channel_id][chip_id].push_back(*it); + break; + default: +@@ -298,6 +299,7 @@ namespace SSD_Components + MappingWriteTRQueue[channel_id][chip_id].push_back(*it); + break; + case Transaction_Source_Type::GC_WL: ++ case Transaction_Source_Type::HBF_MAINTENANCE: + GCWriteTRQueue[channel_id][chip_id].push_back(*it); + break; + default: +@@ -589,4 +591,4 @@ namespace SSD_Components + void TSU_FLIN::Report_results_in_XML(std::string name_prefix, Utils::XmlWriter& xmlwriter) {} + + } +-*/ +\ No newline at end of file ++*/ +diff --git a/src/ssd/TSU_OutofOrder.cpp b/src/ssd/TSU_OutofOrder.cpp +index e5cb545..ff65f58 100644 +--- a/src/ssd/TSU_OutofOrder.cpp ++++ b/src/ssd/TSU_OutofOrder.cpp +@@ -174,6 +174,7 @@ void TSU_OutOfOrder::Schedule() + MappingReadTRQueue[(*it)->Address.ChannelID][(*it)->Address.ChipID].push_back((*it)); + break; + case Transaction_Source_Type::GC_WL: ++ case Transaction_Source_Type::HBF_MAINTENANCE: + GCReadTRQueue[(*it)->Address.ChannelID][(*it)->Address.ChipID].push_back((*it)); + break; + default: +@@ -191,6 +192,7 @@ void TSU_OutOfOrder::Schedule() + MappingWriteTRQueue[(*it)->Address.ChannelID][(*it)->Address.ChipID].push_back((*it)); + break; + case Transaction_Source_Type::GC_WL: ++ case Transaction_Source_Type::HBF_MAINTENANCE: + GCWriteTRQueue[(*it)->Address.ChannelID][(*it)->Address.ChipID].push_back((*it)); + break; + default: +diff --git a/src/ssd/TSU_Priority_OutOfOrder.cpp b/src/ssd/TSU_Priority_OutOfOrder.cpp +index a6b29b7..1673097 100644 +--- a/src/ssd/TSU_Priority_OutOfOrder.cpp ++++ b/src/ssd/TSU_Priority_OutOfOrder.cpp +@@ -235,6 +235,7 @@ void TSU_Priority_OutOfOrder::Schedule() + MappingReadTRQueue[(*it)->Address.ChannelID][(*it)->Address.ChipID].push_back((*it)); + break; + case Transaction_Source_Type::GC_WL: ++ case Transaction_Source_Type::HBF_MAINTENANCE: + GCReadTRQueue[(*it)->Address.ChannelID][(*it)->Address.ChipID].push_back((*it)); + break; + default: +@@ -259,6 +260,7 @@ void TSU_Priority_OutOfOrder::Schedule() + MappingWriteTRQueue[(*it)->Address.ChannelID][(*it)->Address.ChipID].push_back((*it)); + break; + case Transaction_Source_Type::GC_WL: ++ case Transaction_Source_Type::HBF_MAINTENANCE: + GCWriteTRQueue[(*it)->Address.ChannelID][(*it)->Address.ChipID].push_back((*it)); + break; + default: diff --git a/experiments/eq3_maintenance/backend/patches/series.json b/experiments/eq3_maintenance/backend/patches/series.json new file mode 100644 index 0000000..c7fd96a --- /dev/null +++ b/experiments/eq3_maintenance/backend/patches/series.json @@ -0,0 +1,22 @@ +{ + "schema_version": "eq3-isolated-mqsim-patch-series-v1", + "upstream_revision": "51f0f2d3fed92d88ef4a0fa61a38024b07bf9d16", + "patches": [ + { + "path": "../../../../patches/mqsim/0001-online-hbf-api.patch", + "sha256": "e145f8fae5faf55dca89b4b987dc73e010dbe0d8dc06906d89edd61e7a19e167" + }, + { + "path": "../../../../patches/mqsim/0002-qlc-support.patch", + "sha256": "246669fd8d347f695565987d3d0428a7fe28cf382baf5cf1c3f219bcd2b07f8f" + }, + { + "path": "../../../../patches/mqsim/0003-hbf-command-observer.patch", + "sha256": "05fcd2a90e9defce5d21258b0c18b0271dd2285b7e76c4e9e7973ca5257a8724" + }, + { + "path": "0004-eq3-maintenance.patch", + "sha256": "4985a1afc3da2112d3baccb373accf545ca8ed8bfe30c37038e91802522347d5" + } + ] +} diff --git a/experiments/eq3_maintenance/backend/service/hbf_mqsim_maintenance_service.cpp b/experiments/eq3_maintenance/backend/service/hbf_mqsim_maintenance_service.cpp new file mode 100644 index 0000000..e145895 --- /dev/null +++ b/experiments/eq3_maintenance/backend/service/hbf_mqsim_maintenance_service.cpp @@ -0,0 +1,510 @@ +// CPU-only JSON-lines transport for an external causal replay controller. +// One process owns one MQSim engine; read requests share its real service queue. +#include +#include +#include +#include +#include +#include +#include "replay_json.hpp" + +#include +#include +#include +#include +#include +#include +#include +#include +#include + +namespace { +using Json = nlohmann::json; +using hbfsim::eval::integer; + +std::uint64_t integer_value(const Json& value,const char* field) +{ + if(value.is_number_unsigned())return value.get(); + if(!value.is_number_integer()||value.get()<0) + throw std::invalid_argument(std::string(field)+" requires a nonnegative integer"); + return static_cast(value.get()); +} + +hbfsim::eq3_thermal::MqsimStackMapAdapter load_stack_map( + const std::string& path,const hbfsim::Profile& profile) +{ + std::ifstream input(path); + if(!input)throw std::runtime_error("failed to open MQSim stack map"); + const auto config=Json::parse(input); + if(integer(config,"schema_version")!=1||config.at("physical_kind")!="HBF"|| + config.at("route")!="direct"||config.at("address_layout")!="GLOBAL_PAGE_STRIPE_V1"|| + config.at("plane_allocation_scheme")!="CWDP"|| + integer(config,"page_bytes")!=profile.page_bytes|| + integer(config,"channels")!=profile.channels|| + integer(config,"dies_per_channel")!=profile.dies_per_channel) + throw std::invalid_argument("MQSim stack map does not match the HBF profile"); + std::vector groups; + for(const auto& row:config.at("stacks")) { + hbfsim::eq3_thermal::MqsimStackChannelGroup group; + group.stack_id=row.at("id").get(); + const auto declared=integer(row,"declared_dies"); + if(declared>std::numeric_limits::max()) + throw std::invalid_argument("MQSim stack-map die count overflows"); + group.declared_dies=static_cast(declared); + for(const auto& channel:row.at("channels")) { + const auto value=integer_value(channel,"channel"); + if(value>std::numeric_limits::max()) + throw std::invalid_argument("MQSim stack-map channel overflows"); + group.channels.push_back(static_cast(value)); + } + groups.push_back(std::move(group)); + } + return {profile,std::move(groups)}; +} + +using PlacementLedger=std::map; + +struct MaintenanceLedger { + struct Record { + hbfsim::MqsimMaintenanceRequest request; + std::string stack; + std::uint64_t stack_local_page{}; + std::string trigger_reason; + }; + std::map accepted; + std::set completed; +}; + +const char* maintenance_state_name(hbfsim::MqsimMaintenanceState state) +{ + using S=hbfsim::MqsimMaintenanceState; + switch(state) { + case S::Due:return "DUE"; case S::Queued:return "QUEUED"; + case S::Read:return "READ"; case S::ProgramDest:return "PROGRAM_DEST"; + case S::Commit:return "COMMIT"; case S::RetireOld:return "RETIRE_OLD"; + case S::ReclaimErase:return "RECLAIM_ERASE"; case S::Done:return "DONE"; + case S::Failed:return "FAILED"; + } + throw std::logic_error("unknown maintenance state"); +} + +const char* maintenance_status_name(hbfsim::MqsimMaintenanceStatus status) +{ + using S=hbfsim::MqsimMaintenanceStatus; + switch(status) { + case S::Committed:return "COMMITTED"; + case S::CommittedReclaimDeferred:return "COMMITTED_RECLAIM_DEFERRED"; + case S::RejectedUnsupported:return "REJECTED_UNSUPPORTED"; + case S::RejectedInvalidTarget:return "REJECTED_INVALID_TARGET"; + case S::RejectedUnmapped:return "REJECTED_UNMAPPED"; + case S::RejectedNoSpare:return "REJECTED_NO_SPARE"; + case S::FailedRead:return "FAILED_READ"; + case S::FailedProgram:return "FAILED_PROGRAM"; + case S::FailedStaleVersion:return "FAILED_STALE_VERSION"; + case S::FailedAfterCommitNeedsReconcile:return "FAILED_AFTER_COMMIT_NEEDS_RECONCILE"; + case S::RejectedSourceBusy:return "REJECTED_SOURCE_BUSY"; + case S::FailedVersionOverflow:return "FAILED_VERSION_OVERFLOW"; + } + throw std::logic_error("unknown maintenance status"); +} + +hbfsim::MqsimMaintenanceFailurePoint failure_point(const Json& row) +{ + const auto value=row.value("failure_injection",std::string("none")); + if(value=="none")return hbfsim::MqsimMaintenanceFailurePoint::None; + if(value=="read")return hbfsim::MqsimMaintenanceFailurePoint::Read; + if(value=="program")return hbfsim::MqsimMaintenanceFailurePoint::Program; + if(value=="stale_commit")return hbfsim::MqsimMaintenanceFailurePoint::StaleCommit; + if(value=="erase")return hbfsim::MqsimMaintenanceFailurePoint::Erase; + throw std::invalid_argument("unknown maintenance failure injection"); +} + +std::optional requested_stack(const Json& row,bool mapping_enabled) +{ + if(!mapping_enabled)return std::nullopt; + if(!row.contains("stack")||!row.at("stack").is_string()|| + !row.contains("route")||row.at("route")!="direct") + throw std::invalid_argument("enabled MQSim HBF stack map requires stack and direct route"); + const auto stack=row.at("stack").get(); + if(stack.empty())throw std::invalid_argument("empty requested HBF stack identity"); + return stack; +} + +Json placement_json(const hbfsim::eq3_thermal::MqsimStackPlacement& placement) +{ + return {{"external_logical_page",placement.external_page}, + {"backend_logical_page",placement.backend_page}, + {"stack",placement.stack_id ? Json(*placement.stack_id) : Json(nullptr)}, + {"expected_channel",placement.expected_channel ? + Json(*placement.expected_channel) : Json(nullptr)}}; +} + +hbfsim::eq3_thermal::MqsimStackPlacement place_request( + const Json& row,const hbfsim::HbfRequest& external, + const hbfsim::eq3_thermal::MqsimStackMapAdapter& stack_map) +{ + if(!stack_map.enabled())return stack_map.map(external); + const auto requested=requested_stack(row,true); + if(!row.contains("stack_local_page")) + throw std::invalid_argument("enabled MQSim HBF stack map requires stack_local_page"); + const auto local_page=integer(row,"stack_local_page"); + const auto placement=stack_map.map_stack_page(external,*requested,local_page); + if(row.contains("logical_address")&& + integer(row,"logical_address")!=placement.external_page*external.bytes) + throw std::invalid_argument("logical_address conflicts with stack-local placement"); + return placement; +} + +void send(Json reply, hbfsim::MqsimOnlineEngine& engine, + std::vector* native=nullptr, + const hbfsim::eq3_thermal::MqsimStackMapAdapter* stack_map=nullptr, + const PlacementLedger* placements=nullptr, + MaintenanceLedger* maintenance=nullptr) +{ + Json events = Json::array(); + for (const auto& event : engine.take_observations()) { + events.push_back({{"kind", static_cast(event.kind)}, + {"request_id", event.request_id}, {"arrival_ns", event.arrival_ns}, + {"time_ns", event.time_ns}, {"reported_complete", event.modeled_completion_ns}, + {"bytes", event.bytes}, {"device_outstanding", event.device_outstanding}}); + } + reply["events"] = std::move(events); + Json native_events=Json::array(); + if (native) { + if (SSD_Components::NVM_PHY_ONFI_NVDDR2::Hbf_command_observation_failed()) + throw std::runtime_error("native MQSim command observation sink failed"); + for (const auto& event : *native) { + Json transactions=Json::array(); + for (const auto& tr : event.transactions) { + std::optional actual_stack; + const hbfsim::eq3_thermal::MqsimStackPlacement* expected=nullptr; + if(stack_map&&stack_map->enabled())actual_stack=stack_map->stack_for_channel(tr.channel); + if(placements&&tr.external_request_id) { + const auto found=placements->find(tr.external_request_id); + if(found!=placements->end())expected=&found->second; + } + if(expected&&(!actual_stack||!expected->stack_id|| + *actual_stack!=*expected->stack_id|| + !expected->expected_channel||tr.channel!=*expected->expected_channel|| + (tr.logical_page_known&&tr.logical_page!=expected->backend_page))) + throw std::runtime_error("native MQSim command violated configured HBF stack placement"); + transactions.push_back({{"transaction_id",tr.transaction_id}, + {"external_request_id",tr.external_request_id ? Json(tr.external_request_id) : Json(nullptr)}, + {"maintenance_request_id",tr.maintenance_request_id ? Json(tr.maintenance_request_id) : Json(nullptr)}, + {"maintenance_parent_id",tr.maintenance_parent_id ? Json(tr.maintenance_parent_id) : Json(nullptr)}, + {"source",tr.source},{"type",tr.type}, + {"logical_page",tr.logical_page_known ? Json(tr.logical_page) : Json(nullptr)}, + {"bytes",tr.bytes}, + {"stack",actual_stack ? Json(*actual_stack) : Json(nullptr)}, + {"expected_stack",expected&&expected->stack_id ? Json(*expected->stack_id) : Json(nullptr)}, + {"external_logical_page",expected ? Json(expected->external_page) : Json(nullptr)}, + {"backend_logical_page",expected ? Json(expected->backend_page) : Json(nullptr)}, + {"channel",tr.channel},{"chip",tr.chip}, + {"die",tr.die},{"plane",tr.plane},{"block",tr.block},{"page",tr.page}}); + } + native_events.push_back({{"command_id",event.command_id}, + {"phase",static_cast(event.phase)},{"time_ns",event.time}, + {"command_code",event.command_code},{"transactions",std::move(transactions)}}); + } + native->clear(); + } + reply["native_command_events"] = std::move(native_events); + Json maintenance_events=Json::array(); + for(const auto& event:engine.take_maintenance_events()) { + Json destination=nullptr; + if(event.destination_channel) + destination={{"channel",*event.destination_channel},{"chip",*event.destination_chip}, + {"die",*event.destination_die},{"plane",*event.destination_plane}, + {"block",*event.destination_block},{"page",*event.destination_page}}; + maintenance_events.push_back({{"request_id",event.request_id},{"parent_id",event.parent_id}, + {"state",maintenance_state_name(event.state)},{"time_ns",event.time_ns}, + {"transaction_id",event.transaction_id ? Json(event.transaction_id) : Json(nullptr)}, + {"source",{{"channel",event.source_channel},{"chip",event.source_chip}, + {"die",event.source_die},{"plane",event.source_plane}, + {"block",event.source_block},{"page",event.source_page}}}, + {"destination",std::move(destination)}}); + } + reply["maintenance_events"]=std::move(maintenance_events); + Json maintenance_completions=Json::array(); + for(const auto& completion:engine.take_maintenance_completions()) { + if(!maintenance || !maintenance->accepted.contains(completion.request_id) || + !maintenance->completed.insert(completion.request_id).second) + throw std::runtime_error("unknown or duplicate maintenance completion"); + const auto& record=maintenance->accepted.at(completion.request_id); + maintenance_completions.push_back({{"request_id",completion.request_id}, + {"parent_id",completion.parent_id},{"status",maintenance_status_name(completion.status)}, + {"enqueue_ns",completion.enqueue_ns},{"start_ns",completion.start_ns}, + {"end_ns",completion.end_ns},{"logical_page",completion.logical_page}, + {"source_version",completion.source_version}, + {"committed_version",completion.committed_version}, + {"mapping_committed",completion.mapping_committed}, + {"source_retired",completion.source_retired},{"erase_completed",completion.erase_completed}, + {"age_reset_ns",completion.mapping_committed ? Json(completion.end_ns) : Json(nullptr)}, + {"stack",record.stack},{"stack_local_page",record.stack_local_page}, + {"trigger_reason",record.trigger_reason},{"coverage_pages",1}, + {"deadline_ns",record.request.deadline_ns ? Json(record.request.deadline_ns) : Json(nullptr)}, + {"deadline_met",!record.request.deadline_ns || completion.end_ns<=record.request.deadline_ns}, + {"data_semantics","METADATA_VERSION_VALIDITY"}, + {"transaction_ids",completion.transaction_ids}}); + } + reply["maintenance_completions"]=std::move(maintenance_completions); + reply["now_ns"] = engine.current_time_ns(); + reply["pending"] = engine.pending(); + reply["pending_maintenance"] = engine.pending_maintenance(); + std::cout << reply.dump() << '\n' << std::flush; + if (!std::cout) throw std::runtime_error("service response pipe failed"); +} +} + +int main(int argc, char** argv) +{ + try { + std::string profile_path; + std::string stack_map_path; + std::uint64_t requested_units = 0; + std::optional gate_not_before_ns; + std::optional native_command_option; + for (int i=1; i(profile.channels)*profile.dies_per_channel*profile.planes_per_die; + if (requested_units && requested_units!=units) + throw std::invalid_argument("PROJECTED_ANALYTICAL required: no exact configured MQSim topology"); + hbfsim::eq3_thermal::MqsimStackMapAdapter stack_map; + if(!stack_map_path.empty())stack_map=load_stack_map(stack_map_path,profile); + if(stack_map.enabled()&&native_command_option&&!*native_command_option) + throw std::invalid_argument("MQSim stack map requires native command verification"); + const bool native_command_observations=stack_map.enabled()||native_command_option.value_or(false); + std::vector native_events; + hbfsim::MqsimOnlineEngine engine(profile); + if (native_command_observations) { + SSD_Components::NVM_PHY_ONFI_NVDDR2::Set_hbf_command_observation_sink( + [&native_events](const auto& event) { native_events.push_back(event); }); + } + engine.enable_observations(); + hbfsim::eq3_thermal::MqsimSubmissionGateAdapter gate( + engine, + gate_not_before_ns ? hbfsim::eq3_thermal::MqsimGateMode::Enabled + : hbfsim::eq3_thermal::MqsimGateMode::Off, + [gate_not_before_ns](const hbfsim::HbfRequest&,std::uint64_t now) { + if (gate_not_before_ns&&now<*gate_not_before_ns) + return hbfsim::eq3_thermal::MqsimGateDecision{ + hbfsim::eq3_thermal::MqsimGateDisposition::Defer, + *gate_not_before_ns, + "ENGINEERING_FIXTURE target-time gate"}; + return hbfsim::eq3_thermal::MqsimGateDecision{}; + }); + MaintenanceLedger maintenance; + send({{"schema_version", 1}, {"service_source", "MQSIM_SIMULATED"}, {"provenance", "PROJECTED"}, + {"scope", "READ_ONLY_MEDIA_SERVICE_NOT_HARDWARE"}, {"queue_depth", profile.queue_depth}, + {"parallel_units", units}, + {"page_bytes",profile.page_bytes}, + {"aggregate_bandwidth_bytes_per_s",profile.aggregate_bandwidth_bytes_per_s}, + {"stack_mapping",stack_map.enabled() ? + "ACTUAL_MQSIM_CHANNEL_PARTITIONED_HBF_STACKS" : "OFF"}, + {"stack_mapping_evidence",stack_map.enabled() ? + "ENGINEERING_FIXTURE_CONFIGURATION_NOT_RESEARCH_GEOMETRY" : "OFF"}, + {"submission_gate", gate_not_before_ns ? "ENGINEERING_FIXTURE_ENABLED" : "OFF"}, + {"native_command_observations",native_command_observations ? "ON" : "OFF"}, + {"maintenance_backend","EXPERIMENTAL_OUT_OF_PLACE_PAGE_MAINTENANCE"}, + {"maintenance_data_semantics","METADATA_VERSION_VALIDITY"}, + {"maintenance_version_semantics","MONOTONIC_LPA_MAPPING_GENERATION_U64"}, + {"maintenance_payload_validation","UNAVAILABLE"}, + {"maintenance_requires_stack_map",true}}, + engine,native_command_observations ? &native_events : nullptr, + stack_map.enabled() ? &stack_map : nullptr,nullptr,&maintenance); + std::map accepted; + PlacementLedger placements; + std::set completed; + std::uint64_t issued_bytes=0, completed_bytes=0, sequence=0; + std::string line; + while (std::getline(std::cin, line)) { + const auto input=Json::parse(line); + const auto command=input.at("command").get(); + if (command=="submit") { + const auto& rows=input.at("requests"); + if (!rows.is_array() || rows.empty()) throw std::invalid_argument("submit needs a nonempty request array"); + std::vector batch; + std::vector batch_placements; + std::set ids; + auto total=issued_bytes; + // Validate the whole batch before any request is accepted. + for (const auto& row : rows) { + const auto id=integer(row, "request_id"), issue=integer(row, "issue_ns"); + const auto bytes=integer(row, "bytes"); + const auto address=stack_map.enabled() ? 0 : integer(row,"logical_address"); + if (!id || accepted.contains(id) || !ids.insert(id).second) + throw std::invalid_argument("zero or reused request ID"); + if (row.at("operation")!="read" || !bytes || bytes>std::numeric_limits::max() + || bytes%512 || address%512 || bytes>profile.capacity_bytes || address>profile.capacity_bytes-bytes + || issuestd::numeric_limits::max()-bytes) + throw std::overflow_error("request byte total overflow"); + total+=bytes; + const hbfsim::HbfRequest external{ + .request_id=id, .sequence=0, .arrival_ns=issue, + .logical_address=address, .bytes=static_cast(bytes), .operation=0}; + auto placement=place_request(row,external,stack_map); + batch.push_back(placement.backend_request); + batch_placements.push_back(std::move(placement)); + } + if (batch.size()>std::numeric_limits::max()-sequence) + throw std::overflow_error("service sequence exhausted"); + for (std::size_t i=0;istd::numeric_limits::max() || bytes%512 || address%512 || + bytes>profile.capacity_bytes || address>profile.capacity_bytes-bytes || + (!gate_not_before_ns&&issuestd::numeric_limits::max()-bytes || + sequence==std::numeric_limits::max()) + throw std::invalid_argument("invalid gated read extent, identity or arrival"); + const hbfsim::HbfRequest external{.request_id=id,.sequence=sequence+1,.arrival_ns=issue, + .logical_address=address,.bytes=static_cast(bytes),.operation=0}; + auto placement=place_request(row,external,stack_map); + const auto attempt=gate.try_submit(placement.backend_request); + Json gate_result={{"submitted",attempt.submitted}, + {"disposition",static_cast(attempt.decision.disposition)}, + {"reason",attempt.decision.reason}, + {"original_arrival_ns",attempt.original_arrival_ns}, + {"evaluated_ns",attempt.evaluated_ns}}; + gate_result["target_ns"]=attempt.decision.target_time_ns ? + Json(*attempt.decision.target_time_ns) : Json(nullptr); + gate_result["backend_arrival_ns"]=attempt.backend_arrival_ns ? + Json(*attempt.backend_arrival_ns) : Json(nullptr); + gate_result["external_wait_ns"]=attempt.external_wait_ns ? + Json(*attempt.external_wait_ns) : Json(nullptr); + if (attempt.submitted) { + ++sequence;issued_bytes+=bytes; + accepted.emplace(id,static_cast(bytes)); + placement.backend_request.sequence=sequence; + if(stack_map.enabled())placements.emplace(id,placement); + } + send({{"gate",std::move(gate_result)}, + {"placement",stack_map.enabled() ? placement_json(placement) : Json(nullptr)}},engine, + native_command_observations ? &native_events : nullptr, + stack_map.enabled() ? &stack_map : nullptr, + stack_map.enabled() ? &placements : nullptr,&maintenance); + } else if (command=="maintain") { + if(!stack_map.enabled()) + throw std::invalid_argument("maintenance requires an explicit HBF stack map"); + const auto id=integer(input,"request_id"); + const auto parent=input.contains("parent_id") ? integer(input,"parent_id") : id; + const auto due=integer(input,"due_ns"), deadline=integer(input,"deadline_ns"); + const auto local_page=integer(input,"stack_local_page"); + if(!id||!parent||maintenance.accepted.contains(id)|| + (deadline&&deadline(),local_page); + const auto backend_page=placement.backend_page; + const auto channel=static_cast(backend_page%profile.channels); + const auto die=static_cast( + (backend_page/profile.channels)%profile.dies_per_channel); + const auto plane=static_cast( + (backend_page/(static_cast(profile.channels)*profile.dies_per_channel))% + profile.planes_per_die); + if(!placement.expected_channel||channel!=*placement.expected_channel) + throw std::logic_error("maintenance placement disagrees with configured channel group"); + hbfsim::MqsimMaintenanceRequest request{ + .request_id=id,.parent_id=parent,.due_ns=due,.deadline_ns=deadline, + .logical_page=backend_page,.channel=channel,.chip=0,.die=die,.plane=plane, + .plane_is_exact=true, + .reclaim_invalid_source_block=input.value("reclaim_source_block",false), + .failure_point=failure_point(input)}; + const auto reason=input.value("trigger_reason",std::string("UNSPECIFIED")); + if(reason.empty())throw std::invalid_argument("empty maintenance trigger reason"); + engine.submit_maintenance(request); + maintenance.accepted.emplace(id,MaintenanceLedger::Record{ + request,input.at("stack").get(),local_page,reason}); + send({{"maintenance_accepted",1}, + {"placement",placement_json(placement)}, + {"target",{{"channel",channel},{"chip",0},{"die",die},{"plane",plane}}}, + {"failure_injection",input.value("failure_injection",std::string("none"))}, + {"trigger_reason",reason},{"coverage_pages",1}, + {"data_semantics","METADATA_VERSION_VALIDITY"}}, + engine,native_command_observations ? &native_events : nullptr, + &stack_map,&placements,&maintenance); + } else if (command=="until") { + const auto completion=engine.run_next_completion_until(integer(input, "deadline_ns")); + Json value=nullptr; + if (completion) { + if (completion->status!=static_cast(hbfsim::RequestStatus::Ready) + || !accepted.contains(completion->request_id) || !completed.insert(completion->request_id).second) + throw std::runtime_error("unknown, duplicate or failed completion"); + completed_bytes+=accepted.at(completion->request_id); + value={{"request_id", completion->request_id}, {"reported_complete", completion->modeled_completion_ns}, + {"status", completion->status}}; + } + send({{"completion", value}}, engine, + native_command_observations ? &native_events : nullptr, + stack_map.enabled() ? &stack_map : nullptr, + stack_map.enabled() ? &placements : nullptr,&maintenance); + } else if (command=="finish") { + if (engine.pending() || engine.pending_maintenance() || + accepted.size()!=completed.size() || issued_bytes!=completed_bytes || + maintenance.accepted.size()!=maintenance.completed.size()) + throw std::runtime_error("cannot finish with unreturned requests/bytes"); + send({{"status", "FINISHED"}, {"issued", accepted.size()}, {"completed", completed.size()}, + {"issued_bytes", issued_bytes}, {"completed_bytes", completed_bytes}, + {"maintenance_issued",maintenance.accepted.size()}, + {"maintenance_completed",maintenance.completed.size()}}, engine, + native_command_observations ? &native_events : nullptr, + stack_map.enabled() ? &stack_map : nullptr, + stack_map.enabled() ? &placements : nullptr,&maintenance); + if (native_command_observations) + SSD_Components::NVM_PHY_ONFI_NVDDR2::Clear_hbf_command_observation_sink(); + return 0; + } else throw std::invalid_argument("unknown service command"); + } + throw std::runtime_error("service input closed without a complete finish record"); + } catch (const std::exception& error) { + std::cerr << "hbf_mqsim_service: " << error.what() << '\n'; + return 2; + } +} diff --git a/experiments/eq3_maintenance/backend/service/replay_json.hpp b/experiments/eq3_maintenance/backend/service/replay_json.hpp new file mode 100644 index 0000000..354e6b3 --- /dev/null +++ b/experiments/eq3_maintenance/backend/service/replay_json.hpp @@ -0,0 +1,17 @@ +#pragma once + +#include +#include +#include +#include + +namespace hbfsim::eval { +inline std::uint64_t integer(const nlohmann::json& object, const char* key) +{ + const auto& value = object.at(key); + if (value.is_number_unsigned()) return value.get(); + if (!value.is_number_integer() || value.get() < 0) + throw std::invalid_argument(std::string(key)+" requires a nonnegative integer"); + return static_cast(value.get()); +} +} diff --git a/experiments/eq3_maintenance/backend/src/mqsim_online_maintenance.cpp b/experiments/eq3_maintenance/backend/src/mqsim_online_maintenance.cpp new file mode 100644 index 0000000..1bb3f4e --- /dev/null +++ b/experiments/eq3_maintenance/backend/src/mqsim_online_maintenance.cpp @@ -0,0 +1,793 @@ +#include + +#include +#include +#include +#include +#include +#include +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +namespace hbfsim +{ + namespace + { + + constexpr std::uint64_t kSectorBytes = 512; + + Flash_Technology_Type to_mqsim_technology(NandTechnology technology) + { + switch (technology) + { + case NandTechnology::Slc: + return Flash_Technology_Type::SLC; + case NandTechnology::Mlc: + return Flash_Technology_Type::MLC; + case NandTechnology::Tlc: + return Flash_Technology_Type::TLC; + case NandTechnology::Qlc: + return Flash_Technology_Type::QLC; + } + throw std::invalid_argument("unknown NandTechnology"); + } + + SSD_Components::Flash_Plane_Allocation_Scheme_Type to_mqsim_scheme( + PlaneAllocationScheme scheme) + { + using Scheme = SSD_Components::Flash_Plane_Allocation_Scheme_Type; + switch (scheme) + { + case PlaneAllocationScheme::Cwdp: + return Scheme::CWDP; + case PlaneAllocationScheme::Cwpd: + return Scheme::CWPD; + case PlaneAllocationScheme::Cdwp: + return Scheme::CDWP; + case PlaneAllocationScheme::Cdpw: + return Scheme::CDPW; + case PlaneAllocationScheme::Cpwd: + return Scheme::CPWD; + case PlaneAllocationScheme::Cpdw: + return Scheme::CPDW; + case PlaneAllocationScheme::Wcdp: + return Scheme::WCDP; + case PlaneAllocationScheme::Wcpd: + return Scheme::WCPD; + case PlaneAllocationScheme::Wdcp: + return Scheme::WDCP; + case PlaneAllocationScheme::Wdpc: + return Scheme::WDPC; + case PlaneAllocationScheme::Wpcd: + return Scheme::WPCD; + case PlaneAllocationScheme::Wpdc: + return Scheme::WPDC; + case PlaneAllocationScheme::Dcwp: + return Scheme::DCWP; + case PlaneAllocationScheme::Dcpw: + return Scheme::DCPW; + case PlaneAllocationScheme::Dwcp: + return Scheme::DWCP; + case PlaneAllocationScheme::Dwpc: + return Scheme::DWPC; + case PlaneAllocationScheme::Dpcw: + return Scheme::DPCW; + case PlaneAllocationScheme::Dpwc: + return Scheme::DPWC; + case PlaneAllocationScheme::Pcwd: + return Scheme::PCWD; + case PlaneAllocationScheme::Pcdw: + return Scheme::PCDW; + case PlaneAllocationScheme::Pwcd: + return Scheme::PWCD; + case PlaneAllocationScheme::Pwdc: + return Scheme::PWDC; + case PlaneAllocationScheme::Pdcw: + return Scheme::PDCW; + case PlaneAllocationScheme::Pdwc: + return Scheme::PDWC; + } + throw std::invalid_argument("unknown PlaneAllocationScheme"); + } + + std::uint64_t checked_end(std::uint64_t start, std::uint64_t bytes) + { + if (bytes > std::numeric_limits::max() - start) + { + throw std::invalid_argument("HBF request address range overflows"); + } + return start + bytes; + } + + void configure_mqsim(const Profile &profile) + { + Device_Parameter_Set::Seed = 123; + Device_Parameter_Set::Enabled_Preconditioning = false; + Device_Parameter_Set::Memory_Type = NVM::NVM_Type::FLASH; + Device_Parameter_Set::HostInterface_Type = HostInterface_Types::HBF; + Device_Parameter_Set::IO_Queue_Depth = static_cast( + std::min(profile.queue_depth, + std::numeric_limits::max())); + Device_Parameter_Set::Caching_Mechanism = + SSD_Components::Caching_Mechanism::SIMPLE; + Device_Parameter_Set::Address_Mapping = + SSD_Components::Flash_Address_Mapping_Type::PAGE_LEVEL; + Device_Parameter_Set::Plane_Allocation_Scheme = + to_mqsim_scheme(profile.plane_allocation_scheme); + Device_Parameter_Set::Ideal_Mapping_Table = true; + Device_Parameter_Set::Overprovisioning_Ratio = 0.0; + Device_Parameter_Set::Flash_Channel_Count = profile.channels; + Device_Parameter_Set::Flash_Channel_Width = profile.channel_width_bits / 8; + Device_Parameter_Set::Channel_Transfer_Rate = + profile.channel_transfer_rate_mtps; + Device_Parameter_Set::Chip_No_Per_Channel = 1; + Device_Parameter_Set::Flash_Comm_Protocol = + SSD_Components::ONFI_Protocol::NVDDR2; + + Flash_Parameter_Set::Flash_Technology = to_mqsim_technology(profile.nand_technology); + Flash_Parameter_Set::CMD_Suspension_Support = + NVM::FlashMemory::Command_Suspension_Mode::NONE; + Flash_Parameter_Set::Page_Read_Latency_LSB = profile.page_read_latency_lsb_ns; + Flash_Parameter_Set::Page_Read_Latency_CSB = profile.page_read_latency_csb_ns; + Flash_Parameter_Set::Page_Read_Latency_MSB = profile.page_read_latency_msb_ns; + Flash_Parameter_Set::Page_Read_Latency_TSB = profile.page_read_latency_tsb_ns; + Flash_Parameter_Set::Page_Program_Latency_LSB = profile.page_program_latency_lsb_ns; + Flash_Parameter_Set::Page_Program_Latency_CSB = profile.page_program_latency_csb_ns; + Flash_Parameter_Set::Page_Program_Latency_MSB = profile.page_program_latency_msb_ns; + Flash_Parameter_Set::Page_Program_Latency_TSB = profile.page_program_latency_tsb_ns; + Flash_Parameter_Set::Block_Erase_Latency = profile.program_latency_ns * 10; + Flash_Parameter_Set::Suspend_Erase_Time = 0; + Flash_Parameter_Set::Suspend_Program_Time = 0; + Flash_Parameter_Set::Die_No_Per_Chip = profile.dies_per_channel; + Flash_Parameter_Set::Plane_No_Per_Die = profile.planes_per_die; + Flash_Parameter_Set::Block_No_Per_Plane = + static_cast(blocks_per_plane(profile)); + Flash_Parameter_Set::Page_No_Per_Block = profile.pages_per_block; + Flash_Parameter_Set::Page_Capacity = profile.page_bytes; + Flash_Parameter_Set::Page_Metadat_Capacity = 0; + } + + } // namespace + + class ArrivalInjector final : public MQSimEngine::Sim_Object + { + public: + using Handler = std::function; + + explicit ArrivalInjector(Handler handler, const char* name = "HBFSim.ArrivalInjector") + : Sim_Object(name), handler_(std::move(handler)) + { + } + + void Start_simulation() override {} + void Validate_simulation_config() override {} + void Execute_simulator_event(MQSimEngine::Sim_Event *event) override; + + private: + Handler handler_; + }; + + class MqsimOnlineEngine::Impl + { + public: + struct Submission + { + HbfRequest descriptor; + SSD_Components::User_Request *mqsim_request; + bool readmitted{false}; + }; + + struct MaintenanceSubmission + { + MqsimMaintenanceRequest descriptor; + }; + + explicit Impl(const Profile &requested_profile) + : profile(requested_profile), + queue_depth(std::max(1, requested_profile.queue_depth)) + { + validate_profile(profile); + if (profile.channel_width_bits % 8 != 0) + { + throw std::invalid_argument( + "channel_width_bits must be divisible by eight for MQSim"); + } + if (blocks_per_plane(profile) < 4) + { + throw std::invalid_argument( + "MQSim requires at least four blocks per plane"); + } + + Simulator->Reset(); + configure_mqsim(profile); + + flow.Device_Level_Data_Caching_Mode = + SSD_Components::Caching_Mode::TURNED_OFF; + flow.Priority_Class = IO_Flow_Priority_Class::URGENT; + flow.Initial_Occupancy_Percentage = 0; + flows.push_back(&flow); + device = std::make_unique(¶meters, &flows); + host = dynamic_cast( + device->Host_interface); + if (host == nullptr) + { + throw std::runtime_error("MQSim did not construct the HBF interface"); + } + + auto* ftl = static_cast(device->Firmware); + maintenance = std::make_unique( + device->ID() + ".EQ3Maintenance", ftl->Address_Mapping_Unit, + ftl->BlockManager, ftl->TSU, ftl->PHY, profile.page_bytes, + profile.channels, 1, profile.dies_per_channel, + profile.planes_per_die); + maintenance->Set_event_sink([this](const auto& event) { + const auto& source = event.source; + MqsimMaintenanceEvent value{ + .request_id = event.request_id, + .parent_id = event.parent_id, + .state = static_cast(event.state), + .time_ns = static_cast(event.time), + .transaction_id = event.transaction_observation_id, + .source_channel = source.ChannelID, .source_chip = source.ChipID, + .source_die = source.DieID, .source_plane = source.PlaneID, + .source_block = source.BlockID, .source_page = source.PageID, + }; + if (event.destination_known) { + value.destination_channel = event.destination.ChannelID; + value.destination_chip = event.destination.ChipID; + value.destination_die = event.destination.DieID; + value.destination_plane = event.destination.PlaneID; + value.destination_block = event.destination.BlockID; + value.destination_page = event.destination.PageID; + } + maintenance_events.push_back(std::move(value)); + }); + Simulator->AddObject(maintenance.get()); + + injector = std::make_unique( + [this](MQSimEngine::Sim_Event *event) + { + submit_to_device(static_cast(event->Parameters)); + }); + Simulator->AddObject(injector.get()); + maintenance_injector = std::make_unique( + [this](MQSimEngine::Sim_Event* event) { + auto* submission = static_cast(event->Parameters); + submit_maintenance_to_device(submission->descriptor); + delete submission; + }, "HBFSim.EQ3MaintenanceInjector"); + Simulator->AddObject(maintenance_injector.get()); + } + + ~Impl() + { + flush_staged(); + while ((pending_requests != 0 || pending_maintenance != 0) && Simulator->Run_next_event()) + { + } + // Requests still waiting for a queue slot were never handed to + // MQSim, so this object still owns them. + for (auto *waiting : admission_queue) + { + delete waiting->mqsim_request; + delete waiting; + } + admission_queue.clear(); + Simulator->Reset(); + injector.reset(); + clock_injector.reset(); + maintenance_injector.reset(); + maintenance.reset(); + device.reset(); + } + + void submit_maintenance_to_device(const MqsimMaintenanceRequest& descriptor) + { + SSD_Components::HBF_Maintenance_Request request{ + descriptor.request_id, descriptor.parent_id, 0, + descriptor.logical_page, + static_cast(descriptor.channel), + static_cast(descriptor.chip), + static_cast(descriptor.die), + static_cast(descriptor.plane), + descriptor.plane_is_exact, + static_cast(descriptor.due_ns), + static_cast(descriptor.deadline_ns), + descriptor.reclaim_invalid_source_block, + static_cast(descriptor.failure_point)}; + maintenance->Submit(request, [this](const auto& completion) { + maintenance_completions.push_back(MqsimMaintenanceCompletion{ + completion.request_id, completion.parent_id, + static_cast(completion.status), + static_cast(completion.enqueue_time), + static_cast(completion.start_time), + static_cast(completion.end_time), + completion.logical_page, + static_cast(completion.source_version), + static_cast(completion.committed_version), + completion.mapping_committed, completion.source_retired, + completion.erase_completed, completion.transaction_observation_ids}); + --pending_maintenance; + }); + } + + void flush_staged() + { + std::sort(staged.begin(), staged.end(), + [](const Submission *left, const Submission *right) + { + if (left->descriptor.arrival_ns != + right->descriptor.arrival_ns) + { + return left->descriptor.arrival_ns < + right->descriptor.arrival_ns; + } + return left->descriptor.sequence < + right->descriptor.sequence; + }); + for (auto *submission : staged) + { + Simulator->Register_sim_event(submission->descriptor.arrival_ns, + injector.get(), submission); + } + staged.clear(); + } + + // The device accepts at most queue_depth requests concurrently. A + // request whose arrival event fires while the device is full waits in + // admission_queue and is re-registered when a slot frees, so the wait + // shows up in the request's modeled latency. Without this bound the + // adapter handed MQSim every request the moment its arrival event + // fired, which models a device with unlimited outstanding requests. + void submit_to_device(Submission *submission) + { + if (submission->readmitted) + { + // release_admission_slots already took the slot for this + // request; taking a second one here would overcount. + submission->readmitted = false; + dispatch_to_device(submission); + return; + } + observe(MqsimEventKind::Arrival, submission->descriptor, + static_cast(Simulator->Time())); + if (in_device >= queue_depth) + { + admission_queue.push_back(submission); + return; + } + ++in_device; + dispatch_to_device(submission); + } + + // The caller has already taken the slot in in_device. + void dispatch_to_device(Submission *submission) + { + observe(MqsimEventKind::Admission, submission->descriptor, + static_cast(Simulator->Time())); + host->Submit_hbf_request( + submission->mqsim_request, + [this, descriptor = submission->descriptor]( + SSD_Components::User_Request *, sim_time_type completion_ns) + { + auto bounded_completion = static_cast(completion_ns); + const auto transfer_ns = static_cast( + (static_cast(descriptor.bytes) * + 1000000000ULL + + profile.aggregate_bandwidth_bytes_per_s - 1) / + profile.aggregate_bandwidth_bytes_per_s); + bandwidth_cursor_ns = + std::max(bandwidth_cursor_ns, descriptor.arrival_ns) + + transfer_ns; + bounded_completion = + std::max(bounded_completion, bandwidth_cursor_ns); + completions.push_back(HbfCompletion{ + .request_id = descriptor.request_id, + .modeled_completion_ns = bounded_completion, + .modeled_ns = bounded_completion - descriptor.arrival_ns, + .service_ns = 0, + .cache_frame_address = 0, + .page_generation = descriptor.page_generation, + .status = + static_cast(RequestStatus::Ready), + .checksum = 0, + .reserved = 0, + }); + --in_device; + observe(MqsimEventKind::Completion, descriptor, + static_cast(completion_ns), + bounded_completion); + release_admission_slots(); + }); + delete submission; + } + + // Called from inside a completion callback, which runs while MQSim is + // servicing an event. Handing the next request straight to + // Submit_hbf_request here would re-enter request segmentation from + // within request completion, so the waiting request is registered as a + // fresh arrival event at the current simulation time instead. + void release_admission_slots() + { + while (in_device < queue_depth && !admission_queue.empty()) + { + auto *next = admission_queue.front(); + admission_queue.pop_front(); + next->readmitted = true; + ++in_device; + Simulator->Register_sim_event(Simulator->Time(), + injector.get(), next); + } + } + + void observe(MqsimEventKind kind, const HbfRequest &request, + std::uint64_t time_ns, + std::uint64_t modeled_completion_ns = 0) + { + if (!observations_enabled) + { + return; + } + observations.push_back(MqsimObservation{ + .kind = kind, + .request_id = request.request_id, + .arrival_ns = request.arrival_ns, + .time_ns = time_ns, + .modeled_completion_ns = modeled_completion_ns, + .bytes = request.bytes, + .device_outstanding = in_device, + }); + } + + Profile profile; + Device_Parameter_Set parameters; + IO_Flow_Parameter_Set flow; + std::vector flows; + std::unique_ptr device; + std::unique_ptr maintenance; + SSD_Components::Host_Interface_HBF *host{nullptr}; + std::unique_ptr injector; + std::unique_ptr clock_injector; + std::unique_ptr maintenance_injector; + bool horizon_reached{false}; + bool ready_marker_reached{false}; + std::vector staged; + std::deque completions; + std::size_t pending_requests{0}; + std::size_t pending_maintenance{0}; + std::vector maintenance_events; + std::vector maintenance_completions; + std::uint64_t bandwidth_cursor_ns{0}; + // queue_depth is the profile's declared depth. in_device counts the + // requests holding one of those slots: handed to MQSim and not yet + // completed, plus any re-registered at the current time by + // release_admission_slots. admission_queue holds the requests that + // arrived while every slot was taken. + std::size_t queue_depth{1}; + std::size_t in_device{0}; + std::deque admission_queue; + bool has_submitted{false}; + bool observations_enabled{false}; + std::vector observations; + }; + + void ArrivalInjector::Execute_simulator_event(MQSimEngine::Sim_Event *event) + { + handler_(event); + } + + MqsimOnlineEngine::MqsimOnlineEngine(const Profile &profile) + : impl_(std::make_unique(profile)) + { + } + + MqsimOnlineEngine::~MqsimOnlineEngine() = default; + MqsimOnlineEngine::MqsimOnlineEngine(MqsimOnlineEngine &&) noexcept = default; + MqsimOnlineEngine &MqsimOnlineEngine::operator=(MqsimOnlineEngine &&) noexcept = + default; + + void MqsimOnlineEngine::submit(const HbfRequest &request) + { + if (request.bytes == 0 || request.bytes % kSectorBytes != 0 || + request.logical_address % kSectorBytes != 0) + { + throw std::invalid_argument( + "MQSim requests must be non-empty and 512-byte aligned"); + } + if (checked_end(request.logical_address, request.bytes) > + impl_->profile.capacity_bytes) + { + throw std::out_of_range("MQSim request exceeds profile capacity"); + } + if (Simulator->Has_started() && request.arrival_ns < Simulator->Time()) + { + throw std::invalid_argument( + "MQSim request arrival precedes current simulation time"); + } + + auto mqsim_request = std::make_unique(); + mqsim_request->Start_LBA = request.logical_address / kSectorBytes; + mqsim_request->Size_in_byte = request.bytes; + mqsim_request->SizeInSectors = request.bytes / kSectorBytes; + mqsim_request->Type = request.operation == + static_cast( + RequestOperation::Write) + ? SSD_Components::UserRequestType::WRITE + : SSD_Components::UserRequestType::READ; + mqsim_request->Stream_id = 0; + mqsim_request->Priority_class = IO_Flow_Priority_Class::URGENT; + mqsim_request->IO_command_info = nullptr; + mqsim_request->Data = nullptr; + // Patched MQSim carries this immutable identity into native transaction + // and command observations. It is not used by scheduling or completion. + mqsim_request->HBF_External_Request_ID = request.request_id; + + auto submission = new Impl::Submission{ + .descriptor = request, + .mqsim_request = mqsim_request.release(), + }; + impl_->staged.push_back(submission); + impl_->has_submitted = true; + ++impl_->pending_requests; + } + + void MqsimOnlineEngine::submit_maintenance(const MqsimMaintenanceRequest& request) + { + if (!request.request_id || + (request.deadline_ns && request.deadline_ns < request.due_ns) || + request.channel >= impl_->profile.channels || request.chip != 0 || + request.die >= impl_->profile.dies_per_channel || + (request.plane_is_exact && request.plane >= impl_->profile.planes_per_die)) + throw std::invalid_argument("invalid MQSim maintenance request"); + if (request.logical_page >= impl_->profile.capacity_bytes / impl_->profile.page_bytes) + throw std::out_of_range("MQSim maintenance logical page exceeds capacity"); + auto* submission = new Impl::MaintenanceSubmission{request}; + Simulator->Register_sim_event(std::max(request.due_ns, current_time_ns()), + impl_->maintenance_injector.get(), submission); + ++impl_->pending_maintenance; + } + + std::optional MqsimOnlineEngine::run_next_completion() + { + // Ignored clock markers stay in MQSim's event tree. Once a caller has + // opted into external-clock control, an empty legacy poll must not + // advance through those cancelled horizons. Legacy-only use is intact. + if (impl_->clock_injector && impl_->pending_requests == 0) + return std::nullopt; + impl_->flush_staged(); + while (impl_->completions.empty() && Simulator->Run_next_event()) + { + } + if (impl_->completions.empty()) + { + return std::nullopt; + } + + auto completion = impl_->completions.front(); + impl_->completions.pop_front(); + --impl_->pending_requests; + return completion; + } + + std::optional MqsimOnlineEngine::run_next_completion_until( + std::uint64_t deadline_ns) + { + if (deadline_ns < current_time_ns()) + throw std::invalid_argument("MQSim horizon precedes current simulation time"); + impl_->flush_staged(); + if (!impl_->clock_injector) + { + impl_->clock_injector = std::make_unique( + [state = impl_.get()](MQSimEngine::Sim_Event* event) + { + if (event->Type == 1) state->horizon_reached = true; + else state->ready_marker_reached = true; + }, "HBFSim.ExternalClock"); + Simulator->AddObject(impl_->clock_injector.get()); + } + impl_->horizon_reached = false; + impl_->ready_marker_reached = false; + auto* horizon = Simulator->Register_sim_event( + deadline_ns, impl_->clock_injector.get(), nullptr, 1); + MQSimEngine::Sim_Event* ready_marker = nullptr; + const auto cancel_markers = [&] + { + if (!impl_->horizon_reached) Simulator->Ignore_sim_event(horizon); + if (ready_marker != nullptr && !impl_->ready_marker_reached) + Simulator->Ignore_sim_event(ready_marker); + }; + try + { + while (true) + { + if (!impl_->completions.empty() && + impl_->completions.front().modeled_completion_ns <= current_time_ns()) + { + auto completion = impl_->completions.front(); + impl_->completions.pop_front(); + --impl_->pending_requests; + cancel_markers(); + return completion; + } + if (impl_->horizon_reached) + { + cancel_markers(); + return std::nullopt; + } + if (ready_marker == nullptr && !impl_->completions.empty() && + impl_->completions.front().modeled_completion_ns < deadline_ns) + { + ready_marker = Simulator->Register_sim_event( + impl_->completions.front().modeled_completion_ns, + impl_->clock_injector.get(), nullptr, 2); + } + if (!Simulator->Run_next_event()) + throw std::runtime_error("MQSim stopped before external clock horizon"); + } + } + catch (...) + { + cancel_markers(); + throw; + } + } + + std::size_t MqsimOnlineEngine::pending() const noexcept + { + return impl_->pending_requests; + } + + std::size_t MqsimOnlineEngine::pending_maintenance() const noexcept + { + return impl_->pending_maintenance; + } + + std::uint64_t MqsimOnlineEngine::current_time_ns() const noexcept + { + return Simulator->Has_started() + ? static_cast(Simulator->Time()) + : 0; + } + + void MqsimOnlineEngine::enable_observations() + { + if (impl_->has_submitted) + { + throw std::logic_error("enable MQSim observations before submission"); + } + impl_->observations_enabled = true; + } + + std::vector MqsimOnlineEngine::take_observations() + { + std::vector result; + result.swap(impl_->observations); + return result; + } + + std::vector MqsimOnlineEngine::take_maintenance_events() + { + std::vector result; + result.swap(impl_->maintenance_events); + return result; + } + + std::vector MqsimOnlineEngine::take_maintenance_completions() + { + std::vector result; + result.swap(impl_->maintenance_completions); + return result; + } + + std::vector run_mqsim_trace( + const Profile &profile, const std::filesystem::path &trace_path) + { + std::ifstream trace(trace_path); + if (!trace) + { + throw std::runtime_error("failed to open MQSim trace: " + + trace_path.string()); + } + + MqsimOnlineEngine engine(profile); + std::string line; + std::uint64_t request_id = 1; + std::uint64_t previous_arrival = 0; + std::size_t line_number = 0; + while (std::getline(trace, line)) + { + ++line_number; + if (line.empty()) + { + continue; + } + + std::uint64_t arrival_ns = 0; + std::uint64_t device = 0; + std::uint64_t start_sector = 0; + std::uint64_t sector_count = 0; + std::uint32_t operation = 0; + std::string trailing; + std::istringstream fields(line); + if (!(fields >> arrival_ns >> device >> start_sector >> sector_count >> + operation) || + (fields >> trailing)) + { + throw std::invalid_argument("invalid MQSim trace line " + + std::to_string(line_number)); + } + if (device != 0) + { + throw std::invalid_argument( + "MQSim HBF trace device must be zero at line " + + std::to_string(line_number)); + } + if (request_id > 1 && arrival_ns < previous_arrival) + { + throw std::invalid_argument( + "MQSim trace arrivals must be monotonic"); + } + if (start_sector > + std::numeric_limits::max() / kSectorBytes || + sector_count > + std::numeric_limits::max() / kSectorBytes) + { + throw std::overflow_error( + "MQSim trace address or size overflows"); + } + if (operation > 1) + { + throw std::invalid_argument( + "MQSim trace operation must be 0 or 1 at line " + + std::to_string(line_number)); + } + + engine.submit(HbfRequest{ + .request_id = request_id, + .sequence = request_id, + .arrival_ns = arrival_ns, + .logical_address = start_sector * kSectorBytes, + .deadline_ns = 0, + .bytes = static_cast(sector_count * kSectorBytes), + .range_id = 1, + .stream_id = 0, + .operation = operation == 0 + ? static_cast( + RequestOperation::Write) + : static_cast( + RequestOperation::Read), + .page_generation = 1, + .flags = 0, + }); + previous_arrival = arrival_ns; + ++request_id; + } + + std::vector completions; + completions.reserve(request_id - 1); + while (engine.pending() != 0) + { + auto completion = engine.run_next_completion(); + if (!completion.has_value()) + { + throw std::runtime_error( + "MQSim trace ended with pending requests"); + } + completions.push_back(*completion); + } + return completions; + } + +} // namespace hbfsim diff --git a/experiments/eq3_maintenance/backend/tests/maintenance_backend_tests.cpp b/experiments/eq3_maintenance/backend/tests/maintenance_backend_tests.cpp new file mode 100644 index 0000000..3188912 --- /dev/null +++ b/experiments/eq3_maintenance/backend/tests/maintenance_backend_tests.cpp @@ -0,0 +1,189 @@ +#include +#include +#include +#include +#include +#include + +namespace { +void require(bool value, const char* message) { if (!value) throw std::runtime_error(message); } +hbfsim::HbfRequest make_read(std::uint64_t id, std::uint64_t page, std::uint32_t bytes) { + return {.request_id=id,.sequence=id,.arrival_ns=0,.logical_address=page*bytes, + .bytes=bytes,.operation=static_cast(hbfsim::RequestOperation::Read)}; +} +hbfsim::HbfRequest make_write(std::uint64_t id,std::uint64_t page,std::uint32_t bytes, + std::uint64_t arrival) { + return {.request_id=id,.sequence=id,.arrival_ns=arrival,.logical_address=page*bytes, + .bytes=bytes,.operation=static_cast(hbfsim::RequestOperation::Write)}; +} +void submit_read(hbfsim::MqsimOnlineEngine& engine,std::uint64_t id, + std::uint64_t page,std::uint32_t bytes) { + auto request=make_read(id,page,bytes);request.arrival_ns=engine.current_time_ns(); + engine.submit(request); +} +hbfsim::MqsimMaintenanceCompletion maintain(hbfsim::MqsimOnlineEngine& engine, + std::uint64_t id, std::uint64_t page, hbfsim::MqsimMaintenanceFailurePoint failure, + bool reclaim=false) { + engine.submit_maintenance({.request_id=id,.parent_id=9000+id, + .due_ns=engine.current_time_ns(),.deadline_ns=engine.current_time_ns()+1000000000ULL, + .logical_page=page,.channel=0,.chip=0,.die=0,.plane=0,.plane_is_exact=true, + .reclaim_invalid_source_block=reclaim,.failure_point=failure}); + std::vector completions; + while(engine.pending_maintenance()) { + (void)engine.run_next_completion_until(engine.current_time_ns()+1000000000ULL); + auto batch=engine.take_maintenance_completions(); + completions.insert(completions.end(),batch.begin(),batch.end()); + } + require(completions.size()==1,"maintenance did not complete exactly once"); + return completions.front(); +} +} +int main(int argc,char** argv) { + require(argc==2,"profile path required"); + auto profile=hbfsim::load_profile(argv[1]); + profile.capacity_bytes=1ULL<<30;profile.channels=1;profile.dies_per_channel=1; + profile.planes_per_die=1;profile.pages_per_block=1;profile.queue_depth=4; + profile.aggregate_bandwidth_bytes_per_s=1ULL<<40;profile.hbm_cache_bytes=0; + hbfsim::validate_profile(profile); + hbfsim::HbfCompletion baseline{}; + { hbfsim::MqsimOnlineEngine engine(profile);engine.submit(make_read(1,0,profile.page_bytes)); + baseline=*engine.run_next_completion(); } + { hbfsim::MqsimOnlineEngine engine(profile);engine.submit(make_read(1,0,profile.page_bytes)); + auto value=*engine.run_next_completion(); + require(value.modeled_completion_ns==baseline.modeled_completion_ns,"disabled maintenance changed demand timing"); + // Move the one-page data frontier away from page 0's source block so its + // now-invalid source can be reclaimed without erasing an active frontier. + submit_read(engine,3,1,profile.page_bytes); + require(engine.run_next_completion().has_value(),"second mapped page setup failed"); + auto committed=maintain(engine,101,0,hbfsim::MqsimMaintenanceFailurePoint::None,true); + auto events=engine.take_maintenance_events(); + require(committed.status==hbfsim::MqsimMaintenanceStatus::Committed&& + committed.mapping_committed&&committed.source_retired&&committed.erase_completed, + "one-page maintenance did not read/program/commit/reclaim"); + require(committed.source_version!=committed.committed_version&& + committed.transaction_ids.size()==3,"committed version/command identity facts incomplete"); + require(!events.empty()&&events.front().state==hbfsim::MqsimMaintenanceState::Due&& + events.back().state==hbfsim::MqsimMaintenanceState::Done,"maintenance lifecycle facts incomplete"); + submit_read(engine,2,0,profile.page_bytes); + require(engine.run_next_completion().has_value(),"relocated LPA is unreadable"); + auto failed=maintain(engine,102,1,hbfsim::MqsimMaintenanceFailurePoint::Program); + require(failed.status==hbfsim::MqsimMaintenanceStatus::FailedProgram&& + !failed.mapping_committed&&!failed.source_retired,"failed program changed mapping"); + submit_read(engine,4,1,profile.page_bytes);require(engine.run_next_completion().has_value(),"source lost after failed program"); + + submit_read(engine,5,2,profile.page_bytes);require(engine.run_next_completion().has_value(),"read-failure setup failed"); + auto read_failed=maintain(engine,103,2,hbfsim::MqsimMaintenanceFailurePoint::Read); + require(read_failed.status==hbfsim::MqsimMaintenanceStatus::FailedRead&& + !read_failed.mapping_committed&&read_failed.transaction_ids.size()==1, + "failed maintenance read was not retained as a real command"); + submit_read(engine,6,2,profile.page_bytes);require(engine.run_next_completion().has_value(),"source lost after failed read"); + + submit_read(engine,7,3,profile.page_bytes);require(engine.run_next_completion().has_value(),"stale-CAS setup failed"); + auto stale=maintain(engine,104,3,hbfsim::MqsimMaintenanceFailurePoint::StaleCommit); + require(stale.status==hbfsim::MqsimMaintenanceStatus::FailedStaleVersion&& + !stale.mapping_committed&&!stale.source_retired, + "stale CAS changed the authoritative mapping"); + submit_read(engine,8,3,profile.page_bytes);require(engine.run_next_completion().has_value(),"source lost after stale CAS"); + + submit_read(engine,9,4,profile.page_bytes);require(engine.run_next_completion().has_value(),"erase-failure setup failed"); + submit_read(engine,10,5,profile.page_bytes);require(engine.run_next_completion().has_value(),"erase-failure frontier setup failed"); + auto erase_failed=maintain(engine,105,4,hbfsim::MqsimMaintenanceFailurePoint::Erase,true); + require(erase_failed.status==hbfsim::MqsimMaintenanceStatus::FailedAfterCommitNeedsReconcile&& + erase_failed.mapping_committed&&erase_failed.source_retired&&!erase_failed.erase_completed, + "erase failure did not preserve committed destination and reconciliation fact"); + submit_read(engine,11,4,profile.page_bytes);require(engine.run_next_completion().has_value(),"destination lost after erase failure"); + + // A real foreground write arrives while maintenance is reading the old + // source. It executes through MQSim and advances the mapping generation; + // the later maintenance program must be discarded by the generation CAS. + submit_read(engine,12,6,profile.page_bytes); + require(engine.run_next_completion().has_value(),"concurrent-write setup failed"); + const auto overlap_start=engine.current_time_ns(); + engine.submit_maintenance({.request_id=106,.parent_id=9106, + .due_ns=overlap_start,.deadline_ns=overlap_start+1000000000ULL, + .logical_page=6,.channel=0,.chip=0,.die=0,.plane=0,.plane_is_exact=true}); + engine.submit(make_write(13,6,profile.page_bytes,overlap_start+1)); + std::vector foreground; + while(engine.pending_maintenance()) { + auto value=engine.run_next_completion_until(engine.current_time_ns()+1000000000ULL); + if(value)foreground.push_back(*value); + } + while(engine.pending()) { + auto value=engine.run_next_completion();if(value)foreground.push_back(*value); + } + auto overlap=engine.take_maintenance_completions(); + require(overlap.size()==1&& + overlap.front().status==hbfsim::MqsimMaintenanceStatus::FailedStaleVersion&& + !overlap.front().mapping_committed&&foreground.size()==1&& + foreground.front().request_id==13, + "real concurrent foreground write did not win generation CAS exactly once"); + submit_read(engine,14,6,profile.page_bytes); + require(engine.run_next_completion().has_value(),"new foreground mapping lost after stale maintenance"); + auto after_write=maintain(engine,107,6,hbfsim::MqsimMaintenanceFailurePoint::None); + require(after_write.status==hbfsim::MqsimMaintenanceStatus::Committed&& + after_write.source_version==overlap.front().source_version+1&& + after_write.committed_version==after_write.source_version+1, + "mapping generation was not monotonic across foreground write and maintenance commit"); + } + + // The campaign fixture uses one channel per HBF stack and all 16 declared + // dies. Exercise die 15 with the finite 1 GiB engineering workspace. + { auto p16=profile;p16.capacity_bytes=1ULL<<30;p16.channels=1; + p16.dies_per_channel=16;p16.planes_per_die=1;p16.pages_per_block=256; + hbfsim::validate_profile(p16);hbfsim::MqsimOnlineEngine engine(p16); + submit_read(engine,201,15,p16.page_bytes);require(engine.run_next_completion().has_value(),"16-die setup failed"); + engine.submit_maintenance({.request_id=202,.parent_id=9202, + .due_ns=engine.current_time_ns(),.deadline_ns=engine.current_time_ns()+1000000000ULL, + .logical_page=15,.channel=0,.chip=0,.die=15,.plane=0,.plane_is_exact=true}); + while(engine.pending_maintenance())(void)engine.run_next_completion_until(engine.current_time_ns()+1000000000ULL); + auto completion=engine.take_maintenance_completions(); + require(completion.size()==1&&completion.front().status==hbfsim::MqsimMaintenanceStatus::Committed, + "16-die finite-workspace maintenance failed"); + } + + // A thermal guard may hold a logically due job while the external clock + // continues. Preserve its logical due/deadline, inject at the current + // clock, and let the native unit make the existing deadline decision. + { hbfsim::MqsimOnlineEngine engine(profile); + submit_read(engine,301,20,profile.page_bytes); + require(engine.run_next_completion().has_value(),"late-maintenance setup failed"); + const auto due=engine.current_time_ns(); + const auto future_deadline=due+1000000000ULL; + (void)engine.run_next_completion_until(due+1000); + const auto submit_time=engine.current_time_ns(); + engine.submit_maintenance({.request_id=302,.parent_id=9302, + .due_ns=due,.deadline_ns=future_deadline,.logical_page=20, + .channel=0,.chip=0,.die=0,.plane=0,.plane_is_exact=true}); + while(engine.pending_maintenance()) + (void)engine.run_next_completion_until(engine.current_time_ns()+1000000000ULL); + auto late=engine.take_maintenance_completions(); + auto late_events=engine.take_maintenance_events(); + require(late.size()==1&&late.front().status==hbfsim::MqsimMaintenanceStatus::Committed&& + late.front().enqueue_ns>=submit_time&&late.front().transaction_ids.size()==2, + "late but unexpired maintenance did not execute exactly once"); + require(!late_events.empty()&&late_events.front().time_ns>=submit_time, + "late maintenance registered an event in the past"); + + submit_read(engine,303,21,profile.page_bytes); + require(engine.run_next_completion().has_value(),"expired-maintenance setup failed"); + const auto expired_due=engine.current_time_ns(); + const auto expired_deadline=expired_due+100; + (void)engine.run_next_completion_until(expired_deadline+100); + engine.submit_maintenance({.request_id=304,.parent_id=9304, + .due_ns=expired_due,.deadline_ns=expired_deadline,.logical_page=21, + .channel=0,.chip=0,.die=0,.plane=0,.plane_is_exact=true}); + while(engine.pending_maintenance()) + (void)engine.run_next_completion_until(engine.current_time_ns()+1000000000ULL); + auto expired=engine.take_maintenance_completions(); + auto expired_events=engine.take_maintenance_events(); + require(expired.size()==1&& + expired.front().status==hbfsim::MqsimMaintenanceStatus::RejectedInvalidTarget&& + expired.front().transaction_ids.empty()&&!expired.front().mapping_committed, + "late expired maintenance did not reject once without media activity"); + require(expired_events.size()==3&& + expired_events.front().state==hbfsim::MqsimMaintenanceState::Due&& + expired_events.back().state==hbfsim::MqsimMaintenanceState::Failed, + "late expired maintenance lifecycle was not unique and terminal"); + } + std::cout<<"PASS METADATA_VERSION_VALIDITY shared_TSU_PHY maintenance lifecycle\n"; +} diff --git a/experiments/eq3_maintenance/backend/tests/maintenance_service_test.py b/experiments/eq3_maintenance/backend/tests/maintenance_service_test.py new file mode 100644 index 0000000..77eef06 --- /dev/null +++ b/experiments/eq3_maintenance/backend/tests/maintenance_service_test.py @@ -0,0 +1,102 @@ +#!/usr/bin/env python3 +"""Fixed process-level test of the isolated maintenance JSON protocol.""" +from __future__ import annotations + +import json +from pathlib import Path +import sys + + +def main(): + if len(sys.argv) != 4: + raise SystemExit("usage: maintenance_service_test.py MAIN_ROOT BINARY ARTIFACT_DIR") + root, binary, output = map(Path, sys.argv[1:]) + sys.path.insert(0, str(root)) + from experiments.eq3_maintenance.campaign_inputs import configuration + from experiments.eq3_maintenance.backend.client.maintenance_service import MaintenanceMqsimService + + output.mkdir(parents=True, exist_ok=False) + config = configuration("mixed_direct") + profile = output / "profile.json" + stack_map = output / "stack-map.json" + profile.write_text(json.dumps(config["profile"], indent=2) + "\n") + stack_map.write_text(json.dumps(config["stack_map"], indent=2) + "\n") + service_dir = output / "service" + with MaintenanceMqsimService(binary, profile, service_dir, timeout=30, + artifact_root=root.parents[2], stack_map=stack_map, + native_observations=True) as service: + service.submit(dict(request_id=1, issue_ns=0, bytes=16384, + operation="read", stack="hbf0", route="direct", stack_local_page=15)) + while 1 not in service.completions: + service.until(service.now + 1_000_000) + accepted = service.maintain(dict(request_id=101, parent_id=7001, + stack="hbf0", stack_local_page=15, due_ns=service.now, + deadline_ns=service.now + 1_000_000_000, + reclaim_source_block=False, trigger_reason="FIXED_RETENTION_DUE")) + assert accepted["target"] == {"channel": 0, "chip": 0, "die": 15, "plane": 0} + while 101 not in service.maintenance_completions: + service.until(service.now + 1_000_000) + completion = service.maintenance_completions[101] + assert completion["status"] == "COMMITTED" + assert completion["mapping_committed"] and completion["source_retired"] + assert completion["age_reset_ns"] == completion["end_ns"] + assert completion["source_version"] != completion["committed_version"] + assert completion["trigger_reason"] == "FIXED_RETENTION_DUE" + assert completion["coverage_pages"] == 1 and completion["deadline_met"] + phases = [e["state"] for e in service.maintenance_events + if e["request_id"] == 101] + assert phases == ["DUE", "QUEUED", "READ", "PROGRAM_DEST", + "COMMIT", "RETIRE_OLD", "DONE"] + native = [t for event in service.native_observations + for t in event["transactions"] + if t["maintenance_request_id"] == 101] + assert native and all(t["maintenance_parent_id"] == 7001 for t in native) + assert {t["die"] for t in native} == {15} + + service.submit(dict(request_id=2, issue_ns=service.now, bytes=16384, + operation="read", stack="hbf0", route="direct", stack_local_page=16)) + while 2 not in service.completions: + service.until(service.now + 1_000_000) + late_due = service.now + service.until(late_due + 1_000) + service.maintain(dict(request_id=102, parent_id=7002, + stack="hbf0", stack_local_page=16, due_ns=late_due, + deadline_ns=late_due + 1_000_000_000, + reclaim_source_block=False, trigger_reason="THERMAL_GUARD_RELEASE")) + while 102 not in service.maintenance_completions: + service.until(service.now + 1_000_000) + late = service.maintenance_completions[102] + assert late["status"] == "COMMITTED" and late["enqueue_ns"] >= late_due + 1_000 + assert late["deadline_met"] and len(late["transaction_ids"]) == 2 + + service.submit(dict(request_id=3, issue_ns=service.now, bytes=16384, + operation="read", stack="hbf0", route="direct", stack_local_page=17)) + while 3 not in service.completions: + service.until(service.now + 1_000_000) + expired_due = service.now + expired_deadline = expired_due + 100 + service.until(expired_deadline + 100) + service.maintain(dict(request_id=103, parent_id=7003, + stack="hbf0", stack_local_page=17, due_ns=expired_due, + deadline_ns=expired_deadline, reclaim_source_block=False, + trigger_reason="THERMAL_GUARD_RELEASE_AFTER_DEADLINE")) + while 103 not in service.maintenance_completions: + service.until(service.now + 1_000_000) + expired = service.maintenance_completions[103] + assert expired["status"] == "REJECTED_INVALID_TARGET" + assert not expired["mapping_committed"] and not expired["transaction_ids"] + assert not expired["deadline_met"] + expired_phases = [e["state"] for e in service.maintenance_events + if e["request_id"] == 103] + assert expired_phases == ["DUE", "QUEUED", "FAILED"] + expired_native = [t for event in service.native_observations + for t in event["transactions"] + if t["maintenance_request_id"] == 103] + assert not expired_native + receipt = service.finish() + assert receipt["maintenance_issued"] == receipt["maintenance_completed"] == 3 + print("PASS isolated maintain JSON horizon/native-ID protocol") + + +if __name__ == "__main__": + main() diff --git a/experiments/eq3_maintenance/backend/tests/test_ab_process.py b/experiments/eq3_maintenance/backend/tests/test_ab_process.py new file mode 100644 index 0000000..16f380b --- /dev/null +++ b/experiments/eq3_maintenance/backend/tests/test_ab_process.py @@ -0,0 +1,201 @@ +#!/usr/bin/env python3 +"""Strict process-level default-vs-isolated maintenance-off equivalence check.""" +from __future__ import annotations + +import argparse +import hashlib +import json +from pathlib import Path +import random +import subprocess +import sys + + +MAINTENANCE_ONLY_KEYS = { + "maintenance_backend", "maintenance_data_semantics", + "maintenance_payload_validation", "maintenance_events", + "maintenance_completions", "maintenance_issued", "maintenance_completed", + "pending_maintenance", "maintenance_request_id", "maintenance_parent_id", +} +DEFAULT_ENGINE_PATHS = ( + "CMakeLists.txt", "benchmarks/replay/hbf_mqsim_service.cpp", + "include/hbfsim/mqsim_online.hpp", "src/mqsim_adapter/mqsim_online.cpp", +) + + +def sha256(path): + digest = hashlib.sha256() + with path.open("rb") as source: + for chunk in iter(lambda: source.read(1024 * 1024), b""): + digest.update(chunk) + return digest.hexdigest() + + +def without_maintenance(value): + if isinstance(value, dict): + return {key: without_maintenance(item) for key, item in value.items() + if key not in MAINTENANCE_ONLY_KEYS and not key.startswith("maintenance_")} + if isinstance(value, list): + return [without_maintenance(item) for item in value] + return value + + +def workload(page_bytes, stack_ids, seed): + rng = random.Random(seed) + rows = [] + order = [(stack, local) for local in range(3) for stack in stack_ids] + rng.shuffle(order) + for request_id, (stack, local) in enumerate(order, 1): + rows.append({"request_id": request_id, + "issue_ns": (request_id-1)//len(stack_ids) * 1_000, + "bytes": page_bytes, "operation": "read", + "stack": stack, "route": "direct", + "stack_local_page": local}) + return sorted(rows, key=lambda row: (row["issue_ns"], row["request_id"])) + + +def run_service(client_class, binary, profile, stack_map, directory, artifact_root, requests): + with client_class(binary, profile, directory, timeout=30, artifact_root=artifact_root, + stack_map=stack_map, native_observations=True) as service: + header = without_maintenance(service.header) + for request in requests: + target = request["issue_ns"] + while service.now < target: + service.until(target) + if service.now != target: + raise AssertionError("service did not stop at the fixed issue horizon") + service.submit(request) + while len(service.completions) != len(requests): + service.until(service.now + 10_000_000) + receipt = service.finish() + return { + "header": header, + "requests": {str(key): value for key, value in sorted(service.requests.items())}, + "completions": {str(key): value for key, value in sorted(service.completions.items())}, + "events": without_maintenance(service.observations), + "native_command_events": without_maintenance(service.native_observations), + "receipt": without_maintenance(receipt), + } + + +def link_provenance(root, default_binary, isolated_binary): + default_link = default_binary.parent / "CMakeFiles/hbf_mqsim_service.dir/link.txt" + isolated_ninja = isolated_binary.parents[1] / "build.ninja" + isolated_make = isolated_binary.parents[1] / "CMakeFiles/hbf_mqsim_eq3_maint.dir/link.txt" + if not default_link.is_file() or not (isolated_ninja.is_file() or isolated_make.is_file()): + raise ValueError("link provenance files are missing") + default_command = default_link.read_text().strip() + if default_command.split().count("libmqsim_hbf.a") != 1: + raise AssertionError("default service must link exactly one default MQSim engine archive") + if isolated_ninja.is_file(): + isolated_evidence = isolated_ninja + lines = isolated_ninja.read_text().splitlines() + marker = "build bin/hbf_mqsim_eq3_maint:" + index = next((i for i, line in enumerate(lines) if line.startswith(marker)), None) + if index is None: + raise AssertionError("isolated service target missing from build graph") + link_line = next((line.strip() for line in lines[index:index+20] + if line.strip().startswith("LINK_LIBRARIES =")), None) + if link_line is None: + raise AssertionError("isolated service link libraries are missing") + libraries = link_line.split("=", 1)[1].split() + else: + isolated_evidence = isolated_make + libraries = [token for token in isolated_make.read_text().split() + if token.endswith((".a", ".so"))] + if libraries.count("lib/libmqsim_eq3_maint.a") != 1: + raise AssertionError("isolated service must link exactly one isolated MQSim engine archive") + if any("mqsim_eq3_maint" in token for token in default_command.split()): + raise AssertionError("default service unexpectedly links the maintenance engine") + if any(Path(token).name == "libmqsim_hbf.a" for token in libraries): + raise AssertionError("isolated service unexpectedly links the default engine") + default_diff = subprocess.run( + ["git", "-C", str(root), "diff", "--exit-code", "HEAD", "--", *DEFAULT_ENGINE_PATHS], + text=True, stdout=subprocess.PIPE, stderr=subprocess.PIPE, check=False) + if default_diff.returncode != 0: + raise AssertionError("default MQSim/service sources differ from HEAD") + status = subprocess.run( + ["git", "-C", str(root), "status", "--porcelain=v1"], text=True, + stdout=subprocess.PIPE, stderr=subprocess.PIPE, check=True).stdout.splitlines() + return { + "default_binary": str(default_binary), "default_binary_sha256": sha256(default_binary), + "isolated_binary": str(isolated_binary), "isolated_binary_sha256": sha256(isolated_binary), + "default_link_command": default_command, + "default_link_sha256": sha256(default_link), + "isolated_link_libraries": libraries, + "isolated_build_ninja_sha256": sha256(isolated_ninja) if isolated_ninja.is_file() else None, + "isolated_link_evidence": str(isolated_evidence), + "isolated_link_evidence_sha256": sha256(isolated_evidence), + "single_engine_archive_per_process": True, + "default_engine_source_paths": list(DEFAULT_ENGINE_PATHS), + "default_engine_sources_match_head": True, + "git_status_porcelain": status, + "ldd": { + "default": subprocess.run(["ldd", str(default_binary)], text=True, + stdout=subprocess.PIPE, stderr=subprocess.STDOUT, + check=True).stdout.splitlines(), + "isolated": subprocess.run(["ldd", str(isolated_binary)], text=True, + stdout=subprocess.PIPE, stderr=subprocess.STDOUT, + check=True).stdout.splitlines(), + }, + } + + +def main(): + parser = argparse.ArgumentParser() + parser.add_argument("--root", type=Path, required=True) + parser.add_argument("--default-binary", type=Path, required=True) + parser.add_argument("--isolated-binary", type=Path, required=True) + parser.add_argument("--profile", type=Path, required=True) + parser.add_argument("--stack-map", type=Path, required=True) + parser.add_argument("--output", type=Path, required=True) + parser.add_argument("--seed", type=int, default=20260920) + args = parser.parse_args() + root = args.root.resolve(strict=True) + default_binary = args.default_binary.resolve(strict=True) + isolated_binary = args.isolated_binary.resolve(strict=True) + profile = args.profile.resolve(strict=True) + stack_map = args.stack_map.resolve(strict=True) + output = args.output.resolve() + output.mkdir(parents=True, exist_ok=False) + sys.path.insert(0, str(root / "scripts/eval")) + from mqsim_service import MqsimService + + profile_value = json.loads(profile.read_text()) + map_value = json.loads(stack_map.read_text()) + stack_ids = [row["id"] for row in map_value["stacks"]] + requests = workload(profile_value["page_bytes"], stack_ids, args.seed) + provenance = link_provenance(root, default_binary, isolated_binary) + a = run_service(MqsimService, default_binary, profile, stack_map, + output / "default-service", root.parents[2], requests) + b = run_service(MqsimService, isolated_binary, profile, stack_map, + output / "isolated-service", root.parents[2], requests) + (output / "default-result.json").write_text( + json.dumps(a, indent=2, sort_keys=True, allow_nan=False) + "\n") + (output / "isolated-result.json").write_text( + json.dumps(b, indent=2, sort_keys=True, allow_nan=False) + "\n") + if a != b: + keys = [key for key in a if a[key] != b[key]] + failure = {"status": "FAIL", "differing_sections": keys, + "comparison": "STRICT_EXCEPT_EXPLICIT_MAINTENANCE_ONLY_FIELDS"} + (output / "summary.json").write_text(json.dumps(failure, indent=2) + "\n") + raise AssertionError(f"default/isolated foreground mismatch: {keys}") + result = { + "schema_version": "eq3-maintenance-process-ab-v1", "status": "PASS", + "comparison": "STRICT_EXCEPT_EXPLICIT_MAINTENANCE_ONLY_FIELDS", + "time_tolerance_ns": 0, "workload_seed": args.seed, + "request_count": len(requests), + "compared": ["request_and_raw_and_reported_completion", "native_phase_order", + "native_physical_address", "existing_service_statistics"], + "ignored_fields": sorted(MAINTENANCE_ONLY_KEYS), + "profile": str(profile), "profile_sha256": sha256(profile), + "stack_map": str(stack_map), "stack_map_sha256": sha256(stack_map), + "provenance": provenance, + } + (output / "summary.json").write_text( + json.dumps(result, indent=2, sort_keys=True, allow_nan=False) + "\n") + print(json.dumps(result, sort_keys=True)) + + +if __name__ == "__main__": + main() diff --git a/experiments/eq3_maintenance/build_ideal_maintenance_bundle.py b/experiments/eq3_maintenance/build_ideal_maintenance_bundle.py new file mode 100644 index 0000000..7ccb4d0 --- /dev/null +++ b/experiments/eq3_maintenance/build_ideal_maintenance_bundle.py @@ -0,0 +1,211 @@ +#!/usr/bin/env python3 +"""Build a fixed ideal-resource maintenance replay bundle from one completed point. + +The bundle preserves observed maintenance intent, native phases, and terminal +facts. It does not convert them into new backend observations. +""" +from __future__ import annotations + +import argparse +import copy +import hashlib +import json +from pathlib import Path + + +SCHEMA = "eq3-ideal-independent-maintenance-replay-v1" +EVIDENCE = "FIXED_COUNTERFACTUAL_REPLAY_NOT_ACTUAL_SHARED_MQSIM" + + +def sha256(path: Path) -> str: + digest = hashlib.sha256() + with path.open("rb") as source: + for chunk in iter(lambda: source.read(1024 * 1024), b""): + digest.update(chunk) + return digest.hexdigest() + + +def _maintenance_id(event): + identities = { + row.get("maintenance_request_id") + for row in event.get("transactions", ()) + if row.get("maintenance_request_id") not in (None, 0, "UNKNOWN") + } + if len(identities) != 1: + raise ValueError("native replay event must identify exactly one maintenance request") + return next(iter(identities)) + + +def _prefix_native(event, order): + value = copy.deepcopy(event) + original_command = value["command_id"] + value.update( + command_id=f"A2R-C{original_command}", + source_command_id=original_command, + replay_order=order, + evidence=EVIDENCE, + resource_model="IDEAL_INDEPENDENT_MAINTENANCE_REPLAY", + ) + for transaction in value["transactions"]: + original_transaction = transaction["transaction_id"] + maintenance_id = transaction.get("maintenance_request_id") + if maintenance_id in (None, 0, "UNKNOWN"): + raise ValueError("replayed native transaction lacks maintenance identity") + transaction.update( + transaction_id=f"A2R-T{original_transaction}", + source_transaction_id=original_transaction, + # Compatibility alias for the current experimental energy ledger. + # The actual service field remains maintenance_request_id. + maintenance_id=maintenance_id, + evidence=EVIDENCE, + ) + return value + + +def build_bundle(point: Path) -> dict: + point = point.resolve(strict=True) + names = ("manifest.json", "profile.json", "stack-map.json", + "requests-input.json", "maintenance-input.json", "result.json") + paths = {name: point / name for name in names} + if not all(path.is_file() for path in paths.values()): + raise ValueError("source point lacks required frozen artifacts") + manifest = json.loads(paths["manifest.json"].read_text()) + for name in names[1:5]: + expected = manifest.get("input_sha256", {}).get(name) + if expected is None or sha256(paths[name]) != expected: + raise ValueError(f"source point input identity mismatch: {name}") + profile = json.loads(paths["profile.json"].read_text()) + stack_map = json.loads(paths["stack-map.json"].read_text()) + foreground = json.loads(paths["requests-input.json"].read_text()) + intents = json.loads(paths["maintenance-input.json"].read_text()) + result = json.loads(paths["result.json"].read_text()) + if not intents or any(row.get("operation", "read") != "read" + for row in foreground if row.get("stack", "").startswith("hbf")): + raise ValueError("fixed A2 source must have maintenance and read-only HBF foreground") + intent_by_id = {row["request_id"]: row for row in intents} + if len(intent_by_id) != len(intents): + raise ValueError("source maintenance intent IDs are not unique") + + inflight = {} + backend_events = [] + for order, row in enumerate(result["timeline"]["maintenance"]): + phase = row.get("phase") + if phase == "INFLIGHT": + inflight[row["request_id"]] = { + "submit_ns": row["submit_ns"], "placement": row.get("placement"), + "target": row.get("target"), + } + elif phase == "BACKEND_EVENT": + value = copy.deepcopy(row) + value.pop("phase", None) + value.update(replay_order=order, evidence=EVIDENCE, + resource_model="IDEAL_INDEPENDENT_MAINTENANCE_REPLAY") + if value.get("transaction_id") is not None: + original = value["transaction_id"] + value["source_transaction_id"] = original + value["transaction_id"] = f"A2R-T{original}" + backend_events.append(value) + + native = [] + for order, event in enumerate(result["timeline"]["native"]): + try: + identity = _maintenance_id(event) + except ValueError: + continue + if identity not in intent_by_id: + raise ValueError("native maintenance fact has unknown intent") + native.append(_prefix_native(event, order)) + + completions = [] + for row in result["maintenance"]: + request_id = row["request_id"] + if request_id not in intent_by_id or request_id not in inflight: + raise ValueError("maintenance result is absent from intent/submission facts") + completion = copy.deepcopy(row.get("completion")) + if not isinstance(completion, dict): + raise ValueError("maintenance result lacks backend completion") + baseline_source = completion.pop("source_version", None) + baseline_committed = completion.pop("committed_version", None) + baseline_transactions = completion.get("transaction_ids", []) + completion.update( + source_version="UNKNOWN_REPLAY", + committed_version="UNKNOWN_REPLAY", + baseline_source_version=baseline_source, + baseline_committed_version=baseline_committed, + transaction_ids=[f"A2R-T{value}" for value in baseline_transactions], + baseline_transaction_ids=baseline_transactions, + evidence=EVIDENCE, + commit_semantics=("VIRTUAL_METADATA_AGE_LEDGER_ONLY; " + "CURRENT_MQSIM_MAPPING_NOT_MUTATED"), + ) + completions.append(completion) + completions.sort(key=lambda row: (row["end_ns"], row["request_id"])) + + expected = set(intent_by_id) + if (set(inflight) != expected + or {row["request_id"] for row in completions} != expected + or {row["request_id"] for row in backend_events} != expected + or {_maintenance_id(row) for row in native} != expected): + raise ValueError("fixed replay does not cover every maintenance intent") + states = {} + for row in backend_events: + states.setdefault(row["request_id"], []).append(row["state"]) + required_states = ["DUE", "QUEUED", "READ", "PROGRAM_DEST", "COMMIT", + "RETIRE_OLD", "DONE"] + if any(value != required_states for value in states.values()): + raise ValueError("source point lacks the complete maintenance state sequence") + + return { + "schema_version": SCHEMA, + "evidence": EVIDENCE, + "resource_model": "IDEAL_INDEPENDENT_MAINTENANCE_REPLAY", + "source_point": str(point), + "source_files_sha256": {name: sha256(path) for name, path in paths.items()}, + "input_contract": { + "profile_sha256": sha256(paths["profile.json"]), + "stack_map_sha256": sha256(paths["stack-map.json"]), + "foreground_requests_sha256": sha256(paths["requests-input.json"]), + "maintenance_intents_sha256": sha256(paths["maintenance-input.json"]), + "hbf_foreground": "READ_ONLY", + "concurrent_write": "REJECT", + "source_version": "UNKNOWN_REPLAY", + }, + "profile_identity": {key: profile.get(key) for key in + ("name", "channels", "dies_per_channel", "planes_per_die", + "page_bytes", "queue_depth")}, + "stack_map_identity": {key: stack_map.get(key) for key in + ("schema_version", "physical_kind", "route", + "address_layout", "plane_allocation_scheme")}, + "maintenance_intents": intents, + "submission_facts": inflight, + "backend_events": backend_events, + "native_events": native, + "completions": completions, + "counts": {"maintenance": len(intents), "backend_events": len(backend_events), + "native_events": len(native), "completions": len(completions)}, + "limits": [ + "counterfactual replay, not actual shared-MQSim maintenance", + "foreground MQSim mapping is not changed by replayed commit facts", + "valid only for the exact read-only foreground/profile/map/initial-state contract", + "fixed-intent direct-attribution supplement, not the endogenous-trigger A2 arm", + "baseline versions are provenance only; counterfactual source version is unknown", + ], + } + + +def main(): + parser = argparse.ArgumentParser() + parser.add_argument("--source-point", type=Path, required=True) + parser.add_argument("--output", type=Path, required=True) + args = parser.parse_args() + if args.output.exists(): + raise FileExistsError(f"refusing to overwrite {args.output}") + bundle = build_bundle(args.source_point) + args.output.parent.mkdir(parents=True, exist_ok=True) + args.output.write_text(json.dumps(bundle, indent=2, sort_keys=True, + allow_nan=False) + "\n") + print(json.dumps({"status": "GENERATED", **bundle["counts"]}, sort_keys=True)) + + +if __name__ == "__main__": + main() diff --git a/experiments/eq3_maintenance/campaign_inputs.py b/experiments/eq3_maintenance/campaign_inputs.py new file mode 100644 index 0000000..350307f --- /dev/null +++ b/experiments/eq3_maintenance/campaign_inputs.py @@ -0,0 +1,123 @@ +"""Explicit engineering inputs; never launches a run or changes default profiles.""" +from __future__ import annotations +import copy +import json +from pathlib import Path +import sys + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / 'tools')) +from eq3_basic_system import engineering_fixture + +MODES = ('mixed_direct', 'relay', 'dash', 'all_hbf_direct') + + +def configuration(mode, *, pages_per_stack=65536, geometry="legacy16k"): + """16 real MQSim dies/stack; explicit legacy or OCP-sized logical geometry.""" + if mode not in MODES: + raise ValueError('unsupported topology') + n = 8 if mode == 'all_hbf_direct' else 4 + if geometry not in ("legacy16k", "ocp4k16bank"): + raise ValueError("unknown NAND geometry profile") + page = 4096 if geometry == "ocp4k16bank" else 16384 + channel_per_stack = 16 if geometry == "ocp4k16bank" else 1 + die_per_channel = 1 if geometry == "ocp4k16bank" else 16 + planes = 16 if geometry == "ocp4k16bank" else 1 + if pages_per_stack % (16 * planes * 256) or pages_per_stack < 16 * planes * 256 * 8: + raise ValueError('working region must be block aligned and provide >=8 blocks/physical plane') + profile = dict(name=('EQ3_OCP4K_16BANK_FULL_CAPACITY' if geometry=='ocp4k16bank' and pages_per_stack*page==512*1024**3 else 'EQ3_EXPERIMENTAL_16DIE_FINITE_WORKING_REGION'), + capacity_bytes=n * pages_per_stack * page, page_bytes=page, + read_latency_ns=10000, program_latency_ns=100000, nand_technology='SLC', + channels=n*channel_per_stack, dies_per_channel=die_per_channel, planes_per_die=planes, + pages_per_block=256, channel_width_bits=8, + channel_transfer_rate_mtps=1600, queue_depth=256, + aggregate_bandwidth_bytes_per_s=512000000000, + hbm_cache_bytes=67108864, reference_sample_rate=0.01, + reference_warmup_requests=1024, time_scale=1, + timing_tolerance_ns=10000) + mapping = dict(schema_version=1, physical_kind='HBF', route='direct', + address_layout='GLOBAL_PAGE_STRIPE_V1', plane_allocation_scheme='CWDP', + page_bytes=page, channels=n*channel_per_stack, dies_per_channel=die_per_channel, + evidence=('OCP512GiB_LOGICAL_CAPACITY_WITH_ENGINEERING_MAPPING' if pages_per_stack*page==512*1024**3 else 'ENGINEERING_16_DIE_FINITE_WORKING_REGION'), + stacks=[dict(id=f'hbf{i}', declared_dies=16, channels=list(range(i*channel_per_stack,(i+1)*channel_per_stack))) for i in range(n)]) + basic = engineering_fixture(mode, page) + basic.pop('requests') + for item in basic['fabric']['hbf'].values(): + # DASH paper physical 18MiB / usable16MiB split2x8MiB; engineering reuse on other paths. + item['bank_capacity_bytes'] = 8 * 1024 * 1024 + for field in ('fill', 'direct_link', 'relay_link'): + if item[field] is not None: + item[field] = dict(latency_ns=10, bandwidth_bytes_per_s=1600000000000) + for item in basic['fabric']['hbm'].values(): + item['bank_capacity_bytes'] = 4 * 1024 * 1024 + item['gpu_link'] = dict(latency_ns=10, bandwidth_bytes_per_s=2048000000000) + if basic['hbm']: + for item in basic['hbm']['stacks']: + item['media_latency_ns'] = dict(read=50, write=50) + item['media_bandwidth_Bps'] = dict(read=2048000000000, write=2048000000000) + # Energy is accounted once in the experimental phase ledger below. + return dict(profile=profile, stack_map=mapping, fabric=basic['fabric'], hbm=basic['hbm'], + mode=mode, geometry_profile=geometry, + geometry_evidence=dict(nand_cell_mode='EXPLICIT_SLC_SERVICE_PROXY_NOT_HBF_PRODUCT_CELL_MODE',page_bytes='OCP070_SPECIFIED' if page==4096 else 'LEGACY_ENGINEERING', + host_channels_per_stack='OCP070_SPECIFIED16' if channel_per_stack==16 else 'ENGINEERING_AGGREGATION1', + dies_per_channel='SCENARIO_PROJECTION_16_DIES_OVER16_CHANNELS' if die_per_channel==1 else 'ENGINEERING_AGGREGATION16', + planes_per_die='BANK_TO_PLANE_1TO1_SCENARIO_ASSUMPTION' if planes==16 else 'ENGINEERING1', + pages_per_block='SCENARIO_ASSUMPTION256; OCP070_5.7_R3_PRODUCT_DEFINED', + product_stack_capacity_bytes=512*1024**3,product_stack_pages=512*1024**3//4096, + product_average_pages_per_die=512*1024**3//4096//16, + configured_stack_capacity_bytes=pages_per_stack*page, + configured_pages_per_die=pages_per_stack//16, + configured_blocks_per_plane=pages_per_stack//(16*planes*256), + physical_spare='BACKEND_FINITE_ALLOCATOR; NOT_TARGET_PRODUCT_SPECIFIED'), + capacity_scope='FULL_OCP512GiB_LOGICAL_CAPACITY' if pages_per_stack*page==512*1024**3 else 'FINITE_WORKING_REGION; full product capacity not instantiated', + evidence='CONDITIONAL_ENGINEERING_USE', + media_geometry=f'16 MQSim dies per HBF stack; {channel_per_stack} channels/stack, {die_per_channel} dies/channel, {planes} planes/die; existing CWDP', + hbm_geometry='12 physical thermal dies; parametric service die UNKNOWN', + fabric_parameter_scope='DASH buffer anchor; latency10ns assumption, BW target not calibrated', + external_gddr='UNAVAILABLE' if n == 8 else 'NOT_APPLICABLE') + + +def workload(mode, kind, *, total_hbf_rps, active_ns, burst_period_ns=20000000, + hbm_rps_per_stack=10, pages_per_stack=65536): + """Arrivals are independent of service and policy; Q4 equal TOTAL HBF demand.""" + if mode not in MODES or kind not in ('W1', 'W2'): + raise ValueError('unknown topology/workload') + if total_hbf_rps <= 0 or active_ns <= 0 or burst_period_ns <= 0: + raise ValueError('positive rates and durations required') + n = 8 if mode == 'all_hbf_direct' else 4 + requests, local = [], [0] * n + count = int(total_hbf_rps * active_ns // 1000000000) + for i in range(count): + nominal = i * 1000000000 // total_hbf_rps + # Fixed burst at each period, not a temperature-dependent synthetic power. + arrival = nominal // burst_period_ns * burst_period_ns + if kind == 'W1': + s = i % n + else: + # 1/2 of traffic to a rotating hot stack, remaining round-robin. + hot = (arrival // max(burst_period_ns, active_ns // n)) % n + s = hot if i % 2 == 0 else (i // 2) % n + index = local[s]; local[s] += 1 + route = 'relay' if mode == 'relay' or mode == 'dash' and index % 2 else 'direct' + requests.append(dict(request_id=f'hbf-{i}', stack=f'hbf{s}', stack_local_page=index % pages_per_stack, + route=route, bytes=16384, arrival_ns=arrival, operation='read')) + if n == 4 and hbm_rps_per_stack: + for s in range(4): + for i in range(int(hbm_rps_per_stack * active_ns // 1000000000)): + requests.append(dict(request_id=f'hbm-{s}-{i}', stack=f'hbm{s}', route='direct', + bytes=16384, arrival_ns=i*1000000000//hbm_rps_per_stack, operation='read')) + return sorted(requests, key=lambda r:(r['arrival_ns'],r['request_id'])) + + +def energy_profile(): + return dict(schema_version='eq3-stage-energy-engineering-v1', evidence='SCENARIO_ASSUMPTION', + nand_media_w={'0':0.05, '1':0.05, '2':0.05}, nand_data_out_w=0.01, nand_command_transfer_w=0.01, + hbm_array_j_per_byte=40e-12, fabric_endpoint_j_per_byte=2e-12, + gpu_external_w=200.0, standby_w=0.0, + standby_scope='NOT_MODELLED; zero increment is not measured zero device idle power', + media_scope='per active physical die, share among simultaneous planes, not request occupancy', + source_note='0.05W order-of-magnitude engineering coefficient, not HBF calibrated. Micron 2Gb M29B manufacturer datasheet typical15mA at3.3V is only an old-device plausibility anchor; no exact transfer of product current.', + sources=['https://www.micron.com/sales-support/design-tools/nand-system-power-calculator', + 'https://www.mouser.com/datasheet/2/671/2gb_nand_m29b-1879920.pdf'], + hbm_note='40pJ/B array engineering allocation plus separate2pJ/B endpoint increments, informed only in scale by registered H200 aggregate46pJ/B; not calibrated decomposition', + gpu_note='external prescribed compute load <= original200W envelope, no live GPU or inferred token causality') diff --git a/experiments/eq3_maintenance/closed_loop.py b/experiments/eq3_maintenance/closed_loop.py new file mode 100644 index 0000000..0e732c1 --- /dev/null +++ b/experiments/eq3_maintenance/closed_loop.py @@ -0,0 +1,648 @@ +"""Single-threaded experimental EQ3 coordinator. + +This composes existing media, fabric, energy, thermal, and policy objects. It +does not replace their arbitration or infer unavailable reliability facts. +""" +from __future__ import annotations + +import copy +import math +from dataclasses import asdict + +from read_rate_policy import (ByteTokenLedger, Decision, StackDecision, + StackWindowFacts, WindowFacts) + + +MAX_HORIZON_NS = (1 << 64) - 1 +REQUEST_METADATA_FIELDS = ( + "global_page", "global_byte_address", "logical_regions", "scan_index", "local_page", +) + + +class UnsupportedComposition(RuntimeError): + pass + + +def _integer(value, label, *, positive=False): + if isinstance(value, bool) or not isinstance(value, int): + raise ValueError(f"{label} must be an integer") + if value < (1 if positive else 0): + raise ValueError(f"{label} is outside its allowed range") + return value + + +def _percentile95(values): + if not values: + return None + ordered = sorted(values) + return ordered[max(0, math.ceil(0.95 * len(ordered)) - 1)] + + +class ClosedLoopCoordinator: + """Compose actual backend clients without owning their internal schedule.""" + + def __init__(self, mode, mqsim, hbm, fabric, thermal, energy, policy, *, + thermal_window_ns, experiment_end_ns, initial_stack_budget_bytes, + shared_endpoint_caps_bytes=None, resource_probe=None, + drain_submitted=True): + self.mode = mode + self.mqsim, self.hbm, self.fabric = mqsim, hbm, fabric + self.thermal, self.energy, self.policy = thermal, energy, policy + self.window_ns = _integer(thermal_window_ns, "thermal_window_ns", positive=True) + self.end_ns = _integer(experiment_end_ns, "experiment_end_ns", positive=True) + if self.end_ns % self.window_ns: + raise ValueError("experiment_end_ns must contain complete thermal windows") + self.initial_budgets = dict(initial_stack_budget_bytes) + if not self.initial_budgets or any(not isinstance(k, str) or not k or + _integer(v, "initial stack budget") < 0 + for k, v in self.initial_budgets.items()): + raise ValueError("initial stack budgets must cover configured stacks") + self.endpoint_caps = dict(shared_endpoint_caps_bytes or {}) + if any(_integer(v, "shared endpoint cap") < 0 for v in self.endpoint_caps.values()): + raise ValueError("shared endpoint caps must be nonnegative") + self.resource_probe = resource_probe + self.drain_submitted = bool(drain_submitted) + + fabric_facts = fabric.immutable_facts() + config = fabric_facts["config"] + self._hbf, self._hbm = set(config["hbf"]), set(config["hbm"]) + self._pairs = {stack: row["pair"] for stack, row in config["hbf"].items()} + configured = self._hbf | self._hbm + if set(self.initial_budgets) != configured: + raise ValueError("initial stack budget coverage differs from fabric stacks") + + self.now_ns = 0 + self._records, self._waiting, self._mq_pending = {}, [], {} + self._maintenance, self._maintenance_pending = {}, {} + self._next_backend_id = 1 + self._fabric_seen = self._fabric_completion_seen = 0 + self._native_seen = self._observation_seen = 0 + self._maintenance_event_seen = 0 + self._maintenance_completion_seen = set() + self._energy_row_seen = 0 + self._raw_media_ns = {} + self._thermal_start = 0 + self._next_window = self.window_ns + self._current_budgets = dict(self.initial_budgets) + self._last_guard_states = {stack: "normal" for stack in configured} + self._ledger = self._initial_ledger() + self._timeline = {name: [] for name in ( + "requests", "native", "fabric", "hbm", "thermal", "rates", + "control", "maintenance", "resources", "energy")} + + @staticmethod + def resource_probe_contract(): + """Describe the optional observed-only window probe consumed by policy.""" + return { + "call": "resource_probe(start_ns:int,end_ns:int)->mapping[stack_id,mapping]", + "optional_fields": ("backend_busy_fraction", "resource_busy", + "retry_count", "uecc_count"), + "unknown_rule": "omit or None; coordinator never substitutes zero/idle", + "interval": "half-open [start_ns,end_ns)", + } + + def _initial_ledger(self): + enabled = self.policy.profile.enabled + decision = Decision( + enabled=enabled, strategy=self.policy.profile.strategy, + applies_to_window_start_ns=0, + stack_decisions=tuple(StackDecision(stack, budget, "HOLD", "INITIAL", + ("EXPLICIT_INITIAL_BUDGET",)) + for stack, budget in sorted(self.initial_budgets.items())), + shared_endpoint_budget_bytes=dict(self.endpoint_caps)) + return ByteTokenLedger(decision) + + def _route_endpoints(self, row): + stack = row["stack"] + if row["route"] == "relay": + partner = self._pairs.get(stack) + if partner is None: + raise ValueError("relay request lacks a configured HBM partner") + return (f"{stack}->{partner}:relay-link", f"{partner}:gpu-link") + return (f"{stack}:gpu-link",) + + def _validate_request(self, raw, sequence): + required = {"request_id", "stack", "route", "bytes", "arrival_ns", "operation"} + if not isinstance(raw, dict) or not required.issubset(raw): + raise ValueError("request lacks required identity/path/extent/arrival") + request_id, stack = raw["request_id"], raw["stack"] + if not isinstance(request_id, str) or not request_id or request_id in self._records: + raise ValueError("request_id must be unique and nonempty") + if stack not in self.initial_budgets: + raise ValueError("request names an unconfigured stack") + kind = "HBF" if stack in self._hbf else "HBM" + route = raw["route"] + if kind == "HBF": + if raw["operation"] != "read" or "stack_local_page" not in raw: + raise ValueError("HBF requests require read and stack_local_page") + if self.mode == "relay" and route != "relay": + raise ValueError("relay topology requires relay HBF requests") + if self.mode in {"all_hbf_direct", "mixed_direct"} and route != "direct": + raise ValueError("direct topology requires direct HBF requests") + if self.mode == "dash" and route not in {"direct", "relay"}: + raise ValueError("DASH route must be direct or relay") + local_page = _integer(raw["stack_local_page"], "stack_local_page") + else: + if route != "direct" or raw["operation"] not in {"read", "write"}: + raise ValueError("HBM supports direct read/write only") + local_page = None + physical_bytes = _integer(raw["bytes"], "bytes", positive=True) + valid_bytes = _integer(raw.get("valid_weight_bytes", physical_bytes), + "valid_weight_bytes", positive=True) + if valid_bytes > physical_bytes: + raise ValueError("valid_weight_bytes must not exceed physical bytes") + row = {"request_id": request_id, "sequence": sequence, "kind": kind, + "stack": stack, "route": route, + "bytes": physical_bytes, "valid_weight_bytes": valid_bytes, + "arrival_ns": _integer(raw["arrival_ns"], "arrival_ns"), + "operation": raw["operation"], "stack_local_page": local_page, + "route_endpoints": self._route_endpoints(raw), "state": "NOT_ARRIVED", + "backend_submit_ns": None, "backend_completion_ns": None, + "backend_media_ns": None, "fabric_completion_ns": None, + "final_completion_ns": None, "gate_limited": False} + for field in REQUEST_METADATA_FIELDS: + if field in raw: + row[field] = copy.deepcopy(raw[field]) + self._records[request_id] = row + return row + + def _validate_maintenance(self, raw, sequence): + required = {"request_id", "stack", "stack_local_page", "due_ns", "initial_age_s"} + if not isinstance(raw, dict) or not required.issubset(raw): + raise ValueError("maintenance request lacks identity/stack/due/operation/initial_age_s") + mid = _integer(raw["request_id"], "maintenance request_id", positive=True) + if mid in self._maintenance: + raise ValueError("maintenance request_id must be unique") + if raw["stack"] not in self.initial_budgets: + raise ValueError("maintenance stack is unconfigured") + age = raw["initial_age_s"] + if isinstance(age, bool) or not isinstance(age, (int, float)) or not math.isfinite(age) or age < 0: + raise ValueError("initial_age_s must be finite and nonnegative") + value = copy.deepcopy(raw) + value.update(sequence=sequence, due_ns=_integer(raw["due_ns"], "maintenance due_ns"), + state="NOT_DUE", submit_ns=None, completion_ns=None) + self._maintenance[mid] = value + + def _ingest_backend_facts(self): + observations = self.mqsim.observations[self._observation_seen:] + self._observation_seen = len(self.mqsim.observations) + for event in observations: + if event.get("kind") == 2: + self._raw_media_ns[event["request_id"]] = event["time_ns"] + native = self.mqsim.native_observations[self._native_seen:] + self._native_seen = len(self.mqsim.native_observations) + for event in native: + fact = copy.deepcopy(event) + self._timeline["native"].append(fact) + self.energy.native(fact) + + def _accept_mq_completion(self, completion): + if completion is None: + return + backend_id = completion["request_id"] + if backend_id not in self._mq_pending: + raise RuntimeError("duplicate or unknown backend completion") + request = self._records[self._mq_pending.pop(backend_id)] + reported = completion["reported_complete"] + raw = self._raw_media_ns.get(backend_id) + if raw is None: + raise UnsupportedComposition("UNSUPPORTED_COMPOSITION: missing MQSim media completion fact") + if raw != reported: + raise UnsupportedComposition( + "UNSUPPORTED_COMPOSITION: raw media and reported completion differ") + request.update(backend_media_ns=raw, backend_completion_ns=reported, + state="FABRIC_PENDING") + self.fabric.mark_source_ready(request["request_id"], reported) + + def _accept_maintenance_facts(self): + events = getattr(self.mqsim, "maintenance_events", ()) + for event in events[self._maintenance_event_seen:]: + self._timeline["maintenance"].append({"phase": "BACKEND_EVENT", **copy.deepcopy(event)}) + self._maintenance_event_seen = len(events) + completions = getattr(self.mqsim, "maintenance_completions", {}) + for backend_id, completion in completions.items(): + if backend_id in self._maintenance_completion_seen: + continue + self._maintenance_completion_seen.add(backend_id) + if backend_id not in self._maintenance_pending: + raise RuntimeError("unknown maintenance completion") + mid = self._maintenance_pending.pop(backend_id) + row = self._maintenance[mid] + row["completion_ns"] = completion["end_ns"] + status = completion.get("status") + mapping_committed = completion.get("mapping_committed") is True + age_reset_ns = completion.get("age_reset_ns") + if mapping_committed and not isinstance(age_reset_ns, int): + raise UnsupportedComposition( + "mapping commit lacks an observed age_reset_ns") + if not mapping_committed and age_reset_ns is not None: + raise UnsupportedComposition( + "uncommitted maintenance reported an age reset") + normal_commit = status in {"COMMITTED", "COMMITTED_RECLAIM_DEFERRED"} + if normal_commit and not mapping_committed: + raise UnsupportedComposition("maintenance status and mapping commit fact disagree") + if mapping_committed: + row["state"] = "COMMITTED" if normal_commit else "COMMITTED_CLEANUP_FAILED" + else: + row["state"] = "FAILED" + row.update(backend_status=status, mapping_committed=mapping_committed, + age_reset_ns=age_reset_ns, + cleanup_failed=(status == "FAILED_AFTER_COMMIT_NEEDS_RECONCILE")) + row["completion"] = copy.deepcopy(completion) + self._timeline["maintenance"].append({"phase": row["state"], **copy.deepcopy(row)}) + + def _accept_hbm(self): + if self.hbm is None: + return + for completion in self.hbm.take_media_completions(): + request = self._records[completion["request_id"]] + if request["state"] != "BACKEND_PENDING": + raise RuntimeError("duplicate or unknown HBM completion") + request.update(backend_media_ns=completion["time_ns"], + backend_completion_ns=completion["time_ns"], + state="FABRIC_PENDING") + self.fabric.mark_source_ready(request["request_id"], completion["time_ns"]) + for fact in self.hbm.take_facts(): + value = copy.deepcopy(fact) + self._timeline["hbm"].append(value) + self.energy.hbm(value) + + def _accept_fabric(self): + if hasattr(self.fabric, "events_since"): + next_event_offset, new_events = self.fabric.events_since(self._fabric_seen) + else: + events = self.fabric.events() + next_event_offset, new_events = len(events), events[self._fabric_seen:] + for event in new_events: + request = self._records[event["request_id"]] + value = copy.deepcopy(event) + self._timeline["fabric"].append(value) + energy_request = copy.deepcopy(request) + if request["route"] == "relay": + energy_request["partner"] = self._pairs[request["stack"]] + self.energy.fabric(value, energy_request) + self._fabric_seen = next_event_offset + if hasattr(self.fabric, "completions_since"): + next_completion_offset, new_completions = self.fabric.completions_since( + self._fabric_completion_seen) + else: + completions = self.fabric.completions() + next_completion_offset = len(completions) + new_completions = completions[self._fabric_completion_seen:] + for completion in new_completions: + request = self._records[completion["request_id"]] + if request["state"] == "COMPLETE": + continue + if request["state"] != "FABRIC_PENDING": + raise RuntimeError("premature fabric completion") + request["fabric_completion_ns"] = completion["completion_ns"] + request["final_completion_ns"] = max(request["backend_completion_ns"], + completion["completion_ns"]) + request["backend_latency_ns"] = (request["backend_completion_ns"]- + request["backend_submit_ns"]) + request["fabric_latency_ns"] = (request["fabric_completion_ns"]- + request["backend_completion_ns"]) + request["end_to_end_latency_ns"] = (request["final_completion_ns"]- + request["arrival_ns"]) + request["state"] = "COMPLETE" + self._timeline["requests"].append({"phase": "FINAL_COMPLETE", + **copy.deepcopy(request)}) + self._fabric_completion_seen = next_completion_offset + + def _advance_components(self, target_ns): + completion = self.mqsim.until(target_ns) + self.now_ns = self.mqsim.now + if self.hbm is not None: + self.hbm.advance(self.now_ns) + self.fabric.advance(self.now_ns) + self._ingest_backend_facts() + self._accept_mq_completion(completion) + self._accept_maintenance_facts() + self._accept_hbm() + # Ready notifications above may create zero-independent future fabric events. + self.fabric.advance(self.now_ns) + self._accept_fabric() + + def _admit_request(self, request): + if self.now_ns >= self.end_ns or request["arrival_ns"] > self.now_ns: + return False + if request["state"] not in {"EXTERNAL_WAIT", "POLICY_WAIT", "SOURCE_RESERVED", "GATE_WAIT"}: + return False + # Admission reserves physical transfer/media extent. valid_weight_bytes + # is the delivered-work metric consumed by policy feedback below. + if not self._ledger.can_consume(request["stack"], request["route_endpoints"], request["bytes"]): + request["state"] = "POLICY_WAIT" if request["state"] == "EXTERNAL_WAIT" else request["state"] + request["gate_limited"] = True + return False + if request["state"] in {"EXTERNAL_WAIT", "POLICY_WAIT"}: + if not self.fabric.reserve_source(request["request_id"], request["stack"], request["route"], + request["bytes"], request["arrival_ns"]): + request["resource_busy"] = True + return False + request["state"] = "SOURCE_RESERVED" + if request["kind"] == "HBM": + self.hbm.arrival({"request_id": request["request_id"], "stack_id": request["stack"], + "op": request["operation"], "bytes": request["bytes"], + "arrival_ns": self.now_ns}) + self.hbm.submit(request["request_id"], self.now_ns) + submitted_ns = self.now_ns + else: + backend_id = request.get("backend_request_id") + if backend_id is None: + backend_id = self._next_backend_id + self._next_backend_id += 1 + request["backend_request_id"] = backend_id + decision = self.mqsim.try_submit({ + "request_id": backend_id, "issue_ns": self.now_ns, "stack": request["stack"], + "stack_local_page": request["stack_local_page"], "bytes": request["bytes"], + "operation": "read", "route": "direct"}) + request["gate_decision"] = copy.deepcopy(decision) + if not decision["submitted"]: + if decision.get("disposition") == 1 and isinstance(decision.get("target_ns"), int): + if decision["target_ns"] <= self.now_ns: + raise RuntimeError("backend gate returned nonfuture target") + request.update(state="GATE_WAIT", gate_target_ns=decision["target_ns"], + gate_limited=True) + return False + raise UnsupportedComposition( + f"UNSUPPORTED_COMPOSITION: backend rejected reserved request: {decision.get('reason','')}") + submitted_ns = decision["backend_arrival_ns"] + self._mq_pending[backend_id] = request["request_id"] + if not self._ledger.try_consume(request["stack"], request["route_endpoints"], request["bytes"]): + raise AssertionError("token budget changed between check and atomic commit") + request.update(state="BACKEND_PENDING", backend_submit_ns=submitted_ns, + external_wait_ns=submitted_ns-request["arrival_ns"]) + request.pop("gate_target_ns", None) + self._timeline["requests"].append({"phase": "ADMITTED", **copy.deepcopy(request)}) + return True + + def _admit(self): + progress = False + for request in sorted(self._waiting, key=lambda row: row["sequence"]): + if request["state"] == "NOT_ARRIVED" and request["arrival_ns"] <= self.now_ns: + request["state"] = "EXTERNAL_WAIT" + self._timeline["requests"].append({"phase": "ARRIVAL", **copy.deepcopy(request)}) + if request["state"] in {"EXTERNAL_WAIT", "POLICY_WAIT", "SOURCE_RESERVED", "GATE_WAIT"}: + if request.get("gate_target_ns", self.now_ns) <= self.now_ns: + progress = self._admit_request(request) or progress + return progress + + def _submit_due_maintenance(self): + for mid, row in sorted(self._maintenance.items(), key=lambda item: item[1]["sequence"]): + ready_ns = row.get("target_ns", row["due_ns"]) + if row["state"] not in {"NOT_DUE", "DEFERRED", "THERMAL_WAIT"} or ready_ns > self.now_ns or self.now_ns >= self.end_ns: + continue + guard = self._last_guard_states[row["stack"]] + if guard in {"severe", "shutdown"}: + if row["state"] != "THERMAL_WAIT" or row.get("blocked_guard") != guard: + row.update(state="THERMAL_WAIT", blocked_guard=guard) + self._timeline["maintenance"].append( + {"phase": "THERMAL_BLOCKED", **copy.deepcopy(row)}) + continue + row.pop("blocked_guard", None) + if not hasattr(self.mqsim, "maintain"): + row["state"] = "UNSUPPORTED_CAPABILITY" + self._timeline["maintenance"].append({"phase": row["state"], **copy.deepcopy(row)}) + continue + job = {key: row[key] for key in ("request_id", "stack", "stack_local_page", "due_ns")} + job.update(deadline_ns=row.get("deadline_ns", 0), + parent_id=row.get("parent_id"), + reclaim_source_block=row.get("reclaim_source_block", False), + failure_injection=row.get("failure_injection", "none"), + trigger_reason=row.get("trigger_reason", "RETENTION_AGE_DUE")) + response = self.mqsim.maintain(job) + status = response.get("status") + if status == "UNSUPPORTED_CAPABILITY": + row["state"] = status + elif response.get("maintenance_accepted") == 1: + backend_id = row["request_id"] + row.update(state="INFLIGHT", submit_ns=self.now_ns, + backend_request_id=backend_id, placement=response.get("placement"), + target=response.get("target")) + self._maintenance_pending[backend_id] = mid + elif response.get("target_ns", 0) > self.now_ns: + row.update(state="DEFERRED", target_ns=response["target_ns"]) + else: + row["state"] = "FAILED" + self._timeline["maintenance"].append({"phase": row["state"], **copy.deepcopy(row)}) + + def _window_facts(self, start, end, thermal_result): + probes = self.resource_probe(start, end) if self.resource_probe is not None else {} + stack_facts = [] + for stack in sorted(self.initial_budgets): + arrivals = [r for r in self._records.values() + if r["stack"] == stack and start <= r["arrival_ns"] < end] + delivered = [r for r in self._records.values() + if r["stack"] == stack and r["final_completion_ns"] is not None and + start <= r["final_completion_ns"] < end] + backlog = [r for r in self._records.values() if r["stack"] == stack and + r["arrival_ns"] < end and r["state"] != "COMPLETE"] + latencies = [r["final_completion_ns"]-r["arrival_ns"] for r in delivered] + probe = probes.get(stack, {}) + due = [m for m in self._maintenance.values() if m["stack"] == stack and + m["due_ns"] < end and m["state"] not in {"COMMITTED", "UNSUPPORTED_CAPABILITY"}] + stack_facts.append(StackWindowFacts( + stack_id=stack, offered_bytes=sum(r["valid_weight_bytes"] for r in arrivals), + delivered_bytes=sum(r["valid_weight_bytes"] for r in delivered), + backlog_bytes=sum(r["valid_weight_bytes"] for r in backlog), + oldest_wait_ns=max((end-r["arrival_ns"] for r in backlog), default=0), + latency_p95_ns=_percentile95(latencies), censored_requests=len(backlog), + gate_limited=any(r.get("gate_limited") for r in backlog), + backend_busy_fraction=probe.get("backend_busy_fraction"), + resource_busy=probe.get("resource_busy"), + maintenance_due_bytes=sum(int(m.get("bytes", 0)) for m in due), + maintenance_earliest_deadline_ns=min((m.get("deadline_ns", m["due_ns"]) + for m in due), default=None), + retry_count=probe.get("retry_count"), uecc_count=probe.get("uecc_count"))) + guards = thermal_result.get("stack_states", {}) + if set(guards) != set(self.initial_budgets): + raise UnsupportedComposition("thermal stack_states coverage mismatch") + self._last_guard_states = dict(guards) + endpoint_guards = {} + rank = {"normal": 0, "light": 1, "severe": 2, "shutdown": 3} + for endpoint in self.endpoint_caps: + involved = [stack for stack in guards if stack in endpoint] + endpoint_guards[endpoint] = max((guards[s] for s in involved), + key=lambda value: rank[value], default="normal") + routes = {stack: tuple(sorted({endpoint for row in self._records.values() + if row["stack"] == stack + for endpoint in row["route_endpoints"]})) + for stack in self.initial_budgets} + return WindowFacts(start_ns=start, end_ns=end, guard_state="normal", + stacks=tuple(stack_facts), current_budget_bytes=self._current_budgets, + hysteresis_budget_bytes=thermal_result.get("hysteresis_budget_bytes", {}), + shared_endpoint_caps_bytes=self.endpoint_caps, + route_endpoints=routes, + guard_states=guards, endpoint_guard_states=endpoint_guards) + + def _close_window(self, boundary): + component_energy = self.energy.flush(boundary) + rows = getattr(self.energy, "rows", ()) + for row in rows[self._energy_row_seen:]: + self._timeline["energy"].append({"kind": "ACTIVITY", **copy.deepcopy(row)}) + self._energy_row_seen = len(rows) + energy_row = {"kind": "WINDOW_TOTAL", "start_ns": self._thermal_start, "end_ns": boundary, + "component_energy_j": copy.deepcopy(component_energy), + "phase": "OBSERVATION" if boundary <= self.end_ns else "DRAIN"} + self._timeline["energy"].append(energy_row) + thermal_result = self.thermal.advance(self._thermal_start, boundary, component_energy) + if thermal_result.get("start_ns") != self._thermal_start or thermal_result.get("end_ns") != boundary: + raise UnsupportedComposition("thermal wrapper interval mismatch") + self._timeline["thermal"].append(copy.deepcopy(thermal_result)) + self._timeline["resources"].append({"time_ns": boundary, + "phase": energy_row["phase"], + "fabric": self.fabric.resource_state()}) + if boundary <= self.end_ns: + facts = self._window_facts(self._thermal_start, boundary, thermal_result) + decision = self.policy.evaluate(facts) + rate_rows = [] + for fact in facts.stacks: + row = asdict(fact) + arrivals = [r for r in self._records.values() if r["stack"] == fact.stack_id and + facts.start_ns <= r["arrival_ns"] < facts.end_ns] + delivered = [r for r in self._records.values() if r["stack"] == fact.stack_id and + r["final_completion_ns"] is not None and + facts.start_ns <= r["final_completion_ns"] < facts.end_ns] + backlog = [r for r in self._records.values() if r["stack"] == fact.stack_id and + r["arrival_ns"] < facts.end_ns and r["state"] != "COMPLETE"] + row.update(physical_offered_bytes=sum(r["bytes"] for r in arrivals), + physical_delivered_bytes=sum(r["bytes"] for r in delivered), + physical_backlog_bytes=sum(r["bytes"] for r in backlog), + byte_semantics="effective valid_weight_bytes; physical_* are backend extents") + rate_rows.append(row) + self._timeline["rates"].append({"start_ns": facts.start_ns, "end_ns": facts.end_ns, + "stacks": rate_rows}) + self._timeline["control"].append(asdict(decision)) + self._current_budgets = {row.stack_id: row.budget_bytes for row in decision.stack_decisions} + self._ledger = ByteTokenLedger(decision) + self._thermal_start = boundary + self._next_window = boundary + self.window_ns + + def _unfinished_submitted(self): + return (any(r["state"] in {"BACKEND_PENDING", "FABRIC_PENDING"} + for r in self._records.values()) or bool(self._maintenance_pending)) + + def _next_horizon(self): + candidates = [] + if self.now_ns < self.end_ns: + candidates.extend(r["arrival_ns"] for r in self._records.values() + if r["state"] == "NOT_ARRIVED" and r["arrival_ns"] < self.end_ns) + candidates.extend(m["due_ns"] for m in self._maintenance.values() + if m["state"] == "NOT_DUE" and m["due_ns"] < self.end_ns) + candidates.extend(m["target_ns"] for m in self._maintenance.values() + if m["state"] == "DEFERRED" and m["target_ns"] < self.end_ns) + candidates.extend(r["gate_target_ns"] for r in self._records.values() + if r["state"] == "GATE_WAIT" and self.now_ns < self.end_ns) + for source in (self.hbm, self.fabric): + if source is not None and source.next_event_ns() is not None: + candidates.append(source.next_event_ns()) + if (self._next_window <= self.end_ns or self._unfinished_submitted() or + self.now_ns > self._thermal_start): + candidates.append(self._next_window) + if candidates: + return min(value for value in candidates if value >= self.now_ns) + if self._mq_pending or self._maintenance_pending: + return MAX_HORIZON_NS + return None + + def run(self, requests, maintenance=()): + for sequence, raw in enumerate(requests): + self._validate_request(raw, sequence) + for sequence, raw in enumerate(maintenance): + self._validate_maintenance(raw, sequence) + self._waiting = sorted(self._records.values(), key=lambda row: (row["arrival_ns"], row["sequence"])) + while True: + # Component completion/release was processed by the preceding + # advance. At an exact boundary, sample/control precedes any new + # arrival or admission at that timestamp. + if self.now_ns == self._next_window: + self._close_window(self.now_ns) + if self.now_ns < self.end_ns: + self._admit() + self._submit_due_maintenance() + if self.now_ns >= self.end_ns and not self.drain_submitted: + break + if (self.now_ns >= self.end_ns and not self._unfinished_submitted() and + self.now_ns == self._thermal_start): + break + horizon = self._next_horizon() + if horizon is None or horizon < self.now_ns: + raise RuntimeError("closed loop deadlocked with no legal horizon") + self._advance_components(horizon) + + if self._thermal_start != self.now_ns: + raise AssertionError("drain must finish on a complete thermal window") + receipt = self.mqsim.finish() + for row in self._records.values(): + if row["state"] != "COMPLETE": + row["censored"] = True + row["censor_reason"] = ("NOT_ADMITTED_BY_EXPERIMENT_END" if + row["backend_submit_ns"] is None else "DRAIN_DISABLED") + self._timeline["requests"].append({"phase": "CENSORED", **copy.deepcopy(row)}) + resources = self.fabric.resource_state() + self._timeline["resources"].append({"time_ns": self.now_ns, "phase": "FINAL", + "fabric": copy.deepcopy(resources)}) + completed = [row for row in self._records.values() if row["state"] == "COMPLETE"] + observation_completed = [row for row in completed if row["final_completion_ns"] <= self.end_ns] + censored = [row for row in self._records.values() if row["state"] != "COMPLETE"] + latencies = [row["final_completion_ns"]-row["arrival_ns"] for row in observation_completed] + external_waits = [row["external_wait_ns"] for row in observation_completed] + backend_latencies = [row["backend_latency_ns"] for row in observation_completed] + fabric_latencies = [row["fabric_latency_ns"] for row in observation_completed] + effective = lambda rows: sum(row["valid_weight_bytes"] for row in rows) + physical = lambda rows: sum(row["bytes"] for row in rows) + return { + "schema_version": "eq3-isolated-maintenance-closed-loop-v1", + "evidence": "CONDITIONAL_ENGINEERING_COMPOSITION", + "topology_mode": self.mode, "observation_end_ns": self.end_ns, + "drain_end_ns": self.now_ns, "timeline": copy.deepcopy(self._timeline), + "requests": [copy.deepcopy(row) for row in sorted(self._records.values(), + key=lambda value: value["sequence"])], + "maintenance": [copy.deepcopy(row) for row in self._maintenance.values()], + "summary": { + "offered_count": len(self._records), + "offered_bytes": effective(self._records.values()), + "offered_effective_bytes": effective(self._records.values()), + "offered_physical_bytes": physical(self._records.values()), + "submitted_count": sum(row["backend_submit_ns"] is not None for row in self._records.values()), + "observation_completed_count": len(observation_completed), + "observation_completed_bytes": effective(observation_completed), + "observation_completed_effective_bytes": effective(observation_completed), + "observation_completed_physical_bytes": physical(observation_completed), + "drain_completed_count": len(completed)-len(observation_completed), + "drain_completed_effective_bytes": effective( + [row for row in completed if row not in observation_completed]), + "drain_completed_physical_bytes": physical( + [row for row in completed if row not in observation_completed]), + "censored_count": len(censored), + "censored_bytes": effective(censored), + "censored_effective_bytes": effective(censored), + "censored_physical_bytes": physical(censored), + "byte_semantics": ("legacy *_bytes aliases are effective valid_weight_bytes; " + "physical extents are explicit *_physical_bytes"), + "latency_p95_ns": _percentile95(latencies), + "external_wait_p95_ns": _percentile95(external_waits), + "backend_latency_p95_ns": _percentile95(backend_latencies), + "fabric_latency_p95_ns": _percentile95(fabric_latencies), + "maintenance_committed": sum(row.get("mapping_committed") is True + for row in self._maintenance.values()), + "maintenance_failed": sum(row["state"] == "FAILED" + for row in self._maintenance.values()), + "maintenance_cleanup_failed_after_commit": sum( + row["state"] == "COMMITTED_CLEANUP_FAILED" + for row in self._maintenance.values()), + "maintenance_unsupported": sum(row["state"] == "UNSUPPORTED_CAPABILITY" + for row in self._maintenance.values()), + "total_energy_j": getattr(self.energy, "total_j", sum( + sum(row["component_energy_j"].values()) for row in self._timeline["energy"] + if row.get("kind") == "WINDOW_TOTAL")), + "energy_semantics": "timeline.energy preserves observation and drain intervals separately", + }, + "mqsim_receipt": receipt, + "limits": { + "raw_vs_reported": "UNSUPPORTED_COMPOSITION_UNLESS_EQUAL", + "reliability": "OBSERVED_ONLY_NO_ECC_INFERENCE", + "maintenance_age": "INPUT_SECONDS_UNCOMPRESSED", + "unsubmitted_at_cutoff": "CENSORED_NOT_DRAINED", + }, + } diff --git a/experiments/eq3_maintenance/energy_ledger.py b/experiments/eq3_maintenance/energy_ledger.py new file mode 100644 index 0000000..5eb20c5 --- /dev/null +++ b/experiments/eq3_maintenance/energy_ledger.py @@ -0,0 +1,181 @@ +"""Integrate observed activity; engineering coefficients are never command facts. + +This owns no backend resources. Native media intervals, stack-level parametric +HBM intervals and real fabric intervals are disjoint energy scopes. Unknown +physical addresses are rejected, never inferred from a logical address. +""" +from collections import defaultdict +import math + + +class ActivityEnergyLedger: + def __init__(self, profile, stack_map, component_ids, *, hbm_dies=None, gpu_stop_ns=None): + self.profile = profile + self.components = set(component_ids) + self.mapping = stack_map + self.channel = {} + for group in stack_map['stacks']: + for index, channel in enumerate(group['channels']): + self.channel[channel] = (group['id'], index * stack_map['dies_per_channel']) + self.hbm_dies = hbm_dies or {} + self.gpu_stop = gpu_stop_ns + self.now = 0 + self.active = {} + self.closed = [] + self.rows = [] + self.total_j = 0.0 + self.physical_bytes = defaultdict(int) + self._seen = set() + + def _add(self, key, start, end, powers, scope, source): + if key in self._seen: + raise ValueError('duplicate energy activity identity') + if start < self.now or end is not None and end < start: + raise ValueError('activity would rewrite already committed thermal energy') + if not set(powers).issubset(self.components): + raise ValueError('energy target missing from physical thermal model') + if any(not math.isfinite(p) or p < 0 for p in powers.values()): + raise ValueError('nonfinite or negative physical power') + self._seen.add(key) + item = dict(key=key,start=start,end=end,powers=powers,scope=scope,source=source) + if end is None: + self.active[key] = item + else: + self.closed.append(item) + + def _end(self, key, time): + if key not in self.active: + raise ValueError('native phase end without unique begin') + item = self.active.pop(key) + if time < self.now or time < item['start']: + raise ValueError('native activity time moved backward') + item['end'] = time + self.closed.append(item) + + @staticmethod + def _source(tr): + if tr.get('maintenance_request_id',tr.get('maintenance_id')) not in (None,0,'UNKNOWN'): + return 'HBF_MAINTENANCE' + return 'FOREGROUND' if tr.get('external_request_id') not in (None,0,'UNKNOWN') else 'BACKEND_BACKGROUND' + + def _placement(self, tr): + channel, die, chip = tr.get('channel'), tr.get('die'), tr.get('chip') + if channel not in self.channel or type(die) is not int or type(chip) is not int or chip != 0: + raise ValueError('UNKNOWN or unsupported native physical die identity') + if not 0 <= die < self.mapping['dies_per_channel']: + raise ValueError('native die outside declared mapping') + stack, offset = self.channel[channel] + if tr.get('stack') not in (None,'UNKNOWN',stack): + raise ValueError('native stack differs from channel ownership') + return stack, f'{stack}.die{offset+die}' + + def native(self, event): + phase, time, cid = event['phase'],event['time_ns'],event['command_id'] + transactions = event['transactions'] + if phase == 0: + by_channel = {} + for tr in transactions: + stack,_ = self._placement(tr) + by_channel[tr['channel']] = (stack,self._source(tr)) + for channel,(stack,source) in by_channel.items(): + self._add(('command_bus',cid,channel),time,None, + {f'{stack}.base':self.profile['nand_command_transfer_w']}, + 'NAND_COMMAND_ADDRESS_DATA_IN',source) + elif phase in (1,2): + if phase == 1: + for channel in {tr['channel'] for tr in transactions}: + key=('command_bus',cid,channel) + if key in self.active: + self._end(key,time) + groups = defaultdict(list) + for tr in transactions: + stack,component = self._placement(tr) + groups[component].append(tr) + for component, trs in groups.items(): + key=('media',cid,component) + if phase == 1: + types = {str(tr['type']) for tr in trs} + if len(types) != 1: + raise ValueError('one media command contains inconsistent operation types') + op = next(iter(types)) + watt = self.profile['nand_media_w'][op] + sources = {self._source(tr) for tr in trs} + self._add(key,time,None,{component:watt},'NAND_MEDIA', + next(iter(sources)) if len(sources)==1 else 'MIXED_BACKGROUND_FOREGROUND') + for tr in trs: + self.physical_bytes[(self._source(tr),op)] += tr['bytes'] + else: + self._end(key,time) + elif phase in (3,4): + for tr in transactions: + stack,_ = self._placement(tr) + key=('nand_data_out',cid,tr['transaction_id']) + if phase == 3: + self._add(key,time,None,{f'{stack}.base':self.profile['nand_data_out_w']}, + 'NAND_DATA_OUT_BASE',self._source(tr)) + else: + self._end(key,time) + elif phase != 0: + raise ValueError('unsupported native phase') + + def hbm(self, fact): + if fact['phase'] != 'media_start': + return + stack=fact['stack_id']; dies=self.hbm_dies.get(stack) + if not dies: + raise ValueError('HBM thermal die distribution not explicitly configured') + start,end=fact['time_ns'],fact['media_end_ns'] + if end <= start: + raise ValueError('invalid HBM media interval') + energy=fact['bytes']*self.profile['hbm_array_j_per_byte'] + powers={die:energy/len(dies)*1e9/(end-start) for die in dies} + self._add(('hbm',fact['request_id']),start,end,powers, + 'HBM_ARRAY_UNIFORM_SPATIAL_ASSUMPTION','FOREGROUND') + + def fabric(self, event, request): + if event['kind'] != 'start': + return + stage=event['stage']; stack=request['stack'] + if stage in ('HBF_FILL','HBF_DIRECT'): + endpoints=[stack] + elif stage == 'HBF_RELAY': + endpoints=[stack,request['partner']] + elif stage == 'HBM_GPU': + endpoints=[request.get('partner',stack) if stack.startswith('hbf') else stack] + else: + raise ValueError('unsupported fabric energy stage') + start,end=event['start_ns'],event['end_ns'] + if end <= start: + raise ValueError('invalid fabric interval') + power=event['bytes']*self.profile['fabric_endpoint_j_per_byte']*1e9/(end-start) + self._add(('fabric',event['request_id'],stage),start,end, + {f'{x}.base':power for x in set(endpoints)},'FABRIC_'+stage,'FOREGROUND') + + def flush(self, boundary): + if type(boundary) is not int or boundary <= self.now: + raise ValueError('thermal boundary must increase') + totals=defaultdict(float) + for row in [*self.closed,*self.active.values()]: + begin=max(self.now,row['start']) + end=min(boundary,row['end'] if row['end'] is not None else boundary) + if end <= begin: + continue + for component,power in row['powers'].items(): + energy=power*(end-begin)*1e-9 + totals[component]+=energy + self.rows.append(dict(start_ns=begin,end_ns=end,component=component, + energy_j=energy,scope=row['scope'],source=row['source'], + coefficient_evidence=self.profile['evidence'])) + gpu_end=min(boundary,self.gpu_stop) if self.gpu_stop is not None else boundary + if gpu_end>self.now: + energy=self.profile['gpu_external_w']*(gpu_end-self.now)*1e-9 + if 'gpu' not in self.components: + raise ValueError('GPU thermal source missing') + totals['gpu']+=energy + self.rows.append(dict(start_ns=self.now,end_ns=gpu_end,component='gpu',energy_j=energy, + scope='EXTERNAL_COMPUTE_PRESCRIBED',source='GPU_SCENARIO', + coefficient_evidence=self.profile['evidence'])) + self.closed=[row for row in self.closed if row['end']>boundary] + self.now=boundary + self.total_j+=sum(totals.values()) + return dict(totals) diff --git a/experiments/eq3_maintenance/ideal_maintenance_replay.py b/experiments/eq3_maintenance/ideal_maintenance_replay.py new file mode 100644 index 0000000..9d06ac8 --- /dev/null +++ b/experiments/eq3_maintenance/ideal_maintenance_replay.py @@ -0,0 +1,214 @@ +"""Isolated fixed-intent A2 replay wrapper. + +Foreground requests still run on exactly one real MQSim engine. Maintenance +never enters that engine: observed baseline stages are released as an explicit +ideal-independent counterfactual stream for energy and virtual age accounting. +""" +from __future__ import annotations + +import copy +import hashlib +import json +from pathlib import Path + +try: + from .build_ideal_maintenance_bundle import EVIDENCE, SCHEMA +except ImportError: # Direct script-directory import used by run_point.py. + from build_ideal_maintenance_bundle import EVIDENCE, SCHEMA + + +def _sha256(path): + return hashlib.sha256(Path(path).read_bytes()).hexdigest() + + +class IdealIndependentMaintenanceReplay: + """Proxy a foreground-only service and expose frozen maintenance facts.""" + + def __init__(self, inner, bundle_path, *, profile_path, stack_map_path, + foreground_requests_path): + self.inner = inner + self.bundle_path = Path(bundle_path).resolve(strict=True) + self.bundle = json.loads(self.bundle_path.read_text()) + if (self.bundle.get("schema_version") != SCHEMA + or self.bundle.get("evidence") != EVIDENCE + or self.bundle.get("resource_model") != + "IDEAL_INDEPENDENT_MAINTENANCE_REPLAY"): + raise ValueError("invalid ideal-independent maintenance replay bundle") + contract = self.bundle["input_contract"] + actual = { + "profile_sha256": _sha256(profile_path), + "stack_map_sha256": _sha256(stack_map_path), + "foreground_requests_sha256": _sha256(foreground_requests_path), + } + if any(actual[key] != contract[key] for key in actual): + raise ValueError("A2 replay input identity differs from frozen source point") + foreground = json.loads(Path(foreground_requests_path).read_text()) + if any(row.get("operation") != "read" + for row in foreground if row.get("stack", "").startswith("hbf")): + raise ValueError("A2 fixed replay rejects concurrent HBF writes") + + self._intents = {row["request_id"]: copy.deepcopy(row) + for row in self.bundle["maintenance_intents"]} + self._submissions = {int(key): copy.deepcopy(value) + for key, value in self.bundle["submission_facts"].items()} + self._accepted = set() + self._inner_native_seen = 0 + self._event_index = self._native_index = self._completion_index = 0 + self.maintenance_events = [] + self.maintenance_completions = {} + self.native_observations = [] + self.header = copy.deepcopy(inner.header) + self.header.update( + maintenance_resource_model="IDEAL_INDEPENDENT_MAINTENANCE_REPLAY", + maintenance_replay_evidence=EVIDENCE, + actual_backend_maintenance="DISABLED_BY_WRAPPER", + replay_bundle_sha256=_sha256(self.bundle_path), + ) + + @property + def now(self): + return self.inner.now + + @property + def requests(self): + return self.inner.requests + + @property + def completions(self): + return self.inner.completions + + @property + def observations(self): + return self.inner.observations + + def _sync(self): + now = self.inner.now + new_native = [copy.deepcopy(row) for row in + self.inner.native_observations[self._inner_native_seen:]] + self._inner_native_seen = len(self.inner.native_observations) + replay_native = [] + rows = self.bundle["native_events"] + while self._native_index < len(rows) and rows[self._native_index]["time_ns"] <= now: + row = rows[self._native_index] + request_ids = {tr["maintenance_request_id"] for tr in row["transactions"]} + if not request_ids.issubset(self._accepted): + break + replay_native.append(copy.deepcopy(row)) + self._native_index += 1 + combined = [(row["time_ns"], 0, index, row) + for index, row in enumerate(new_native)] + combined += [(row["time_ns"], 1, row["replay_order"], row) + for row in replay_native] + self.native_observations.extend(row for *_, row in sorted(combined)) + + events = self.bundle["backend_events"] + while self._event_index < len(events) and events[self._event_index]["time_ns"] <= now: + row = events[self._event_index] + if row["request_id"] not in self._accepted: + break + self.maintenance_events.append(copy.deepcopy(row)) + self._event_index += 1 + completions = self.bundle["completions"] + while (self._completion_index < len(completions) + and completions[self._completion_index]["end_ns"] <= now): + row = copy.deepcopy(completions[self._completion_index]) + request_id = row["request_id"] + if request_id not in self._accepted or request_id in self.maintenance_completions: + raise RuntimeError("invalid replay maintenance completion identity") + self.maintenance_completions[request_id] = row + self._completion_index += 1 + + def submit(self, request): + if request.get("operation") != "read": + raise ValueError("A2 replay foreground is read-only") + answer = self.inner.submit(request) + self._sync() + return answer + + def try_submit(self, request): + if request.get("operation") != "read": + raise ValueError("A2 replay foreground is read-only") + answer = self.inner.try_submit(request) + self._sync() + return answer + + def until(self, horizon): + completion = self.inner.until(horizon) + self._sync() + return completion + + def maintain(self, job=None, **kwargs): + if job is not None: + if not isinstance(job, dict) or kwargs: + raise TypeError("maintain accepts one job dict or keyword arguments") + kwargs = dict(job) + request_id = kwargs.get("request_id") + if request_id not in self._intents or request_id in self._accepted: + raise ValueError("unknown or duplicate fixed replay maintenance request") + intent = self._intents[request_id] + normalized = { + "request_id": request_id, + "stack": kwargs.get("stack"), + "stack_local_page": kwargs.get("stack_local_page"), + "due_ns": kwargs.get("due_ns"), + "deadline_ns": kwargs.get("deadline_ns", 0), + "parent_id": request_id if kwargs.get("parent_id") is None else kwargs["parent_id"], + "reclaim_source_block": kwargs.get("reclaim_source_block", False), + "failure_injection": kwargs.get("failure_injection", "none"), + "trigger_reason": kwargs.get("trigger_reason", "UNSPECIFIED"), + } + expected = { + "request_id": request_id, + "stack": intent["stack"], + "stack_local_page": intent["stack_local_page"], + "due_ns": intent["due_ns"], + "deadline_ns": intent.get("deadline_ns", 0), + "parent_id": intent.get("parent_id", request_id), + "reclaim_source_block": intent.get("reclaim_source_block", False), + "failure_injection": intent.get("failure_injection", "none"), + "trigger_reason": intent.get("trigger_reason", "UNSPECIFIED"), + } + if normalized != expected: + raise ValueError("maintenance call differs from frozen A2 intent") + submit = self._submissions[request_id] + if self.now != submit["submit_ns"]: + raise RuntimeError("fixed replay is incompatible with changed maintenance submit time") + self._accepted.add(request_id) + return { + "maintenance_accepted": 1, + "status": "IDEAL_INDEPENDENT_REPLAY_ACCEPTED", + "placement": copy.deepcopy(submit.get("placement")), + "target": copy.deepcopy(submit.get("target")), + "actual_backend_submitted": False, + "evidence": EVIDENCE, + } + + def finish(self): + receipt = self.inner.finish() + self._sync() + expected = set(self._intents) + if (self._accepted != expected + or set(self.maintenance_completions) != expected + or self._event_index != len(self.bundle["backend_events"]) + or self._native_index != len(self.bundle["native_events"])): + raise ValueError("A2 replay finish conservation failed") + answer = copy.deepcopy(receipt) + answer["ideal_independent_maintenance_replay"] = { + "evidence": EVIDENCE, + "bundle_sha256": _sha256(self.bundle_path), + "actual_backend_maintenance_issued": receipt.get("maintenance_issued", 0), + "replay_maintenance_issued": len(expected), + "replay_maintenance_completed": len(self.maintenance_completions), + "current_mqsim_mapping_mutated_by_replay": False, + "source_version": "UNKNOWN_REPLAY", + } + return answer + + def close(self): + return self.inner.close() + + def __enter__(self): + return self + + def __exit__(self, *args): + self.close() diff --git a/experiments/eq3_maintenance/incremental_fabric.py b/experiments/eq3_maintenance/incremental_fabric.py new file mode 100644 index 0000000..adfb496 --- /dev/null +++ b/experiments/eq3_maintenance/incremental_fabric.py @@ -0,0 +1,29 @@ +"""Additive incremental observation adapter for the experimental fabric path. + +Scheduling, clocks, resource ownership, and the default BasicFabric API remain +owned by BasicFabric. These methods only deepcopy an already-produced suffix. +""" +from __future__ import annotations + +import copy + +from eq3_basic_fabric import BasicFabric + + +def _offset(value, size, label): + if isinstance(value, bool) or not isinstance(value, int) or not 0 <= value <= size: + raise ValueError(f"{label} must be an integer in [0,{size}]") + return value + + +class IncrementalBasicFabric(BasicFabric): + """BasicFabric with cursor-based, read-only observation access.""" + + def events_since(self, offset): + start = _offset(offset, len(self._events), "event offset") + return len(self._events), tuple(copy.deepcopy(self._events[start:])) + + def completions_since(self, offset): + start = _offset(offset, len(self._completions), "completion offset") + return len(self._completions), tuple(copy.deepcopy(self._completions[start:])) + diff --git a/experiments/eq3_maintenance/launch_point.py b/experiments/eq3_maintenance/launch_point.py new file mode 100644 index 0000000..4cce8cf --- /dev/null +++ b/experiments/eq3_maintenance/launch_point.py @@ -0,0 +1,42 @@ +#!/usr/bin/env python3 +"""Bound one owned experiment process group, preserving startup/failure evidence.""" +import argparse +import datetime +import json +import os +from pathlib import Path +import signal +import shutil +import subprocess +import time + +p=argparse.ArgumentParser();p.add_argument('--receipt',type=Path,required=True);p.add_argument('--wall-s',type=float,default=600);p.add_argument('--memory-reserve-gib',type=float,default=0);p.add_argument('--disk-reserve-gib',type=float,default=0);p.add_argument('command',nargs=argparse.REMAINDER);a=p.parse_args() +if not a.command or a.wall_s<=0:p.error('positive finite wall and explicit command required') +a.receipt.mkdir(parents=True,exist_ok=False) +(a.receipt/'launch.json').write_text(json.dumps(dict(command=a.command,wall_s=a.wall_s, + timestamp_utc=datetime.datetime.now(datetime.timezone.utc).isoformat(), + authority='USER_CONFIRMED EQ3-ISOLATED-MAINTENANCE-CAMPAIGN-v1; point manifest supplied by consumer'),indent=2)) +started=time.monotonic();reason=None +with (a.receipt/'stdout.log').open('wb') as stdout,(a.receipt/'stderr.log').open('wb') as stderr: + child=subprocess.Popen(a.command,stdout=stdout,stderr=stderr,start_new_session=True, + env=dict(os.environ,PYTHONDONTWRITEBYTECODE='1')) + while True: + remaining=a.wall_s-(time.monotonic()-started) + if remaining<=0:reason='WATCHDOG';break + try: + code=child.wait(timeout=min(10,remaining));break + except subprocess.TimeoutExpired: + mem=dict(line.split(':',1) for line in Path('/proc/meminfo').read_text().splitlines()) + available=int(mem['MemAvailable'].split()[0])*1024 + if available int: + value = row.get(field) + if value in (None, ""): + raise ValueError(f"commands row lacks {field}") + try: + parsed = int(value) + except (TypeError, ValueError) as exc: + raise ValueError(f"commands row has non-integer {field}") from exc + if parsed < 0: + raise ValueError(f"commands row has negative {field}") + return parsed + + +def _enum(row: Mapping[str, str], field: str, values: Mapping[int, str]) -> tuple[int, str]: + code = _integer(row, field) + if code not in values: + raise ValueError(f"commands row has unsupported {field}={code}") + return code, values[code] + + +def _empty_operation() -> dict: + return { + "media_command_starts": 0, + "media_command_ends": 0, + "paired_media_commands": 0, + "unpaired_media_starts": 0, + "unpaired_media_ends": 0, + "child_transactions": 0, + "media_start_first_ns": None, + "media_start_last_ns": None, + "media_end_first_ns": None, + "media_end_last_ns": None, + "paired_media_duration_ns": { + "count": 0, "minimum": None, "maximum": None, "sum": 0, + }, + "source_counts": {}, + } + + +def summarize_rows(rows: Iterable[Mapping[str, str]], *, source_name: str = "commands.csv") -> dict: + """Summarize flattened command rows without inferring unobserved facts.""" + + # COMMAND_ISSUED/MEDIA_BEGIN/MEDIA_END are command-level callbacks whose + # rows are flattened over children. DATA_OUT callbacks are emitted once per + # child transaction and may legitimately have distinct timestamps. + phase_events: dict[tuple[int, int, int | None], dict] = {} + transactions: dict[int, dict] = {} + row_count = 0 + + for row in rows: + row_count += 1 + missing = REQUIRED_FIELDS.difference(row) + if missing: + raise ValueError(f"commands row lacks fields: {sorted(missing)}") + command_id = _integer(row, "command_id") + phase_code, phase_name = _enum(row, "phase", PHASE) + time_ns = _integer(row, "time_ns") + transaction_id = _integer(row, "transaction_id") + operation_code, operation = _enum(row, "type", OPERATION) + source_code, source = _enum(row, "source", SOURCE) + placement = tuple(_integer(row, key) for key in ("channel", "chip", "die", "plane")) + + transaction = { + "operation_code": operation_code, + "operation": operation, + "source_code": source_code, + "source": source, + "placement": placement, + } + previous = transactions.setdefault(transaction_id, transaction) + if previous != transaction: + raise ValueError(f"transaction {transaction_id} changed identity") + + key = (command_id, phase_code, transaction_id if phase_code in (3, 4) else None) + event = phase_events.setdefault(key, { + "command_id": command_id, + "phase_code": phase_code, + "phase": phase_name, + "time_ns": time_ns, + "transaction_ids": set(), + "operations": set(), + "sources": set(), + }) + if event["time_ns"] != time_ns: + raise ValueError(f"command {command_id} phase {phase_name} has multiple timestamps") + event["transaction_ids"].add(transaction_id) + event["operations"].add(operation) + event["sources"].add(source) + + operations = {name: _empty_operation() for name in OPERATION.values()} + transaction_sources: Counter[str] = Counter() + placement_transactions: defaultdict[tuple[int, int, int, int], set[int]] = defaultdict(set) + for transaction_id, transaction in transactions.items(): + operation = transaction["operation"] + source = transaction["source"] + operations[operation]["child_transactions"] += 1 + transaction_sources[source] += 1 + placement_transactions[transaction["placement"]].add(transaction_id) + + command_sources: Counter[str] = Counter() + operation_source_starts: defaultdict[str, Counter[str]] = defaultdict(Counter) + operation_source_ends: defaultdict[str, Counter[str]] = defaultdict(Counter) + starts: dict[int, dict] = {} + ends: dict[int, dict] = {} + + for (_, phase_code, _), event in phase_events.items(): + if phase_code not in (1, 2): + continue + if len(event["operations"]) != 1: + raise ValueError( + f"command {event['command_id']} {event['phase']} contains mixed operations") + operation = next(iter(event["operations"])) + source = next(iter(event["sources"])) if len(event["sources"]) == 1 else "MIXED" + target = starts if phase_code == 1 else ends + if event["command_id"] in target: + raise ValueError(f"duplicate deduplicated {event['phase']} command") + target[event["command_id"]] = event + field = "media_command_starts" if phase_code == 1 else "media_command_ends" + operations[operation][field] += 1 + times = [x["time_ns"] for x in target.values() + if next(iter(x["operations"])) == operation] + prefix = "media_start" if phase_code == 1 else "media_end" + operations[operation][f"{prefix}_first_ns"] = min(times) + operations[operation][f"{prefix}_last_ns"] = max(times) + if phase_code == 1: + command_sources[source] += 1 + operation_source_starts[operation][source] += 1 + else: + operation_source_ends[operation][source] += 1 + + for operation, summary in operations.items(): + start_ids = {cid for cid, event in starts.items() + if next(iter(event["operations"])) == operation} + end_ids = {cid for cid, event in ends.items() + if next(iter(event["operations"])) == operation} + paired = start_ids & end_ids + durations = [] + for command_id in paired: + start, end = starts[command_id], ends[command_id] + if start["operations"] != end["operations"]: + raise ValueError(f"command {command_id} changed operation across media phases") + if end["time_ns"] < start["time_ns"]: + raise ValueError(f"command {command_id} media end precedes media begin") + durations.append(end["time_ns"] - start["time_ns"]) + summary["paired_media_commands"] = len(paired) + summary["unpaired_media_starts"] = len(start_ids - end_ids) + summary["unpaired_media_ends"] = len(end_ids - start_ids) + if durations: + summary["paired_media_duration_ns"] = { + "count": len(durations), "minimum": min(durations), + "maximum": max(durations), "sum": sum(durations), + } + sources = set(operation_source_starts[operation]) | set(operation_source_ends[operation]) + summary["source_counts"] = { + name: { + "media_command_starts": operation_source_starts[operation][name], + "media_command_ends": operation_source_ends[operation][name], + } + for name in sorted(sources) + } + + tuple_rows = [{ + "channel": key[0], "chip": key[1], "die": key[2], "plane": key[3], + "child_transactions": len(transaction_ids), + } for key, transaction_ids in sorted(placement_transactions.items())] + triple_transactions: defaultdict[tuple[int, int, int], set[int]] = defaultdict(set) + for (channel, _chip, die, plane), transaction_ids in placement_transactions.items(): + triple_transactions[(channel, die, plane)].update(transaction_ids) + triple_rows = [{ + "channel": key[0], "die": key[1], "plane": key[2], + "child_transactions": len(transaction_ids), + } for key, transaction_ids in sorted(triple_transactions.items())] + + return { + "schema_version": "eq3-native-operation-summary-v1", + "source": source_name, + "evidence": "ACTUAL_MQSIM_COMMAND_OBSERVATIONS", + "enum_contract": { + "phase": {str(key): value for key, value in PHASE.items()}, + "operation": {str(key): value for key, value in OPERATION.items()}, + "source": {str(key): value for key, value in SOURCE.items()}, + "verified_source": { + "phase": "experiments/eq3_maintenance/backend/patches/0004-eq3-maintenance.patch:NVM_PHY_ONFI_NVDDR2.h::HBF_Command_Observation_Phase", + "operation_and_source": "experiments/eq3_maintenance/backend/patches/0004-eq3-maintenance.patch:NVM_Transaction.h::{Transaction_Type,Transaction_Source_Type}", + }, + }, + "deduplication": { + "input_rows": row_count, + "native_phase_events": len(phase_events), + "child_transactions": len(transactions), + "media_commands_are_unique_by": ["command_id", "phase"], + "child_transactions_are_unique_by": ["transaction_id"], + }, + "operations": operations, + "source_child_transaction_counts": dict(sorted(transaction_sources.items())), + "source_media_command_start_counts": dict(sorted(command_sources.items())), + "actual_channel_die_plane_tuple_count": len(triple_rows), + "actual_channel_die_plane_tuples": triple_rows, + "actual_channel_chip_die_plane_tuple_count": len(tuple_rows), + "actual_channel_chip_die_plane_tuples": tuple_rows, + "completion_semantics": { + "media_end": "NAND_MEDIA_PHASE_ENDED_NOT_COMMITTED_SUCCESS", + "maintenance_commit": "REQUIRES_SEPARATE_MAINTENANCE_COMPLETION_FACT", + "failure_injection": "MEDIA_ACTIVITY_BEFORE_A_TERMINAL_FAILURE_REMAINS_OBSERVED", + }, + "unavailable": { + "program_erase_lifetime_cycles": "UNAVAILABLE_NOT_OBSERVED", + "maximum_physical_spare_capacity_bytes": "UNAVAILABLE_NOT_OBSERVED", + "ecc_corrected_bits": "UNAVAILABLE_NOT_OBSERVED", + "ecc_uncorrectable_events": "UNAVAILABLE_NOT_OBSERVED", + "payload_or_byte_integrity": "UNAVAILABLE_NOT_OBSERVED_METADATA_VALIDITY_ONLY", + }, + } + + +def summarize_commands_csv(path: str | Path) -> dict: + source = Path(path) + with source.open(newline="") as handle: + reader = csv.DictReader(handle) + if reader.fieldnames is None: + raise ValueError("commands CSV has no header") + missing = REQUIRED_FIELDS.difference(reader.fieldnames) + if missing: + raise ValueError(f"commands CSV lacks fields: {sorted(missing)}") + return summarize_rows(reader, source_name=str(source)) + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("commands_csv", type=Path) + parser.add_argument("--output", type=Path) + args = parser.parse_args() + result = summarize_commands_csv(args.commands_csv) + text = json.dumps(result, indent=2, sort_keys=True) + "\n" + if args.output: + args.output.parent.mkdir(parents=True, exist_ok=True) + args.output.write_text(text) + else: + print(text, end="") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/experiments/eq3_maintenance/pilot_energy_replay.py b/experiments/eq3_maintenance/pilot_energy_replay.py new file mode 100644 index 0000000..93b89e6 --- /dev/null +++ b/experiments/eq3_maintenance/pilot_energy_replay.py @@ -0,0 +1,151 @@ +#!/usr/bin/env python3 +"""Convert committed maintenance-pilot thermal windows to RC node events. + +Only ``WINDOW_TOTAL`` rows are consumed. The ACTIVITY rows in the same ledger +are evidence for those totals and must not be integrated a second time. +""" +import argparse +import csv +import hashlib +import json +import math +from pathlib import Path + + +def _sha256(path): + return hashlib.sha256(Path(path).read_bytes()).hexdigest() + + +def _seconds(nanoseconds): + return format(nanoseconds / 1_000_000_000, ".17g") + + +def read_windows(path): + windows = [] + with Path(path).open(newline="") as stream: + reader = csv.DictReader(stream) + required = {"kind", "start_ns", "end_ns", "component_energy_j", "phase"} + if not reader.fieldnames or not required.issubset(reader.fieldnames): + raise ValueError("energy ledger lacks the WINDOW_TOTAL contract") + for row_number, row in enumerate(reader, 2): + if row["kind"] != "WINDOW_TOTAL": + continue + try: + start = int(row["start_ns"]) + end = int(row["end_ns"]) + totals = json.loads(row["component_energy_j"]) + except (TypeError, ValueError, json.JSONDecodeError) as error: + raise ValueError(f"malformed WINDOW_TOTAL at CSV row {row_number}") from error + if start < 0 or end <= start or not isinstance(totals, dict): + raise ValueError(f"invalid WINDOW_TOTAL at CSV row {row_number}") + clean = {} + for component, energy in totals.items(): + if not isinstance(component, str) or not component: + raise ValueError(f"invalid component at CSV row {row_number}") + if isinstance(energy, bool) or not isinstance(energy, (int, float)): + raise ValueError(f"invalid energy at CSV row {row_number}") + energy = float(energy) + if not math.isfinite(energy) or energy < 0: + raise ValueError(f"invalid energy at CSV row {row_number}") + if energy: + clean[component] = energy + windows.append({"start_ns": start, "end_ns": end, + "phase": row["phase"] or "UNKNOWN", + "component_energy_j": clean}) + if not windows: + raise ValueError("energy ledger contains no WINDOW_TOTAL rows") + for index, window in enumerate(windows): + if index and window["start_ns"] != windows[index - 1]["end_ns"]: + raise ValueError("WINDOW_TOTAL intervals must be ordered and contiguous") + return windows + + +def convert(energy_csv, grid_json): + windows = read_windows(energy_csv) + grid = json.loads(Path(grid_json).read_text()) + cells = grid.get("cells") + component_cells = grid.get("component_cells") + if not isinstance(cells, list) or not isinstance(component_cells, dict): + raise ValueError("grid lacks cells/component_cells") + + lines = ["HBFSIM_EQ3_THERMAL_EVENTS 1"] + source_by_component = {} + emitted_by_component = {} + event_id = 0 + phase_windows = [] + for window in windows: + phase_windows.append({key: window[key] for key in ("start_ns", "end_ns", "phase")}) + for component, energy in sorted(window["component_energy_j"].items()): + if component not in component_cells or not component_cells[component]: + raise ValueError(f"energy component is absent from grid: {component}") + indices = component_cells[component] + try: + volumes = [float(cells[index]["volume_m3"]) for index in indices] + node_ids = [str(cells[index]["id"]) for index in indices] + except (IndexError, KeyError, TypeError, ValueError) as error: + raise ValueError(f"malformed grid mapping for {component}") from error + if any(not math.isfinite(value) or value <= 0 for value in volumes): + raise ValueError(f"non-positive cell volume for {component}") + total_volume = math.fsum(volumes) + assignments = [energy * volume / total_volume for volume in volumes] + # Put floating division residue on the last cell so the serialized + # event conserves the ledger total to normal double precision. + assignments[-1] += energy - math.fsum(assignments) + event_id += 1 + payload = " ".join(f"{node} {value:.17g}" + for node, value in zip(node_ids, assignments)) + start_s = _seconds(window["start_ns"]) + end_s = _seconds(window["end_ns"]) + lines.append(f"activity {event_id} replay-{event_id} external_heat external " + f"0 -1 -1 -1 {start_s} {end_s} {end_s} {payload}") + source_by_component[component] = source_by_component.get(component, 0.0) + energy + emitted_by_component[component] = ( + emitted_by_component.get(component, 0.0) + math.fsum(assignments)) + + for component, source in source_by_component.items(): + if not math.isclose(source, emitted_by_component[component], + rel_tol=1e-13, abs_tol=1e-15): + raise AssertionError(f"energy conversion failed for {component}") + events = "\n".join(lines) + "\n" + receipt = { + "schema_version": "eq3-maintenance-pilot-energy-replay-v1", + "status": "GENERATED_NOT_SOLVED", + "source_energy_csv_sha256": _sha256(energy_csv), + "grid_sha256": _sha256(grid_json), + "events_sha256": hashlib.sha256(events.encode()).hexdigest(), + "window_count": len(windows), + "event_count": event_id, + "start_ns": windows[0]["start_ns"], + "end_ns": windows[-1]["end_ns"], + "phase_windows": phase_windows, + "source_energy_by_component_j": source_by_component, + "emitted_energy_by_component_j": emitted_by_component, + "source_total_energy_j": math.fsum(source_by_component.values()), + "emitted_total_energy_j": math.fsum(emitted_by_component.values()), + "activity_rows_consumed": False, + "solver_started": False, + } + return events, receipt + + +def main(): + parser = argparse.ArgumentParser() + parser.add_argument("--energy-csv", type=Path, required=True) + parser.add_argument("--grid", type=Path, required=True) + parser.add_argument("--events-output", type=Path, required=True) + parser.add_argument("--receipt-output", type=Path, required=True) + args = parser.parse_args() + for output in (args.events_output, args.receipt_output): + output.parent.mkdir(parents=True, exist_ok=True) + if output.exists(): + raise FileExistsError(f"refusing to overwrite {output}") + events, receipt = convert(args.energy_csv, args.grid) + args.events_output.write_text(events) + args.receipt_output.write_text(json.dumps(receipt, indent=2, sort_keys=True) + "\n") + print(json.dumps({"status": receipt["status"], "window_count": receipt["window_count"], + "event_count": receipt["event_count"], + "total_energy_j": receipt["source_total_energy_j"]})) + + +if __name__ == "__main__": + main() diff --git a/experiments/eq3_maintenance/plot_closed_loop.py b/experiments/eq3_maintenance/plot_closed_loop.py new file mode 100644 index 0000000..434ed9b --- /dev/null +++ b/experiments/eq3_maintenance/plot_closed_loop.py @@ -0,0 +1,66 @@ +#!/usr/bin/env python3 +"""Read-only six-panel timeline of actual raw observations; no solver calls.""" +import argparse +import json +import math +from pathlib import Path +from analyze_points import analyze_point, _rows, _integer, _cell + +def plot(point, output): + import matplotlib + matplotlib.use('Agg') + import matplotlib.pyplot as plt + evidence = analyze_point(point) + rows = evidence['time_series'] + x = [(r['start_ns']+r['end_ns'])/2e9 for r in rows] + def values(key, scale=1): + out=[] + for r in rows: + value=r + for field in key.split('.'): value=value.get(field) if isinstance(value,dict) else None + out.append(float(value)/scale if value is not None else math.nan) + return out + latest={} + for row in _rows(point/'maintenance.csv'): + if row.get('request_id'): latest[row['request_id']]=row + maintenance=[] + for row in latest.values(): + completion=_cell(row.get('completion'));completion=completion if isinstance(completion,dict) else {} + reset=_integer(row.get('age_reset_ns')) + if reset is None:reset=_integer(completion.get('age_reset_ns')) + maintenance.append((_integer(row.get('due_ns')),reset,_integer(row.get('bytes')))) + backlog=[];rate=[] + for row in rows: + a,b=row['start_ns'],row['end_ns'] + backlog.append(sum(d is not None and d<=b and (r is None or r>b) for d,r,_ in maintenance)) + rate.append(sum(size for _,r,size in maintenance if size is not None and r is not None and a<=r original_budget else "HOLD") + decisions.append(StackDecision(observed.stack_id, current, + action, + "THERMAL_GUARD", (thermal_reason,))) + self._previous[observed.stack_id] = (observed, current, "guard") + continue + if self.profile.strategy == "guard_only": + budget = self._bounded(current) + action = "INCREASE" if budget > original_budget else "DECREASE" if budget < original_budget else "HOLD" + decisions.append(StackDecision(observed.stack_id, budget, action, "GUARD_ONLY", + ("NO_ORDINARY_RATE_FEEDBACK",))) + continue + if self.profile.strategy == "thermal_hysteresis_guard": + budget = self._bounded(current) + decisions.append(StackDecision(observed.stack_id, budget, + "INCREASE" if budget > original_budget else + "DECREASE" if budget < original_budget else "HOLD", + "THERMAL_HYSTERESIS", ("EXISTING_HYSTERESIS_CAP",))) + continue + + # New arrivals alone are not the demand population: after an input + # cutoff, previously offered requests can remain queued. max() + # avoids adding overlapping offered and delivered/backlog views. + demand_bytes = max(observed.offered_bytes, + observed.delivered_bytes + observed.backlog_bytes) + enough_demand = demand_bytes >= target + latency_met = (self.profile.target_latency_p95_ns is None or + (observed.latency_p95_ns is not None and + observed.latency_p95_ns <= self.profile.target_latency_p95_ns)) + met_rate = (observed.delivered_bytes >= target*(1-self.profile.tolerance_fraction) + and latency_met) + reliability_bad = ((observed.uecc_count or 0) > 0) + previous = self._previous.get(observed.stack_id) + budget, action, outcome, reasons = current, "HOLD", "STABLE", [] + if not enough_demand: + outcome, reasons = "INSUFFICIENT_DEMAND", ["DEMAND_BELOW_TARGET"] + elif reliability_bad: + outcome, reasons = "UNMET_TARGET", ["OBSERVED_UECC"] + elif met_rate: + excess = observed.delivered_bytes > target*(1+self.profile.smoothing_excess_fraction) + if excess and current > self.profile.minimum_budget_bytes: + budget, action = self._bounded(current-self.profile.step_bytes), "DECREASE" + reasons = ["OVERDELIVERY_SMOOTHING"] + else: + reasons = ["TARGET_MET_HOLD"] + else: + outcome = "UNMET_TARGET" + backend_idle = (observed.backend_busy_fraction is not None and + observed.backend_busy_fraction < 1 and observed.resource_busy is False) + facts_unknown = observed.backend_busy_fraction is None or observed.resource_busy is None + rollback = False + if previous and previous[2] == "increase": + prior, prior_budget, _ = previous + rollback = (current > prior_budget and observed.delivered_bytes <= prior.delivered_bytes and + (observed.backlog_bytes > prior.backlog_bytes or + (observed.latency_p95_ns is not None and prior.latency_p95_ns is not None and + observed.latency_p95_ns > prior.latency_p95_ns))) + if rollback: + budget, action, reasons = self._bounded(current-self.profile.step_bytes), "DECREASE", ["ROLLBACK_NO_DELIVERY_GAIN"] + elif observed.gate_limited and observed.backlog_bytes > 0 and backend_idle: + budget, action, reasons = self._bounded(current+self.profile.step_bytes), "INCREASE", ["GATE_LIMITED_BACKEND_IDLE"] + elif facts_unknown: + reasons = ["BOTTLENECK_UNKNOWN_HOLD"] + elif observed.backend_busy_fraction == 1 or observed.resource_busy: + reasons = ["BACKEND_OR_RESOURCE_BUSY_HOLD"] + else: + reasons = ["NO_EVIDENCE_TO_INCREASE_HOLD"] + if observed.maintenance_due_bytes: + reasons.append("MAINTENANCE_DUE_VISIBLE") + decisions.append(StackDecision(observed.stack_id, int(budget), action, outcome, tuple(reasons))) + self._previous[observed.stack_id] = (observed, current, action.lower()) + + endpoint = dict(facts.shared_endpoint_caps_bytes) + if self.profile.enabled: + for key, value in tuple(endpoint.items()): + guard = facts.endpoint_guard_states.get(key, facts.guard_state) + if guard == "shutdown": + endpoint[key] = 0 + elif guard == "severe": + endpoint[key] = min(value, self.profile.severe_budget_bytes) + elif guard == "light" and self.profile.strategy != "guard_only": + endpoint[key] = int(value*self.profile.light_fraction) + return Decision(self.profile.enabled, self.profile.strategy, facts.end_ns, + tuple(decisions), endpoint) + + +class ByteTokenLedger: + """Atomic consumer helper; duplicate route endpoints are charged once.""" + def __init__(self, decision: Decision): + self.enabled = decision.enabled + self.stack = {row.stack_id: row.budget_bytes for row in decision.stack_decisions} + self.endpoint = dict(decision.shared_endpoint_budget_bytes) + + def can_consume(self, stack_id: str, route_endpoints, byte_count: int): + if byte_count < 0: + raise ValueError("byte_count must be nonnegative") + if not self.enabled: + return True + names = set(route_endpoints) + if stack_id not in self.stack or any(name not in self.endpoint for name in names): + raise ValueError("unknown stack or route endpoint") + return (self.stack[stack_id] >= byte_count and + all(self.endpoint[name] >= byte_count for name in names)) + + def try_consume(self, stack_id: str, route_endpoints, byte_count: int): + if not self.can_consume(stack_id, route_endpoints, byte_count): + return False + if not self.enabled: + return True + names = set(route_endpoints) + # All checks precede all mutations: failed admission leaves no partial reservation. + self.stack[stack_id] -= byte_count + for name in names: + self.endpoint[name] -= byte_count + return True diff --git a/experiments/eq3_maintenance/repro_maintenance_energy_source_label.py b/experiments/eq3_maintenance/repro_maintenance_energy_source_label.py new file mode 100644 index 0000000..2a0933a --- /dev/null +++ b/experiments/eq3_maintenance/repro_maintenance_energy_source_label.py @@ -0,0 +1,80 @@ +#!/usr/bin/env python3 +"""Fixed reproduction for the maintenance native-source classification mismatch.""" +from __future__ import annotations + +import argparse +import csv +import json +from pathlib import Path + +try: + from .energy_ledger import ActivityEnergyLedger +except ImportError: # Direct invocation from the experiment directory. + from energy_ledger import ActivityEnergyLedger + + +def diagnose(result_path=None, activity_path=None): + actual_field = ActivityEnergyLedger._source({"maintenance_request_id": 7}) + legacy_field = ActivityEnergyLedger._source({"maintenance_id": 7}) + if actual_field == legacy_field == "HBF_MAINTENANCE": + status = "FIXED" + elif (actual_field == "BACKEND_BACKGROUND" + and legacy_field == "HBF_MAINTENANCE"): + status = "CONFIRMED_BUG" + else: + status = "UNEXPECTED_CLASSIFICATION" + receipt = { + "schema_version": "eq3-maintenance-energy-source-repro-v1", + "status": status, + "actual_service_field": "maintenance_request_id", + "actual_field_classification": actual_field, + "legacy_alias_classification": legacy_field, + "physics_effect": "NONE; source is an evidence label, not a power coefficient", + "minimal_patch": ( + "In ActivityEnergyLedger._source, read maintenance_request_id first; " + "fall back to maintenance_id only for legacy/replay compatibility." + ), + "foreground_classification": ActivityEnergyLedger._source( + {"external_request_id": 9}), + "unknown_background_classification": ActivityEnergyLedger._source({}), + } + if result_path is not None and activity_path is not None: + result = json.loads(Path(result_path).read_text()) + native = [event for event in result["timeline"]["native"] + if any(row.get("maintenance_request_id") not in (None, 0, "UNKNOWN") + for row in event["transactions"])] + with Path(activity_path).open(newline="") as stream: + activity = list(csv.DictReader(stream)) + background = [row for row in activity if row["source"] == "BACKEND_BACKGROUND"] + labelled = [row for row in activity if row["source"] == "HBF_MAINTENANCE"] + receipt["fixed_point"] = { + "maintenance_native_events": len(native), + "maintenance_request_ids": len({row["maintenance_request_id"] + for event in native for row in event["transactions"] + if row.get("maintenance_request_id") not in (None, 0, "UNKNOWN")}), + "backend_background_activity_rows": len(background), + "backend_background_energy_j": sum(float(row["energy_j"]) for row in background), + "hbf_maintenance_activity_rows": len(labelled), + "hbf_maintenance_energy_j": sum(float(row["energy_j"]) for row in labelled), + } + return receipt + + +def main(): + parser = argparse.ArgumentParser() + parser.add_argument("--result", type=Path) + parser.add_argument("--energy-activity", type=Path) + parser.add_argument("--output", type=Path, required=True) + args = parser.parse_args() + if (args.result is None) != (args.energy_activity is None): + raise ValueError("result and energy-activity must be supplied together") + if args.output.exists(): + raise FileExistsError(f"refusing to overwrite {args.output}") + receipt = diagnose(args.result, args.energy_activity) + args.output.parent.mkdir(parents=True, exist_ok=True) + args.output.write_text(json.dumps(receipt, indent=2, sort_keys=True) + "\n") + print(json.dumps(receipt, sort_keys=True)) + + +if __name__ == "__main__": + main() diff --git a/experiments/eq3_maintenance/run_campaign.py b/experiments/eq3_maintenance/run_campaign.py new file mode 100644 index 0000000..2bf621f --- /dev/null +++ b/experiments/eq3_maintenance/run_campaign.py @@ -0,0 +1,38 @@ +#!/usr/bin/env python3 +"""Execute an explicit frozen conditional campaign in owned independent processes.""" +import argparse,concurrent.futures,datetime,json,os,pathlib,shutil,subprocess,time +p=argparse.ArgumentParser();p.add_argument('--stage',type=pathlib.Path,required=True);p.add_argument('--artifact-root',type=pathlib.Path,required=True);p.add_argument('--binary',type=pathlib.Path,required=True);p.add_argument('--thermal-binary',type=pathlib.Path,required=True);a=p.parse_args() +a.stage=a.stage.resolve();root=pathlib.Path(__file__).resolve().parents[2];here=pathlib.Path(__file__).resolve().parent +lock_path=a.stage/'MAIN_CAMPAIGN_LOCK_v1.json';lock=json.loads(lock_path.read_text());cpus=sorted(os.sched_getaffinity(0))[:3] +if len(cpus)!=3:raise RuntimeError('frozen3 independent CPUs unavailable') +started=time.monotonic();status=[] +queue=[] +for workload in ('W1','W2'): + for qi,mode in enumerate(lock['topologies'],1): + for scene,config in lock['scenarios'].items(): + for pi,policy in enumerate(lock['policies']): + queue.append(dict(id=f'MAIN-{workload}-Q{qi}-{scene}-P{pi}',workload=workload,mode=mode,scene=scene,policy=policy,config=config,cpu=cpus[pi])) +queue_path=a.stage/'MAIN_QUEUE_v1.json' +if queue_path.exists():raise RuntimeError('create-only campaign; existing queue requires explicit resume review') +queue_path.write_text(json.dumps(dict(created_utc=datetime.datetime.now(datetime.timezone.utc).isoformat(),points=queue,lock=str(lock_path)),indent=2)) +def run_point(row): + if time.monotonic()-started>lock['resources']['campaign_wall_s']:return dict(row,status='NOT_STARTED_CAMPAIGN_WALL') + if shutil.disk_usage(a.stage).free<100*1024**3:return dict(row,status='NOT_STARTED_DISK_RESERVE') + c=row['config'];out=a.stage/'points'/row['id'];model=(a.stage/'points/Q4-THERMAL-STATIC01/generated_2mm' if row['mode']=='all_hbf_direct' else a.artifact_root/'generated/campaign-RC2MM-train') + cmd=['taskset','-c',str(row['cpu']),'python3',str(here/'launch_point.py'),'--receipt',str(out)+'-launch','--wall-s','600','python3',str(here/'run_point.py'),'--binary',str(a.binary.resolve()),'--thermal-binary',str(a.thermal_binary.resolve()),'--model-dir',str(model.resolve()),'--output',str(out),'--artifact-root',str(a.artifact_root.resolve()),'--mode',row['mode'],'--workload',row['workload'],'--policy',row['policy'],'--active-s',str(c['active_s']),'--recovery-s',str(c['recovery_s']),'--weight-model',c['weight_model'],'--scan-period-s',str(lock['scan_period_s']),'--maintenance','--maintenance-due-s',str(lock['maintenance']['due_s']),'--campaign-lock',str(lock_path)] + begin=time.monotonic() + with (a.stage/(row['id']+'-driver.log')).open('w') as f:r=subprocess.run(cmd,cwd=root,stdout=f,stderr=subprocess.STDOUT) + return dict(row,status='COMPLETED' if r.returncode==0 else 'FAILED',exit_code=r.returncode,wall_s=time.monotonic()-begin,point=str(out)) +# Each wave has same workload/topology/scenario,3policies, independent singletons. +for start in range(0,len(queue),3): + wave=queue[start:start+3] + with concurrent.futures.ThreadPoolExecutor(max_workers=3) as pool: + for answer in pool.map(run_point,wave): + status.append(answer) + (a.stage/'MAIN_STATUS_v1.json').write_text(json.dumps(dict(elapsed_s=time.monotonic()-started,points=status),indent=2)) + print(json.dumps({k:answer[k] for k in ('id','status','wall_s') if k in answer}),flush=True) + if any(r['status']!='COMPLETED' for r in status[-3:]): + (a.stage/'MAIN_PAUSED_v1.json').write_text(json.dumps(dict(reason='wave failure; requires causal review, independent work continues',remaining=[r['id'] for r in queue[start+3:]]),indent=2));break + if sum(p.stat().st_size for p in (a.stage/'points').rglob('*') if p.is_file() and p.parts[-2].startswith('MAIN-'))>32*1024**3: + raise RuntimeError('campaign output budget reached') +else:(a.stage/'MAIN_DONE_v1.json').write_text(json.dumps(dict(completed=len(status),wall_s=time.monotonic()-started),indent=2)) diff --git a/experiments/eq3_maintenance/run_frozen_queue.py b/experiments/eq3_maintenance/run_frozen_queue.py new file mode 100644 index 0000000..b38120a --- /dev/null +++ b/experiments/eq3_maintenance/run_frozen_queue.py @@ -0,0 +1,24 @@ +#!/usr/bin/env python3 +"""Run a frozen explicit queue; isolated engines, finite resource guards, no sweep inference.""" +import argparse,concurrent.futures,hashlib,json,os,pathlib,shutil,subprocess,time +p=argparse.ArgumentParser();p.add_argument('--queue',type=pathlib.Path,required=True);p.add_argument('--root',type=pathlib.Path,required=True);a=p.parse_args();a.queue=a.queue.resolve();a.root=a.root.resolve();stage=a.queue.parent;queue=json.loads(a.queue.read_text());lock=json.loads((stage/'ENGINEERING_LOCK.json').read_text());began=time.monotonic();status=[] +if (stage/'STATUS.json').exists():raise RuntimeError('create-only execution; explicit reviewed resume/new IDs required') +def run(row): + target=stage/'points'/row['id'];cmd=row['command'];t=time.monotonic() + with (stage/(row['id']+'-driver.log')).open('w') as f:proc=subprocess.run(cmd,cwd=a.root,stdout=f,stderr=subprocess.STDOUT) + state='COMPLETED' if proc.returncode==0 else 'FAILED' + if (target/'NOT_STARTED.json').exists():state='NOT_STARTED' + return dict(row,status=state,exit_code=proc.returncode,wall_s=time.monotonic()-t,point=str(target)) +for wave in queue['waves']: + if time.monotonic()-began>lock['resources']['campaign_wall_s']:raise RuntimeError('bounded campaign wall reached before next wave') + for relative,expected in lock['consumer_sha256'].items(): + if hashlib.sha256((a.root/relative).read_bytes()).hexdigest()!=expected:raise RuntimeError('frozen consumer changed: '+relative) + if shutil.disk_usage(stage).free<100*1024**3:raise RuntimeError('host disk reserve before wave') + if sum(q.stat().st_size for q in (stage/'points').rglob('*') if q.is_file())>32*1024**3:raise RuntimeError('own retained campaign budget reached') + with concurrent.futures.ThreadPoolExecutor(max_workers=wave['concurrency']) as pool: + for item in pool.map(run,wave['points']): + status.append(item);print(json.dumps({k:item[k] for k in ('id','status','wall_s')}),flush=True) + (stage/'STATUS.json').write_text(json.dumps(dict(elapsed_s=time.monotonic()-began,points=status),indent=2)) + if any(v['status']!='COMPLETED' for v in status[-len(wave['points']):]): + (stage/'PAUSED.json').write_text(json.dumps(dict(reason='actual failure; preserve and diagnose before affected continuation',wave=wave['id'],completed=len([x for x in status if x['status']=='COMPLETED'])),indent=2));break +else:(stage/'DONE.json').write_text(json.dumps(dict(completed=len(status),wall_s=time.monotonic()-began),indent=2)) diff --git a/experiments/eq3_maintenance/run_point.py b/experiments/eq3_maintenance/run_point.py new file mode 100644 index 0000000..0c1bfd1 --- /dev/null +++ b/experiments/eq3_maintenance/run_point.py @@ -0,0 +1,334 @@ +#!/usr/bin/env python3 +"""Explicit CPU-only experiment entry. No default-backend fallback or auto sweep.""" +import argparse +from dataclasses import asdict +import csv +import hashlib +import json +import os +from pathlib import Path +import resource +import shutil +import subprocess +import sys +import time + +HERE=Path(__file__).resolve().parent +ROOT=HERE.parents[1] +for p in (ROOT,ROOT/'tools',HERE,HERE/'backend/client'): + sys.path.insert(0,str(p)) +from campaign_inputs import configuration,workload,energy_profile +from closed_loop import ClosedLoopCoordinator +from energy_ledger import ActivityEnergyLedger +from thermal_client import ThermalService +from read_rate_policy import EngineeringProfile,ReadRatePolicy +from maintenance_service import MaintenanceMqsimService +from ideal_maintenance_replay import IdealIndependentMaintenanceReplay +from weight_workloads import weight_extent,generate_weight_requests +from eq3_basic_hbm import BasicHbm +from incremental_fabric import IncrementalBasicFabric as BasicFabric + + +FROZEN_INPUT_FILES=('configuration.json','profile.json','stack-map.json','requests-input.json', + 'maintenance-input.json','energy-profile.json','policy-profile.json', + 'weight-model-extent.json','weight-workload.json') + + +def workload_burst_period_ns(kind): + if kind=='W1':return 20000000 + if kind=='W2':return 100000000 + raise ValueError('unknown workload') + + +def write_json(path,value): + path.write_text(json.dumps(value,indent=2,allow_nan=False)+'\n') + + +def write_csv(path,rows): + keys=list(dict.fromkeys(k for row in rows for k in row)) + with path.open('w',newline='') as f: + writer=csv.DictWriter(f,fieldnames=keys);writer.writeheader() + for row in rows: + writer.writerow({k:json.dumps(v,sort_keys=True) if isinstance(v,(dict,list,tuple)) else v + for k,v in row.items()}) + + +def load_frozen_inputs(source,args): + source=source.resolve(strict=True) + manifest=json.loads((source/'manifest.json').read_text()) + expected=dict(mode=args.mode,workload=args.workload,policy=args.policy, + active_ns=round(args.active_s*1e9), + end_ns=round((args.active_s+args.recovery_s)*1e9), + weight_model=args.weight_model) + mismatch={key:(manifest.get(key),value) for key,value in expected.items() + if manifest.get(key)!=value} + if mismatch:raise ValueError(f'frozen input CLI differs from source manifest: {mismatch}') + if args.geometry!='legacy16k' or args.capacity_scope!='working-region' or not args.maintenance: + raise ValueError('PILOT03 frozen replay requires legacy16k working-region maintenance mode') + missing=[name for name in FROZEN_INPUT_FILES if not (source/name).is_file()] + if missing:raise ValueError(f'frozen input files missing: {missing}') + values={name:json.loads((source/name).read_text()) for name in FROZEN_INPUT_FILES} + if args.gpu_external_w != values['energy-profile.json']['gpu_external_w']: + raise ValueError('frozen replay GPU external power differs from source input') + return source,manifest,values + + +def run(args): + output=args.output.resolve();output.mkdir(parents=True,exist_ok=False) + started=time.monotonic() + if bool(args.ideal_maintenance_bundle) != bool(args.frozen_inputs): + raise ValueError('ideal maintenance bundle and frozen inputs must be enabled together') + frozen_source=frozen_manifest=frozen_values=None + if args.frozen_inputs: + frozen_source,frozen_manifest,frozen_values=load_frozen_inputs(args.frozen_inputs,args) + if args.campaign_lock and (args.campaign_lock.parent/"CAMPAIGN_PAUSE.json").exists(): + write_json(output/"NOT_STARTED.json",dict(execution_status="NOT_STARTED",reason=json.loads((args.campaign_lock.parent/"CAMPAIGN_PAUSE.json").read_text()))) + raise SystemExit(75) + n=8 if args.mode=='all_hbf_direct' else 4 + page_bytes=4096 if args.geometry=='ocp4k16bank' else 16384 + alignment=65536 if args.geometry=='ocp4k16bank' else 4096 + weight=weight_extent(args.weight_model,stacks=n,page_bytes=page_bytes) if args.weight_model else None + capacity_extent=weight_extent(args.capacity_model,stacks=n,page_bytes=page_bytes) if weight else None + pages_per_stack=max(alignment*8,((capacity_extent['global_page_count']+n-1)//n+alignment-1)//alignment*alignment) if weight else alignment*16 + if args.capacity_scope=='full-product':pages_per_stack=512*1024**3//page_bytes + if weight and weight['global_page_count']>pages_per_stack*n:raise ValueError('weight model exceeds fixed modeled capacity') + if args.scan_period_s is not None: + if not weight or args.scan_period_s<=0:raise ValueError('scan period requires model and positive seconds') + args.total_hbf_rps=max(1,int(weight['global_page_count']/args.scan_period_s)) + cfg=configuration(args.mode,pages_per_stack=pages_per_stack,geometry=args.geometry) + active_ns=round(args.active_s*1e9);end_ns=active_ns+round(args.recovery_s*1e9) + window_ns=20000000 + if active_ns%window_ns or end_ns%window_ns: + raise ValueError('duration must align to thermal20ms') + burst_period_ns=workload_burst_period_ns(args.workload) + requests=workload(args.mode,args.workload,total_hbf_rps=args.total_hbf_rps, + active_ns=active_ns,burst_period_ns=burst_period_ns) + weight_input=None + if weight: + region='model.layers.9' if args.workload=='W2' else None + start_page=next(r['first_global_page'] for r in weight['regions'] if r['id'].startswith(region+'.')) if region else 0 + weight_input=generate_weight_requests(args.weight_model,args.total_hbf_rps*active_ns//1000000000,start_page, + max(1,1000000000//args.total_hbf_rps),stacks=n,region=region,page_bytes=page_bytes) + hbf=[] + for row in weight_input['requests']: + route='relay' if args.mode=='relay' or args.mode=='dash' and row['local_page']%2 else 'direct' + hbf.append(dict(row,request_id='weight-'+str(row['request_id']),stack_local_page=row['local_page'], + route=route,arrival_ns=row['arrival_ns']//burst_period_ns*burst_period_ns)) + requests=sorted(hbf+[r for r in requests if r['stack'].startswith('hbm')],key=lambda r:(r['arrival_ns'],r['request_id'])) + if frozen_values: + cfg=frozen_values['configuration.json'] + requests=frozen_values['requests-input.json'] + weight=frozen_values['weight-model-extent.json'] + weight_input=frozen_values['weight-workload.json'] + n=len(cfg['fabric']['hbf']) + page_bytes=frozen_values['profile.json']['page_bytes'] + args.total_hbf_rps=frozen_manifest['total_hbf_rps'] + coefficient=energy_profile() + coefficient['gpu_external_w']=args.gpu_external_w + if frozen_values:coefficient=frozen_values['energy-profile.json'] + # Same GPU load in every arm: demand scenarios never modify an operation coefficient. + stacks=list(cfg['fabric']['hbf'])+list(cfg['fabric']['hbm']) + budget=16384*2048 # Allows102400 pages/s/stack, limited further by actual service. + budgets={s:budget for s in stacks} + endpoint_caps={f'{s}:gpu-link':budget for s in stacks} + for stack,item in cfg['fabric']['hbf'].items(): + if item['pair']: + endpoint_caps[f'{stack}->{item["pair"]}:relay-link']=budget + policy_profile=EngineeringProfile(profile_id='eq3-pilot-v1',enabled=True,strategy=args.policy, + window_ns=window_ns,target_bytes_per_s=max(1,args.total_hbf_rps//len(cfg['fabric']['hbf']))*page_bytes, + target_latency_p95_ns=window_ns,step_bytes=page_bytes,minimum_budget_bytes=page_bytes, + maximum_budget_bytes=budget,severe_budget_bytes=0) + if frozen_values:policy_profile=EngineeringProfile(**frozen_values['policy-profile.json']) + maintenance=[] + if args.maintenance: + due=round(args.maintenance_due_s*1e9) if args.maintenance_due_s is not None else active_ns//2//window_ns*window_ns + if due>=end_ns or due<0:raise ValueError('maintenance due outside experiment') + first_global=weight_input['selected_global_page_range'][0] if weight_input else 0 + for offset in range(64): + page=first_global+offset + maintenance.append(dict(request_id=1000000+offset,stack=f'hbf{page%n}', + stack_local_page=page//n,due_ns=due,deadline_ns=due+window_ns, + initial_age_s=86400-due*1e-9,bytes=page_bytes,reclaim_source_block=True,trigger_reason='RETENTION_AGE_DUE')) + if frozen_values:maintenance=frozen_values['maintenance-input.json'] + meminfo=dict(line.split(':',1) for line in Path('/proc/meminfo').read_text().splitlines()) + available=int(meminfo['MemAvailable'].split()[0])*1024 + free=shutil.disk_usage(output).free + expected_peak_gib=(n*5.5+6) if args.capacity_scope=='full-product' else 4 + if available<(expected_peak_gib+32)*1024**3 or free<100*1024**3: + raise RuntimeError('host reserve insufficient; no backend started') + source=subprocess.check_output(['git','-C',str(ROOT),'rev-parse','HEAD'],text=True).strip() + meta=dict(task='EQ3-ISOLATED-MAINTENANCE-CAMPAIGN-v1',kind='PILOT_CONDITIONAL_ENGINEERING_USE', + environment_id='eq3-thermal-cpu-v1',source_head=source, + source_status=subprocess.check_output(['git','-C',str(ROOT),'status','--porcelain'],text=True), + mode=args.mode,workload=args.workload,policy=args.policy,active_ns=active_ns,end_ns=end_ns, + workload_shape=dict(burst_period_ns=(None if frozen_values else burst_period_ns), + evidence=('FROZEN_SOURCE_INPUT' if frozen_values else + ('W2_100MS_PHASE_BURSTS' if args.workload=='W2' else 'W1_20MS_BURSTS'))), + gpu_external_w=coefficient['gpu_external_w'],gpu_external_w_evidence='EXPLICIT_OAT_ENGINEERING_INPUT', + total_hbf_rps=args.total_hbf_rps,request_count=len(requests),maintenance_count=len(maintenance), + thermal_model_dir=str(args.model_dir.resolve()),binary=str(args.binary.resolve()), + thermal_binary=str(args.thermal_binary.resolve()),observed_memory_available_bytes=available, + observed_disk_available_bytes=free,geometry=args.geometry,capacity_scope=args.capacity_scope,page_bytes=page_bytes,limits=dict(process_address_gib=args.address_limit_gib,cpu=1,gpu=0, + wall_s=600,expected_output_gib=1,host_memory_reserve_gib=32,host_disk_reserve_gib=100), + resource_reason=('Full logical capacity4K probe measured20.52GiB for4stacks; Q4 projected41.1GiB; thermal<0.4GiB, oneCPU per point.' if args.capacity_scope=='full-product' else 'Finite working region medium-event pilot; complete-model thermal factorRSS<0.4GiB; owned backend and thermal processes.'), + scientific_scope='Finite observed weight working set inside explicitly declared NAND logical namespace, true16die MQSim topology, complete2mm thermal model; no product timing/physical spare/energy calibration or token causality') + if frozen_values: + meta.update(kind='IDEAL_INDEPENDENT_MAINTENANCE_REPLAY', + evidence='FIXED_COUNTERFACTUAL_REPLAY_NOT_ACTUAL_SHARED_MQSIM', + frozen_source_point=str(frozen_source),frozen_source_manifest_sha256=hashlib.sha256((frozen_source/'manifest.json').read_bytes()).hexdigest(), + frozen_source_head=frozen_manifest.get('source_head'), + ideal_maintenance_bundle=str(args.ideal_maintenance_bundle.resolve(strict=True)), + ideal_maintenance_bundle_sha256=hashlib.sha256(args.ideal_maintenance_bundle.read_bytes()).hexdigest(), + actual_backend_maintenance='DISABLED_BY_WRAPPER',mapping_semantics='UNKNOWN_REPLAY_CURRENT_ENGINE_UNCHANGED', + virtual_age_semantics='FROZEN_MAPPING_COMMIT_FACT_REPLAY_ONLY') + if args.campaign_lock: + lock=args.campaign_lock.resolve(strict=True) + meta['kind']='FROZEN_CONDITIONAL_ENGINEERING_CAMPAIGN' + meta['campaign_lock']=str(lock) + meta['campaign_lock_sha256']=hashlib.sha256(lock.read_bytes()).hexdigest() + inputs=[('configuration.json',cfg),('profile.json',cfg['profile']),('stack-map.json',cfg['stack_map']), + ('requests-input.json',requests),('maintenance-input.json',maintenance), + ('energy-profile.json',coefficient),('policy-profile.json',asdict(policy_profile))] + for name,obj in inputs: + if frozen_values:shutil.copyfile(frozen_source/name,output/name) + else:write_json(output/name,obj) + if weight_input: + if frozen_values: + shutil.copyfile(frozen_source/'weight-model-extent.json',output/'weight-model-extent.json') + shutil.copyfile(frozen_source/'weight-workload.json',output/'weight-workload.json') + else: + write_json(output/'weight-model-extent.json',weight) + write_json(output/'weight-workload.json',weight_input) + meta['weight_model']=args.weight_model + meta['capacity_model']=args.capacity_model + meta['scan_period_s']=args.scan_period_s + meta['requested_weight_bytes_per_s']=weight['tensor_payload_bytes']/args.scan_period_s if args.scan_period_s else None + meta['effective_page_scan_period_s']=weight['global_page_count']/args.total_hbf_rps + meta['read_only_weight_workload']=True + meta['weight_coverage']=weight_input['coverage'] + meta['executable_sha256']={str(p.resolve()):hashlib.sha256(p.read_bytes()).hexdigest() for p in (args.binary,args.thermal_binary)} + meta['input_sha256']={p.name:hashlib.sha256(p.read_bytes()).hexdigest() for p in output.glob('*.json')} + write_json(output/'manifest.json',meta) + resource.setrlimit(resource.RLIMIT_AS,(args.address_limit_gib*1024**3,args.address_limit_gib*1024**3)) + resource.setrlimit(resource.RLIMIT_CORE,(0,0)) + os.sched_setaffinity(0,{min(os.sched_getaffinity(0))}) + for key in ('OMP_NUM_THREADS','OPENBLAS_NUM_THREADS','MKL_NUM_THREADS','NUMEXPR_NUM_THREADS'): + os.environ[key]='1' + os.environ['CUDA_VISIBLE_DEVICES']='' + normalized=json.loads((args.model_dir/'normalized.json').read_text()) + components=[c['id'] for c in normalized['components']] + hbm_dies={stack:[c['id'] for c in normalized['components'] + if c['device_id']==stack and c['role']=='array_die'] for stack in cfg['fabric']['hbm']} + energy=ActivityEnergyLedger(coefficient,cfg['stack_map'],components,hbm_dies=hbm_dies,gpu_stop_ns=active_ns) + fabric=BasicFabric(cfg['fabric']);hbm=BasicHbm(cfg['hbm']) if cfg['hbm'] else None + def resource_probe(start,end): + state=fabric.resource_state();answer={} + for stack in stacks: + rows=[r for r in energy.rows if start<=r['start_ns'] \ + -DEQ3_BACKEND_BUILD= +cmake --build --parallel 1 +``` + +The output directory is create-only. Use a frozen profile and stack map, bind a +source label and SHA-256 in the outer preflight, and retain the emitted CSV and +JSON files as raw evidence. + +`--outstanding-window N` is a rolling window: every returned completion +immediately permits one refill until all writes or reads have been issued. +`--verify-pages N` defaults to all loaded pages. A smaller value chooses an +evenly stratified read sample containing the first and last loaded page (one +sample selects the last page); the receipt reports the exact coverage. Sampling +reads never changes the number of actual startup programs. diff --git a/experiments/eq3_maintenance/startup/analyze_concurrency.py b/experiments/eq3_maintenance/startup/analyze_concurrency.py new file mode 100644 index 0000000..0bc1dab --- /dev/null +++ b/experiments/eq3_maintenance/startup/analyze_concurrency.py @@ -0,0 +1,90 @@ +#!/usr/bin/env python3 +"""Verify bounded-window write concurrency from immutable probe CSV facts.""" +import argparse,csv,json +from collections import defaultdict +from pathlib import Path + +def analyze(raw: Path): + with (raw/'requests.csv').open() as stream: requests=list(csv.DictReader(stream)) + with (raw/'request-observations.csv').open() as stream: observations=list(csv.DictReader(stream)) + with (raw/'native-events.csv').open() as stream: events=list(csv.DictReader(stream)) + writes={int(r['request_id']) for r in requests if r['phase']=='STARTUP_WRITE'} + write_rows=[r for r in requests if r['phase']=='STARTUP_WRITE'] + by_command=defaultdict(dict) + resources=defaultdict(set) + transaction_ids=set() + for row in events: + rid=int(row['external_request_id']) + if rid not in writes or int(row['maintenance_request_id']) or int(row['type'])!=1: + continue + phase=int(row['phase']);command=int(row['command_id']) + if phase in (1,2): + if phase in by_command[command] and by_command[command][phase]!=int(row['time_ns']): + raise ValueError('one command has inconsistent media phase times') + by_command[command][phase]=int(row['time_ns']) + resources[command].add((int(row['channel']),int(row['chip']),int(row['die']),int(row['plane']))) + transaction_ids.add(int(row['transaction_id'])) + intervals=[] + for command,phase in by_command.items(): + if set(phase)!={1,2} or phase[2]1 and max_active(lambda r:r[:3])>1, + 'count_semantics':'operations=unique transaction IDs; commands=unique command IDs; CSV rows repeat identities across phases', + 'interval_semantics':'half-open MEDIA_BEGIN to MEDIA_END; endings processed before starts at equal timestamps'} + if len(writes)!=len(transaction_ids) or not intervals:raise ValueError('program operation conservation failed') + return result + +def main(): + p=argparse.ArgumentParser();p.add_argument('--raw',type=Path,required=True);p.add_argument('--output',type=Path,required=True);a=p.parse_args() + if a.output.exists():raise FileExistsError(a.output) + result=analyze(a.raw);a.output.write_text(json.dumps(result,indent=2,sort_keys=True)+'\n');print(json.dumps(result,sort_keys=True)) +if __name__=='__main__':main() diff --git a/experiments/eq3_maintenance/startup/replay_native_heat.py b/experiments/eq3_maintenance/startup/replay_native_heat.py new file mode 100644 index 0000000..deb229d --- /dev/null +++ b/experiments/eq3_maintenance/startup/replay_native_heat.py @@ -0,0 +1,143 @@ +#!/usr/bin/env python3 +"""Replay immutable startup native facts through unchanged energy/full thermal APIs. + +This is a source-only open-loop diagnostic, not controller reclosure or a +complete host-to-HBF upload transport model. No MQSim engine is run here. +""" +import argparse +import csv +from collections import defaultdict +import hashlib +import json +import math +import platform +import shutil +from datetime import datetime, timezone +import os +from pathlib import Path +import resource +import sys +import time + +sys.path.insert(0, str(Path(__file__).resolve().parents[1])) +from campaign_inputs import energy_profile +from energy_ledger import ActivityEnergyLedger +from thermal_client import ThermalService + + +def replay(args): + out = args.output.resolve(); out.mkdir(parents=True, exist_ok=False) + native = args.native.resolve(strict=True) + summary = json.loads(args.summary.read_text()) + mapping = json.loads(args.stack_map.read_text()) + coefficient = energy_profile(); coefficient['gpu_external_w'] = 0 + coefficient['gpu_note'] = 'SOURCE_ONLY_UPLOAD_DIAGNOSTIC; idle GPU/host ingress not modelled' + normalized = json.loads((args.model_dir/'normalized.json').read_text()) + components = [c['id'] for c in normalized['components']] + model_files={name:hashlib.sha256((args.model_dir/name).read_bytes()).hexdigest() + for name in ('model.txt','rc_grid.json','rc_sensors.json','normalized.json')} + manifest = dict(kind='NATIVE_STARTUP_OPEN_LOOP_THERMAL_REPLAY', + started_utc=datetime.now(timezone.utc).isoformat(),python=sys.version,platform=platform.platform(), + address_limit_gib=4,threads=1,gpu_count=0,disk_free_bytes=shutil.disk_usage(out).free, + source_native=str(native), source_summary_path=str(args.summary.resolve()), + native_sha256=hashlib.sha256(native.read_bytes()).hexdigest(), + source_summary_sha256=hashlib.sha256(args.summary.read_bytes()).hexdigest(), + stack_map_sha256=hashlib.sha256(args.stack_map.read_bytes()).hexdigest(), + thermal_binary_sha256=hashlib.sha256(args.thermal_binary.read_bytes()).hexdigest(), + model_dir=str(args.model_dir.resolve()),model_file_sha256=model_files, + source_summary=summary, energy_profile=coefficient, + step_ns=20_000_000, recovery_ns=args.recovery_ns, + no_mqsim_rerun=True, no_controller_reclosure=True, + host_ingress_energy='UNAVAILABLE', background_idle_power='NOT_MODELLED', + physical_qualification='CONDITIONAL_ENGINEERING_USE_NOT_MODEL_FREEZE') + (out/'manifest.json').write_text(json.dumps(manifest, indent=2)) + for key in ('OMP_NUM_THREADS','OPENBLAS_NUM_THREADS','MKL_NUM_THREADS','NUMEXPR_NUM_THREADS'): os.environ[key]='1' + os.environ['CUDA_VISIBLE_DEVICES']='' + resource.setrlimit(resource.RLIMIT_AS, (4*1024**3,4*1024**3)) + resource.setrlimit(resource.RLIMIT_CORE, (0,0)) + window = 20_000_000 + ledger = ActivityEnergyLedger(coefficient, mapping, components, gpu_stop_ns=0) + began = time.monotonic(); native_count=0; previous=-1; peaks={}; worst_residual=0 + energy_totals=[]; last_frame=None + with (out/'thermal.jsonl').open('w') as frames: + with ThermalService(args.thermal_binary,args.model_dir,out/'thermal-process', + artifact_root=args.artifact_root) as thermal: + def flush(): + nonlocal last_frame, worst_residual + start=ledger.now; totals=ledger.flush(start+window) + last_frame=thermal.advance(start,start+window,totals) + frames.write(json.dumps(last_frame)+'\n') + for entity,value in last_frame['temperatures'].items(): + peaks[entity]=max(peaks.get(entity,-math.inf),value) + cumulative=last_frame.get('energy_j',{}).get('cumulative',{}) + input_j=cumulative.get('total_input_j',0) + residual=cumulative.get('energy_residual_j') + if residual is not None and input_j:worst_residual=max(worst_residual,abs(residual/input_j)) + energy_totals.append(dict(start_ns=start,end_ns=start+window,component_energy_j=totals)) + with native.open() as stream: + for line in stream: + event=json.loads(line); stamp=event['time_ns'] + if stamp=ledger.now+window:flush() + ledger.native(event); native_count+=1; previous=stamp + if ledger.active:raise ValueError('native media/transfer has no end; do not fabricate completion energy') + stop=((max(previous,0)+window-1)//window)*window+args.recovery_ns + while ledger.now1e-10*max(1.0,ledger.total_j): + raise ValueError('stage energy must conserve the observed source ledger') + window_energy=sum(sum(item['component_energy_j'].values()) for item in energy_totals) + row_energy=sum(row['energy_j'] for row in ledger.rows) + thermal_input=(last_frame or {}).get('energy_j',{}).get('cumulative',{}).get('total_input_j') + checks={ + 'ledger_vs_activity_rows_abs_j':abs(ledger.total_j-row_energy), + 'ledger_vs_thermal_windows_abs_j':abs(ledger.total_j-window_energy), + 'ledger_vs_stage_split_abs_j':abs(ledger.total_j-sum(stage_energy.values())), + 'ledger_vs_source_split_abs_j':abs(ledger.total_j-sum(source_energy.values())), + 'ledger_vs_scope_split_abs_j':abs(ledger.total_j-sum(scope_energy.values())), + 'ledger_vs_thermal_service_input_abs_j':None if thermal_input is None else abs(ledger.total_j-thermal_input), + } + tolerance=1e-10*max(1.0,ledger.total_j) + if any(value is None or value>tolerance for value in checks.values()): + raise ValueError('energy conservation failed across source, stage, window, or thermal receipt') + result=dict(execution_status='COMPLETED',capability_status='BACKEND_NATIVE_FACTS_PLUS_OPEN_LOOP_THERMAL', + native_event_count=native_count,simulated_end_ns=ledger.now,source_only_energy_j=ledger.total_j, + source_energy_j_by_stage=stage_energy, + source_energy_j_by_native_source=dict(sorted(source_energy.items())), + source_energy_j_by_scope=dict(sorted(scope_energy.items())), + energy_conservation_abs_j=checks,energy_conservation_tolerance_j=tolerance, + peak_k_by_entity=peaks,peak_k=max(peaks.values()),energy_relative_residual_max=worst_residual, + wall_s=time.monotonic()-began,child_peak_rss_kib=resource.getrusage(resource.RUSAGE_CHILDREN).ru_maxrss, + sub_20ms_peak='UNRESOLVED_BY_FIXED_MACROSTEP',weight_upload_coverage='SEE_SOURCE_SUMMARY_NOT_FULL_MODEL', + physical_status='CONDITIONAL_ENGINEERING_USE',controller_status='NOT_RECLOSED', + host_ingress_energy='UNAVAILABLE',fabric_upload_transport='NOT_MODELLED') + (out/'DONE.json').write_text(json.dumps(result,indent=2)) + print(json.dumps(result)) + + +if __name__=='__main__': + p=argparse.ArgumentParser();p.add_argument('--native',type=Path,required=True);p.add_argument('--summary',type=Path,required=True) + p.add_argument('--stack-map',type=Path,required=True);p.add_argument('--model-dir',type=Path,required=True) + p.add_argument('--thermal-binary',type=Path,required=True);p.add_argument('--artifact-root',type=Path,required=True) + p.add_argument('--output',type=Path,required=True);p.add_argument('--recovery-ns',type=int,default=200_000_000) + a=p.parse_args() + if a.recovery_ns<0 or a.recovery_ns%20_000_000:raise ValueError('recovery must be whole20ms windows') + try:replay(a) + except BaseException as error: + if a.output.exists():(a.output/'FAILED.json').write_text(json.dumps(dict(type=type(error).__name__,error=str(error)),indent=2)) + raise diff --git a/experiments/eq3_maintenance/startup/startup_write_probe.cpp b/experiments/eq3_maintenance/startup/startup_write_probe.cpp new file mode 100644 index 0000000..ae85d11 --- /dev/null +++ b/experiments/eq3_maintenance/startup/startup_write_probe.cpp @@ -0,0 +1,272 @@ +#include +#include +#include +#include +#include +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +namespace { +using Json=nlohmann::json; +namespace fs=std::filesystem; + +struct Options { + fs::path profile,stack_map,output; + std::uint64_t pages{},verify_pages{},maintenance_pages{}; + std::size_t outstanding_window{256}; + std::string source_label,source_sha256; +}; + +std::uint64_t u64(std::string_view text,const char* name) { + if(text.empty()||text.find_first_not_of("0123456789")!=std::string_view::npos) + throw std::invalid_argument(std::string("invalid ")+name); + return std::stoull(std::string(text)); +} +Options parse(int argc,char** argv) { + Options o; + for(int i=1;i=argc)throw std::invalid_argument("missing option value"); + const std::string k=argv[i],v=argv[i+1]; + if(k=="--profile")o.profile=v; else if(k=="--stack-map")o.stack_map=v; + else if(k=="--output-dir")o.output=v; else if(k=="--pages")o.pages=u64(v,"pages"); + else if(k=="--verify-pages")o.verify_pages=u64(v,"verify-pages"); + else if(k=="--maintenance-pages")o.maintenance_pages=u64(v,"maintenance-pages"); + else if(k=="--outstanding-window"||k=="--batch-size")o.outstanding_window=static_cast(u64(v,"outstanding-window")); + else if(k=="--source-label")o.source_label=v; else if(k=="--source-sha256")o.source_sha256=v; + else throw std::invalid_argument("unknown option: "+k); + } + if(!o.verify_pages)o.verify_pages=o.pages; + if(o.profile.empty()||o.stack_map.empty()||o.output.empty()||!o.pages||!o.outstanding_window|| + o.verify_pages>o.pages||o.maintenance_pages>o.pages||o.source_label.empty()||o.source_sha256.size()!=64|| + o.source_sha256.find_first_not_of("0123456789abcdef")!=std::string::npos) + throw std::invalid_argument("required: --profile --stack-map --output-dir --pages N " + "[--verify-pages N] --maintenance-pages N --outstanding-window N --source-label TEXT --source-sha256 64-lower-hex"); + return o; +} + +Json read_json(const fs::path& path) { + std::ifstream in(path);if(!in)throw std::runtime_error("cannot open "+path.string()); + return Json::parse(in); +} +std::uint64_t integer(const Json& v,const char* field) { + if(v.is_number_unsigned())return v.get(); + if(!v.is_number_integer()||v.get()<0)throw std::invalid_argument(std::string("invalid ")+field); + return static_cast(v.get()); +} + +struct Mapping { + hbfsim::eq3_thermal::MqsimStackMapAdapter adapter; + std::vector stacks; +}; +Mapping load_map(const fs::path& path,const hbfsim::Profile& profile) { + const auto j=read_json(path); + if(integer(j.at("schema_version"),"schema_version")!=1||j.at("physical_kind")!="HBF"|| + j.at("route")!="direct"||j.at("address_layout")!="GLOBAL_PAGE_STRIPE_V1"|| + integer(j.at("page_bytes"),"page_bytes")!=profile.page_bytes|| + integer(j.at("channels"),"channels")!=profile.channels|| + integer(j.at("dies_per_channel"),"dies_per_channel")!=profile.dies_per_channel) + throw std::invalid_argument("stack map/profile identity mismatch"); + std::vector groups; + std::vector names; + for(const auto& row:j.at("stacks")) { + hbfsim::eq3_thermal::MqsimStackChannelGroup g; + g.stack_id=row.at("id").get();names.push_back(g.stack_id); + g.declared_dies=static_cast(integer(row.at("declared_dies"),"declared_dies")); + for(const auto& c:row.at("channels"))g.channels.push_back(static_cast(integer(c,"channel"))); + groups.push_back(std::move(g)); + } + return {hbfsim::eq3_thermal::MqsimStackMapAdapter(profile,std::move(groups)),std::move(names)}; +} + +std::string csv(std::string value) { + std::string out="\"";for(char c:value){if(c=='\"')out+="\"\"";else out+=c;}return out+="\""; +} +const char* maintenance_status(hbfsim::MqsimMaintenanceStatus s) { + using S=hbfsim::MqsimMaintenanceStatus; + switch(s) { + case S::Committed:return "COMMITTED";case S::CommittedReclaimDeferred:return "COMMITTED_RECLAIM_DEFERRED"; + case S::RejectedUnsupported:return "REJECTED_UNSUPPORTED";case S::RejectedInvalidTarget:return "REJECTED_INVALID_TARGET"; + case S::RejectedUnmapped:return "REJECTED_UNMAPPED";case S::RejectedNoSpare:return "REJECTED_NO_SPARE"; + case S::FailedRead:return "FAILED_READ";case S::FailedProgram:return "FAILED_PROGRAM"; + case S::FailedStaleVersion:return "FAILED_STALE_VERSION"; + case S::FailedAfterCommitNeedsReconcile:return "FAILED_AFTER_COMMIT_NEEDS_RECONCILE"; + case S::RejectedSourceBusy:return "REJECTED_SOURCE_BUSY";case S::FailedVersionOverflow:return "FAILED_VERSION_OVERFLOW"; + } throw std::logic_error("unknown maintenance status"); +} + +struct RequestRecord { + std::uint64_t id{},external_page{},backend_page{},arrival{},completion{}; + std::string phase,stack;std::uint32_t expected_channel{}; +}; + +void publish(const fs::path& path,const Json& value) { + std::ofstream out(path,std::ios::out|std::ios::trunc);if(!out)throw std::runtime_error("cannot create "+path.string()); + out<profile.capacity_bytes/profile.page_bytes)throw std::out_of_range("page count exceeds profile capacity"); + if(hbfsim::blocks_per_plane(profile)<=10) + throw std::invalid_argument("write diagnostic requires more than 10 blocks per plane for MQSim spare/frontier safety"); + std::vector native; + SSD_Components::NVM_PHY_ONFI_NVDDR2::Set_hbf_command_observation_sink( + [&native](const auto& event){native.push_back(event);}); + std::vector records;records.reserve(static_cast(o.pages+o.verify_pages)); + std::map placements; + std::vector observations; + std::vector maintenance_events; + std::vector maintenance_completions; + std::uint64_t write_end=0,read_end=0,maintenance_end=0; + { + hbfsim::MqsimOnlineEngine engine(profile);engine.enable_observations(); + const auto run_phase=[&](std::string phase,std::uint64_t id_base,hbfsim::RequestOperation operation, + const std::vector& pages) { + std::size_t next=0;std::map pending; + const auto refill=[&]() { + while(next(operation)}; + auto p=mapping.adapter.map_stack_page(external,stack,page/mapping.stacks.size()); + if(!p.expected_channel)throw std::logic_error("mapped request lacks expected channel"); + if(placements.contains(id))throw std::logic_error("duplicate request identity"); + placements.emplace(id,p);engine.submit(p.backend_request); + records.push_back({id,p.external_page,p.backend_page,p.backend_request.arrival_ns,0,phase,stack,*p.expected_channel}); + if(!pending.emplace(id,records.size()-1).second)throw std::logic_error("duplicate pending identity"); + } + }; + refill(); + while(!pending.empty()) { + auto c=engine.run_next_completion(); + if(!c)throw std::runtime_error("pending rolling window returned no completion"); + const auto found=pending.find(c->request_id); + if(found==pending.end())throw std::runtime_error( + "unknown/duplicate completion id="+std::to_string(c->request_id)+ + " pending_first="+(pending.empty()?std::string("NONE"):std::to_string(pending.begin()->first))); + records.at(found->second).completion=c->modeled_completion_ns;pending.erase(found); + auto batch=engine.take_observations();observations.insert(observations.end(),batch.begin(),batch.end()); + refill(); + } + }; + std::vector write_pages(static_cast(o.pages)); + for(std::uint64_t page=0;page verify_pages;verify_pages.reserve(static_cast(o.verify_pages)); + if(o.verify_pages==1)verify_pages.push_back(o.pages-1); + else for(std::uint64_t index=0;index((static_cast(index)*(o.pages-1))/(o.verify_pages-1))); + if(std::adjacent_find(verify_pages.begin(),verify_pages.end())!=verify_pages.end())throw std::logic_error("verification sample is not unique"); + run_phase("STARTUP_WRITE",1,hbfsim::RequestOperation::Write,write_pages); + write_end=engine.current_time_ns(); + run_phase("VERIFY_READ",1000000001ULL,hbfsim::RequestOperation::Read,verify_pages); + read_end=engine.current_time_ns(); + for(std::uint64_t page=0;page(backend%profile.channels); + const auto die=static_cast((backend/profile.channels)%profile.dies_per_channel); + const auto plane=static_cast((backend/(static_cast(profile.channels)*profile.dies_per_channel))%profile.planes_per_die); + engine.submit_maintenance({.request_id=2000000001ULL+page,.parent_id=3000000001ULL+page, + .due_ns=read_end,.deadline_ns=read_end+1000000000ULL,.logical_page=backend, + .channel=channel,.chip=0,.die=die,.plane=plane,.plane_is_exact=true}); + } + while(engine.pending_maintenance()) { + (void)engine.run_next_completion_until(engine.current_time_ns()+1000000000ULL); + auto e=engine.take_maintenance_events();maintenance_events.insert(maintenance_events.end(),e.begin(),e.end()); + auto c=engine.take_maintenance_completions();maintenance_completions.insert(maintenance_completions.end(),c.begin(),c.end()); + auto obs=engine.take_observations();observations.insert(observations.end(),obs.begin(),obs.end()); + } + auto e=engine.take_maintenance_events();maintenance_events.insert(maintenance_events.end(),e.begin(),e.end()); + auto c=engine.take_maintenance_completions();maintenance_completions.insert(maintenance_completions.end(),c.begin(),c.end()); + if(engine.pending())throw std::runtime_error("foreground work remains pending"); + maintenance_end=engine.current_time_ns(); + if(SSD_Components::NVM_PHY_ONFI_NVDDR2::Hbf_command_observation_failed())throw std::runtime_error("native observation sink failed"); + } + SSD_Components::NVM_PHY_ONFI_NVDDR2::Clear_hbf_command_observation_sink(); + if(records.size()!=o.pages+o.verify_pages||maintenance_completions.size()!=o.maintenance_pages) + throw std::runtime_error("final request/maintenance conservation failed"); + if(std::any_of(records.begin(),records.end(),[](const auto& r){return !r.completion;}))throw std::runtime_error("missing completion time"); + if(std::any_of(maintenance_completions.begin(),maintenance_completions.end(),[](const auto& c){return !c.mapping_committed;})) + throw std::runtime_error("maintenance did not commit every requested page"); + + std::ofstream req(o.output/"requests.csv");req<<"phase,request_id,stack,external_page,backend_page,expected_channel,arrival_ns,completion_ns,bytes\n"; + for(const auto& r:records)req<(e.kind)<<','<(e.phase)<<','<(e.command_code)<<','<(e.phase)}, + {"time_ns",e.time},{"command_code",static_cast(e.command_code)}, + {"transactions",std::move(transactions)}}.dump()<<'\n'; + } + std::ofstream m(o.output/"maintenance.csv");m<<"request_id,status,enqueue_ns,start_ns,end_ns,logical_page,source_version,committed_version,mapping_committed,source_retired,erase_completed,transaction_count\n"; + for(const auto& c:maintenance_completions)m<flush();if(!*stream)throw std::runtime_error("raw evidence write failed");} + const auto wall=std::chrono::duration(std::chrono::steady_clock::now()-started).count(); + rusage usage{};getrusage(RUSAGE_SELF,&usage); + std::set write_ids,read_ids;for(const auto& r:records)(r.phase=="STARTUP_WRITE"?write_ids:read_ids).insert(r.id); + Json summary={{"schema_version","eq3-startup-write-probe-v1"},{"classification","BACKEND_FIXED_TRACE_DIAGNOSTIC"}, + {"source_provenance",{{"label",o.source_label},{"sha256",o.source_sha256}}}, + {"profile",fs::absolute(o.profile).string()},{"stack_map",fs::absolute(o.stack_map).string()}, + {"payload_semantics","METADATA_AND_NATIVE_COMMANDS_ONLY_PAYLOAD_BYTES_UNAVAILABLE"}, + {"maintenance_semantics","SOFTWARE_LIFECYCLE_TEST_ON_FRESHLY_PROGRAMMED_PAGES_NOT_RETENTION_DUE"}, + {"page_age_origin","EACH_STARTUP_PROGRAM_COMPLETION"}, + {"request_id_ranges",{{"startup_write","1..pages"},{"verify_read","1000000001..1000000000+pages"}, + {"software_lifecycle_maintenance","2000000001..2000000000+maintenance_pages"}}}, + {"host_ingress_energy","UNAVAILABLE_NOT_FABRIC_RETURN_DELIVERY"},{"page_bytes",profile.page_bytes},{"pages",o.pages}, + {"submission_policy","BOUNDED_ROLLING_REFILL_ON_EACH_COMPLETION"},{"outstanding_window",o.outstanding_window}, + {"write_requests",write_ids.size()},{"read_requests",read_ids.size()}, + {"write_mapping_validation",{{"same_page_reads_completed",read_ids.size()}, + {"read_coverage",{{"method",o.verify_pages==o.pages?"ALL_LOADED_PAGES": + (o.verify_pages==1?"LAST_PAGE_ONLY":"EVENLY_STRATIFIED_INCLUDING_FIRST_AND_LAST")}, + {"loaded_pages",o.pages},{"verified_pages",o.verify_pages}, + {"includes_last_page",true}}}, + {"fresh_pages_maintenance_committed",maintenance_completions.size()}, + {"payload_validation","UNAVAILABLE"}}}, + {"phase_boundaries_ns",{{"startup_write_start",0},{"startup_write_end",write_end}, + {"verify_read_start",write_end},{"verify_read_end",read_end}, + {"software_lifecycle_maintenance_start",read_end}, + {"software_lifecycle_maintenance_end",maintenance_end}}}, + {"maintenance_requests",o.maintenance_pages},{"maintenance_committed",maintenance_completions.size()}, + {"native_command_events",native.size()},{"request_observations",observations.size()}, + {"wall_seconds",wall},{"max_rss_kib",usage.ru_maxrss},{"status","PASS"}}; + publish(o.output/"summary.json",summary);publish(o.output/"DONE.json",{{"status","PASS"},{"wall_seconds",wall},{"max_rss_kib",usage.ru_maxrss}}); + std::cout< \ + -DEQ3_THERMAL_ARCHIVE=/libhbfsim_eq3_thermal.a +cmake --build -j2 +python3 experiments/eq3_maintenance/thermal/tests/test_thermal_service.py \ + --binary /eq3_maintenance_thermal_service +``` + +The full-model paired test is deliberately separate because it factors the +64,512-node model. It feeds the same one-step node energy to the existing +campaign runner and the persistent service, then compares all 275 sensors and +both energy receipts: + +```text +python3 experiments/eq3_maintenance/thermal/tests/pair_with_campaign_runner.py \ + --service /eq3_maintenance_thermal_service \ + --campaign-runner /eq3_campaign_rc_runner \ + --runner-source tools/eq3_campaign_rc_runner.cpp \ + --model /model.txt --grid /rc_grid.json \ + --sensors /rc_sensors.json --output .json +``` + +Run the full check only through the campaign's coordinated CPU/resource entry. diff --git a/experiments/eq3_maintenance/thermal/client_schema.json b/experiments/eq3_maintenance/thermal/client_schema.json new file mode 100644 index 0000000..7fa4b66 --- /dev/null +++ b/experiments/eq3_maintenance/thermal/client_schema.json @@ -0,0 +1,88 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "urn:hbfsim:eq3:maintenance:persistent-thermal-client:v1", + "title": "EQ3 persistent thermal ADVANCE response", + "type": "object", + "required": [ + "type", "from_ns", "time_ns", "steps", "entity_temperatures_k", + "sensor_temperatures_k", "temperature_range_k", "energy_j", "timing", + "factorization_count" + ], + "properties": { + "type": {"const": "ADVANCE"}, + "from_ns": {"type": "integer", "minimum": 0}, + "time_ns": {"type": "integer", "minimum": 1}, + "steps": {"type": "integer", "minimum": 1}, + "entity_temperatures_k": { + "type": "object", + "description": "Keys are locked non-background component IDs; the production lock has 255.", + "additionalProperties": { + "type": "object", + "required": ["mean_k", "hotspot_k"], + "properties": { + "mean_k": {"type": "number", "exclusiveMinimum": 0}, + "hotspot_k": {"type": "number", "exclusiveMinimum": 0} + }, + "additionalProperties": false + } + }, + "sensor_temperatures_k": { + "type": "object", + "description": "Keys are the locked sensor IDs; the production lock has 275.", + "additionalProperties": {"type": "number", "exclusiveMinimum": 0} + }, + "temperature_range_k": { + "type": "array", "minItems": 2, "maxItems": 2, + "items": {"type": "number", "exclusiveMinimum": 0} + }, + "energy_j": { + "type": "object", + "required": ["window", "cumulative", "accepted_activity", + "accepted_not_yet_applied_activity"], + "properties": { + "window": {"$ref": "#/$defs/energyBalance"}, + "cumulative": {"$ref": "#/$defs/energyBalance"}, + "accepted_activity": {"type": "number", "minimum": 0}, + "accepted_not_yet_applied_activity": {"type": "number", "minimum": 0} + }, + "additionalProperties": false + }, + "timing": { + "type": "object", + "required": ["advance_seconds", "observation_seconds", + "serialization_probe_seconds", "previous_stdout_write_seconds", + "input_parse_seconds"], + "properties": { + "advance_seconds": {"type": "number", "minimum": 0}, + "observation_seconds": {"type": "number", "minimum": 0}, + "serialization_probe_seconds": {"type": "number", "minimum": 0}, + "previous_stdout_write_seconds": {"type": "number", "minimum": 0}, + "input_parse_seconds": {"type": "number", "minimum": 0} + }, + "additionalProperties": true + }, + "factorization_count": {"const": 1} + }, + "$defs": { + "energyBalance": { + "type": "object", + "required": [ + "activity_input_j", "static_input_j", "total_input_j", + "stored_energy_change_j", "boundary_loss_j", "energy_residual_j", + "node_mapped_activity_input_j", "component_mapping_error_j" + ], + "properties": { + "activity_input_j": {"type": "number", "minimum": 0}, + "static_input_j": {"type": "number", "minimum": 0}, + "total_input_j": {"type": "number", "minimum": 0}, + "stored_energy_change_j": {"type": "number"}, + "boundary_loss_j": {"type": "number"}, + "energy_residual_j": {"type": "number"}, + "node_mapped_activity_input_j": {"type": "number", "minimum": 0}, + "component_mapping_error_j": {"type": "number"} + }, + "additionalProperties": false + } + }, + "additionalProperties": false +} diff --git a/experiments/eq3_maintenance/thermal/tests/pair_with_campaign_runner.py b/experiments/eq3_maintenance/thermal/tests/pair_with_campaign_runner.py new file mode 100644 index 0000000..b5767ac --- /dev/null +++ b/experiments/eq3_maintenance/thermal/tests/pair_with_campaign_runner.py @@ -0,0 +1,110 @@ +#!/usr/bin/env python3 +"""One-step full-model equivalence check; intended for coordinated execution.""" +import argparse +import hashlib +import json +import math +import subprocess +import tempfile +from pathlib import Path + + +def sha(path): + return hashlib.sha256(path.read_bytes()).hexdigest() + + +def observe(definitions, values): + result = {} + for sensor in definitions: + if sensor["reduction"] == "max": + value = max(values[index] for index in sensor["cell_indices"]) + else: + value = math.fsum(values[index] * weight + for index, weight in sensor["cell_weights"]) + result[sensor["id"]] = value + return result + + +def main(): + parser = argparse.ArgumentParser() + parser.add_argument("--service", type=Path, required=True) + parser.add_argument("--campaign-runner", type=Path, required=True) + parser.add_argument("--runner-source", type=Path, required=True) + parser.add_argument("--model", type=Path, required=True) + parser.add_argument("--grid", type=Path, required=True) + parser.add_argument("--sensors", type=Path, required=True) + parser.add_argument("--component", default="hbm0.base") + parser.add_argument("--energy-j", type=float, default=0.02) + parser.add_argument("--step-ns", type=int, default=20_000_000) + parser.add_argument("--output", type=Path, required=True) + args = parser.parse_args() + if args.energy_j < 0 or args.step_ns <= 0: + raise ValueError("positive step and non-negative energy required") + grid = json.loads(args.grid.read_text()) + sensors = json.loads(args.sensors.read_text()) + indices = grid["component_cells"][args.component] + volumes = [grid["cells"][index]["volume_m3"] for index in indices] + total_volume = math.fsum(volumes) + assignments = [(grid["cells"][index]["id"], args.energy_j * volume / total_volume) + for index, volume in zip(indices, volumes)] + event = "activity 1 pair external_heat external 0 -1 -1 -1 0 {0} {0} {1}\n".format( + args.step_ns * 1e-9, + " ".join(f"{node} {energy:.17g}" for node, energy in assignments)) + with tempfile.TemporaryDirectory() as temporary: + root = Path(temporary) + events = root / "events.txt" + events.write_text("HBFSIM_EQ3_THERMAL_EVENTS 1\n" + event) + runner = subprocess.run([ + str(args.campaign_runner), "--run", "--model", str(args.model), + "--events", str(events), "--step-s", str(args.step_ns * 1e-9), + "--slot-s", str(args.step_ns * 1e-9), "--end-s", str(args.step_ns * 1e-9), + "--sample-s", str(args.step_ns * 1e-9), "--min-k", "300", "--max-k", "400", + "--model-sha256", sha(args.model), "--events-sha256", sha(events), + "--runner-source-sha256", sha(args.runner_source), + "--domain-version", "eq3-maintenance-paired-test-v1"], + cwd=root, text=True, stdout=subprocess.PIPE, stderr=subprocess.PIPE, + timeout=600, check=True) + values = {} + for line in runner.stdout.splitlines()[1:]: + time_s, node, temperature = line.split(",") + if math.isclose(float(time_s), args.step_ns * 1e-9, abs_tol=1e-14): + values[node] = float(temperature) + node_values = [values[cell["id"]] for cell in grid["cells"]] + expected = observe(sensors, node_values) + service_input = (f"ENERGY 0 {args.step_ns} {args.component} {args.energy_j:.17g}\n" + f"ADVANCE {args.step_ns}\nQUIT\n") + service = subprocess.run([ + str(args.service), "--model", str(args.model), "--grid", str(args.grid), + "--sensors", str(args.sensors), "--model-sha256", sha(args.model), + "--grid-sha256", sha(args.grid), "--sensors-sha256", sha(args.sensors), + "--step-ns", str(args.step_ns), "--min-k", "300", "--max-k", "400"], + input=service_input, text=True, stdout=subprocess.PIPE, stderr=subprocess.PIPE, + timeout=600, check=True) + advanced = next(json.loads(line) for line in service.stdout.splitlines() + if json.loads(line)["type"] == "ADVANCE") + actual = advanced["sensor_temperatures_k"] + worst = max((abs(actual[key] - value), key, value, actual[key]) + for key, value in expected.items()) + runner_receipt = json.loads((root / "rc_energy_receipt.json").read_text()) + result = { + "status": "PASS" if worst[0] <= 1e-10 else "FAIL", + "schema_version": "eq3-maintenance-thermal-paired-v1", + "component": args.component, + "energy_j": args.energy_j, + "step_ns": args.step_ns, + "sensor_count": len(expected), + "max_sensor_abs_difference_k": worst[0], + "worst_sensor": {"id": worst[1], "campaign_k": worst[2], "service_k": worst[3]}, + "campaign_energy": runner_receipt, + "service_energy": advanced["energy_j"]["cumulative"], + "identities": {"model_sha256": sha(args.model), "grid_sha256": sha(args.grid), + "sensors_sha256": sha(args.sensors)}, + } + args.output.write_text(json.dumps(result, indent=2) + "\n") + print(json.dumps(result)) + if result["status"] != "PASS": + raise SystemExit(1) + + +if __name__ == "__main__": + main() diff --git a/experiments/eq3_maintenance/thermal/tests/stream_campaign_replay.py b/experiments/eq3_maintenance/thermal/tests/stream_campaign_replay.py new file mode 100644 index 0000000..fb659b8 --- /dev/null +++ b/experiments/eq3_maintenance/thermal/tests/stream_campaign_replay.py @@ -0,0 +1,331 @@ +#!/usr/bin/env python3 +"""Stream a full campaign-runner field into registered sensor observations. + +The campaign runner emits one CSV row per node and sample. This harness keeps +only one node frame in memory, reduces it with the registered sensor +definitions, and compares it with an immutable maintenance-service thermal CSV. +""" +import argparse +import csv +import hashlib +import json +import math +import resource +import subprocess +import time +from pathlib import Path + + +def sha256(path): + digest = hashlib.sha256() + with Path(path).open("rb") as stream: + for chunk in iter(lambda: stream.read(1024 * 1024), b""): + digest.update(chunk) + return digest.hexdigest() + + +def model_nodes(path): + result = [] + with Path(path).open() as stream: + for line in stream: + fields = line.split() + if fields and fields[0] == "node": + result.append({"id": fields[1], "capacity": float(fields[6]), + "initial": float(fields[7]), + "static_w": float(fields[8]), + "boundary_g": float(fields[9]), + "boundary_k": float(fields[10])}) + if not result: + raise ValueError("model contains no nodes") + return result + + +def reference_rows(path): + with Path(path).open(newline="") as stream: + for row in csv.DictReader(stream): + if row.get("type") != "ADVANCE": + continue + yield { + "time_ns": int(row["time_ns"]), + "start_ns": int(row["start_ns"]), + "end_ns": int(row["end_ns"]), + "sensors": json.loads(row["sensor_temperatures_k"]), + "range": json.loads(row["temperature_range_k"]), + "energy": json.loads(row["energy_j"])["cumulative"], + } + + +def window_rows(path): + with Path(path).open(newline="") as stream: + for row in csv.DictReader(stream): + if row.get("kind") == "WINDOW_TOTAL": + totals = json.loads(row["component_energy_j"]) + yield {"start_ns": int(row["start_ns"]), + "end_ns": int(row["end_ns"]), + "phase": row["phase"] or "UNKNOWN", + "energy_j": math.fsum(float(value) for value in totals.values())} + + +def observe(definitions, values): + output = {} + for sensor in definitions: + if sensor["reduction"] == "max": + value = max(values[index] for index in sensor["cell_indices"]) + elif sensor["reduction"] == "weighted_mean": + value = math.fsum(values[index] * weight + for index, weight in sensor["cell_weights"]) + else: + raise ValueError(f"unsupported sensor reduction: {sensor['reduction']}") + output[sensor["id"]] = value + return output + + +def main(): + parser = argparse.ArgumentParser() + parser.add_argument("--runner", type=Path, required=True) + parser.add_argument("--runner-source", type=Path, required=True) + parser.add_argument("--model", type=Path, required=True) + parser.add_argument("--grid", type=Path, required=True) + parser.add_argument("--sensors", type=Path, required=True) + parser.add_argument("--events", type=Path, required=True) + parser.add_argument("--reference-thermal", type=Path, required=True) + parser.add_argument("--energy-csv", type=Path, required=True) + parser.add_argument("--output-dir", type=Path, required=True) + parser.add_argument("--step-s", type=float, default=0.02) + parser.add_argument("--end-s", type=float, default=10.0) + parser.add_argument("--active-end-ns", type=int, default=8_000_000_000) + parser.add_argument("--temperature-tolerance-k", type=float, default=1e-10) + parser.add_argument("--energy-tolerance-j", type=float, default=1e-8) + args = parser.parse_args() + for name in ("runner", "runner_source", "model", "grid", "sensors", "events", + "reference_thermal", "energy_csv"): + setattr(args, name, getattr(args, name).resolve(strict=True)) + args.output_dir = args.output_dir.resolve() + args.output_dir.mkdir(parents=True, exist_ok=True) + for name in ("derived_sensors.csv", "frame_comparison.csv", "runner.stderr.log", + "result.json"): + if (args.output_dir / name).exists(): + raise FileExistsError(f"refusing to overwrite {args.output_dir / name}") + + nodes = model_nodes(args.model) + grid = json.loads(args.grid.read_text()) + definitions = json.loads(args.sensors.read_text()) + grid_ids = [str(cell["id"]) for cell in grid["cells"]] + node_ids = [node["id"] for node in nodes] + if grid_ids != node_ids: + raise ValueError("model node order differs from rc_grid cell order") + references = reference_rows(args.reference_thermal) + windows = window_rows(args.energy_csv) + expected_frames = round(args.end_s / args.step_s) + command = [str(args.runner), "--run", "--model", str(args.model), + "--events", str(args.events), "--step-s", format(args.step_s, ".17g"), + "--slot-s", format(args.step_s, ".17g"), + "--sample-s", format(args.step_s, ".17g"), + "--end-s", format(args.end_s, ".17g"), "--min-k", "300", "--max-k", "400", + "--model-sha256", sha256(args.model), "--events-sha256", sha256(args.events), + "--runner-source-sha256", sha256(args.runner_source), + "--domain-version", "EQ3_MAINTENANCE_PILOT_REPLAY_2MM_V1"] + + sensor_path = args.output_dir / "derived_sensors.csv" + frame_path = args.output_dir / "frame_comparison.csv" + stderr_path = args.output_dir / "runner.stderr.log" + start_wall = time.monotonic() + max_sensor_diff = 0.0 + max_range_diff = 0.0 + max_energy_diff = 0.0 + worst_sensor = None + worst_energy = None + cumulative_input = 0.0 + cumulative_boundary = 0.0 + frame_count = 0 + stdout_rows = 0 + global_min = math.inf + global_max = -math.inf + initial_stored = math.fsum(node["capacity"] * node["initial"] for node in nodes) + static_power = math.fsum(node["static_w"] for node in nodes) + + with sensor_path.open("w", newline="") as sensor_stream, \ + frame_path.open("w", newline="") as frame_stream, \ + stderr_path.open("w") as stderr_stream: + sensor_writer = csv.writer(sensor_stream) + sensor_writer.writerow(["time_ns", "source_phase", "derived_segment", "sensor_id", + "campaign_k", "service_k", "abs_difference_k"]) + frame_writer = csv.writer(frame_stream) + frame_writer.writerow(["time_ns", "source_phase", "derived_segment", + "campaign_min_k", "service_min_k", "campaign_max_k", + "service_max_k", "stored_energy_j", "service_stored_energy_j", + "boundary_loss_j", "service_boundary_loss_j", "input_energy_j", + "service_input_energy_j", "energy_residual_j", + "service_energy_residual_j"]) + process = subprocess.Popen(command, cwd=args.output_dir, text=True, + stdout=subprocess.PIPE, stderr=stderr_stream, bufsize=1) + assert process.stdout is not None + header = process.stdout.readline().rstrip("\n") + if header != "time_s,node_id,temperature_k": + process.kill() + raise ValueError(f"unexpected runner stdout header: {header!r}") + values = [0.0] * len(nodes) + current_time = None + index = 0 + try: + for line in process.stdout: + stdout_rows += 1 + fields = line.rstrip("\n").split(",") + if len(fields) != 3: + raise ValueError(f"malformed runner stdout row {stdout_rows}") + time_s = float(fields[0]) + if current_time is None: + current_time = time_s + elif time_s != current_time: + raise ValueError("runner changed frame time before emitting every model node") + if index >= len(nodes) or fields[1] != node_ids[index]: + raise ValueError(f"runner node order mismatch at stdout row {stdout_rows}") + values[index] = float(fields[2]) + index += 1 + if index != len(nodes): + continue + + if math.isclose(current_time, 0.0, abs_tol=1e-15): + if max(abs(value - node["initial"]) for value, node in zip(values, nodes)) > 1e-12: + raise ValueError("runner t0 differs from model initial state") + else: + frame_count += 1 + reference = next(references) + window = next(windows) + time_ns = round(current_time * 1_000_000_000) + if (time_ns != reference["time_ns"] or time_ns != window["end_ns"] or + reference["start_ns"] != window["start_ns"] or + reference["end_ns"] != window["end_ns"]): + raise ValueError(f"window identity mismatch at frame {frame_count}") + campaign = observe(definitions, values) + if set(campaign) != set(reference["sensors"]): + raise ValueError("sensor identity mismatch") + segment = "ACTIVE" if time_ns <= args.active_end_ns else "RECOVERY" + for sensor_id, campaign_k in campaign.items(): + service_k = float(reference["sensors"][sensor_id]) + difference = abs(campaign_k - service_k) + sensor_writer.writerow([time_ns, window["phase"], segment, sensor_id, + format(campaign_k, ".17g"), + format(service_k, ".17g"), + format(difference, ".17g")]) + if difference > max_sensor_diff: + max_sensor_diff = difference + worst_sensor = {"time_ns": time_ns, "sensor_id": sensor_id, + "campaign_k": campaign_k, "service_k": service_k} + frame_min, frame_max = min(values), max(values) + global_min = min(global_min, frame_min) + global_max = max(global_max, frame_max) + range_difference = max(abs(frame_min - float(reference["range"][0])), + abs(frame_max - float(reference["range"][1]))) + max_range_diff = max(max_range_diff, range_difference) + cumulative_input += window["energy_j"] + static_power * args.step_s + stored = (math.fsum(node["capacity"] * value + for node, value in zip(nodes, values)) - initial_stored) + cumulative_boundary += args.step_s * math.fsum( + node["boundary_g"] * (value - node["boundary_k"]) + for node, value in zip(nodes, values)) + residual = cumulative_input - stored - cumulative_boundary + energy_pairs = { + "input_energy_j": (cumulative_input, + float(reference["energy"]["total_input_j"])), + "stored_energy_j": (stored, + float(reference["energy"]["stored_energy_change_j"])), + "boundary_loss_j": (cumulative_boundary, + float(reference["energy"]["boundary_loss_j"])), + "energy_residual_j": (residual, + float(reference["energy"]["energy_residual_j"])), + } + for field, pair in energy_pairs.items(): + difference = abs(pair[0] - pair[1]) + if difference > max_energy_diff: + max_energy_diff = difference + worst_energy = {"time_ns": time_ns, "field": field, + "campaign": pair[0], "service": pair[1]} + frame_writer.writerow([ + time_ns, window["phase"], segment, format(frame_min, ".17g"), + format(float(reference["range"][0]), ".17g"), format(frame_max, ".17g"), + format(float(reference["range"][1]), ".17g"), format(stored, ".17g"), + format(float(reference["energy"]["stored_energy_change_j"]), ".17g"), + format(cumulative_boundary, ".17g"), + format(float(reference["energy"]["boundary_loss_j"]), ".17g"), + format(cumulative_input, ".17g"), + format(float(reference["energy"]["total_input_j"]), ".17g"), + format(residual, ".17g"), + format(float(reference["energy"]["energy_residual_j"]), ".17g")]) + current_time = None + index = 0 + except Exception: + process.kill() + process.wait() + raise + return_code = process.wait() + if return_code: + raise RuntimeError(f"campaign runner failed with status {return_code}") + if index or frame_count != expected_frames: + raise ValueError(f"runner emitted {frame_count} complete non-t0 frames; expected {expected_frames}") + try: + next(references) + raise ValueError("reference thermal CSV has extra ADVANCE frames") + except StopIteration: + pass + try: + next(windows) + raise ValueError("energy CSV has extra WINDOW_TOTAL rows") + except StopIteration: + pass + + runner_receipt_path = args.output_dir / "rc_energy_receipt.json" + runner_receipt = json.loads(runner_receipt_path.read_text()) + final_ledger_difference = max( + abs(float(runner_receipt["total_input_energy_j"]) - cumulative_input), + abs(float(runner_receipt["stored_energy_change_j"]) - stored), + abs(float(runner_receipt["boundary_loss_j"]) - cumulative_boundary), + abs(float(runner_receipt["energy_residual_j"]) - residual)) + status = "PASS" if (max_sensor_diff <= args.temperature_tolerance_k and + max_range_diff <= args.temperature_tolerance_k and + max_energy_diff <= args.energy_tolerance_j and + final_ledger_difference <= args.energy_tolerance_j) else "FAIL" + result = { + "schema_version": "eq3-maintenance-a3-streamed-pair-v1", + "status": status, + "classification": "FULL_2MM_DISCRETE_EQUIVALENCE", + "limitations": ["NOT_INDEPENDENT_PHYSICAL_REFERENCE", "NO_CONTROL_CHANGE"], + "frame_count": frame_count, + "sensor_count": len(definitions), + "sensor_comparison_count": frame_count * len(definitions), + "stdout_node_rows_streamed": stdout_rows, + "stdout_all_node_artifact_retained": False, + "max_sensor_abs_difference_k": max_sensor_diff, + "max_global_range_abs_difference_k": max_range_diff, + "max_cumulative_energy_abs_difference_j": max_energy_diff, + "final_runner_vs_streamed_energy_abs_difference_j": final_ledger_difference, + "worst_sensor": worst_sensor, + "worst_energy": worst_energy, + "runner_observed_range_k": [global_min, global_max], + "thresholds": {"temperature_abs_k": args.temperature_tolerance_k, + "cumulative_energy_abs_j": args.energy_tolerance_j}, + "runner_command": command, + "identities": {"runner_sha256": sha256(args.runner), + "runner_source_sha256": sha256(args.runner_source), + "model_sha256": sha256(args.model), "grid_sha256": sha256(args.grid), + "sensors_sha256": sha256(args.sensors), + "events_sha256": sha256(args.events), + "reference_thermal_sha256": sha256(args.reference_thermal), + "source_energy_sha256": sha256(args.energy_csv), + "runner_receipt_sha256": sha256(runner_receipt_path)}, + "artifacts": {"derived_sensors": str(sensor_path), + "frame_comparison": str(frame_path), + "runner_energy_receipt": str(runner_receipt_path), + "runner_stderr": str(stderr_path)}, + "resources": {"wall_s": time.monotonic() - start_wall, + "maxrss_kib_self_and_waited_children": resource.getrusage(resource.RUSAGE_SELF).ru_maxrss + + resource.getrusage(resource.RUSAGE_CHILDREN).ru_maxrss}, + } + (args.output_dir / "result.json").write_text(json.dumps(result, indent=2) + "\n") + print(json.dumps(result)) + if status != "PASS": + raise SystemExit(1) + + +if __name__ == "__main__": + main() diff --git a/experiments/eq3_maintenance/thermal/tests/test_thermal_service.py b/experiments/eq3_maintenance/thermal/tests/test_thermal_service.py new file mode 100644 index 0000000..1be95c5 --- /dev/null +++ b/experiments/eq3_maintenance/thermal/tests/test_thermal_service.py @@ -0,0 +1,128 @@ +#!/usr/bin/env python3 +"""Fixed small-model tests for the isolated persistent thermal service.""" +import argparse +import hashlib +import json +import subprocess +import tempfile +import unittest +from pathlib import Path + +BINARY = None + + +def digest(path): + return hashlib.sha256(path.read_bytes()).hexdigest() + + +class ServiceTests(unittest.TestCase): + def setUp(self): + self.temporary = tempfile.TemporaryDirectory() + root = Path(self.temporary.name) + self.model = root / "model.txt" + self.grid = root / "rc_grid.json" + self.sensors = root / "rc_sensors.json" + self.model.write_text( + "HBFSIM_EQ3_THERMAL_MODEL 1\n" + "coupling on\n" + "node n0 gpu compute a -1 2 300 0 0.5 300\n" + "node n1 hbm fast_memory b 0 1 300 0 0.25 300\n" + "edge n0 n1 1 component\n") + cells = [ + {"index": 0, "id": "n0", "component": "a", "volume_m3": 2.0, + "capacity_j_k": 2.0}, + {"index": 1, "id": "n1", "component": "b", "volume_m3": 1.0, + "capacity_j_k": 1.0}, + ] + self.grid.write_text(json.dumps({"shape": [2, 1, 1], "cells": cells, + "component_cells": {"a": [0], "b": [1]}})) + self.sensors.write_text(json.dumps([ + {"id": "component:a:hotspot", "reduction": "max", "cell_indices": [0]}, + {"id": "component:a:mean", "reduction": "weighted_mean", + "cell_weights": [[0, 1.0]]}, + {"id": "component:b:hotspot", "reduction": "max", "cell_indices": [1]}, + {"id": "component:b:mean", "reduction": "weighted_mean", + "cell_weights": [[1, 1.0]]}, + ])) + + def tearDown(self): + self.temporary.cleanup() + + def command(self, maximum="400", model_hash=None): + return [str(BINARY), "--model", str(self.model), "--grid", str(self.grid), + "--sensors", str(self.sensors), "--model-sha256", + model_hash or digest(self.model), "--grid-sha256", digest(self.grid), + "--sensors-sha256", digest(self.sensors), "--step-ns", "20000000", + "--min-k", "300", "--max-k", maximum] + + def run_service(self, text, **kwargs): + return subprocess.run(self.command(**kwargs), input=text, text=True, + stdout=subprocess.PIPE, stderr=subprocess.PIPE, timeout=20) + + def test_two_components_persistent_energy_and_observations(self): + result = self.run_service( + "ENERGY 0 20000000 a 2\n" + "ENERGY 0 20000000 b 1\n" + "ADVANCE 20000000\nQUIT\n") + self.assertEqual(result.returncode, 0, result.stderr) + rows = [json.loads(line) for line in result.stdout.splitlines()] + self.assertEqual([row["type"] for row in rows], + ["READY", "ENERGY_ACK", "ENERGY_ACK", "ADVANCE", "BYE"]) + self.assertEqual(rows[0]["nodes"], 2) + self.assertEqual(rows[0]["entities"], 2) + self.assertEqual(rows[0]["sensors"], 4) + advanced = rows[3] + self.assertEqual(advanced["factorization_count"], 1) + self.assertEqual(len(advanced["entity_temperatures_k"]), 2) + self.assertEqual(len(advanced["sensor_temperatures_k"]), 4) + energy = advanced["energy_j"]["cumulative"] + self.assertAlmostEqual(energy["activity_input_j"], 3.0, places=12) + self.assertLess(abs(energy["energy_residual_j"]), 1e-10) + self.assertGreater(advanced["temperature_range_k"][1], 300) + self.assertIn("advance_seconds", advanced["timing"]) + self.assertIn("observation_seconds", advanced["timing"]) + self.assertIn("previous_stdout_write_seconds", advanced["timing"]) + + def test_duplicate_energy_is_rejected_before_double_count(self): + result = self.run_service( + "ENERGY 0 20000000 a 1\nENERGY 0 20000000 a 1\n") + self.assertEqual(result.returncode, 2) + rows = [json.loads(line) for line in result.stdout.splitlines()] + self.assertEqual(rows[-1]["status"], "INPUT_OR_PROTOCOL_FAILURE") + self.assertIn("double count", rows[-1]["reason"]) + + def test_committed_past_and_partial_step_are_rejected(self): + for invalid in ("ENERGY 0 10000000 a 1\n", "ADVANCE 10000000\n"): + with self.subTest(invalid=invalid): + result = self.run_service(invalid) + self.assertEqual(result.returncode, 2) + self.assertEqual(json.loads(result.stdout.splitlines()[-1])["status"], + "INPUT_OR_PROTOCOL_FAILURE") + + def test_domain_failure_preserves_last_valid_and_unknown_trial_energy(self): + result = self.run_service("ENERGY 0 20000000 a 100\nADVANCE 20000000\n", + maximum="300.01") + self.assertEqual(result.returncode, 2) + failure = json.loads(result.stdout.splitlines()[-1]) + self.assertEqual(failure["status"], "DOMAIN_FAILURE") + self.assertEqual(failure["last_valid_time_ns"], 0) + self.assertEqual(failure["trial_target_time_ns"], 20000000) + self.assertEqual(failure["failed_trial_step_energy"]["status"], + "UNKNOWN_NOT_INTEGRATED") + self.assertEqual(failure["completed_energy_j"]["activity_input_j"], 0) + self.assertTrue(failure["offending_nodes"]) + + def test_hash_mismatch_fails_closed(self): + result = self.run_service("", model_hash="0" * 64) + self.assertEqual(result.returncode, 2) + failure = json.loads(result.stdout.splitlines()[-1]) + self.assertEqual(failure["status"], "INPUT_OR_PROTOCOL_FAILURE") + self.assertIn("SHA-256 mismatch", failure["reason"]) + + +if __name__ == "__main__": + parser = argparse.ArgumentParser() + parser.add_argument("--binary", type=Path, required=True) + arguments = parser.parse_args() + BINARY = arguments.binary.resolve() + unittest.main(argv=[__file__]) diff --git a/experiments/eq3_maintenance/thermal/thermal_service.cpp b/experiments/eq3_maintenance/thermal/thermal_service.cpp new file mode 100644 index 0000000..b3faa68 --- /dev/null +++ b/experiments/eq3_maintenance/thermal/thermal_service.cpp @@ -0,0 +1,650 @@ +// Persistent, experiment-only sparse thermal service for EQ3 maintenance studies. +// It intentionally does not alter the public thermal ABI or any default target. +#include "hbfsim/eq3_thermal/thermal.hpp" + +#include +#include +#include +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +namespace { +using Clock = std::chrono::steady_clock; +using Json = nlohmann::json; +using SparseMatrix = Eigen::SparseMatrix; +using SparseSolver = Eigen::SimplicialLDLT>; +using ThermalModelConfig = hbfsim::eq3_thermal::ThermalModelConfig; +using ConductanceKind = hbfsim::eq3_thermal::ConductanceKind; +using Tick = std::uint64_t; + +[[noreturn]] void fail(const std::string& message) { + throw std::invalid_argument(message); +} +void require(bool condition, const std::string& message) { + if (!condition) fail(message); +} +double seconds(Clock::duration duration) { + return std::chrono::duration(duration).count(); +} +std::string read_file(const std::string& path) { + std::ifstream input(path, std::ios::binary); + if (!input) throw std::runtime_error("cannot open input: " + path); + std::ostringstream output; + output << input.rdbuf(); + if (input.bad()) throw std::runtime_error("failed reading input: " + path); + return output.str(); +} +std::string sha256(const std::string& bytes) { + std::unique_ptr context( + EVP_MD_CTX_new(), EVP_MD_CTX_free); + require(context != nullptr, "cannot allocate SHA-256 context"); + require(EVP_DigestInit_ex(context.get(), EVP_sha256(), nullptr) == 1 && + EVP_DigestUpdate(context.get(), bytes.data(), bytes.size()) == 1, + "SHA-256 calculation failed"); + unsigned char digest[EVP_MAX_MD_SIZE]; + unsigned int size = 0; + require(EVP_DigestFinal_ex(context.get(), digest, &size) == 1 && size == 32, + "SHA-256 finalization failed"); + std::ostringstream output; + output << std::hex << std::setfill('0'); + for (unsigned int index = 0; index < size; ++index) + output << std::setw(2) << static_cast(digest[index]); + return output.str(); +} +Tick unsigned_value(const std::string& token, const std::string& label) { + Tick result{}; + const auto [end, error] = + std::from_chars(token.data(), token.data() + token.size(), result); + require(error == std::errc{} && end == token.data() + token.size(), + "invalid " + label); + return result; +} +double finite_nonnegative(const std::string& token, const std::string& label) { + std::size_t used = 0; + double result{}; + try { + result = std::stod(token, &used); + } catch (const std::exception&) { + fail("invalid " + label); + } + require(used == token.size() && std::isfinite(result) && result >= 0, + label + " must be finite and non-negative"); + return result; +} + +struct Options { + std::string model_path, grid_path, sensors_path; + std::string model_sha256, grid_sha256, sensors_sha256; + Tick step_ns{}; + double minimum_k{}, maximum_k{}; +}; + +Options parse_options(int argc, char** argv) { + std::map values; + for (int index = 1; index < argc; index += 2) { + require(index + 1 < argc, "every option requires a value"); + const std::string key = argv[index]; + require(key.starts_with("--") && values.emplace(key, argv[index + 1]).second, + "unknown or duplicate option: " + key); + } + for (const char* key : {"--model", "--grid", "--sensors", "--model-sha256", + "--grid-sha256", "--sensors-sha256", "--step-ns", + "--min-k", "--max-k"}) + require(values.contains(key), std::string("missing option: ") + key); + require(values.size() == 9, "unknown CLI option"); + Options result; + result.model_path = values["--model"]; + result.grid_path = values["--grid"]; + result.sensors_path = values["--sensors"]; + result.model_sha256 = values["--model-sha256"]; + result.grid_sha256 = values["--grid-sha256"]; + result.sensors_sha256 = values["--sensors-sha256"]; + result.step_ns = unsigned_value(values["--step-ns"], "step-ns"); + result.minimum_k = finite_nonnegative(values["--min-k"], "min-k"); + result.maximum_k = finite_nonnegative(values["--max-k"], "max-k"); + require(result.step_ns > 0, "step-ns must be positive"); + require(result.minimum_k > 0 && result.minimum_k < result.maximum_k, + "temperature domain must be positive and increasing"); + return result; +} + +SparseMatrix system_matrix(const ThermalModelConfig& config, double step_s) { + require(config.nodes.size() <= static_cast(std::numeric_limits::max()), + "node count exceeds Eigen int range"); + const int count = static_cast(config.nodes.size()); + std::vector> entries; + entries.reserve(config.nodes.size() + 4 * config.edges.size()); + for (int index = 0; index < count; ++index) { + const auto& node = config.nodes[static_cast(index)]; + entries.emplace_back(index, index, node.heat_capacity_j_per_k / step_s + + node.boundary_conductance_w_per_k); + } + for (const auto& edge : config.edges) { + if (!config.direct_intercomponent_edges_enabled && + edge.kind == ConductanceKind::InterComponent) + continue; + const int left = static_cast(edge.node_a); + const int right = static_cast(edge.node_b); + entries.emplace_back(left, left, edge.conductance_w_per_k); + entries.emplace_back(right, right, edge.conductance_w_per_k); + entries.emplace_back(left, right, -edge.conductance_w_per_k); + entries.emplace_back(right, left, -edge.conductance_w_per_k); + } + SparseMatrix result(count, count); + result.setFromTriplets(entries.begin(), entries.end()); + result.makeCompressed(); + return result; +} + +struct Component { + std::string id; + std::vector cells; + std::vector volume_weights; +}; +struct Sensor { + std::string id; + bool maximum{}; + std::vector cells; + std::vector weights; +}; +struct Interval { + Tick start_step{}, end_step{}; + std::size_t component{}; + double energy_j{}; +}; + +class Service { + public: + Service(Options options, std::string model_text, std::string grid_text, + std::string sensors_text) + : options_(std::move(options)), model_text_(std::move(model_text)), + grid_text_(std::move(grid_text)), sensors_text_(std::move(sensors_text)), + config_(hbfsim::eq3_thermal::model_config_from_text(model_text_)), + step_s_(static_cast(options_.step_ns) * 1e-9), + matrix_(system_matrix(config_, step_s_)) { + verify_hash(model_text_, options_.model_sha256, "model"); + verify_hash(grid_text_, options_.grid_sha256, "grid"); + verify_hash(sensors_text_, options_.sensors_sha256, "sensors"); + parse_grid(Json::parse(grid_text_)); + parse_sensors(Json::parse(sensors_text_)); + const Eigen::Index count = static_cast(config_.nodes.size()); + temperature_.resize(count); + theta_.resize(count); + capacity_over_dt_.resize(count); + static_and_boundary_.resize(count); + applied_node_energy_ = Eigen::VectorXd::Zero(count); + origin_k_ = config_.nodes.front().initial_temperature_k; + for (Eigen::Index index = 0; index < count; ++index) { + const auto& node = config_.nodes[static_cast(index)]; + require(node.initial_temperature_k >= options_.minimum_k && + node.initial_temperature_k <= options_.maximum_k, + "initial temperature outside declared domain: " + node.id); + temperature_[index] = node.initial_temperature_k; + theta_[index] = node.initial_temperature_k - origin_k_; + capacity_over_dt_[index] = node.heat_capacity_j_per_k / step_s_; + static_and_boundary_[index] = + node.static_power_w + node.boundary_conductance_w_per_k * + (node.boundary_temperature_k - origin_k_); + static_power_w_ += node.static_power_w; + } + const auto begin = Clock::now(); + solver_.compute(matrix_); + factor_seconds_ = seconds(Clock::now() - begin); + require(solver_.info() == Eigen::Success, "sparse LDLT factorization failed"); + } + + Json ready() const { + return {{"type", "READY"}, + {"schema_version", "eq3-maintenance-persistent-thermal-v1"}, + {"backend", "Eigen::SimplicialLDLT_AMD"}, + {"state_variable", "theta=T-origin"}, + {"temperature_clamping", false}, + {"model_sha256", options_.model_sha256}, + {"grid_sha256", options_.grid_sha256}, + {"sensors_sha256", options_.sensors_sha256}, + {"step_ns", options_.step_ns}, + {"domain_k", {options_.minimum_k, options_.maximum_k}}, + {"nodes", config_.nodes.size()}, + {"entities", components_.size()}, + {"sensors", sensors_.size()}, + {"matrix_structural_nnz", matrix_.nonZeros()}, + {"factor_L_nnz", solver_.matrixL().nestedExpression().nonZeros()}, + {"factorization_count", 1}, + {"timing", {{"factor_seconds", factor_seconds_}}}}; + } + + Json energy(Tick start_ns, Tick end_ns, const std::string& id, double energy_j) { + require(start_ns < end_ns, "ENERGY interval must be positive"); + require(start_ns % options_.step_ns == 0 && end_ns % options_.step_ns == 0, + "ENERGY interval must align with the fixed thermal step"); + const Tick start = start_ns / options_.step_ns; + const Tick end = end_ns / options_.step_ns; + require(start >= current_step_, "ENERGY cannot enter committed thermal time"); + const auto component = component_index_.find(id); + require(component != component_index_.end(), "unknown or unpowered component: " + id); + const auto identity = std::tuple{start, end, component->second}; + require(interval_identities_.insert(identity).second, + "duplicate component ENERGY interval would double count"); + intervals_.push_back({start, end, component->second, energy_j}); + accepted_activity_j_ += energy_j; + return {{"type", "ENERGY_ACK"}, {"start_ns", start_ns}, {"end_ns", end_ns}, + {"component", id}, {"energy_j", energy_j}, + {"accepted_activity_energy_j", static_cast(accepted_activity_j_)}, + {"pending_intervals", intervals_.size()}}; + } + + Json advance(Tick target_ns) { + require(target_ns % options_.step_ns == 0, + "ADVANCE target must align with the fixed thermal step"); + const Tick target_step = target_ns / options_.step_ns; + require(target_step > current_step_, "ADVANCE must move forward"); + const Tick initial_step = current_step_; + const long double initial_activity = applied_activity_j_; + const long double initial_node_activity = applied_node_energy_.sum(); + const long double initial_static = static_input_j_; + const long double initial_boundary = boundary_loss_j_; + const long double initial_stored = stored_energy(); + const auto advance_start = Clock::now(); + for (; current_step_ < target_step;) advance_one(); + const double advance_seconds = seconds(Clock::now() - advance_start); + intervals_.erase(std::remove_if(intervals_.begin(), intervals_.end(), + [&](const Interval& value) { + return value.end_step <= current_step_; + }), intervals_.end()); + const auto observation_start = Clock::now(); + Json entities = entity_temperatures(); + Json sensors = sensor_temperatures(); + const double observation_seconds = seconds(Clock::now() - observation_start); + const long double stored = stored_energy(); + Json window_energy = energy_json(applied_activity_j_ - initial_activity, + static_input_j_ - initial_static, + stored - initial_stored, + boundary_loss_j_ - initial_boundary); + const long double mapped_window = applied_node_energy_.sum() - initial_node_activity; + window_energy["node_mapped_activity_input_j"] = static_cast(mapped_window); + window_energy["component_mapping_error_j"] = + static_cast(mapped_window - (applied_activity_j_ - initial_activity)); + Json cumulative_energy = energy_json(applied_activity_j_, static_input_j_, stored, + boundary_loss_j_); + cumulative_energy["node_mapped_activity_input_j"] = applied_node_energy_.sum(); + cumulative_energy["component_mapping_error_j"] = + static_cast(applied_node_energy_.sum() - applied_activity_j_); + return {{"type", "ADVANCE"}, + {"from_ns", initial_step * options_.step_ns}, + {"time_ns", current_step_ * options_.step_ns}, + {"steps", current_step_ - initial_step}, + {"entity_temperatures_k", std::move(entities)}, + {"sensor_temperatures_k", std::move(sensors)}, + {"temperature_range_k", {temperature_.minCoeff(), temperature_.maxCoeff()}}, + {"energy_j", + {{"window", std::move(window_energy)}, + {"cumulative", std::move(cumulative_energy)}, + {"accepted_activity", static_cast(accepted_activity_j_)}, + {"accepted_not_yet_applied_activity", + static_cast(accepted_activity_j_ - applied_activity_j_)}}}, + {"timing", {{"advance_seconds", advance_seconds}, + {"observation_seconds", observation_seconds}}}, + {"factorization_count", 1}}; + } + + Json quit() const { + require(intervals_.empty() && + std::abs(static_cast(accepted_activity_j_ - applied_activity_j_)) <= + 1e-12 * std::max(1.0, std::abs(static_cast(accepted_activity_j_))), + "QUIT would discard accepted but unapplied ENERGY"); + Json energy = energy_json(applied_activity_j_, static_input_j_, + stored_energy(), boundary_loss_j_); + energy["node_mapped_activity_input_j"] = applied_node_energy_.sum(); + energy["component_mapping_error_j"] = + static_cast(applied_node_energy_.sum() - applied_activity_j_); + return {{"type", "BYE"}, {"time_ns", current_step_ * options_.step_ns}, + {"energy_j", std::move(energy)}}; + } + + private: + void verify_hash(const std::string& text, const std::string& expected, + const std::string& label) { + require(expected.size() == 64 && + std::all_of(expected.begin(), expected.end(), [](unsigned char value) { + return (value >= '0' && value <= '9') || (value >= 'a' && value <= 'f'); + }), label + " SHA-256 must be lowercase hexadecimal"); + require(sha256(text) == expected, label + " SHA-256 mismatch"); + } + + void parse_grid(const Json& grid) { + const auto& cells = grid.at("cells"); + require(cells.is_array() && cells.size() == config_.nodes.size(), + "grid/model node count mismatch"); + cell_volumes_.resize(cells.size()); + for (std::size_t index = 0; index < cells.size(); ++index) { + const auto& cell = cells.at(index); + require(cell.at("index").get() == index && + cell.at("id").get() == config_.nodes[index].id, + "grid/model node ordering mismatch"); + const double volume = cell.at("volume_m3").get(); + const double capacity = cell.at("capacity_j_k").get(); + require(std::isfinite(volume) && volume > 0, "invalid grid cell volume"); + require(std::isfinite(capacity) && + std::abs(capacity - config_.nodes[index].heat_capacity_j_per_k) <= + 1e-12 * std::max(1.0, std::abs(capacity)), + "grid/model capacity mismatch"); + cell_volumes_[index] = volume; + } + std::vector ownership(cells.size()); + for (const auto& [id, indices] : grid.at("component_cells").items()) { + if (id == "__background__") continue; + Component component; + component.id = id; + double volume = 0; + for (const auto& item : indices) { + const std::size_t index = item.get(); + require(index < cells.size(), "component cell index outside grid"); + require(cells.at(index).at("component") == id, + "component ownership disagrees with grid cell"); + require(++ownership[index] == 1, "grid cell has duplicate component ownership"); + component.cells.push_back(index); + volume += cell_volumes_[index]; + } + require(!component.cells.empty() && std::isfinite(volume) && volume > 0, + "empty or invalid component mapping"); + for (const auto index : component.cells) + component.volume_weights.push_back(cell_volumes_[index] / volume); + component_index_.emplace(id, components_.size()); + components_.push_back(std::move(component)); + } + require(!components_.empty(), "grid contains no observable entities"); + } + + void parse_sensors(const Json& definitions) { + require(definitions.is_array(), "sensor mapping must be an array"); + std::set ids; + for (const auto& definition : definitions) { + Sensor sensor; + sensor.id = definition.at("id").get(); + require(!sensor.id.empty() && ids.insert(sensor.id).second, + "duplicate or empty sensor id"); + const std::string reduction = definition.at("reduction").get(); + if (reduction == "max") { + sensor.maximum = true; + for (const auto& item : definition.at("cell_indices")) + sensor.cells.push_back(item.get()); + } else if (reduction == "weighted_mean") { + double total = 0; + for (const auto& item : definition.at("cell_weights")) { + require(item.is_array() && item.size() == 2, + "sensor cell weight must be [index,weight]"); + sensor.cells.push_back(item.at(0).get()); + const double weight = item.at(1).get(); + require(std::isfinite(weight) && weight >= 0, "invalid sensor weight"); + sensor.weights.push_back(weight); + total += weight; + } + require(std::abs(total - 1.0) <= 1e-9, "sensor weights do not sum to one"); + } else { + fail("unsupported sensor reduction: " + reduction); + } + require(!sensor.cells.empty(), "empty sensor mapping"); + for (const auto index : sensor.cells) + require(index < config_.nodes.size(), "sensor cell outside model"); + sensors_.push_back(std::move(sensor)); + } + } + + void advance_one() { + const Tick trial_step = current_step_ + 1; + Eigen::VectorXd power = Eigen::VectorXd::Zero(theta_.size()); + long double trial_activity = 0; + for (const auto& interval : intervals_) { + if (interval.start_step > current_step_ || interval.end_step <= current_step_) continue; + const double duration_s = + static_cast(interval.end_step - interval.start_step) * step_s_; + const double component_power = interval.energy_j / duration_s; + const auto& component = components_[interval.component]; + for (std::size_t position = 0; position < component.cells.size(); ++position) + power[static_cast(component.cells[position])] += + component_power * component.volume_weights[position]; + trial_activity += component_power * step_s_; + } + const Eigen::VectorXd rhs = + capacity_over_dt_.cwiseProduct(theta_) + static_and_boundary_ + power; + const Eigen::VectorXd trial_theta = solver_.solve(rhs); + if (solver_.info() != Eigen::Success || !trial_theta.allFinite()) + throw_trial("NUMERICAL_FAILURE", "sparse LDLT solve failed", nullptr, power, + trial_activity, trial_step); + const Eigen::VectorXd trial_temperature = trial_theta.array() + origin_k_; + if (!trial_temperature.allFinite() || trial_temperature.minCoeff() < options_.minimum_k || + trial_temperature.maxCoeff() > options_.maximum_k) + throw_trial("DOMAIN_FAILURE", "temperature left required domain", + &trial_temperature, power, trial_activity, trial_step); + theta_ = trial_theta; + temperature_ = trial_temperature; + applied_node_energy_ += power * step_s_; + applied_activity_j_ += trial_activity; + static_input_j_ += static_power_w_ * step_s_; + for (Eigen::Index index = 0; index < theta_.size(); ++index) { + const auto& node = config_.nodes[static_cast(index)]; + boundary_loss_j_ += static_cast(step_s_) * + node.boundary_conductance_w_per_k * + (static_cast(theta_[index]) + + (origin_k_ - node.boundary_temperature_k)); + } + current_step_ = trial_step; + } + + [[noreturn]] void throw_trial(const std::string& status, const std::string& reason, + const Eigen::VectorXd* trial, + const Eigen::VectorXd& power, + long double trial_activity, Tick trial_step) const { + Json completed_energy = energy_json(applied_activity_j_, static_input_j_, + stored_energy(), boundary_loss_j_); + completed_energy["node_mapped_activity_input_j"] = applied_node_energy_.sum(); + completed_energy["component_mapping_error_j"] = + static_cast(applied_node_energy_.sum() - applied_activity_j_); + Json evidence = {{"type", "ERROR"}, {"status", status}, {"reason", reason}, + {"failure_returned_to_caller", true}, + {"temperature_clamping", false}, + {"last_valid_time_ns", current_step_ * options_.step_ns}, + {"trial_target_time_ns", trial_step * options_.step_ns}, + {"last_valid_temperature_range_k", + {temperature_.minCoeff(), temperature_.maxCoeff()}}, + {"completed_energy_j", std::move(completed_energy)}, + {"failed_trial_step_energy", + {{"status", "UNKNOWN_NOT_INTEGRATED"}, + {"activity_input_j", nullptr}, {"static_input_j", nullptr}, + {"stored_energy_change_j", nullptr}, {"boundary_loss_j", nullptr}, + {"energy_residual_j", nullptr}}}, + {"trial_declared_activity_energy_j", static_cast(trial_activity)}, + {"trial_activity_power_w", power.sum()}, + {"domain_k", {options_.minimum_k, options_.maximum_k}}}; + if (trial) { + evidence["trial_temperature_range_k"] = {trial->minCoeff(), trial->maxCoeff()}; + evidence["offending_nodes"] = Json::array(); + for (Eigen::Index index = 0; index < trial->size(); ++index) { + const double value = (*trial)[index]; + if (!std::isfinite(value) || value < options_.minimum_k || value > options_.maximum_k) + evidence["offending_nodes"].push_back( + {{"index", index}, + {"id", config_.nodes[static_cast(index)].id}, + {"temperature_k", std::isfinite(value) ? Json(value) : Json(nullptr)}}); + } + } else { + evidence["trial_temperature_range_k"] = nullptr; + evidence["offending_nodes"] = nullptr; + } + throw TrialFailure(std::move(evidence)); + } + + Json entity_temperatures() const { + Json result = Json::object(); + for (const auto& component : components_) { + double mean = 0; + double hotspot = -std::numeric_limits::infinity(); + for (std::size_t position = 0; position < component.cells.size(); ++position) { + const double value = temperature_[static_cast(component.cells[position])]; + mean += value * component.volume_weights[position]; + hotspot = std::max(hotspot, value); + } + result[component.id] = {{"mean_k", mean}, {"hotspot_k", hotspot}}; + } + return result; + } + + Json sensor_temperatures() const { + Json result = Json::object(); + for (const auto& sensor : sensors_) { + double value = sensor.maximum ? -std::numeric_limits::infinity() : 0; + for (std::size_t position = 0; position < sensor.cells.size(); ++position) { + const double temperature = + temperature_[static_cast(sensor.cells[position])]; + if (sensor.maximum) value = std::max(value, temperature); + else value += temperature * sensor.weights[position]; + } + result[sensor.id] = value; + } + return result; + } + + long double stored_energy() const { + long double result = 0; + for (Eigen::Index index = 0; index < theta_.size(); ++index) { + const auto& node = config_.nodes[static_cast(index)]; + result += node.heat_capacity_j_per_k * + (static_cast(theta_[index]) - + (node.initial_temperature_k - origin_k_)); + } + return result; + } + + Json energy_json(long double activity, long double static_input, + long double stored, long double boundary) const { + return {{"activity_input_j", static_cast(activity)}, + {"static_input_j", static_cast(static_input)}, + {"total_input_j", static_cast(activity + static_input)}, + {"stored_energy_change_j", static_cast(stored)}, + {"boundary_loss_j", static_cast(boundary)}, + {"energy_residual_j", + static_cast(activity + static_input - stored - boundary)}}; + } + + public: + class TrialFailure : public std::exception { + public: + explicit TrialFailure(Json evidence) : evidence_(std::move(evidence)) {} + const char* what() const noexcept override { return "thermal trial failed"; } + const Json& evidence() const noexcept { return evidence_; } + private: + Json evidence_; + }; + + private: + Options options_; + std::string model_text_, grid_text_, sensors_text_; + ThermalModelConfig config_; + double step_s_{}, origin_k_{}, static_power_w_{}, factor_seconds_{}; + SparseMatrix matrix_; + SparseSolver solver_; + Eigen::VectorXd temperature_, theta_, capacity_over_dt_, static_and_boundary_; + Eigen::VectorXd applied_node_energy_; + std::vector cell_volumes_; + std::vector components_; + std::map component_index_; + std::vector sensors_; + std::vector intervals_; + std::set> interval_identities_; + Tick current_step_{}; + long double accepted_activity_j_{}, applied_activity_j_{}, static_input_j_{}; + long double boundary_loss_j_{}; +}; + +double write_response(Json response, double previous_write_seconds, + double input_parse_seconds = 0) { + response["timing"]["input_parse_seconds"] = input_parse_seconds; + response["timing"]["previous_stdout_write_seconds"] = previous_write_seconds; + const auto serialization_start = Clock::now(); + (void)response.dump(); + response["timing"]["serialization_probe_seconds"] = + seconds(Clock::now() - serialization_start); + const std::string line = response.dump(); + const auto write_start = Clock::now(); + std::cout << line << '\n' << std::flush; + if (!std::cout) throw std::runtime_error("stdout write failed"); + return seconds(Clock::now() - write_start); +} + +} // namespace + +int main(int argc, char** argv) try { + const Options options = parse_options(argc, argv); + Service service(options, read_file(options.model_path), read_file(options.grid_path), + read_file(options.sensors_path)); + double previous_write = write_response(service.ready(), 0); + for (std::string line; std::getline(std::cin, line);) { + if (line.empty()) continue; + const auto parse_start = Clock::now(); + std::istringstream input(line); + std::string command; + input >> command; + Json response; + bool done = false; + double parse_seconds = 0; + if (command == "ENERGY") { + std::string start, end, component, energy, extra; + require(static_cast(input >> start >> end >> component >> energy) && !(input >> extra), + "ENERGY syntax: ENERGY START_NS END_NS COMPONENT ENERGY_J"); + parse_seconds = seconds(Clock::now() - parse_start); + response = service.energy(unsigned_value(start, "ENERGY start"), + unsigned_value(end, "ENERGY end"), component, + finite_nonnegative(energy, "ENERGY joules")); + } else if (command == "ADVANCE") { + std::string target, extra; + require(static_cast(input >> target) && !(input >> extra), + "ADVANCE syntax: ADVANCE TARGET_NS"); + parse_seconds = seconds(Clock::now() - parse_start); + response = service.advance(unsigned_value(target, "ADVANCE target")); + } else if (command == "QUIT") { + std::string extra; + require(!(input >> extra), "QUIT takes no arguments"); + parse_seconds = seconds(Clock::now() - parse_start); + response = service.quit(); + done = true; + } else { + fail("unknown protocol command: " + command); + } + previous_write = write_response(std::move(response), previous_write, parse_seconds); + if (done) return 0; + } + fail("stdin ended without QUIT"); +} catch (const Service::TrialFailure& failure) { + try { + (void)write_response(failure.evidence(), 0); + } catch (...) { + } + return 2; +} catch (const std::exception& error) { + try { + (void)write_response({{"type", "ERROR"}, {"status", "INPUT_OR_PROTOCOL_FAILURE"}, + {"reason", error.what()}, {"state_commit", "NO_TRIAL_COMMIT"}}, 0); + } catch (...) { + } + std::cerr << "eq3_maintenance_thermal_service: " << error.what() << '\n'; + return 2; +} diff --git a/experiments/eq3_maintenance/thermal_client.py b/experiments/eq3_maintenance/thermal_client.py new file mode 100644 index 0000000..73e4779 --- /dev/null +++ b/experiments/eq3_maintenance/thermal_client.py @@ -0,0 +1,123 @@ +"""Owned experimental thermal process plus explicit research guard semantics.""" +from __future__ import annotations +import hashlib +import json +import os +from pathlib import Path +import select +import subprocess +import time + + +class ThermalService: + def __init__(self, binary, model_dir, directory, *, window_ns=20000000, + timeout=180, limits=None, action_delay_ns=20000000, recovery_dwell_ns=100000000, + artifact_root=None, baseline_budgets=None, light_fraction=0.5): + self.directory=Path(directory);self.directory.mkdir(parents=True,exist_ok=False) + self.binary=Path(binary).resolve(strict=True) + if artifact_root is not None and not self.binary.is_relative_to(Path(artifact_root).resolve()): + raise ValueError('thermal binary outside explicit experimental artifact root') + self.timeout=timeout;self.buffer=b'';self.now=0;self.window=window_ns + self.action_delay=action_delay_ns;self.dwell=recovery_dwell_ns + self.limits=limits or {'hbf':(353.15,363.15,378.15),'hbm':(353.15,363.15,378.15),'gpu':(363.15,373.15,383.15)} + self.states={};self.pending={};self.last_change={};self.hysteresis_k=2.0 + self.baseline_budgets=dict(baseline_budgets or {});self.light_fraction=light_fraction + self.transcript=(self.directory/'thermal-transcript.jsonl').open('w') + self.stderr=(self.directory/'thermal-stderr.log').open('w') + self.argv=[str(self.binary)] + names={'model':'model.txt','grid':'rc_grid.json','sensors':'rc_sensors.json'} + self.lock={} + for key,name in names.items(): + p=Path(model_dir)/name + digest=hashlib.sha256(p.read_bytes()).hexdigest() + self.lock[name]=digest + self.argv += ['--'+key,str(p),'--'+key+'-sha256',digest] + self.argv += ['--step-ns',str(window_ns),'--min-k','300','--max-k','400'] + self.process=subprocess.Popen(self.argv,stdin=subprocess.PIPE,stdout=subprocess.PIPE,stderr=self.stderr,bufsize=0) + try: + self.header=self.read() + if self.header.get('type')!='READY':raise ValueError('thermal startup did not return READY') + except BaseException: + self.close();raise + + def read(self): + deadline=time.monotonic()+self.timeout + while b'\n' not in self.buffer: + if not select.select([self.process.stdout],[],[],max(0,deadline-time.monotonic()))[0]: + raise TimeoutError('thermal service response timeout') + chunk=os.read(self.process.stdout.fileno(),65536) + if not chunk:raise RuntimeError('thermal service closed without response') + self.buffer+=chunk + if len(self.buffer)>16*1024*1024:raise ValueError('thermal response exceeds bound') + line,self.buffer=self.buffer.split(b'\n',1) + result=json.loads(line) + self.transcript.write(json.dumps({'response':result})+'\n');self.transcript.flush() + if result.get('type')=='ERROR':raise RuntimeError('thermal failure preserved in transcript: '+str(result)) + return result + + def command(self, text): + self.transcript.write(json.dumps({'command':text})+'\n');self.transcript.flush() + self.process.stdin.write((text+'\n').encode());self.process.stdin.flush() + return self.read() + + def _guard(self, entity_temperatures, time_ns): + temperatures={} + for entity, values in entity_temperatures.items(): + owner=entity.split('.')[0] + if owner=='gpu' or owner.startswith(('hbf','hbm')): + temperatures[owner]=max(temperatures.get(owner,float('-inf')),values['hotspot_k']) + names=('normal','light','severe','shutdown') + for stack,temp in temperatures.items(): + prefix='gpu' if stack=='gpu' else stack[:3] + thresholds=self.limits[prefix] + current=self.states.setdefault(stack,0) + desired=sum(temp>=limit for limit in thresholds) + if desired=thresholds[current-1]-self.hysteresis_k: + desired=current + if desired==current: + self.pending.pop(stack,None) + continue + previous=self.pending.get(stack) + if previous is None or previous[0]!=desired: + self.pending[stack]=(desired,time_ns+self.action_delay) + desired,effective=self.pending[stack] + recovery_ready=desired>current or time_ns-self.last_change.get(stack,0)>=self.dwell + if time_ns>=effective and recovery_ready: + self.states[stack]=desired;self.last_change[stack]=time_ns;self.pending.pop(stack) + return temperatures,{stack:names[state] for stack,state in self.states.items()} + + def advance(self,start_ns,end_ns,component_energy_j): + if start_ns!=self.now or end_ns-start_ns!=self.window: + raise ValueError('thermal client requires contiguous fixed windows') + for component,energy in sorted(component_energy_j.items()): + if energy: + self.command(f'ENERGY {start_ns} {end_ns} {component} {energy:.17g}') + result=self.command(f'ADVANCE {end_ns}') + self.now=end_ns + temperatures,states=self._guard(result['entity_temperatures_k'],end_ns) + # The shared compute-die research guard constrains future package traffic. + if states.get('gpu') in ('light','severe','shutdown'): + rank={'normal':0,'light':1,'severe':2,'shutdown':3} + for stack in states: + if rank[states[stack]] int: + if isinstance(value, bool) or not isinstance(value, int): + raise ValueError(f"{name} must be an integer") + if value < (0 if allow_zero else 1): + raise ValueError(f"{name} must be {'non-negative' if allow_zero else 'positive'}") + return value + + +def _catalog() -> dict[str, Any]: + data = json.loads(CATALOG_PATH.read_text(encoding="utf-8")) + return data["models"] + + +def model_metadata(model_id: str) -> dict[str, Any]: + try: + return json.loads(json.dumps(_catalog()[model_id])) + except KeyError as error: + raise ValueError(f"unknown model_id {model_id!r}") from error + + +def _logical_regions(metadata: dict[str, Any]) -> list[dict[str, Any]]: + """Derive exact BF16 tensor bytes in a documented logical packing order.""" + arch = metadata["architecture"] + scalar = metadata["bytes_per_tensor_element"] + hidden = arch["hidden_size"] + intermediate = arch["intermediate_size"] + layers = arch["num_hidden_layers"] + heads = arch["num_attention_heads"] + kv_heads = arch["num_key_value_heads"] + vocab = arch["vocab_size"] + if hidden % heads: + raise ValueError("hidden size must be divisible by attention heads") + head_dim = hidden // heads + + definitions: list[tuple[str, int, str]] = [ + ("model.embed_tokens", vocab * hidden * scalar, "embedding"), + ] + for layer in range(layers): + prefix = f"model.layers.{layer}" + definitions.extend(( + (prefix + ".input_layernorm", hidden * scalar, "layer_norm"), + (prefix + ".self_attn", + (hidden * hidden + hidden + + 2 * (hidden * kv_heads * head_dim + kv_heads * head_dim) + + hidden * hidden) * scalar, + "attention"), + (prefix + ".post_attention_layernorm", hidden * scalar, "layer_norm"), + (prefix + ".mlp", 3 * hidden * intermediate * scalar, "mlp"), + )) + definitions.extend(( + ("model.norm", hidden * scalar, "final_norm"), + ("lm_head", vocab * hidden * scalar, "output_embedding"), + )) + + regions = [] + cursor = 0 + for identity, size, kind in definitions: + regions.append({ + "id": identity, + "kind": kind, + "start_byte": cursor, + "end_byte": cursor + size, + "size_bytes": size, + }) + cursor += size + if cursor != metadata["tensor_payload_bytes"]: + raise ValueError( + f"architecture-derived tensor bytes {cursor} do not match official " + f"metadata.total_size {metadata['tensor_payload_bytes']}" + ) + return regions + + +def _stack_page_counts(page_count: int, stacks: int) -> list[int]: + quotient, remainder = divmod(page_count, stacks) + return [quotient + (index < remainder) for index in range(stacks)] + + +def weight_extent(model_id: str, page_bytes: int = 16384, stacks: int = 4) -> dict[str, Any]: + """Return full-model page extent, logical regions, and static stripe layout.""" + page_bytes = _positive_integer(page_bytes, "page_bytes") + stacks = _positive_integer(stacks, "stacks") + if stacks not in {4, 8}: + raise ValueError("stacks must be 4 or 8 for the current EQ3 topologies") + metadata = model_metadata(model_id) + payload = _positive_integer(metadata["tensor_payload_bytes"], "tensor_payload_bytes") + pages = math.ceil(payload / page_bytes) + regions = _logical_regions(metadata) + for region in regions: + region["first_global_page"] = region["start_byte"] // page_bytes + region["last_global_page_inclusive"] = (region["end_byte"] - 1) // page_bytes + region["overlapping_page_count"] = ( + region["last_global_page_inclusive"] - region["first_global_page"] + 1 + ) + return { + "schema_version": "eq3-weight-extent-v1", + "model_id": model_id, + "revision": metadata["revision"], + "resolved_commit": metadata["resolved_commit"], + "dtype": metadata["dtype"], + "tensor_payload_bytes": payload, + "page_bytes": page_bytes, + "global_page_count": pages, + "allocated_page_bytes": pages * page_bytes, + "last_page_padding_bytes": pages * page_bytes - payload, + "stacks": stacks, + "placement": "global_page_modulo_stack_static_stripe", + "stack_page_counts": { + f"hbf{index}": count + for index, count in enumerate(_stack_page_counts(pages, stacks)) + }, + "mapping": { + "stack_index": "global_page % stacks", + "local_page": "global_page // stacks", + }, + "logical_region_packing": "DERIVED_FROM_DOCUMENTED_QWEN2_ARCHITECTURE", + "regions": regions, + "sources": metadata["sources"], + "limitations": [ + "logical regions are reconstructable derived packing, not safetensors file offsets", + "token/s and inference-stage causality are UNAVAILABLE", + ], + } + + +def _selected_page_range(extent: dict[str, Any], region: str | None) -> tuple[int, int]: + if region is None: + return 0, extent["global_page_count"] + matches = [ + item for item in extent["regions"] + if item["id"] == region or item["id"].startswith(region + ".") + ] + if not matches: + raise ValueError(f"unknown logical region {region!r}") + first = min(item["first_global_page"] for item in matches) + end = max(item["last_global_page_inclusive"] for item in matches) + 1 + return first, end + + +def generate_weight_requests( + model_id: str, + request_count: int, + start_page: int, + period_ns: int, + *, + page_bytes: int = 16384, + stacks: int = 4, + start_time_ns: int = 0, + region: str | None = None, + request_id_start: int = 0, +) -> dict[str, Any]: + """Generate deterministic read requests in the complete global page space. + + ``region`` can create a layer/region hotspot. ``start_page`` is still a + global model page and must lie in that region; returned addresses are never + rebased to zero. Requests wrap only at the selected full-model/region + boundary. ``scan_index`` records the canonical range cycle containing the + request; aggregate equivalent scans are reported separately. + """ + request_count = _positive_integer(request_count, "request_count", allow_zero=True) + start_page = _positive_integer(start_page, "start_page", allow_zero=True) + period_ns = _positive_integer(period_ns, "period_ns") + start_time_ns = _positive_integer(start_time_ns, "start_time_ns", allow_zero=True) + request_id_start = _positive_integer(request_id_start, "request_id_start", allow_zero=True) + extent = weight_extent(model_id, page_bytes, stacks) + range_start, range_end = _selected_page_range(extent, region) + if not range_start <= start_page < range_end: + raise ValueError( + f"start_page {start_page} outside selected global range [{range_start}, {range_end})" + ) + span = range_end - range_start + regions = extent["regions"] + requests = [] + unique_pages: set[int] = set() + unique_valid_bytes = 0 + for index in range(request_count): + sequence_offset = start_page - range_start + index + global_page = range_start + sequence_offset % span + scan_index = sequence_offset // span + stack_index = global_page % stacks + byte_address = global_page * page_bytes + valid_bytes = min(page_bytes, extent["tensor_payload_bytes"] - byte_address) + overlap = [ + item["id"] for item in regions + if item["start_byte"] < byte_address + valid_bytes + and byte_address < item["end_byte"] + ] + requests.append({ + "request_id": request_id_start + index, + "arrival_ns": start_time_ns + index * period_ns, + "operation": "read", + "bytes": page_bytes, + "valid_weight_bytes": valid_bytes, + "stack": f"hbf{stack_index}", + "global_page": global_page, + "global_byte_address": byte_address, + "local_page": global_page // stacks, + "scan_index": scan_index, + "logical_regions": overlap, + }) + if global_page not in unique_pages: + unique_pages.add(global_page) + unique_valid_bytes += valid_bytes + return { + "schema_version": "eq3-weight-read-requests-v1", + "model_id": model_id, + "operation_mix": {"read_fraction": 1.0, "write_fraction": 0.0}, + "address_space": "FULL_MODEL_GLOBAL_PAGES", + "selected_region": region, + "selected_global_page_range": [range_start, range_end], + "request_count": request_count, + "period_ns": period_ns, + "requests": requests, + "coverage": { + "unique_global_pages": len(unique_pages), + "unique_model_payload_bytes": unique_valid_bytes, + "model_payload_fraction": unique_valid_bytes / extent["tensor_payload_bytes"], + "equivalent_full_page_scans": request_count / extent["global_page_count"], + "equivalent_selected_range_scans": request_count / span, + }, + "extent_summary": { + key: extent[key] for key in ( + "tensor_payload_bytes", "page_bytes", "global_page_count", "stacks", + "placement", "stack_page_counts", "revision", "resolved_commit" + ) + }, + "limitations": [ + "read-only weight traffic; cache hits, activations, KV cache, and writes excluded", + "request period is an explicit scenario input, not a token-rate derivation", + "token/s is UNAVAILABLE", + ], + } + + +__all__ = ["generate_weight_requests", "model_metadata", "weight_extent"] diff --git a/experiments/eq3_rate_thermal/CONTROLLED_ANALYSIS.md b/experiments/eq3_rate_thermal/CONTROLLED_ANALYSIS.md new file mode 100644 index 0000000..f75081a --- /dev/null +++ b/experiments/eq3_rate_thermal/CONTROLLED_ANALYSIS.md @@ -0,0 +1,64 @@ +# Controlled campaign analysis + +`analyze_controlled_campaign.py` consumes completed `run_controlled.py` point +directories. By default it requires the frozen 39-point design: + +- 2 topologies × 3 model sizes × 2 demand patterns × 3 policies at 16 scans/s; +- 3 additional all-HBF, 235B, continuous points at 32 scans/s. + +Run it after the campaign is complete: + +```sh +python3 -B experiments/eq3_rate_thermal/analyze_controlled_campaign.py \ + --campaign CAMPAIGN_DIR --output NEW_DERIVED_DIR +``` + +`--allow-partial` is intended for diagnostic or fixed-fixture analysis and does +not label an incomplete campaign complete. + +When `CAMPAIGN_DIR/RUN_INDEX.json` exists, point discovery uses only its 39 +entries with `phase=main`. Every registered output must have a complete DONE +point contract, a unique path, and matching topology/model/pattern/strategy, +scan rate, duration, offered-byte total, and profile/workload/scenario hashes. +Pilot entries, stage DONE receipts and derived diagnostic links are excluded. +Without a RUN_INDEX, fake fixtures and isolated pilot bundles use recursive +discovery only for directories containing all eight required point files; a +standalone stage receipt therefore cannot become a point. + +The analysis rereads aligned `rates.jsonl`, `control.jsonl`, `energy.jsonl` and +`thermal.jsonl` streams. It validates per-point byte conservation, 50 pJ/B +served-energy accounting, component/window energy sums, the final thermal +energy receipt, and the byte-weighted delay histograms recorded by the runner. +It reports total and per-stack offered/delivered/backlog bytes, total-duration +and active-window delivery rates, peak/final temperatures, state residence +times, P95/P99 fluid delay, and incremental energy. + +Service stability uses two preregistered window sets: every complete 20 ms +window in the full active interval, and every complete 20 ms window in the +fixed second half of that active interval (10--20 s for the full campaign). +For total service and each stack it reports population CV, nearest-rank +P5/P50/P95, and the fraction of zero-service windows. These metrics expose both +stop/start delivery and steady but lower service; they add no PASS threshold +and no windows are selected from observed outcomes. + +Policy comparisons are paired within topology, model, pattern and scan rate. +Every delta is `right - left`; no sign is named a benefit. The separate +cross-topology table records: + +- 4-stack 16 scans/s versus 8-stack 16 scans/s as equal total offered demand; +- 4-stack 16 scans/s versus 8-stack 32 scans/s as equal offered bytes per stack. + +Nonzero backlog is a saturation result, not an execution failure. Backend +latency, token/s and maintenance remain unavailable because this campaign uses +the modelled fluid FIFO rather than MQSim or fabric completion. Energy is only +the user-confirmed incremental 50 pJ/B scenario; idle and GPU self-power remain +unknown in this path. + +Outputs include a machine-readable JSON report, point and comparison CSVs, one +maximum-temperature/rate/backlog trajectory plot per workload group, one +three-policy per-stack-temperature panel figure per workload group, and a +campaign summary figure. Rate trajectories show offered demand as a dashed +line beside delivered service, and all trajectory panels mark the fixed active +cutoff. The complete 39-point design therefore produces 13 grouped per-stack +figures rather than 39 separate large figures. All figures are derived from the +same checked in analysis and immutable point streams. diff --git a/experiments/eq3_rate_thermal/MODEL_WORKLOADS.md b/experiments/eq3_rate_thermal/MODEL_WORKLOADS.md new file mode 100644 index 0000000..f58207d --- /dev/null +++ b/experiments/eq3_rate_thermal/MODEL_WORKLOADS.md @@ -0,0 +1,59 @@ +# Model-size offered-byte workload + +`model_workloads.py` generates demand for the rate/fluid/thermal path. It does +not run MQSim, read model tensors, cap demand to service capacity, or estimate +tokens per second. + +The catalog freezes official metadata payload sizes: Qwen2.5-7B-Instruct is +15,231,233,024 B, Qwen2.5-72B-Instruct is 145,412,407,296 B, and pinned +Qwen3-235B-A22B is 470,187,269,120 B. The latter is a synthetic complete stored +weight scan. Its model card says 235B total and 22B active; it must not be read +as all 235B weights fetched for each MoE token. Token/s remains `UNKNOWN`. + +The validated input follows `model_workload_config.schema.json`. A full pilot +configuration is: + +```json +{ + "schema_version": "eq3-rate-model-workload-config-v1", + "model_id": "Qwen/Qwen2.5-72B-Instruct", + "full_scans_per_s": 16, + "pattern": "continuous", + "stack_count": 4, + "channels_per_stack": 16, + "step_ns": 20000000, + "active_ns": 8000000000, + "recovery_ns": 4000000000 +} +``` + +The full-duration form uses 20 s active plus 10 s recovery. The bounded pilot +uses 8 s plus 4 s. For the equal-mean burst form, set `pattern` to +`burst_equal_mean`, `burst_period_ns` to 200,000,000, and `burst_on_ns` to +100,000,000. Its on windows use twice the continuous demand and its off windows +use zero, preserving the same active-period mean. + +Generate a canonical JSON artifact with: + +```sh +python3 -B experiments/eq3_rate_thermal/model_workloads.py \ + --config CONFIG.json --output WORKLOAD.json +``` + +The output has contiguous 20 ms half-open windows. Every window contains all +4 or 8 `hbfN` stacks and channels `"0"` through `"15"` under +`stack_channel_offered_bytes`; recovery windows explicitly contain zero. +Bytes are nonnegative integers. A global exact fractional carry plus a rotating +uniform remainder preserves the requested complete-scan byte total while +keeping cumulative channel totals within one byte. + +The defining demand is + +```text +mean offered B/s = tensor_payload_bytes * full_scans_per_s +``` + +The generator deliberately applies no OCP/channel service cap. Offered demand +may exceed capacity and must enter the downstream fluid backlog. Any downstream +consumer that clips these bytes without accounting for the remainder violates +this interface. diff --git a/experiments/eq3_rate_thermal/README.md b/experiments/eq3_rate_thermal/README.md new file mode 100644 index 0000000..d8a1e21 --- /dev/null +++ b/experiments/eq3_rate_thermal/README.md @@ -0,0 +1,133 @@ +# Prescribed read-rate thermal inputs + +`rate_inputs.py` is a default-disconnected, pure converter. It turns an +externally prescribed HBF read-rate schedule into the component-energy windows +already accepted by the thermal service. It issues no NAND or HBM requests and +does not claim the requested rate was achieved by a backend. + +```python +from rate_inputs import build_windows + +inputs = build_windows(profile, schedule, normalized, step_ns=20_000_000) +``` + +The approved profile is explicit: + +```json +{ + "schema_version": 1, + "reference_read_Bps": 1600000000000.0, + "array_j_per_byte": 4e-11, + "base_j_per_byte": 1e-11, + "provenance": "SCENARIO_ASSUMPTION_USER_CONFIRMED" +} +``` + +This maps the 1.6 TB/s scenario endpoint to 64 W array plus 16 W base. These are +incremental read-energy assumptions, not measured product power. Idle power is +unknown and omitted. Rates above the envelope are rejected rather than clamped. + +The schedule uses integer-nanosecond, half-open, contiguous intervals covering +`[0, end_ns)`: + +```json +{ + "schema_version": 1, + "end_ns": 40000000, + "segments": [ + {"start_ns": 0, "end_ns": 20000000, + "read_Bps": {"hbf0": 384000000000.0}}, + {"start_ns": 20000000, "end_ns": 40000000, + "read_Bps": {"hbf0": 1600000000000.0}} + ] +} +``` + +An omitted discovered stack means zero prescribed read rate for that segment; +it does not mean zero physical idle power. Unknown stack IDs, gaps, overlaps, +negative or non-finite values, and rates above 1.6 TB/s fail explicitly. + +Powered HBF stacks, their base component, and their array-die components are +discovered from `normalized.components` using `physical_type`, `device_id`, and +`role`. No stack count, die count, or component-name pattern is assumed. Array +energy is uniform over each stack's discovered dies unless `die_weights` gives +an exact component map summing to one for that stack. Base energy is assigned +only to that stack's discovered `base_die`. + +For a channel-locality scenario, the profile may additionally declare an +explicit physical mapping and capacity for each named channel: + +```json +{ + "channel_map": {"hbf0": {"c0": "hbf0.die0", "c1": "hbf0.die1"}}, + "channel_capacity_Bps": {"hbf0": {"c0": 400000000000.0, + "c1": 400000000000.0}} +} +``` + +A segment can then include +`"channel_read_Bps":{"hbf0":{"c0":400000000000.0}}`. If the segment also +declares `read_Bps.hbf0`, its value must equal the channel sum. Each channel is +checked against its own capacity; unknown channels, stacks, or die components +fail and no value is clamped. Multiple channels may map to one die, in which +case their rates accumulate there. Array energy follows the mapped die rates, +while base energy uses the stack total once. Thus concentrated and distributed +channel activity can have equal total joules but different spatial sources. +`active_channel_count` counts positive prescribed channel rates only. Without +explicit channel rates, the older uniform/static-weight path remains available +and is labelled `UNIFORM_ASSUMED_CHANNEL_ACTIVITY_UNKNOWN`; it does not infer +which NAND channels were actually active. + +The output contains each window's `component_energy_j` plus per-stack requested +and modelled bytes, source energy, and window-average source power. It records +the entity mapping, weights, parameter provenance, prescribed-rate semantics, +and `ZERO_READ_ONLY_NOT_ZERO_IDLE`. A final partial window is supported; all +segment overlaps are integrated before energy is assigned. + +Run the small fixed validation without invoking a thermal solve: + +```sh +python3 -B experiments/eq3_rate_thermal/test_rate_inputs.py +``` + +## Current explicit OCP Grade 2 scenario + +The active user-selected interface profile has16channels/stack and96GB/s effective capacity/channel, totaling1.536TB/s. Source: OCP v0.7.0 p16 Tables2/4:64-bit×16GT/s×75%. This is the host-interface effective envelope, not a calibrated NAND bus, individual request simulator or guaranteed achieved throughput. A one-channel/one-thermal-die map is a separate explicit scenario assumption. + +The energy anchor remains80W at1.6TB/s,40pJ/B array+10pJ/B base. Thus full OCP Grade2 is76.8W, not80W; there is no hidden coefficient renormalization. Input schedules must obey each channel capacity, aggregate capacity and the approved energy envelope; excess demand is rejected as an invalid prescribed-delivery scenario rather than silently clamped. No queue or backpressure is simulated in this open-loop converter. + +`run_rate_thermal.py` consumes explicit profile/schedule/model/binary paths and writes energy windows, complete thermal entity frames, stack temperature CSV and DONE/FAILED receipts. It uses existing ThermalService ENERGY/ADVANCE on the unchanged full coupled network. No MQSim process is created. `plot_rate_thermal.py POINT` reproduces the rate/power/incremental-temperature figure. Initial300K is the model reference; without an idle/GPU background the output is incremental heating, not the product's absolute operating temperature. Existing 300..400K domain checks remain; failure evidence is retained. + +Input time segments are integrated exactly across20ms windows. Sources are per actual discovered die and base; temperature uses the existing2mm lateral grid and layered geometry. Unaccessed dies have no incremental read power but remain thermally coupled. Per-page/block/plane heat-source geometry, channel activation/static energy, UCIe scheduling and actual device throughput are not represented. + +## Fluid feedback runner + +`run_controlled.py` is a separate, default-disconnected closed-loop path. Each +20 ms window enqueues integer offered bytes, serves the persistent per-channel +FIFO under the budget chosen from the preceding completed thermal window, maps +only served bytes to 40/10 pJ/B array/base energy, and advances the unchanged +coupled thermal network. The resulting control decision applies to the next +window. Recovery windows admit zero new bytes while retaining and serving old +backlog. + +```sh +python3 experiments/eq3_rate_thermal/run_controlled.py \ + --profile PROFILE.json --model-dir MODEL_DIR --workload WORKLOAD.json \ + --scenario SCENARIO.json --thermal-binary THERMAL_SERVICE \ + --artifact-root ARTIFACT_ROOT --output NEW_OUTPUT \ + --strategy read_rate_feedback_thermal_guard_v1 --address-limit-gib 4 +``` + +The scenario schema is `controlled_scenario.schema.json`. Its explicit target +must equal `min(mean active offered B/s per stack, 0.8 * 1.536 TB/s)`. The +initial and maximum budget is the physical 1.536 TB/s window capacity; minimum +and step are 0.10 and 0.05 of that budget. Existing 20 ms action delay, 100 ms +recovery dwell, 2 K hysteresis and thermal thresholds are reused unchanged. + +All delivery, queue and delay facts are labelled `MODELLED_FLUID`. The reported +byte-weighted delay is quantized to thermal-window ends; backend latency remains +`UNKNOWN`. This path creates no MQSim request or fabric completion, has no +maintenance or endpoint-group arbitration, and does not infer retry, ECC, age, +retention, idle power, or GPU self power. It therefore supports only the +mixed-direct and all-HBF thermal studies; relay and DASH behavior remains a +separate capability gap. diff --git a/experiments/eq3_rate_thermal/analyze_controlled_campaign.py b/experiments/eq3_rate_thermal/analyze_controlled_campaign.py new file mode 100644 index 0000000..2ffe759 --- /dev/null +++ b/experiments/eq3_rate_thermal/analyze_controlled_campaign.py @@ -0,0 +1,783 @@ +#!/usr/bin/env python3 +"""Reproducible analysis and plots for the 36-point controlled-rate campaign.""" + +from __future__ import annotations + +import argparse +import csv +from itertools import combinations, zip_longest +import json +import math +from pathlib import Path +from typing import Any, Iterable + + +STRATEGIES = ( + "guard_only", + "thermal_hysteresis_guard", + "read_rate_feedback_thermal_guard_v1", +) +TOPOLOGIES = ("mixed_direct", "all_hbf_direct") +PATTERNS = ("continuous", "burst_equal_mean") +STATE_RANK = {"normal": 0, "light": 1, "severe": 2, "shutdown": 3} +EXPECTED_POINT_COUNT = 39 +J_PER_BYTE = 50e-12 +POINT_FILES = ( + "DONE.json", "manifest.json", "workload.json", "scenario.json", + "rates.jsonl", "control.jsonl", "energy.jsonl", "thermal.jsonl", +) + + +def _load(path: Path) -> dict: + value = json.loads(path.read_text()) + if not isinstance(value, dict): + raise ValueError(f"{path} must contain a JSON object") + return value + + +def _rows(path: Path) -> Iterable[dict]: + with path.open() as stream: + for line_number, line in enumerate(stream, 1): + if not line.strip(): + continue + value = json.loads(line) + if not isinstance(value, dict): + raise ValueError(f"{path}:{line_number} must contain a JSON object") + yield value + + +def _weighted_percentile(histogram: dict[int, int], percentile: int) -> int | None: + total = sum(histogram.values()) + if total == 0: + return None + rank = (percentile * total + 99) // 100 + cumulative = 0 + for delay, byte_count in sorted(histogram.items()): + cumulative += byte_count + if cumulative >= rank: + return delay + raise AssertionError("weighted percentile rank was not reached") + + +def _nearest_rank(values: list[float], percentile: int) -> float | None: + if not values: + return None + ordered = sorted(values) + rank = max(1, (percentile * len(ordered) + 99) // 100) + return ordered[rank - 1] + + +def _rate_stability(values: list[float]) -> dict: + if not values: + raise ValueError("stability interval has no complete windows") + mean = sum(values) / len(values) + variance = sum((value - mean) ** 2 for value in values) / len(values) + return { + "window_count": len(values), + "mean_Bps": mean, + "population_cv": math.sqrt(variance) / mean if mean > 0.0 else None, + "p5_Bps": _nearest_rank(values, 5), + "p50_Bps": _nearest_rank(values, 50), + "p95_Bps": _nearest_rank(values, 95), + "zero_service_window_fraction": sum(value == 0.0 for value in values) / len(values), + } + + +def _stability_intervals(trace: dict, stacks: list[str], active_ns: int) -> dict: + if active_ns % 2: + raise ValueError("active_ns must divide exactly for the fixed second-half interval") + midpoint = active_ns // 2 + if midpoint not in trace["start_ns"]: + raise ValueError("fixed active midpoint is not a service-window boundary") + definitions = { + "active_full": (0, active_ns), + "active_second_half": (midpoint, active_ns), + } + result = {} + for name, (start_ns, end_ns) in definitions.items(): + indices = [index for index, (start, end) in enumerate(zip(trace["start_ns"], trace["end_ns"])) + if start >= start_ns and end <= end_ns] + result[name] = { + "start_ns": start_ns, + "end_ns": end_ns, + "selection": "ALL_COMPLETE_20MS_WINDOWS_IN_FIXED_INTERVAL", + "total": _rate_stability([trace["delivered_Bps"][index] for index in indices]), + "per_stack": { + stack: _rate_stability( + [trace["delivered_Bps_by_stack"][stack][index] for index in indices] + ) for stack in stacks + }, + } + return result + + +def _finite(value: Any, path: str) -> float: + if isinstance(value, bool) or not isinstance(value, (int, float)): + raise ValueError(f"{path} must be numeric") + result = float(value) + if not math.isfinite(result): + raise ValueError(f"{path} must be finite") + return result + + +def _interval(row: dict, source: str) -> tuple[int, int]: + start = row.get("start_ns") + end = row.get("end_ns") + if isinstance(start, bool) or not isinstance(start, int) or isinstance(end, bool) or not isinstance(end, int): + raise ValueError(f"{source} interval must use integer ns") + if end <= start: + raise ValueError(f"{source} interval must be positive") + return start, end + + +def _histogram_from_done(done: dict, stack: str) -> dict[int, int]: + wait = done["summary"]["fluid_wait_by_stack"][stack] + histogram = {} + for row in wait.get("delivered_delay_histogram_bytes", []): + delay, byte_count = row.get("delay_ns"), row.get("bytes") + if not isinstance(delay, int) or not isinstance(byte_count, int) or delay < 0 or byte_count < 0: + raise ValueError("delay histogram must contain nonnegative integer ns/bytes") + histogram[delay] = histogram.get(delay, 0) + byte_count + if _weighted_percentile(histogram, 95) != wait.get("delivered_delay_p95_ns"): + raise ValueError(f"{stack} p95 differs from its byte histogram") + if _weighted_percentile(histogram, 99) != wait.get("delivered_delay_p99_ns"): + raise ValueError(f"{stack} p99 differs from its byte histogram") + return histogram + + +def analyze_point(point: Path) -> dict: + done = _load(point / "DONE.json") + manifest = _load(point / "manifest.json") + workload = _load(point / "workload.json") + scenario = _load(point / "scenario.json") + if done.get("execution_status") != "COMPLETED": + raise ValueError(f"{point} is not COMPLETED") + strategy = done.get("strategy") + if strategy not in STRATEGIES or manifest.get("strategy") != strategy: + raise ValueError(f"{point} has an unsupported or inconsistent strategy") + topology = scenario.get("topology") + model_id = workload.get("metadata", {}).get("model_id") + pattern = workload.get("metadata", {}).get("pattern") + full_scans_per_s = workload.get("metadata", {}).get("full_scans_per_s") + if (topology not in TOPOLOGIES or not isinstance(model_id, str) or pattern not in PATTERNS or + isinstance(full_scans_per_s, bool) or not isinstance(full_scans_per_s, int) or + full_scans_per_s <= 0): + raise ValueError(f"{point} has invalid topology/model/pattern identity") + stacks = workload.get("metadata", {}).get("stack_ids") + if not isinstance(stacks, list) or not stacks or any(not isinstance(x, str) for x in stacks): + raise ValueError(f"{point} workload must declare stack_ids") + active_ns = workload["metadata"].get("active_ns") + if not isinstance(active_ns, int) or active_ns <= 0: + raise ValueError(f"{point} workload must declare active_ns") + + per_stack = { + stack: {"offered_bytes": 0, "delivered_bytes": 0, "active_delivered_bytes": 0, + "final_backlog_bytes": 0, "peak_k": -math.inf, "final_k": None, + "state_time_ns": {state: 0 for state in STATE_RANK}} + for stack in stacks + } + totals = {"offered_bytes": 0, "delivered_bytes": 0, "active_delivered_bytes": 0, + "energy_j": 0.0} + any_stack_state_time_ns = {state: 0 for state in STATE_RANK} + trace = {"start_ns": [], "end_ns": [], "max_temperature_k": [], + "temperature_k_by_stack": {stack: [] for stack in stacks}, + "offered_Bps": [], "delivered_Bps": [], + "delivered_Bps_by_stack": {stack: [] for stack in stacks}, "backlog_bytes": []} + expected_start = 0 + final_thermal = None + final_rate = None + count = 0 + + streams = ( + _rows(point / "rates.jsonl"), + _rows(point / "control.jsonl"), + _rows(point / "energy.jsonl"), + _rows(point / "thermal.jsonl"), + ) + for index, aligned in enumerate(zip_longest(*streams)): + if any(row is None for row in aligned): + raise ValueError(f"{point} JSONL streams have different row counts") + rate, control, energy, thermal = aligned + interval = _interval(rate, "rates") + if interval[0] != expected_start: + raise ValueError(f"{point} rates are not contiguous") + for name, row in (("energy", energy), ("thermal", thermal)): + if _interval(row, name) != interval: + raise ValueError(f"{point} {name} interval differs from rates") + if (control.get("observed_window_start_ns"), control.get("observed_window_end_ns")) != interval: + raise ValueError(f"{point} control observation interval differs from rates") + duration_ns = interval[1] - interval[0] + expected_start = interval[1] + count += 1 + receipts = rate.get("stacks") + if not isinstance(receipts, dict) or set(receipts) != set(stacks): + raise ValueError(f"{point} rate stack coverage differs from workload") + window_offered = window_delivered = window_backlog = 0 + for stack in stacks: + receipt = receipts[stack] + offered = int(receipt["offered_bytes"]) + delivered = int(receipt["delivered_bytes"]) + backlog = int(receipt["backlog_bytes"]) + if min(offered, delivered, backlog) < 0: + raise ValueError("rate receipts contain negative bytes") + per_stack[stack]["offered_bytes"] += offered + per_stack[stack]["delivered_bytes"] += delivered + if interval[0] < active_ns: + per_stack[stack]["active_delivered_bytes"] += delivered + per_stack[stack]["final_backlog_bytes"] = backlog + window_offered += offered + window_delivered += delivered + window_backlog += backlog + if rate.get("window_arrived_bytes") != window_offered or rate.get("window_served_bytes") != window_delivered: + raise ValueError(f"{point} window byte totals differ from stack receipts") + totals["offered_bytes"] += window_offered + totals["delivered_bytes"] += window_delivered + if interval[0] < active_ns: + totals["active_delivered_bytes"] += window_delivered + + component_energy = energy.get("component_energy_j") + if not isinstance(component_energy, dict): + raise ValueError("energy row lacks component_energy_j") + window_energy = sum(_finite(value, "component energy") for value in component_energy.values()) + if not math.isclose(window_energy, _finite(energy.get("window_total_j"), "window_total_j"), + rel_tol=1e-12, abs_tol=1e-15): + raise ValueError(f"{point} component energy does not sum to window total") + if not math.isclose(window_energy, window_delivered * J_PER_BYTE, + rel_tol=1e-12, abs_tol=1e-15): + raise ValueError(f"{point} served bytes do not map to 50 pJ/B once") + totals["energy_j"] += window_energy + + temperatures = thermal.get("temperatures") + states = thermal.get("stack_states") + if not isinstance(temperatures, dict) or not isinstance(states, dict): + raise ValueError("thermal row lacks temperatures/stack_states") + if set(stacks) - set(temperatures) or set(stacks) - set(states): + raise ValueError("thermal row lacks a workload stack") + worst = "normal" + for stack in stacks: + temperature = _finite(temperatures[stack], f"temperature {stack}") + state = states[stack] + if state not in STATE_RANK: + raise ValueError(f"unknown thermal state {state!r}") + per_stack[stack]["peak_k"] = max(per_stack[stack]["peak_k"], temperature) + per_stack[stack]["final_k"] = temperature + per_stack[stack]["state_time_ns"][state] += duration_ns + if STATE_RANK[state] > STATE_RANK[worst]: + worst = state + any_stack_state_time_ns[worst] += duration_ns + trace["start_ns"].append(interval[0]) + trace["end_ns"].append(interval[1]) + trace["max_temperature_k"].append(max(float(temperatures[stack]) for stack in stacks)) + for stack in stacks: + trace["temperature_k_by_stack"][stack].append(float(temperatures[stack])) + trace["offered_Bps"].append(window_offered * 1_000_000_000 / duration_ns) + trace["delivered_Bps"].append(window_delivered * 1_000_000_000 / duration_ns) + for stack in stacks: + trace["delivered_Bps_by_stack"][stack].append( + int(receipts[stack]["delivered_bytes"]) * 1_000_000_000 / duration_ns + ) + trace["backlog_bytes"].append(window_backlog) + final_thermal, final_rate = thermal, rate + + workload_windows = workload.get("windows", []) + if count != len(workload_windows) or expected_start == 0: + raise ValueError(f"{point} JSONL count differs from workload") + duration_ns = expected_start + final_backlog = sum(row["final_backlog_bytes"] for row in per_stack.values()) + byte_error = totals["offered_bytes"] - totals["delivered_bytes"] - final_backlog + if byte_error != 0: + raise ValueError(f"{point} end-to-end byte conservation failed by {byte_error}") + expected_offered = sum(int(window["total_offered_bytes"]) for window in workload_windows) + if expected_offered != totals["offered_bytes"]: + raise ValueError(f"{point} workload and rate arrivals differ") + + aggregate_histogram = {} + for stack in stacks: + histogram = _histogram_from_done(done, stack) + if sum(histogram.values()) != per_stack[stack]["delivered_bytes"]: + raise ValueError(f"{stack} delay histogram bytes differ from delivered bytes") + for delay, byte_count in histogram.items(): + aggregate_histogram[delay] = aggregate_histogram.get(delay, 0) + byte_count + stack_result = per_stack[stack] + stack_result["mean_delivered_Bps_total_duration"] = ( + stack_result["delivered_bytes"] * 1_000_000_000 / duration_ns + ) + stack_result["mean_delivered_Bps_active_windows"] = ( + stack_result["active_delivered_bytes"] * 1_000_000_000 / active_ns + ) + stack_result["delivered_delay_p95_ns"] = _weighted_percentile(histogram, 95) + stack_result["delivered_delay_p99_ns"] = _weighted_percentile(histogram, 99) + + thermal_receipt_j = _finite(done["thermal_energy_receipt"]["total_input_j"], "thermal receipt") + done_energy_j = _finite(done["summary"]["energy_j"], "DONE energy") + energy_stream_error_j = totals["energy_j"] - totals["delivered_bytes"] * J_PER_BYTE + thermal_energy_error_j = thermal_receipt_j - totals["energy_j"] + if not math.isclose(done_energy_j, totals["energy_j"], rel_tol=1e-12, abs_tol=1e-12): + raise ValueError(f"{point} DONE and energy stream totals differ") + if abs(thermal_energy_error_j) > 1e-9 * max(1.0, totals["energy_j"]): + raise ValueError(f"{point} thermal receipt and energy stream differ") + + stability = _stability_intervals(trace, stacks, active_ns) + return { + "point_id": point.name, + "point_path": str(point.resolve()), + "topology": topology, + "model_id": model_id, + "weight_bytes": workload["metadata"].get("weight_bytes"), + "pattern": pattern, + "full_scans_per_s": full_scans_per_s, + "strategy": strategy, + "window_count": count, + "duration_ns": duration_ns, + "active_ns": active_ns, + "totals": { + **totals, + "final_backlog_bytes": final_backlog, + "mean_delivered_Bps_total_duration": totals["delivered_bytes"] * 1_000_000_000 / duration_ns, + "mean_delivered_Bps_active_windows": totals["active_delivered_bytes"] * 1_000_000_000 / active_ns, + "delivery_fraction": ( + totals["delivered_bytes"] / totals["offered_bytes"] if totals["offered_bytes"] else None + ), + "byte_conservation_error": byte_error, + "served_energy_error_j": energy_stream_error_j, + "thermal_energy_error_j": thermal_energy_error_j, + "byte_weighted_delay_p95_ns": _weighted_percentile(aggregate_histogram, 95), + "byte_weighted_delay_p99_ns": _weighted_percentile(aggregate_histogram, 99), + "peak_temperature_k": max(row["peak_k"] for row in per_stack.values()), + "final_peak_temperature_k": max(row["final_k"] for row in per_stack.values()), + }, + "per_stack": per_stack, + "any_stack_state_time_ns": any_stack_state_time_ns, + "service_rate_stability": stability, + "final_guard_states": done["summary"].get("final_guard_states"), + "semantics": done.get("semantics", {}), + "limitations": { + "backend_latency": "UNKNOWN", + "token_per_s": "UNKNOWN", + "maintenance": "UNAVAILABLE_IN_THIS_FLUID_PATH", + "service": "MODELLED_FLUID_BYTE_FIFO_NOT_MQSIM_OR_FABRIC_COMPLETION", + "energy": "INCREMENTAL_SERVED_BYTES_AT_USER_CONFIRMED_50_PJ_PER_BYTE_IDLE_UNKNOWN", + }, + "input_identity": manifest.get("input_sha256", {}), + "_trace": trace, + } + + +def _delta(right: Any, left: Any) -> float | int | None: + if right is None or left is None: + return None + return right - left + + +def pairwise_costs(points: list[dict]) -> list[dict]: + grouped = {} + for point in points: + key = (point["topology"], point["model_id"], point["pattern"], point["full_scans_per_s"]) + strategy = point["strategy"] + if strategy in grouped.setdefault(key, {}): + raise ValueError(f"duplicate strategy {strategy!r} for {key}") + grouped[key][strategy] = point + rows = [] + for key, by_strategy in sorted(grouped.items()): + for left_name, right_name in combinations(STRATEGIES, 2): + if left_name not in by_strategy or right_name not in by_strategy: + continue + left, right = by_strategy[left_name], by_strategy[right_name] + if left["input_identity"].get("workload.json") != right["input_identity"].get("workload.json"): + raise ValueError(f"paired policies use different workloads for {key}") + lt, rt = left["totals"], right["totals"] + rows.append({ + "topology": key[0], "model_id": key[1], "pattern": key[2], + "full_scans_per_s": key[3], + "left_strategy": left_name, "right_strategy": right_name, + "delta_semantics": "RIGHT_MINUS_LEFT_NO_BENEFIT_DIRECTION_ASSUMED", + "delivered_bytes_delta": _delta(rt["delivered_bytes"], lt["delivered_bytes"]), + "final_backlog_bytes_delta": _delta(rt["final_backlog_bytes"], lt["final_backlog_bytes"]), + "mean_delivered_Bps_delta": _delta(rt["mean_delivered_Bps_total_duration"], + lt["mean_delivered_Bps_total_duration"]), + "peak_temperature_k_delta": _delta(rt["peak_temperature_k"], lt["peak_temperature_k"]), + "final_peak_temperature_k_delta": _delta(rt["final_peak_temperature_k"], + lt["final_peak_temperature_k"]), + "byte_weighted_delay_p95_ns_delta": _delta(rt["byte_weighted_delay_p95_ns"], + lt["byte_weighted_delay_p95_ns"]), + "byte_weighted_delay_p99_ns_delta": _delta(rt["byte_weighted_delay_p99_ns"], + lt["byte_weighted_delay_p99_ns"]), + "energy_j_delta": _delta(rt["energy_j"], lt["energy_j"]), + "active_full_service_cv_delta": _delta( + right["service_rate_stability"]["active_full"]["total"]["population_cv"], + left["service_rate_stability"]["active_full"]["total"]["population_cv"], + ), + "active_second_half_service_cv_delta": _delta( + right["service_rate_stability"]["active_second_half"]["total"]["population_cv"], + left["service_rate_stability"]["active_second_half"]["total"]["population_cv"], + ), + "active_full_zero_service_fraction_delta": _delta( + right["service_rate_stability"]["active_full"]["total"]["zero_service_window_fraction"], + left["service_rate_stability"]["active_full"]["total"]["zero_service_window_fraction"], + ), + "light_or_worse_stack_ns_delta": _delta( + sum(rt_state for state, rt_state in right["any_stack_state_time_ns"].items() + if STATE_RANK[state] >= STATE_RANK["light"]), + sum(lt_state for state, lt_state in left["any_stack_state_time_ns"].items() + if STATE_RANK[state] >= STATE_RANK["light"]), + ), + }) + return rows + + +def cross_topology_pressure_costs(points: list[dict]) -> list[dict]: + """Compare the 235B continuous points under two explicit demand identities.""" + index = {(p["topology"], p["model_id"], p["pattern"], p["full_scans_per_s"], + p["strategy"]): p for p in points} + model = "Qwen/Qwen3-235B-A22B" + rows = [] + comparisons = ( + ("SAME_TOTAL_OFFERED_DEMAND", 16, 16), + ("SAME_PER_STACK_OFFERED_PRESSURE", 16, 32), + ) + for identity, mixed_scans, all_hbf_scans in comparisons: + for strategy in STRATEGIES: + left = index.get(("mixed_direct", model, "continuous", mixed_scans, strategy)) + right = index.get(("all_hbf_direct", model, "continuous", all_hbf_scans, strategy)) + if left is None or right is None: + continue + lt, rt = left["totals"], right["totals"] + left_per_stack = lt["offered_bytes"] / len(left["per_stack"]) + right_per_stack = rt["offered_bytes"] / len(right["per_stack"]) + if identity == "SAME_TOTAL_OFFERED_DEMAND" and lt["offered_bytes"] != rt["offered_bytes"]: + raise ValueError("same-total-demand comparison does not have equal offered bytes") + if identity == "SAME_PER_STACK_OFFERED_PRESSURE" and left_per_stack != right_per_stack: + raise ValueError("same-per-stack-pressure comparison does not have equal offered bytes/stack") + rows.append({ + "comparison_identity": identity, + "strategy": strategy, + "left": "mixed_direct:4stack:16scans_per_s", + "right": f"all_hbf_direct:8stack:{all_hbf_scans}scans_per_s", + "delta_semantics": "RIGHT_MINUS_LEFT_NO_BENEFIT_DIRECTION_ASSUMED", + "left_total_offered_bytes": lt["offered_bytes"], + "right_total_offered_bytes": rt["offered_bytes"], + "left_offered_bytes_per_stack": left_per_stack, + "right_offered_bytes_per_stack": right_per_stack, + "delivered_bytes_delta": _delta(rt["delivered_bytes"], lt["delivered_bytes"]), + "final_backlog_bytes_delta": _delta(rt["final_backlog_bytes"], lt["final_backlog_bytes"]), + "peak_temperature_k_delta": _delta(rt["peak_temperature_k"], lt["peak_temperature_k"]), + "energy_j_delta": _delta(rt["energy_j"], lt["energy_j"]), + "byte_weighted_delay_p99_ns_delta": _delta(rt["byte_weighted_delay_p99_ns"], + lt["byte_weighted_delay_p99_ns"]), + }) + return rows + + +def _is_point_directory(path: Path) -> bool: + return all((path / name).is_file() for name in POINT_FILES) + + +def _discover_point_dirs(campaign: Path) -> tuple[list[Path], list[dict] | None, str]: + """Discover only registered main points or validated standalone point dirs.""" + run_index_path = campaign / "RUN_INDEX.json" + if run_index_path.is_file(): + run_index = _load(run_index_path) + entries = run_index.get("points") + if not isinstance(entries, list): + raise ValueError("RUN_INDEX.points must be an array") + main_entries = [entry for entry in entries + if isinstance(entry, dict) and entry.get("phase") == "main"] + if run_index.get("main_count") != EXPECTED_POINT_COUNT or len(main_entries) != EXPECTED_POINT_COUNT: + raise ValueError( + f"RUN_INDEX must declare exactly {EXPECTED_POINT_COUNT} phase=main points" + ) + outputs = [] + for index, entry in enumerate(main_entries): + raw_output = entry.get("output") + if not isinstance(raw_output, str) or not raw_output: + raise ValueError(f"RUN_INDEX main point {index} has no output path") + output = Path(raw_output) + if not output.is_absolute(): + output = campaign / output + output = output.resolve() + if not _is_point_directory(output): + missing = [name for name in POINT_FILES if not (output / name).is_file()] + raise ValueError(f"RUN_INDEX main point {output} is incomplete; missing={missing}") + outputs.append(output) + if len(set(outputs)) != len(outputs): + raise ValueError("RUN_INDEX main point outputs must be unique") + return outputs, main_entries, "RUN_INDEX_PHASE_MAIN_WHITELIST" + + # Standalone fake fixtures and pilot-only diagnostic bundles do not have a + # RUN_INDEX. Require the complete point file contract so stage-level DONE + # receipts cannot be mistaken for experiment points. + outputs = sorted({path.parent.resolve() for path in campaign.rglob("DONE.json") + if _is_point_directory(path.parent)}) + if not outputs: + raise ValueError("no complete controlled-rate point directories found") + return outputs, None, "COMPLETE_POINT_FILE_DISCOVERY_WITHOUT_RUN_INDEX" + + +def _index_model_label(model_id: str) -> str: + labels = { + "Qwen/Qwen2.5-7B-Instruct": "7B", + "Qwen/Qwen2.5-72B-Instruct": "72B", + "Qwen/Qwen3-235B-A22B": "235B", + } + if model_id not in labels: + raise ValueError(f"no RUN_INDEX model label for {model_id!r}") + return labels[model_id] + + +def _validate_index_identity(entry: dict, point: dict) -> None: + checks = { + "point_id": point["point_id"], + "topology": point["topology"], + "model": _index_model_label(point["model_id"]), + "pattern": point["pattern"], + "strategy": point["strategy"], + "full_scans_per_s": point["full_scans_per_s"], + "expected_active_offered_bytes": point["totals"]["offered_bytes"], + "active_s": point["active_ns"] // 1_000_000_000, + "recovery_s": (point["duration_ns"] - point["active_ns"]) // 1_000_000_000, + } + for key, observed in checks.items(): + if entry.get(key) != observed: + raise ValueError( + f"RUN_INDEX identity {key} differs for {point['point_id']}: " + f"index={entry.get(key)!r}, point={observed!r}" + ) + output = Path(entry["output"]).resolve() + if output != Path(point["point_path"]).resolve(): + raise ValueError(f"RUN_INDEX output path differs for {point['point_id']}") + expected_hashes = entry.get("input_sha256") + if not isinstance(expected_hashes, dict): + raise ValueError(f"RUN_INDEX input hashes missing for {point['point_id']}") + actual_hashes = point["input_identity"] + for index_key, manifest_key in (("profile", "profile.json"), + ("workload", "workload.json"), + ("scenario", "scenario.json")): + if expected_hashes.get(index_key) != actual_hashes.get(manifest_key): + raise ValueError( + f"RUN_INDEX {index_key} input identity differs for {point['point_id']}" + ) + + +def analyze_campaign(campaign: Path, *, require_complete: bool = True) -> dict: + point_dirs, index_entries, discovery_mode = _discover_point_dirs(campaign) + points = [analyze_point(path) for path in point_dirs] + if index_entries is not None: + for entry, point in zip(index_entries, points): + _validate_index_identity(entry, point) + if require_complete: + if len(points) != EXPECTED_POINT_COUNT: + raise ValueError(f"expected {EXPECTED_POINT_COUNT} completed points, found {len(points)}") + design = {(p["topology"], p["model_id"], p["pattern"], p["full_scans_per_s"], + p["strategy"]) for p in points} + models = sorted({p["model_id"] for p in points}) + expected = {(t, m, w, 16, s) for t in TOPOLOGIES for m in models + for w in PATTERNS for s in STRATEGIES} + expected |= {("all_hbf_direct", "Qwen/Qwen3-235B-A22B", "continuous", 32, s) + for s in STRATEGIES} + if len(models) != 3 or design != expected: + raise ValueError("completed points do not form the frozen 2x3x2x3 design") + return { + "schema_version": "eq3-controlled-campaign-analysis-v1", + "campaign": str(campaign.resolve()), + "point_discovery": discovery_mode, + "completed_point_count": len(points), + "expected_point_count": EXPECTED_POINT_COUNT, + "design_complete": len(points) == EXPECTED_POINT_COUNT, + "points": points, + "pairwise_policy_costs": pairwise_costs(points), + "cross_topology_pressure_costs": cross_topology_pressure_costs(points), + "global_limitations": { + "backend_latency": "UNKNOWN", + "token_per_s": "UNKNOWN", + "maintenance": "UNAVAILABLE_IN_THIS_FLUID_PATH", + "backend": "NO_MQSIM_OR_FABRIC_COMPLETION_IN_THIS_CAMPAIGN", + "thermal": "CONDITIONAL_INCREMENTAL_READ_HEATING_IDLE_AND_GPU_SELF_POWER_UNKNOWN", + }, + } + + +def _serializable(analysis: dict) -> dict: + return { + **analysis, + "points": [{key: value for key, value in point.items() if not key.startswith("_")} + for point in analysis["points"]], + } + + +def _slug(value: str) -> str: + return "".join(character.lower() if character.isalnum() else "-" for character in value).strip("-") + + +def write_plots(analysis: dict, output: Path) -> list[str]: + import matplotlib + matplotlib.use("Agg") + import matplotlib.pyplot as plt + + output.mkdir(parents=True, exist_ok=True) + paths = [] + grouped = {} + for point in analysis["points"]: + grouped.setdefault((point["topology"], point["model_id"], point["pattern"], + point["full_scans_per_s"]), []).append(point) + for key, points in sorted(grouped.items()): + fig, axes = plt.subplots(3, 1, figsize=(10, 8), sharex=True, constrained_layout=True) + ordered_points = sorted(points, key=lambda row: STRATEGIES.index(row["strategy"])) + for point_index, point in enumerate(ordered_points): + trace = point["_trace"] + x = [value / 1e9 for value in trace["end_ns"]] + label = point["strategy"] + axes[0].plot(x, trace["max_temperature_k"], label=label) + axes[1].plot(x, [value / 1e12 for value in trace["delivered_Bps"]], label=label) + axes[2].plot(x, [value / 1e12 for value in trace["backlog_bytes"]], label=label) + if point_index == 0: + axes[1].plot(x, [value / 1e12 for value in trace["offered_Bps"]], + color="black", linestyle="--", linewidth=1.2, + label="offered demand") + active_end_s = ordered_points[0]["active_ns"] / 1e9 + for axis in axes: + axis.axvline(active_end_s, color="gray", linestyle=":", linewidth=1, + label="active end" if axis is axes[0] else None) + axes[0].set_ylabel("Max HBF K") + axes[1].set_ylabel("Rate (TB/s)") + axes[2].set_ylabel("Backlog TB") + axes[2].set_xlabel("Time (s)") + axes[0].legend(fontsize=8) + axes[1].legend(fontsize=8) + for axis in axes: + axis.grid(alpha=.2) + fig.suptitle(f"{key[0]} | {key[1]} | {key[2]} | {key[3]} scans/s\n" + "modelled fluid service; backend/token latency unavailable") + path = output / f"trajectory-{_slug(key[0])}-{_slug(key[1])}-{_slug(key[2])}-{key[3]}sps.png" + fig.savefig(path, dpi=150) + plt.close(fig) + paths.append(str(path)) + + # One compact figure per workload group: strategy panels, stack lines. + stack_fig, stack_axes = plt.subplots( + len(STRATEGIES), 1, figsize=(10, 8), sharex=True, sharey=True, + constrained_layout=True, + ) + by_strategy = {point["strategy"]: point for point in points} + for axis, strategy in zip(stack_axes, STRATEGIES): + point = by_strategy.get(strategy) + if point is None: + axis.text(.5, .5, "point unavailable", ha="center", va="center", + transform=axis.transAxes) + axis.set_title(strategy) + axis.axvline(active_end_s, color="gray", linestyle=":", linewidth=1) + continue + trace = point["_trace"] + x = [value / 1e9 for value in trace["end_ns"]] + for stack, temperatures in sorted(trace["temperature_k_by_stack"].items()): + axis.plot(x, temperatures, linewidth=1, label=stack) + axis.axvline(point["active_ns"] / 1e9, color="gray", linestyle=":", linewidth=1) + axis.set_title(strategy) + axis.set_ylabel("HBF K") + axis.grid(alpha=.2) + axis.legend(ncol=4, fontsize=7, frameon=False) + stack_axes[-1].set_xlabel("Time (s)") + stack_fig.suptitle( + f"Per-stack temperature | {key[0]} | {key[1]} | {key[2]} | {key[3]} scans/s" + ) + stack_path = output / ( + f"stack-temperatures-{_slug(key[0])}-{_slug(key[1])}-{_slug(key[2])}-{key[3]}sps.png" + ) + stack_fig.savefig(stack_path, dpi=150) + plt.close(stack_fig) + paths.append(str(stack_path)) + + points = sorted(analysis["points"], key=lambda p: (p["topology"], p["model_id"], p["pattern"], + p["full_scans_per_s"], + STRATEGIES.index(p["strategy"]))) + fig, axes = plt.subplots(3, 1, figsize=(14, 10), sharex=True, constrained_layout=True) + labels = [f"{p['topology']}\n{p['model_id'].split('/')[-1]}\n{p['pattern']} " + f"{p['full_scans_per_s']}sps\n{p['strategy']}" for p in points] + x = list(range(len(points))) + axes[0].bar(x, [p["totals"]["peak_temperature_k"] for p in points]) + axes[1].bar(x, [p["totals"]["delivery_fraction"] or 0.0 for p in points]) + axes[2].bar(x, [p["totals"]["final_backlog_bytes"] / 1e12 for p in points]) + axes[0].set_ylabel("Peak HBF K") + axes[1].set_ylabel("Delivered/offered") + axes[2].set_ylabel("Final backlog TB") + axes[2].set_xticks(x, labels, rotation=90, fontsize=6) + for axis in axes: + axis.grid(axis="y", alpha=.2) + fig.suptitle("Controlled-rate campaign summary (no policy benefit direction assumed)") + path = output / "controlled-campaign-summary.png" + fig.savefig(path, dpi=150) + plt.close(fig) + paths.append(str(path)) + return paths + + +def write_outputs(analysis: dict, output: Path, *, plots: bool = True) -> None: + output.mkdir(parents=True, exist_ok=False) + serializable = _serializable(analysis) + (output / "CONTROLLED_CAMPAIGN_ANALYSIS.json").write_text( + json.dumps(serializable, indent=2, sort_keys=True, allow_nan=False) + "\n" + ) + point_fields = [ + "point_id", "topology", "model_id", "pattern", "strategy", "offered_bytes", + "full_scans_per_s", + "delivered_bytes", "final_backlog_bytes", "mean_delivered_Bps_total_duration", + "delivery_fraction", "peak_temperature_k", "final_peak_temperature_k", + "byte_weighted_delay_p95_ns", "byte_weighted_delay_p99_ns", "energy_j", + "byte_conservation_error", "thermal_energy_error_j", + "active_full_service_cv", "active_second_half_service_cv", + "active_full_zero_service_fraction", "active_second_half_zero_service_fraction", + ] + with (output / "point-summary.csv").open("w", newline="") as stream: + writer = csv.DictWriter(stream, fieldnames=point_fields) + writer.writeheader() + for point in analysis["points"]: + row = {key: point.get(key, point["totals"].get(key)) for key in point_fields} + row.update({ + "active_full_service_cv": point["service_rate_stability"]["active_full"]["total"]["population_cv"], + "active_second_half_service_cv": point["service_rate_stability"]["active_second_half"]["total"]["population_cv"], + "active_full_zero_service_fraction": point["service_rate_stability"]["active_full"]["total"]["zero_service_window_fraction"], + "active_second_half_zero_service_fraction": point["service_rate_stability"]["active_second_half"]["total"]["zero_service_window_fraction"], + }) + writer.writerow(row) + pair_rows = analysis["pairwise_policy_costs"] + if pair_rows: + with (output / "pairwise-policy-costs.csv").open("w", newline="") as stream: + writer = csv.DictWriter(stream, fieldnames=list(pair_rows[0])) + writer.writeheader() + writer.writerows(pair_rows) + cross_rows = analysis["cross_topology_pressure_costs"] + if cross_rows: + with (output / "cross-topology-pressure-costs.csv").open("w", newline="") as stream: + writer = csv.DictWriter(stream, fieldnames=list(cross_rows[0])) + writer.writeheader() + writer.writerows(cross_rows) + plot_paths = write_plots(analysis, output) if plots else [] + lines = [ + "# Controlled rate campaign analysis", "", + f"Completed points: {analysis['completed_point_count']} / {analysis['expected_point_count']}", "", + "Metrics were recomputed from aligned rates/control/energy/thermal streams. Byte and energy", + "conservation were checked per point. Pairwise deltas are always right minus left and do not", + "encode a preferred policy or assume that lower temperature offsets lost delivery.", "", + "Nonzero backlog is a measured saturation cost, not an execution failure. Cross-topology rows", + "separately identify equal-total-demand and equal-per-stack-pressure comparisons.", "", + "Backend latency, token/s and maintenance behavior remain unavailable. Delivery and byte-weighted", + "delay are properties of the window-quantized fluid FIFO, not MQSim or fabric completion. Energy", + "is incremental served-byte energy at the user-confirmed 50 pJ/B scenario; idle and GPU self-power", + "are unknown in this path.", "", + f"Generated plots: {len(plot_paths)}", + ] + (output / "CONTROLLED_CAMPAIGN_ANALYSIS.md").write_text("\n".join(lines) + "\n") + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--campaign", type=Path, required=True) + parser.add_argument("--output", type=Path, required=True) + parser.add_argument("--allow-partial", action="store_true") + args = parser.parse_args(argv) + analysis = analyze_campaign(args.campaign.resolve(), require_complete=not args.allow_partial) + write_outputs(analysis, args.output.resolve()) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/experiments/eq3_rate_thermal/analyze_pair.py b/experiments/eq3_rate_thermal/analyze_pair.py new file mode 100644 index 0000000..1fef3f1 --- /dev/null +++ b/experiments/eq3_rate_thermal/analyze_pair.py @@ -0,0 +1,298 @@ +#!/usr/bin/env python3 +"""Compare equal-energy uniform and channel-concentrated rate-thermal runs.""" +from __future__ import annotations + +import argparse +import hashlib +import json +from pathlib import Path + + +BASELINE_K = 300.0 +SPLIT_NS = 1_400_000_000 +RECOVERY_NS = 2_200_000_000 +END_NS = 3_200_000_000 + + +def load_json(path: Path): + return json.loads(path.read_text()) + + +def load_jsonl(path: Path): + return [json.loads(line) for line in path.read_text().splitlines() if line] + + +def sha256(path: Path) -> str: + return hashlib.sha256(path.read_bytes()).hexdigest() + + +def maximum(records): + return max(records, key=lambda item: item["delta_k"]) + + +def minimum(records): + return min(records, key=lambda item: item["delta_k"]) + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--uniform", type=Path, required=True) + parser.add_argument("--concentrated", type=Path, required=True) + parser.add_argument("--output", type=Path, required=True) + args = parser.parse_args() + out = args.output.resolve() + out.mkdir(parents=True, exist_ok=False) + + points = {"uniform": args.uniform.resolve(), "concentrated": args.concentrated.resolve()} + done = {arm: load_json(path / "DONE.json") for arm, path in points.items()} + manifests = {arm: load_json(path / "manifest.json") for arm, path in points.items()} + profiles = {arm: load_json(path / "profile.json") for arm, path in points.items()} + loads = {arm: load_json(path / "load-windows.json") for arm, path in points.items()} + frames = {arm: load_jsonl(path / "thermal.jsonl") for arm, path in points.items()} + for arm in points: + if done[arm]["execution_status"] != "COMPLETED": + raise ValueError(f"{arm} is not complete") + if len(frames[arm]) != 160 or frames[arm][-1]["end_ns"] != END_NS: + raise ValueError(f"{arm} frame count or end time differs from the registered pair") + if profiles["uniform"] != profiles["concentrated"]: + raise ValueError("profiles differ") + identity_keys = ("thermal_binary_sha256", "model_file_sha256", "model_dir", "step_ns") + identity = {key: manifests["uniform"].get(key) for key in identity_keys} + if any(manifests["concentrated"].get(key) != value for key, value in identity.items()): + raise ValueError("thermal model or executable identity differs") + times = {arm: [frame["end_ns"] for frame in rows] for arm, rows in frames.items()} + if times["uniform"] != times["concentrated"]: + raise ValueError("frame timestamps differ") + + # Same requested bytes and stack energy at every window; only die placement may differ. + energy_windows = {} + for arm, payload in loads.items(): + energy_windows[arm] = [sum(w["component_energy_j"].values()) for w in payload["windows"]] + byte_mismatch = [] + for wu, wc in zip(loads["uniform"]["windows"], loads["concentrated"]["windows"]): + for stack in wu["stacks"]: + if wu["stacks"][stack]["requested_bytes"] != wc["stacks"][stack]["requested_bytes"]: + byte_mismatch.append((wu["end_ns"], stack)) + max_window_energy_delta = max(abs(a - b) for a, b in zip(*energy_windows.values())) + total_energy = {arm: sum(values) for arm, values in energy_windows.items()} + if byte_mismatch or max_window_energy_delta > 1e-12 or abs(total_energy["uniform"] - total_energy["concentrated"]) > 1e-10: + raise ValueError("pair does not conserve the same requested bytes and energy") + placement_windows = 0 + for wu, wc in zip(loads["uniform"]["windows"], loads["concentrated"]["windows"]): + differing = wu["start_ns"] >= SPLIT_NS and wu["end_ns"] <= RECOVERY_NS + if not differing: + if wu["component_energy_j"] != wc["component_energy_j"]: + raise ValueError("component placement differs outside the registered phase") + continue + placement_windows += 1 + for component, energy in wu["component_energy_j"].items(): + if not component.startswith("hbf0.die") and wc["component_energy_j"][component] != energy: + raise ValueError("a source outside HBF0 dies differs in the placement phase") + for arm, window, expected_active in (("uniform", wu, 16), ("concentrated", wc, 4)): + channel_bytes = window["stacks"]["hbf0"]["channel_requested_bytes"] + active = [value for value in channel_bytes.values() if value > 0] + if len(active) != expected_active or max(active) != min(active): + raise ValueError(f"{arm} channel placement does not match the registered pair") + if placement_windows != 40: + raise ValueError("unexpected number of channel-placement windows") + + indexed = {arm: {row["end_ns"]: row for row in rows} for arm, rows in frames.items()} + pre_records = [] + for t in times["uniform"]: + if t > SPLIT_NS: + continue + u, c = indexed["uniform"][t], indexed["concentrated"][t] + for name in u["entity_temperatures_k"]: + for metric in ("hotspot_k", "mean_k"): + pre_records.append(abs(c["entity_temperatures_k"][name][metric] - u["entity_temperatures_k"][name][metric])) + for name in u["sensor_temperatures_k"]: + pre_records.append(abs(c["sensor_temperatures_k"][name] - u["sensor_temperatures_k"][name])) + pre_max = max(pre_records, default=0.0) + if pre_max > 1e-12: + raise ValueError(f"pre-divergence states differ by {pre_max} K") + + compared = [] + names = [f"hbf0.die{i}" for i in range(16)] + ["hbf0.base"] + for t in times["uniform"]: + if SPLIT_NS < t <= RECOVERY_NS: + for name in names: + for metric in ("hotspot_k", "mean_k"): + compared.append({ + "time_ns": t, "entity": name, "metric": metric, + "delta_k": indexed["concentrated"][t]["entity_temperatures_k"][name][metric] + - indexed["uniform"][t]["entity_temperatures_k"][name][metric], + }) + stack_deltas = [] + for t in times["uniform"]: + if SPLIT_NS < t <= RECOVERY_NS: + stack_deltas.append({"time_ns": t, "entity": "hbf0", "metric": "stack_hotspot_k", + "delta_k": indexed["concentrated"][t]["temperatures"]["hbf0"] + - indexed["uniform"][t]["temperatures"]["hbf0"]}) + + at_22 = {} + for name in names: + at_22[name] = {} + for arm in points: + state = indexed[arm][RECOVERY_NS]["entity_temperatures_k"][name] + at_22[name][arm + "_hotspot_rise_k"] = state["hotspot_k"] - BASELINE_K + at_22[name][arm + "_mean_rise_k"] = state["mean_k"] - BASELINE_K + at_22[name]["hotspot_delta_k"] = (at_22[name]["concentrated_hotspot_rise_k"] + - at_22[name]["uniform_hotspot_rise_k"]) + at_22[name]["mean_delta_k"] = (at_22[name]["concentrated_mean_rise_k"] + - at_22[name]["uniform_mean_rise_k"]) + + phase_peak = {} + for arm in points: + candidates = [(t, indexed[arm][t]["temperatures"]["hbf0"]) + for t in times[arm] if SPLIT_NS < t <= RECOVERY_NS] + t, value = max(candidates, key=lambda item: item[1]) + phase_peak[arm] = {"time_ns": t, "hbf0_stack_hotspot_rise_k": value - BASELINE_K} + + recovery = {} + for arm in points: + recovery[arm] = {} + for stack in indexed[arm][END_NS]["temperatures"]: + recovery[arm][stack] = { + "rise_at_2p2s_k": indexed[arm][RECOVERY_NS]["temperatures"][stack] - BASELINE_K, + "rise_at_3p2s_k": indexed[arm][END_NS]["temperatures"][stack] - BASELINE_K, + "change_during_recovery_k": indexed[arm][END_NS]["temperatures"][stack] + - indexed[arm][RECOVERY_NS]["temperatures"][stack], + } + + hbf3_zero = {} + for arm in points: + requested = sum(w["stacks"]["hbf3"]["requested_bytes"] for w in loads[arm]["windows"]) + hbf3_zero[arm] = { + "requested_bytes": requested, + "peak_rise_k": done[arm]["peak_k_by_stack"]["hbf3"] - BASELINE_K, + "rise_at_2p2s_k": indexed[arm][RECOVERY_NS]["temperatures"]["hbf3"] - BASELINE_K, + "final_rise_k": indexed[arm][END_NS]["temperatures"]["hbf3"] - BASELINE_K, + } + if requested != 0: + raise ValueError("hbf3 is not an unpowered coupling observation") + + aligned = [] + for t in times["uniform"]: + if t <= SPLIT_NS: + continue + u, c = indexed["uniform"][t], indexed["concentrated"][t] + aligned.append({ + "time_ns": t, + "stack_hotspot_delta_k": {name: c["temperatures"][name] - value + for name, value in u["temperatures"].items()}, + "hbf0_component_delta_k": { + name: {metric: c["entity_temperatures_k"][name][metric] - + u["entity_temperatures_k"][name][metric] + for metric in ("hotspot_k", "mean_k")} + for name in names + }, + }) + + conservation = {} + for arm in points: + cumulative = indexed[arm][END_NS]["energy_j"]["cumulative"] + conservation[arm] = { + **cumulative, + "absolute_residual_fraction": abs(cumulative["energy_residual_j"]) / + max(cumulative["total_input_j"], 1.0), + } + + result = { + "schema_version": 1, + "status": "PASS", + "classification": "CONDITIONAL_SIMULATED_INCREMENTAL_HEATING", + "interpretation": "300 K is a common reference initial condition; values are simulated rises, not measured product temperatures.", + "inputs": { + arm: {"path": str(path), "done_sha256": sha256(path / "DONE.json"), + "thermal_sha256": sha256(path / "thermal.jsonl"), + "load_windows_sha256": sha256(path / "load-windows.json"), + "schedule_sha256": sha256(path / "schedule.json"), + "profile_sha256": sha256(path / "profile.json")} + for arm, path in points.items() + }, + "identity": identity, + "frames": 160, + "time_step_ns": 20_000_000, + "total_energy_j": total_energy, + "max_per_window_total_energy_difference_j": max_window_energy_delta, + "pre_1p4s_equivalence": {"max_temperature_difference_k": pre_max, "passed": True}, + "differing_phase_ns": [SPLIT_NS, RECOVERY_NS], + "placement_windows": placement_windows, + "phase_hbf0_stack_peak": phase_peak, + "phase_hbf0_stack_delta_extrema": {"maximum": maximum(stack_deltas), "minimum": minimum(stack_deltas)}, + "phase_hbf0_component_delta_extrema": {"maximum": maximum(compared), "minimum": minimum(compared)}, + "snapshot_2p2s": at_22, + "hbf3_unpowered_coupling": hbf3_zero, + "recovery_to_3p2s": recovery, + "energy_conservation": conservation, + } + + aligned_path = out / "aligned-differences.jsonl" + aligned_path.write_text("".join(json.dumps(row, allow_nan=False) + "\n" for row in aligned)) + result["aligned_differences"] = { + "path": aligned_path.name, + "sha256": sha256(aligned_path), + "frames_after_1p4s": len(aligned), + "contents": "all stack hotspots plus HBF0 die/base hotspot and mean deltas at each aligned timestamp", + } + + # Plot after all assertions, so a figure cannot outlive a failed comparison. + import matplotlib + matplotlib.use("Agg") + import matplotlib.pyplot as plt + import numpy as np + + fig, axes = plt.subplots(1, 3, figsize=(15.2, 4.5)) + t_s = np.array(times["uniform"]) / 1e9 + rate = [] + for w in loads["uniform"]["windows"]: + dt = (w["end_ns"] - w["start_ns"]) / 1e9 + rate.append(w["stacks"]["hbf0"]["requested_bytes"] / dt / 1e12) + axes[0].step(t_s, rate, where="post", color="black", label="both arms") + axes[0].set(title="Identical HBF0 stack rate", xlabel="Time (s)", ylabel="Prescribed rate (TB/s)") + axes[0].legend(frameon=False) + for arm, color in (("uniform", "#2878B5"), ("concentrated", "#D95319")): + rise = [row["temperatures"]["hbf0"] - BASELINE_K for row in frames[arm]] + axes[1].plot(t_s, rise, color=color, label=arm) + axes[1].set(title="HBF0 stack hotspot rise", xlabel="Time (s)", ylabel="Rise from 300 K (K)") + axes[1].legend(frameon=False) + x = np.arange(17) + width = 0.4 + for offset, arm, color in ((-width/2, "uniform", "#2878B5"), (width/2, "concentrated", "#D95319")): + values = [at_22[f"hbf0.die{i}"][arm + "_hotspot_rise_k"] for i in range(16)] + values.append(at_22["hbf0.base"][arm + "_hotspot_rise_k"]) + axes[2].bar(x + offset, values, width, color=color, label=arm) + axes[2].set(title="HBF0 component hotspot rise at 2.2 s", xlabel="Thermal component", ylabel="Rise from 300 K (K)") + axes[2].set_xticks(x, [str(i) for i in range(16)] + ["base"], rotation=55) + axes[2].legend(frameon=False) + for axis in axes: + axis.axvspan(1.4, 2.2, color="#888888", alpha=0.09) + axis.grid(alpha=0.2) + fig.suptitle("Equal-rate, equal-energy channel placement comparison (conditional simulation)") + fig.tight_layout() + fig.savefig(out / "paired.png", dpi=180) + plt.close(fig) + + result["analyzer_sha256"] = sha256(Path(__file__)) + (out / "PAIR_RESULT.json").write_text(json.dumps(result, indent=2, allow_nan=False) + "\n") + + smax = result["phase_hbf0_stack_delta_extrema"]["maximum"] + smin = result["phase_hbf0_stack_delta_extrema"]["minimum"] + cmax = result["phase_hbf0_component_delta_extrema"]["maximum"] + cmin = result["phase_hbf0_component_delta_extrema"]["minimum"] + report = f"""# OCP Grade 2 equal-energy channel-placement pair + +Both arms completed 160 aligned 20 ms frames through 3.2 s. They use the same model, executable, profile, stack-rate schedule, 300 K initial condition and **{total_energy['uniform']:.6f} J** incremental read energy. Before 1.4 s their full entity/sensor temperature state agrees within **{pre_max:.3e} K**. From 1.4–2.2 s HBF0 remains at 0.384 TB/s: the uniform arm assigns 24 GB/s to each of 16 channels, while the concentrated arm assigns 96 GB/s to four channels. + +The concentrated-minus-uniform HBF0 stack-hotspot difference spans **{smin['delta_k']:+.6f} to {smax['delta_k']:+.6f} K** during that phase. Across HBF0 dies/base and both hotspot/mean observations, the extrema are **{cmin['delta_k']:+.6f} K** ({cmin['entity']}, {cmin['metric']}, {cmin['time_ns']/1e9:.2f} s) and **{cmax['delta_k']:+.6f} K** ({cmax['entity']}, {cmax['metric']}, {cmax['time_ns']/1e9:.2f} s). This reports the simulated spatial response without assuming beforehand which placement must be hotter. + +HBF3 receives zero requested bytes in both arms, yet reaches peak rises of **{hbf3_zero['uniform']['peak_rise_k']:.6f} K** and **{hbf3_zero['concentrated']['peak_rise_k']:.6f} K**, respectively; this is coupled-package heating. At 3.2 s HBF0's stack-hotspot rise is **{recovery['uniform']['hbf0']['rise_at_3p2s_k']:.6f} K** (uniform) and **{recovery['concentrated']['hbf0']['rise_at_3p2s_k']:.6f} K** (concentrated), after changes of **{recovery['uniform']['hbf0']['change_during_recovery_k']:+.6f} K** and **{recovery['concentrated']['hbf0']['change_during_recovery_k']:+.6f} K** during recovery. + +Final thermal energy residuals are **{conservation['uniform']['energy_residual_j']:.3e} J** and **{conservation['concentrated']['energy_residual_j']:.3e} J**. These are conditional incremental-heating simulations: 300 K is a shared reference initial state, not a claim about absolute product temperature. See `PAIR_RESULT.json` for every die/base value at 2.2 s and [the paired figure](paired.png). +""" + (out / "PAIR_RESULT.md").write_text(report) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/experiments/eq3_rate_thermal/controlled_scenario.schema.json b/experiments/eq3_rate_thermal/controlled_scenario.schema.json new file mode 100644 index 0000000..35477da --- /dev/null +++ b/experiments/eq3_rate_thermal/controlled_scenario.schema.json @@ -0,0 +1,21 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "eq3-rate-controlled-scenario-v1", + "title": "EQ3 fluid rate/thermal control scenario", + "type": "object", + "additionalProperties": false, + "required": [ + "schema_version", + "scenario_id", + "topology", + "window_ns", + "target_read_Bps_per_stack" + ], + "properties": { + "schema_version": {"const": "eq3-rate-controlled-scenario-v1"}, + "scenario_id": {"type": "string", "minLength": 1}, + "topology": {"enum": ["mixed_direct", "all_hbf_direct"]}, + "window_ns": {"const": 20000000}, + "target_read_Bps_per_stack": {"type": "integer", "minimum": 1} + } +} diff --git a/experiments/eq3_rate_thermal/fluid_service.py b/experiments/eq3_rate_thermal/fluid_service.py new file mode 100644 index 0000000..899d515 --- /dev/null +++ b/experiments/eq3_rate_thermal/fluid_service.py @@ -0,0 +1,423 @@ +"""Window-quantized byte service for rate-driven thermal experiments. + +This module is deliberately independent of MQSim and the thermal solver. It +models only finite byte service and queued byte cohorts. A caller supplies a +future service budget for every stack on every window; the service never +changes that budget based on temperatures or queue state. +""" + +from __future__ import annotations + +from collections import deque +from dataclasses import dataclass +from typing import Deque, Dict, Mapping, Optional, Sequence, Tuple + + +NANOSECONDS_PER_SECOND = 1_000_000_000 +DEFAULT_WINDOW_NS = 20_000_000 +LATENCY_SEMANTICS = "FLUID_WINDOW_QUANTIZED_DELAY" +BACKEND_LATENCY = "UNKNOWN" + + +@dataclass +class _Cohort: + arrival_ns: int + remaining_bytes: int + + +def _integer_bytes(value: object, field: str, *, positive: bool = False) -> int: + if isinstance(value, bool) or not isinstance(value, int): + raise ValueError(f"{field} must be an integer byte count") + if value < 0 or (positive and value == 0): + qualifier = "positive" if positive else "non-negative" + raise ValueError(f"{field} must be {qualifier}") + return value + + +def _weighted_nearest_rank_p95(samples: Sequence[Tuple[int, int]]) -> Optional[int]: + """Return the nearest-rank p95 over byte-weighted integer delays.""" + + total = sum(byte_count for _, byte_count in samples) + if total == 0: + return None + rank = (95 * total + 99) // 100 + cumulative = 0 + for delay_ns, byte_count in sorted(samples): + cumulative += byte_count + if cumulative >= rank: + return delay_ns + raise AssertionError("weighted percentile did not reach its rank") + + +def _delay_histogram(samples: Sequence[Tuple[int, int]]) -> list[dict]: + """Return a JSON-friendly byte histogram sorted by quantized delay.""" + + totals: Dict[int, int] = {} + for delay_ns, byte_count in samples: + totals[delay_ns] = totals.get(delay_ns, 0) + byte_count + return [ + {"delay_ns": delay_ns, "bytes": totals[delay_ns]} + for delay_ns in sorted(totals) + ] + + +def _max_min_allocation(demand: Mapping[str, int], budget: int) -> Dict[str, int]: + """Deterministic integer max-min allocation, bounded by channel demand.""" + + allocation = {channel: 0 for channel in demand} + remaining = min(budget, sum(demand.values())) + active = sorted(channel for channel, value in demand.items() if value > 0) + while remaining and active: + share, remainder = divmod(remaining, len(active)) + if share == 0: + for channel in active[:remainder]: + allocation[channel] += 1 + remaining = 0 + break + + granted = 0 + for channel in active: + room = demand[channel] - allocation[channel] + amount = min(share, room) + allocation[channel] += amount + granted += amount + remaining -= granted + active = [ + channel + for channel in active + if allocation[channel] < demand[channel] + ] + return allocation + + +class FluidService: + """Persistent per-channel FIFO queues served in fixed time windows. + + ``budget_by_stack`` is a future-only byte budget for the supplied window. + Channel capacities and the stack budget jointly limit service. The stack + budget is distributed with integer max-min fairness among channels that + have serviceable backlog. Endpoint-group arbitration is intentionally not + represented in schema v1. + """ + + schema_version = "eq3-fluid-service-v1" + + def __init__( + self, + stack_channels: Mapping[str, Sequence[str]], + channel_capacity_Bps: Mapping[str, Mapping[str, int]], + window_ns: int = DEFAULT_WINDOW_NS, + ) -> None: + self.window_ns = _integer_bytes(window_ns, "window_ns", positive=True) + if not stack_channels: + raise ValueError("stack_channels must not be empty") + + self._channels: Dict[str, Tuple[str, ...]] = {} + self._capacity_bps: Dict[str, Dict[str, int]] = {} + self._capacity_bytes: Dict[str, Dict[str, int]] = {} + self._queues: Dict[Tuple[str, str], Deque[_Cohort]] = {} + self._cumulative_offered: Dict[Tuple[str, str], int] = {} + self._cumulative_delivered: Dict[Tuple[str, str], int] = {} + + declared_stacks = set(stack_channels) + if set(channel_capacity_Bps) != declared_stacks: + raise ValueError("channel_capacity_Bps must exactly match declared stacks") + for stack in sorted(declared_stacks): + if not isinstance(stack, str) or not stack: + raise ValueError("stack identifiers must be non-empty strings") + raw_channels = tuple(stack_channels[stack]) + if not raw_channels or len(set(raw_channels)) != len(raw_channels): + raise ValueError(f"{stack} must declare unique, non-empty channels") + if any(not isinstance(channel, str) or not channel for channel in raw_channels): + raise ValueError(f"{stack} channel identifiers must be non-empty strings") + channels = tuple(sorted(raw_channels)) + if set(channel_capacity_Bps[stack]) != set(channels): + raise ValueError(f"channel_capacity_Bps[{stack}] must match its channels") + + self._channels[stack] = channels + self._capacity_bps[stack] = {} + self._capacity_bytes[stack] = {} + for channel in channels: + rate = _integer_bytes( + channel_capacity_Bps[stack][channel], + f"channel_capacity_Bps[{stack}][{channel}]", + positive=True, + ) + self._capacity_bps[stack][channel] = rate + self._capacity_bytes[stack][channel] = ( + rate * self.window_ns // NANOSECONDS_PER_SECOND + ) + key = (stack, channel) + self._queues[key] = deque() + self._cumulative_offered[key] = 0 + self._cumulative_delivered[key] = 0 + + self._now_ns = 0 + + @property + def now_ns(self) -> int: + return self._now_ns + + def _validate_offered( + self, offered_by_channel: Mapping[str, Mapping[str, int]] + ) -> Dict[str, Dict[str, int]]: + unknown_stacks = set(offered_by_channel) - set(self._channels) + if unknown_stacks: + raise ValueError(f"unknown offered stacks: {sorted(unknown_stacks)}") + result: Dict[str, Dict[str, int]] = {} + for stack, channels in self._channels.items(): + supplied = offered_by_channel.get(stack, {}) + unknown_channels = set(supplied) - set(channels) + if unknown_channels: + raise ValueError( + f"unknown offered channels for {stack}: {sorted(unknown_channels)}" + ) + result[stack] = { + channel: _integer_bytes( + supplied.get(channel, 0), + f"offered_by_channel[{stack}][{channel}]", + ) + for channel in channels + } + return result + + def _validate_budget(self, budget_by_stack: Mapping[str, int]) -> Dict[str, int]: + if set(budget_by_stack) != set(self._channels): + raise ValueError("budget_by_stack must exactly match declared stacks") + return { + stack: _integer_bytes(budget_by_stack[stack], f"budget_by_stack[{stack}]") + for stack in self._channels + } + + @staticmethod + def _queue_bytes(queue: Deque[_Cohort]) -> int: + return sum(cohort.remaining_bytes for cohort in queue) + + def advance( + self, + start_ns: int, + end_ns: int, + offered_by_channel: Mapping[str, Mapping[str, int]], + budget_by_stack: Mapping[str, int], + ) -> dict: + """Enqueue arrivals and serve exactly one fixed window. + + Omitted declared channels in ``offered_by_channel`` mean zero arrivals, + which permits cooling/recovery windows to continue draining old bytes. + All delivery timestamps are quantized to ``end_ns``. + """ + + start_ns = _integer_bytes(start_ns, "start_ns") + end_ns = _integer_bytes(end_ns, "end_ns") + if start_ns != self._now_ns: + raise ValueError(f"start_ns must equal service now_ns {self._now_ns}") + if end_ns - start_ns != self.window_ns: + raise ValueError(f"window must be exactly {self.window_ns} ns") + + offered = self._validate_offered(offered_by_channel) + budgets = self._validate_budget(budget_by_stack) + previous_backlog: Dict[str, Dict[str, int]] = { + stack: { + channel: self._queue_bytes(self._queues[(stack, channel)]) + for channel in channels + } + for stack, channels in self._channels.items() + } + + for stack, channels in self._channels.items(): + for channel in channels: + amount = offered[stack][channel] + if amount: + self._queues[(stack, channel)].append(_Cohort(start_ns, amount)) + self._cumulative_offered[(stack, channel)] += amount + + served_by_channel: Dict[str, Dict[str, int]] = {} + channel_rows: Dict[str, Dict[str, dict]] = {} + stack_rows: Dict[str, dict] = {} + window_samples_by_stack: Dict[str, list[Tuple[int, int]]] = {} + + for stack, channels in self._channels.items(): + demand = { + channel: min( + self._queue_bytes(self._queues[(stack, channel)]), + self._capacity_bytes[stack][channel], + ) + for channel in channels + } + allocation = _max_min_allocation(demand, budgets[stack]) + served_by_channel[stack] = {} + channel_rows[stack] = {} + window_samples_by_stack[stack] = [] + + for channel in channels: + key = (stack, channel) + queue = self._queues[key] + remaining_service = allocation[channel] + samples: list[Tuple[int, int]] = [] + while remaining_service: + cohort = queue[0] + amount = min(remaining_service, cohort.remaining_bytes) + delay_ns = end_ns - cohort.arrival_ns + samples.append((delay_ns, amount)) + window_samples_by_stack[stack].append((delay_ns, amount)) + cohort.remaining_bytes -= amount + remaining_service -= amount + if cohort.remaining_bytes == 0: + queue.popleft() + + delivered = allocation[channel] + self._cumulative_delivered[key] += delivered + backlog = self._queue_bytes(queue) + oldest_wait = end_ns - queue[0].arrival_ns if queue else None + served_by_channel[stack][channel] = delivered + channel_rows[stack][channel] = { + "offered_bytes": offered[stack][channel], + "delivered_bytes": delivered, + "backlog_bytes": backlog, + "oldest_wait_ns": oldest_wait, + "latency_p95_ns": _weighted_nearest_rank_p95(samples), + "delivered_delay_histogram_bytes": _delay_histogram(samples), + "capacity_bytes": self._capacity_bytes[stack][channel], + "capacity_Bps": self._capacity_bps[stack][channel], + } + + stack_offered = sum(offered[stack].values()) + stack_delivered = sum(served_by_channel[stack].values()) + stack_backlog = sum( + channel_rows[stack][channel]["backlog_bytes"] for channel in channels + ) + oldest_values = [ + channel_rows[stack][channel]["oldest_wait_ns"] + for channel in channels + if channel_rows[stack][channel]["oldest_wait_ns"] is not None + ] + stack_rows[stack] = { + "offered_bytes": stack_offered, + "delivered_bytes": stack_delivered, + "backlog_bytes": stack_backlog, + "oldest_wait_ns": max(oldest_values) if oldest_values else None, + "latency_p95_ns": _weighted_nearest_rank_p95( + window_samples_by_stack[stack] + ), + "delivered_delay_histogram_bytes": _delay_histogram( + window_samples_by_stack[stack] + ), + "budget_bytes": budgets[stack], + "latency_semantics": LATENCY_SEMANTICS, + "backend_latency_ns": BACKEND_LATENCY, + } + + per_stack_conservation = {} + for stack, channels in self._channels.items(): + per_channel_conservation = {} + for channel in channels: + key = (stack, channel) + previous_channel = previous_backlog[stack][channel] + arrivals_channel = offered[stack][channel] + delivered_channel = served_by_channel[stack][channel] + backlog_channel = channel_rows[stack][channel]["backlog_bytes"] + if previous_channel + arrivals_channel != delivered_channel + backlog_channel: + raise AssertionError( + f"window byte conservation failed for {stack}/{channel}" + ) + if ( + self._cumulative_offered[key] + != self._cumulative_delivered[key] + backlog_channel + ): + raise AssertionError( + f"cumulative byte conservation failed for {stack}/{channel}" + ) + per_channel_conservation[channel] = { + "previous_backlog_bytes": previous_channel, + "offered_bytes": arrivals_channel, + "delivered_bytes": delivered_channel, + "backlog_bytes": backlog_channel, + "window_conserved": True, + "cumulative_offered_bytes": self._cumulative_offered[key], + "cumulative_delivered_bytes": self._cumulative_delivered[key], + "cumulative_conserved": True, + } + previous = sum(previous_backlog[stack].values()) + arrivals = stack_rows[stack]["offered_bytes"] + delivered = stack_rows[stack]["delivered_bytes"] + backlog = stack_rows[stack]["backlog_bytes"] + cumulative_offered = sum( + self._cumulative_offered[(stack, channel)] for channel in channels + ) + cumulative_delivered = sum( + self._cumulative_delivered[(stack, channel)] for channel in channels + ) + if previous + arrivals != delivered + backlog: + raise AssertionError(f"window byte conservation failed for {stack}") + if cumulative_offered != cumulative_delivered + backlog: + raise AssertionError(f"cumulative byte conservation failed for {stack}") + per_stack_conservation[stack] = { + "previous_backlog_bytes": previous, + "offered_bytes": arrivals, + "delivered_bytes": delivered, + "backlog_bytes": backlog, + "window_conserved": True, + "cumulative_offered_bytes": cumulative_offered, + "cumulative_delivered_bytes": cumulative_delivered, + "cumulative_conserved": True, + "per_channel": per_channel_conservation, + } + + self._now_ns = end_ns + return { + "schema_version": self.schema_version, + "start_ns": start_ns, + "end_ns": end_ns, + "window_ns": self.window_ns, + "served_by_channel": served_by_channel, + "stacks": stack_rows, + "channels": channel_rows, + "conservation": { + "per_stack": per_stack_conservation, + "window_conserved": True, + "cumulative_conserved": True, + }, + "semantics": { + "latency": LATENCY_SEMANTICS, + "delivered_delay_histogram": "BYTE_WEIGHTED_WINDOW_END_DELAY_NS", + "undelivered_delay": "CENSORED_AS_BACKLOG_NO_DELAY_ASSIGNED", + "backend_latency_ns": BACKEND_LATENCY, + "capacity_conversion": "FLOOR_INTEGER_BYTES_PER_WINDOW", + "stack_budget_allocation": "INTEGER_MAX_MIN_FAIR", + "service_scope": "STACK_AND_CHANNEL_ONLY", + "endpoint_group_arbitration": "UNSUPPORTED_IN_V1", + "transaction_model": "NONE_FLUID_BYTES_ONLY", + }, + } + + def snapshot(self) -> dict: + """Return current queue facts without changing service state.""" + + stacks = {} + channels = {} + for stack, stack_channels in self._channels.items(): + channels[stack] = {} + stack_backlog = 0 + oldest_values = [] + for channel in stack_channels: + queue = self._queues[(stack, channel)] + backlog = self._queue_bytes(queue) + oldest_wait = self._now_ns - queue[0].arrival_ns if queue else None + channels[stack][channel] = { + "backlog_bytes": backlog, + "oldest_wait_ns": oldest_wait, + "cohort_count": len(queue), + } + stack_backlog += backlog + if oldest_wait is not None: + oldest_values.append(oldest_wait) + stacks[stack] = { + "backlog_bytes": stack_backlog, + "oldest_wait_ns": max(oldest_values) if oldest_values else None, + } + return { + "schema_version": self.schema_version, + "now_ns": self._now_ns, + "stacks": stacks, + "channels": channels, + } diff --git a/experiments/eq3_rate_thermal/launch_control_stage.py b/experiments/eq3_rate_thermal/launch_control_stage.py new file mode 100644 index 0000000..4fc48e1 --- /dev/null +++ b/experiments/eq3_rate_thermal/launch_control_stage.py @@ -0,0 +1,295 @@ +#!/usr/bin/env python3 +"""Run the frozen rate/thermal control index serially with bounded resources.""" + +from __future__ import annotations + +import argparse +from datetime import datetime, timezone +import hashlib +import json +import math +import os +from pathlib import Path +import signal +import shutil +import subprocess +import sys +import time +from typing import Any + + +HERE = Path(__file__).resolve().parent +ROOT = HERE.parents[1] +GIB = 1024 ** 3 +MIB = 1024 ** 2 + + +def _sha256(path: Path) -> str: + return hashlib.sha256(path.read_bytes()).hexdigest() + + +def _save(path: Path, value: Any) -> None: + temporary = path.with_suffix(path.suffix + ".tmp") + temporary.write_text(json.dumps(value, indent=2, sort_keys=True, allow_nan=False) + "\n") + temporary.replace(path) + + +def _tree_bytes(path: Path) -> int: + if not path.exists(): + return 0 + return sum(item.stat().st_size for item in path.rglob("*") if item.is_file()) + + +def _memory_available_bytes() -> int: + values = dict(line.split(":", 1) for line in Path("/proc/meminfo").read_text().splitlines()) + return int(values["MemAvailable"].split()[0]) * 1024 + + +def _check_locks(index: dict) -> None: + if _sha256(Path(index["preflight"])) != index["preflight_sha256"]: + raise RuntimeError("preflight lock mismatch") + for relative, expected in index["source_locks"].items(): + if _sha256(Path(index["source_root"]) / relative) != expected: + raise RuntimeError(f"source lock mismatch: {relative}") + if _sha256(Path(index["points"][0]["thermal_binary"])) != index["thermal_binary_sha256"]: + raise RuntimeError("thermal binary lock mismatch") + for topology, locks in index["model_locks"].items(): + row = next(point for point in index["points"] if point["topology"] == topology) + directory = Path(row["model_dir"]) + for name, expected in locks.items(): + if _sha256(directory / name) != expected: + raise RuntimeError(f"thermal model lock mismatch: {topology}/{name}") + for point in index["points"]: + for field, expected in point["input_sha256"].items(): + if _sha256(Path(point[field])) != expected: + raise RuntimeError(f"point input lock mismatch: {point['point_id']}/{field}") + + +def _command(point: dict, resources: dict) -> list[str]: + return [ + sys.executable, "-B", str(HERE / "run_controlled.py"), + "--profile", point["profile"], + "--model-dir", point["model_dir"], + "--workload", point["workload"], + "--scenario", point["scenario"], + "--thermal-binary", point["thermal_binary"], + "--artifact-root", point["artifact_root"], + "--output", point["output"], + "--strategy", point["strategy"], + "--address-limit-gib", str(resources["address_limit_gib"]), + ] + + +def _run_point(stage: Path, point: dict, resources: dict, stage_deadline: float, + effective_point_limit: int | None, effective_stage_limit: int | None) -> dict: + output = Path(point["output"]) + launch = stage / "launch" / point["point_id"] + if output.exists() or launch.exists(): + raise FileExistsError(f"refusing to overwrite existing point evidence: {point['point_id']}") + if _memory_available_bytes() < resources["host_memory_reserve_gib"] * GIB: + raise RuntimeError("host memory reserve unavailable before point start") + if shutil.disk_usage(stage).free < resources["host_disk_reserve_gib"] * GIB: + raise RuntimeError("host disk reserve unavailable before point start") + launch.mkdir(parents=True, exist_ok=False) + command = _command(point, resources) + started_utc = datetime.now(timezone.utc).isoformat() + _save(launch / "launch.json", { + "point_id": point["point_id"], "started_utc": started_utc, + "command": command, "point_watchdog_s": resources["per_point_watchdog_s"], + "stage_deadline_remaining_s": max(0.0, stage_deadline - time.monotonic()), + "host_memory_reserve_gib": resources["host_memory_reserve_gib"], + "host_disk_reserve_gib": resources["host_disk_reserve_gib"], + "effective_point_output_limit_bytes": effective_point_limit, + "effective_stage_output_limit_bytes": effective_stage_limit, + }) + reason = None + started = time.monotonic() + env = dict(os.environ, PYTHONDONTWRITEBYTECODE="1", CUDA_VISIBLE_DEVICES="", + OMP_NUM_THREADS="1", OPENBLAS_NUM_THREADS="1", MKL_NUM_THREADS="1", + NUMEXPR_NUM_THREADS="1") + with (launch / "stdout.log").open("wb") as stdout, (launch / "stderr.log").open("wb") as stderr: + child = subprocess.Popen(command, cwd=ROOT, env=env, stdout=stdout, stderr=stderr, + start_new_session=True) + while True: + point_remaining = resources["per_point_watchdog_s"] - (time.monotonic() - started) + stage_remaining = stage_deadline - time.monotonic() + if point_remaining <= 0: + reason = "POINT_WATCHDOG" + break + if stage_remaining <= 0: + reason = "STAGE_WATCHDOG" + break + try: + return_code = child.wait(timeout=min(10.0, point_remaining, stage_remaining)) + break + except subprocess.TimeoutExpired: + if _memory_available_bytes() < resources["host_memory_reserve_gib"] * GIB: + reason = "HOST_MEMORY_RESERVE" + break + if shutil.disk_usage(stage).free < resources["host_disk_reserve_gib"] * GIB: + reason = "HOST_DISK_RESERVE" + break + if effective_point_limit is not None and _tree_bytes(output) > effective_point_limit: + reason = "EVIDENCE_ADJUSTED_POINT_OUTPUT_LIMIT" + break + if effective_stage_limit is not None and _tree_bytes(stage) > effective_stage_limit: + reason = "EVIDENCE_ADJUSTED_STAGE_OUTPUT_LIMIT" + break + if reason: + os.killpg(child.pid, signal.SIGTERM) + try: + return_code = child.wait(timeout=5) + except subprocess.TimeoutExpired: + os.killpg(child.pid, signal.SIGKILL) + return_code = child.wait() + result = { + "point_id": point["point_id"], + "execution_status": "COMPLETED" if return_code == 0 else "FAILED", + "exit_code": return_code, + "reason": reason, + "wall_s": time.monotonic() - started, + "output_bytes": _tree_bytes(output), + "finished_utc": datetime.now(timezone.utc).isoformat(), + } + _save(launch / "result.json", result) + if return_code != 0 or not (output / "DONE.json").is_file(): + raise RuntimeError(f"point failed with retained evidence: {point['point_id']}") + return result + + +def _review_pilots(stage: Path, points: list[dict], main_points: list[dict], + resources: dict, launch_results: list[dict]) -> tuple[int, int, float, dict]: + rows = [] + observed_max = 0 + for point, launch_result in zip(points, launch_results): + done = json.loads((Path(point["output"]) / "DONE.json").read_text()) + summary = done.get("summary", {}) + peaks = summary.get("peak_k_by_stack", {}) + if done.get("execution_status") != "COMPLETED": + raise RuntimeError(f"pilot did not complete: {point['point_id']}") + if summary.get("byte_conservation_error") != 0: + raise RuntimeError(f"pilot byte conservation failed: {point['point_id']}") + energy_tolerance = 1e-9 * max(1.0, float(summary.get("energy_j", 0.0))) + if abs(done.get("served_to_thermal_energy_error_j", math.inf)) > energy_tolerance: + raise RuntimeError(f"pilot thermal energy mapping failed: {point['point_id']}") + if not peaks or min(peaks.values()) < 300 or max(peaks.values()) >= 400: + raise RuntimeError(f"pilot temperature domain review failed: {point['point_id']}") + if max(peaks.values()) <= 300: + raise RuntimeError(f"pilot showed no thermal response: {point['point_id']}") + if launch_result["wall_s"] > resources["per_point_watchdog_s"]: + raise RuntimeError(f"pilot exceeded point watchdog: {point['point_id']}") + observed_max = max(observed_max, launch_result["output_bytes"]) + rows.append({ + "point_id": point["point_id"], + "byte_conservation_error": 0, + "served_to_thermal_energy_error_j": done["served_to_thermal_energy_error_j"], + "served_to_thermal_energy_tolerance_j": energy_tolerance, + "peak_k": max(peaks.values()), + "wall_s": launch_result["wall_s"], + "output_bytes": launch_result["output_bytes"], + "final_backlog_bytes": summary.get("final_backlog_bytes"), + "backlog_interpretation": "EXPECTED_EVIDENCE_NOT_EXECUTION_FAILURE", + }) + estimate = resources["per_point_output_mib"] * MIB + # Main points cover 30 simulated seconds versus 12 seconds for pilots. + # Scale measured pilot evidence by that exact duration ratio, then retain + # 25% headroom for transcript/state-size variation. + effective_point = max(estimate, math.ceil(observed_max * 30 / 12 * 1.25)) + # The old values are reviewed estimates, not restored hard ceilings. The + # evidence-adjusted finite bounds remain subordinate to host free-space reserve. + effective_stage = max( + resources["stage_new_retained_gib"] * GIB, + _tree_bytes(stage) + effective_point * 39, + ) + pilot_wall_by_topology = { + point["topology"]: result["wall_s"] + for point, result in zip(points, launch_results) + } + projected_main_wall = sum( + pilot_wall_by_topology[point["topology"]] * 30 / 12 + for point in main_points + ) + effective_stage_wall = max( + float(resources["stage_wall_s"]), + sum(row["wall_s"] for row in launch_results) + projected_main_wall * 1.25, + ) + review = { + "status": "PILOT_REVIEW_PASS_MAIN_RELEASED", + "reviewed_utc": datetime.now(timezone.utc).isoformat(), + "rows": rows, + "preflight_output_estimate_bytes_per_point": estimate, + "observed_max_output_bytes": observed_max, + "pilot_to_main_duration_ratio": 30 / 12, + "output_headroom_fraction": 0.25, + "effective_point_output_limit_bytes": effective_point, + "effective_stage_output_limit_bytes": effective_stage, + "projected_main_wall_s_from_topology_pilots": projected_main_wall, + "effective_stage_watchdog_s": effective_stage_wall, + "resource_semantics": "FINITE_EVIDENCE_ADJUSTED_NOT_OLD_FIXED_HARD_LIMIT", + } + _save(stage / "PILOT_REVIEW.json", review) + return effective_point, effective_stage, effective_stage_wall, review + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--stage", type=Path, required=True) + args = parser.parse_args() + stage = args.stage.resolve(strict=True) + index_path = stage / "RUN_INDEX.json" + index = json.loads(index_path.read_text()) + if index.get("schema_version") != "eq3-rate-thermal-control-run-index-v1": + raise ValueError("unsupported run index") + if index.get("point_count") != 41 or index.get("pilot_count") != 2 or index.get("main_count") != 39: + raise ValueError("run index does not contain frozen 2+39 points") + _check_locks(index) + resources = index["resources"] + stage_started = time.monotonic() + stage_deadline = stage_started + resources["stage_wall_s"] + status = { + "execution_status": "RUNNING", "started_utc": datetime.now(timezone.utc).isoformat(), + "run_index_sha256": _sha256(index_path), "completed": [], "failed": None, + } + _save(stage / "STATUS.json", status) + try: + pilots = [row for row in index["points"] if row["phase"] == "pilot"] + mains = [row for row in index["points"] if row["phase"] == "main"] + pilot_results = [] + for point in pilots: + result = _run_point(stage, point, resources, stage_deadline, None, None) + pilot_results.append(result) + status["completed"].append(point["point_id"]) + _save(stage / "STATUS.json", status) + point_limit, stage_limit, effective_stage_wall, review = _review_pilots( + stage, pilots, mains, resources, pilot_results + ) + stage_deadline = stage_started + effective_stage_wall + status["pilot_review"] = review["status"] + _save(stage / "STATUS.json", status) + for point in mains: + result = _run_point(stage, point, resources, stage_deadline, point_limit, stage_limit) + status["completed"].append(point["point_id"]) + _save(stage / "STATUS.json", status) + status.update( + execution_status="COMPLETED", finished_utc=datetime.now(timezone.utc).isoformat(), + wall_s=time.monotonic() - stage_started, + retained_bytes=_tree_bytes(stage), + ) + _save(stage / "DONE.json", status) + _save(stage / "STATUS.json", status) + print(json.dumps(status, indent=2)) + return 0 + except BaseException as exc: + status.update( + execution_status="FAILED_PAUSED", failed=repr(exc), + finished_utc=datetime.now(timezone.utc).isoformat(), + wall_s=time.monotonic() - stage_started, + retained_bytes=_tree_bytes(stage), + ) + _save(stage / "FAILED.json", status) + _save(stage / "STATUS.json", status) + raise + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/experiments/eq3_rate_thermal/model_workload_config.schema.json b/experiments/eq3_rate_thermal/model_workload_config.schema.json new file mode 100644 index 0000000..67db0f6 --- /dev/null +++ b/experiments/eq3_rate_thermal/model_workload_config.schema.json @@ -0,0 +1,50 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "eq3-rate-model-workload-config-v1", + "title": "EQ3 model-size offered-byte workload configuration", + "type": "object", + "additionalProperties": false, + "required": [ + "schema_version", + "model_id", + "full_scans_per_s", + "pattern", + "stack_count", + "channels_per_stack", + "step_ns", + "active_ns", + "recovery_ns" + ], + "properties": { + "schema_version": {"const": "eq3-rate-model-workload-config-v1"}, + "model_id": { + "enum": [ + "Qwen/Qwen2.5-7B-Instruct", + "Qwen/Qwen2.5-72B-Instruct", + "Qwen/Qwen3-235B-A22B" + ] + }, + "full_scans_per_s": {"enum": [16, 32]}, + "pattern": {"enum": ["continuous", "burst_equal_mean"]}, + "stack_count": {"enum": [4, 8]}, + "channels_per_stack": {"const": 16}, + "step_ns": {"const": 20000000}, + "active_ns": {"type": "integer", "minimum": 1, "multipleOf": 500000000}, + "recovery_ns": {"type": "integer", "minimum": 0, "multipleOf": 20000000}, + "burst_period_ns": {"const": 200000000}, + "burst_on_ns": {"const": 100000000} + }, + "allOf": [ + { + "if": {"properties": {"pattern": {"const": "burst_equal_mean"}}}, + "then": { + "required": ["burst_period_ns", "burst_on_ns"], + "properties": {"active_ns": {"multipleOf": 1000000000}} + }, + "else": {"not": {"anyOf": [ + {"required": ["burst_period_ns"]}, + {"required": ["burst_on_ns"]} + ]}} + } + ] +} diff --git a/experiments/eq3_rate_thermal/model_workloads.py b/experiments/eq3_rate_thermal/model_workloads.py new file mode 100644 index 0000000..a3bc92f --- /dev/null +++ b/experiments/eq3_rate_thermal/model_workloads.py @@ -0,0 +1,269 @@ +#!/usr/bin/env python3 +"""Generate canonical model-size-driven offered-byte workload windows. + +The output is demand, not delivered service. It deliberately permits offered +rates above a channel's service capacity so a separate fluid model can retain +the excess as backlog rather than silently dropping it here. +""" + +from __future__ import annotations + +import argparse +from fractions import Fraction +import json +from pathlib import Path +from typing import Any + + +CONFIG_SCHEMA = "eq3-rate-model-workload-config-v1" +OUTPUT_SCHEMA = "eq3-rate-model-workload-v1" +CATALOG_VERSION = "eq3-official-weight-metadata-2026-09-20-v1" +STEP_NS = 20_000_000 +CHANNELS_PER_STACK = 16 +ALLOWED_STACK_COUNTS = (4, 8) +ALLOWED_SCANS_PER_S = (16, 32) +PATTERNS = ("continuous", "burst_equal_mean") + +MODEL_CATALOG = { + "Qwen/Qwen2.5-7B-Instruct": { + "tensor_payload_bytes": 15_231_233_024, + "revision": "main", + "resolved_commit": "UNKNOWN", + "source_class": "DOC_DERIVED_OFFICIAL_METADATA", + "source": "experiments/eq3_maintenance/sources/qwen2_5_weight_models.json", + "workload_interpretation": "SYNTHETIC_FULL_STORED_WEIGHT_SCAN", + }, + "Qwen/Qwen2.5-72B-Instruct": { + "tensor_payload_bytes": 145_412_407_296, + "revision": "main", + "resolved_commit": "UNKNOWN", + "source_class": "DOC_DERIVED_OFFICIAL_METADATA", + "source": "experiments/eq3_maintenance/sources/qwen2_5_weight_models.json", + "workload_interpretation": "SYNTHETIC_FULL_STORED_WEIGHT_SCAN", + }, + "Qwen/Qwen3-235B-A22B": { + "tensor_payload_bytes": 470_187_269_120, + "revision": "8efa61729e24bd65b1d152b5ab5409052aa80e65", + "resolved_commit": "8efa61729e24bd65b1d152b5ab5409052aa80e65", + "source_class": "DOC_DERIVED_OFFICIAL_METADATA", + "source": ( + "eq3_thermal/plans/isolated-maintenance-campaign-v1/" + "large-model-sources/EXTENT_AUDIT.json" + ), + "workload_interpretation": "SYNTHETIC_FULL_STORED_WEIGHT_SCAN_NOT_MOE_TOKEN_TRACE", + "model_card_parameters": "235B_TOTAL_22B_ACTIVE", + }, +} + + +def _mapping(value: Any, path: str) -> dict: + if not isinstance(value, dict): + raise ValueError(f"{path} must be an object") + return value + + +def _integer(value: Any, path: str, *, positive: bool = False) -> int: + if isinstance(value, bool) or not isinstance(value, int): + raise ValueError(f"{path} must be an integer") + if positive and value <= 0: + raise ValueError(f"{path} must be positive") + return value + + +def _validate_config(config: dict) -> dict: + config = _mapping(config, "config") + allowed_keys = { + "schema_version", "model_id", "full_scans_per_s", "pattern", + "stack_count", "channels_per_stack", "step_ns", "active_ns", + "recovery_ns", "burst_period_ns", "burst_on_ns", + } + unknown = sorted(set(config) - allowed_keys) + if unknown: + raise ValueError(f"config contains unknown fields: {unknown}") + if config.get("schema_version") != CONFIG_SCHEMA: + raise ValueError(f"config.schema_version must equal {CONFIG_SCHEMA!r}") + model_id = config.get("model_id") + if model_id not in MODEL_CATALOG: + raise ValueError(f"config.model_id is not in catalog {CATALOG_VERSION!r}") + scans = _integer(config.get("full_scans_per_s"), "config.full_scans_per_s", positive=True) + if scans not in ALLOWED_SCANS_PER_S: + raise ValueError(f"config.full_scans_per_s must be one of {ALLOWED_SCANS_PER_S}") + pattern = config.get("pattern") + if pattern not in PATTERNS: + raise ValueError(f"config.pattern must be one of {PATTERNS}") + stack_count = _integer(config.get("stack_count"), "config.stack_count", positive=True) + if stack_count not in ALLOWED_STACK_COUNTS: + raise ValueError(f"config.stack_count must be one of {ALLOWED_STACK_COUNTS}") + channel_count = _integer( + config.get("channels_per_stack"), "config.channels_per_stack", positive=True + ) + if channel_count != CHANNELS_PER_STACK: + raise ValueError(f"config.channels_per_stack must equal {CHANNELS_PER_STACK}") + step_ns = _integer(config.get("step_ns"), "config.step_ns", positive=True) + if step_ns != STEP_NS: + raise ValueError(f"config.step_ns must equal {STEP_NS}") + active_ns = _integer(config.get("active_ns"), "config.active_ns", positive=True) + recovery_ns = _integer(config.get("recovery_ns"), "config.recovery_ns") + if recovery_ns < 0: + raise ValueError("config.recovery_ns must be nonnegative") + if active_ns % step_ns or recovery_ns % step_ns: + raise ValueError("config active_ns and recovery_ns must align to step_ns") + + period_ns = config.get("burst_period_ns") + on_ns = config.get("burst_on_ns") + if pattern == "continuous": + if period_ns is not None or on_ns is not None: + raise ValueError("continuous config must omit burst_period_ns and burst_on_ns") + period_ns = None + on_ns = None + else: + period_ns = _integer(period_ns, "config.burst_period_ns", positive=True) + on_ns = _integer(on_ns, "config.burst_on_ns", positive=True) + if period_ns != 200_000_000 or on_ns != 100_000_000: + raise ValueError("burst_equal_mean requires a 200ms period and 100ms on interval") + if period_ns % step_ns or on_ns % step_ns or active_ns % period_ns: + raise ValueError("burst timing and active_ns must align to complete thermal windows/periods") + + equivalent_scans = Fraction(scans * active_ns, 1_000_000_000) + if equivalent_scans.denominator != 1: + raise ValueError("active_ns must produce an integer number of equivalent full scans") + canonical = { + "schema_version": CONFIG_SCHEMA, + "model_id": model_id, + "full_scans_per_s": scans, + "pattern": pattern, + "stack_count": stack_count, + "channels_per_stack": channel_count, + "step_ns": step_ns, + "active_ns": active_ns, + "recovery_ns": recovery_ns, + } + if pattern == "burst_equal_mean": + canonical["burst_period_ns"] = period_ns + canonical["burst_on_ns"] = on_ns + return canonical + + +def _multiplier(config: dict, start_ns: int) -> int: + if start_ns >= config["active_ns"]: + return 0 + if config["pattern"] == "continuous": + return 1 + return 2 if start_ns % config["burst_period_ns"] < config["burst_on_ns"] else 0 + + +def build_workload(config: dict) -> dict: + """Return canonical per-window, per-stack, per-channel offered bytes.""" + + config = _validate_config(config) + model = dict(MODEL_CATALOG[config["model_id"]]) + payload_bytes = model["tensor_payload_bytes"] + stack_ids = [f"hbf{index}" for index in range(config["stack_count"])] + channel_ids = [str(index) for index in range(config["channels_per_stack"])] + destinations = [(stack_id, channel_id) for stack_id in stack_ids for channel_id in channel_ids] + destination_count = len(destinations) + end_ns = config["active_ns"] + config["recovery_ns"] + exact_carry = Fraction(0, 1) + stripe_cursor = 0 + windows = [] + total_emitted = 0 + + for start_ns in range(0, end_ns, config["step_ns"]): + end_window_ns = start_ns + config["step_ns"] + multiplier = _multiplier(config, start_ns) + exact_window_bytes = Fraction( + payload_bytes * config["full_scans_per_s"] * config["step_ns"] * multiplier, + 1_000_000_000, + ) + exact_carry += exact_window_bytes + emitted_window_bytes = exact_carry.numerator // exact_carry.denominator + exact_carry -= emitted_window_bytes + + base, remainder = divmod(emitted_window_bytes, destination_count) + flat = [base] * destination_count + for offset in range(remainder): + flat[(stripe_cursor + offset) % destination_count] += 1 + stripe_cursor = (stripe_cursor + remainder) % destination_count + offered = {stack_id: {channel_id: 0 for channel_id in channel_ids} for stack_id in stack_ids} + for (stack_id, channel_id), byte_count in zip(destinations, flat): + offered[stack_id][channel_id] = byte_count + if sum(sum(channels.values()) for channels in offered.values()) != emitted_window_bytes: + raise AssertionError("uniform stripe did not conserve offered bytes") + windows.append({ + "start_ns": start_ns, + "end_ns": end_window_ns, + "stack_channel_offered_bytes": offered, + "total_offered_bytes": emitted_window_bytes, + "demand_multiplier": multiplier, + }) + total_emitted += emitted_window_bytes + + expected_active_bytes = payload_bytes * int( + Fraction(config["full_scans_per_s"] * config["active_ns"], 1_000_000_000) + ) + if exact_carry != 0 or total_emitted != expected_active_bytes: + raise AssertionError("offered-byte generation did not conserve full-scan demand") + + offered_mean_bps = payload_bytes * config["full_scans_per_s"] + metadata = { + "catalog_version": CATALOG_VERSION, + "model_id": config["model_id"], + "model_revision": model["revision"], + "model_resolved_commit": model["resolved_commit"], + "model_source_class": model["source_class"], + "model_source": model["source"], + "weight_bytes": payload_bytes, + "page_bytes": 4096, + "full_scans_per_s": config["full_scans_per_s"], + "pattern": config["pattern"], + "active_ns": config["active_ns"], + "recovery_ns": config["recovery_ns"], + "step_ns": config["step_ns"], + "stack_count": config["stack_count"], + "channels_per_stack": config["channels_per_stack"], + "stack_ids": stack_ids, + "channel_ids": channel_ids, + "burst_period_ns": config.get("burst_period_ns"), + "burst_on_ns": config.get("burst_on_ns"), + "mean_active_offered_Bps": offered_mean_bps, + "demand_formula": "tensor_payload_bytes * full_scans_per_s", + "equivalent_full_scans": expected_active_bytes // payload_bytes, + "expected_active_offered_bytes": expected_active_bytes, + "actual_total_offered_bytes": total_emitted, + "distribution": "UNIFORM_STRIPED_GLOBAL_FRACTIONAL_CARRY_ROTATING_REMAINDER", + "semantics": { + "bytes": "OFFERED_DEMAND_NOT_DELIVERED_SERVICE", + "capacity": "NOT_APPLIED_HERE_EXCESS_MUST_ENTER_FLUID_BACKLOG", + "scan": model["workload_interpretation"], + "token_per_s": "UNKNOWN", + "idle_or_recovery": "ZERO_OFFERED_READ_NOT_ZERO_PHYSICAL_IDLE_POWER", + }, + "provenance": "USER_CONFIRMED_METHOD_WITH_DOC_DERIVED_OFFICIAL_WEIGHT_BYTES", + } + if "model_card_parameters" in model: + metadata["model_card_parameters"] = model["model_card_parameters"] + metadata["moe_limit"] = ( + "FULL_235B_SCAN_IS_SYNTHETIC_PRESSURE; 22B_ACTIVE_MODEL_CARD_LABEL; " + "DO_NOT_INTERPRET_AS_ALL_WEIGHTS_PER_TOKEN" + ) + return { + "schema_version": OUTPUT_SCHEMA, + "metadata": metadata, + "canonical_config": config, + "windows": windows, + } + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--config", required=True, type=Path) + parser.add_argument("--output", required=True, type=Path) + args = parser.parse_args(argv) + config = json.loads(args.config.read_text()) + result = build_workload(config) + args.output.write_text(json.dumps(result, indent=2, sort_keys=True, allow_nan=False) + "\n") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/experiments/eq3_rate_thermal/plot_rate_thermal.py b/experiments/eq3_rate_thermal/plot_rate_thermal.py new file mode 100644 index 0000000..e9a32d2 --- /dev/null +++ b/experiments/eq3_rate_thermal/plot_rate_thermal.py @@ -0,0 +1,51 @@ +#!/usr/bin/env python3 +"""Reproduce the rate, source-power and incremental-temperature figure.""" +import argparse +import csv +import json +import os +from pathlib import Path + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument('point', type=Path) + args = parser.parse_args() + point = args.point + os.environ.setdefault('MPLCONFIGDIR', str(point / 'plot-cache')) + schedule = json.loads((point / 'schedule.json').read_text()) + profile = json.loads((point / 'profile.json').read_text()) + with (point / 'stack-temperatures.csv').open() as stream: + temperatures = list(csv.DictReader(stream)) + stacks = sorted({row['stack'] for row in temperatures if row['stack'].startswith('hbf')}) + import matplotlib + matplotlib.use('Agg') + import matplotlib.pyplot as plt + fig, axes = plt.subplots(3, 1, figsize=(10, 8), sharex=True, constrained_layout=True) + for stack in stacks: + x, rates = [], [] + for segment in schedule['segments']: + x += [segment['start_ns']/1e9, segment['end_ns']/1e9] + rates += [segment['read_Bps'].get(stack, 0)] * 2 + line, = axes[0].plot(x, [rate/1e12 for rate in rates], label=stack) + coefficient = profile['array_j_per_byte'] + profile['base_j_per_byte'] + axes[1].plot(x, [coefficient*rate for rate in rates], color=line.get_color()) + selected = [row for row in temperatures if row['stack'] == stack] + axes[2].plot([0] + [int(row['time_ns'])/1e9 for row in selected], + [0] + [float(row['increment_above_300k']) for row in selected], + color=line.get_color()) + axes[0].set_ylabel('Scenario read rate (TB/s)') + axes[1].set_ylabel('Read-induced power (W)') + axes[2].set_ylabel('Stack hotspot rise (K)') + axes[2].set_xlabel('Time (s)') + axes[0].legend(ncol=4, frameon=False) + for axis in axes: + axis.grid(alpha=.2) + fig.suptitle('Conditional rate-driven thermal simulation: 50 pJ/B\n' + 'Coupled package; idle/GPU baseline excluded; no MQSim transactions') + fig.savefig(point / 'rate-power-temperature.png', dpi=160) + plt.close(fig) + + +if __name__ == '__main__': + main() diff --git a/experiments/eq3_rate_thermal/prepare_control_stage.py b/experiments/eq3_rate_thermal/prepare_control_stage.py new file mode 100644 index 0000000..06cc995 --- /dev/null +++ b/experiments/eq3_rate_thermal/prepare_control_stage.py @@ -0,0 +1,247 @@ +#!/usr/bin/env python3 +"""Prepare immutable inputs and a run index for the approved rate/thermal stage.""" + +from __future__ import annotations + +import argparse +from datetime import datetime, timezone +import hashlib +import json +from pathlib import Path +import subprocess +import sys +from typing import Any + + +HERE = Path(__file__).resolve().parent +ROOT = HERE.parents[1] +sys.path.insert(0, str(HERE)) + +from model_workloads import build_workload + + +MODEL_IDS = { + "7B": "Qwen/Qwen2.5-7B-Instruct", + "72B": "Qwen/Qwen2.5-72B-Instruct", + "235B": "Qwen/Qwen3-235B-A22B", +} +TOPOLOGIES = { + "mixed_direct": {"tag": "mixed", "stack_count": 4}, + "all_hbf_direct": {"tag": "allhbf", "stack_count": 8}, +} +POLICIES = { + "guard_only": "P0", + "thermal_hysteresis_guard": "P1", + "read_rate_feedback_thermal_guard_v1": "P2", +} + + +def _sha256(path: Path) -> str: + return hashlib.sha256(path.read_bytes()).hexdigest() + + +def _encoded(value: Any) -> bytes: + return (json.dumps(value, indent=2, sort_keys=True, allow_nan=False) + "\n").encode() + + +def _write_immutable(path: Path, value: Any) -> None: + data = _encoded(value) + path.parent.mkdir(parents=True, exist_ok=True) + if path.exists(): + if path.read_bytes() != data: + raise FileExistsError(f"refusing to replace different frozen input: {path}") + return + path.write_bytes(data) + + +def _workload_config(*, model: str, topology: str, pattern: str, + scans: int, active_s: int, recovery_s: int) -> dict: + result = { + "schema_version": "eq3-rate-model-workload-config-v1", + "model_id": MODEL_IDS[model], + "full_scans_per_s": scans, + "pattern": pattern, + "stack_count": TOPOLOGIES[topology]["stack_count"], + "channels_per_stack": 16, + "step_ns": 20_000_000, + "active_ns": active_s * 1_000_000_000, + "recovery_ns": recovery_s * 1_000_000_000, + } + if pattern == "burst_equal_mean": + result.update(burst_period_ns=200_000_000, burst_on_ns=100_000_000) + return result + + +def _input_set(stage: Path, *, phase: str, topology: str, model: str, + pattern: str, scans: int, active_s: int, recovery_s: int) -> tuple[Path, Path, Path, dict]: + key = f"{phase}-{TOPOLOGIES[topology]['tag']}-{model}-{pattern}-{scans}scan" + config_path = stage / "inputs" / "configs" / f"{key}.json" + workload_path = stage / "inputs" / "workloads" / f"{key}.json" + scenario_path = stage / "inputs" / "scenarios" / f"{key}.json" + config = _workload_config(model=model, topology=topology, pattern=pattern, + scans=scans, active_s=active_s, recovery_s=recovery_s) + workload = build_workload(config) + mean_bps = workload["metadata"]["mean_active_offered_Bps"] + stack_count = TOPOLOGIES[topology]["stack_count"] + target_bps = max(1, min(mean_bps // stack_count, 1_536_000_000_000 * 4 // 5)) + scenario = { + "schema_version": "eq3-rate-controlled-scenario-v1", + "scenario_id": key, + "topology": topology, + "window_ns": 20_000_000, + "target_read_Bps_per_stack": target_bps, + } + _write_immutable(config_path, config) + _write_immutable(workload_path, workload) + _write_immutable(scenario_path, scenario) + return config_path, workload_path, scenario_path, workload + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--stage", type=Path, required=True) + args = parser.parse_args() + stage = args.stage.resolve(strict=True) + preflight_path = stage / "PREFLIGHT.json" + preflight = json.loads(preflight_path.read_text()) + if preflight.get("task") != "EQ3-RATE-THERMAL-CONTROL-CONTINUATION-v1": + raise ValueError("unexpected stage preflight task") + if preflight.get("pilot", {}).get("points") != 2 or preflight.get("main", {}).get("points") != 39: + raise ValueError("preflight must freeze two pilots and 39 main points") + resources = preflight.get("resources", {}) + expected_resources = { + "cpu_processes": 1, "OMP_BLAS_threads": 1, "address_limit_gib": 4, + "per_point_watchdog_s": 600, "stage_wall_s": 5400, + "host_memory_reserve_gib": 32, "host_disk_reserve_gib": 100, + } + for field, expected in expected_resources.items(): + if resources.get(field) != expected: + raise ValueError(f"preflight resource {field} differs from {expected}") + + binary = Path(preflight["thermal_binary"]).resolve(strict=True) + artifact_root = binary.parents[2] + model_dirs = {key: Path(value).resolve(strict=True) + for key, value in preflight["model_dirs"].items()} + profiles = { + topology: (stage / f"profile-{topology}.json").resolve(strict=True) + for topology in TOPOLOGIES + } + source_files = [ + HERE / "prepare_control_stage.py", HERE / "launch_control_stage.py", + HERE / "model_workloads.py", HERE / "fluid_service.py", + HERE / "run_controlled.py", + ROOT / "experiments" / "eq3_maintenance" / "read_rate_policy.py", + ROOT / "experiments" / "eq3_maintenance" / "thermal_client.py", + ] + source_locks = {str(path.relative_to(ROOT)): _sha256(path) for path in source_files} + source_revision = subprocess.run( + ["git", "rev-parse", "HEAD"], cwd=ROOT, check=True, + text=True, capture_output=True, + ).stdout.strip() + + points = [] + + def add_points(*, phase: str, topology: str, model: str, pattern: str, + scans: int, active_s: int, recovery_s: int, + policies: list[str]) -> None: + config_path, workload_path, scenario_path, workload = _input_set( + stage, phase=phase, topology=topology, model=model, pattern=pattern, + scans=scans, active_s=active_s, recovery_s=recovery_s, + ) + for policy in policies: + point_id = ( + f"RT-{phase.upper()}-{TOPOLOGIES[topology]['tag']}-{model}-" + f"{pattern}-{scans}scan-{POLICIES[policy]}" + ) + points.append({ + "ordinal": len(points) + 1, + "point_id": point_id, + "phase": phase, + "topology": topology, + "model": model, + "pattern": pattern, + "full_scans_per_s": scans, + "strategy": policy, + "active_s": active_s, + "recovery_s": recovery_s, + "profile": str(profiles[topology]), + "config": str(config_path), + "workload": str(workload_path), + "scenario": str(scenario_path), + "model_dir": str(model_dirs[topology]), + "thermal_binary": str(binary), + "artifact_root": str(artifact_root), + "output": str(stage / "points" / point_id), + "input_sha256": { + "profile": _sha256(profiles[topology]), + "config": _sha256(config_path), + "workload": _sha256(workload_path), + "scenario": _sha256(scenario_path), + }, + "expected_active_offered_bytes": workload["metadata"]["expected_active_offered_bytes"], + "expected_backlog_is_failure": False, + }) + + for topology in ("mixed_direct", "all_hbf_direct"): + add_points( + phase="pilot", topology=topology, model="235B", pattern="continuous", + scans=16, active_s=8, recovery_s=4, + policies=["read_rate_feedback_thermal_guard_v1"], + ) + for topology in ("mixed_direct", "all_hbf_direct"): + for model in ("7B", "72B", "235B"): + for pattern in ("continuous", "burst_equal_mean"): + add_points( + phase="main", topology=topology, model=model, pattern=pattern, + scans=16, active_s=20, recovery_s=10, + policies=list(POLICIES), + ) + add_points( + phase="main", topology="all_hbf_direct", model="235B", pattern="continuous", + scans=32, active_s=20, recovery_s=10, policies=list(POLICIES), + ) + if len(points) != 41 or sum(row["phase"] == "pilot" for row in points) != 2: + raise AssertionError("prepared point count does not match the frozen stage") + + model_locks = {} + for topology, directory in model_dirs.items(): + model_locks[topology] = { + name: _sha256(directory / name) + for name in ("normalized.json", "model.txt", "rc_grid.json", "rc_sensors.json") + } + index = { + "schema_version": "eq3-rate-thermal-control-run-index-v1", + "prepared_utc": datetime.now(timezone.utc).isoformat(), + "stage": str(stage), + "authority": "USER_CONFIRMED_RATE_THERMAL_CONTINUATION_AND_SATURATION_AMENDMENT", + "preflight": str(preflight_path), + "preflight_sha256": _sha256(preflight_path), + "source_root": str(ROOT), + "source_revision": source_revision, + "source_locks": source_locks, + "thermal_binary_sha256": _sha256(binary), + "model_locks": model_locks, + "resources": resources, + "point_count": len(points), + "pilot_count": 2, + "main_count": 39, + "points": points, + } + index_path = stage / "RUN_INDEX.json" + _write_immutable(index_path, index) + receipt = { + "status": "PREPARED_NOT_STARTED", + "run_index": str(index_path), + "run_index_sha256": _sha256(index_path), + "point_count": 41, + "pilot_count": 2, + "main_count": 39, + "source_revision": source_revision, + } + _write_immutable(stage / "PREPARED.json", receipt) + print(json.dumps(receipt, indent=2)) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/experiments/eq3_rate_thermal/rate_inputs.py b/experiments/eq3_rate_thermal/rate_inputs.py new file mode 100644 index 0000000..3dfd09b --- /dev/null +++ b/experiments/eq3_rate_thermal/rate_inputs.py @@ -0,0 +1,466 @@ +"""Convert prescribed HBF read-rate schedules to incremental thermal energy. + +This module is deliberately disconnected from every runtime by default. It +does not issue storage requests or estimate achieved backend throughput. +""" + +from __future__ import annotations + +import math +from typing import Any + + +SCHEMA_VERSION = 1 +THERMAL_STEP_NS = 20_000_000 +REFERENCE_READ_BPS = 1.6e12 +ARRAY_J_PER_BYTE = 40e-12 +BASE_J_PER_BYTE = 10e-12 +PROVENANCE = "SCENARIO_ASSUMPTION_USER_CONFIRMED" + + +def _mapping(value: Any, path: str) -> dict: + if not isinstance(value, dict): + raise ValueError(f"{path} must be an object") + return value + + +def _integer(value: Any, path: str, *, positive: bool = False) -> int: + if isinstance(value, bool) or not isinstance(value, int): + raise ValueError(f"{path} must be an integer") + if positive and value <= 0: + raise ValueError(f"{path} must be positive") + return value + + +def _finite_number(value: Any, path: str, *, nonnegative: bool = False) -> float: + if isinstance(value, bool) or not isinstance(value, (int, float)): + raise ValueError(f"{path} must be a finite number") + result = float(value) + if not math.isfinite(result): + raise ValueError(f"{path} must be a finite number") + if nonnegative and result < 0.0: + raise ValueError(f"{path} must be nonnegative") + return result + + +def _approved_parameter(profile: dict, key: str, expected: float) -> float: + value = _finite_number(profile.get(key), f"profile.{key}") + if value != expected: + raise ValueError(f"profile.{key} must equal the approved value {expected!r}") + return value + + +def _discover_hbf_entities(normalized: dict) -> dict[str, dict[str, Any]]: + components = normalized.get("components") + if not isinstance(components, list): + raise ValueError("normalized.components must be an array") + + stacks: dict[str, dict[str, Any]] = {} + seen_component_ids: set[str] = set() + for index, component in enumerate(components): + if not isinstance(component, dict): + raise ValueError(f"normalized.components[{index}] must be an object") + if component.get("physical_type") != "HBF" or component.get("powered") is not True: + continue + component_id = component.get("id") + stack_id = component.get("device_id") + role = component.get("role") + if not isinstance(component_id, str) or not component_id: + raise ValueError(f"normalized.components[{index}].id must be a nonempty string") + if not isinstance(stack_id, str) or not stack_id: + raise ValueError(f"normalized.components[{index}].device_id must be a nonempty string") + if component_id in seen_component_ids: + raise ValueError(f"duplicate powered HBF component id {component_id!r}") + seen_component_ids.add(component_id) + stack = stacks.setdefault(stack_id, {"base": None, "array_dies": []}) + if role == "base_die": + if stack["base"] is not None: + raise ValueError(f"HBF stack {stack_id!r} has more than one powered base_die") + stack["base"] = component_id + elif role == "array_die": + stack["array_dies"].append(component_id) + else: + raise ValueError( + f"powered HBF component {component_id!r} has unsupported role {role!r}" + ) + + if not stacks: + raise ValueError("normalized model has no powered HBF stacks") + for stack_id, stack in stacks.items(): + if stack["base"] is None: + raise ValueError(f"HBF stack {stack_id!r} has no powered base_die") + if not stack["array_dies"]: + raise ValueError(f"HBF stack {stack_id!r} has no powered array_die") + stack["array_dies"].sort() + return dict(sorted(stacks.items())) + + +def _resolve_weights(profile: dict, stacks: dict[str, dict[str, Any]]) -> dict[str, dict[str, float]]: + configured = profile.get("die_weights", {}) + configured = _mapping(configured, "profile.die_weights") + unknown = sorted(set(configured) - set(stacks)) + if unknown: + raise ValueError(f"profile.die_weights contains unknown HBF stacks: {unknown}") + + result: dict[str, dict[str, float]] = {} + for stack_id, stack in stacks.items(): + dies = stack["array_dies"] + if stack_id not in configured: + weight = 1.0 / len(dies) + result[stack_id] = {component_id: weight for component_id in dies} + continue + raw = _mapping(configured[stack_id], f"profile.die_weights.{stack_id}") + if set(raw) != set(dies): + missing = sorted(set(dies) - set(raw)) + extra = sorted(set(raw) - set(dies)) + raise ValueError( + f"profile.die_weights.{stack_id} must cover exactly the discovered dies; " + f"missing={missing}, extra={extra}" + ) + weights = { + component_id: _finite_number( + raw[component_id], + f"profile.die_weights.{stack_id}.{component_id}", + nonnegative=True, + ) + for component_id in dies + } + if not math.isclose(sum(weights.values()), 1.0, rel_tol=0.0, abs_tol=1e-12): + raise ValueError(f"profile.die_weights.{stack_id} must sum to 1") + result[stack_id] = weights + return result + + +def _resolve_channels(profile: dict, stacks: dict[str, dict[str, Any]]) -> dict[str, dict[str, Any]]: + raw_map = profile.get("channel_map") + raw_caps = profile.get("channel_capacity_Bps") + if raw_map is None and raw_caps is None: + return {} + if raw_map is None or raw_caps is None: + raise ValueError("profile.channel_map and profile.channel_capacity_Bps must be provided together") + raw_map = _mapping(raw_map, "profile.channel_map") + raw_caps = _mapping(raw_caps, "profile.channel_capacity_Bps") + if set(raw_map) != set(raw_caps): + raise ValueError("profile.channel_map and profile.channel_capacity_Bps must cover the same stacks") + unknown_stacks = sorted(set(raw_map) - set(stacks)) + if unknown_stacks: + raise ValueError(f"profile.channel_map contains unknown HBF stacks: {unknown_stacks}") + + result = {} + for stack_id in sorted(raw_map): + mapping = _mapping(raw_map[stack_id], f"profile.channel_map.{stack_id}") + capacities = _mapping(raw_caps[stack_id], f"profile.channel_capacity_Bps.{stack_id}") + if not mapping: + raise ValueError(f"profile.channel_map.{stack_id} must not be empty") + if set(mapping) != set(capacities): + raise ValueError( + f"profile channel map and capacities for {stack_id!r} must cover the same channels" + ) + actual_dies = set(stacks[stack_id]["array_dies"]) + checked_mapping = {} + checked_capacities = {} + for channel_id, component_id in mapping.items(): + if not isinstance(channel_id, str) or not channel_id: + raise ValueError(f"profile.channel_map.{stack_id} channel IDs must be nonempty strings") + if component_id not in actual_dies: + raise ValueError( + f"profile.channel_map.{stack_id}.{channel_id} refers to unknown array die " + f"{component_id!r}" + ) + capacity = _finite_number( + capacities[channel_id], + f"profile.channel_capacity_Bps.{stack_id}.{channel_id}", + nonnegative=True, + ) + if capacity == 0.0: + raise ValueError( + f"profile.channel_capacity_Bps.{stack_id}.{channel_id} must be positive" + ) + checked_mapping[channel_id] = component_id + checked_capacities[channel_id] = capacity + result[stack_id] = {"map": checked_mapping, "capacities": checked_capacities} + return result + + +def _validate_schedule( + schedule: dict, + stack_ids: set[str], + reference_read_bps: float, + channels: dict[str, dict[str, Any]], +) -> tuple[int, list]: + if schedule.get("schema_version") != SCHEMA_VERSION: + raise ValueError("schedule.schema_version must equal 1") + end_ns = _integer(schedule.get("end_ns"), "schedule.end_ns", positive=True) + segments = schedule.get("segments") + if not isinstance(segments, list) or not segments: + raise ValueError("schedule.segments must be a nonempty array") + + validated = [] + expected_start = 0 + for index, segment in enumerate(segments): + segment = _mapping(segment, f"schedule.segments[{index}]") + start_ns = _integer(segment.get("start_ns"), f"schedule.segments[{index}].start_ns") + segment_end_ns = _integer(segment.get("end_ns"), f"schedule.segments[{index}].end_ns") + if start_ns != expected_start: + raise ValueError("schedule segments must continuously cover [0, end_ns) without gaps or overlap") + if segment_end_ns <= start_ns or segment_end_ns > end_ns: + raise ValueError(f"schedule.segments[{index}] has an invalid half-open interval") + read_bps = _mapping(segment.get("read_Bps", {}), f"schedule.segments[{index}].read_Bps") + unknown = sorted(set(read_bps) - stack_ids) + if unknown: + raise ValueError(f"schedule.segments[{index}].read_Bps has unknown HBF stacks: {unknown}") + channel_field_present = "channel_read_Bps" in segment + raw_channel_rates = _mapping( + segment.get("channel_read_Bps", {}), + f"schedule.segments[{index}].channel_read_Bps", + ) + unknown_channel_stacks = sorted(set(raw_channel_rates) - set(channels)) + if unknown_channel_stacks: + raise ValueError( + f"schedule.segments[{index}].channel_read_Bps has stacks without a channel map: " + f"{unknown_channel_stacks}" + ) + rates = {} + segment_channel_rates = {} + for stack_id in stack_ids: + explicit_channels = None + if channel_field_present and stack_id in raw_channel_rates: + configured = channels[stack_id] + raw_stack_channels = _mapping( + raw_channel_rates[stack_id], + f"schedule.segments[{index}].channel_read_Bps.{stack_id}", + ) + unknown_channels = sorted(set(raw_stack_channels) - set(configured["map"])) + if unknown_channels: + raise ValueError( + f"schedule.segments[{index}].channel_read_Bps.{stack_id} has unknown " + f"channels: {unknown_channels}" + ) + explicit_channels = {} + for channel_id in configured["map"]: + channel_rate = _finite_number( + raw_stack_channels.get(channel_id, 0.0), + f"schedule.segments[{index}].channel_read_Bps.{stack_id}.{channel_id}", + nonnegative=True, + ) + if channel_rate > configured["capacities"][channel_id]: + raise ValueError( + f"schedule.segments[{index}].channel_read_Bps.{stack_id}.{channel_id} " + "exceeds its configured channel capacity" + ) + explicit_channels[channel_id] = channel_rate + channel_total = sum(explicit_channels.values()) + if stack_id in read_bps: + requested_total = _finite_number( + read_bps[stack_id], + f"schedule.segments[{index}].read_Bps.{stack_id}", + nonnegative=True, + ) + if not math.isclose(requested_total, channel_total, rel_tol=1e-12, abs_tol=1e-6): + raise ValueError( + f"schedule.segments[{index}] total read_Bps and channel_read_Bps " + f"disagree for {stack_id!r}" + ) + rate = channel_total + else: + rate = _finite_number( + read_bps.get(stack_id, 0.0), + f"schedule.segments[{index}].read_Bps.{stack_id}", + nonnegative=True, + ) + if rate > reference_read_bps: + raise ValueError( + f"schedule.segments[{index}].read_Bps.{stack_id} exceeds the approved " + f"{reference_read_bps:g} B/s envelope" + ) + if stack_id in channels and rate > sum(channels[stack_id]["capacities"].values()): + raise ValueError(f"read_Bps.{stack_id} exceeds the configured aggregate channel capacity") + rates[stack_id] = rate + segment_channel_rates[stack_id] = explicit_channels + validated.append((start_ns, segment_end_ns, rates, segment_channel_rates)) + expected_start = segment_end_ns + if expected_start != end_ns: + raise ValueError("schedule segments must continuously cover [0, end_ns) without gaps or overlap") + return end_ns, validated + + +def build_windows( + profile: dict, + schedule: dict, + normalized: dict, + step_ns: int = THERMAL_STEP_NS, +) -> dict: + """Build 20 ms component-energy windows from a prescribed rate schedule. + + All time intervals are integer-nanosecond, half-open intervals. A stack + omitted from a segment has zero *read* energy in that segment; no idle + power is inferred. + """ + + profile = _mapping(profile, "profile") + schedule = _mapping(schedule, "schedule") + normalized = _mapping(normalized, "normalized") + if profile.get("schema_version") != SCHEMA_VERSION: + raise ValueError("profile.schema_version must equal 1") + if profile.get("provenance") != PROVENANCE: + raise ValueError(f"profile.provenance must equal {PROVENANCE!r}") + reference_read_bps = _approved_parameter(profile, "reference_read_Bps", REFERENCE_READ_BPS) + array_j_per_byte = _approved_parameter(profile, "array_j_per_byte", ARRAY_J_PER_BYTE) + base_j_per_byte = _approved_parameter(profile, "base_j_per_byte", BASE_J_PER_BYTE) + step_ns = _integer(step_ns, "step_ns", positive=True) + if step_ns != THERMAL_STEP_NS: + raise ValueError(f"step_ns must equal the approved thermal window {THERMAL_STEP_NS}") + + stacks = _discover_hbf_entities(normalized) + weights = _resolve_weights(profile, stacks) + channels = _resolve_channels(profile, stacks) + end_ns, segments = _validate_schedule( + schedule, set(stacks), reference_read_bps, channels + ) + + component_ids = sorted( + component_id + for stack in stacks.values() + for component_id in [stack["base"], *stack["array_dies"]] + ) + windows = [] + segment_index = 0 + for window_start_ns in range(0, end_ns, step_ns): + window_end_ns = min(window_start_ns + step_ns, end_ns) + duration_s = (window_end_ns - window_start_ns) * 1e-9 + bytes_by_stack = {stack_id: 0.0 for stack_id in stacks} + die_bytes_by_stack = { + stack_id: {component_id: 0.0 for component_id in stack["array_dies"]} + for stack_id, stack in stacks.items() + } + channel_bytes_by_stack = { + stack_id: {channel_id: 0.0 for channel_id in config["map"]} + for stack_id, config in channels.items() + } + uniform_positive = {stack_id: False for stack_id in stacks} + while segment_index < len(segments) and segments[segment_index][1] <= window_start_ns: + segment_index += 1 + scan_index = segment_index + while scan_index < len(segments): + segment_start_ns, segment_end_ns, rates, segment_channel_rates = segments[scan_index] + if segment_start_ns >= window_end_ns: + break + overlap_ns = min(window_end_ns, segment_end_ns) - max(window_start_ns, segment_start_ns) + if overlap_ns > 0: + for stack_id, rate in rates.items(): + # Divide the exact integer duration before multiplying by a + # potentially large B/s value. This avoids the avoidable + # rounding introduced by a ~1e19 intermediate product. + overlap_s = overlap_ns / 1_000_000_000 + interval_bytes = rate * overlap_s + bytes_by_stack[stack_id] += interval_bytes + explicit_channels = segment_channel_rates[stack_id] + if explicit_channels is None: + if rate > 0.0: + uniform_positive[stack_id] = True + for component_id, weight in weights[stack_id].items(): + die_bytes_by_stack[stack_id][component_id] += interval_bytes * weight + else: + for channel_id, channel_rate in explicit_channels.items(): + channel_bytes = channel_rate * overlap_s + channel_bytes_by_stack[stack_id][channel_id] += channel_bytes + component_id = channels[stack_id]["map"][channel_id] + die_bytes_by_stack[stack_id][component_id] += channel_bytes + scan_index += 1 + + component_energy_j = {component_id: 0.0 for component_id in component_ids} + stack_receipts = {} + for stack_id, stack in stacks.items(): + byte_count = bytes_by_stack[stack_id] + array_energy_j = byte_count * array_j_per_byte + base_energy_j = byte_count * base_j_per_byte + component_energy_j[stack["base"]] = base_energy_j + for component_id, die_bytes in die_bytes_by_stack[stack_id].items(): + component_energy_j[component_id] = die_bytes * array_j_per_byte + assigned_array_energy_j = sum( + component_energy_j[component_id] for component_id in stack["array_dies"] + ) + if not math.isclose( + assigned_array_energy_j, array_energy_j, rel_tol=1e-12, abs_tol=1e-15 + ): + raise AssertionError(f"array energy assignment does not conserve {stack_id!r}") + total_energy_j = array_energy_j + base_energy_j + if uniform_positive[stack_id]: + activity_mode = "UNIFORM_ASSUMED_CHANNEL_ACTIVITY_UNKNOWN" + active_channel_count = None + channel_requested_bytes = None + elif stack_id in channel_bytes_by_stack: + activity_mode = "EXPLICIT_PRESCRIBED_CHANNEL_RATES" + active_channel_count = sum( + value > 0.0 for value in channel_bytes_by_stack[stack_id].values() + ) + channel_requested_bytes = channel_bytes_by_stack[stack_id] + else: + activity_mode = "UNIFORM_ASSUMED_CHANNEL_ACTIVITY_UNKNOWN" + active_channel_count = None + channel_requested_bytes = None + stack_receipts[stack_id] = { + "requested_bytes": byte_count, + "modelled_bytes": byte_count, + "channel_activity": activity_mode, + "active_channel_count": active_channel_count, + "channel_requested_bytes": channel_requested_bytes, + "source_energy_j": { + "array": array_energy_j, + "base": base_energy_j, + "total": total_energy_j, + }, + "source_power_w": { + "array": array_energy_j / duration_s, + "base": base_energy_j / duration_s, + "total": total_energy_j / duration_s, + }, + } + windows.append( + { + "start_ns": window_start_ns, + "end_ns": window_end_ns, + "component_energy_j": component_energy_j, + "stacks": stack_receipts, + } + ) + + entity_mapping = {} + for stack_id, stack in stacks.items(): + entity_mapping[stack_id] = { + "base": stack["base"], + "array_dies": list(stack["array_dies"]), + "die_weights": weights[stack_id], + "weight_mode": "EXPLICIT" if stack_id in profile.get("die_weights", {}) else "UNIFORM", + "channel_map": channels.get(stack_id, {}).get("map"), + "channel_capacity_Bps": channels.get(stack_id, {}).get("capacities"), + "channel_fallback": "UNIFORM_ASSUMED_CHANNEL_ACTIVITY_UNKNOWN", + } + return { + "schema_version": "eq3-rate-thermal-windows-v1", + "metadata": { + "mode": "DEFAULT_OFF_PURE_RATE_TO_INCREMENTAL_ENERGY", + "provenance": PROVENANCE, + "step_ns": step_ns, + "end_ns": end_ns, + "interval_semantics": "INTEGER_NS_HALF_OPEN", + "rate_semantics": "PRESCRIBED_READ_RATE_NOT_BACKEND_THROUGHPUT", + "idle_semantics": "ZERO_READ_ONLY_NOT_ZERO_IDLE", + "energy_semantics": "INCREMENTAL_READ_ENERGY_ONLY", + "entity_mapping": entity_mapping, + }, + "parameters": { + "reference_read_Bps": reference_read_bps, + "array_j_per_byte": array_j_per_byte, + "base_j_per_byte": base_j_per_byte, + "reference_power_w": { + "array": reference_read_bps * array_j_per_byte, + "base": reference_read_bps * base_j_per_byte, + "total": reference_read_bps * (array_j_per_byte + base_j_per_byte), + }, + }, + "windows": windows, + } + + +__all__ = ["build_windows"] diff --git a/experiments/eq3_rate_thermal/run_controlled.py b/experiments/eq3_rate_thermal/run_controlled.py new file mode 100644 index 0000000..0a5babe --- /dev/null +++ b/experiments/eq3_rate_thermal/run_controlled.py @@ -0,0 +1,490 @@ +#!/usr/bin/env python3 +"""Default-disconnected fluid read-rate/thermal feedback experiment runner. + +This runner models byte arrivals, FIFO service and incremental read energy. It +does not issue MQSim commands and does not represent backend completion times. +""" +from __future__ import annotations + +import argparse +from dataclasses import asdict, is_dataclass +from datetime import datetime, timezone +import hashlib +import json +import os +from pathlib import Path +import platform +import resource +import shutil +import subprocess +import sys +import time +from typing import Any + + +HERE = Path(__file__).resolve().parent +ROOT = HERE.parents[1] +MAINTENANCE = ROOT / "experiments" / "eq3_maintenance" +sys.path.insert(0, str(HERE)) +sys.path.insert(0, str(MAINTENANCE)) + +from read_rate_policy import EngineeringProfile, ReadRatePolicy, StackWindowFacts, WindowFacts +from thermal_client import ThermalService + + +WINDOW_NS = 20_000_000 +BASELINE_STACK_BPS = 1_536_000_000_000 +ARRAY_J_PER_BYTE = 40e-12 +BASE_J_PER_BYTE = 10e-12 +ALLOWED_TOPOLOGIES = {"mixed_direct", "all_hbf_direct"} +ALLOWED_STRATEGIES = {"guard_only", "thermal_hysteresis_guard", + "read_rate_feedback_thermal_guard_v1"} + + +def _save(path: Path, value: Any) -> None: + path.write_text(json.dumps(value, indent=2, allow_nan=False) + "\n") + + +def _sha256(path: Path) -> str: + return hashlib.sha256(path.read_bytes()).hexdigest() + + +def _plain(value: Any) -> Any: + if is_dataclass(value): + return asdict(value) + if isinstance(value, dict): + return {key: _plain(item) for key, item in value.items()} + if isinstance(value, (list, tuple)): + return [_plain(item) for item in value] + return value + + +def _field(value: Any, key: str) -> Any: + return value[key] if isinstance(value, dict) else getattr(value, key) + + +def _validate_inputs(profile: dict, workload: dict, scenario: dict, normalized: dict, + strategy: str) -> tuple[list[str], dict[str, dict[str, str]], dict[str, dict[str, int]]]: + if strategy not in ALLOWED_STRATEGIES: + raise ValueError("unsupported policy strategy") + if scenario.get("schema_version") != "eq3-rate-controlled-scenario-v1": + raise ValueError("unsupported controlled scenario schema") + if not isinstance(scenario.get("scenario_id"), str) or not scenario["scenario_id"]: + raise ValueError("scenario_id must be nonempty") + if set(scenario) != {"schema_version", "scenario_id", "topology", "window_ns", + "target_read_Bps_per_stack"}: + raise ValueError("controlled scenario contains missing or unknown fields") + if scenario.get("topology") not in ALLOWED_TOPOLOGIES: + raise ValueError("only mixed_direct and all_hbf_direct are currently supported") + if profile.get("schema_version") != 1: + raise ValueError("profile schema_version must equal 1") + if profile.get("provenance") != "SCENARIO_ASSUMPTION_USER_CONFIRMED": + raise ValueError("profile provenance is not the frozen user-confirmed rate-energy scenario") + if profile.get("reference_read_Bps") != 1_600_000_000_000: + raise ValueError("profile reference_read_Bps must remain 1.6 TB/s") + if profile.get("array_j_per_byte") != ARRAY_J_PER_BYTE or profile.get("base_j_per_byte") != BASE_J_PER_BYTE: + raise ValueError("runner requires the user-confirmed 40/10 pJ/B incremental energy profile") + channel_map = profile.get("channel_map") + capacities = profile.get("channel_capacity_Bps") + if not isinstance(channel_map, dict) or not channel_map or set(channel_map) != set(capacities or {}): + raise ValueError("profile must provide matching channel maps and capacities") + stacks = sorted(channel_map) + components = {row.get("id") for row in normalized.get("components", [])} + checked_capacities = {} + for stack in stacks: + if set(channel_map[stack]) != set(capacities[stack]): + raise ValueError(f"channel map/capacity mismatch for {stack}") + if set(channel_map[stack].values()) - components or f"{stack}.base" not in components: + raise ValueError(f"profile mapping is not present in the thermal model for {stack}") + checked_capacities[stack] = {} + for channel, raw in capacities[stack].items(): + value = int(raw) + if value <= 0 or value != raw: + raise ValueError("channel capacities must be positive integer B/s") + checked_capacities[stack][channel] = value + windows = workload.get("windows") + if workload.get("schema_version") != "eq3-rate-model-workload-v1": + raise ValueError("unsupported model workload schema") + workload_metadata = workload.get("metadata", {}) + if (workload_metadata.get("stack_ids") not in (None, stacks) or + workload_metadata.get("step_ns", WINDOW_NS) != WINDOW_NS): + raise ValueError("workload metadata stack or step identity differs from the profile") + if not isinstance(windows, list) or not windows: + raise ValueError("workload.windows must be a nonempty array") + expected = 0 + for index, window in enumerate(windows): + if window.get("start_ns") != expected or window.get("end_ns") != expected + WINDOW_NS: + raise ValueError(f"workload window {index} is not contiguous 20 ms") + offered = window.get("stack_channel_offered_bytes") + if not isinstance(offered, dict) or set(offered) != set(stacks): + raise ValueError(f"workload window {index} must exactly cover profile stacks") + for stack, channels in offered.items(): + if set(channels) != set(channel_map[stack]): + raise ValueError(f"workload window {index} must exactly cover profile channels") + if any(isinstance(value, bool) or not isinstance(value, int) or value < 0 + for value in channels.values()): + raise ValueError("offered byte counts must be nonnegative integers") + if window.get("total_offered_bytes") != sum(sum(row.values()) for row in offered.values()): + raise ValueError(f"workload window {index} total_offered_bytes does not conserve") + expected += WINDOW_NS + if (workload_metadata.get("active_ns") is not None and + workload_metadata.get("active_ns") + workload_metadata.get("recovery_ns", 0) != expected): + raise ValueError("workload active/recovery duration does not match its windows") + actual_total = sum(window["total_offered_bytes"] for window in windows) + if workload_metadata.get("actual_total_offered_bytes") not in (None, actual_total): + raise ValueError("workload metadata total offered bytes does not match its windows") + target = scenario.get("target_read_Bps_per_stack") + if isinstance(target, bool) or not isinstance(target, int) or target <= 0: + raise ValueError("scenario.target_read_Bps_per_stack must be an explicit positive integer") + if scenario.get("window_ns", WINDOW_NS) != WINDOW_NS: + raise ValueError("scenario window must remain 20 ms") + physical = {stack: sum(checked_capacities[stack].values()) for stack in stacks} + if len(set(physical.values())) != 1 or next(iter(physical.values())) != BASELINE_STACK_BPS: + raise ValueError("v1 requires the frozen OCP Grade 2 physical capacity of 1.536 TB/s per stack") + mean_total = workload.get("metadata", {}).get("mean_active_offered_Bps") + if isinstance(mean_total, bool) or not isinstance(mean_total, int) or mean_total <= 0: + raise ValueError("workload metadata must state positive integer mean_active_offered_Bps") + expected_target = max(1, min(mean_total // len(stacks), BASELINE_STACK_BPS * 4 // 5)) + if target != expected_target: + raise ValueError( + "scenario target must equal min(mean active offered rate per stack, 0.8 physical capacity)" + ) + return stacks, channel_map, checked_capacities + + +def _served_energy(served: dict[str, dict[str, int]], channel_map: dict[str, dict[str, str]]) -> tuple[dict, dict]: + component = {} + stacks = {} + for stack in sorted(channel_map): + expected = set(channel_map[stack]) + actual = served.get(stack, {}) + if set(actual) != expected: + raise ValueError(f"fluid served channel coverage mismatch for {stack}") + total = 0 + array_j = 0.0 + for channel, target in channel_map[stack].items(): + byte_count = actual[channel] + if isinstance(byte_count, bool) or not isinstance(byte_count, int) or byte_count < 0: + raise ValueError("fluid served bytes must be nonnegative integers") + total += byte_count + value = byte_count * ARRAY_J_PER_BYTE + component[target] = component.get(target, 0.0) + value + array_j += value + base_j = total * BASE_J_PER_BYTE + component[f"{stack}.base"] = base_j + stacks[stack] = {"served_bytes": total, "array_energy_j": array_j, + "base_energy_j": base_j, "total_energy_j": array_j + base_j} + return component, stacks + + +def _fluid_stack_receipts(result: Any) -> dict: + receipts = _field(result, "stacks") + return {key: _plain(value) for key, value in receipts.items()} + + +def _weighted_percentile(histogram: dict[int, int], percentile: int) -> int | None: + total = sum(histogram.values()) + if total == 0: + return None + rank = (percentile * total + 99) // 100 + cumulative = 0 + for delay, byte_count in sorted(histogram.items()): + cumulative += byte_count + if cumulative >= rank: + return delay + raise AssertionError("delay histogram percentile did not reach rank") + + +def execute_loop(*, profile: dict, workload: dict, scenario: dict, normalized: dict, + strategy: str, fluid: Any, thermal: Any, + sinks: dict[str, Any] | None = None) -> dict: + """Execute a closed loop with injected fluid and thermal services for fixed tests.""" + stacks, channel_map, _ = _validate_inputs(profile, workload, scenario, normalized, strategy) + baseline = BASELINE_STACK_BPS * WINDOW_NS // 1_000_000_000 + minimum = baseline // 10 + step = baseline // 20 + target_bps = scenario["target_read_Bps_per_stack"] + policy_profile = EngineeringProfile( + profile_id=scenario.get("scenario_id", "UNNAMED") + ":" + strategy, + enabled=True, strategy=strategy, window_ns=WINDOW_NS, + target_bytes_per_s=target_bps, target_latency_p95_ns=None, + tolerance_fraction=0.05, step_bytes=step, + minimum_budget_bytes=minimum, maximum_budget_bytes=baseline, + severe_budget_bytes=0, light_fraction=0.5) + policy = ReadRatePolicy(policy_profile) + budgets = {stack: baseline for stack in stacks} + cumulative = {"arrived_bytes": 0, "served_bytes": 0, "energy_j": 0.0} + peaks = {} + records = {"rates": [], "control": [], "energy": [], "thermal": []} + delay_histograms = {stack: {} for stack in stacks} + maximum_oldest_wait = {stack: 0 for stack in stacks} + last_thermal = None + + def emit(name: str, row: dict) -> None: + records[name].append(row) + if sinks and name in sinks: + sinks[name].write(json.dumps(row, allow_nan=False) + "\n") + sinks[name].flush() + + for window in workload["windows"]: + start_ns, end_ns = window["start_ns"], window["end_ns"] + offered = window.get("stack_channel_offered_bytes", {}) + fluid_result = fluid.advance(start_ns, end_ns, offered, budgets) + served = _plain(_field(fluid_result, "served_by_channel")) + stack_receipts = _fluid_stack_receipts(fluid_result) + component_energy, stack_energy = _served_energy(served, channel_map) + offered_total = sum(sum(channels.values()) for channels in offered.values()) + served_total = sum(row["served_bytes"] for row in stack_energy.values()) + for stack, receipt in stack_receipts.items(): + maximum_oldest_wait[stack] = max(maximum_oldest_wait[stack], receipt["oldest_wait_ns"] or 0) + for bucket in receipt.get("delivered_delay_histogram_bytes", []): + delay = int(bucket["delay_ns"]) + delay_histograms[stack][delay] = delay_histograms[stack].get(delay, 0) + int(bucket["bytes"]) + cumulative["arrived_bytes"] += offered_total + cumulative["served_bytes"] += served_total + window_energy = sum(component_energy.values()) + cumulative["energy_j"] += window_energy + thermal_result = thermal.advance(start_ns, end_ns, component_energy) + last_thermal = thermal_result + for stack, temperature in thermal_result["temperatures"].items(): + peaks[stack] = max(peaks.get(stack, temperature), temperature) + + policy_adapter = {} + for stack, receipt in stack_receipts.items(): + utilization = receipt["delivered_bytes"] / baseline + policy_adapter[stack] = { + "modelled_fluid_capacity_utilization": utilization, + "modelled_fluid_capacity_saturated": utilization >= 1.0, + "fluid_byte_weighted_latency_p95_ns": receipt["latency_p95_ns"], + "utilization_semantics": "MODELLED_FLUID_PHYSICAL_CHANNEL_CAPACITY_NOT_NATIVE_BACKEND_BUSY", + "latency_semantics": "FLUID_WINDOW_QUANTIZED_BYTE_WEIGHTED_NOT_BACKEND_LATENCY", + } + rate_row = { + "start_ns": start_ns, "end_ns": end_ns, + "semantics": "MODELLED_FLUID_BYTE_FIFO_NOT_MQSIM_COMPLETION", + "backend_latency_ns": None, + "budget_by_stack": dict(budgets), + "offered_by_channel": offered, + "served_by_channel": served, + "stacks": stack_receipts, + "fluid_receipt": _plain(fluid_result), + "policy_fact_adapter_by_stack": policy_adapter, + "window_arrived_bytes": offered_total, + "window_served_bytes": served_total, + "cumulative": dict(cumulative), + } + emit("rates", rate_row) + emit("energy", {"start_ns": start_ns, "end_ns": end_ns, + "component_energy_j": component_energy, + "stack_energy_j": stack_energy, + "window_total_j": window_energy, + "cumulative_total_j": cumulative["energy_j"]}) + emit("thermal", thermal_result) + + facts = [] + for stack in stacks: + receipt = stack_receipts[stack] + facts.append(StackWindowFacts( + stack_id=stack, + offered_bytes=int(receipt["offered_bytes"]), + delivered_bytes=int(receipt["delivered_bytes"]), + backlog_bytes=int(receipt["backlog_bytes"]), + oldest_wait_ns=int(receipt["oldest_wait_ns"] or 0), + latency_p95_ns=receipt["latency_p95_ns"], + censored_requests=0, + gate_limited=bool(receipt["backlog_bytes"] > 0 and + receipt["delivered_bytes"] == budgets[stack]), + backend_busy_fraction=policy_adapter[stack]["modelled_fluid_capacity_utilization"], + resource_busy=policy_adapter[stack]["modelled_fluid_capacity_saturated"], + maintenance_due_bytes=0, + maintenance_earliest_deadline_ns=None, + retry_count=None, + uecc_count=None)) + guard_states = {stack: thermal_result["stack_states"][stack] for stack in stacks} + rank = {"normal": 0, "light": 1, "severe": 2, "shutdown": 3} + worst = max(guard_states.values(), key=rank.get) + decision = policy.evaluate(WindowFacts( + start_ns=start_ns, end_ns=end_ns, guard_state=worst, + stacks=tuple(facts), current_budget_bytes=dict(budgets), + hysteresis_budget_bytes={stack: thermal_result["hysteresis_budget_bytes"][stack] + for stack in stacks}, + guard_states=guard_states)) + next_budgets = {row.stack_id: row.budget_bytes for row in decision.stack_decisions} + serialized_decision = _plain(decision) + serialized_decision["library_fact_semantics"] = serialized_decision["fact_semantics"] + serialized_decision["fact_semantics"] = "MODELLED_FLUID_BYTE_DELIVERY_NOT_ACTUAL_FABRIC_DELIVERY" + emit("control", {"observed_window_start_ns": start_ns, + "observed_window_end_ns": end_ns, + "applies_to_window_start_ns": end_ns, + "causality": "COMPLETED_WINDOW_FACTS_AFFECT_NEXT_WINDOW_ONLY", + "policy_fact_adapter_semantics": { + "backend_busy_fraction": "MODELLED_FLUID_CAPACITY_UTILIZATION_NOT_NATIVE_BACKEND_BUSY", + "resource_busy": "MODELLED_FLUID_PHYSICAL_CAPACITY_SATURATION_NOT_NATIVE_RESOURCE_OBSERVATION", + "latency_p95_ns": "FLUID_WINDOW_QUANTIZED_BYTE_WEIGHTED_NOT_BACKEND_LATENCY", + }, + "guard_states": guard_states, + "current_budget_bytes": dict(budgets), + "decision": serialized_decision, + "next_budget_bytes": next_budgets}) + budgets = next_budgets + + final_backlog = sum(int(row["backlog_bytes"]) for row in _fluid_stack_receipts(fluid_result).values()) + if cumulative["arrived_bytes"] != cumulative["served_bytes"] + final_backlog: + raise AssertionError("end-to-end byte conservation failed") + return { + "summary": { + **cumulative, + "final_backlog_bytes": final_backlog, + "byte_conservation_error": cumulative["arrived_bytes"] - cumulative["served_bytes"] - final_backlog, + "peak_k_by_stack": peaks, + "final_k_by_stack": last_thermal["temperatures"] if last_thermal else {}, + "final_guard_states": last_thermal["stack_states"] if last_thermal else {}, + "final_budget_bytes": budgets, + "fluid_wait_by_stack": { + stack: { + "delay_semantics": "FLUID_WINDOW_QUANTIZED_BYTE_WEIGHTED_NOT_BACKEND_LATENCY", + "delivered_delay_p95_ns": _weighted_percentile(delay_histograms[stack], 95), + "delivered_delay_p99_ns": _weighted_percentile(delay_histograms[stack], 99), + "maximum_oldest_backlog_wait_ns": maximum_oldest_wait[stack], + "delivered_delay_histogram_bytes": [ + {"delay_ns": delay, "bytes": byte_count} + for delay, byte_count in sorted(delay_histograms[stack].items()) + ], + } for stack in stacks + }, + }, + "policy_profile": _plain(policy_profile), + "semantics": { + "service": "MODELLED_FLUID_BYTE_FIFO", + "policy_fact_semantics": "MODELLED_FLUID_BYTE_DELIVERY_NOT_ACTUAL_FABRIC_DELIVERY", + "thermal_energy": "SERVED_BYTES_TIMES_USER_CONFIRMED_40_10_PJ_PER_BYTE", + "backend_latency": "UNKNOWN", + "policy_capacity_facts": "MODELLED_FLUID_NOT_NATIVE_BACKEND_OBSERVATION", + "mqsim_or_fabric_completion": False, + "idle_power": "UNKNOWN_NOT_INCLUDED", + "gpu_self_power": "ZERO_INCREMENT_NOT_PHYSICAL_IDLE", + "age_retention_ecc": "UNKNOWN_NOT_INFERRED", + "maintenance": "UNAVAILABLE_IN_THIS_FLUID_PATH", + }, + "records": records, + } + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--profile", type=Path, required=True) + parser.add_argument("--model-dir", type=Path, required=True) + parser.add_argument("--workload", type=Path, required=True) + parser.add_argument("--scenario", type=Path, required=True) + parser.add_argument("--thermal-binary", type=Path, required=True) + parser.add_argument("--artifact-root", type=Path, required=True) + parser.add_argument("--output", type=Path, required=True) + parser.add_argument("--strategy", choices=sorted(ALLOWED_STRATEGIES), required=True) + parser.add_argument("--address-limit-gib", type=int, default=4) + args = parser.parse_args() + output = args.output.resolve() + output.mkdir(parents=True, exist_ok=False) + started = time.monotonic() + try: + if args.address_limit_gib <= 0: + raise ValueError("address limit must be positive") + for key in ("OMP_NUM_THREADS", "OPENBLAS_NUM_THREADS", "MKL_NUM_THREADS", "NUMEXPR_NUM_THREADS"): + os.environ[key] = "1" + os.environ["CUDA_VISIBLE_DEVICES"] = "" + profile = json.loads(args.profile.read_text()) + workload = json.loads(args.workload.read_text()) + scenario = json.loads(args.scenario.read_text()) + normalized = json.loads((args.model_dir / "normalized.json").read_text()) + stacks, channel_map, capacities = _validate_inputs(profile, workload, scenario, normalized, args.strategy) + for source, name in ((args.profile, "profile.json"), (args.workload, "workload.json"), + (args.scenario, "scenario.json")): + shutil.copyfile(source, output / name) + manifest = { + "experiment_kind": "MODELLED_FLUID_READ_RATE_COUPLED_THERMAL_CONTROL", + "started_utc": datetime.now(timezone.utc).isoformat(), + "python": sys.version, "platform": platform.platform(), + "strategy": args.strategy, + "environment_id": "eq3-thermal-cpu-v1", + "source_sha256": { + "experiments/eq3_rate_thermal/run_controlled.py": _sha256(HERE / "run_controlled.py"), + "experiments/eq3_rate_thermal/rate_inputs.py": _sha256(HERE / "rate_inputs.py"), + "experiments/eq3_rate_thermal/fluid_service.py": _sha256(HERE / "fluid_service.py"), + "experiments/eq3_maintenance/read_rate_policy.py": _sha256(MAINTENANCE / "read_rate_policy.py"), + "experiments/eq3_maintenance/thermal_client.py": _sha256(MAINTENANCE / "thermal_client.py"), + }, + "input_sha256": {name: _sha256(output / name) for name in + ("profile.json", "workload.json", "scenario.json")}, + "model_dir": str(args.model_dir.resolve()), + "thermal_binary": str(args.thermal_binary.resolve()), + "thermal_binary_sha256": _sha256(args.thermal_binary), + "window_ns": WINDOW_NS, "cpu_threads": 1, "gpu_count": 0, + "address_limit_gib": args.address_limit_gib, + "scope": "CONDITIONAL_MODELLED_FLUID_NOT_ACTUAL_MQSIM_OR_FABRIC", + "backend_latency": "UNKNOWN", + "supported_topologies": sorted(ALLOWED_TOPOLOGIES), + "relay_dash": "UNAVAILABLE_IN_THIS_RUNNER", + "thermal_limits_k": {"hbf": [353.15, 363.15, 378.15], + "gpu": [363.15, 373.15, 383.15]}, + "guard_timing": {"action_delay_ns": WINDOW_NS, + "recovery_dwell_ns": 100_000_000, "hysteresis_k": 2.0}, + "disk_free_bytes": shutil.disk_usage(output).free, + } + revision = subprocess.run(["git", "rev-parse", "HEAD"], cwd=ROOT, + check=True, text=True, capture_output=True).stdout.strip() + dirty = subprocess.run(["git", "status", "--porcelain"], cwd=ROOT, + check=True, text=True, capture_output=True).stdout + manifest["source_revision"] = revision + manifest["source_dirty"] = bool(dirty) + manifest["source_dirty_paths"] = [line[3:] for line in dirty.splitlines() if len(line) >= 4] + _save(output / "manifest.json", manifest) + resource.setrlimit(resource.RLIMIT_AS, (args.address_limit_gib * 1024**3,) * 2) + resource.setrlimit(resource.RLIMIT_CORE, (0, 0)) + from fluid_service import FluidService + fluid = FluidService(channel_map, capacities, window_ns=WINDOW_NS) + baseline = BASELINE_STACK_BPS * WINDOW_NS // 1_000_000_000 + sinks = {name: (output / f"{name}.jsonl").open("w") + for name in ("rates", "control", "energy", "thermal")} + try: + with ThermalService(args.thermal_binary, args.model_dir, output / "thermal-process", + window_ns=WINDOW_NS, artifact_root=args.artifact_root, + baseline_budgets={stack: baseline for stack in stacks}, + action_delay_ns=WINDOW_NS, recovery_dwell_ns=100_000_000, + light_fraction=0.5) as thermal: + manifest["thermal_model_lock"] = thermal.lock + manifest["thermal_header"] = thermal.header + _save(output / "manifest.json", manifest) + result = execute_loop(profile=profile, workload=workload, scenario=scenario, + normalized=normalized, strategy=args.strategy, + fluid=fluid, thermal=thermal, sinks=sinks) + finally: + for stream in sinks.values(): + stream.close() + _save(output / "manifest.json", manifest) + final_receipt = result["records"]["thermal"][-1]["energy_j"]["cumulative"] + mapping_error = final_receipt["total_input_j"] - result["summary"]["energy_j"] + if abs(mapping_error) > 1e-9 * max(1.0, result["summary"]["energy_j"]): + raise AssertionError("served-byte energy differs from thermal receipt") + done = { + "execution_status": "COMPLETED", + "capability_status": "MODELLED_FLUID_RATE_TO_INCREMENTAL_ENERGY_TO_COUPLED_THERMAL_CONTROL", + "scientific_scope": "CONDITIONAL_SIMULATED", + "strategy": args.strategy, + "summary": result["summary"], + "policy_profile": result["policy_profile"], + "semantics": result["semantics"], + "thermal_energy_receipt": final_receipt, + "served_to_thermal_energy_error_j": mapping_error, + "wall_s": time.monotonic() - started, + "child_peak_rss_kib": resource.getrusage(resource.RUSAGE_CHILDREN).ru_maxrss, + } + _save(output / "DONE.json", done) + return 0 + except BaseException as exc: + _save(output / "FAILED.json", {"execution_status": "FAILED", "error": repr(exc), + "wall_s": time.monotonic() - started, + "evidence": "partial JSONL and thermal transcript retained"}) + raise + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/experiments/eq3_rate_thermal/run_rate_thermal.py b/experiments/eq3_rate_thermal/run_rate_thermal.py new file mode 100644 index 0000000..3661521 --- /dev/null +++ b/experiments/eq3_rate_thermal/run_rate_thermal.py @@ -0,0 +1,121 @@ +#!/usr/bin/env python3 +"""Explicit read-rate thermal scenario; never creates MQSim transactions.""" +from __future__ import annotations + +import argparse +import csv +from datetime import datetime, timezone +import hashlib +import json +import os +from pathlib import Path +import platform +import resource +import shutil +import sys +import time + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / 'experiments' / 'eq3_maintenance')) +from thermal_client import ThermalService +from rate_inputs import build_windows + + +def save(path, value): + path.write_text(json.dumps(value, indent=2, allow_nan=False) + '\n') + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + for name in ('profile', 'schedule', 'model-dir', 'thermal-binary', 'artifact-root', 'output'): + parser.add_argument('--' + name, type=Path, required=True) + parser.add_argument('--address-limit-gib', type=int, default=4) + args = parser.parse_args() + out = args.output.resolve() + out.mkdir(parents=True, exist_ok=False) + started = time.monotonic() + try: + for key in ('OMP_NUM_THREADS', 'OPENBLAS_NUM_THREADS', 'MKL_NUM_THREADS', 'NUMEXPR_NUM_THREADS'): + os.environ[key] = '1' + os.environ['CUDA_VISIBLE_DEVICES'] = '' + profile = json.loads(args.profile.read_text()) + schedule = json.loads(args.schedule.read_text()) + normalized = json.loads((args.model_dir / 'normalized.json').read_text()) + inputs = build_windows(profile, schedule, normalized) + save(out / 'profile.json', profile) + save(out / 'schedule.json', schedule) + save(out / 'load-windows.json', inputs) + manifest = dict( + experiment_kind='READ_RATE_DRIVEN_INCREMENTAL_THERMAL', + started_utc=datetime.now(timezone.utc).isoformat(), + python=sys.version, platform=platform.platform(), + source_sha256={name: hashlib.sha256((Path(__file__).parent / name).read_bytes()).hexdigest() + for name in ('run_rate_thermal.py', 'rate_inputs.py')}, + thermal_binary=str(args.thermal_binary.resolve()), + thermal_binary_sha256=hashlib.sha256(args.thermal_binary.read_bytes()).hexdigest(), + model_dir=str(args.model_dir.resolve()), + model_domain_k=[300, 400], step_ns=20_000_000, + parameter_provenance=profile.get('provenance'), + requested_rate_is='SCENARIO_INPUT_NOT_MEASURED_BACKEND_THROUGHPUT', + idle_power='UNKNOWN_NOT_INCLUDED', gpu_external_power='ZERO_INCREMENT_NOT_PHYSICAL_IDLE', + mqsim_used=False, active_thermal_control=False, + physical_qualification='CONDITIONAL_ENGINEERING_USE_NOT_MODEL_FREEZE', + address_limit_gib=args.address_limit_gib, cpu_threads=1, gpu_count=0, + disk_free_bytes=shutil.disk_usage(out).free) + save(out / 'manifest.json', manifest) + if args.address_limit_gib <= 0: + raise ValueError('address limit must be positive') + resource.setrlimit(resource.RLIMIT_AS, (args.address_limit_gib * 1024**3,) * 2) + resource.setrlimit(resource.RLIMIT_CORE, (0, 0)) + total_input = 0.0 + last = None + peaks = {} + maximum_energy_error = 0.0 + with (out / 'thermal.jsonl').open('w') as frames, (out / 'stack-temperatures.csv').open('w') as csvfile: + writer = csv.DictWriter(csvfile, fieldnames=['time_ns', 'stack', 'temperature_k', + 'increment_above_300k', 'guard_observation']) + writer.writeheader() + with ThermalService(args.thermal_binary, args.model_dir, out / 'thermal-process', + artifact_root=args.artifact_root) as thermal: + manifest['model_file_sha256'] = thermal.lock + save(out / 'manifest.json', manifest) + for window in inputs['windows']: + energy = window['component_energy_j'] + total_input += sum(energy.values()) + last = thermal.advance(window['start_ns'], window['end_ns'], energy) + frames.write(json.dumps(last, allow_nan=False) + '\n') + cumulative = last['energy_j']['cumulative'] + error = abs(cumulative['total_input_j'] - total_input) + maximum_energy_error = max(maximum_energy_error, error) + if error > 1e-9 * max(1.0, total_input): + raise ValueError('load energy differs from thermal input receipt') + for stack, temperature in last['temperatures'].items(): + peaks[stack] = max(peaks.get(stack, temperature), temperature) + writer.writerow(dict(time_ns=window['end_ns'], stack=stack, + temperature_k=temperature, increment_above_300k=temperature - 300.0, + guard_observation=last['stack_states'].get(stack, 'UNKNOWN'))) + if last is None: + raise ValueError('no thermal window') + save(out / 'DONE.json', dict( + execution_status='COMPLETED', capability_status='RATE_TO_ENERGY_TO_COUPLED_THERMAL', + scientific_scope='CONDITIONAL_INCREMENTAL_READ_HEATING', + simulated_end_ns=last['end_ns'], input_energy_j=total_input, + thermal_input_error_max_j=maximum_energy_error, + thermal_energy_receipt=last['energy_j'], + peak_k_by_stack=peaks, + peak_increment_k_by_stack={key: value - 300.0 for key, value in peaks.items()}, + final_k_by_stack=last['temperatures'], + wall_s=time.monotonic() - started, + child_peak_rss_kib=resource.getrusage(resource.RUSAGE_CHILDREN).ru_maxrss, + actual_reads_or_writes=False, idle_power_included=False, + thermal_control='OBSERVED_ONLY_NOT_APPLIED', source_hashes=manifest['source_sha256'])) + return 0 + except BaseException as exc: + save(out / 'FAILED.json', dict(execution_status='FAILED', error=repr(exc), + wall_s=time.monotonic() - started, + evidence='thermal-process transcript retains trial failure; no clamp')) + raise + + +if __name__ == '__main__': + raise SystemExit(main()) diff --git a/experiments/eq3_rate_thermal/test_analyze_controlled_campaign.py b/experiments/eq3_rate_thermal/test_analyze_controlled_campaign.py new file mode 100644 index 0000000..8f199a9 --- /dev/null +++ b/experiments/eq3_rate_thermal/test_analyze_controlled_campaign.py @@ -0,0 +1,207 @@ +import json +from pathlib import Path +import tempfile +import unittest + +from analyze_controlled_campaign import ( + EXPECTED_POINT_COUNT, + POINT_FILES, + _discover_point_dirs, + _validate_index_identity, + analyze_campaign, + analyze_point, + write_outputs, +) + + +def save(path, value): + path.write_text(json.dumps(value) + "\n") + + +def jsonl(path, rows): + path.write_text("".join(json.dumps(row) + "\n" for row in rows)) + + +def make_point(root: Path, strategy: str, served=(60, 20)) -> Path: + point = root / strategy + point.mkdir() + stacks = ["hbf0", "hbf1"] + workload = { + "metadata": {"model_id": "Qwen/Qwen2.5-7B-Instruct", "pattern": "continuous", + "stack_ids": stacks, "active_ns": 40, "weight_bytes": 100, + "full_scans_per_s": 16}, + "windows": [{"start_ns": 0, "end_ns": 20, "total_offered_bytes": 100}, + {"start_ns": 20, "end_ns": 40, "total_offered_bytes": 0}], + } + save(point / "workload.json", workload) + save(point / "scenario.json", {"topology": "mixed_direct"}) + save(point / "manifest.json", {"strategy": strategy, + "input_sha256": {"workload.json": "same"}}) + delivered_per_stack = sum(served) // 2 + histories = {stack: [{"delay_ns": 20, "bytes": delivered_per_stack // 2}, + {"delay_ns": 40, "bytes": delivered_per_stack - delivered_per_stack // 2}] + for stack in stacks} + per_stack_wait = {stack: {"delivered_delay_histogram_bytes": rows, + "delivered_delay_p95_ns": 40, + "delivered_delay_p99_ns": 40} + for stack, rows in histories.items()} + energy = sum(served) * 50e-12 + save(point / "DONE.json", { + "execution_status": "COMPLETED", "strategy": strategy, + "summary": {"energy_j": energy, "fluid_wait_by_stack": per_stack_wait, + "final_guard_states": {"hbf0": "normal", "hbf1": "light"}}, + "thermal_energy_receipt": {"total_input_j": energy}, + "semantics": {"backend_latency": "UNKNOWN", + "maintenance": "UNAVAILABLE_IN_THIS_FLUID_PATH"}, + }) + rates = [ + {"start_ns": 0, "end_ns": 20, "window_arrived_bytes": 100, + "window_served_bytes": served[0], + "stacks": {"hbf0": {"offered_bytes": 50, "delivered_bytes": served[0] // 2, + "backlog_bytes": 50 - served[0] // 2}, + "hbf1": {"offered_bytes": 50, "delivered_bytes": served[0] // 2, + "backlog_bytes": 50 - served[0] // 2}}}, + {"start_ns": 20, "end_ns": 40, "window_arrived_bytes": 0, + "window_served_bytes": served[1], + "stacks": {"hbf0": {"offered_bytes": 0, "delivered_bytes": served[1] // 2, + "backlog_bytes": 50 - sum(served) // 2}, + "hbf1": {"offered_bytes": 0, "delivered_bytes": served[1] // 2, + "backlog_bytes": 50 - sum(served) // 2}}}, + ] + controls = [{"observed_window_start_ns": row["start_ns"], + "observed_window_end_ns": row["end_ns"]} for row in rates] + energies = [{"start_ns": row["start_ns"], "end_ns": row["end_ns"], + "component_energy_j": {"array": row["window_served_bytes"] * 40e-12, + "base": row["window_served_bytes"] * 10e-12}, + "window_total_j": row["window_served_bytes"] * 50e-12} + for row in rates] + thermals = [ + {"start_ns": 0, "end_ns": 20, "temperatures": {"hbf0": 350.0, "hbf1": 351.0}, + "stack_states": {"hbf0": "normal", "hbf1": "light"}}, + {"start_ns": 20, "end_ns": 40, "temperatures": {"hbf0": 349.0, "hbf1": 350.0}, + "stack_states": {"hbf0": "normal", "hbf1": "normal"}}, + ] + jsonl(point / "rates.jsonl", rates) + jsonl(point / "control.jsonl", controls) + jsonl(point / "energy.jsonl", energies) + jsonl(point / "thermal.jsonl", thermals) + return point + + +class ControlledAnalysisTests(unittest.TestCase): + def test_recomputes_metrics_conservation_states_and_weighted_latency(self): + with tempfile.TemporaryDirectory() as directory: + point = make_point(Path(directory), "guard_only") + result = analyze_point(point) + self.assertEqual(result["totals"]["offered_bytes"], 100) + self.assertEqual(result["totals"]["delivered_bytes"], 80) + self.assertEqual(result["totals"]["final_backlog_bytes"], 20) + self.assertEqual(result["totals"]["byte_conservation_error"], 0) + self.assertEqual(result["totals"]["byte_weighted_delay_p95_ns"], 40) + self.assertEqual(result["totals"]["peak_temperature_k"], 351.0) + self.assertEqual(result["any_stack_state_time_ns"]["light"], 20) + self.assertEqual(result["per_stack"]["hbf0"]["state_time_ns"]["normal"], 40) + self.assertAlmostEqual(result["totals"]["energy_j"], 80 * 50e-12) + self.assertEqual(result["limitations"]["token_per_s"], "UNKNOWN") + trace = result["_trace"] + self.assertEqual(trace["offered_Bps"], [5e9, 0.0]) + self.assertEqual(trace["temperature_k_by_stack"]["hbf0"], [350.0, 349.0]) + self.assertEqual(len(trace["temperature_k_by_stack"]["hbf1"]), len(trace["end_ns"])) + active = result["service_rate_stability"]["active_full"]["total"] + self.assertEqual(active["window_count"], 2) + self.assertAlmostEqual(active["population_cv"], 0.5) + self.assertEqual(active["p5_Bps"], 1e9) + self.assertEqual(active["p50_Bps"], 1e9) + self.assertEqual(active["p95_Bps"], 3e9) + self.assertEqual(active["zero_service_window_fraction"], 0.0) + latter = result["service_rate_stability"]["active_second_half"]["total"] + self.assertEqual(latter["window_count"], 1) + self.assertEqual(latter["population_cv"], 0.0) + + def test_partial_campaign_pairs_all_policies_without_benefit_label(self): + with tempfile.TemporaryDirectory() as directory: + root = Path(directory) + make_point(root, "guard_only", (60, 20)) + make_point(root, "thermal_hysteresis_guard", (50, 10)) + make_point(root, "read_rate_feedback_thermal_guard_v1", (40, 10)) + result = analyze_campaign(root, require_complete=False) + self.assertEqual(result["completed_point_count"], 3) + self.assertEqual(len(result["pairwise_policy_costs"]), 3) + row = result["pairwise_policy_costs"][0] + self.assertEqual(row["delta_semantics"], "RIGHT_MINUS_LEFT_NO_BENEFIT_DIRECTION_ASSUMED") + self.assertLess(row["delivered_bytes_delta"], 0) + output = root / "derived" + write_outputs(result, output, plots=True) + self.assertTrue((output / "CONTROLLED_CAMPAIGN_ANALYSIS.json").is_file()) + self.assertTrue((output / "pairwise-policy-costs.csv").is_file()) + self.assertTrue((output / "controlled-campaign-summary.png").is_file()) + self.assertEqual(len(list(output.glob("trajectory-*.png"))), 1) + self.assertEqual(len(list(output.glob("stack-temperatures-*.png"))), 1) + + def test_detects_byte_conservation_failure(self): + with tempfile.TemporaryDirectory() as directory: + point = make_point(Path(directory), "guard_only") + rows = list(json.loads(line) for line in (point / "rates.jsonl").read_text().splitlines()) + rows[-1]["stacks"]["hbf0"]["backlog_bytes"] += 1 + jsonl(point / "rates.jsonl", rows) + with self.assertRaisesRegex(ValueError, "byte conservation"): + analyze_point(point) + + def test_run_index_whitelists_main_and_excludes_pilot_and_stage_done(self): + with tempfile.TemporaryDirectory() as directory: + root = Path(directory) + save(root / "DONE.json", {"stage": "complete"}) + pilot = root / "pilot" + pilot.mkdir() + for name in POINT_FILES: + (pilot / name).write_text("{}\n") + entries = [{"phase": "pilot", "output": str(pilot)}] + expected = [] + for index in range(EXPECTED_POINT_COUNT): + point = root / f"main-{index:02d}" + point.mkdir() + for name in POINT_FILES: + (point / name).write_text("{}\n") + expected.append(point.resolve()) + entries.append({"phase": "main", "output": str(point)}) + save(root / "RUN_INDEX.json", { + "main_count": EXPECTED_POINT_COUNT, + "pilot_count": 1, + "points": entries, + }) + found, main_entries, mode = _discover_point_dirs(root) + self.assertEqual(found, expected) + self.assertEqual(len(main_entries), EXPECTED_POINT_COUNT) + self.assertEqual(mode, "RUN_INDEX_PHASE_MAIN_WHITELIST") + self.assertNotIn(pilot.resolve(), found) + (expected[-1] / "DONE.json").unlink() + with self.assertRaisesRegex(ValueError, "incomplete"): + _discover_point_dirs(root) + + def test_run_index_identity_checks_point_inputs(self): + point_path = Path("/tmp/fixed-main-point") + point = { + "point_id": "RT-MAIN-fixed", "point_path": str(point_path), + "topology": "mixed_direct", "model_id": "Qwen/Qwen2.5-7B-Instruct", + "pattern": "continuous", "strategy": "guard_only", "full_scans_per_s": 16, + "active_ns": 20_000_000_000, "duration_ns": 30_000_000_000, + "totals": {"offered_bytes": 123}, + "input_identity": {"profile.json": "profile-hash", "workload.json": "workload-hash", + "scenario.json": "scenario-hash"}, + } + entry = { + "phase": "main", "point_id": "RT-MAIN-fixed", "output": str(point_path), + "topology": "mixed_direct", "model": "7B", "pattern": "continuous", + "strategy": "guard_only", "full_scans_per_s": 16, + "active_s": 20, "recovery_s": 10, "expected_active_offered_bytes": 123, + "input_sha256": {"profile": "profile-hash", "workload": "workload-hash", + "scenario": "scenario-hash"}, + } + _validate_index_identity(entry, point) + entry["input_sha256"]["workload"] = "wrong" + with self.assertRaisesRegex(ValueError, "workload input identity"): + _validate_index_identity(entry, point) + + +if __name__ == "__main__": + unittest.main() diff --git a/experiments/eq3_rate_thermal/test_fluid_service.py b/experiments/eq3_rate_thermal/test_fluid_service.py new file mode 100644 index 0000000..dcf14e6 --- /dev/null +++ b/experiments/eq3_rate_thermal/test_fluid_service.py @@ -0,0 +1,136 @@ +#!/usr/bin/env python3 +"""Fixed, solver-free checks for the rate-thermal fluid byte service.""" + +import unittest + +from fluid_service import ( + BACKEND_LATENCY, + LATENCY_SEMANTICS, + FluidService, +) + + +WINDOW_NS = 20_000_000 + + +class FluidServiceTests(unittest.TestCase): + def make_service(self, channels=("c0",), capacity_bps=100): + return FluidService( + {"hbf0": list(channels)}, + {"hbf0": {channel: capacity_bps for channel in channels}}, + ) + + def test_fifo_conservation_and_cooling_drain(self): + service = self.make_service(capacity_bps=100) # 2 bytes/window + first = service.advance(0, WINDOW_NS, {"hbf0": {"c0": 5}}, {"hbf0": 99}) + self.assertEqual(first["served_by_channel"]["hbf0"]["c0"], 2) + self.assertEqual(first["stacks"]["hbf0"]["backlog_bytes"], 3) + self.assertEqual(first["stacks"]["hbf0"]["latency_p95_ns"], WINDOW_NS) + + second = service.advance(WINDOW_NS, 2 * WINDOW_NS, {}, {"hbf0": 99}) + self.assertEqual(second["served_by_channel"]["hbf0"]["c0"], 2) + self.assertEqual(second["stacks"]["hbf0"]["backlog_bytes"], 1) + self.assertEqual(second["stacks"]["hbf0"]["latency_p95_ns"], 2 * WINDOW_NS) + + third = service.advance(2 * WINDOW_NS, 3 * WINDOW_NS, {}, {"hbf0": 99}) + self.assertEqual(third["served_by_channel"]["hbf0"]["c0"], 1) + self.assertEqual(third["stacks"]["hbf0"]["backlog_bytes"], 0) + self.assertIsNone(third["stacks"]["hbf0"]["oldest_wait_ns"]) + conservation = third["conservation"]["per_stack"]["hbf0"] + self.assertEqual(conservation["cumulative_offered_bytes"], 5) + self.assertEqual(conservation["cumulative_delivered_bytes"], 5) + self.assertTrue(conservation["cumulative_conserved"]) + + def test_stack_budget_is_max_min_fair_across_channels(self): + service = self.make_service(("c0", "c1"), capacity_bps=1_000) + result = service.advance( + 0, + WINDOW_NS, + {"hbf0": {"c0": 20, "c1": 20}}, + {"hbf0": 11}, + ) + served = result["served_by_channel"]["hbf0"] + self.assertEqual(sum(served.values()), 11) + self.assertLessEqual(abs(served["c0"] - served["c1"]), 1) + self.assertGreater(served["c0"], 0) + self.assertGreater(served["c1"], 0) + self.assertEqual(result["semantics"]["stack_budget_allocation"], "INTEGER_MAX_MIN_FAIR") + + def test_byte_weighted_p95_uses_served_cohorts(self): + service = self.make_service(capacity_bps=10_000) + service.advance(0, WINDOW_NS, {"hbf0": {"c0": 6}}, {"hbf0": 0}) + result = service.advance( + WINDOW_NS, + 2 * WINDOW_NS, + {"hbf0": {"c0": 94}}, + {"hbf0": 100}, + ) + # Bytes 1..94 have a 20 ms delay and bytes 95..100 have 40 ms. + self.assertEqual(result["stacks"]["hbf0"]["latency_p95_ns"], 2 * WINDOW_NS) + self.assertEqual( + result["stacks"]["hbf0"]["delivered_delay_histogram_bytes"], + [ + {"delay_ns": WINDOW_NS, "bytes": 94}, + {"delay_ns": 2 * WINDOW_NS, "bytes": 6}, + ], + ) + self.assertEqual(result["stacks"]["hbf0"]["latency_semantics"], LATENCY_SEMANTICS) + self.assertEqual(result["stacks"]["hbf0"]["backend_latency_ns"], BACKEND_LATENCY) + channel_conservation = result["conservation"]["per_stack"]["hbf0"]["per_channel"]["c0"] + self.assertTrue(channel_conservation["window_conserved"]) + self.assertTrue(channel_conservation["cumulative_conserved"]) + + def test_integer_capacity_floor_and_future_budget(self): + service = self.make_service(capacity_bps=149) # floor(2.98) = 2 bytes/window + first = service.advance(0, WINDOW_NS, {"hbf0": {"c0": 8}}, {"hbf0": 1}) + self.assertEqual(first["channels"]["hbf0"]["c0"]["capacity_bytes"], 2) + self.assertEqual(first["stacks"]["hbf0"]["delivered_bytes"], 1) + second = service.advance(WINDOW_NS, 2 * WINDOW_NS, {}, {"hbf0": 2}) + self.assertEqual(second["stacks"]["hbf0"]["delivered_bytes"], 2) + self.assertEqual(second["stacks"]["hbf0"]["backlog_bytes"], 5) + + def test_stacks_are_independent_and_snapshot_is_read_only(self): + stacks = {f"hbf{i}": ["c0", "c1"] for i in range(8)} + capacities = { + stack: {"c0": 1_000, "c1": 1_000} for stack in stacks + } + service = FluidService(stacks, capacities) + offered = { + stack: {"c0": index + 1, "c1": index + 2} + for index, stack in enumerate(stacks) + } + budgets = {stack: index for index, stack in enumerate(stacks)} + result = service.advance(0, WINDOW_NS, offered, budgets) + for index, stack in enumerate(stacks): + self.assertEqual(result["stacks"][stack]["delivered_bytes"], index) + self.assertEqual( + result["stacks"][stack]["backlog_bytes"], + (index + 1) + (index + 2) - index, + ) + snapshot = service.snapshot() + snapshot["stacks"]["hbf0"]["backlog_bytes"] = -1 + self.assertGreaterEqual(service.snapshot()["stacks"]["hbf0"]["backlog_bytes"], 0) + + def test_invalid_inputs_fail_without_advancing_time(self): + service = self.make_service() + invalid_calls = [ + lambda: service.advance(1, WINDOW_NS + 1, {}, {"hbf0": 0}), + lambda: service.advance(0, WINDOW_NS - 1, {}, {"hbf0": 0}), + lambda: service.advance(0, WINDOW_NS, {"hbf9": {}}, {"hbf0": 0}), + lambda: service.advance(0, WINDOW_NS, {"hbf0": {"bad": 1}}, {"hbf0": 0}), + lambda: service.advance(0, WINDOW_NS, {"hbf0": {"c0": -1}}, {"hbf0": 0}), + lambda: service.advance(0, WINDOW_NS, {"hbf0": {"c0": True}}, {"hbf0": 0}), + lambda: service.advance(0, WINDOW_NS, {}, {}), + ] + for call in invalid_calls: + with self.assertRaises(ValueError): + call() + self.assertEqual(service.now_ns, 0) + + def test_constructor_rejects_incomplete_capacity_map(self): + with self.assertRaises(ValueError): + FluidService({"hbf0": ["c0", "c1"]}, {"hbf0": {"c0": 96_000_000_000}}) + + +if __name__ == "__main__": + unittest.main() diff --git a/experiments/eq3_rate_thermal/test_model_workloads.py b/experiments/eq3_rate_thermal/test_model_workloads.py new file mode 100644 index 0000000..5927e5d --- /dev/null +++ b/experiments/eq3_rate_thermal/test_model_workloads.py @@ -0,0 +1,104 @@ +import json +from pathlib import Path +import tempfile +import unittest + +from model_workloads import build_workload, main + + +def config(**updates): + value = { + "schema_version": "eq3-rate-model-workload-config-v1", + "model_id": "Qwen/Qwen2.5-7B-Instruct", + "full_scans_per_s": 16, + "pattern": "continuous", + "stack_count": 4, + "channels_per_stack": 16, + "step_ns": 20_000_000, + "active_ns": 1_000_000_000, + "recovery_ns": 40_000_000, + } + value.update(updates) + return value + + +class ModelWorkloadTests(unittest.TestCase): + def test_continuous_exact_bytes_uniform_stripe_and_recovery(self): + result = build_workload(config()) + expected = 15_231_233_024 * 16 + self.assertEqual(result["metadata"]["actual_total_offered_bytes"], expected) + self.assertEqual(result["metadata"]["equivalent_full_scans"], 16) + self.assertEqual(len(result["windows"]), 52) + self.assertTrue(all(window["total_offered_bytes"] == 0 for window in result["windows"][-2:])) + cumulative = {f"hbf{s}": {str(c): 0 for c in range(16)} for s in range(4)} + for window in result["windows"]: + self.assertEqual(window["end_ns"] - window["start_ns"], 20_000_000) + for stack, channels in window["stack_channel_offered_bytes"].items(): + self.assertEqual(set(channels), {str(c) for c in range(16)}) + for channel, byte_count in channels.items(): + self.assertIsInstance(byte_count, int) + self.assertGreaterEqual(byte_count, 0) + cumulative[stack][channel] += byte_count + totals = [value for channels in cumulative.values() for value in channels.values()] + self.assertLessEqual(max(totals) - min(totals), 1) + self.assertEqual(sum(totals), expected) + + def test_burst_has_same_mean_and_twice_then_zero_pattern(self): + continuous = build_workload(config(recovery_ns=0)) + burst = build_workload(config(pattern="burst_equal_mean", recovery_ns=0, + burst_period_ns=200_000_000, + burst_on_ns=100_000_000)) + self.assertEqual(continuous["metadata"]["actual_total_offered_bytes"], + burst["metadata"]["actual_total_offered_bytes"]) + multipliers = [window["demand_multiplier"] for window in burst["windows"][:10]] + self.assertEqual(multipliers, [2] * 5 + [0] * 5) + + def test_all_models_stack_counts_and_high_pressure_are_canonical(self): + expected = { + "Qwen/Qwen2.5-7B-Instruct": 15_231_233_024, + "Qwen/Qwen2.5-72B-Instruct": 145_412_407_296, + "Qwen/Qwen3-235B-A22B": 470_187_269_120, + } + for model_id, payload in expected.items(): + for stack_count in (4, 8): + with self.subTest(model=model_id, stacks=stack_count): + result = build_workload(config(model_id=model_id, stack_count=stack_count, + full_scans_per_s=32, recovery_ns=0)) + self.assertEqual(result["metadata"]["weight_bytes"], payload) + self.assertEqual(result["metadata"]["actual_total_offered_bytes"], payload * 32) + self.assertEqual(len(result["windows"][0]["stack_channel_offered_bytes"]), + stack_count) + qwen3 = build_workload(config(model_id="Qwen/Qwen3-235B-A22B", recovery_ns=0)) + self.assertEqual(qwen3["metadata"]["semantics"]["token_per_s"], "UNKNOWN") + self.assertIn("DO_NOT_INTERPRET", qwen3["metadata"]["moe_limit"]) + self.assertEqual(qwen3["metadata"]["semantics"]["capacity"], + "NOT_APPLIED_HERE_EXCESS_MUST_ENTER_FLUID_BACKLOG") + + def test_rejects_unfrozen_or_malformed_inputs(self): + cases = [ + {"full_scans_per_s": 17}, + {"stack_count": 5}, + {"channels_per_stack": 8}, + {"step_ns": 10_000_000}, + {"pattern": "burst_equal_mean", "burst_period_ns": 180_000_000, + "burst_on_ns": 100_000_000}, + {"unknown": 1}, + ] + for update in cases: + with self.subTest(update=update), self.assertRaises(ValueError): + build_workload(config(**update)) + + def test_cli_writes_canonical_json(self): + with tempfile.TemporaryDirectory() as directory: + directory = Path(directory) + source = directory / "config.json" + output = directory / "workload.json" + source.write_text(json.dumps(config(recovery_ns=0))) + self.assertEqual(main(["--config", str(source), "--output", str(output)]), 0) + result = json.loads(output.read_text()) + self.assertEqual(result["schema_version"], "eq3-rate-model-workload-v1") + self.assertEqual(result["canonical_config"], config(recovery_ns=0)) + + +if __name__ == "__main__": + unittest.main() diff --git a/experiments/eq3_rate_thermal/test_rate_inputs.py b/experiments/eq3_rate_thermal/test_rate_inputs.py new file mode 100644 index 0000000..96c7496 --- /dev/null +++ b/experiments/eq3_rate_thermal/test_rate_inputs.py @@ -0,0 +1,183 @@ +import copy +import math +import unittest + +from rate_inputs import build_windows + + +def profile(**updates): + value = { + "schema_version": 1, + "reference_read_Bps": 1.6e12, + "array_j_per_byte": 40e-12, + "base_j_per_byte": 10e-12, + "provenance": "SCENARIO_ASSUMPTION_USER_CONFIRMED", + } + value.update(updates) + return value + + +def normalized(): + def component(component_id, stack, role, die_index): + return { + "id": component_id, + "device_id": stack, + "physical_type": "HBF", + "powered": True, + "role": role, + "die_index": die_index, + } + + return { + "components": [ + {"id": "gpu", "device_id": "gpu", "physical_type": "GPU", "powered": True}, + component("array-A", "stack-A", "array_die", 1), + component("base-A", "stack-A", "base_die", -1), + component("array-B0", "stack-B", "array_die", 0), + component("array-A0", "stack-A", "array_die", 0), + component("base-B", "stack-B", "base_die", -1), + component("array-B2", "stack-B", "array_die", 2), + component("array-B1", "stack-B", "array_die", 1), + ] + } + + +class RateInputsTests(unittest.TestCase): + def test_ocp_grade2_aggregate_limit_applies_without_explicit_channel_rates(self): + settings = profile( + channel_map={"stack-A": {str(i): "array-A" for i in range(16)}}, + channel_capacity_Bps={"stack-A": {str(i): 96e9 for i in range(16)}}) + schedule = {"schema_version": 1, "end_ns": 20_000_000, + "segments": [{"start_ns": 0, "end_ns": 20_000_000, + "read_Bps": {"stack-A": 1.536e12}}]} + result = build_windows(settings, schedule, normalized()) + self.assertAlmostEqual(result["windows"][0]["stacks"]["stack-A"]["source_power_w"]["total"], 76.8) + schedule["segments"][0]["read_Bps"]["stack-A"] = 1.6e12 + with self.assertRaisesRegex(ValueError, "aggregate channel capacity"): + build_windows(settings, schedule, normalized()) + + def test_discovers_geometry_and_full_rate_maps_to_80w(self): + schedule = { + "schema_version": 1, + "end_ns": 20_000_000, + "segments": [{"start_ns": 0, "end_ns": 20_000_000, + "read_Bps": {"stack-A": 1.6e12}}], + } + original = copy.deepcopy(schedule) + result = build_windows(profile(), schedule, normalized()) + self.assertEqual(schedule, original) + window = result["windows"][0] + stack_a = window["stacks"]["stack-A"] + self.assertAlmostEqual(stack_a["requested_bytes"], 32e9) + self.assertEqual(stack_a["requested_bytes"], stack_a["modelled_bytes"]) + self.assertAlmostEqual(stack_a["source_energy_j"]["array"], 1.28) + self.assertAlmostEqual(stack_a["source_energy_j"]["base"], 0.32) + self.assertAlmostEqual(stack_a["source_power_w"]["total"], 80.0) + self.assertAlmostEqual(window["component_energy_j"]["array-A"], 0.64) + self.assertAlmostEqual(window["component_energy_j"]["array-A0"], 0.64) + self.assertAlmostEqual(window["component_energy_j"]["base-A"], 0.32) + self.assertEqual(window["stacks"]["stack-B"]["requested_bytes"], 0.0) + self.assertEqual(result["metadata"]["idle_semantics"], "ZERO_READ_ONLY_NOT_ZERO_IDLE") + self.assertEqual(len(result["metadata"]["entity_mapping"]["stack-B"]["array_dies"]), 3) + + def test_integrates_piecewise_segments_across_window_boundaries(self): + schedule = { + "schema_version": 1, + "end_ns": 40_000_000, + "segments": [ + {"start_ns": 0, "end_ns": 10_000_000, "read_Bps": {}}, + {"start_ns": 10_000_000, "end_ns": 30_000_000, + "read_Bps": {"stack-A": 0.8e12}}, + {"start_ns": 30_000_000, "end_ns": 40_000_000, + "read_Bps": {"stack-A": 1.6e12}}, + ], + } + result = build_windows(profile(), schedule, normalized()) + self.assertEqual(len(result["windows"]), 2) + self.assertAlmostEqual(result["windows"][0]["stacks"]["stack-A"]["requested_bytes"], 8e9) + self.assertAlmostEqual(result["windows"][1]["stacks"]["stack-A"]["requested_bytes"], 24e9) + self.assertAlmostEqual(result["windows"][1]["stacks"]["stack-A"]["source_energy_j"]["total"], 1.2) + + def test_explicit_weights_are_applied_to_discovered_dies(self): + p = profile(die_weights={"stack-A": {"array-A": 0.25, "array-A0": 0.75}}) + schedule = {"schema_version": 1, "end_ns": 20_000_000, + "segments": [{"start_ns": 0, "end_ns": 20_000_000, + "read_Bps": {"stack-A": 1.6e12}}]} + result = build_windows(p, schedule, normalized()) + energy = result["windows"][0]["component_energy_j"] + self.assertAlmostEqual(energy["array-A"], 0.32) + self.assertAlmostEqual(energy["array-A0"], 0.96) + self.assertEqual(result["metadata"]["entity_mapping"]["stack-A"]["weight_mode"], "EXPLICIT") + self.assertEqual(result["metadata"]["entity_mapping"]["stack-B"]["weight_mode"], "UNIFORM") + + def test_explicit_channels_change_die_locality_without_changing_total_energy(self): + p = profile( + channel_map={"stack-A": {"c0": "array-A0", "c1": "array-A0", + "c2": "array-A", "c3": "array-A"}}, + channel_capacity_Bps={"stack-A": {"c0": 0.4e12, "c1": 0.4e12, + "c2": 0.4e12, "c3": 0.4e12}}, + ) + concentrated = {"schema_version": 1, "end_ns": 20_000_000, + "segments": [{"start_ns": 0, "end_ns": 20_000_000, + "read_Bps": {"stack-A": 0.4e12}, + "channel_read_Bps": {"stack-A": {"c0": 0.4e12}}}]} + distributed = copy.deepcopy(concentrated) + distributed["segments"][0]["channel_read_Bps"]["stack-A"] = { + "c0": 0.1e12, "c1": 0.1e12, "c2": 0.1e12, "c3": 0.1e12, + } + one = build_windows(p, concentrated, normalized())["windows"][0] + four = build_windows(p, distributed, normalized())["windows"][0] + self.assertEqual(one["stacks"]["stack-A"]["active_channel_count"], 1) + self.assertEqual(four["stacks"]["stack-A"]["active_channel_count"], 4) + self.assertEqual(one["stacks"]["stack-A"]["source_energy_j"], + four["stacks"]["stack-A"]["source_energy_j"]) + self.assertAlmostEqual(one["component_energy_j"]["array-A0"], 0.32) + self.assertAlmostEqual(one["component_energy_j"]["array-A"], 0.0) + self.assertAlmostEqual(four["component_energy_j"]["array-A0"], 0.16) + self.assertAlmostEqual(four["component_energy_j"]["array-A"], 0.16) + + too_fast = copy.deepcopy(concentrated) + too_fast["segments"][0]["read_Bps"]["stack-A"] = 0.5e12 + too_fast["segments"][0]["channel_read_Bps"]["stack-A"]["c0"] = 0.5e12 + with self.assertRaisesRegex(ValueError, "channel capacity"): + build_windows(p, too_fast, normalized()) + inconsistent = copy.deepcopy(concentrated) + inconsistent["segments"][0]["read_Bps"]["stack-A"] = 0.3e12 + with self.assertRaisesRegex(ValueError, "disagree"): + build_windows(p, inconsistent, normalized()) + bad_die = copy.deepcopy(p) + bad_die["channel_map"]["stack-A"]["c0"] = "not-a-die" + with self.assertRaisesRegex(ValueError, "unknown array die"): + build_windows(bad_die, concentrated, normalized()) + + def test_rejects_invalid_rates_timeline_profile_weights_and_entities(self): + base_schedule = {"schema_version": 1, "end_ns": 20_000_000, + "segments": [{"start_ns": 0, "end_ns": 20_000_000, + "read_Bps": {"stack-A": 1.0}}]} + bad_rates = [-1.0, math.nan, math.inf, 1.6e12 + 1.0] + for rate in bad_rates: + with self.subTest(rate=rate), self.assertRaises(ValueError): + schedule = copy.deepcopy(base_schedule) + schedule["segments"][0]["read_Bps"]["stack-A"] = rate + build_windows(profile(), schedule, normalized()) + with self.assertRaisesRegex(ValueError, "unknown HBF"): + schedule = copy.deepcopy(base_schedule) + schedule["segments"][0]["read_Bps"] = {"unknown": 1.0} + build_windows(profile(), schedule, normalized()) + with self.assertRaisesRegex(ValueError, "continuously cover"): + schedule = copy.deepcopy(base_schedule) + schedule["segments"][0]["start_ns"] = 1 + build_windows(profile(), schedule, normalized()) + with self.assertRaisesRegex(ValueError, "approved value"): + build_windows(profile(array_j_per_byte=41e-12), base_schedule, normalized()) + with self.assertRaisesRegex(ValueError, "sum to 1"): + p = profile(die_weights={"stack-A": {"array-A": 0.2, "array-A0": 0.7}}) + build_windows(p, base_schedule, normalized()) + with self.assertRaisesRegex(ValueError, "no powered base_die"): + entities = normalized() + entities["components"] = [x for x in entities["components"] if x.get("id") != "base-A"] + build_windows(profile(), base_schedule, entities) + + +if __name__ == "__main__": + unittest.main() diff --git a/experiments/eq3_rate_thermal/test_run_controlled.py b/experiments/eq3_rate_thermal/test_run_controlled.py new file mode 100644 index 0000000..4741322 --- /dev/null +++ b/experiments/eq3_rate_thermal/test_run_controlled.py @@ -0,0 +1,123 @@ +#!/usr/bin/env python3 +"""Solver-free causal/conservation tests for the controlled fluid runner.""" +import unittest + +from fluid_service import FluidService +from run_controlled import BASELINE_STACK_BPS, WINDOW_NS, execute_loop + + +class FakeThermal: + def __init__(self, states): + self.states = iter(states) + self.energy = 0.0 + + def advance(self, start_ns, end_ns, component_energy_j): + state = next(self.states) + self.energy += sum(component_energy_j.values()) + temperature = {"normal": 320.0, "light": 355.0, + "severe": 365.0, "shutdown": 380.0}[state] + return { + "start_ns": start_ns, "end_ns": end_ns, + "temperatures": {"hbf0": temperature}, + "stack_states": {"hbf0": state}, + "hysteresis_budget_bytes": {"hbf0": BASELINE_STACK_BPS * WINDOW_NS // 2_000_000_000}, + "energy_j": {"cumulative": {"total_input_j": self.energy}}, + } + + +def fixtures(): + channels = {str(index): f"hbf0.die{index}" for index in range(16)} + capacities = {channel: 96_000_000_000 for channel in channels} + offered = {"hbf0": {channel: 5_000_000_000 for channel in channels}} + workload = { + "schema_version": "eq3-rate-model-workload-v1", + "metadata": {"mean_active_offered_Bps": 4_000_000_000_000}, + "windows": [ + {"start_ns": index * WINDOW_NS, "end_ns": (index + 1) * WINDOW_NS, + "stack_channel_offered_bytes": offered, + "total_offered_bytes": sum(sum(row.values()) for row in offered.values())} + for index in range(3) + ], + } + profile = { + "schema_version": 1, "array_j_per_byte": 40e-12, "base_j_per_byte": 10e-12, + "reference_read_Bps": 1_600_000_000_000, + "provenance": "SCENARIO_ASSUMPTION_USER_CONFIRMED", + "channel_map": {"hbf0": channels}, + "channel_capacity_Bps": {"hbf0": capacities}, + } + normalized = {"components": ([{"id": "hbf0.base"}] + + [{"id": value} for value in channels.values()])} + scenario = { + "schema_version": "eq3-rate-controlled-scenario-v1", + "scenario_id": "FIXED_CAUSAL_TEST", "topology": "mixed_direct", + "window_ns": WINDOW_NS, + "target_read_Bps_per_stack": BASELINE_STACK_BPS * 4 // 5, + } + return profile, workload, scenario, normalized, channels, capacities + + +class ControlledRunnerTests(unittest.TestCase): + def test_guard_action_applies_only_to_next_window_and_bytes_conserve(self): + profile, workload, scenario, normalized, channels, capacities = fixtures() + fluid = FluidService({"hbf0": list(channels)}, {"hbf0": capacities}, WINDOW_NS) + result = execute_loop( + profile=profile, workload=workload, scenario=scenario, normalized=normalized, + strategy="guard_only", fluid=fluid, + thermal=FakeThermal(["severe", "normal", "normal"])) + baseline = BASELINE_STACK_BPS * WINDOW_NS // 1_000_000_000 + rates = result["records"]["rates"] + controls = result["records"]["control"] + self.assertEqual(rates[0]["budget_by_stack"]["hbf0"], baseline) + self.assertEqual(controls[0]["next_budget_bytes"]["hbf0"], 0) + self.assertEqual(rates[1]["budget_by_stack"]["hbf0"], 0) + self.assertEqual(controls[1]["next_budget_bytes"]["hbf0"], baseline) + self.assertEqual(rates[2]["budget_by_stack"]["hbf0"], baseline) + self.assertEqual(result["summary"]["byte_conservation_error"], 0) + self.assertEqual(result["semantics"]["backend_latency"], "UNKNOWN") + self.assertFalse(result["semantics"]["mqsim_or_fabric_completion"]) + self.assertEqual(controls[0]["decision"]["fact_semantics"], + "MODELLED_FLUID_BYTE_DELIVERY_NOT_ACTUAL_FABRIC_DELIVERY") + + def test_all_strategies_use_the_same_severe_guard(self): + for strategy in ("guard_only", "thermal_hysteresis_guard", + "read_rate_feedback_thermal_guard_v1"): + profile, workload, scenario, normalized, channels, capacities = fixtures() + workload["windows"] = workload["windows"][:1] + fluid = FluidService({"hbf0": list(channels)}, {"hbf0": capacities}, WINDOW_NS) + result = execute_loop(profile=profile, workload=workload, scenario=scenario, + normalized=normalized, strategy=strategy, fluid=fluid, + thermal=FakeThermal(["severe"])) + self.assertEqual(result["records"]["control"][0]["next_budget_bytes"]["hbf0"], 0) + + def test_feedback_recovers_using_explicit_modelled_fluid_spare_capacity(self): + profile, workload, scenario, normalized, channels, capacities = fixtures() + fluid = FluidService({"hbf0": list(channels)}, {"hbf0": capacities}, WINDOW_NS) + result = execute_loop( + profile=profile, workload=workload, scenario=scenario, normalized=normalized, + strategy="read_rate_feedback_thermal_guard_v1", fluid=fluid, + thermal=FakeThermal(["severe", "normal", "normal"])) + controls = result["records"]["control"] + minimum = BASELINE_STACK_BPS * WINDOW_NS // 1_000_000_000 // 10 + step = BASELINE_STACK_BPS * WINDOW_NS // 1_000_000_000 // 20 + self.assertEqual(controls[0]["next_budget_bytes"]["hbf0"], 0) + self.assertEqual(controls[1]["next_budget_bytes"]["hbf0"], minimum + step) + self.assertEqual(controls[2]["next_budget_bytes"]["hbf0"], minimum + 2 * step) + adapter = result["records"]["rates"][2]["policy_fact_adapter_by_stack"]["hbf0"] + self.assertLess(adapter["modelled_fluid_capacity_utilization"], 1.0) + self.assertFalse(adapter["modelled_fluid_capacity_saturated"]) + self.assertIsNotNone(adapter["fluid_byte_weighted_latency_p95_ns"]) + self.assertIn("MODELLED_FLUID", controls[1]["policy_fact_adapter_semantics"]["backend_busy_fraction"]) + + def test_rejects_unfrozen_target_formula(self): + profile, workload, scenario, normalized, channels, capacities = fixtures() + scenario["target_read_Bps_per_stack"] -= 1 + fluid = FluidService({"hbf0": list(channels)}, {"hbf0": capacities}, WINDOW_NS) + with self.assertRaisesRegex(ValueError, "scenario target"): + execute_loop(profile=profile, workload=workload, scenario=scenario, + normalized=normalized, strategy="guard_only", fluid=fluid, + thermal=FakeThermal(["normal"] * 3)) + + +if __name__ == "__main__": + unittest.main() diff --git a/experiments/eq3_system_thermal/MAINTENANCE_MAIN_CANDIDATE.md b/experiments/eq3_system_thermal/MAINTENANCE_MAIN_CANDIDATE.md new file mode 100644 index 0000000..07c2cb4 --- /dev/null +++ b/experiments/eq3_system_thermal/MAINTENANCE_MAIN_CANDIDATE.md @@ -0,0 +1,62 @@ +# Maintenance-main candidate contract + +`prepare_maintenance_main.py` is a default-off input generator. It neither +freezes nor launches a run. Its destination must be a new directory named +`maintenance-main-v1`, so pilot receipts and earlier raw data cannot be +overwritten. + +## Pilot gate + +Generation requires exactly one pilot for each of `mixed_direct`, `relay`, +`dash`, and `all_hbf_direct`. Each pilot must have: + +- a `DONE.json` with `COMPLETED` status; +- a completed `analyze_extensions.py` result with matching point identity; +- passing timeline, byte-conservation, and energy-to-thermal checks; +- exact completion identities rather than only grouped hashes. +- at least one terminal maintenance operation, proving that the pilot exercised + maintenance rather than only foreground traffic. + +It also requires a separate root review receipt with status +`APPROVED_FOR_MAIN_INPUT_GENERATION`. That receipt binds the pilot-index hash, +every analysis hash, and a positive measured resource budget. This is an +engineering review gate under the already authorized four-topology plan; it is +not represented as a new user signature. + +The measured budget records point/stage time and output limits, address-space +limit, CPU thread count, and host RAM/disk reserves. GPU count remains zero. + +## Candidate matrix + +The generator creates 19 inputs, all at 1.536 TB/s offered per stack for 20 s, +followed by 10 s recovery: + +- 12 shared-resource points: four topologies by the three existing policies; +- three feedback-policy ideal-independent resource points for mixed, relay, + and DASH; +- four mixed-topology shared-resource Ea endpoints: 1.01 and 1.08 eV by + guard-only and feedback. + +Ea 1.04 eV is already present in the 12-point shared matrix and is not +duplicated. Ideal-independent changes maintenance resource contention only; +it retains the actual controller states, admission guards, age, workload, and +energy assumptions. + +The generated inputs copy the reviewed pilot's initial wall/equivalent age, +4 GiB aged subset per HBF stack, channel-owned spare geometry, program/erase +proxies, foreground energy profile, and topology service parameters. The +generator rejects pilot templates if these invariants or the offered bandwidth +differ. + +## Output and next gate + +The output contains `inputs/`, `CANDIDATE_INDEX.json`, and preflight candidate +documents. Its status is +`PILOT_REVIEW_PASSED_PREPARED_NOT_FROZEN_NOT_LAUNCHABLE`. A later root-owned +freeze must bind current source, model, thermal-binary, input, and resource +identities into a runnable index. No launcher consumes the candidate directly. + +Results remain `CONDITIONAL_SIMULATED`: the service is an aggregate fluid +model rather than native MQSim NAND timing, maintenance validity is metadata +only, program/erase parameters are engineering proxies, and input generation +does not establish a scientific PASS. diff --git a/experiments/eq3_system_thermal/PARAMETER_SOURCES.md b/experiments/eq3_system_thermal/PARAMETER_SOURCES.md new file mode 100644 index 0000000..0f69675 --- /dev/null +++ b/experiments/eq3_system_thermal/PARAMETER_SOURCES.md @@ -0,0 +1,27 @@ +# EQ3 system/thermal causal and reliability parameter sources + +This table records source scope before the new modules are used. Every output +is conditional; none of these entries calibrates target-HBF RBER or lifetime. + +| Parameter | Value used or allowed | Evidence class and source | Consumer and boundary | +|---|---:|---|---| +| Arrhenius activation energy | 1.01, 1.04, 1.08 eV | `PROXY`: HeatWatch reports nominal 1.04 eV and 95% CI 1.01–1.08 for old 30–40-layer 3D charge-trap MLC. Registered audit: `docs/eq3_thermal/PARAMETER_SOURCE_AUDIT.md` lines 219–260; original `docs/ref_article/luo2018-heatwatch-nand-temperature.pdf`. | `ReliabilityLedger`; conditional equivalent age only. HeatWatch fit domain was 20–70 °C and 1k–10k P/E, so extrapolation and target-HBF transfer are not physical validation. | +| Reference temperature | 358.15 K (85 °C) | `USER_CONFIRMED_SCENARIO_REFERENCE`; OCP fixed retention condition is powered-on 85 °C/24 h, recorded in `PARAMETER_SOURCE_AUDIT.md` line 38 and `ocp-verification.md` line 13. | Arrhenius reference rate is one at 358.15 K. This reference choice is not claimed to be HeatWatch's fitted reference. | +| Maintenance period | 24 h wall clock | `SCENARIO_ASSUMPTION_USER_CONFIRMED`; OCP says refresh is typically every 24–48 h and product-specific (`PARAMETER_SOURCE_AUDIT.md` line 41). | `wall_only` schedules by wall cadence. `equivalent_age_or_wall` also schedules conservatively when conditional reference-age reaches 24 h, giving Ea an actual policy consumer. Neither trigger is a failure, RBER, ECC, endurance, or throttle threshold. Equivalent initial age does not shorten the wall cadence. | +| Initial age | caller-supplied equivalent ns | `SCENARIO_ASSUMPTION`; no product initial-age distribution exists. | Literal fair initial pressure shared by arms; never divided by Arrhenius rate and never used to compress the 24 h cadence. | +| Program/erase wear | observed phase starts and successful completions, per physical block | `ACTUAL_EVENT_FACT` when supplied by the isolated maintenance backend. Its lifecycle is documented in `experiments/eq3_maintenance/backend/docs/IMPLEMENTATION_BOUNDARY.md`. | Program and erase counters remain separate. A failed operation after phase start remains an exposure fact but not a successful completion; no damage severity or P/E limit is inferred. | +| Age reset | successful refresh mapping commit only | `ACTUAL_EVENT_FACT`; narrow lifecycle in `docs/eq3_thermal/MQSIM_DIE_MAINTENANCE_NARROW_DESIGN.md`. | Failed read/program/CAS/erase-before-commit retains age. A post-commit erase failure does not undo the already committed refresh. | +| Retry/ECC | caller-supplied count, latency, energy | `SCENARIO_ASSUMPTION_NO_RBER_CLAIM`. OCP leaves ECC strength, retry cost and RBER unavailable (`PARAMETER_SOURCE_AUDIT.md` line 38). | Stored independently and never generated from temperature/age. | +| Maintenance read/program timing | 10 us / 100 us, 4096 B | `ENGINEERING_PROXY`: `experiments/eq3_maintenance/campaign_inputs.py::configuration`. | Existing isolated backend scenario only, not causal executor timing and not calibrated HBF. | +| Maintenance program energy | 0.05 W × 100 us / 4096 B = 1.220703125 nJ/B | `DERIVED_ENGINEERING_PROXY`: media power and timing in `campaign_inputs.py::energy_profile/configuration`; the 0.05 W anchor is explicitly old-device/order-of-magnitude. | May classify actual program intervals. It must not be replaced by the read-rate 50 pJ/B scenario or called target-HBF program energy. | +| Program service work | 16 planes/channel × 4096 B / 100 us = 655,360,000 B/s/channel; read-equivalent work ratio at 96 GB/s is 9375/64 | `DERIVED_ENGINEERING_PROXY` from explicit OCP projection and maintenance timing. | Topology-service `operation_media_cost.program`; a scheduling cost, not achieved product program bandwidth. | +| Erase latency/energy | native experimental adapter latency = 10 × 100 us = 1 ms; 0.05 W × 1 ms = 50 µJ/erase | `DERIVED_ENGINEERING_PROXY`: `backend/src/mqsim_online_maintenance.cpp` line 150 plus registered 0.05 W media-power scenario. | Count-based maintenance-driver energy only; not calibrated HBF erase energy. Failed-after-service energy remains. | +| Read-rate thermal energy | array 40 pJ/B; base 10 pJ/B | `SCENARIO_ASSUMPTION_USER_CONFIRMED`, implemented separately in `experiments/eq3_rate_thermal`. | Default HBF read increment is 50 pJ/B total. Relay adds 2 pJ/B partner receive plus 2 pJ/B partner send; it does not add HBM-array energy. Not program energy or backend throughput. | +| Fabric relay partner energy | 2 pJ/B receive + 2 pJ/B send | `SCENARIO_ASSUMPTION` in the system energy profile. | Relay partner base/link increments only; counted once and kept distinct from HBF array/base energy. | +| Qwen2.5 structure/payload | 7B: 15,231,233,024 B, 28 layers; 72B: 145,412,407,296 B, 80 layers | `DOC_DERIVED` official config and safetensors-index metadata registered in `experiments/eq3_maintenance/sources/qwen2_5_weight_models.json`. | Tensor sizes and architecture dependency identities. The produced trace is synthetic architecture-derived, not a PyTorch/runtime capture. | + +Native backend program/erase facts and aggregate rate-service modelled wear facts +must remain separately labelled. Unavailable by design: target-HBF RBER, ECC +strength, retry probability, failure probability, lifetime, calibrated +program/erase energy, and calibrated product token/s. Synthetic causal-token +completion is available only after an exact subwindow service is connected. diff --git a/experiments/eq3_system_thermal/README.md b/experiments/eq3_system_thermal/README.md new file mode 100644 index 0000000..c8748bf --- /dev/null +++ b/experiments/eq3_system_thermal/README.md @@ -0,0 +1,110 @@ +# Isolated four-topology conditional thermal system + +Default-off CPU experiment modules. The original MQSim source, default backend, +production scheduler/ABI and GPU paths are unchanged. Results are +`CONDITIONAL_SIMULATED`, not measured HBF throughput, calibrated temperature, +error probability or token performance. Original P2 failures and400K domain +checks remain; no model freeze or blind-test opening is implied. + +## Actual consumers + +- `run_system_point.py`: fixed20ms byte-cohort service, four real route variants, + per-source energy, existing complete coupled thermal network and future gates. + This is the frozen60-point rate matrix consumer. Tokens areUNAVAILABLE. +- `run_endpoint_guard_point.py`: same isolated chain plus HBM endpoint recovery + adapter. HBF feedback is unchanged; an HBM forwarding endpoint with no local + demand must not retain a half quota after its temperature state recovers. +- `run_maintenance_point.py`: rate foreground and real aggregate + read/program/version-commit/erase jobs compete in the same resource service; + temperature-history age produces future maintenance. The explicitly ideal + independent-resource ablation preserves maintenance/Shutdown semantics. +- `run_causal_point.py`: exact subwindow event service and a dependency executor. + Actual service completion unlocks compute; only final compute completes a + simulated token. Thermal/control windows stay20ms. Same-ledger maintenance is + supported in physical-channel mode. Uniform-group mode explicitly rejects + physical maintenance extent claims. + +All are behavioral fluid consumers. The small isolated native MQSim receipts +validate lifecycle/arbitration correspondence, notTB/s performance. See +`docs/eq3_thermal/NATIVE_PROXY_COMPARISON.md` and +`docs/eq3_thermal/SYSTEM_THERMAL_INTERFACE_COVERAGE.md` for evidence boundaries. + +## Routes, capacity and energy + +OCPGrade2 scenario:16channels×96GB/s perHBF=1.536TB/s, with explicit channel→die +mapping. Offered demand can exceed this; delivered bytes remain constrained by +media, source/partner base buffers, direct/relay links, endpoint quotas and +thermal guard. DASH direct/relay routes share media supply and unique byte +identities. Relay uses both hops and partnerHBM resources. No phantomHBM array +access is added to forwarding. + +The energy baseline is user-confirmed40pJ/B HBFarray+10pJ/B HBFbase. HBM, +forwarding, program and erase coefficients are separate documented engineering +proxies. Each activity phase is counted once. Unknown idle/selfheat is excluded +from this incremental scenario, not claimed physically zero. External GDDR +thermal/service areUNAVAILABLE in8HBF package-only experiments. + +Uniform causal mode groups16 physical channels into aggregate capacity and +spreads energy over all16 die. DASH children retain one shared media group. +Fixed tests compare capacity/bytes/per-die energy with physical mode. Nonuniform +experiments use the physical16-channel rate consumer. Buffers use finite +occupancy bounds and continuous fluid turnover, not a per-page NAND timing model. + +## Causal trace and mechanisms + +`tiny_cpu_trace.py` records an actual deterministic random-weight NumPy +Qwen2-style forward. `build_architecture_trace(dependency_mode='tiny_cpu_template')` +validates its operation/access order, GQA shapes and canonical checksum, then +regenerates target7B/72B layer counts, BF16 tensor payloads, scenario addresses +and analytical MAC counts from registered official architecture metadata. +Generated tasks carry the observed dependency template identities. Explicit +compute durations remain scenario inputs; tinyCPU elapsed time is never +scaled intoGPU timing. `synthetic_metadata_dag` remains a separate classified +mode. KV/activation traffic and pretrained model quality are not modeled. + +`CausalExecutor.poll(now)` returns jobs once. The consumer submits these to +`CausalTopologyService`, advances to the next service/compute/window event and +calls`complete` only from actual byte-complete service receipts. Streaming DAG +retirement bounds memory while preserving arrival times. Coalescing, one-layer +prefetch, issue-stall/consumption-wait, finiteLRU with actualHBM fill/read, +version-safe migration with program/commit/erase and0/1/4 named retry scenarios +have actual consumers. Retry completion does not duplicate useful bytes. + +Uniform causal workload payloads are continuous byte traffic, not NAND +page+OOB/ECC-encoded wire bytes. Page/block/plane evidence and tiny tail-padding +approximation are in`docs/eq3_thermal/GEOMETRY_AND_BYTE_SCOPE.md`. + +## Retention and maintenance + +`ReliabilityLedger` integrates contiguous temperature intervals into equivalent +age at358.15K, using separately sourced Ea proxies1.01/1.04/1.08eV. The24h +cadence is not compressed. `wall_only` leaves Ea observational; +`equivalent_age_or_wall` makes it causal through the declared conservative +refresh policy, not a predicted device failure threshold. + +Only successful per-extent program+version commit resets that extent's age. +Failed/conflicted destinations need cleanup erase; failed erases quarantine +blocks. Source data stays valid until commit and newer writes cannot be +superseded. Finite spares are channel-owned. Successful physical block erase +increments wear counts; there is no unsupported damage/lifetime model. + +The thermal rate maintenance scenario uses a4GiB aged subset perHBF, +4096×1MiB blocks with256×4KiB pages and16spares/channel. It is not a whole512GiB +capacity simulation. Shared/ideal arms receive identical initial age. Rate +age integration uses observed window endpoint midpoint temperature; exact +causal maintenance uses the previous known die temperature until its next +thermal observation, integrating to the exact commit before resetting age. + +## Execution and evidence + +Stage artifacts are outside source under`eq3_thermal/plans/four-topology-system-v1`. +Inputs, actual consumer/model/binary locks and finite resources precede every +run. CPU experiments are serial, BLAS1/GPU0. Watchdogs preserve failedraw; +status files and estimated completion checks avoid frequentAI polling. +`prepare_*` writes inputs only. `freeze_extension.py` binds reviewed extensions +to the serial launcher. Pilots gate main runs. Scope authorization comes from +the user's explicit plan, not generated approval files. + +Fixed software tests prove specific contracts; actual run receipts determine +thermal closure and system results. No qualified thermal fast path exists; +service native/proxy comparisons are not substitutes for a thermalROM ablation. diff --git a/experiments/eq3_system_thermal/__init__.py b/experiments/eq3_system_thermal/__init__.py new file mode 100644 index 0000000..23ca1c6 --- /dev/null +++ b/experiments/eq3_system_thermal/__init__.py @@ -0,0 +1,13 @@ +"""Independent causal-workload and conditional reliability helpers for EQ3.""" + +from .causal_workload import CausalExecutor, build_architecture_trace, load_architecture +from .maintenance_driver import MaintenanceDriver +from .reliability import ReliabilityLedger + +__all__ = [ + "CausalExecutor", + "MaintenanceDriver", + "ReliabilityLedger", + "build_architecture_trace", + "load_architecture", +] diff --git a/experiments/eq3_system_thermal/aggregate_maintenance_campaign.py b/experiments/eq3_system_thermal/aggregate_maintenance_campaign.py new file mode 100644 index 0000000..7c009b8 --- /dev/null +++ b/experiments/eq3_system_thermal/aggregate_maintenance_campaign.py @@ -0,0 +1,165 @@ +#!/usr/bin/env python3 +"""Aggregate a frozen 19-point maintenance campaign from validated receipts.""" + +from __future__ import annotations + +import argparse +import csv +import hashlib +import json +from pathlib import Path + + +WINDOW_NS = 20_000_000 + + +def digest(path): + return hashlib.sha256(Path(path).read_bytes()).hexdigest() + + +def observed_max(values): + observed = [value for value in values if value is not None] + return max(observed) if observed else None + + +def timing(point): + first_due = first_commit = last_commit = None + max_due = max_outstanding = 0 + with (point / "windows.jsonl").open() as stream: + for line in stream: + row = json.loads(line) + due = row["reliability"]["due_block_count"] + max_due = max(max_due, due) + if due and first_due is None: + first_due = row["end_ns"] + delta = row["maintenance_delta"] + max_outstanding = max(max_outstanding, delta["summary"]["outstanding_job_count"]) + committed = [value for value in delta["terminal_results"] + if value["status"] == "COMMITTED"] + if committed: + first_commit = row["end_ns"] if first_commit is None else first_commit + last_commit = row["end_ns"] + return {"first_due_ns": first_due, "first_commit_ns": first_commit, + "last_commit_ns": last_commit, "max_due_blocks": max_due, + "max_outstanding_jobs": max_outstanding} + + +def analyze(index_path: Path, point_analysis: Path, output: Path): + if output.exists(): + raise FileExistsError(output) + index = json.loads(index_path.read_text()) + if len(index["points"]) != 19: + raise ValueError("maintenance main index must contain 19 points") + receipt = json.loads((index_path.parent / "DONE.json").read_text()) + if receipt["status"] != "COMPLETED" or len(receipt["completed"]) != 19 or receipt["domain_failures"]: + raise ValueError("maintenance stage terminal coverage mismatch") + execution = {row["point_id"]: row for row in receipt["completed"]} + points = [] + for spec in index["points"]: + point = Path(spec["output"]) + config = json.loads(Path(spec["config"]).read_text()) + if digest(spec["config"]) != spec["config_sha256"]: + raise ValueError("config identity mismatch") + derived = json.loads((point_analysis / spec["point_id"] / "analysis.json").read_text()) + if derived["analysis_status"] != "VALIDATED_COMPLETE_RECEIPTS" or any( + value != "PASS" for key, value in derived["checks"].items() + if key in ("byte_conservation", "energy_to_thermal", "timeline")): + raise ValueError("strict point analysis did not pass") + done = json.loads((point / "DONE.json").read_text())["summary"] + panels = derived["panels"] + maintenance = derived["maintenance"] + active_rates = panels["physical_effective_hbf_traffic"]["effective_useful_Bps"][:1000] + row = {"point_id": spec["point_id"], "topology": config["topology"], + "strategy": config["strategy"], "mode": config["maintenance"]["mode"], + "ea_ev": config["maintenance"]["ea_ev"], + "active_mean_delivered_Bps": sum(active_rates)/len(active_rates), + "delivered_bytes": done["delivered_bytes"], "final_backlog_bytes": done["backlog_bytes"], + "total_energy_j": done["energy_j"], + "peak_hbf_k": observed_max(panels["owner_temperature_k"]["hbf_max"]), + "peak_hbm_k": observed_max(panels["owner_temperature_k"]["hbm_max"]), + "peak_gpu_k": observed_max(panels["owner_temperature_k"]["gpu"]), + "max_logical_backlog_bytes": max(panels["queue_backlog_bytes"]["logical_backlog"]), + "terminal_operations": maintenance["terminal_operation_count"], + "terminal_extents": maintenance["terminal_extent_count"], + "committed_operations": maintenance["terminal_status_counts"].get("COMMITTED", 0), + "successful_age_resets": maintenance["age"]["successful_age_resets"], + "free_spares": maintenance["free_spares"], + "quarantined_blocks": maintenance["quarantined_blocks"], + "wall_s": execution[spec["point_id"]]["wall_s"], + "output_bytes": execution[spec["point_id"]]["output_bytes"]} + row.update(timing(point)) + points.append(row) + by = {(p["topology"], p["strategy"], p["mode"], p["ea_ev"]): p for p in points} + metrics = ("active_mean_delivered_Bps", "delivered_bytes", "final_backlog_bytes", + "peak_hbf_k", "max_logical_backlog_bytes", "total_energy_j", + "first_due_ns", "first_commit_ns", "last_commit_ns") + def compare(left, right, identity): + return {"identity": identity, "left_point_id": left["point_id"], + "right_point_id": right["point_id"], + "right_minus_left": {name: right[name]-left[name] for name in metrics}} + policy = [] + for topology in ("mixed_direct", "relay", "dash", "all_hbf_direct"): + guard = by[(topology, "guard_only", "shared", 1.04)] + for strategy in ("thermal_hysteresis_guard", "read_rate_feedback_thermal_guard_v1"): + policy.append(compare(guard, by[(topology, strategy, "shared", 1.04)], + f"{topology}: {strategy} minus guard_only")) + contention = [] + for topology in ("mixed_direct", "relay", "dash"): + shared = by[(topology, "read_rate_feedback_thermal_guard_v1", "shared", 1.04)] + ideal = by[(topology, "read_rate_feedback_thermal_guard_v1", "ideal_independent", 1.04)] + contention.append(compare(shared, ideal, f"{topology}: ideal_independent minus shared")) + ea = [] + for strategy in ("guard_only", "read_rate_feedback_thermal_guard_v1"): + center = by[("mixed_direct", strategy, "shared", 1.04)] + for value in (1.01, 1.08): + ea.append(compare(center, by[("mixed_direct", strategy, "shared", value)], + f"mixed_direct {strategy}: Ea {value} minus 1.04 eV")) + result = {"schema_version": "eq3-maintenance-main-aggregate-v1", "status": "PASS", + "point_count": 19, "points": points, "policy_costs": policy, + "shared_vs_ideal": contention, "ea_sensitivity": ea, + "resource_receipt": {"stage_wall_s": receipt["wall_s"], + "retained_output_bytes": sum(p["output_bytes"] for p in points)}, + "claim_scope": ("CONDITIONAL_AGGREGATE_RATE_SERVICE; DELTAS_HAVE_NO_PRESET_WINNER; " + "NO_ECC_OR_NATIVE_NAND_THROUGHPUT_BENEFIT_CLAIM")} + output.mkdir(parents=True) + (output / "MAINTENANCE_MAIN_ANALYSIS.json").write_text( + json.dumps(result, indent=2, sort_keys=True, allow_nan=False)+"\n") + with (output / "point-summary.csv").open("w", newline="") as stream: + writer = csv.DictWriter(stream, fieldnames=list(points[0])); writer.writeheader(); writer.writerows(points) + lines = ["# Maintenance main result", "", + "All 19 points passed strict timeline, byte, energy, and unique-completion checks. Deltas below are observational costs, with no preferred policy direction encoded.", "", + "| Comparison | Δ active TB/s | Δ delivered TB | Δ final backlog TB | Δ peak HBF K |", + "|---|---:|---:|---:|---:|"] + for item in policy: + delta = item["right_minus_left"] + lines.append(f"| {item['identity']} | {delta['active_mean_delivered_Bps']/1e12:.6f} | {delta['delivered_bytes']/1e12:.6f} | {delta['final_backlog_bytes']/1e12:.6f} | {delta['peak_hbf_k']:.6f} |") + lines += ["", "## Shared versus ideal-independent maintenance", "", + "| Comparison | Δ delivered TB | Δ backlog TB | Δ peak HBF K | Δ first commit ms |", + "|---|---:|---:|---:|---:|"] + for item in contention: + delta = item["right_minus_left"] + lines.append(f"| {item['identity']} | {delta['delivered_bytes']/1e12:.6f} | {delta['final_backlog_bytes']/1e12:.6f} | {delta['peak_hbf_k']:.9f} | {delta['first_commit_ns']/1e6:.3f} |") + lines += ["", "## Arrhenius activation-energy sensitivity", "", + "| Comparison | Δ delivered GB | Δ backlog GB | Δ peak HBF K | Δ first commit ms | Δ last commit ms |", + "|---|---:|---:|---:|---:|---:|"] + for item in ea: + delta = item["right_minus_left"] + lines.append(f"| {item['identity']} | {delta['delivered_bytes']/1e9:.6f} | {delta['final_backlog_bytes']/1e9:.6f} | {delta['peak_hbf_k']:.9f} | {delta['first_commit_ns']/1e6:.3f} | {delta['last_commit_ns']/1e6:.3f} |") + lines += ["", "All 19 points committed every declared operation, reset only the successful extents, returned every spare, and quarantined no block. Exact point counts and deltas are stored in `MAINTENANCE_MAIN_ANALYSIS.json`. Token throughput, native NAND timing, ECC benefit, and lifetime remain unavailable."] + (output / "RESULT.md").write_text("\n".join(lines)+"\n") + return result + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--index", type=Path, required=True) + parser.add_argument("--point-analysis", type=Path, required=True) + parser.add_argument("--output", type=Path, required=True) + args = parser.parse_args() + value = analyze(args.index.resolve(strict=True), args.point_analysis.resolve(strict=True), + args.output.resolve()) + print(json.dumps({"status": value["status"], "point_count": value["point_count"]})) + + +if __name__ == "__main__": + main() diff --git a/experiments/eq3_system_thermal/analyze_campaign.py b/experiments/eq3_system_thermal/analyze_campaign.py new file mode 100644 index 0000000..0ec3a06 --- /dev/null +++ b/experiments/eq3_system_thermal/analyze_campaign.py @@ -0,0 +1,571 @@ +#!/usr/bin/env python3 +"""Strict, read-only analysis for the 60-point four-topology campaign.""" + +from __future__ import annotations + +import argparse +import csv +import hashlib +import json +import math +from itertools import combinations +from pathlib import Path +from typing import Any, Iterable + + +EXPECTED_POINT_COUNT = 60 +TOPOLOGIES = ("mixed_direct", "relay", "dash", "all_hbf_direct") +STRATEGIES = ("guard_only", "thermal_hysteresis_guard", + "read_rate_feedback_thermal_guard_v1") +RATES_BPS = (384_000_000_000, 768_000_000_000, 1_152_000_000_000, + 1_536_000_000_000, 1_920_000_000_000) +STATE_RANK = {"normal": 0, "light": 1, "severe": 2, "shutdown": 3} +POINT_FILES = ("config.json", "manifest.json", "DONE.json", "windows.jsonl") + + +def _load(path: Path) -> dict: + value = json.loads(path.read_text()) + if not isinstance(value, dict): + raise ValueError(f"{path} must contain a JSON object") + return value + + +def _rows(path: Path) -> Iterable[dict]: + with path.open() as stream: + for line_number, line in enumerate(stream, 1): + if not line.strip(): + continue + value = json.loads(line) + if not isinstance(value, dict): + raise ValueError(f"{path}:{line_number} must be a JSON object") + yield value + + +def _finite(value: Any, label: str) -> float: + if isinstance(value, bool) or not isinstance(value, (int, float)): + raise ValueError(f"{label} must be numeric") + result = float(value) + if not math.isfinite(result): + raise ValueError(f"{label} must be finite") + return result + + +def _integer(value: Any, label: str, *, positive: bool = False) -> int: + if isinstance(value, bool) or not isinstance(value, int): + raise ValueError(f"{label} must be integer") + if value < (1 if positive else 0): + raise ValueError(f"{label} must be {'positive' if positive else 'nonnegative'}") + return value + + +def _identity(config: dict, manifest: dict, point: Path) -> dict: + required = ("point_id", "topology", "strategy", "workload", "recovery_ns") + missing = [key for key in required if key not in config] + if missing: + raise ValueError(f"{point} config identity missing {missing}") + topology, strategy = config["topology"], config["strategy"] + workload = config["workload"] + if not isinstance(workload, dict): + raise ValueError("config.workload must be an object") + rate = _integer(workload.get("per_stack_Bps"), "workload.per_stack_Bps", positive=True) + active_ns = _integer(workload.get("active_ns"), "workload.active_ns", positive=True) + recovery_ns = _integer(config["recovery_ns"], "recovery_ns", positive=True) + duration_ns = active_ns + recovery_ns + if topology not in TOPOLOGIES or strategy not in STRATEGIES or rate not in RATES_BPS: + raise ValueError(f"{point} manifest is outside the frozen design") + if duration_ns <= active_ns: + raise ValueError("duration_ns must include a recovery interval") + if not isinstance(config["point_id"], str) or not config["point_id"]: + raise ValueError("point_id must be nonempty") + config_sha = hashlib.sha256((point / "config.json").read_bytes()).hexdigest() + if manifest.get("input_sha256") != config_sha: + raise ValueError(f"{point} manifest input hash differs from config.json") + return {"point_id": config["point_id"], "topology": topology, + "offered_rate_Bps": rate, "strategy": strategy, + "active_ns": active_ns, "duration_ns": duration_ns, + "model_variant": Path(str(manifest.get("model_dir", "UNDECLARED"))).name, + "input_sha256": config_sha} + + +def _token_fact(causal: Any) -> tuple[int | None, str]: + if causal is None: + return None, "UNAVAILABLE_CAUSAL_NULL" + if not isinstance(causal, dict): + raise ValueError("causal must be null or object") + # Never derive tokens from bytes or layers. Only consume a named completion fact. + for key in ("completed_tokens", "token_completions"): + if key in causal: + value = _integer(causal[key], f"causal.{key}") + return value, f"OBSERVED_FIELD:{key}" + return None, "UNAVAILABLE_NO_EXPLICIT_TOKEN_COMPLETION_FIELD" + + +def _sum_energy(mapping: Any, label: str) -> float: + if not isinstance(mapping, dict): + raise ValueError(f"{label} must be an object") + result = sum(_finite(value, f"{label}.{key}") for key, value in mapping.items()) + if result < 0: + raise ValueError(f"{label} cannot sum negative") + return result + + +def _media_activity_bytes(receipt: dict) -> int: + """Consume the final service field; retain one explicit fixture alias.""" + present = [key for key in ("media_activity_bytes", "media_payload_bytes") if key in receipt] + if len(present) != 1: + raise ValueError("stack receipt must contain exactly one media activity byte field") + return _integer(receipt[present[0]], present[0]) + + +def analyze_point(point: Path) -> dict: + config, manifest, done = (_load(point / "config.json"), _load(point / "manifest.json"), + _load(point / "DONE.json")) + identity = _identity(config, manifest, point) + if done.get("execution_status", done.get("status")) not in ("COMPLETED", "DONE"): + raise ValueError(f"{point} is not complete") + for key in ("point_id", "topology", "strategy"): + if key in done and done[key] != identity[key]: + raise ValueError(f"{point} DONE {key} differs from manifest") + + per_stack: dict[str, dict[str, Any]] = {} + traces = {"end_ns": [], "offered_Bps": [], "delivered_Bps": [], "backlog_bytes": [], + "max_temperature_k": [], "worst_state_rank": [], "maintenance_pending_bytes": [], + "maintenance_completions": [], "token_completions": [], + "temperatures_k_by_owner": {}} + totals = {"offered_effective_bytes": 0, "delivered_effective_bytes": 0, + "media_payload_bytes": 0, "energy_j": 0.0, + "maintenance_completion_count": 0} + expected_start = 0 + service_stacks: set[str] | None = None + thermal_owners: set[str] | None = None + completed_maintenance: set[str] = set() + token_semantics: set[str] = set() + reliability_statuses: set[str] = set() + final_backlog = 0 + count = 0 + + for row in _rows(point / "windows.jsonl"): + start = _integer(row.get("start_ns"), "start_ns") + end = _integer(row.get("end_ns"), "end_ns", positive=True) + if start != expected_start or end <= start: + raise ValueError(f"{point} windows are not positive and contiguous") + if end > identity["duration_ns"]: + raise ValueError(f"{point} window exceeds declared duration") + expected_start = end + duration = end - start + count += 1 + service, energy, thermal = row.get("service"), row.get("energy"), row.get("thermal") + if not all(isinstance(value, dict) for value in (service, energy, thermal)): + raise ValueError("window lacks service/energy/thermal objects") + if service.get("start_ns", start) != start or service.get("end_ns", end) != end: + raise ValueError("service interval differs from window") + stacks = service.get("stacks") + if not isinstance(stacks, dict) or not stacks: + raise ValueError("service.stacks must be nonempty") + if service_stacks is None: + service_stacks = set(stacks) + for stack in sorted(service_stacks): + per_stack[stack] = { + "offered_effective_bytes": 0, "delivered_effective_bytes": 0, + "media_activity_bytes": 0, "final_backlog_effective_bytes": 0, + "peak_temperature_k": None, "final_temperature_k": None, + "peak_oldest_wait_ns": None, + "state_time_ns": {state: 0 for state in STATE_RANK}, + } + elif set(stacks) != service_stacks: + raise ValueError("service stack coverage changed across windows") + + window_offered = window_delivered = window_media = window_backlog = 0 + for stack in sorted(service_stacks): + receipt = stacks[stack] + if not isinstance(receipt, dict): + raise ValueError("stack receipt must be object") + offered = _integer(receipt.get("offered_effective_bytes"), "offered bytes") + delivered = _integer(receipt.get("delivered_effective_bytes"), "delivered bytes") + backlog = _integer(receipt.get("backlog_effective_bytes"), "backlog bytes") + media = _media_activity_bytes(receipt) + dest = per_stack[stack] + dest["offered_effective_bytes"] += offered + dest["delivered_effective_bytes"] += delivered + dest["media_activity_bytes"] += media + dest["final_backlog_effective_bytes"] = backlog + wait = receipt.get("oldest_wait_ns") + if wait is not None: + wait = _integer(wait, "oldest_wait_ns") + dest["peak_oldest_wait_ns"] = max(dest["peak_oldest_wait_ns"] or 0, wait) + window_offered += offered; window_delivered += delivered + window_backlog += backlog; window_media += media + totals["offered_effective_bytes"] += window_offered + totals["delivered_effective_bytes"] += window_delivered + totals["media_payload_bytes"] += window_media + final_backlog = window_backlog + + component_j = _sum_energy(energy.get("component_energy_j"), "component_energy_j") + scope_j = _sum_energy(energy.get("scope_energy_j"), "scope_energy_j") + total_j = _finite(energy.get("total_j"), "energy.total_j") + if not math.isclose(component_j, total_j, rel_tol=1e-12, abs_tol=1e-12): + raise ValueError("component energy does not conserve to total_j") + if not math.isclose(scope_j, total_j, rel_tol=1e-12, abs_tol=1e-12): + raise ValueError("scope energy does not conserve to total_j") + thermal_energy = thermal.get("energy_j") + if not isinstance(thermal_energy, dict) or not isinstance(thermal_energy.get("window"), dict): + raise ValueError("thermal.energy_j.window must be an object") + thermal_activity_j = _finite(thermal_energy["window"].get("activity_input_j"), + "thermal window activity_input_j") + thermal_total_j = _finite(thermal_energy["window"].get("total_input_j"), + "thermal window total_input_j") + if not math.isclose(thermal_activity_j, total_j, rel_tol=1e-12, abs_tol=1e-12): + raise ValueError("thermal injected energy differs from mapped energy") + if not math.isclose(thermal_total_j, total_j, rel_tol=1e-12, abs_tol=1e-12): + raise ValueError("thermal total input includes energy outside mapped baseline") + totals["energy_j"] += total_j + + temperatures, states = thermal.get("temperatures"), thermal.get("stack_states") + if not isinstance(temperatures, dict) or not isinstance(states, dict): + raise ValueError("thermal temperatures/states must be objects") + if not service_stacks.issubset(temperatures) or not service_stacks.issubset(states): + raise ValueError("thermal facts do not cover every service stack") + if thermal_owners is None: + thermal_owners = set(temperatures) + if "gpu" not in thermal_owners: + raise ValueError("thermal facts do not contain the package GPU owner") + traces["temperatures_k_by_owner"] = {owner: [] for owner in sorted(thermal_owners)} + elif set(temperatures) != thermal_owners: + raise ValueError("thermal owner coverage changed across windows") + for owner in sorted(thermal_owners): + traces["temperatures_k_by_owner"][owner].append( + _finite(temperatures[owner], f"temperature.{owner}")) + worst_rank = 0 + for stack in sorted(service_stacks): + temperature = _finite(temperatures[stack], f"temperature.{stack}") + state = states[stack] + if state not in STATE_RANK: + raise ValueError(f"unsupported thermal state {state!r}") + dest = per_stack[stack] + dest["peak_temperature_k"] = max(dest["peak_temperature_k"] or -math.inf, temperature) + dest["final_temperature_k"] = temperature + dest["state_time_ns"][state] += duration + worst_rank = max(worst_rank, STATE_RANK[state]) + + progress = service.get("job_progress", []) + if not isinstance(progress, list): + raise ValueError("service.job_progress must be an array") + pending_maintenance = 0 + for job in progress: + if not isinstance(job, dict): + raise ValueError("job_progress entry must be an object") + if job.get("maintenance_id") is not None: + pending_maintenance += _integer(job.get("remaining_bytes"), "maintenance remaining") + completions = service.get("maintenance_completion_ids", []) + if not isinstance(completions, list) or any(not isinstance(value, str) for value in completions): + raise ValueError("maintenance completion IDs must be strings") + if completed_maintenance.intersection(completions) or len(set(completions)) != len(completions): + raise ValueError("maintenance completion identity is not unique") + completed_maintenance.update(completions) + totals["maintenance_completion_count"] += len(completions) + + token_count, token_kind = _token_fact(row.get("causal")) + token_semantics.add(token_kind) + reliability = row.get("reliability") + if not isinstance(reliability, dict) or not isinstance(reliability.get("status"), str): + raise ValueError("reliability must carry an explicit status") + reliability_statuses.add(reliability["status"]) + traces["end_ns"].append(end) + traces["offered_Bps"].append(window_offered * 1e9 / duration) + traces["delivered_Bps"].append(window_delivered * 1e9 / duration) + traces["backlog_bytes"].append(window_backlog) + traces["max_temperature_k"].append(max(float(temperatures[s]) for s in thermal_owners)) + traces["worst_state_rank"].append(worst_rank) + traces["maintenance_pending_bytes"].append(pending_maintenance) + traces["maintenance_completions"].append(len(completions)) + traces["token_completions"].append(token_count) + + if count == 0 or expected_start != identity["duration_ns"]: + raise ValueError(f"{point} window coverage differs from declared duration") + conservation_error = totals["offered_effective_bytes"] - totals["delivered_effective_bytes"] - final_backlog + if conservation_error != 0: + raise ValueError(f"{point} foreground byte conservation failed by {conservation_error}") + for stack, result in per_stack.items(): + error = (result["offered_effective_bytes"] - result["delivered_effective_bytes"] - + result["final_backlog_effective_bytes"]) + if error != 0: + raise ValueError(f"{point} {stack} byte conservation failed by {error}") + result["byte_conservation_error"] = error + result["mean_delivered_Bps"] = result["delivered_effective_bytes"] * 1e9 / identity["duration_ns"] + totals.update({ + "final_backlog_effective_bytes": final_backlog, + "byte_conservation_error": conservation_error, + "mean_delivered_Bps": totals["delivered_effective_bytes"] * 1e9 / identity["duration_ns"], + "delivery_fraction": (totals["delivered_effective_bytes"] / totals["offered_effective_bytes"] + if totals["offered_effective_bytes"] else None), + "peak_temperature_k": max(traces["max_temperature_k"]), + "final_peak_temperature_k": max(values[-1] for values in traces["temperatures_k_by_owner"].values()), + }) + token_available = all(value is not None for value in traces["token_completions"]) + no_maintenance = reliability_statuses == {"NO_MAINTENANCE_DEMAND_IN_BASE_RATE_WORKLOAD"} + return {**identity, "point_path": str(point.resolve()), "window_count": count, + "service_stacks": sorted(service_stacks), "totals": totals, "per_stack": per_stack, + "token_metric": {"status": "AVAILABLE" if token_available else "UNAVAILABLE", + "semantics": sorted(token_semantics)}, + "maintenance_metric": { + "status": ("NOT_EXERCISED_NO_DEMAND" if no_maintenance else + "AVAILABLE_MODELLED_JOB_PROGRESS"), + "reliability_statuses": sorted(reliability_statuses), + }, + "limitations": {"backend_latency": "UNKNOWN", + "service": "MODELLED_AGGREGATED_FLUID_NOT_MQSIM_COMPLETION", + "token_per_s": "AVAILABLE_ONLY_FROM_EXPLICIT_CAUSAL_TOKEN_COMPLETION_FACT", + "maintenance": "MODELLED_MAINTENANCE_JOB_PROGRESS_AND_UNIQUE_COMPLETIONS"}, + "_trace": traces} + + +def _delta(right: Any, left: Any) -> Any: + return None if right is None or left is None else right - left + + +def policy_comparisons(points: list[dict]) -> list[dict]: + groups: dict[tuple, dict[str, dict]] = {} + for point in points: + key = (point["topology"], point["offered_rate_Bps"], point["model_variant"]) + if point["strategy"] in groups.setdefault(key, {}): + raise ValueError(f"duplicate strategy for {key}") + groups[key][point["strategy"]] = point + rows = [] + for key, by_strategy in sorted(groups.items()): + for left_name, right_name in combinations(STRATEGIES, 2): + if left_name not in by_strategy or right_name not in by_strategy: + continue + left, right = by_strategy[left_name], by_strategy[right_name] + if left["totals"]["offered_effective_bytes"] != right["totals"]["offered_effective_bytes"]: + raise ValueError(f"policy pair offered demand differs for {key}") + lt, rt = left["totals"], right["totals"] + rows.append({"topology": key[0], "offered_rate_Bps": key[1], + "model_variant": key[2], "left_strategy": left_name, + "right_strategy": right_name, + "delta_semantics": "RIGHT_MINUS_LEFT_NO_BENEFIT_DIRECTION_ASSUMED", + "delivered_effective_bytes_delta": _delta(rt["delivered_effective_bytes"], lt["delivered_effective_bytes"]), + "final_backlog_effective_bytes_delta": _delta(rt["final_backlog_effective_bytes"], lt["final_backlog_effective_bytes"]), + "peak_temperature_k_delta": _delta(rt["peak_temperature_k"], lt["peak_temperature_k"]), + "energy_j_delta": _delta(rt["energy_j"], lt["energy_j"]), + "maintenance_completion_count_delta": _delta(rt["maintenance_completion_count"], lt["maintenance_completion_count"])}) + return rows + + +def _is_point(path: Path) -> bool: + return all((path / name).is_file() for name in POINT_FILES) + + +def _discover(campaign: Path) -> tuple[list[Path], str]: + index_path = campaign / "RUN_INDEX.json" + if index_path.is_file(): + index = _load(index_path) + entries = index.get("points") + if not isinstance(entries, list): + raise ValueError("RUN_INDEX.points must be an array") + main = [row for row in entries if isinstance(row, dict) and row.get("kind") == "base"] + if len(main) != EXPECTED_POINT_COUNT: + raise ValueError("RUN_INDEX must whitelist exactly 60 kind=base points") + paths = [] + for row in main: + raw = row.get("output") + if not isinstance(raw, str) or not raw: + raise ValueError("RUN_INDEX point lacks output") + path = Path(raw) + if not path.is_absolute(): + path = campaign / path + path = path.resolve() + if not _is_point(path): + raise ValueError(f"indexed point incomplete: {path}") + config_path = Path(row.get("config", "")) + if not config_path.is_file(): + raise ValueError(f"RUN_INDEX config missing: {config_path}") + if row.get("config_sha256") != hashlib.sha256(config_path.read_bytes()).hexdigest(): + raise ValueError(f"RUN_INDEX config hash differs: {config_path}") + paths.append(path) + if len(set(paths)) != len(paths): + raise ValueError("RUN_INDEX contains duplicate output paths") + return paths, "RUN_INDEX_KIND_BASE_WHITELIST" + paths = sorted({path.parent.resolve() for path in campaign.rglob("DONE.json") if _is_point(path.parent)}) + if not paths: + raise ValueError("no complete point directories found") + return paths, "COMPLETE_POINT_DISCOVERY_WITHOUT_RUN_INDEX" + + +def analyze_campaign(campaign: Path, *, require_complete: bool = True) -> dict: + paths, discovery = _discover(campaign) + points = [analyze_point(path) for path in paths] + identities = {(p["topology"], p["offered_rate_Bps"], p["strategy"]) for p in points} + expected = {(topology, rate, strategy) for topology in TOPOLOGIES + for rate in RATES_BPS for strategy in STRATEGIES} + if len(identities) != len(points): + raise ValueError("duplicate campaign design identity") + if require_complete and identities != expected: + raise ValueError("completed points do not form the frozen 4x5x3 design") + return {"schema_version": "eq3-system-thermal-campaign-analysis-v1", + "campaign": str(campaign.resolve()), "point_discovery": discovery, + "completed_point_count": len(points), "expected_point_count": EXPECTED_POINT_COUNT, + "design_complete": identities == expected, "points": points, + "policy_comparisons": policy_comparisons(points), + "global_limitations": { + "backend_latency": "UNKNOWN", + "service": "MODELLED_AGGREGATED_FLUID_NOT_NATIVE_MQSIM", + "token_per_s": "UNAVAILABLE_WHEN_CAUSAL_IS_NULL; NEVER_INFERRED_FROM_BYTES", + "thermal": "CONDITIONAL_MODEL_WITH_REGISTERED_VARIANTS_NOT_PRODUCT_MEASUREMENT"}} + + +def _serializable(analysis: dict) -> dict: + return {**analysis, "points": [{k: v for k, v in p.items() if k != "_trace"} + for p in analysis["points"]]} + + +def _slug(value: str) -> str: + return "".join(c.lower() if c.isalnum() else "-" for c in value).strip("-") + + +def write_plots(analysis: dict, output: Path) -> list[str]: + import matplotlib + matplotlib.use("Agg") + import matplotlib.pyplot as plt + + paths = [] + for topology in TOPOLOGIES: + points = [p for p in analysis["points"] if p["topology"] == topology] + if not points: + continue + fig, axes = plt.subplots(3, 2, figsize=(14, 11), sharex=True, constrained_layout=True) + axes = axes.ravel() + colors = {rate: plt.cm.viridis(i / max(1, len(RATES_BPS)-1)) + for i, rate in enumerate(RATES_BPS)} + styles = {STRATEGIES[0]: "-", STRATEGIES[1]: "--", STRATEGIES[2]: ":"} + for point in sorted(points, key=lambda p: (p["offered_rate_Bps"], p["strategy"])): + trace = point["_trace"] + x = [value / 1e9 for value in trace["end_ns"]] + label = f"{point['offered_rate_Bps']/1e12:.3f} TB/s | {point['strategy']}" + kwargs = {"color": colors[point["offered_rate_Bps"]], + "linestyle": styles[point["strategy"]], "linewidth": 1, "label": label} + axes[0].plot(x, [v / 1e12 for v in trace["offered_Bps"]], **kwargs) + axes[0].plot(x, [v / 1e12 for v in trace["delivered_Bps"]], alpha=.7, + color=kwargs["color"], linestyle=kwargs["linestyle"], linewidth=.8) + axes[1].plot(x, trace["max_temperature_k"], **kwargs) + axes[2].step(x, trace["worst_state_rank"], where="post", **kwargs) + axes[3].plot(x, [v / 1e6 for v in trace["maintenance_pending_bytes"]], **kwargs) + axes[3].scatter(x, trace["maintenance_completions"], color=kwargs["color"], s=2) + axes[4].plot(x, [v / 1e12 for v in trace["backlog_bytes"]], **kwargs) + axes[0].set_ylabel("Offered / delivered TB/s") + axes[1].set_ylabel("Max stack K") + axes[2].set_ylabel("Worst state rank") + axes[2].set_yticks(list(STATE_RANK.values()), list(STATE_RANK)) + axes[3].set_ylabel("Maint pending MB\n(points=window commits)") + axes[4].set_ylabel("Foreground backlog TB") + if any(p["token_metric"]["status"] == "AVAILABLE" for p in points): + axes[5].text(.5, .5, "Explicit token facts available; see JSON/CSV\n" + "Token-rate aggregation intentionally requires frozen causal semantics", + ha="center", va="center", transform=axes[5].transAxes) + else: + axes[5].text(.5, .5, "TOKEN/S UNAVAILABLE\ncausal=null or no explicit token completion fact\n" + "Never inferred from bytes", ha="center", va="center", + transform=axes[5].transAxes) + axes[5].set_axis_off() + if all(p["maintenance_metric"]["status"] == "NOT_EXERCISED_NO_DEMAND" for p in points): + axes[3].text(.5, .5, "NO MAINTENANCE DEMAND\nqueue/commit path not exercised", + ha="center", va="center", transform=axes[3].transAxes, + bbox={"facecolor": "white", "alpha": .8, "edgecolor": "gray"}) + for axis in axes[:5]: + axis.axvline(points[0]["active_ns"] / 1e9, color="gray", linestyle="-.", linewidth=.7) + axis.grid(alpha=.2) + axis.set_xlabel("Time (s)") + axes[0].legend(fontsize=5, ncol=2) + fig.suptitle(f"{topology}: conditional modelled system/thermal campaign\n" + "line color=offered rate; style=policy; delivered is the lighter rate trace") + path = output / f"six-panel-{_slug(topology)}.png" + fig.savefig(path, dpi=150) + plt.close(fig) + paths.append(str(path)) + + # The overview intentionally uses a single worst-temperature trace. + # These representative plots retain every actual thermal owner instead + # of hiding GPU or HBM/HBF spatial asymmetry behind that maximum. + representative = [p for p in points if p["offered_rate_Bps"] == 1_536_000_000_000] + for point in sorted(representative, key=lambda p: STRATEGIES.index(p["strategy"])): + trace = point["_trace"] + x = [value / 1e9 for value in trace["end_ns"]] + owner_fig, axis = plt.subplots(figsize=(11, 6), constrained_layout=True) + for owner, values in sorted(trace["temperatures_k_by_owner"].items()): + if owner == "gpu": + style, width = "-", 2.0 + elif owner.startswith("hbm"): + style, width = "--", 1.0 + else: + style, width = "-", 1.0 + axis.plot(x, values, linestyle=style, linewidth=width, label=owner) + axis.axvline(point["active_ns"] / 1e9, color="gray", linestyle="-.", linewidth=.8) + axis.set_xlabel("Time (s)"); axis.set_ylabel("Owner hotspot K") + axis.grid(alpha=.2); axis.legend(ncol=5, fontsize=7) + axis.set_title(f"All package thermal owners | {topology} | 1.536 TB/s/stack | {point['strategy']}\n" + "GPU thick; HBM dashed; HBF solid; conditional model temperatures") + owner_path = output / f"owner-temperatures-{_slug(topology)}-{_slug(point['strategy'])}-1536.png" + owner_fig.savefig(owner_path, dpi=150) + plt.close(owner_fig) + paths.append(str(owner_path)) + return paths + + +def write_outputs(analysis: dict, output: Path, *, plots: bool = True) -> None: + output.mkdir(parents=True, exist_ok=False) + (output / "SYSTEM_THERMAL_CAMPAIGN_ANALYSIS.json").write_text( + json.dumps(_serializable(analysis), indent=2, sort_keys=True, allow_nan=False) + "\n") + point_fields = ("point_id", "topology", "offered_rate_Bps", "strategy", "model_variant", + "offered_effective_bytes", "delivered_effective_bytes", + "final_backlog_effective_bytes", "delivery_fraction", "mean_delivered_Bps", + "peak_temperature_k", "final_peak_temperature_k", "energy_j", + "maintenance_completion_count", "byte_conservation_error", "token_status") + with (output / "point-summary.csv").open("w", newline="") as stream: + writer = csv.DictWriter(stream, fieldnames=point_fields); writer.writeheader() + for p in analysis["points"]: + writer.writerow({**{key: p.get(key, p["totals"].get(key)) for key in point_fields}, + "token_status": p["token_metric"]["status"]}) + stack_fields = ("point_id", "topology", "offered_rate_Bps", "strategy", "stack", + "offered_effective_bytes", "delivered_effective_bytes", + "final_backlog_effective_bytes", "mean_delivered_Bps", "peak_temperature_k", + "final_temperature_k", "peak_oldest_wait_ns", "normal_ns", "light_ns", + "severe_ns", "shutdown_ns", "byte_conservation_error") + with (output / "per-stack-summary.csv").open("w", newline="") as stream: + writer = csv.DictWriter(stream, fieldnames=stack_fields); writer.writeheader() + for p in analysis["points"]: + for stack, row in sorted(p["per_stack"].items()): + output_row = {"point_id": p["point_id"], "topology": p["topology"], + "offered_rate_Bps": p["offered_rate_Bps"], + "strategy": p["strategy"], "stack": stack} + for field in stack_fields: + if field in row: + output_row[field] = row[field] + output_row.update({f"{state}_ns": row["state_time_ns"][state] + for state in STATE_RANK}) + writer.writerow(output_row) + comparisons = analysis["policy_comparisons"] + if comparisons: + with (output / "policy-comparisons.csv").open("w", newline="") as stream: + writer = csv.DictWriter(stream, fieldnames=list(comparisons[0])); writer.writeheader(); writer.writerows(comparisons) + plot_paths = write_plots(analysis, output) if plots else [] + lines = ["# Four-topology system/thermal campaign analysis", "", + f"Completed main points: {analysis['completed_point_count']} / {EXPECTED_POINT_COUNT}.", "", + "All byte, component/scope energy, thermal-input energy, time-axis, stack-coverage, and unique maintenance completion checks passed before output generation.", "", + "Rates and queues are properties of the modelled aggregate topology service, not native MQSim completion. Token/s remains unavailable whenever the causal field is null or lacks an explicit token completion fact; it is never reconstructed from bytes. Policy deltas are right minus left and do not encode a preferred direction.", "", + f"Generated topology six-panel figures: {len(plot_paths)}."] + (output / "SYSTEM_THERMAL_CAMPAIGN_ANALYSIS.md").write_text("\n".join(lines) + "\n") + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--campaign", type=Path, required=True) + parser.add_argument("--output", type=Path, required=True) + parser.add_argument("--allow-partial", action="store_true") + parser.add_argument("--no-plots", action="store_true") + args = parser.parse_args(argv) + analysis = analyze_campaign(args.campaign.resolve(), require_complete=not args.allow_partial) + write_outputs(analysis, args.output.resolve(), plots=not args.no_plots) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/experiments/eq3_system_thermal/analyze_ecc_pilots.py b/experiments/eq3_system_thermal/analyze_ecc_pilots.py new file mode 100644 index 0000000..3c5e2e7 --- /dev/null +++ b/experiments/eq3_system_thermal/analyze_ecc_pilots.py @@ -0,0 +1,120 @@ +#!/usr/bin/env python3 +"""Replay the optional reliability receipts, then compare frozen null/weak pairs.""" +import argparse +from collections import defaultdict +import json +import math +from pathlib import Path +from analyze_extensions import analyze_point,plot_panels +from ecc_cost_proxy import ReadCostProxy + + +def analyze(point): + generic=analyze_point(point) + config=json.loads((point/'config.json').read_text()) + option=config['hbf_read_cost_proxy'] + provider=ReadCostProxy(option['profile'],option['initial_by_stack']) + manifest=json.loads((point/'manifest.json').read_text()) + normalized=json.loads((Path(manifest['model_dir'])/'normalized.json').read_text()) + die_ids={stack:[c['id'] for c in normalized['components'] + if c.get('device_id')==stack and c.get('role')=='array_die'] + for stack in provider.states} + if any(not ids for ids in die_ids.values()):raise ValueError('missing actual array mapping') + decisions=set();physical=defaultdict(int);retry=defaultdict(int);decode=defaultdict(int) + energy=0.;peaks={};rows=0;times=[];owner_series=defaultdict(list) + for line in (point/'windows.jsonl').open(): + row=json.loads(line);rows+=1 + receipt=row['hbf_read_cost_proxy'] + for d in receipt['admission_cost_decisions']: + if d['job_id'] in decisions:raise ValueError('duplicate cost sampling') + decisions.add(d['job_id']) + if not row['start_ns']<=d['at_ns']<=row['end_ns']:raise ValueError('noncausal decision time') + expected=provider.cost(d['stack'],d['at_ns']) + if expected['attempt_work_milli']!=d['attempt_work_milli']:raise ValueError('unreproducible retry effort') + if not math.isclose(expected['equivalent_age_days_30c'],d['equivalent_age_days_30c'],rel_tol=1e-12): + raise ValueError('retention age differs from previous-temperature replay') + for a in row['service']['activities']: + s=a['stack'] + if a['phase']=='media_read' and s in provider.states: + physical[s]+=a['bytes'] + if a['operation']=='retry_internal': + if s not in provider.states or a['phase']!='media_read':raise ValueError('retry leaks out of HBF media scope') + retry[s]+=a['bytes'] + if a['phase']=='ecc_decode':decode[s]+=a['bytes'] + energy+=sum(v for k,v in row['energy']['scope_energy_j'].items() if k.startswith('retry_internal:')) + entities=row['thermal']['entity_temperatures_k'] + temperatures={s:max(entities[k]['hotspot_k'] for k in die_ids[s]) + for s in provider.states} + provider.observe(row['start_ns'],row['end_ns'],temperatures) + for s,v in provider.states.items(): + observed=receipt['state']['states'][s] + if not math.isclose(v['equivalent_age_ns'],observed['equivalent_age_ns'],rel_tol=1e-12): + raise ValueError('window age mismatch') + times.append(row['end_ns']/1e9) + for s,v in row['thermal']['temperatures'].items(): + peaks[s]=max(peaks.get(s,v),v);owner_series[s].append(v) + if dict(physical)!=dict(decode):raise ValueError('physical media/decoder activity byte mismatch') + if not math.isclose(energy,sum(retry.values())*50e-12,rel_tol=1e-10,abs_tol=1e-12): + raise ValueError('retry phase energy must match original50pJ/B once') + strength=option['profile']['transfer_strength'] + if strength==0 and sum(retry.values()):raise ValueError('null transfer emitted retry work') + summary=json.loads((point/'DONE.json').read_text())['summary'] + return generic,{'point_id':config['point_id'],'topology':config['topology'],'strength':strength, + 'status':'PASS_FIXED_INPUT_CAUSAL_REPLAY_AND_ENERGY','sampled_jobs':len(decisions), + 'windows':rows,'physical_read_bytes':dict(physical),'internal_retry_bytes':dict(retry), + 'retry_energy_j':energy,'total_energy_j':summary['energy_j'], + 'useful_bytes':summary['delivered_useful_bytes_by_stack'], + 'completed_simulated_tokens':summary['completed_tokens'],'peak_k_by_owner':peaks, + 'final_age_state':provider.snapshot()['states'],'temperature_series':{'time_s':times,'owners_k':dict(owner_series)},'scope':'CONDITIONAL_PROXY_NOT_HBF_MEASUREMENT_OR_POLICY_BENEFIT'} + + +def owner_plot(result,path): + import matplotlib + matplotlib.use('Agg') + from matplotlib import pyplot as plt + fig,axes=plt.subplots(3,1,figsize=(10,8),sharex=True) + data=result['temperature_series'] + for axis,prefix in zip(axes,('gpu','hbm','hbf')): + count=0 + for owner,values in data['owners_k'].items(): + if owner.startswith(prefix):axis.plot(data['time_s'],values,label=owner);count+=1 + axis.set_ylabel(prefix.upper()+' hotspot K');axis.grid(alpha=.2) + if count:axis.legend(ncol=4,fontsize=8) + else:axis.text(.1,.5,'UNAVAILABLE / NOT PRESENT',transform=axis.transAxes) + axes[-1].set_xlabel('Simulation time (s)') + fig.suptitle(result['point_id']+' — conditional proxy') + fig.tight_layout();fig.savefig(path,dpi=140);plt.close(fig) + + +def main(): + ap=argparse.ArgumentParser();ap.add_argument('--index',type=Path,required=True);ap.add_argument('--output',type=Path,required=True) + a=ap.parse_args();a.output.mkdir(parents=True,exist_ok=False) + index=json.loads(a.index.read_text());results=[];pairs=defaultdict(dict) + for p in index['points']: + path=Path(p['output']);generic,result=analyze(path) + (a.output/(p['point_id']+'.json')).write_text(json.dumps(result,indent=2)+'\n') + plot_panels(generic,a.output/(p['point_id']+'.png')) + owner_plot(result,a.output/(p['point_id']+'-owners.png')) + results.append(result);pairs[result['topology']][result['strength']]=result + comparisons=[] + for topology,pair in pairs.items(): + if set(pair)!={0,.1}:raise ValueError('incomplete paired topology') + null,weak=pair[0],pair[.1] + comparable=[] + for strength in (0,.1): + source=next(Path(p['output']) for p in index['points'] if p['point_id']==pair[strength]['point_id']) + cfg=json.loads((source/'config.json').read_text());cfg.pop('point_id') + cfg['hbf_read_cost_proxy']['profile'].pop('transfer_strength') + comparable.append(cfg) + if comparable[0]!=comparable[1]:raise ValueError('paired scientific inputs differ beyond proxy strength') + if sum(weak['internal_retry_bytes'].values())<=0:raise ValueError('enabled proxy has no retry consumer') + comparisons.append({'topology':topology, + 'null_useful_bytes':sum(null['useful_bytes'].values()),'weak_useful_bytes':sum(weak['useful_bytes'].values()), + 'null_tokens':null['completed_simulated_tokens'],'weak_tokens':weak['completed_simulated_tokens'], + 'null_peak_HBF_K':max(v for s,v in null['peak_k_by_owner'].items() if s.startswith('hbf')), + 'weak_peak_HBF_K':max(v for s,v in weak['peak_k_by_owner'].items() if s.startswith('hbf')), + 'weak_retry_energy_j':weak['retry_energy_j']}) + (a.output/'SUMMARY.json').write_text(json.dumps({'status':'PASS_PAIRED_INTEGRATION', + 'points':results,'pairs':comparisons,'policy_benefit':'NOT_TESTED'},indent=2)+'\n') + +if __name__=='__main__':main() diff --git a/experiments/eq3_system_thermal/analyze_extensions.py b/experiments/eq3_system_thermal/analyze_extensions.py new file mode 100644 index 0000000..4769e50 --- /dev/null +++ b/experiments/eq3_system_thermal/analyze_extensions.py @@ -0,0 +1,551 @@ +#!/usr/bin/env python3 +"""Validate and summarize maintenance/causal point receipts without rerunning them.""" +from __future__ import annotations + +import argparse +from collections import Counter, defaultdict +import hashlib +import json +import math +import os +from pathlib import Path +from typing import Any + + +STATE_RANK = {"normal": 0, "light": 1, "severe": 2, "shutdown": 3} +MEDIA_PHASES = {"media_read", "media_program", "media_erase", "media_fill"} + + +def _load(path: Path) -> dict: + value = json.loads(path.read_text(encoding="utf-8")) + if not isinstance(value, dict): + raise ValueError(f"{path} must contain a JSON object") + return value + + +def _rows(path: Path): + with path.open(encoding="utf-8") as stream: + for line_number, line in enumerate(stream, 1): + if not line.strip(): + continue + value = json.loads(line) + if not isinstance(value, dict): + raise ValueError(f"{path}:{line_number} must contain an object") + yield value + + +def _sha256(path: Path) -> str: + return hashlib.sha256(path.read_bytes()).hexdigest() + + +def _canonical_sha256(value: Any) -> str: + return hashlib.sha256(json.dumps( + value, sort_keys=True, separators=(",", ":"), allow_nan=False + ).encode()).hexdigest() + + +def _number(value: Any, label: str) -> float: + if isinstance(value, bool) or not isinstance(value, (int, float)) \ + or not math.isfinite(float(value)): + raise ValueError(f"{label} must be finite numeric") + return float(value) + + +def _nonnegative_int(value: Any, label: str) -> int: + if isinstance(value, bool) or not isinstance(value, int) or value < 0: + raise ValueError(f"{label} must be a nonnegative integer") + return value + + +def _interval(row: dict, label: str) -> tuple[int, int]: + start, end = row.get("start_ns"), row.get("end_ns") + if type(start) is not int or type(end) is not int or start < 0 or end <= start: + raise ValueError(f"{label} has invalid integer interval") + return start, end + + +def _identity(point: Path, config: dict, manifest: dict) -> dict: + input_hash = manifest.get("input_sha256") + if not isinstance(input_hash, str) or len(input_hash) != 64: + raise ValueError("manifest lacks input_sha256 identity") + saved_hash = _sha256(point / "config.json") + if saved_hash != input_hash: + raise ValueError("saved config differs from manifest input identity") + sources = manifest.get("source_sha256") + if not isinstance(sources, dict) or not sources \ + or any(not isinstance(v, str) or len(v) != 64 for v in sources.values()): + raise ValueError("manifest lacks complete source hashes") + thermal_hash = manifest.get("thermal_binary_sha256") + if not isinstance(thermal_hash, str) or len(thermal_hash) != 64: + raise ValueError("manifest lacks thermal binary identity") + trace_config = config.get("trace", {}) + trace_identity = None + if trace_config.get("dependency_mode") == "tiny_cpu_template": + checksum = trace_config.get("tiny_trace_sha256") + path = trace_config.get("tiny_trace_path") + if not isinstance(checksum, str) or len(checksum) != 64 or not isinstance(path, str): + raise ValueError("tiny trace consumer lacks bound path/checksum") + trace_identity = {"mode": "tiny_cpu_template", "path": path, + "trace_sha256": checksum} + return { + "point": str(point), "point_id": config.get("point_id"), + "runner_input_sha256": input_hash, "saved_config_sha256": saved_hash, + "saved_config_canonical_sha256": _canonical_sha256(config), + "source_revision": manifest.get("source_revision"), + "source_sha256": sources, "thermal_binary_sha256": thermal_hash, + "thermal_model_lock": manifest.get("thermal_model_lock"), + "trace_identity": trace_identity, + } + + +def _activities(row: dict, runner: str) -> list[dict]: + if runner == "causal": + values = row["service"].get("activities", []) + else: + receipts = [row["service"]] + if row.get("independent_maintenance_service") is not None: + receipts.append(row["independent_maintenance_service"]) + values = [item for receipt in receipts for item in receipt.get("activities", [])] + if not isinstance(values, list): + raise ValueError("service activities must be a list") + for item in values: + _nonnegative_int(item.get("bytes"), "activity bytes") + return values + + +def _temperature_groups(temperatures: dict) -> dict[str, float | None]: + groups = {"gpu": [], "hbf": [], "hbm": []} + for owner, value in temperatures.items(): + number = _number(value, f"temperature {owner}") + lower = owner.lower() + if lower == "gpu" or lower.startswith("gpu"): + groups["gpu"].append(number) + elif lower.startswith("hbf"): + groups["hbf"].append(number) + elif lower.startswith("hbm"): + groups["hbm"].append(number) + return {key: max(values) if values else None for key, values in groups.items()} + + +def _state_group(states: dict, prefix: str) -> int | None: + selected = [] + for owner, state in states.items(): + if owner.lower().startswith(prefix): + if state not in STATE_RANK: + raise ValueError("unknown controller state") + selected.append(STATE_RANK[state]) + return max(selected) if selected else None + + +def _driver_summary(final: dict | None, raw_terminal: list[dict]) -> dict: + if final is None: + return {"mode": "disabled", "capability": "NO_MAINTENANCE_FINAL_RECEIPT"} + driver = final["driver"] + reliability = final["reliability"] + wear = driver.get("physical_wear", {}) + blocks = reliability.get("blocks", {}) + statuses = Counter(row["status"] for row in raw_terminal) + terminal = driver.get("terminal_summary", {}) + if sum(statuses.values()) != terminal.get("operation_count", 0): + raise ValueError("raw maintenance terminal count differs from final driver summary") + if dict(sorted(statuses.items())) != terminal.get("status_counts", {}): + raise ValueError("raw maintenance terminal statuses differ from final driver summary") + if sum(len(row.get("extent_ids", ())) for row in raw_terminal) != \ + terminal.get("extent_count", 0): + raise ValueError("raw maintenance terminal extents differ from final driver summary") + free = driver.get("free_spares_by_stack_channel", {}) + return { + "mode": final.get("mode", "shared_exact_service"), + "terminal_status_counts": dict(sorted(statuses.items())), + "terminal_operation_count": terminal.get("operation_count", 0), + "terminal_extent_count": terminal.get("extent_count", 0), + "free_spares_by_stack_channel": { + stack: {channel: len(values) for channel, values in channels.items()} + for stack, channels in free.items()}, + "free_spares": sum(len(values) for channels in free.values() + for values in channels.values()), + "quarantined_blocks": len(driver.get("quarantined_blocks", [])), + "age": { + "block_count": len(blocks), + "min_equivalent_age_ns": min((row["equivalent_age_ns"] for row in blocks.values()), + default=None), + "max_equivalent_age_ns": max((row["equivalent_age_ns"] for row in blocks.values()), + default=None), + "successful_age_resets": sum(row.get("last_refresh_commit_ns") is not None + for row in blocks.values()), + }, + "wear": {key: sum(row.get(key, 0) for row in wear.values()) for key in ( + "block_program_work_started", "block_program_work_completed", + "nand_page_programs_started", "nand_page_programs_completed", + "erase_phase_started", "erase_completed")}, + "limitations": driver.get("limitations", []), + } + + +def analyze_point(point: Path) -> dict: + point = Path(point) + config_path, manifest_path = point / "config.json", point / "manifest.json" + done_path, failed_path = point / "DONE.json", point / "FAILED.json" + if done_path.exists() and failed_path.exists(): + raise ValueError("point has both DONE and FAILED terminal receipts") + if failed_path.exists() and (not config_path.is_file() or not manifest_path.is_file()): + return {"schema_version": "eq3-extension-analysis-v1", + "analysis_status": "FAILED_RUN_PRESERVED_NOT_ANALYZED", + "scientific_pass": "NOT_ASSESSED", + "identity": {"point": str(point), "availability": "PARTIAL_STARTUP_IDENTITY", + "config_present": config_path.is_file(), + "manifest_present": manifest_path.is_file()}, + "failure_receipt": _load(failed_path), "panels": None} + if not config_path.is_file() or not manifest_path.is_file(): + raise ValueError("point lacks config.json or manifest.json") + config, manifest = _load(config_path), _load(manifest_path) + identity = _identity(point, config, manifest) + if failed_path.exists(): + return {"schema_version": "eq3-extension-analysis-v1", + "analysis_status": "FAILED_RUN_PRESERVED_NOT_ANALYZED", + "scientific_pass": "NOT_ASSESSED", "identity": identity, + "failure_receipt": _load(failed_path), "panels": None} + if not done_path.exists(): + return {"schema_version": "eq3-extension-analysis-v1", + "analysis_status": "INCOMPLETE_NO_TERMINAL_RECEIPT", + "scientific_pass": "NOT_ASSESSED", "identity": identity, + "panels": None} + done = _load(done_path) + if done.get("status") != "COMPLETED": + raise ValueError("DONE receipt is not COMPLETED") + windows_path = point / "windows.jsonl" + if not windows_path.is_file(): + raise ValueError("completed point lacks windows.jsonl") + rows = _rows(windows_path) + try: + first = next(rows) + except StopIteration as error: + raise ValueError("completed point has no raw windows") from error + runner = "causal" if "executor" in first else "maintenance" + all_rows = [first, *rows] + active_ns = config.get("active_ns", config.get("workload", {}).get("active_ns", 0)) + recovery_ns = config.get("recovery_ns", 0) + total_end = _nonnegative_int(active_ns, "configured active_ns") + \ + _nonnegative_int(recovery_ns, "configured recovery_ns") + if total_end == 0: + raise ValueError("configured duration must be positive") + expected_start = 0 + times, physical_hbf, effective_hbf = [], [], [] + gpu_k, hbf_k, hbm_k = [], [], [] + hbf_state, hbm_state = [], [] + refresh, retry, migration = [], [], [] + backlog_series, queue_series, token_rate = [], [], [] + total_energy = 0.0 + prior_thermal_energy = 0.0 + completion_ids = set() + completion_group_hashes = set() + completion_count = 0 + uniqueness = "EXACT_JOB_IDS" + raw_terminal, causal_events = [], Counter() + causal_retry_count = 0 + offered_by_stack, delivered_by_stack = defaultdict(int), defaultdict(int) + last_backlog_by_stack = {} + cumulative_tokens = 0 + for index, row in enumerate(all_rows): + start, end = _interval(row, f"window {index}") + if start != expected_start: + raise ValueError("raw windows are not contiguous from zero") + expected_start = end + duration = end - start + times.append((start + end) / 2e9) + thermal = row.get("thermal") + if not isinstance(thermal, dict) or _interval(thermal, "thermal") != (start, end): + raise ValueError("thermal timeline differs from raw window") + energy = row.get("energy") + if not isinstance(energy, dict): + raise ValueError("window lacks energy receipt") + component = energy.get("component_energy_j") + if not isinstance(component, dict): + raise ValueError("energy receipt lacks component map") + component_values = [_number(value, "component energy") for value in component.values()] + if any(value < 0 for value in component_values): + raise ValueError("component energy must be nonnegative") + window_energy = sum(component_values) + if not math.isclose(window_energy, _number(energy.get("total_j"), "energy total"), + rel_tol=1e-12, abs_tol=1e-12): + raise ValueError("component energy does not sum to total_j") + total_energy += window_energy + cumulative = _number(thermal["energy_j"]["cumulative"]["total_input_j"], + "thermal cumulative energy") + if cumulative + 1e-12 < prior_thermal_energy \ + or not math.isclose(cumulative, total_energy, rel_tol=1e-10, abs_tol=1e-9): + raise ValueError("thermal cumulative energy differs from activity energy timeline") + prior_thermal_energy = cumulative + temperature = _temperature_groups(thermal.get("temperatures", {})) + gpu_k.append(temperature["gpu"]); hbf_k.append(temperature["hbf"]) + hbm_k.append(temperature["hbm"]) + states = row.get("control", {}).get("observed_states") + if not isinstance(states, dict): + raise ValueError("control receipt lacks observed states") + hbf_state.append(_state_group(states, "hbf")); hbm_state.append(_state_group(states, "hbm")) + activities = _activities(row, runner) + media_hbf_bytes = sum(item["bytes"] for item in activities + if str(item.get("stack", "")).startswith("hbf") + and item.get("phase") in MEDIA_PHASES) + physical_hbf.append(media_hbf_bytes * 1e9 / duration) + refresh_bytes = sum(item["bytes"] for item in activities + if item.get("operation") in {"refresh_read", "program", "erase"}) + retry_bytes = sum(item["bytes"] for item in activities + if item.get("operation") in {"retry", "retry_internal"} + and item.get("phase") == "media_read") + migration_bytes = sum(item["bytes"] for item in activities + if item.get("operation") in {"migration", "migration_program"}) + refresh.append(refresh_bytes * 1e9 / duration) + retry.append(retry_bytes * 1e9 / duration) + migration.append(migration_bytes * 1e9 / duration) + if runner == "maintenance": + service = row["service"] + effective = 0; backlog = 0; queue = 0 + for stack, receipt in service["stacks"].items(): + offered = _nonnegative_int(receipt["offered_effective_bytes"], "offered bytes") + delivered = _nonnegative_int(receipt["delivered_effective_bytes"], "delivered bytes") + remaining = _nonnegative_int(receipt["backlog_effective_bytes"], "backlog bytes") + if receipt["cumulative_offered_effective_bytes"] != \ + receipt["cumulative_delivered_effective_bytes"] + remaining: + raise ValueError("maintenance service byte conservation failed") + offered_by_stack[stack] += offered; delivered_by_stack[stack] += delivered + last_backlog_by_stack[stack] = remaining + if stack.startswith("hbf"): + effective += delivered; backlog += remaining + for receipt in (service, row.get("independent_maintenance_service")): + if receipt is None: + continue + for job_id in receipt.get("completion_ids", []): + if job_id in completion_ids: + raise ValueError("duplicate maintenance completion ID") + completion_ids.add(job_id); completion_count += 1 + queue += sum(item.get("remaining_bytes", 0) + for item in receipt.get("job_progress", [])) + delta = row.get("maintenance_delta") or {} + terminals = delta.get("terminal_results", []) + for terminal in terminals: + if any(existing["operation_id"] == terminal["operation_id"] + for existing in raw_terminal): + raise ValueError("duplicate maintenance terminal operation") + raw_terminal.append(terminal) + effective_hbf.append(effective * 1e9 / duration) + backlog_series.append(backlog); queue_series.append(queue); token_rate.append(None) + else: + facts = row["control"].get("stack_facts", {}) + effective = backlog = 0 + for stack, fact in facts.items(): + offered = _nonnegative_int(fact["offered_bytes"], "causal offered bytes") + delivered = _nonnegative_int(fact["delivered_bytes"], "causal delivered bytes") + remaining = _nonnegative_int(fact["backlog_bytes"], "causal backlog bytes") + offered_by_stack[stack] += offered; delivered_by_stack[stack] += delivered + last_backlog_by_stack[stack] = remaining + if stack.startswith("hbf"): + effective += delivered; backlog += remaining + progress = row["service"].get("changed_job_progress", []) + queue = sum(_nonnegative_int(item.get("remaining_bytes", 0), "job remaining") + for item in progress) + grouped = row.get("output_granularity", "").startswith("UNIFORM_") + for item in row["service"].get("completions", []): + if grouped: + group_hash = item.get("job_ids_sha256") + if not isinstance(group_hash, str) or len(group_hash) != 64: + raise ValueError("grouped causal completion lacks identity hash") + if group_hash in completion_group_hashes: + raise ValueError("duplicate grouped causal completion identity hash") + completion_group_hashes.add(group_hash) + completion_count += _nonnegative_int(item["count"], "completion count") + uniqueness = "AGGREGATED_JOB_ID_HASH_ONLY" + else: + job_id = item["job_id"] + if job_id in completion_ids: + raise ValueError("duplicate causal completion ID") + completion_ids.add(job_id); completion_count += 1 + events = row["executor"].get("events", []) + event_migration_bytes = 0 + for event in events: + causal_events[event["kind"]] += int(event.get("count", 1)) + causal_retry_count += int(event.get("retry_count", 0)) + if event["kind"].startswith("migration"): + event_migration_bytes += int(event.get("completed_bytes", + event.get("bytes", 0))) + if event_migration_bytes: + migration_bytes = event_migration_bytes + maintenance_row = row.get("maintenance", {}) + for delta in [*maintenance_row.get("receipt_deltas", []), + maintenance_row.get("driver", {})]: + for terminal in delta.get("terminal_results", []): + if any(existing["operation_id"] == terminal["operation_id"] + for existing in raw_terminal): + raise ValueError("duplicate causal maintenance terminal operation") + raw_terminal.append(terminal) + completed = _nonnegative_int(row["executor"]["completed_tokens"], "completed tokens") + cumulative = _nonnegative_int(row["executor"]["cumulative_completed_tokens"], + "cumulative tokens") + if cumulative != cumulative_tokens + completed: + raise ValueError("causal token cumulative timeline is inconsistent") + cumulative_tokens = cumulative + effective_hbf.append(effective * 1e9 / duration) + backlog_series.append(backlog); queue_series.append(queue) + token_rate.append(completed * 1e9 / duration) + migration[-1] = migration_bytes * 1e9 / duration + if expected_start != total_end: + raise ValueError("raw timeline does not cover configured duration") + summary = done.get("summary", {}) + if not math.isclose(total_energy, _number(summary.get("energy_j"), "summary energy"), + rel_tol=1e-10, abs_tol=1e-9): + raise ValueError("DONE energy differs from raw timeline") + hbf = [stack for stack in offered_by_stack if stack.startswith("hbf")] + if runner == "maintenance": + offered = sum(offered_by_stack[s] for s in hbf) + delivered = sum(delivered_by_stack[s] for s in hbf) + backlog = sum(last_backlog_by_stack[s] for s in hbf) + if (offered, delivered, backlog) != ( + summary.get("offered_bytes"), summary.get("delivered_bytes"), + summary.get("backlog_bytes")): + raise ValueError("maintenance DONE bytes differ from raw HBF timeline") + final_path = point / "maintenance-final.json" + final = None + if config.get("maintenance", {}).get("mode") != "disabled": + if not final_path.is_file() or summary.get("maintenance_final_sha256") != _sha256(final_path): + raise ValueError("maintenance final receipt identity mismatch") + final = _load(final_path) + maintenance = _driver_summary(final, raw_terminal) + tokens = {"availability": "UNAVAILABLE_RATE_WORKLOAD_HAS_NO_TOKEN_DEPENDENCY_DAG"} + consumers = {"availability": "NOT_APPLICABLE_RATE_WORKLOAD"} + else: + expected_offered = summary.get("offered_useful_bytes_by_stack", {}) + expected_delivered = summary.get("delivered_useful_bytes_by_stack", {}) + stacks = set(offered_by_stack) | set(expected_offered) | set(expected_delivered) + if ({stack: offered_by_stack[stack] for stack in stacks} != + {stack: expected_offered.get(stack, 0) for stack in stacks} or + {stack: delivered_by_stack[stack] for stack in stacks} != + {stack: expected_delivered.get(stack, 0) for stack in stacks}): + raise ValueError("causal DONE useful bytes differ from raw control facts") + if cumulative_tokens != summary.get("completed_tokens"): + raise ValueError("causal DONE token count differs from raw timeline") + trace_origin = summary.get("trace_origin") + if not isinstance(trace_origin, str) or not trace_origin: + raise ValueError("causal DONE lacks trace origin") + provenance = summary.get("structure_provenance") + if identity["trace_identity"] is not None: + if not isinstance(provenance, dict) or provenance.get("trace_sha256") != \ + identity["trace_identity"]["trace_sha256"]: + raise ValueError("causal DONE trace provenance differs from bound input") + if summary.get("pending_external_jobs") == 0 and summary.get("uninstantiated_batches") == 0 \ + and any(delivered_by_stack[s] != offered_by_stack[s] for s in offered_by_stack): + raise ValueError("closed causal point does not conserve useful bytes") + final_maintenance = summary.get("maintenance", {"mode": "disabled"}) + maintenance = (_driver_summary(final_maintenance, raw_terminal) + if final_maintenance.get("mode") != "disabled" else + {"mode": "disabled"}) + tokens = {"availability": "ACTUAL_DAG_TERMINAL_COMPLETIONS", + "completed_tokens": cumulative_tokens, + "average_tokens_per_s": cumulative_tokens * 1e9 / total_end} + identity["trace_result"] = { + "trace_origin": trace_origin, "structure_provenance": provenance} + consumers = { + "cache": {key: causal_events[key] for key in sorted(causal_events) + if key.startswith("cache_")}, + "prefetch": {key: causal_events[key] for key in sorted(causal_events) + if key.startswith("prefetch_")}, + "migration": {key: causal_events[key] for key in sorted(causal_events) + if key.startswith("migration_")}, + "retry": { + "observed_retry_count": ( + None if config.get("hbf_read_cost_proxy", {}).get("mode") + == "conditional_nand_history_v1" else causal_retry_count), + "count_semantics": ( + "UNKNOWN_INTEGER_COUNT_EXPECTED_WORK_PROXY" + if config.get("hbf_read_cost_proxy", {}).get("mode") + == "conditional_nand_history_v1" + else "RECORDED_EXECUTOR_RETRY_EVENTS"), + }, + "semantics": "ACTUAL_RECORDED_CONSUMER_EVENTS_ZERO_MEANS_NOT_OBSERVED_IN_POINT", + } + return { + "schema_version": "eq3-extension-analysis-v1", "runner": runner, + "analysis_status": "VALIDATED_COMPLETE_RECEIPTS", + "scientific_pass": "NOT_ASSESSED", + "identity": identity, + "checks": {"timeline": "PASS", "byte_conservation": "PASS", + "energy_to_thermal": "PASS", "terminal_uniqueness": uniqueness, + "completion_count": completion_count}, + "maintenance": maintenance, "causal_tokens": tokens, + "causal_consumers": consumers, + "panels": { + "time_s": times, + "physical_effective_hbf_traffic": { + "physical_media_Bps": physical_hbf, "effective_useful_Bps": effective_hbf}, + "owner_temperature_k": {"gpu": gpu_k, "hbm_max": hbm_k, "hbf_max": hbf_k}, + "controller_state_rank": {"labels": STATE_RANK, "hbm_worst": hbm_state, + "hbf_worst": hbf_state}, + "maintenance_retry_migration_Bps": {"refresh": refresh, "retry": retry, + "migration": migration}, + "queue_backlog_bytes": {"logical_backlog": backlog_series, + "changed_external_remaining": queue_series}, + "causal_token_rate": {"tokens_per_s": token_rate, + "maintenance_semantics": ( + "UNAVAILABLE" if runner == "maintenance" else "NOT_APPLICABLE")}, + }, + "interpretation_limits": [ + "VALIDATED_COMPLETE_RECEIPTS_IS_NOT_A_SCIENTIFIC_MODEL_PASS", + "PHYSICAL_TRAFFIC_IS_RECORDED_MODEL_MEDIA_ACTIVITY_NOT_NATIVE_NAND_WIRE_BYTES", + "CHANGED_EXTERNAL_REMAINING_IS_A_DELTA_DIAGNOSTIC_NOT_ALWAYS_TOTAL_QUEUE", + ], + } + + +def plot_panels(analysis: dict, output: Path) -> None: + if analysis.get("analysis_status") != "VALIDATED_COMPLETE_RECEIPTS": + raise ValueError("plots require validated complete receipts") + os.environ.setdefault("MPLCONFIGDIR", "/tmp/eq3-matplotlib-cache") + import matplotlib + matplotlib.use("Agg") + import matplotlib.pyplot as plt + p, time = analysis["panels"], analysis["panels"]["time_s"] + fig, axes = plt.subplots(6, 1, figsize=(11, 15), sharex=True) + for name, values in p["physical_effective_hbf_traffic"].items(): axes[0].plot(time, values, label=name) + for name, values in p["owner_temperature_k"].items(): + if any(value is not None for value in values): axes[1].plot(time, values, label=name) + for name in ("hbf_worst", "hbm_worst"): + values = p["controller_state_rank"][name] + if any(value is not None for value in values): axes[2].step(time, values, where="mid", label=name) + for name, values in p["maintenance_retry_migration_Bps"].items(): axes[3].plot(time, values, label=name) + for name, values in p["queue_backlog_bytes"].items(): axes[4].plot(time, values, label=name) + tokens = p["causal_token_rate"]["tokens_per_s"] + if any(value is not None for value in tokens): axes[5].plot(time, tokens, label="causal tokens/s") + else: axes[5].text(.5, .5, "UNAVAILABLE for rate workload", ha="center", transform=axes[5].transAxes) + labels = ("HBF traffic (B/s)", "Owner temperature (K)", "Controller state rank", + "Refresh/retry/migration (B/s)", "Queue/backlog (bytes)", "Causal token rate") + for axis, label in zip(axes, labels): + axis.set_ylabel(label); axis.grid(True, alpha=.25) + if axis.lines: axis.legend(loc="best", fontsize=8) + axes[-1].set_xlabel("Simulation time (s)") + fig.tight_layout(); fig.savefig(output, dpi=160); plt.close(fig) + + +def main() -> None: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--point", type=Path, required=True) + parser.add_argument("--output", type=Path, required=True) + args = parser.parse_args() + args.output.mkdir(parents=True, exist_ok=False) + try: + result = analyze_point(args.point) + (args.output / "analysis.json").write_text( + json.dumps(result, indent=2, sort_keys=True, allow_nan=False) + "\n") + if result["analysis_status"] == "VALIDATED_COMPLETE_RECEIPTS": + plot_panels(result, args.output / "six-panel.png") + terminal = "DONE.json" + except BaseException as error: + (args.output / "ANALYSIS_FAILED.json").write_text(json.dumps({ + "status": "ANALYSIS_FAILED", "error": repr(error), + "source_point_preserved": True}, indent=2) + "\n") + raise + (args.output / terminal).write_text(json.dumps({ + "status": "COMPLETED", "analysis_status": result["analysis_status"], + "scientific_pass": "NOT_ASSESSED"}, indent=2) + "\n") + + +if __name__ == "__main__": + main() diff --git a/experiments/eq3_system_thermal/analyze_uncontrolled_first_constraint.py b/experiments/eq3_system_thermal/analyze_uncontrolled_first_constraint.py new file mode 100644 index 0000000..936cba3 --- /dev/null +++ b/experiments/eq3_system_thermal/analyze_uncontrolled_first_constraint.py @@ -0,0 +1,192 @@ +#!/usr/bin/env python3 +"""Strict analysis of a frozen uncontrolled first-constraint subset.""" + +from __future__ import annotations + +import argparse +from collections import Counter +import hashlib +import json +from pathlib import Path + + +WINDOW_NS = 20_000_000 +LABELS = ("light", "severe", "shutdown") + + +def digest(path): + return hashlib.sha256(Path(path).read_bytes()).hexdigest() + + +def last_error_response(path): + for line in reversed(Path(path).read_text().splitlines()): + value = json.loads(line) + response = value.get("response") + if isinstance(response, dict) and response.get("type") == "ERROR": + return response + raise ValueError("thermal transcript has no ERROR response") + + +def first_crossings(rows, limits): + first = {} + for row in rows: + end = row["end_ns"] + for owner, temperature in row["thermal"]["temperatures"].items(): + kind = "gpu" if owner == "gpu" else owner[:3] + for label, threshold in zip(LABELS, limits[kind]): + if temperature >= threshold: + first.setdefault((label, owner), end) + result = {} + for label in LABELS: + observed = [(time, owner) for (kind, owner), time in first.items() if kind == label] + if observed: + time = min(time for time, _ in observed) + result[label] = {"status": "CROSSED", "end_ns": time, + "interval_ns": {"lower_exclusive": max(0, time-WINDOW_NS), + "upper_inclusive": time}, + "tied_owners": sorted(owner for value, owner in observed + if value == time)} + else: + result[label] = None + return result + + +def analyze(index_path: Path, output: Path): + if output.exists(): + raise FileExistsError(output) + index = json.loads(index_path.read_text()) + if index["point_count"] != 18 or any( + row["execution_mode"] != "uncontrolled_first_constraint" + for row in index["points"]): + raise ValueError("index is not the frozen 18-point uncontrolled subset") + stage = index_path.parent + receipt = json.loads((stage / "DONE.json").read_text()) + completed_ids = {row["point_id"] for row in receipt["completed"]} + domain_ids = {row["point_id"] for row in receipt["domain_failures"]} + if len(completed_ids) + len(domain_ids) != 18 or completed_ids & domain_ids: + raise ValueError("stage receipt coverage mismatch") + + points = [] + for spec in index["points"]: + point = Path(spec["output"]) + config = json.loads(Path(spec["config"]).read_text()) + if digest(spec["config"]) != spec["config_sha256"]: + raise ValueError("config hash mismatch: " + spec["point_id"]) + manifest = json.loads((point / "manifest.json").read_text()) + if manifest["input_sha256"] != spec["config_sha256"]: + raise ValueError("manifest input mismatch: " + spec["point_id"]) + if not config.get("control_disabled"): + raise ValueError("controlled point in uncontrolled analysis") + rows, previous_end = [], 0 + with (point / "windows.jsonl").open() as stream: + for line in stream: + row = json.loads(line) + if row["start_ns"] != previous_end or row["end_ns"]-row["start_ns"] != WINDOW_NS: + raise ValueError("non-contiguous window: " + spec["point_id"]) + previous_end = row["end_ns"] + if any(not value["cumulative_conserved"] + for value in row["service"]["stacks"].values()): + raise ValueError("byte conservation failure: " + spec["point_id"]) + expected = row["energy"]["total_j"] + actual = row["thermal"]["energy_j"]["window"]["total_input_j"] + if abs(expected-actual) > 1e-9*max(1, abs(expected)): + raise ValueError("energy conservation failure: " + spec["point_id"]) + rows.append(row) + if not rows: + raise ValueError("point has no completed thermal windows") + status = "COMPLETED" if spec["point_id"] in completed_ids else "DOMAIN_FAILURE" + failure = None + if status == "COMPLETED": + if not (point / "DONE.json").is_file() or (point / "FAILED.json").exists(): + raise ValueError("completed point terminal evidence mismatch") + else: + if not (point / "FAILED.json").is_file(): + raise ValueError("domain failure lacks FAILED evidence") + last = last_error_response(point / "thermal-process" / "thermal-transcript.jsonl") + if last.get("status") != "DOMAIN_FAILURE" or not last.get("failure_returned_to_caller"): + raise ValueError("failure is not a preserved DOMAIN_FAILURE") + if last["last_valid_time_ns"] != previous_end: + raise ValueError("last-valid/raw coverage mismatch") + failure = {key: last.get(key) for key in + ("status", "reason", "last_valid_time_ns", "trial_target_time_ns", + "last_valid_temperature_range_k", "trial_temperature_range_k", + "trial_declared_activity_energy_j", "failed_trial_step_energy")} + crossings = first_crossings(rows, config["thermal_limits_k"]) + for label in LABELS: + if crossings[label] is None: + crossings[label] = {"status": ("NO_CROSSING_COMPLETED" if status == "COMPLETED" + else "RIGHT_CENSORED_DOMAIN_FAILURE")} + points.append({"point_id": spec["point_id"], "topology": spec["topology"], + "parameter_values": spec["sensitivity_values"], "status": status, + "completed_window_count": len(rows), "last_valid_time_ns": previous_end, + "crossings": crossings, "failure": failure, + "tested_parameter_set_weight": {"numerator": 1, "denominator": 18}}) + + summaries = {} + for label in LABELS: + states = Counter(point["crossings"][label]["status"] for point in points) + summaries[label] = {"counts": dict(sorted(states.items())), + "tested_parameter_set_denominator": 18, + "crossed_proportion": states["CROSSED"] / 18, + "tie_point_count": sum(len(point["crossings"][label].get("tied_owners", ())) > 1 + for point in points)} + result = {"schema_version": "eq3-uncontrolled-first-constraint-analysis-v1", + "status": "PASS", "index": str(index_path.resolve()), + "index_sha256": digest(index_path), "point_count": 18, + "terminal_counts": dict(sorted(Counter(p["status"] for p in points).items())), + "crossing_summaries": summaries, "points": points, + "claim_scope": ("PROPORTIONS_OF_THE_18_TESTED_PARAMETER_SETS_ONLY; " + "NO_POPULATION_OR_POLICY_BENEFIT_INFERENCE"), + "limitations": ["20MS_CROSSING_INTERVAL", "DOMAIN_FAILURE_RIGHT_CENSORING", + "CONDITIONAL_ENGINEERING_INPUTS", "P2_REFERENCE_NOT_QUALIFIED", + "NO_ECC_RETRY_OR_MAINTENANCE_BENEFIT_CLAIM"]} + failed = [point for point in points if point["status"] == "DOMAIN_FAILURE"] + result["domain_failure_summary"] = { + "count": len(failed), + "reasons": dict(sorted(Counter(point["failure"]["reason"] for point in failed).items())), + "failed_trial_energy_statuses": dict(sorted(Counter( + point["failure"]["failed_trial_step_energy"]["status"] for point in failed).items())), + "configured_gpu_external_w_counts": dict(sorted(Counter( + str(point["parameter_values"]["gpu_external_w"]) for point in failed).items())), + "interpretation": ("Every crossing reported for a failed point occurred in a preserved completed " + "window before failure. The trajectory after the last-valid frame is unobserved."), + } + output.mkdir(parents=True) + (output / "UNCONTROLLED_FIRST_CONSTRAINT_ANALYSIS.json").write_text( + json.dumps(result, indent=2, sort_keys=True, allow_nan=False) + "\n") + lines = ["# Uncontrolled first-constraint analysis", "", + "Scope: the 18 tested parameter sets only; no policy, ECC/retry, maintenance, or population inference.", "", + f"Terminal results: {result['terminal_counts']}.", ""] + for label in LABELS: + row = summaries[label] + lines.append(f"- {label}: {row['counts']}; crossed proportion {row['crossed_proportion']:.3f} of 18; tie points {row['tie_point_count']}.") + lines += ["", "All 16 failures explicitly report `temperature left required domain`; the failed trial step was not integrated and its energy terms remain UNKNOWN. Thirteen of those 16 configs prescribe 0 W external GPU heat, one prescribes 100 W, and two prescribe 200 W. The failure is therefore the registered 400 K domain response under uncontrolled conditional heat inputs, not a GPU-only failure.", + "", "Each crossing is localized only to `(previous frame, reported frame]`, a 20 ms interval. Every reported failed-point crossing occurred in a preserved pre-failure window. A missing crossing in a DOMAIN_FAILURE point is right-censored, and no post-failure trajectory or later constraint ordering is inferred.", "", + "| Point | Terminal | Light s / owners | Severe s / owners | Shutdown s / owners | Last valid s |", + "|---|---|---|---|---|---:|"] + def cell(crossing): + if crossing["status"] != "CROSSED": + return crossing["status"] + return f"{crossing['end_ns']/1e9:.3f} / {','.join(crossing['tied_owners'])}" + for point in points: + lines.append("| {point_id} | {status} | {light} | {severe} | {shutdown} | {last:.3f} |".format( + point_id=point["point_id"], status=point["status"], + light=cell(point["crossings"]["light"]), + severe=cell(point["crossings"]["severe"]), + shutdown=cell(point["crossings"]["shutdown"]), + last=point["last_valid_time_ns"]/1e9)) + (output / "RESULT.md").write_text("\n".join(lines) + "\n") + return result + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--index", type=Path, required=True) + parser.add_argument("--output", type=Path, required=True) + args = parser.parse_args() + result = analyze(args.index.resolve(strict=True), args.output.resolve()) + print(json.dumps({"status": result["status"], "point_count": result["point_count"]})) + + +if __name__ == "__main__": + main() diff --git a/experiments/eq3_system_thermal/causal_maintenance_age.py b/experiments/eq3_system_thermal/causal_maintenance_age.py new file mode 100644 index 0000000..778fc5c --- /dev/null +++ b/experiments/eq3_system_thermal/causal_maintenance_age.py @@ -0,0 +1,106 @@ +"""Exact-completion age bridge for maintenance on CausalTopologyService. + +Temperatures are known only at thermal-window boundaries. The previously +observed per-die temperature is therefore applied piecewise-constantly through +the current window. Maintenance completions advance only their affected +extents before CAS commit; all other extents catch up at the window boundary. +""" +from __future__ import annotations + +from collections import defaultdict +from typing import Mapping + + +class CausalMaintenanceAgeAdapter: + def __init__(self, ledger, driver, previous_temperature_k_by_stack_channel: Mapping): + self.ledger = ledger + self.driver = driver + self._temperature = self._normalize_temperatures( + previous_temperature_k_by_stack_channel) + self._validate_coverage() + + @staticmethod + def _normalize_temperatures(raw: Mapping) -> dict[tuple[str, str], float]: + result = {} + if not isinstance(raw, Mapping): + raise ValueError("temperature map must be a mapping") + for stack, channels in raw.items(): + if not isinstance(stack, str) or not isinstance(channels, Mapping): + raise ValueError("temperature map must be stack -> channel -> kelvin") + for channel, value in channels.items(): + temperature = float(value) + if not 0 < temperature < float("inf"): + raise ValueError("temperature must be finite and positive") + result[(stack, str(channel))] = temperature + return result + + def _validate_coverage(self) -> None: + required = {(row["stack"], row["channel"]) + for row in self.driver.extents.values()} + if not required.issubset(self._temperature): + raise ValueError("temperature map does not cover every maintenance extent") + + def _advance_extents(self, extent_ids, end_ns: int) -> None: + groups = defaultdict(list) + for extent_id in sorted(set(extent_ids)): + row = self.driver.extents[extent_id] + start_ns = self.ledger.accounted_through_ns(extent_id) + if end_ns < start_ns: + raise ValueError("completion precedes extent age frontier") + groups[(start_ns, self._temperature[(row["stack"], row["channel"])] )].append( + extent_id) + for (start_ns, temperature), identities in sorted(groups.items()): + self.ledger.advance_temperature_many( + identities, start_ns, end_ns, temperature) + + def start_window(self, start_ns: int) -> list[dict]: + """Discover due work only after every extent is aligned to the boundary.""" + if any(self.ledger.accounted_through_ns(extent_id) != start_ns + for extent_id in self.driver.extents): + raise ValueError("all extents must be age-aligned at window start") + return self.driver.poll(start_ns, discover_due=True) + + def consume_receipt(self, receipt: dict, *, failed_job_ids=()) -> dict: + """Apply exact maintenance completions and return delta plus next phases.""" + progress = {row["job_id"]: row for row in receipt.get("job_progress", [])} + affected_by_time = defaultdict(list) + for job_id in receipt.get("completion_ids", []): + if job_id not in self.driver.outstanding: + continue + row = progress.get(job_id) + if row is None or not isinstance(row.get("completion_ns"), int): + raise ValueError("maintenance completion lacks exact progress timestamp") + affected_by_time[row["completion_ns"]].extend( + row.get("metadata", {}).get("extent_ids", ())) + for completion_ns, extent_ids in sorted(affected_by_time.items()): + self._advance_extents(extent_ids, completion_ns) + delta = self.driver.consume_receipt(receipt, failed_job_ids=failed_job_ids) + delta["reliability_events"] = self.ledger.drain_events() + delta["next_phase_jobs"] = self.driver.poll( + receipt["end_ns"], discover_due=False) + return delta + + def finish_window(self, end_ns: int, + observed_temperature_k_by_stack_channel: Mapping) -> dict: + """Catch every extent up with prior temperature, then install new facts.""" + self._advance_extents(self.driver.extents, end_ns) + events = self.ledger.drain_events() + next_temperature = self._normalize_temperatures( + observed_temperature_k_by_stack_channel) + prior = self._temperature + self._temperature = next_temperature + try: + self._validate_coverage() + except Exception: + self._temperature = prior + raise + return { + "end_ns": end_ns, + "reliability_events": events, + "temperature_semantics": ( + "PREVIOUS_KNOWN_WINDOW_TEMPERATURE_PIECEWISE_CONSTANT_NO_RETROACTIVE_REAGE" + ), + } + + +__all__ = ["CausalMaintenanceAgeAdapter"] diff --git a/experiments/eq3_system_thermal/causal_service.py b/experiments/eq3_system_thermal/causal_service.py new file mode 100644 index 0000000..1926712 --- /dev/null +++ b/experiments/eq3_system_thermal/causal_service.py @@ -0,0 +1,684 @@ +#!/usr/bin/env python3 +"""Exact-event aggregate fluid service for causal batch consumers. + +Control state and endpoint byte quotas are frozen at thermal-window boundaries, +while storage completions occur at integer-nanosecond event horizons inside the +window. Shared bandwidth uses exact Fraction max-min rates. Two-bank +occupancy is finite, though buffer turnover remains a continuous-fluid +approximation rather than a page transaction model. +""" + +from __future__ import annotations + +from collections import deque +import copy +from dataclasses import dataclass +from fractions import Fraction +import heapq +from typing import Deque, Dict, Mapping, Optional, Sequence, Tuple + +from topology_service import ( + MAINTENANCE_OPERATIONS, + MAINTENANCE_POLICY, + OPERATIONS, + READ_OPERATIONS, + STATES, + _integer, + _normalize, + TopologyService, +) + + +NS_PER_SECOND = 1_000_000_000 +CAUSAL_OPERATIONS = set(OPERATIONS) | {"hbm_fill", "migration_program"} + + +def _ceil_fraction(value: Fraction) -> int: + return (value.numerator + value.denominator - 1) // value.denominator + + +@dataclass +class _CausalJob: + job_id: str + stack: str + channel: str + operation: str + route: Optional[str] + arrival_ns: int + total_bytes: int + completed_bytes: int + unadmitted_bytes: int + sequence: int + maintenance_id: Optional[str] + metadata: dict + state: str = "QUEUED" + slice_total_bytes: int = 0 + slice_transferred_bytes: int = 0 + rate_credit: Fraction = Fraction(0, 1) + first_service_ns: Optional[int] = None + last_service_ns: Optional[int] = None + completion_ns: Optional[int] = None + + +class CausalTopologyService: + """Event-driven v2 service sharing the v1 topology/configuration contract.""" + + schema_version = "eq3-causal-topology-service-v2" + + def __init__(self, config: dict): + raw_config = copy.deepcopy(config) + self._channel_groups = raw_config.pop("causal_channel_groups", None) + self._config = _normalize(raw_config) + # Reuse the reviewed route/resource/activity rules without sharing its + # mutable rate-window state. + self._rules = TopologyService(raw_config) + if self._channel_groups is not None: + if set(self._channel_groups) != set(self._config["channels"]): + raise ValueError("causal_channel_groups must exactly cover stacks") + for stack, channels in self._config["channels"].items(): + groups = self._channel_groups[stack] + if set(groups) != set(channels): + raise ValueError("causal_channel_groups must exactly cover channels") + for channel, row in groups.items(): + if (not isinstance(row, dict) or set(row) != { + "resource_id", "bandwidth_bytes_per_s"}): + raise ValueError("invalid causal channel group row") + if (not isinstance(row["resource_id"], str) or not row["resource_id"] + or type(row["bandwidth_bytes_per_s"]) is not int + or row["bandwidth_bytes_per_s"] <= 0): + raise ValueError("invalid causal channel group identity/rate") + self._now = 0 + self._window_start: Optional[int] = None + self._window_end: Optional[int] = None + self._states: Dict[str, str] = {} + self._quota_remaining: Dict[str, int] = {} + self._sequence = 0 + self._jobs: Dict[str, _CausalJob] = {} + self._known_ids = set() + self._queues: Dict[Tuple[str, str], Deque[str]] = { + (stack, channel): deque() + for stack, channels in self._config["channels"].items() + for channel in channels + } + self._active = set() + self._draining: list[Tuple[int, int, str, int]] = [] + self._window_activities: list[dict] = [] + self._window_resource_work: Dict[str, int] = {} + self._window_completion_ids: list[str] = [] + self._window_maintenance_ids: list[str] = [] + self._completed_since_read: Dict[str, int] = {} + self._activity_offset = 0 + self._completion_offset = 0 + self._maintenance_offset = 0 + + @property + def now_ns(self) -> int: + return self._now + + def immutable_facts(self) -> dict: + facts = self._rules.immutable_facts() + return { + "schema_version": "eq3-causal-topology-facts-v2", + "config": facts["config"], + "control": "FROZEN_PER_THERMAL_WINDOW_FUTURE_ONLY", + "bandwidth": "EXACT_FRACTION_PROGRESSIVE_MAX_MIN", + "endpoint_quota": "INTEGER_BYTES_RESERVED_ON_SLICE_ADMISSION_ONCE", + "buffer": "FINITE_TWO_BANK_ANALYTICAL_OCCUPANCY_BOUND_CONTINUOUS_TURNOVER", + "pipeline": "SHARED_BANDWIDTH_EXACT_PROPAGATION_LATENCY_APPROXIMATE", + "completion": "INTEGER_NS_EVENT_NO_THERMAL_WINDOW_FLOOR", + "completion_query": "NONE_UNTIL_NEXT_STATE_CHANGING_EVENT_IS_TERMINAL_USE_NEXT_EVENT_NS", + "backend_latency": "UNKNOWN_FLUID_MODEL", + "maintenance_guard_policy": MAINTENANCE_POLICY, + "causal_operations": sorted(CAUSAL_OPERATIONS), + "hbm_fill": "HBM_MEDIA_PLUS_SHARED_HALF_DUPLEX_GPU_LINK_ENGINEERING_PROXY", + "migration_program": ( + "HBF_PROGRAM_MEDIA_WORK_PLUS_REVERSE_DESTINATION_FABRIC_PATH_ENGINEERING_PROXY" + ), + "causal_channel_groups": copy.deepcopy(self._channel_groups), + } + + def begin_window(self, start_ns: int, end_ns: int, + budgets: Mapping[str, int], + endpoint_states: Mapping[str, str]) -> dict: + start = _integer(start_ns, "start_ns") + end = _integer(end_ns, "end_ns") + if start != self._now or end <= start: + raise ValueError("control window must begin at current service time") + if self._window_end is not None and self._now != self._window_end: + raise ValueError("previous control window has not reached its boundary") + stacks = set(self._config["channels"]) + if set(budgets) != stacks or set(endpoint_states) != stacks: + raise ValueError("budgets and endpoint_states must exactly cover stacks") + if any(endpoint_states[stack] not in STATES for stack in stacks): + raise ValueError("unknown endpoint state") + scale = self._config["work_scale"] + self._window_start, self._window_end = start, end + self._states = dict(endpoint_states) + self._quota_remaining = { + stack: _integer(budgets[stack], f"budgets.{stack}") * scale + for stack in stacks + } + self._window_activities = [] + self._window_resource_work = {} + self._window_completion_ids = [] + self._window_maintenance_ids = [] + self._activity_offset = self._completion_offset = self._maintenance_offset = 0 + self._activate() + return { + "start_ns": start, "end_ns": end, + "budgets": dict(budgets), "endpoint_states": dict(endpoint_states), + "semantics": "FROZEN_FUTURE_CONTROL_NO_REEVALUATION_OF_ACTIVE_SLICES", + } + + def _default_route(self, stack: str, channel: str) -> Optional[str]: + return self._rules._default_route(stack, channel) + + def _endpoints(self, job: _CausalJob) -> Tuple[str, ...]: + return self._rules._endpoints(job) + + def _phase_resources(self, job: _CausalJob) -> list[Tuple[str, dict, int]]: + if job.operation == "hbm_fill": + proxy = copy.copy(job) + proxy.operation = "read" + proxy.route = "direct" + phases = self._rules._phase_resources(proxy) + elif job.operation == "migration_program": + read_proxy = copy.copy(job) + read_proxy.operation = "read" + program_proxy = copy.copy(job) + program_proxy.operation = "program" + media = self._rules._phase_resources(program_proxy)[0] + phases = [media] + self._rules._phase_resources(read_proxy)[1:] + else: + phases = self._rules._phase_resources(job) + if self._channel_groups is not None: + group = self._channel_groups[job.stack][job.channel] + _, _, coefficient = phases[0] + phases[0] = ( + group["resource_id"], + {"latency_ns": 0, + "bandwidth_bytes_per_s": group["bandwidth_bytes_per_s"]}, + coefficient, + ) + return phases + + def _buffer_stacks(self, job: _CausalJob) -> Tuple[str, ...]: + if job.operation not in READ_OPERATIONS and job.operation not in { + "hbm_fill", "migration_program", + }: + return () + return tuple(row[0] for row in self._rules._buffer_specs(job)) + + def _buffer_capacity(self, job: _CausalJob) -> int: + """Maximum payload resident in each reserved ping-pong bank.""" + specs = self._rules._buffer_specs(job) + return min((capacity for _, _, capacity in specs), default=job.unadmitted_bytes) + + def _pipeline_latency_ns(self, job: _CausalJob) -> int: + return sum(stage["latency_ns"] for _, stage, _ in self._phase_resources(job)) + + def _slice_drain_latency_ns(self, job: _CausalJob) -> int: + # Shared rates already model overlapped stage throughput. Finite-bank + # turnover therefore pays propagation on the final slice only. + return self._pipeline_latency_ns(job) if job.unadmitted_bytes == 0 else 0 + + def _class(self, job: _CausalJob) -> str: + maintenance = job.operation in MAINTENANCE_OPERATIONS or job.maintenance_id is not None + return f"maintenance:{job.operation}" if maintenance else f"foreground:{job.route}" + + def submit_jobs(self, jobs: Sequence[dict]) -> dict: + if self._window_end is None: + raise RuntimeError("begin_window is required before submit_jobs") + accepted = [] + for raw in jobs: + required = {"job_id", "stack", "channel", "operation", "bytes", "arrival_ns"} + optional = {"route", "maintenance_id", "metadata"} + if not isinstance(raw, dict) or not required.issubset(raw) or set(raw) - required - optional: + raise ValueError("job has missing or unknown fields") + job_id = raw["job_id"] + if not isinstance(job_id, str) or not job_id or job_id in self._known_ids: + raise ValueError("job_id must be globally unique and nonempty") + stack, channel = raw["stack"], raw["channel"] + if stack not in self._config["channels"] or channel not in self._config["channels"][stack]: + raise ValueError("job names unknown stack/channel") + operation = raw["operation"] + if operation not in CAUSAL_OPERATIONS: + raise ValueError("unsupported operation") + arrival = _integer(raw["arrival_ns"], "arrival_ns") + if arrival < self._now or arrival >= self._window_end: + raise ValueError("job arrival must lie between current time and control-window end") + route = raw.get("route", self._default_route(stack, channel)) + if operation == "hbm_fill": + if stack not in self._config["fabric"]["hbm"]: + raise ValueError("hbm_fill requires an HBM destination stack") + route = None + elif operation == "migration_program": + if stack not in self._config["fabric"]["hbf"]: + raise ValueError("migration_program requires an HBF destination stack") + route = self._rules._validate_route(stack, channel, route) + else: + route = self._rules._validate_route(stack, channel, route) + if operation not in READ_OPERATIONS and operation not in { + "hbm_fill", "migration_program", + }: + route = None + metadata = raw.get("metadata", {}) + if not isinstance(metadata, dict): + raise ValueError("metadata must be an object") + maintenance_id = raw.get("maintenance_id") + if maintenance_id is not None and (not isinstance(maintenance_id, str) or not maintenance_id): + raise ValueError("maintenance_id must be a nonempty string") + byte_count = _integer(raw["bytes"], "bytes", positive=True) + job = _CausalJob( + job_id=job_id, stack=stack, channel=channel, operation=operation, + route=route, arrival_ns=arrival, total_bytes=byte_count, + completed_bytes=0, unadmitted_bytes=byte_count, + sequence=self._sequence, maintenance_id=maintenance_id, + metadata=copy.deepcopy(metadata), + ) + self._sequence += 1 + self._known_ids.add(job_id) + self._jobs[job_id] = job + self._queues[(stack, channel)].append(job_id) + accepted.append(job_id) + self._activate() + return {"accepted_job_ids": accepted, "time_ns": self._now} + + def _eligible_heads(self) -> list[_CausalJob]: + result = [] + for key in sorted(self._queues): + queue = self._queues[key] + if queue: + self._queues[key] = queue = deque( + job_id for job_id in queue if self._jobs[job_id].state != "DONE" + ) + first = {} + for job_id in queue: + job = self._jobs[job_id] + if job.state == "QUEUED": + group = self._class(job) + prior = first.get(group) + if prior is None or (job.arrival_ns, job.sequence) < (prior.arrival_ns, prior.sequence): + first[group] = job + result.extend(first.values()) + return sorted(result, key=lambda job: job.sequence) + + def _state_allows(self, job: _CausalJob) -> bool: + maintenance = job.operation in MAINTENANCE_OPERATIONS or job.maintenance_id is not None + states = [self._states[endpoint] for endpoint in self._endpoints(job)] + if maintenance: + return "shutdown" not in states + return not any(state in {"severe", "shutdown"} for state in states) + + def _buffers_available(self, job: _CausalJob) -> bool: + # Continuous aggregate turnover does not bind one whole bank to one + # channel flow. Occupancy remains bounded analytically in receipts. + return True + + def _reserve_buffers(self, job: _CausalJob) -> None: + pass + + def _release_buffers(self, job: _CausalJob) -> None: + pass + + def _buffer_bounds(self) -> dict: + bounds = {} + fabric = self._config["fabric"] + for stack in self._config["channels"]: + row = fabric["hbf"].get(stack) or fabric["hbm"].get(stack) + if row is None: + continue + active = [job for job in self._jobs.values() + if job.state == "ACTIVE" and stack in self._buffer_stacks(job)] + capacity = row["bank_count"] * row["bank_capacity_bytes"] + demand_bound = sum( + min(job.slice_total_bytes - job.slice_transferred_bytes, + row["bank_capacity_bytes"]) + for job in active + ) + bounds[stack] = { + "bank_count": row["bank_count"], + "bank_capacity_bytes": row["bank_capacity_bytes"], + "maximum_occupancy_bytes": min(capacity, demand_bound), + "active_flow_count": len(active), + "semantics": "ANALYTICAL_UPPER_BOUND_NOT_OBSERVED_BANK_OWNERSHIP", + } + return bounds + + def _activate(self) -> None: + if self._window_end is None or self._now >= self._window_end: + return + scale = self._config["work_scale"] + while True: + heads = [job for job in self._eligible_heads() + if job.arrival_ns <= self._now and self._state_allows(job) + and self._buffers_available(job)] + if not heads: + return + progress = False + for job in heads: + if job.state != "QUEUED" or not self._buffers_available(job): + continue + maintenance = job.operation in MAINTENANCE_OPERATIONS or job.maintenance_id is not None + if maintenance: + slice_bytes = job.unadmitted_bytes + else: + endpoints = self._endpoints(job) + peers = { + endpoint: sum(1 for other in heads if endpoint in self._endpoints(other) + and not (other.operation in MAINTENANCE_OPERATIONS + or other.maintenance_id is not None)) + for endpoint in endpoints + } + slice_bytes = min( + job.unadmitted_bytes, + *(self._quota_remaining[endpoint] // (scale * max(1, peers[endpoint])) + for endpoint in endpoints), + ) + if slice_bytes <= 0: + continue + if not maintenance: + for endpoint in self._endpoints(job): + self._quota_remaining[endpoint] -= slice_bytes * scale + job.unadmitted_bytes -= slice_bytes + job.slice_total_bytes = slice_bytes + job.slice_transferred_bytes = 0 + job.rate_credit = Fraction(0, 1) + job.state = "ACTIVE" + job.first_service_ns = self._now if job.first_service_ns is None else job.first_service_ns + self._active.add(job.job_id) + self._reserve_buffers(job) + progress = True + if not progress: + return + + def _rates(self) -> Dict[str, Fraction]: + active = [self._jobs[job_id] for job_id in sorted(self._active)] + if not active: + return {} + resource_rates: Dict[str, Fraction] = {} + coefficients: Dict[str, Dict[str, int]] = {} + for job in active: + coefficients[job.job_id] = {} + for resource, stage, coefficient in self._phase_resources(job): + coefficients[job.job_id][resource] = coefficient + resource_rates[resource] = Fraction( + stage["bandwidth_bytes_per_s"] * self._config["work_scale"], 1 + ) + residual = dict(resource_rates) + rates = {job.job_id: Fraction(0, 1) for job in active} + unfrozen = set(rates) + while unfrozen: + bounds = [] + for resource, capacity in residual.items(): + total_coefficient = sum( + coefficients[job_id].get(resource, 0) for job_id in unfrozen + ) + if total_coefficient: + bounds.append((capacity / total_coefficient, resource)) + if not bounds: + break + increment = min(bound for bound, _ in bounds) + for job_id in unfrozen: + rates[job_id] += increment + for resource in residual: + used = increment * sum( + coefficients[job_id].get(resource, 0) for job_id in unfrozen + ) + residual[resource] -= used + saturated = {resource for resource, value in residual.items() if value == 0} + frozen = { + job_id for job_id in unfrozen + if any(resource in saturated for resource in coefficients[job_id]) + } + if not frozen: + raise AssertionError("max-min allocation made no progress") + unfrozen -= frozen + return rates + + def _next_transfer_finish(self, rates: Mapping[str, Fraction]) -> Optional[int]: + times = [] + for job_id, rate in rates.items(): + if rate <= 0: + continue + job = self._jobs[job_id] + remaining = Fraction(job.slice_total_bytes - job.slice_transferred_bytes, 1) - job.rate_credit + duration_ns = _ceil_fraction(remaining * NS_PER_SECOND / rate) + times.append(self._now + max(1, duration_ns)) + return min(times) if times else None + + def next_completion_ns(self) -> Optional[int]: + events = [] + events.extend((completion_ns, "drain", job_id) + for completion_ns, _, job_id, _ in self._draining) + rates = self._rates() + for job_id, rate in rates.items(): + job = self._jobs[job_id] + if rate <= 0: + continue + remaining = Fraction(job.slice_total_bytes - job.slice_transferred_bytes, 1) - job.rate_credit + finish = self._now + max(1, _ceil_fraction(remaining * NS_PER_SECOND / rate)) + events.append((finish, "transfer", job_id)) + events.extend((job.arrival_ns, "arrival", job.job_id) + for job in self._jobs.values() + if job.state == "QUEUED" and job.arrival_ns > self._now) + if not events: + return None + earliest = min(time_ns for time_ns, _, _ in events) + # Only a terminal drain at the earliest state-changing event is an + # exact next completion. Other events may change max-min rates first. + terminal = [ + time_ns for time_ns, kind, job_id in events + if time_ns == earliest and kind == "drain" + and self._jobs[job_id].unadmitted_bytes == 0 + ] + return min(terminal) if terminal else None + + def next_event_ns(self) -> Optional[int]: + if self._window_end is None: + return None + values = [self._window_end] + if self._draining: + values.append(self._draining[0][0]) + transfer = self._next_transfer_finish(self._rates()) + if transfer is not None: + values.append(transfer) + arrivals = [job.arrival_ns for job in self._jobs.values() + if job.state == "QUEUED" and job.arrival_ns > self._now] + if arrivals: + values.append(min(arrivals)) + return min(value for value in values if value >= self._now) + + def _progress(self, target_ns: int, rates: Mapping[str, Fraction]) -> None: + duration = target_ns - self._now + if duration <= 0: + return + for job_id, rate in rates.items(): + job = self._jobs[job_id] + exact = job.rate_credit + rate * duration / NS_PER_SECOND + integer_bytes = min( + job.slice_total_bytes - job.slice_transferred_bytes, + exact.numerator // exact.denominator, + ) + job.rate_credit = exact - integer_bytes + if integer_bytes <= 0: + continue + job.slice_transferred_bytes += integer_bytes + job.last_service_ns = target_ns + if job.operation in {"hbm_fill", "migration_program"}: + rows = [] + base = { + "operation": job.operation, "stack": job.stack, + "channel": job.channel, "bytes": integer_bytes, + "route": job.route, "start_ns": self._now, "end_ns": target_ns, + } + if job.operation == "hbm_fill": + phases = ( + ("media_fill", f"{job.stack}:channel:{job.channel}:media", None), + ("hbm_base_fill", f"{job.stack}:base", None), + ("gpu_link_fill", f"{job.stack}:gpu-link", None), + ) + elif job.route == "direct": + phases = ( + ("media_program", f"{job.stack}:channel:{job.channel}:media", None), + ("destination_base", f"{job.stack}:base", None), + ("destination_fill", f"{job.stack}:fill", None), + ("reverse_gpu_link", f"{job.stack}:gpu-link", None), + ) + else: + partner = self._config["fabric"]["hbf"][job.stack]["pair"] + phases = ( + ("media_program", f"{job.stack}:channel:{job.channel}:media", None), + ("destination_base", f"{job.stack}:base", None), + ("destination_fill", f"{job.stack}:fill", None), + ("reverse_relay", f"{job.stack}->{partner}:relay-link", partner), + ("partner_gpu_receive", f"{partner}:gpu-link", partner), + ) + for phase, resource, partner in phases: + row = dict(base, phase=phase, resource=resource) + if partner is not None: + row["partner"] = partner + rows.append(row) + else: + rows = self._rules._activity_rows(job, integer_bytes, self._now, target_ns) + self._window_activities.extend(rows) + for resource, _, coefficient in self._phase_resources(job): + self._window_resource_work[resource] = ( + self._window_resource_work.get(resource, 0) + integer_bytes * coefficient + ) + + def _finish_transfers(self) -> None: + for job_id in list(self._active): + job = self._jobs[job_id] + if job.slice_transferred_bytes != job.slice_total_bytes: + continue + self._active.remove(job_id) + job.state = "DRAINING" + completion_ns = self._now + self._slice_drain_latency_ns(job) + heapq.heappush( + self._draining, + (completion_ns, self._sequence, job_id, job.slice_total_bytes), + ) + self._sequence += 1 + + def _finish_drains(self) -> None: + while self._draining and self._draining[0][0] <= self._now: + completion_ns, _, job_id, byte_count = heapq.heappop(self._draining) + job = self._jobs[job_id] + job.completed_bytes += byte_count + self._completed_since_read[job_id] = self._completed_since_read.get(job_id, 0) + byte_count + self._release_buffers(job) + job.slice_total_bytes = 0 + job.slice_transferred_bytes = 0 + job.rate_credit = Fraction(0, 1) + if job.unadmitted_bytes: + job.state = "QUEUED" + else: + job.state = "DONE" + job.completion_ns = completion_ns + self._window_completion_ids.append(job_id) + if job.maintenance_id is not None: + self._window_maintenance_ids.append(job.maintenance_id) + + def _progress_rows(self) -> list[dict]: + rows = [] + for job in sorted(self._jobs.values(), key=lambda row: row.sequence): + rows.append({ + "job_id": job.job_id, "maintenance_id": job.maintenance_id, + "metadata": copy.deepcopy(job.metadata), "operation": job.operation, + "stack": job.stack, "channel": job.channel, "route": job.route, + "arrival_ns": job.arrival_ns, "total_bytes": job.total_bytes, + "served_since_last_advance_bytes": self._completed_since_read.get(job.job_id, 0), + "cumulative_served_bytes": job.completed_bytes, + "remaining_bytes": job.total_bytes - job.completed_bytes, + "unadmitted_bytes": job.unadmitted_bytes, + "slice_total_bytes": job.slice_total_bytes, + "slice_transferred_bytes": job.slice_transferred_bytes, + "inflight_transferred_bytes": ( + job.slice_transferred_bytes if job.state in {"ACTIVE", "DRAINING"} else 0 + ), + "state": job.state, "first_service_ns": job.first_service_ns, + "last_service_ns": job.last_service_ns, + "completion_ns": job.completion_ns, + }) + return rows + + def _retire_completed(self, completion_ids: Sequence[str]) -> None: + """Drop emitted terminal records while retaining global ID uniqueness.""" + completed = set(completion_ids) + if not completed: + return + for key, queue in self._queues.items(): + self._queues[key] = deque(job_id for job_id in queue if job_id not in completed) + for job_id in completed: + self._jobs.pop(job_id, None) + + def advance_to(self, horizon_ns: int) -> dict: + if self._window_end is None: + raise RuntimeError("begin_window is required before advance_to") + horizon = _integer(horizon_ns, "horizon_ns") + if horizon < self._now or horizon > self._window_end: + raise ValueError("horizon must lie within the current control window") + start = self._now + self._completed_since_read = {} + while self._now < horizon: + self._finish_drains() + self._activate() + rates = self._rates() + candidates = [horizon] + transfer = self._next_transfer_finish(rates) + if transfer is not None: + candidates.append(transfer) + if self._draining: + candidates.append(self._draining[0][0]) + arrivals = [job.arrival_ns for job in self._jobs.values() + if job.state == "QUEUED" and job.arrival_ns > self._now] + if arrivals: + candidates.append(min(arrivals)) + target = min(value for value in candidates if value >= self._now) + if target == self._now: + # Only same-time completion/release/admission is allowed here. + self._finish_transfers() + self._finish_drains() + self._activate() + next_value = [value for value in candidates if value > self._now] + if not next_value: + break + target = min(next_value) + self._progress(target, rates) + self._now = target + self._finish_transfers() + self._finish_drains() + self._activate() + activities = self._window_activities[self._activity_offset:] + completion_ids = self._window_completion_ids[self._completion_offset:] + maintenance_ids = self._window_maintenance_ids[self._maintenance_offset:] + self._activity_offset = len(self._window_activities) + self._completion_offset = len(self._window_completion_ids) + self._maintenance_offset = len(self._window_maintenance_ids) + receipt = { + "schema_version": self.schema_version, + "start_ns": start, "end_ns": self._now, + "control_window_start_ns": self._window_start, + "control_window_end_ns": self._window_end, + "activities": copy.deepcopy(activities), + "resource_work_units_scaled": dict(self._window_resource_work), + "endpoint_quota_remaining_scaled": dict(self._quota_remaining), + "buffer_occupancy_bounds": self._buffer_bounds(), + "job_progress": self._progress_rows(), + "completion_ids": list(completion_ids), + "maintenance_completion_ids": list(maintenance_ids), + "next_completion_ns": self.next_completion_ns(), + "next_event_ns": self.next_event_ns(), + "semantics": { + "time": "EXACT_INTEGER_NS_EVENTS_WITHIN_FROZEN_THERMAL_CONTROL_WINDOW", + "bandwidth": "EXACT_FRACTION_PROGRESSIVE_MAX_MIN_INTEGER_BYTE_EMISSION", + "buffer": "FINITE_TWO_BANK_ANALYTICAL_OCCUPANCY_BOUND_CONTINUOUS_TURNOVER", + "pipeline": "BANDWIDTH_SHARED_EXACT_PROPAGATION_LATENCY_APPROXIMATE", + "energy": "ACTIVITY_FACTS_SHARE_THE_SAME_PROGRESS_LEDGER", + }, + } + if self._now == self._window_end: + receipt["window_complete"] = True + self._retire_completed(completion_ids) + return receipt diff --git a/experiments/eq3_system_thermal/causal_workload.py b/experiments/eq3_system_thermal/causal_workload.py new file mode 100644 index 0000000..4eca4b5 --- /dev/null +++ b/experiments/eq3_system_thermal/causal_workload.py @@ -0,0 +1,935 @@ +"""Online, architecture-derived causal weight-consumption model. + +This is deliberately not a NAND simulator. Storage jobs are offered to an +external service and dependencies unlock only when that service reports their +actual completion timestamp. +""" + +from __future__ import annotations + +from collections import OrderedDict +from copy import deepcopy +import hashlib +import json +from pathlib import Path +from typing import Any, Callable + + +CATALOG = Path(__file__).parents[1] / "eq3_maintenance" / "sources" / "qwen2_5_weight_models.json" +TRACE_ORIGIN = "SYNTHETIC_ARCHITECTURE_DEPENDENCY_FROM_OFFICIAL_METADATA" +TINY_TRACE_ORIGIN = "TRACE_DERIVED_TINY_CPU_FORWARD_TEMPLATE" +DEPENDENCY_MODES = {"synthetic_metadata_dag", "tiny_cpu_template"} +REPO_ROOT = Path(__file__).resolve().parents[2] + + +def load_architecture(model_id: str) -> dict[str, Any]: + models = json.loads(CATALOG.read_text(encoding="utf-8"))["models"] + if model_id not in models: + raise ValueError(f"unsupported official model metadata {model_id!r}") + item = deepcopy(models[model_id]) + item["model_id"] = model_id + return item + + +def _tensor_groups(meta: dict[str, Any]) -> list[dict[str, Any]]: + a, scalar = meta["architecture"], meta["bytes_per_tensor_element"] + h, inter = a["hidden_size"], a["intermediate_size"] + heads, kv, layers, vocab = (a["num_attention_heads"], a["num_key_value_heads"], + a["num_hidden_layers"], a["vocab_size"]) + if h % heads: + raise ValueError("hidden size is not divisible by attention heads") + hd = h // heads + groups = [{"tensor_id": "model.embed_tokens", "kind": "embedding", + "bytes": vocab * h * scalar, "layer": None}] + for layer in range(layers): + groups.extend([ + {"tensor_id": f"model.layers.{layer}.attention_bundle", "kind": "attention", + "bytes": ((2 * h * h + h) + 2 * (h * kv * hd + kv * hd) + + h) * scalar, "layer": layer}, + {"tensor_id": f"model.layers.{layer}.mlp_bundle", "kind": "mlp", + "bytes": (3 * h * inter + h) * scalar, "layer": layer}, + ]) + groups.extend([ + {"tensor_id": "model.norm", "kind": "final_norm", "bytes": h * scalar, + "layer": None}, + {"tensor_id": "lm_head", "kind": "output_embedding", + "bytes": vocab * h * scalar, "layer": None}, + ]) + if sum(x["bytes"] for x in groups) != meta["tensor_payload_bytes"]: + raise ValueError("derived tensor groups do not match official payload bytes") + address = 0 + for group in groups: + group['logical_address_bytes'] = address + address = ((address + group['bytes'] + 1048575) // 1048576) * 1048576 + return groups + + +def _canonical(value: Any) -> bytes: + return json.dumps(value, sort_keys=True, separators=(",", ":"), + allow_nan=False).encode("utf-8") + + +def _target_projection(meta: dict[str, Any], context_tokens: int) -> dict[str, Any]: + """Regenerate target shapes, addresses and analytical cost from metadata.""" + if isinstance(context_tokens, bool) or not isinstance(context_tokens, int) \ + or context_tokens <= 0: + raise ValueError("projection_context_tokens must be a positive integer") + a = meta["architecture"] + h, inter = a["hidden_size"], a["intermediate_size"] + heads, kv, layers = (a["num_attention_heads"], a["num_key_value_heads"], + a["num_hidden_layers"]) + if h % heads: + raise ValueError("hidden size is not divisible by attention heads") + hd = h // heads + groups = _tensor_groups(meta) + projected = [] + for group in groups: + row = deepcopy(group) + if row["kind"] == "embedding": + row["weight_shapes"] = [[a["vocab_size"], h]] + row["analytical_macs_per_token"] = 0 + elif row["kind"] == "attention": + row["weight_shapes"] = [[h], [h, h], [h], [kv * hd, h], [kv * hd], + [kv * hd, h], [kv * hd], [h, h]] + row["analytical_macs_per_token"] = ( + 2 * h * h + 2 * h * kv * hd + 2 * h * context_tokens) + elif row["kind"] == "mlp": + row["weight_shapes"] = [[h], [inter, h], [inter, h], [h, inter]] + row["analytical_macs_per_token"] = 3 * h * inter + elif row["kind"] == "final_norm": + row["weight_shapes"] = [[h]] + row["analytical_macs_per_token"] = 0 + else: + row["weight_shapes"] = [[a["vocab_size"], h]] + row["analytical_macs_per_token"] = a["vocab_size"] * h + projected.append(row) + transformer_macs = sum(row["analytical_macs_per_token"] for row in projected + if row["kind"] in {"attention", "mlp"}) + head_macs = next(row["analytical_macs_per_token"] for row in projected + if row["kind"] == "output_embedding") + return { + "layer_count": layers, "hidden_size": h, "head_dim": hd, + "num_attention_heads": heads, "num_key_value_heads": kv, + "intermediate_size": inter, "context_tokens": context_tokens, + "logical_regions": projected, + "tensor_payload_bytes": meta["tensor_payload_bytes"], + "analytical_macs_per_token_at_context": transformer_macs, + "analytical_output_head_macs_per_token": head_macs, + "analytical_total_macs_per_token_at_context": transformer_macs + head_macs, + "compute_cost_semantics": ( + "ANALYTICAL_DENSE_MAC_COUNT_EXCLUDES_NORMS_ROPE_SOFTMAX_AND_RUNTIME" + ), + "address_semantics": "DERIVED_1MIB_ALIGNED_SCENARIO_NOT_SAFETENSORS_FILE_OFFSETS", + } + + +def _resolve_trace_path(value: Any) -> tuple[Path, str]: + if not isinstance(value, str) or not value: + raise ValueError("tiny_trace_path must be a nonempty path string") + path = Path(value) + resolved = path.resolve() if path.is_absolute() else (REPO_ROOT / path).resolve() + if not resolved.is_file(): + raise ValueError("tiny trace artifact does not exist") + try: + portable = str(resolved.relative_to(REPO_ROOT)) + except ValueError: + portable = "EXTERNAL_FIXED_TEST_ARTIFACT" + return resolved, portable + + +def _validate_tiny_template(config: dict[str, Any], meta: dict[str, Any]) -> dict[str, Any]: + path, portable = _resolve_trace_path(config.get("tiny_trace_path")) + document = json.loads(path.read_text(encoding="utf-8")) + if document.get("schema_version") != "eq3-tiny-qwen2-cpu-trace-v1" \ + or document.get("classification") != "TRACE_DERIVED_TINY_RANDOM_WEIGHT_CPU_FORWARD": + raise ValueError("tiny trace has unsupported schema or classification") + stored = document.get("trace_sha256") + payload = dict(document) + payload.pop("trace_sha256", None) + actual = hashlib.sha256(_canonical(payload)).hexdigest() + expected = config.get("tiny_trace_sha256") + if not isinstance(expected, str) or expected != stored or actual != stored: + raise ValueError("tiny trace checksum/provenance mismatch") + operations = document.get("operations") + accesses = document.get("weight_accesses") + if not isinstance(operations, list) or not operations or not isinstance(accesses, list) \ + or not accesses: + raise ValueError("tiny trace lacks captured operations or weight accesses") + by_id = {} + for sequence, row in enumerate(operations): + if row.get("sequence") != sequence or row.get("op_id") in by_id: + raise ValueError("tiny operation order or identity is invalid") + for dependency in row.get("depends_on", []): + if dependency != "token_ids" and dependency not in by_id: + raise ValueError("tiny operation dependency is absent or not causal") + by_id[row["op_id"]] = row + access_by_op = {} + for sequence, row in enumerate(accesses): + if row.get("sequence") != sequence or row.get("op_id") not in by_id: + raise ValueError("tiny weight access order or owner is invalid") + shape = row.get("access_shape") + storage_shape = row.get("storage_shape") + if not isinstance(shape, list) or not shape or any(type(x) is not int or x <= 0 for x in shape) \ + or not isinstance(storage_shape, list) or not storage_shape \ + or any(type(x) is not int or x <= 0 for x in storage_shape): + raise ValueError("tiny weight access shape is invalid") + count = 1 + for dimension in shape: + count *= dimension + if row.get("storage_dtype") != "float32" or row.get("access_bytes") != count * 4: + raise ValueError("tiny weight access byte count is not the captured float32 array") + access_by_op.setdefault(row["op_id"], []).append(row["weight_name"]) + prefill = [row for row in operations if row.get("forward_id") == "prefill"] + layer_count = document.get("config", {}).get("layers") + tiny_config = document.get("config", {}) + if type(layer_count) is not int or layer_count <= 0: + raise ValueError("tiny trace layer count is invalid") + tiny_heads = tiny_config.get("num_attention_heads") + tiny_kv = tiny_config.get("num_key_value_heads") + tiny_hd = tiny_config.get("head_dim") + if any(type(value) is not int or value <= 0 for value in (tiny_heads, tiny_kv, tiny_hd)) \ + or tiny_heads % tiny_kv: + raise ValueError("tiny trace GQA geometry is invalid") + layers = [] + prior = next((row for row in prefill if row["name"] == "embedding_lookup"), None) + if prior is None: + raise ValueError("tiny trace lacks embedding root") + for layer in range(layer_count): + names = { + role: next((row for row in prefill if row["name"] == f"layer{layer}.{suffix}"), None) + for role, suffix in ( + ("input_norm", "input_rmsnorm"), ("q", "q_projection"), + ("k", "k_projection"), ("v", "v_projection"), + ("attention", "rope_gqa_causal_attention"), ("o", "o_projection"), + ("post_norm", "post_attention_rmsnorm"), ("gate", "gate_projection"), + ("up", "up_projection"), ("mlp", "swiglu_down_projection"))} + if any(row is None for row in names.values()): + raise ValueError("tiny trace layer structure is incomplete") + if names["input_norm"]["depends_on"] != [prior["op_id"]] \ + or set(names["attention"]["depends_on"]) != { + names["q"]["op_id"], names["k"]["op_id"], names["v"]["op_id"]} \ + or names["o"]["depends_on"] != [names["attention"]["op_id"]] \ + or names["post_norm"]["depends_on"] != [names["o"]["op_id"]] \ + or set(names["mlp"]["depends_on"]) != { + names["gate"]["op_id"], names["up"]["op_id"]}: + raise ValueError("tiny trace attention-to-MLP dependency structure is invalid") + expected_access_roles = ("input_norm", "q", "k", "v", "o", "post_norm", + "gate", "up", "mlp") + if any(not access_by_op.get(names[role]["op_id"]) for role in expected_access_roles): + raise ValueError("tiny trace layer operation lacks an actual weight access") + details = names["attention"].get("details", {}) + if details.get("q_shape", [None, None, None])[-2:] != [tiny_heads, tiny_hd] \ + or details.get("kv_shape", [None, None, None])[-2:] != [tiny_kv, tiny_hd] \ + or details.get("kv_repeat_groups") != tiny_heads // tiny_kv: + raise ValueError("tiny trace observed GQA shapes disagree with its config") + layers.append({"template_layer": layer, + "attention_op_ids": [names[x]["op_id"] for x in ( + "input_norm", "q", "k", "v", "attention", "o")], + "mlp_op_ids": [names[x]["op_id"] for x in ( + "post_norm", "gate", "up", "mlp")]}) + prior = names["mlp"] + final_norm = next((row for row in prefill if row["name"] == "final_rmsnorm"), None) + head = next((row for row in prefill if row["name"] == "lm_head"), None) + if final_norm is None or head is None or final_norm["depends_on"] != [prior["op_id"]] \ + or head["depends_on"] != [final_norm["op_id"]] \ + or not access_by_op.get(final_norm["op_id"]) or not access_by_op.get(head["op_id"]): + raise ValueError("tiny trace final-norm/head dependency structure is invalid") + context = int(config.get("projection_context_tokens", 0)) + target = _target_projection(meta, context) + recorded = document.get("target_projection", {}).get(meta["model_id"]) + compact = [{"name": row["tensor_id"], "logical_address_bytes": row["logical_address_bytes"], + "payload_bytes": row["bytes"]} for row in target["logical_regions"]] + if not isinstance(recorded, dict) or recorded.get("tensor_payload_bytes") != target[ + "tensor_payload_bytes"] or recorded.get("logical_regions") != compact \ + or recorded.get("analytical_macs_per_token_at_context") != target[ + "analytical_macs_per_token_at_context"] \ + or recorded.get("context_tokens") != context: + raise ValueError("tiny artifact target projection disagrees with regenerated metadata") + return { + "trace_origin": TINY_TRACE_ORIGIN, "artifact_path": portable, + "trace_sha256": stored, "trace_file_sha256": hashlib.sha256(path.read_bytes()).hexdigest(), + "source_sha256": document["source_sha256"], "config_sha256": document["config_sha256"], + "tiny_layer_count": layer_count, "template_layers": layers, + "final_op_ids": [final_norm["op_id"], head["op_id"]], + "target_projection": target, + "validation": "DEPENDENCIES_ACCESS_ORDER_SHAPES_BYTES_AND_TARGET_PROJECTION_VALIDATED", + "scope": "ARCHITECTURE_ORDERING_ONLY_COMPUTE_TIMING_REMAINS_EXPLICIT_SCENARIO", + } + + +def build_architecture_trace(config: dict[str, Any]) -> dict[str, Any]: + """Build a compact dependency DAG; it is not a captured framework trace.""" + model_id = config["model_id"] + meta = load_architecture(model_id) + dependency_mode = config.get("dependency_mode", "synthetic_metadata_dag") + if dependency_mode not in DEPENDENCY_MODES: + raise ValueError("unsupported dependency_mode") + required = ("batch_intervals", "batch_size", "batch_interval_ns", "prefetch_layers", + "attention_compute_ns_per_token", "mlp_compute_ns_per_token", + "output_compute_ns_per_token", "embedding_access") + missing = [key for key in required if key not in config] + if missing: + raise ValueError(f"explicit causal scenario fields required: {missing}") + intervals = int(config["batch_intervals"]) + batch_size = int(config["batch_size"]) + interval_ns = int(config["batch_interval_ns"]) + attention_compute_ns = int(config["attention_compute_ns_per_token"]) + mlp_compute_ns = int(config["mlp_compute_ns_per_token"]) + output_compute_ns = int(config["output_compute_ns_per_token"]) + prefetch = int(config["prefetch_layers"]) + embedding_access = config["embedding_access"] + if embedding_access not in {"selected_token_rows", "full_weight_stress"}: + raise ValueError("embedding_access must be selected_token_rows or full_weight_stress") + if min(intervals, batch_size, attention_compute_ns, mlp_compute_ns, + output_compute_ns) <= 0 or interval_ns < 0 or prefetch < 0: + raise ValueError("trace counts/durations must be positive; interval/prefetch non-negative") + prefetch_mode = config.get("prefetch_mode", "layer_lookahead") + if prefetch_mode not in {"on_demand", "layer_lookahead"}: + raise ValueError("unknown prefetch_mode") + if prefetch_mode == "on_demand" and prefetch != 0: + raise ValueError("on-demand weights cannot also request layer lookahead") + provenance = None + if dependency_mode == "tiny_cpu_template": + provenance = _validate_tiny_template(config, meta) + groups = provenance["target_projection"]["logical_regions"] + else: + groups = _tensor_groups(meta) + by_id = {x["tensor_id"]: x for x in groups} + layers = meta["architecture"]["num_hidden_layers"] + batches = [] + first_interval = int(config.get('first_interval', 0)) + for interval in range(first_interval, first_interval + intervals): + prefix = f"interval{interval}" + tasks: list[dict[str, Any]] = [] + embedding = deepcopy(by_id["model.embed_tokens"]) + if embedding_access == "selected_token_rows": + unique_rows = int(config.get("embedding_unique_rows_per_interval", batch_size)) + if not 1 <= unique_rows <= batch_size: + raise ValueError("embedding unique rows must be within the batch size") + embedding["full_tensor_bytes"] = embedding["bytes"] + embedding["bytes"] = (meta["architecture"]["hidden_size"] + * meta["bytes_per_tensor_element"] * unique_rows) + embedding["tensor_id"] += f":interval{interval}:selected_rows" + embedding["access_semantics"] = "SELECTED_TOKEN_ROWS_SYNTHETIC_IDENTITIES" + else: + embedding["access_semantics"] = "EXPLICIT_SYNTHETIC_FULL_WEIGHT_STRESS" + tasks.append({"task_id": prefix + ":embed_read", "type": "storage", + "tensor": embedding, "issue_after": [], + "consume_after": [], "consumer_count": batch_size}) + previous = prefix + ":embed_read" + layer_compute_ids: list[str] = [] + for layer in range(layers): + template = (None if provenance is None else + provenance["template_layers"][layer % provenance["tiny_layer_count"]]) + issue_parent_index = layer - prefetch - 1 + issue_parent = ([layer_compute_ids[issue_parent_index]] + if issue_parent_index >= 0 else []) + attn_read = prefix + f":l{layer}:attn_read" + attn_compute = prefix + f":l{layer}:attn_compute" + mlp_read = prefix + f":l{layer}:mlp_read" + mlp_compute = prefix + f":l{layer}:mlp_compute" + tasks.extend([ + {"task_id": attn_read, "type": "storage", + "tensor": by_id[f"model.layers.{layer}.attention_bundle"], + "issue_after": ([previous] if prefetch_mode == "on_demand" else issue_parent), "consume_after": [previous], + "consumer_count": batch_size, + "structure_template_op_ids": ( + None if template is None else template["attention_op_ids"])}, + {"task_id": attn_compute, "type": "compute", + "duration_ns": attention_compute_ns * batch_size, + "depends_on": [previous, attn_read], + "structure_role": "ATTENTION_AFTER_INPUT_AND_WEIGHT_READ"}, + {"task_id": mlp_read, "type": "storage", + "tensor": by_id[f"model.layers.{layer}.mlp_bundle"], + "issue_after": ([attn_compute] if prefetch_mode == "on_demand" else [previous]), "consume_after": [attn_compute], + "consumer_count": batch_size, + "structure_template_op_ids": ( + None if template is None else template["mlp_op_ids"])}, + {"task_id": mlp_compute, "type": "compute", + "duration_ns": mlp_compute_ns * batch_size, + "depends_on": [attn_compute, mlp_read], + "structure_role": "MLP_AFTER_ATTENTION_AND_WEIGHT_READ"}, + ]) + previous = mlp_compute + layer_compute_ids.append(mlp_compute) + for suffix, tensor_id in (("norm_read", "model.norm"), ("head_read", "lm_head")): + task_id = prefix + ":" + suffix + tasks.append({"task_id": task_id, "type": "storage", "tensor": by_id[tensor_id], + "issue_after": [previous], "consume_after": [previous], + "consumer_count": batch_size, + "structure_template_op_ids": ( + None if provenance is None else [provenance["final_op_ids"][ + 0 if suffix == "norm_read" else 1]])}) + previous = task_id + final = prefix + ":token_complete" + tasks.append({"task_id": final, "type": "compute", + "duration_ns": output_compute_ns * batch_size, + "depends_on": [previous], "is_token_terminal": True}) + for task in tasks: + task["batch_interval_id"] = interval + if provenance is None: + task.pop("structure_template_op_ids", None) + task.pop("structure_role", None) + batches.append({"interval_id": interval, "arrival_ns": interval * interval_ns, + "batch_size": batch_size, "token_ids": [ + f"{prefix}:token{i}" for i in range(batch_size)], "tasks": tasks, + "terminal_task_id": final}) + return { + "schema_version": "eq3-causal-architecture-trace-v1", + "trace_origin": (TRACE_ORIGIN if provenance is None else TINY_TRACE_ORIGIN), + "dependency_mode": dependency_mode, "model_id": model_id, + "embedding_access": embedding_access, + "compute_cost_evidence": "EXPLICIT_SCENARIO_INPUT_NOT_RUNTIME_TRACE", + "official_metadata": {k: meta[k] for k in ( + "revision", "resolved_commit", "tensor_payload_bytes", "architecture", "sources")}, + "prefetch_layers": prefetch, "prefetch_mode": prefetch_mode, "batches": batches, + "structure_provenance": provenance, + "limitations": ["not a PyTorch or hardware runtime trace", "no NAND timing model", + "activation and KV-cache traffic unavailable"], + } + + +class CausalExecutor: + """Incrementally connect a causal trace to an external topology service.""" + + def __init__(self, trace: dict[str, Any], config: dict[str, Any], + placement_provider: Callable[[dict[str, Any], str, int], dict[str, Any]] | None = None): + if trace.get("trace_origin") not in {TRACE_ORIGIN, TINY_TRACE_ORIGIN}: + raise ValueError("unclassified causal trace") + self.trace = deepcopy(trace) + self.cache_capacity = int(config.get("cache_capacity_bytes", 0)) + self.migration_mode = config.get("migration_mode", "fixed") + self.migration_threshold = int(config.get("migration_access_threshold", 2)) + self.cache_mode = config.get("cache_mode") + self.coalescing_enabled = config.get("coalescing_enabled") + self.prefetch_wait_mode = config.get("prefetch_wait_mode") + self.retry_count = config.get('retry_count_per_source_read', 0) + if type(self.retry_count) is not int or self.retry_count not in (0,1,4): + raise ValueError('retry count must be explicit conditional scenario 0, 1 or 4') + if self.cache_capacity < 0 or self.migration_mode not in {"fixed", "basic"}: + raise ValueError("invalid cache or migration mode") + self.migration_capacity_bytes = int(config.get('migration_capacity_bytes', 0)) + self.migration_used_bytes = 0 + self.tensor_versions = {} + if self.migration_mode == 'basic' and self.migration_capacity_bytes <= 0: + raise ValueError('basic migration requires finite explicit destination capacity') + if self.migration_mode == 'basic' and (not config.get('fast_stripe_targets') or + any(not r['stack'].startswith('hbf') for r in config['fast_stripe_targets'])): + raise ValueError('basic placement migration currently requires explicit HBF destinations') + if self.cache_mode not in {"disabled", "ideal_metadata_only", "external_hbm"}: + raise ValueError("unsupported cache_mode") + if self.cache_mode == "disabled" and self.cache_capacity != 0: + raise ValueError("disabled cache requires zero capacity") + if not isinstance(self.coalescing_enabled, bool): + raise ValueError("coalescing_enabled must be explicit boolean") + if self.prefetch_wait_mode not in {"wait_at_consumption", "stall_at_issue"}: + raise ValueError("explicit prefetch_wait_mode is required") + self.stripe_unit_bytes = int(config.get("stripe_unit_bytes", 0)) + if self.stripe_unit_bytes <= 0: + raise ValueError("stripe_unit_bytes must be an explicit positive value") + if placement_provider is None: + self._placement_provider, self.target_count = self._default_placement(config) + else: + self._placement_provider = placement_provider + self.target_count = int(config.get("placement_target_count", 0)) + if self.target_count <= 0: + raise ValueError("custom placement_provider requires placement_target_count") + self.tasks = {t["task_id"]: deepcopy(t) for b in trace["batches"] for t in b["tasks"]} + self.tensor_specs = {t['tensor']['tensor_id']:deepcopy(t['tensor']) + for t in self.tasks.values() if t['type']=='storage'} + self.arrival = {t["task_id"]: b["arrival_ns"] for b in trace["batches"] for t in b["tasks"]} + self.done: dict[str, int] = {} + self.offered: set[str] = set() + self.job_group: dict[str, str] = {} + self.groups: dict[str, dict[str, Any]] = {} + self.job_bytes: dict[str, int] = {} + self.job_arrival: dict[str, int] = {} + self.pending_tensor: dict[str, str] = {} + self.cache: OrderedDict[str, int] = OrderedDict() + self.cache_bytes = 0 + self.access_count: dict[str, int] = {} + self.tensor_tier: dict[str, str] = {} + self.pending_migrations: dict[str, str] = {} + self.jobs: list[dict[str, Any]] = [] + self.events: list[dict[str, Any]] = [] + self.compute_running = None + self.compute_available_ns = 0 + self.now_ns = 0 + self.deferred_jobs = [] + self.cache_reservations = {} + self.deferred_source_erase = [] + self.completed_batches = [] + self.known_batch_ids = {b.get('interval_id', i) for i,b in enumerate(trace['batches'])} + if self.cache_mode == "external_hbm": + targets = config.get("fast_stripe_targets", []) + if not targets or any(not row["stack"].startswith("hbm") for row in targets): + raise ValueError("service-backed cache requires explicit HBM targets") + + def append_trace(self, trace): + """Append an arrived batch; callers retain uninstantiated arrival backlog.""" + for key in ('trace_origin','dependency_mode','model_id','embedding_access','prefetch_layers','prefetch_mode'): + if trace.get(key) != self.trace.get(key): + raise ValueError('streaming trace scientific identity changed') + left = (self.trace.get("structure_provenance") or {}).get("trace_sha256") + right = (trace.get("structure_provenance") or {}).get("trace_sha256") + if left != right: + raise ValueError("streaming structure provenance changed") + for batch in trace['batches']: + if batch['interval_id'] in self.known_batch_ids: + raise ValueError('duplicate streaming batch') + self.known_batch_ids.add(batch['interval_id']) + for task in batch['tasks']: + task_id=task['task_id'] + if task_id in self.tasks: + raise ValueError('duplicate streaming batch/task') + self.tasks[task_id]=deepcopy(task);self.arrival[task_id]=batch['arrival_ns'] + if task['type']=='storage':self.tensor_specs[task['tensor']['tensor_id']]=deepcopy(task['tensor']) + self.trace['batches'].append(deepcopy(batch)) + + def retire_completed_batches(self, now_ns): + """Retire finished DAG storage, preserving token completion and cache state.""" + self._resolve_ready_storage(now_ns) + retained=[]; completed=[] + referenced={task for group in self.groups.values() for task in group.get('task_ids',[])} + for batch in self.trace['batches']: + task_ids={task['task_id'] for task in batch['tasks']} + terminal=batch['terminal_task_id'] + if terminal not in self.done or task_ids & referenced: + retained.append(batch);continue + fact={'interval_id':batch['interval_id'],'arrival_ns':batch['arrival_ns'], + 'completion_ns':self.done[terminal],'token_count':len(batch['token_ids'])} + self.completed_batches.append(fact);completed.append(fact) + for task_id in task_ids: + self.tasks.pop(task_id);self.arrival.pop(task_id);self.done.pop(task_id,None) + self.offered.discard(task_id) + self.trace['batches']=retained + return completed + + def drain_observations(self): + """Return immutable-window facts once; avoid retaining full run duplicates.""" + result={'events':self.events,'submissions':self.jobs} + self.events=[];self.jobs=[] + return result + + @staticmethod + def _default_placement(config: dict[str, Any]): + targets = deepcopy(config.get("stripe_targets")) + if targets is None and config.get("default_placement") is not None: + targets = [deepcopy(config["default_placement"])] + if not targets: + raise ValueError("explicit stripe_targets are required") + fast_targets = deepcopy(config.get("fast_stripe_targets", targets)) + if not fast_targets: + raise ValueError("fast_stripe_targets cannot be empty") + if len(fast_targets) != len(targets): + raise ValueError("source and fast stripe target counts must match") + def provider(tensor: dict[str, Any], tier: str, partition: int) -> dict[str, Any]: + selected = fast_targets if tier == "fast" else targets + value = selected[partition % len(selected)] + required = {"stack", "channel", "route"} + if not required.issubset(value): + raise ValueError("placement lacks stack/channel/route") + return deepcopy(value) + return provider, len(targets) + + def _stripe_parts(self, size: int) -> list[tuple[int, int]]: + """Aggregate round-robin stripe units into one child per physical target.""" + full, tail = divmod(size, self.stripe_unit_bytes) + base, extra = divmod(full, self.target_count) + parts = [] + for index in range(self.target_count): + child = (base + (index < extra)) * self.stripe_unit_bytes + if tail and index == full % self.target_count: + child += tail + if child: + parts.append((index, child)) + if sum(value for _, value in parts) != size: + raise AssertionError("stripe partition did not conserve tensor bytes") + return parts + + def _deps_time(self, ids: list[str]) -> int | None: + if any(item not in self.done for item in ids): + return None + return max((self.done[item] for item in ids), default=0) + + def _insert_cache(self, tensor: str, size: int) -> None: + if size > self.cache_capacity or self.cache_capacity == 0: + return + if tensor in self.cache: + self.cache_bytes -= self.cache.pop(tensor) + while self.cache and self.cache_bytes + size > self.cache_capacity: + _, removed = self.cache.popitem(last=False) + self.cache_bytes -= removed + self.cache[tensor] = size + self.cache_bytes += size + + def _offer_cache_fill(self, tensor: str, size: int, now_ns: int) -> None: + if self.cache_mode != "external_hbm" or size > self.cache_capacity: + return + if tensor in self.cache or tensor in self.cache_reservations: + return + reserved = sum(self.cache_reservations.values()) + while self.cache and self.cache_bytes + reserved + size > self.cache_capacity: + pinned = {group['tensor_id'] for group in self.groups.values() if group.get('cache_hit')} + evicted = next((key for key in self.cache if key not in pinned), None) + if evicted is None: + return + removed = self.cache.pop(evicted) + self.cache_bytes -= removed + self.events.append({"kind":"cache_evict","tensor_id":evicted,"at_ns":now_ns,"bytes":removed}) + if reserved + self.cache_bytes + size > self.cache_capacity: + return # Pending fills already own the finite capacity. + self.cache_reservations[tensor] = size + group_id = f"cache-fill:{tensor}:{now_ns}" + group = {"operation":"cache_fill", "tensor_id":tensor,"tensor_bytes":size, + "pending":set(),"task_ids":[],"completion_ns":now_ns} + self.groups[group_id] = group + for partition, child_bytes in self._stripe_parts(size): + place = self._placement_provider({"tensor_id":tensor,"bytes":size},"fast",partition) + job_id = group_id + f":part{partition}" + job = {"job_id":job_id,"operation":"hbm_fill","bytes":child_bytes, + "arrival_ns":now_ns,**place, + "metadata":{"parent_group_id":group_id,"tensor_id":tensor, + "purpose":"CACHE_FILL_AFTER_SOURCE_DELIVERY"}} + group["pending"].add(job_id) + self.job_group[job_id] = group_id + self.job_bytes[job_id] = child_bytes + self.job_arrival[job_id] = now_ns + self.jobs.append(deepcopy(job)) + self.deferred_jobs.append(job) + + def _settle_compute(self, now_ns: int) -> None: + if self.compute_running is not None: + task_id, start, finish = self.compute_running + if finish <= now_ns: + self.done[task_id] = finish + self.compute_available_ns = finish + self.compute_running = None + self.events.append({"kind":"compute_complete", "task_id":task_id, + "start_ns":start,"completion_ns":finish, + "duration_ns":finish-start}) + + def _start_compute(self, now_ns: int) -> None: + if self.compute_running is not None: + return + if any(group.get("blocks_compute") for group in self.groups.values()): + return + candidates = [] + for task_id, task in self.tasks.items(): + if task_id in self.done or task["type"] != "compute": + continue + ready = self._deps_time(task["depends_on"]) + if ready is not None and max(ready, self.arrival[task_id]) <= now_ns: + candidates.append((max(ready,self.arrival[task_id]), task_id)) + if candidates: + _, task_id = min(candidates) + start = max(now_ns, self.compute_available_ns) + finish = start + self.tasks[task_id]["duration_ns"] + self.compute_running = (task_id,start,finish) + self.events.append({"kind":"compute_start","task_id":task_id, + "start_ns":start,"scheduled_end_ns":finish, + "resource":"single_scenario_gpu_compute"}) + + def next_internal_event_ns(self) -> int | None: + candidates = [self.compute_running[2]] if self.compute_running is not None else [] + for task_id, task in self.tasks.items(): + if task_id in self.done: + continue + if task["type"] == "storage" and task_id not in self.offered: + ready = self._deps_time(task["issue_after"]) + if ready is not None: + candidates.append(max(ready, self.arrival[task_id])) + return min(candidates) if candidates else None + + def poll(self, now_ns: int) -> list[dict[str, Any]]: + """Return newly eligible jobs at exact ``now_ns``; never self-complete them.""" + if not isinstance(now_ns, int) or now_ns < 0: + raise ValueError("now_ns must be non-negative integer ns") + if now_ns < self.now_ns: + raise ValueError("causal clock cannot go backward") + self.now_ns = now_ns + self._resolve_ready_storage(now_ns) + result, self.deferred_jobs = self.deferred_jobs, [] + pending = [] + for transfer in self.deferred_source_erase: + pinned = any(g.get('operation')=='read' and g.get('tier')=='source' and + g['tensor_id']==transfer['tensor_id'] for g in self.groups.values()) + if pinned:pending.append(transfer) + else:result.extend(self._migration_jobs(transfer,now_ns)) + self.deferred_source_erase = pending + for task_id, task in self.tasks.items(): + if task_id in self.done or task_id in self.offered or task["type"] != "storage": + continue + issue_ready = self._deps_time(task["issue_after"]) + if issue_ready is None: + continue + logical_issue_ns = max(issue_ready, self.arrival[task_id]) + if logical_issue_ns > now_ns: + continue + issue_ns = now_ns # A queued batch cannot submit retrospectively. + tensor = task["tensor"]["tensor_id"] + consume_ready = self._deps_time(task["consume_after"]) + if self.cache_mode == "ideal_metadata_only" and tensor in self.cache: + self.cache.move_to_end(tensor) + self.offered.add(task_id) + if consume_ready is None: + task["external_ready_ns"] = issue_ns + task["ready_source"] = "cache" + else: + self.done[task_id] = max(issue_ns, consume_ready) + self.events.append({"kind": "storage_consumed", "task_id": task_id, + "tensor_id": tensor, "ready_ns": issue_ns, + "consume_ns": self.done[task_id], "source": "cache"}) + self.events.append({"kind": "cache_hit", "task_id": task_id, + "tensor_id": tensor, "ready_ns": issue_ns, + "completion_ns": self.done.get(task_id)}) + continue + if self.coalescing_enabled and tensor in self.pending_tensor: + group_id = self.pending_tensor[tensor] + self.groups[group_id]["task_ids"].append(task_id) + self.offered.add(task_id) + self.events.append({"kind": "coalesced", "task_id": task_id, + "group_id": group_id, "logical_issue_ns": issue_ns}) + continue + external_hit = self.cache_mode == "external_hbm" and tensor in self.cache + if external_hit: + self.cache.move_to_end(tensor) + self.events.append({"kind":"cache_hit","task_id":task_id,"tensor_id":tensor, + "at_ns":now_ns,"requires_hbm_service":True}) + tier = "fast" if external_hit else self.tensor_tier.get(tensor, "source") + group_id = "causal:" + task_id + parts = self._stripe_parts(task["tensor"]["bytes"]) + self.offered.add(task_id) + self.pending_tensor[tensor] = group_id + self.groups[group_id] = {"task_ids": [task_id], "pending": set(), + "completion_ns": issue_ns, "tensor_id": tensor, + "tensor_bytes": task["tensor"]["bytes"], + "child_count":len(parts), + "retry_remaining":0 if external_hit else self.retry_count, + "retry_sequence":0,"retry_parts":[], + "operation": "read", + "cache_hit": external_hit, + "tier": tier, + "blocks_compute": ( + self.prefetch_wait_mode == "stall_at_issue" + and consume_ready is None)} + for partition, child_bytes in parts: + place = self._placement_provider(task["tensor"], tier, partition) + job_id = group_id + f":part{partition}" + job = {"job_id": job_id, "operation": "read", "bytes": child_bytes, + "arrival_ns": issue_ns, **place, + "metadata": {"logical_issue_ns": logical_issue_ns, "tensor_id": tensor, + "parent_group_id": group_id, + "partition_index": partition, + "partition_count": len(parts), + "stripe_unit_bytes": self.stripe_unit_bytes, + "stripe_semantics": "ROUND_ROBIN_UNITS_AGGREGATED_PER_TARGET", + "batch_consumer_count": task["consumer_count"], + "batch_interval_id": task["batch_interval_id"]}} + self.groups[group_id]["pending"].add(job_id) + self.groups[group_id]['retry_parts'].append((partition,child_bytes,deepcopy(place))) + self.job_group[job_id] = group_id + self.job_bytes[job_id] = child_bytes + self.job_arrival[job_id] = issue_ns + self.jobs.append(deepcopy(job)); result.append(job) + self._resolve_ready_storage(now_ns) + self._start_compute(now_ns) + return result + + def complete(self, job_id: str, completion_ns: int, completed_bytes: int) -> None: + """Consume an actual external completion and unlock dependent work.""" + if job_id not in self.job_bytes or completed_bytes != self.job_bytes[job_id]: + raise ValueError("unknown job or byte-incomplete external completion") + if not isinstance(completion_ns, int) or completion_ns < 0: + raise ValueError("completion_ns must be non-negative integer ns") + if completion_ns < self.job_arrival[job_id]: + raise ValueError("external completion precedes causal job arrival") + self.job_bytes.pop(job_id) + self.job_arrival.pop(job_id) + group_id = self.job_group.pop(job_id) + group = self.groups[group_id] + group["pending"].remove(job_id) + group["completion_ns"] = max(group["completion_ns"], completion_ns) + self.events.append({"kind": "storage_child_complete", "job_id": job_id, + "group_id": group_id, "completion_ns": completion_ns, + "completed_bytes": completed_bytes}) + if group["pending"]: + return + if group['operation']=='read' and group['retry_remaining']: + group['retry_remaining']-=1;group['retry_sequence']+=1 + for partition,size,place in group['retry_parts']: + retry_id=f"{group_id}:retry{group['retry_sequence']}:part{partition}" + job={'job_id':retry_id,'operation':'retry','bytes':size,'arrival_ns':completion_ns,**place, + 'metadata':{'parent_group_id':group_id,'tensor_id':group['tensor_id'], + 'retry_sequence':group['retry_sequence'], + 'evidence':'SSD_INSPIRED_FIXED_RETRY_COST_SCENARIO_NOT_HBF_RBER'}} + group['pending'].add(retry_id);self.job_group[retry_id]=group_id + self.job_bytes[retry_id]=size;self.job_arrival[retry_id]=completion_ns + self.jobs.append(deepcopy(job));self.deferred_jobs.append(job) + return + if group['operation'].startswith('migration_'): + self._complete_migration_phase(group_id, group, completion_ns) + return + if group["operation"] == "cache_fill": + tensor = group["tensor_id"] + self.cache_reservations.pop(tensor) + self._insert_cache(tensor, group["tensor_bytes"]) + self.events.append({"kind":"cache_fill_complete","tensor_id":tensor, + "completion_ns":completion_ns,"bytes":group["tensor_bytes"]}) + self.groups.pop(group_id) + return + tasks = group["task_ids"] + tensor = group["tensor_id"] + if self.pending_tensor.get(tensor) == group_id: + self.pending_tensor.pop(tensor) + completion_ns = group["completion_ns"] + if group["blocks_compute"]: + self.events.append({"kind": "prefetch_issue_stall_released", + "group_id": group_id, "completion_ns": completion_ns}) + if self.cache_mode == "ideal_metadata_only": + self._insert_cache(tensor, group["tensor_bytes"]) + elif not group["cache_hit"]: + self._offer_cache_fill(tensor,group["tensor_bytes"],completion_ns) + for task_id in tasks: + consume_ready = self._deps_time(self.tasks[task_id]["consume_after"]) + if consume_ready is None: + # Storage is ready; consumption is finalized when its dependency resolves. + self.tasks[task_id]["external_ready_ns"] = completion_ns + self.tasks[task_id]["ready_source"] = "external_service" + else: + self.done[task_id] = max(completion_ns, consume_ready) + self.events.append({"kind": "storage_consumed", "task_id": task_id, + "tensor_id": tensor, "ready_ns": completion_ns, + "consume_ns": self.done[task_id], + "source": "external_service"}) + self.access_count[tensor] = self.access_count.get(tensor, 0) + 1 + self.events.append({"kind": "storage_complete", "group_id": group_id, + "tensor_id": tensor, "completion_ns": completion_ns, + "completed_bytes": group["tensor_bytes"], + "retry_count":group['retry_sequence'], + "child_count": group['child_count'], + "task_ids": list(tasks)}) + self.groups.pop(group_id) + self._resolve_ready_storage(completion_ns) + + def _resolve_ready_storage(self, now_ns: int) -> None: + changed = True + while changed: + changed = False + self._settle_compute(now_ns) + for task_id, task in self.tasks.items(): + if task_id in self.done or "external_ready_ns" not in task: + continue + consume_ready = self._deps_time(task["consume_after"]) + if consume_ready is not None: + ready_ns = task.pop("external_ready_ns") + source = task.pop("ready_source") + self.done[task_id] = max(ready_ns, consume_ready) + self.events.append({"kind": "storage_consumed", "task_id": task_id, + "tensor_id": task["tensor"]["tensor_id"], + "ready_ns": ready_ns, "consume_ns": self.done[task_id], + "source": source}) + changed = True + + def offer_migrations(self, now_ns: int) -> list[dict[str, Any]]: + """Move repeatedly accessed immutable weights through explicit copy phases.""" + if self.migration_mode != "basic": + return [] + result = [] + tensors = self.tensor_specs + for tensor, count in sorted(self.access_count.items()): + if count < self.migration_threshold or self.tensor_tier.get(tensor) == "fast" \ + or tensor in self.pending_migrations: + continue + spec = tensors[tensor] + allocation = sum(((size+1048575)//1048576)*1048576 for _,size in self._stripe_parts(spec['bytes'])) + if self.migration_used_bytes + allocation > self.migration_capacity_bytes: + continue + group_id = "causal:migration:" + tensor + self.pending_migrations[tensor] = group_id + self.migration_used_bytes += allocation + transfer = {'tensor_id':tensor,'tensor_bytes':spec['bytes'], + 'allocation_bytes':allocation, + 'spec':deepcopy(spec),'expected_version':self.tensor_versions.get(tensor,0), + 'root_id':group_id,'operation':'migration_source'} + result.extend(self._migration_jobs(transfer,now_ns)) + return result + + def _migration_jobs(self, transfer, now_ns): + phase = transfer['operation'] + group_id = transfer['root_id'] + ':' + phase + group = {**transfer,'task_ids':[],'pending':set(),'completion_ns':now_ns} + self.groups[group_id] = group + jobs = [] + tier = 'source' if phase in ('migration_source','migration_erase_old') else 'fast' + for partition,size in self._stripe_parts(transfer['tensor_bytes']): + place = self._placement_provider(transfer['spec'],tier,partition) + if phase == 'migration_source':operation='read' + elif phase == 'migration_destination': + operation='hbm_fill' if place['stack'].startswith('hbm') else 'migration_program' + else:operation='erase' + block_count = (size + 1048575)//1048576 + if operation=='erase':size=block_count*1048576 + job_id=group_id+f':part{partition}' + job={'job_id':job_id,'maintenance_id':job_id,'operation':operation, + 'bytes':size,'arrival_ns':now_ns,**place, + 'metadata':{'purpose':'MIGRATION','parent_group_id':group_id, + 'tensor_id':transfer['tensor_id'],'phase':phase, + 'logical_address_bytes':transfer['spec'].get('logical_address_bytes'), + 'block_count':block_count,'block_bytes':1048576, + 'placement_evidence':'CONDITIONAL_DEDICATED_BLOCK_RANGES_NO_NATIVE_FTL_CLAIM'}} + group['pending'].add(job_id) + self.job_group[job_id]=group_id;self.job_bytes[job_id]=size;self.job_arrival[job_id]=now_ns + self.jobs.append(deepcopy(job));jobs.append(job) + return jobs + + def _complete_migration_phase(self, group_id, group, now_ns): + phase=group['operation'];tensor=group['tensor_id'] + self.groups.pop(group_id) + self.events.append({'kind':'migration_phase_complete','tensor_id':tensor,'phase':phase, + 'completion_ns':now_ns,'bytes':group['tensor_bytes']}) + if phase=='migration_source': + group['operation']='migration_destination' + elif phase=='migration_destination': + valid=self.tensor_versions.get(tensor,0)==group['expected_version'] + if valid: + self.tensor_tier[tensor]='fast' + self.tensor_versions[tensor]=group['expected_version']+1 + group['operation']='migration_erase_old' + else: + group['operation']='migration_erase_destination' + self.events.append({'kind':'migration_commit','tensor_id':tensor,'at_ns':now_ns, + 'committed':valid,'source_valid_until_commit':True}) + else: + if phase=='migration_erase_destination':self.migration_used_bytes-=group['allocation_bytes'] + self.pending_migrations.pop(tensor,None) + return + if group['operation']=='migration_erase_old': + self.deferred_source_erase.append(group) + else: + self.deferred_jobs.extend(self._migration_jobs(group,now_ns)) + + def result(self, now_ns: int) -> dict[str, Any]: + self._resolve_ready_storage(now_ns) + tokens = [] + for batch in self.trace["batches"]: + terminal = batch["terminal_task_id"] + for token_id in batch["token_ids"]: + tokens.append({"token_id": token_id, + "completion_ns": self.done.get(terminal), + "complete": terminal in self.done}) + return {"schema_version": "eq3-causal-execution-v1", + "trace_origin": TRACE_ORIGIN, "tokens": tokens, + "retired_completed_batches":deepcopy(self.completed_batches), + "jobs": deepcopy(self.jobs), "events": deepcopy(self.events), + "pending_external_jobs": sorted(self.job_bytes), + "capabilities": {"coalescing_enabled": self.coalescing_enabled, + "prefetch_wait_mode": self.prefetch_wait_mode, + "cache": {"ideal_metadata_only":"IDEAL_DIAGNOSTIC_NOT_EXTERNAL_HBM_SERVICE", + "external_hbm":"SERVICE_BACKED_HBM_FILL_AND_HIT", + "disabled":"DISABLED"}[self.cache_mode], + "basic_migration": "SOURCE_READ_DESTINATION_WRITE_VERSION_COMMIT_OLD_ERASE"}, + "unavailable": ["TOKEN_PER_SECOND_CALIBRATION", "NAND_COMMAND_TIMING"]} + + +__all__ = ["CausalExecutor", "build_architecture_trace", "load_architecture"] diff --git a/experiments/eq3_system_thermal/ecc_cost_proxy.py b/experiments/eq3_system_thermal/ecc_cost_proxy.py new file mode 100644 index 0000000..941c4ec --- /dev/null +++ b/experiments/eq3_system_thermal/ecc_cost_proxy.py @@ -0,0 +1,78 @@ +"""Temperature-history to equivalent read effort, explicitly conditional. + +No bits are invented, no HBF error probability is predicted, and lowering +instantaneous read temperature never erases already accumulated retention age. +""" +from copy import deepcopy +import math + +DAY_NS=86_400_000_000_000 + +class ProxyDomainError(ValueError): + pass + +class ReadCostProxy: + def __init__(self, profile, initial_by_stack): + self.profile=deepcopy(profile) + if profile['classification']!='CONDITIONAL_NAND_BEHAVIOR_PROXY_NOT_HBF_RBER': + raise ValueError('explicit conditional classification required') + strength=profile['transfer_strength'] + if not math.isfinite(strength) or strength<0: + raise ValueError('invalid transfer strength') + p=profile + # Interpolant goes through19.9 at365days/2kPE and14.5 at90days/2kPE. + at90=p['retention_retry_coefficient_at_zero_pe']+2*p['retention_retry_coefficient_per_1000_pe'] + extra=p['mean_retry_anchor_at_365_days_2000_pe']-2*p['fresh_pe_retry_per_1000_cycles'] + self.exponent=math.log(extra/at90)/math.log(365/p['mean_retry_anchor_age_days']) + if self.exponent<=0:raise ValueError('retention exponent must be positive') + self.states={} + for stack,row in initial_by_stack.items(): + if not stack.startswith('hbf'):raise ValueError('NAND proxy cannot apply to HBM') + self.states[stack]={'last_ns':0,'equivalent_age_ns':float(row['equivalent_age_days_30c'])*DAY_NS, + 'temperature_k':float(row['temperature_k']),'pe_cycles':float(row['pe_cycles']), + 'extrapolated_temperature_ns':0} + self.cost(stack,0) + + def acceleration(self,temp): + if not math.isfinite(temp) or temp<=0:raise ValueError('invalid Kelvin temperature') + p=self.profile + return math.exp(p['activation_energy_ev']/p['boltzmann_ev_per_k']* + (1/p['temperature_reference_k']-1/temp)) + + def observe(self, start_ns, end_ns, temperatures): + if end_ns original else + "DECREASE" if budget < original else "HOLD") + decision = StackDecision(self.stack_id, budget, action, outcome, reasons) + return Decision( + self.profile.enabled, + self.profile.strategy, + facts.end_ns, + (decision,), + dict(facts.shared_endpoint_caps_bytes), + fact_semantics=("HBM shared-endpoint cap consumes thermal guard state only; " + "no HBM foreground delivery is inferred"), + effect_semantics="future shared-endpoint byte admission only", + ) diff --git a/experiments/eq3_system_thermal/energy.py b/experiments/eq3_system_thermal/energy.py new file mode 100644 index 0000000..fbfe73d --- /dev/null +++ b/experiments/eq3_system_thermal/energy.py @@ -0,0 +1,103 @@ +"""Disjoint incremental-energy scopes for the isolated system experiment. + +Coefficients are scenario/proxy inputs, never facts inferred from an address. +The caller supplies observed/modelled activity identity and explicit placement. +""" +from __future__ import annotations + +from collections import defaultdict +import math + + +def engineering_energy_profile(): + return { + "read_array_j_per_byte": 40e-12, + "read_base_j_per_byte": 10e-12, + "program_array_j_per_byte": 0.05 * 100e-6 / 4096, + "program_base_j_per_byte": 0.01 * 100e-6 / 4096, + "erase_array_j_per_operation": None, + "relay_partner_receive_j_per_byte": 2e-12, + "relay_partner_send_j_per_byte": 2e-12, + "hbm_array_j_per_byte": 40e-12, + "hbm_base_j_per_byte": 2e-12, + "ecc_j_per_byte": 0.0, + "ecc_scope": "ZERO_INCREMENT_BASELINE_NOT_MEASURED_ZERO; scenario overrides explicit", + "evidence": { + "read": "USER_CONFIRMED_40_ARRAY_10_BASE_PJ_PER_B", + "program": "DERIVED_ENGINEERING_PROXY_OLD_0.05W_0.01W_100US_4KIB; NOT_HBF_CALIBRATION", + "relay_hbm": "REUSED_EQ3_MAINTENANCE_ENERGY_PROFILE_SCENARIO", + "erase": "UNKNOWN_REQUIRES_EXPLICIT_DURATION_AND_POWER", + }, + } + + +class EnergyMapper: + def __init__(self, normalized, channel_map, profile): + self.components = {c["id"] for c in normalized["components"]} + self.channels = channel_map + self.hbm_dies = defaultdict(list) + for row in normalized["components"]: + if str(row.get("physical_type", "")).startswith("HBM") and row.get("role") == "array_die": + self.hbm_dies[row["device_id"]].append(row["id"]) + self.profile = profile + + def map(self, activities, *, gpu_compute_j=0.0, gpu_external_j=0.0): + sources = defaultdict(float) + scopes = defaultdict(float) + seen = set() + + def add(component, joules, scope): + if component not in self.components: + raise ValueError(f"energy component absent: {component}") + if not math.isfinite(joules) or joules < 0: + raise ValueError("invalid energy") + sources[component] += joules + scopes[scope] += joules + + for row in activities: + if "activity_id" in row: + if row["activity_id"] in seen: + raise ValueError("duplicate activity identity in energy window") + seen.add(row["activity_id"]) + op, stack, size = row["operation"], row["stack"], row["bytes"] + if type(size) is not int or size < 0: + raise ValueError("activity bytes must be nonnegative integer") + if op in {"read", "retry", "refresh_read", "migration_read"}: + channel = str(row["channel"]) + die = self.channels[stack][channel] + add(die, size * self.profile["read_array_j_per_byte"], op + ":array") + add(stack + ".base", size * self.profile["read_base_j_per_byte"], op + ":base") + elif op in {"program", "refresh_program", "migration_program"}: + die = self.channels[stack][str(row["channel"])] + add(die, size * self.profile["program_array_j_per_byte"], op + ":array") + add(stack + ".base", size * self.profile["program_base_j_per_byte"], op + ":base") + elif op == "erase": + coefficient = self.profile["erase_array_j_per_operation"] + if coefficient is None: + raise ValueError("erase energy UNKNOWN; explicit operation coefficient required") + count = row["operations"] + if type(count) is not int or count < 0: + raise ValueError("invalid erase count") + add(self.channels[stack][str(row["channel"])], count * coefficient, "erase:array") + elif op == "relay_receive": + add(stack + ".base", size * self.profile["relay_partner_receive_j_per_byte"], "relay:receive") + elif op == "relay_send": + add(stack + ".base", size * self.profile["relay_partner_send_j_per_byte"], "relay:send") + elif op == "hbm_read": + dies = self.hbm_dies[stack] + if not dies: + raise ValueError("HBM array source unavailable") + # Parametric service observes stack only; uniform thermal allocation + # is a declared scenario, not claimed observed physical die identity. + for die in dies: + add(die, size * self.profile["hbm_array_j_per_byte"] / len(dies), "hbm:array_uniform_proxy") + add(stack + ".base", size * self.profile["hbm_base_j_per_byte"], "hbm:base") + elif op == "ecc": + add(stack + ".base", size * self.profile["ecc_j_per_byte"], "ecc:base") + else: + raise ValueError(f"unsupported energy activity: {op}") + add("gpu", gpu_compute_j, "gpu:causal_compute") + add("gpu", gpu_external_j, "gpu:independent_external") + return {"component_energy_j": dict(sources), "scope_energy_j": dict(scopes), + "total_j": sum(sources.values()), + "evidence": "CONDITIONAL_INCREMENTAL_MODELLED_ACTIVITY_NOT_CALIBRATED_POWER"} diff --git a/experiments/eq3_system_thermal/freeze_extension.py b/experiments/eq3_system_thermal/freeze_extension.py new file mode 100644 index 0000000..cc9b7e1 --- /dev/null +++ b/experiments/eq3_system_thermal/freeze_extension.py @@ -0,0 +1,64 @@ +#!/usr/bin/env python3 +"""Bind a reviewed extension input list to the existing serial launcher.""" +import argparse +from copy import deepcopy +import hashlib +import json +from pathlib import Path + +HERE=Path(__file__).resolve().parent +ROOT=HERE.parents[1] + +def digest(path):return hashlib.sha256(Path(path).read_bytes()).hexdigest() +def save(path,obj):Path(path).write_text(json.dumps(obj,indent=2,allow_nan=False)+'\n') + +def freeze(stage,destination,configs,runner,*,point_wall_s=900,dependency=None): + destination.mkdir(parents=True,exist_ok=True) + base=json.loads((stage/'RUN_INDEX.json').read_text()) + templates={r['topology']:r for r in base['points']} + names=['endpoint_policy.py','topology_service.py','energy.py','rate_workload.py','run_system_point.py', + 'maintenance_driver.py','reliability.py',runner.name] + if runner.name=='run_causal_point.py': + names+=['causal_service.py','causal_workload.py','causal_maintenance_age.py','tiny_cpu_trace.py'] + sources={str((HERE/n).relative_to(ROOT)):digest(HERE/n) for n in sorted(set(names))} + for n in ('thermal_client.py','read_rate_policy.py'): + p=ROOT/'experiments'/'eq3_maintenance'/n + sources[str(p.relative_to(ROOT))]=digest(p) + sources['tools/eq3_basic_fabric.py']=digest(ROOT/'tools'/'eq3_basic_fabric.py') + if runner.name=='run_causal_point.py': + catalogue=ROOT/'experiments/eq3_maintenance/sources/qwen2_5_weight_models.json' + sources[str(catalogue.relative_to(ROOT))]=digest(catalogue) + points=[] + for path in configs: + cfg=json.loads(path.read_text());row=deepcopy(templates[cfg['topology']]) + if cfg.get('hbf_read_cost_proxy',{}).get('mode','disabled')!='disabled': + for name in ('ecc_cost_proxy.py','ecc_service_adapter.py'): + sources[str((HERE/name).relative_to(ROOT))]=digest(HERE/name) + row.update(point_id=cfg['point_id'],config=str(path.resolve()),config_sha256=digest(path), + output=str(destination/'points'/cfg['point_id'])) + points.append(row) + if cfg.get('trace',{}).get('dependency_mode')=='tiny_cpu_template': + trace_path=ROOT/cfg['trace']['tiny_trace_path'] + sources[str(trace_path.relative_to(ROOT))]=digest(trace_path) + locks={m:{n:digest(Path(m)/n) for n in ('model.txt','normalized.json','rc_grid.json','rc_sensors.json')} + for m in sorted({r['model_dir'] for r in points})} + result={'schema_version':'eq3-extension-frozen-index-v1','status':'PENDING_DEPENDENCIES_BASE_MATRIX', + 'authorization':'USER_EXPLICIT_FOUR_TOPOLOGY_PLAN','runner':str(runner),'runner_sha256':digest(runner), + 'dependencies':{'base_done':str(dependency or stage/'BASE_DONE.json')}, + 'runtime_source_locks_sha256':sources,'model_locks_sha256':locks, + 'thermal_binary_sha256':digest(points[0]['thermal_binary']),'points':points, + 'resources':{'point_wall_s':point_wall_s,'stage_wall_s':21600,'point_output_gib':2, + 'sensitivity_output_gib':30,'parent_combined_output_gib':100, + 'host_ram_reserve_gib':32,'host_disk_reserve_gib':100}, + 'failure_contract':'DOMAIN_FAILURE_RETAINED_CONTINUE; OTHER_FAILURE_STOP_AND_DIAGNOSE', + 'cpu_ownership':'ROOT_AFTER_P2_DIAGNOSTICS_QUEUE_RELEASE_ONLY'} + return result + +if __name__=='__main__': + p=argparse.ArgumentParser(description=__doc__) + p.add_argument('--stage',type=Path,required=True);p.add_argument('--destination',type=Path,required=True) + p.add_argument('--runner',type=Path,required=True);p.add_argument('--config',type=Path,action='append',required=True) + p.add_argument('--point-wall-s',type=int,default=900);p.add_argument('--output',type=Path,required=True) + a=p.parse_args() + if a.output.exists():raise FileExistsError(a.output) + save(a.output,freeze(a.stage.resolve(),a.destination.resolve(),a.config,a.runner.resolve(),point_wall_s=a.point_wall_s)) diff --git a/experiments/eq3_system_thermal/launch_sensitivity.py b/experiments/eq3_system_thermal/launch_sensitivity.py new file mode 100644 index 0000000..788538e --- /dev/null +++ b/experiments/eq3_system_thermal/launch_sensitivity.py @@ -0,0 +1,171 @@ +#!/usr/bin/env python3 +"""Serial launcher for the separately frozen sensitivity index.""" + +from __future__ import annotations + +import argparse +import hashlib +import json +import os +from pathlib import Path +import resource +import shutil +import signal +import subprocess +import sys +import time + + +GIB = 1024**3 +ROOT = Path(__file__).resolve().parents[2] + + +def save(path: Path, value: dict) -> None: + temp = path.with_suffix(path.suffix + ".tmp") + temp.write_text(json.dumps(value, indent=2, sort_keys=True) + "\n") + temp.replace(path) + + +def digest(path: Path) -> str: + return hashlib.sha256(path.read_bytes()).hexdigest() + + +def size(path: Path) -> int: + return sum(row.stat().st_size for row in path.rglob("*") if row.is_file()) + + +def available_memory() -> int: + rows = dict(line.split(":", 1) for line in Path("/proc/meminfo").read_text().splitlines()) + return int(rows["MemAvailable"].split()[0]) * 1024 + + +def domain_failure(output: Path) -> bool: + transcript = output / "thermal-process" / "thermal-transcript.jsonl" + if not transcript.is_file() or not (output / "FAILED.json").is_file(): + return False + for line in transcript.read_text().splitlines(): + value = json.loads(line) + response = value.get("response") + if isinstance(response, dict) and response.get("type") == "ERROR": + if response.get("status") == "DOMAIN_FAILURE": + return True + return False + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--index", type=Path, required=True) + args = parser.parse_args() + index_path = args.index.resolve(strict=True) + index = json.loads(index_path.read_text()) + stage = index_path.parent + parent_stage = stage.parent + resources = index["resources"] + if index.get("status") != "PENDING_DEPENDENCIES_BASE_MATRIX": + raise ValueError("sensitivity index status is not the frozen prelaunch state") + dependency = Path(index["dependencies"]["base_done"]) + if not dependency.is_file() or json.loads(dependency.read_text()).get("status") != "COMPLETED": + raise RuntimeError("base matrix dependency is not complete") + runner = Path(index["runner"]) + if digest(runner) != index["runner_sha256"]: + raise ValueError("frozen sensitivity runner changed") + for relative, expected in index["runtime_source_locks_sha256"].items(): + if digest(ROOT / relative) != expected: + raise ValueError(f"runtime source changed: {relative}") + for directory, locks in index["model_locks_sha256"].items(): + for name, expected in locks.items(): + if digest(Path(directory) / name) != expected: + raise ValueError(f"thermal model changed: {directory}/{name}") + binary = Path(index["points"][0]["thermal_binary"]) + if digest(binary) != index["thermal_binary_sha256"]: + raise ValueError("thermal binary changed") + + env = dict(os.environ, OMP_NUM_THREADS="1", OPENBLAS_NUM_THREADS="1", + MKL_NUM_THREADS="1", NUMEXPR_NUM_THREADS="1", CUDA_VISIBLE_DEVICES="", + PYTHONDONTWRITEBYTECODE="1") + started = time.monotonic() + completed, domain_failures = [], [] + for row in index["points"]: + output = Path(row["output"]) + if output.exists(): + raise FileExistsError(f"preserve existing sensitivity evidence: {output}") + config = Path(row["config"]) + if digest(config) != row["config_sha256"]: + raise ValueError(f"input freeze mismatch: {row['point_id']}") + if available_memory() < resources["host_ram_reserve_gib"] * GIB: + raise RuntimeError("host memory reserve") + if shutil.disk_usage(parent_stage).free < resources["host_disk_reserve_gib"] * GIB: + raise RuntimeError("host disk reserve") + if size(stage) > resources["sensitivity_output_gib"] * GIB: + raise RuntimeError("sensitivity retained-output budget") + if size(parent_stage) > resources["parent_combined_output_gib"] * GIB: + raise RuntimeError("parent combined retained-output budget") + if time.monotonic() - started > resources["stage_wall_s"]: + raise RuntimeError("sensitivity stage watchdog") + + launch = stage / "launch" / row["point_id"] + launch.mkdir(parents=True, exist_ok=False) + command = [sys.executable, "-B", str(runner)] + for key in ("config", "model_dir", "thermal_binary", "artifact_root", "output"): + command += ["--" + key.replace("_", "-"), row[key]] + save(launch / "launch.json", {"command": command, "point": row, + "resources": resources}) + save(stage / "STATUS.json", {"status": "RUNNING", "active": row["point_id"], + "completed": completed, "domain_failures": domain_failures}) + wall = time.monotonic(); reason = None + with (launch / "stdout.log").open("w") as stdout, (launch / "stderr.log").open("w") as stderr: + child = subprocess.Popen(command, stdout=stdout, stderr=stderr, env=env, + start_new_session=True) + while True: + try: + code = child.wait(timeout=min(15, max(.1, resources["point_wall_s"] - + (time.monotonic() - wall)))) + break + except subprocess.TimeoutExpired: + if time.monotonic() - wall >= resources["point_wall_s"]: + reason = "POINT_WATCHDOG" + elif time.monotonic() - started >= resources["stage_wall_s"]: + reason = "STAGE_WATCHDOG" + elif size(output) > resources["point_output_gib"] * GIB: + reason = "POINT_OUTPUT_BUDGET" + elif size(stage) > resources["sensitivity_output_gib"] * GIB: + reason = "SENSITIVITY_OUTPUT_BUDGET" + elif size(parent_stage) > resources["parent_combined_output_gib"] * GIB: + reason = "PARENT_COMBINED_OUTPUT_BUDGET" + elif available_memory() < resources["host_ram_reserve_gib"] * GIB: + reason = "HOST_RAM_RESERVE" + elif shutil.disk_usage(parent_stage).free < resources["host_disk_reserve_gib"] * GIB: + reason = "HOST_DISK_RESERVE" + if reason: + os.killpg(child.pid, signal.SIGTERM) + try: + code = child.wait(timeout=5) + except subprocess.TimeoutExpired: + os.killpg(child.pid, signal.SIGKILL); code = child.wait() + break + result = {"point_id": row["point_id"], "exit_code": code, "reason": reason, + "wall_s": time.monotonic() - wall, + "output_bytes": size(output) if output.exists() else 0} + if code == 0 and (output / "DONE.json").is_file(): + result["classification"] = "COMPLETED" + completed.append(result) + elif reason is None and domain_failure(output): + result["classification"] = "DOMAIN_FAILURE" + domain_failures.append(result) + else: + result["classification"] = "EXECUTION_FAILURE" + save(launch / "result.json", result) + save(stage / "FAILED.json", {"failed": result, "completed": completed, + "domain_failures": domain_failures}) + return 1 + save(launch / "result.json", result) + receipt = {"status": "COMPLETED_WITH_RETAINED_DOMAIN_FAILURES" if domain_failures else "COMPLETED", + "completed": completed, "domain_failures": domain_failures, + "wall_s": time.monotonic() - started} + save(stage / "DONE.json", receipt) + save(stage / "STATUS.json", receipt) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/experiments/eq3_system_thermal/launch_stage.py b/experiments/eq3_system_thermal/launch_stage.py new file mode 100644 index 0000000..01f3d4a --- /dev/null +++ b/experiments/eq3_system_thermal/launch_stage.py @@ -0,0 +1,90 @@ +#!/usr/bin/env python3 +"""Serial bounded stage executor; status files, no AI polling required.""" +import argparse +import hashlib +import json +import os +from pathlib import Path +import shutil +import signal +import subprocess +import sys +import time + +HERE=Path(__file__).resolve().parent +GIB=1024**3 + + +def save(path,obj): + temp=path.with_suffix('.tmp');temp.write_text(json.dumps(obj,indent=2)+'\n');temp.replace(path) + + +def size(path):return sum(p.stat().st_size for p in path.rglob('*') if p.is_file()) + + +def main(): + p=argparse.ArgumentParser(description=__doc__) + p.add_argument('--index',type=Path,required=True) + p.add_argument('--kind',choices=['pilot','base'],required=True) + a=p.parse_args();index=json.loads(a.index.read_text());stage=a.index.parent + limits=index['resources'];started=time.monotonic();results=[] + for name,expected in index['source_locks'].items(): + if hashlib.sha256((HERE/name).read_bytes()).hexdigest()!=expected: + raise ValueError('source changed after freeze: '+name) + env=dict(os.environ,OMP_NUM_THREADS='1',OPENBLAS_NUM_THREADS='1',MKL_NUM_THREADS='1', + NUMEXPR_NUM_THREADS='1',CUDA_VISIBLE_DEVICES='',PYTHONDONTWRITEBYTECODE='1') + if a.kind=='base': + review=json.loads((stage/'PILOT_REVIEW.json').read_text()) + if review.get('status')!='PASS':raise RuntimeError('pilot review not passed') + points=[r for r in index['points'] if r['kind']==a.kind] + for row in points: + output=Path(row['output']) + if output.exists():raise FileExistsError('preserve existing evidence: '+str(output)) + if hashlib.sha256(Path(row['config']).read_bytes()).hexdigest()!=row['config_sha256']: + raise ValueError('input freeze mismatch') + if shutil.disk_usage(stage).freelimits['stage_wall_s']:raise RuntimeError('stage watchdog') + launch=stage/'launch'/row['point_id'];launch.mkdir(parents=True,exist_ok=False) + command=[sys.executable,'-B',str(HERE/index.get('runner','run_system_point.py'))] + for key in ('config','model_dir','thermal_binary','artifact_root','output'): + command += ['--'+key.replace('_','-'),row[key]] + save(launch/'launch.json',{'command':command,'point':row,'limits':limits}) + save(stage/'STATUS.json',{'status':'RUNNING','active':row['point_id'],'completed':results}) + wall=time.monotonic();reason=None + with (launch/'stdout.log').open('w') as out,(launch/'stderr.log').open('w') as err: + child=subprocess.Popen(command,stdout=out,stderr=err,env=env,start_new_session=True) + while True: + try: + code=child.wait(timeout=min(15,max(.1,limits['point_wall_s']-(time.monotonic()-wall)))) + break + except subprocess.TimeoutExpired: + if time.monotonic()-wall>=limits['point_wall_s']:reason='POINT_WATCHDOG' + elif time.monotonic()-started>=limits['stage_wall_s']:reason='STAGE_WATCHDOG' + elif size(output)>limits['point_output_gib']*GIB:reason='POINT_OUTPUT_BUDGET' + elif size(stage)>limits['stage_output_gib']*GIB:reason='STAGE_OUTPUT_BUDGET' + elif shutil.disk_usage(stage).freelimits['point_output_gib']*GIB or size(stage)>limits['stage_output_gib']*GIB: + raise RuntimeError('retained output exceeds finite budget') + save(stage/(a.kind.upper()+'_DONE.json'),{'status':'COMPLETED','results':results,'wall_s':time.monotonic()-started}) + save(stage/'STATUS.json',{'status':'COMPLETED_'+a.kind.upper(),'completed':results}) + return 0 + + +if __name__=='__main__':raise SystemExit(main()) diff --git a/experiments/eq3_system_thermal/maintenance_driver.py b/experiments/eq3_system_thermal/maintenance_driver.py new file mode 100644 index 0000000..ec927de --- /dev/null +++ b/experiments/eq3_system_thermal/maintenance_driver.py @@ -0,0 +1,391 @@ +"""Bounded-spare maintenance state machine over TopologyService extra jobs. + +The driver owns metadata versions and physical-block allocation only. It does +not emulate NAND timing: read/program/erase work completes solely from service +receipts. Callers advance the ReliabilityLedger through the receipt end time +before consuming that receipt so commit-time age reset remains exact. +""" + +from __future__ import annotations + +from collections import defaultdict, deque +from copy import deepcopy +import math +from typing import Any + +try: + from .reliability import ReliabilityLedger +except ImportError: # Direct fixed-test execution from this directory. + from reliability import ReliabilityLedger + + +class MaintenanceDriver: + def __init__(self, reliability: ReliabilityLedger, config: dict[str, Any]): + self.reliability = reliability + self.block_bytes = self._positive(config.get("block_bytes"), "block_bytes") + self.pages_per_block = self._positive(config.get("pages_per_block"), "pages_per_block") + self.max_cohort = self._positive(config.get("max_blocks_per_cohort"), + "max_blocks_per_cohort") + raw_spares = config.get("spare_block_ids_by_stack_channel") + if not isinstance(raw_spares, dict) or not raw_spares: + raise ValueError("explicit spare_block_ids_by_stack_channel is required") + self.free_spares: dict[tuple[str, str], deque[str]] = {} + all_spares = set() + for stack, channels in raw_spares.items(): + if not isinstance(stack, str) or not stack or not isinstance(channels, dict) or not channels: + raise ValueError("each stack needs explicit per-channel spare pools") + for channel, values in channels.items(): + channel = str(channel) + if not channel or not isinstance(values, list) or not values: + raise ValueError("each configured channel needs nonempty spare block IDs") + if any(not isinstance(value, str) or not value for value in values): + raise ValueError("spare block IDs must be nonempty strings") + if len(set(values)) != len(values) or all_spares.intersection(values): + raise ValueError("spare block IDs must be globally unique") + all_spares.update(values) + self.free_spares[(stack, channel)] = deque(values) + self.program_j_per_byte = self._optional_nonnegative( + config.get("program_energy_j_per_byte"), "program_energy_j_per_byte") + self.erase_j_per_operation = self._optional_nonnegative( + config.get("erase_energy_j_per_operation"), "erase_energy_j_per_operation") + self.energy_evidence = config.get("energy_evidence") + if not isinstance(self.energy_evidence, dict): + raise ValueError("energy_evidence must explicitly classify program and erase") + if set(self.energy_evidence) != {"program", "erase"}: + raise ValueError("energy_evidence must exactly cover program and erase") + self.extents: dict[str, dict[str, Any]] = {} + self.physical_wear: dict[str, dict[str, int]] = defaultdict( + lambda: {"block_program_work_started": 0, "block_program_work_completed": 0, + "nand_page_programs_started": 0, "nand_page_programs_completed": 0, + "erase_phase_started": 0, "erase_completed": 0}) + self.operations: dict[str, dict[str, Any]] = {} + self.outstanding: dict[str, tuple[str, str]] = {} + self.active_extents: set[str] = set() + self.quarantined_blocks: set[str] = set() + self.events: list[dict[str, Any]] = [] + self.energy_facts: list[dict[str, Any]] = [] + self.results: list[dict[str, Any]] = [] + self._terminal_status_counts: dict[str, int] = defaultdict(int) + self._terminal_extent_count = 0 + self._sequence = 0 + + @staticmethod + def _positive(value: Any, label: str) -> int: + if isinstance(value, bool) or not isinstance(value, int) or value <= 0: + raise ValueError(f"{label} must be a positive integer") + return value + + @staticmethod + def _optional_nonnegative(value: Any, label: str) -> float | None: + if value is None: + return None + if not isinstance(value, (int, float)) or not math.isfinite(value) or value < 0: + raise ValueError(f"{label} must be finite non-negative or null") + return float(value) + + def register_extents(self, rows: list[dict[str, Any]]) -> None: + """Register exact refresh extents and their current physical mappings.""" + spare_ids = {item for values in self.free_spares.values() for item in values} + mapped_sources = {row["physical_block_id"] for row in self.extents.values()} + for raw in rows: + required = {"extent_id", "stack", "channel", "source_block_id", "version"} + if not isinstance(raw, dict) or set(raw) != required: + raise ValueError("extent row has missing or unknown fields") + extent, stack, channel, source = (raw["extent_id"], raw["stack"], + str(raw["channel"]), raw["source_block_id"]) + version = raw["version"] + if any(not isinstance(value, str) or not value + for value in (extent, stack, channel, source)): + raise ValueError("extent identities must be nonempty strings") + if (stack, channel) not in self.free_spares: + raise ValueError("extent stack/channel lacks a same-channel spare pool") + if source in spare_ids or source in mapped_sources or extent in self.extents: + raise ValueError("source overlaps spare pool or extent is duplicated") + if isinstance(version, bool) or not isinstance(version, int) or version < 0: + raise ValueError("version must be a non-negative integer") + self.extents[extent] = {"extent_id": extent, "stack": stack, + "channel": channel, "physical_block_id": source, + "version": version, "maintenance_attempt_count": 0} + mapped_sources.add(source) + # Establish the age identity without changing its configured initial age. + self.reliability.due_reasons(extent, self.reliability.epoch_ns) + + def record_foreground_write(self, extent_id: str, at_ns: int) -> int: + """Model an explicit mapping-generation conflict; read-only runs never call this.""" + row = self.extents[extent_id] + if not isinstance(at_ns, int) or at_ns < 0: + raise ValueError("foreground write time must be non-negative integer ns") + if row["version"] == (1 << 64) - 1: + raise OverflowError("mapping generation overflow") + row["version"] += 1 + self.events.append({"kind": "foreground_version_write", "extent_id": extent_id, + "at_ns": at_ns, "new_version": row["version"]}) + return row["version"] + + def _begin_due(self, now_ns: int) -> None: + groups: dict[tuple[str, str], list[str]] = defaultdict(list) + for extent_id, row in sorted(self.extents.items()): + if extent_id in self.active_extents: + continue + if self.reliability.accounted_through_ns(extent_id) != now_ns: + raise ValueError("reliability age must be integrated exactly through poll time") + reasons = self.reliability.due_reasons(extent_id, now_ns) + if reasons: + groups[(row["stack"], row["channel"])].append(extent_id) + for (stack, channel), extent_ids in sorted(groups.items()): + pool = self.free_spares[(stack, channel)] + extent_ids.sort(key=lambda item: ( + self.extents[item]["maintenance_attempt_count"], item)) + while extent_ids and pool: + count = min(len(extent_ids), len(pool), self.max_cohort) + selected = extent_ids[:count] + del extent_ids[:count] + items = [] + for extent_id in selected: + row = self.extents[extent_id] + items.append({"extent_id": extent_id, + "source_block_id": row["physical_block_id"], + "destination_block_id": pool.popleft(), + "expected_version": row["version"]}) + row["maintenance_attempt_count"] += 1 + self.active_extents.add(extent_id) + operation_id = f"refresh:{self._sequence}" + self._sequence += 1 + self.operations[operation_id] = { + "operation_id": operation_id, "stack": stack, "channel": channel, + "items": items, "phase": "refresh_read", "submitted": False, + "erase_queue": deque(), "committed_extents": [], + "conflicted_extents": [], "failures": [], "start_ns": now_ns, + } + self.events.append({"kind": "maintenance_started", "operation_id": operation_id, + "at_ns": now_ns, "extent_ids": list(selected), + "due_reasons": {item: self.reliability.due_reasons(item, now_ns) + for item in selected}}) + + def poll(self, now_ns: int, *, discover_due: bool = True) -> list[dict[str, Any]]: + """Offer each new phase once; jobs are suitable for TopologyService.extra_jobs.""" + if not isinstance(now_ns, int) or now_ns < 0: + raise ValueError("now_ns must be non-negative integer ns") + if discover_due: + self._begin_due(now_ns) + jobs = [] + for operation_id, operation in sorted(self.operations.items()): + if operation["phase"] == "terminal" or operation["submitted"]: + continue + phase = operation["phase"] + if phase in {"refresh_read", "program"}: + blocks = ([item["source_block_id"] for item in operation["items"]] + if phase == "refresh_read" else + [item["destination_block_id"] for item in operation["items"]]) + extents = [item["extent_id"] for item in operation["items"]] + expected = [item["expected_version"] for item in operation["items"]] + else: + erase = operation["current_erase"] + blocks, extents = erase["block_ids"], erase["extent_ids"] + expected = [] + job_id = f"{operation_id}:{phase}:{len(self.outstanding)}" + job = { + "job_id": job_id, "maintenance_id": job_id, + "stack": operation["stack"], "channel": operation["channel"], + "operation": "erase" if phase.startswith("erase_") else phase, + "bytes": self.block_bytes * len(blocks), "arrival_ns": now_ns, + "metadata": {"parent_maintenance_id": operation_id, "phase": phase, + "extent_ids": list(extents), + "physical_block_ids": list(blocks), + "expected_versions": list(expected), + "block_count": len(blocks)}, + } + operation["submitted"] = True + self.outstanding[job_id] = (operation_id, phase) + jobs.append(job) + return jobs + + def _cost_fact(self, operation_id: str, phase: str, completion_ns: int, + block_count: int, failed: bool) -> None: + if phase == "program": + quantity, unit, coefficient = self.block_bytes * block_count, "bytes", self.program_j_per_byte + elif phase.startswith("erase_"): + quantity, unit, coefficient = block_count, "operations", self.erase_j_per_operation + else: + return + self.energy_facts.append({ + "operation_id": operation_id, "phase": phase, "completion_ns": completion_ns, + "quantity": quantity, "unit": unit, + "energy_j": None if coefficient is None else quantity * coefficient, + "coefficient": coefficient, "evidence": self.energy_evidence[ + "program" if phase == "program" else "erase"], + "failed_after_service": failed, + "accounting_scope": "DRIVER_OPERATION_COST_FACT_CONSUME_ONCE", + }) + + def consume_receipt(self, receipt: dict[str, Any], *, failed_job_ids=()) -> dict[str, Any]: + """Advance phases from unique service completions. + + ``failed_job_ids`` is a fixed fault-injection/test input. Service work + and energy before that terminal failure remain counted. + """ + failed = set(failed_job_ids) + progress = {row["job_id"]: row for row in receipt.get("job_progress", [])} + for job_id in receipt.get("completion_ids", []): + if job_id not in self.outstanding: + continue + if job_id not in progress or progress[job_id]["remaining_bytes"] != 0: + raise ValueError("maintenance completion lacks byte-complete progress") + operation_id, phase = self.outstanding.pop(job_id) + operation = self.operations[operation_id] + completion_ns = progress[job_id]["completion_ns"] + if not isinstance(completion_ns, int): + raise ValueError("maintenance completion timestamp unavailable") + block_count = progress[job_id]["metadata"]["block_count"] + did_fail = job_id in failed + self._cost_fact(operation_id, phase, completion_ns, block_count, did_fail) + operation["submitted"] = False + if phase == "refresh_read": + if did_fail: + operation["failures"].append("refresh_read") + for item in operation["items"]: + self.free_spares[(operation["stack"], operation["channel"])].append( + item["destination_block_id"]) + self._terminal(operation, completion_ns, "FAILED_READ") + else: + operation["phase"] = "program" + elif phase == "program": + for item in operation["items"]: + wear = self.physical_wear[item["destination_block_id"]] + wear["block_program_work_started"] += 1 + wear["nand_page_programs_started"] += self.pages_per_block + if not did_fail: + wear["block_program_work_completed"] += 1 + wear["nand_page_programs_completed"] += self.pages_per_block + if did_fail: + operation["failures"].append("program") + operation["erase_queue"].append({ + "kind": "cleanup", "block_ids": [x["destination_block_id"] + for x in operation["items"]], + "extent_ids": [x["extent_id"] for x in operation["items"]]}) + else: + old_blocks, old_extents, cleanup_blocks, cleanup_extents = [], [], [], [] + for item in operation["items"]: + extent = self.extents[item["extent_id"]] + if (extent["version"] != item["expected_version"] + or extent["version"] == (1 << 64) - 1): + operation["conflicted_extents"].append(item["extent_id"]) + if extent["version"] == (1 << 64) - 1: + operation["failures"].append("version_overflow") + cleanup_blocks.append(item["destination_block_id"]) + cleanup_extents.append(item["extent_id"]) + continue + old_blocks.append(item["source_block_id"]) + old_extents.append(item["extent_id"]) + extent["physical_block_id"] = item["destination_block_id"] + extent["version"] += 1 + operation["committed_extents"].append(item["extent_id"]) + self.reliability.record_refresh_terminal( + item["extent_id"], completion_ns, True, operation_id) + if old_blocks: + operation["erase_queue"].append({"kind": "old", "block_ids": old_blocks, + "extent_ids": old_extents}) + if cleanup_blocks: + operation["erase_queue"].append({"kind": "cleanup", + "block_ids": cleanup_blocks, + "extent_ids": cleanup_extents}) + self._next_erase_or_terminal(operation, completion_ns) + else: + erase = operation["current_erase"] + for block in erase["block_ids"]: + wear = self.physical_wear[block] + wear["erase_phase_started"] += 1 + if not did_fail: + wear["erase_completed"] += 1 + self.free_spares[(operation["stack"], operation["channel"])].append(block) + else: + self.quarantined_blocks.add(block) + if did_fail: + operation["failures"].append(phase) + operation.pop("current_erase") + self._next_erase_or_terminal(operation, completion_ns) + unknown_failed = failed - set(receipt.get("completion_ids", [])) + if unknown_failed: + raise ValueError("failure injection names a non-completed job") + return self.drain_delta(receipt.get("end_ns")) + + def _next_erase_or_terminal(self, operation: dict[str, Any], at_ns: int) -> None: + if operation["erase_queue"]: + erase = operation["erase_queue"].popleft() + operation["current_erase"] = erase + operation["phase"] = "erase_old" if erase["kind"] == "old" else "erase_cleanup" + operation["submitted"] = False + return + if operation["committed_extents"]: + status = "COMMITTED" if not operation["failures"] and not operation["conflicted_extents"] \ + else "COMMITTED_WITH_POST_PROGRAM_EXCEPTION" + else: + status = "FAILED_OR_VERSION_CONFLICT" + self._terminal(operation, at_ns, status) + + def _terminal(self, operation: dict[str, Any], at_ns: int, status: str) -> None: + operation["phase"] = "terminal" + operation["submitted"] = False + for item in operation["items"]: + self.active_extents.discard(item["extent_id"]) + result = {"operation_id": operation["operation_id"], "status": status, + "start_ns": operation["start_ns"], "end_ns": at_ns, + "extent_ids": [x["extent_id"] for x in operation["items"]], + "committed_extents": list(operation["committed_extents"]), + "conflicted_extents": list(operation["conflicted_extents"]), + "failures": list(operation["failures"]), + "age_reset_extent_ids": list(operation["committed_extents"])} + self.results.append(result) + self.events.append({"kind": "maintenance_terminal", **deepcopy(result)}) + self._terminal_status_counts[status] += 1 + self._terminal_extent_count += len(result["extent_ids"]) + del self.operations[operation["operation_id"]] + + def drain_delta(self, end_ns: int | None = None) -> dict[str, Any]: + """Return compact new facts since the prior drain, without cumulative maps.""" + events, energy, results = self.events, self.energy_facts, self.results + self.events, self.energy_facts, self.results = [], [], [] + return { + "schema_version": "eq3-maintenance-driver-window-delta-v1", + "end_ns": end_ns, "events": events, "energy_facts": energy, + "terminal_results": results, + "summary": {"active_operation_count": sum( + op["phase"] != "terminal" for op in self.operations.values()), + "outstanding_job_count": len(self.outstanding), + "free_spare_count": sum(len(pool) for pool in self.free_spares.values()), + "quarantined_block_count": len(self.quarantined_blocks), + "committed_extent_count": sum(len(row["committed_extents"]) + for row in results), + "failed_terminal_count": sum(row["status"] != "COMMITTED" + for row in results)}, + } + + def snapshot(self) -> dict[str, Any]: + return { + "schema_version": "eq3-maintenance-driver-v1", + "service_semantics": "EXTERNAL_TOPOLOGY_SERVICE_COMPLETION_DRIVEN", + "version_semantics": "MONOTONIC_LOGICAL_EXTENT_GENERATION_CAS", + "age_reset_semantics": "SUCCESSFUL_COMMIT_EXACT_EXTENTS_ONLY", + "extents": deepcopy(self.extents), + "free_spares_by_stack_channel": { + stack: {channel: list(self.free_spares[(stack, channel)]) + for owner, channel in sorted(self.free_spares) if owner == stack} + for stack in sorted({owner for owner, _ in self.free_spares})}, + "quarantined_blocks": sorted(self.quarantined_blocks), + "physical_wear": deepcopy(dict(self.physical_wear)), + "outstanding_job_ids": sorted(self.outstanding), + "retained_terminal_results": deepcopy(self.results), + "retained_events": deepcopy(self.events), + "retained_energy_facts": deepcopy(self.energy_facts), + "terminal_summary": { + "operation_count": sum(self._terminal_status_counts.values()), + "extent_count": self._terminal_extent_count, + "status_counts": dict(sorted(self._terminal_status_counts.items())), + }, + "limitations": ["METADATA_VERSION_VALIDITY_NO_PAYLOAD_INTEGRITY", + "NO_RBER_OR_WEAR_DAMAGE_CURVE", + "RATE_SERVICE_NOT_NATIVE_NAND_MAINTENANCE"], + } + + +__all__ = ["MaintenanceDriver"] diff --git a/experiments/eq3_system_thermal/maintenance_inputs.py b/experiments/eq3_system_thermal/maintenance_inputs.py new file mode 100644 index 0000000..5ec7243 --- /dev/null +++ b/experiments/eq3_system_thermal/maintenance_inputs.py @@ -0,0 +1,52 @@ +"""Freeze bounded maintenance-pilot additions without launching a run.""" +from __future__ import annotations + +import copy + +try: + from .reliability import DAY_NS +except ImportError: + from reliability import DAY_NS + + +def add_maintenance(base_config: dict, mode: str) -> dict: + if mode not in {"disabled", "shared", "ideal_independent"}: + raise ValueError("unknown maintenance mode") + result = copy.deepcopy(base_config) + result["point_id"] = result["point_id"] + ":maintenance:" + mode + result["maintenance"] = { + "mode": mode, + "ea_ev": 1.04, + "refresh_trigger": "equivalent_age_or_wall", + # Every arm gets the same literal near-due scenario input. Disabled is + # still a fresh no-maintenance baseline and does not instantiate age. + "initial_equivalent_age_ns": DAY_NS - 2_000_000_000, + "initial_wall_age_ns": 0, + "initial_temperature_k": 300.0, + "block_bytes": 4096 * 256, + "pages_per_block": 256, + "aged_blocks_per_stack": 4096, + "spares_per_channel": 16, + "max_blocks_per_cohort": 16, + "program_energy_j_per_byte": 0.05 * 100e-6 / 4096, + "erase_energy_j_per_operation": 0.05 * 1e-3, + "energy_evidence": { + "program": "DERIVED_ENGINEERING_PROXY_0.05W_100US_NOT_HBF_CALIBRATION", + "erase": "DERIVED_ENGINEERING_PROXY_0.05W_NATIVE_ADAPTER_1MS_NOT_HBF_CALIBRATION", + }, + } + result["maintenance_scope"] = { + "aged_subset_bytes_per_hbf_stack": 4 * 1024**3, + "aged_subset_semantics": "EXPLICIT_SUBSET_NOT_WHOLE_MODEL_OR_PRODUCT_CAPACITY", + "spares_per_stack": 16 * 16, + "spare_ownership": "16_PER_EACH_OF_16_CHANNELS_NO_CROSS_CHANNEL_MOVE", + "backend": "AGGREGATE_RATE_SERVICE_NOT_NATIVE_MQSIM", + "initial_age_fairness": ( + "SHARED_AND_IDEAL_ARMS_USE_IDENTICAL_DAY_MINUS_2S_EQUIVALENT_AGE;" + "DISABLED_IS_FRESH_NO_MAINTENANCE_DIAGNOSTIC;NOT_24H_TIME_COMPRESSION" + ), + } + return result + + +__all__ = ["add_maintenance"] diff --git a/experiments/eq3_system_thermal/maintenance_pilot_inputs_v2/INDEX.json b/experiments/eq3_system_thermal/maintenance_pilot_inputs_v2/INDEX.json new file mode 100644 index 0000000..62db237 --- /dev/null +++ b/experiments/eq3_system_thermal/maintenance_pilot_inputs_v2/INDEX.json @@ -0,0 +1,51 @@ +{ + "comparison_scope": "FOUR_TOPOLOGY_SHARED_FEASIBILITY_PILOT_NO_DISABLED_OR_IDEAL_ARMS", + "duration": { + "active_ns": 8000000000, + "recovery_ns": 4000000000, + "windows": 600 + }, + "point_count": 4, + "points": [ + { + "config": "maint-pilot-v2-mixed_direct-shared-01.json", + "mode": "shared", + "point_id": "maint-pilot-v2-mixed_direct-shared-01", + "sha256": "05beecb5a62069a7c7d45c04f293340e82aba62bc1101308874310ffded1775b", + "status": "PREPARED_NOT_RUN", + "strategy": "read_rate_feedback_thermal_guard_v1", + "topology": "mixed_direct" + }, + { + "config": "maint-pilot-v2-relay-shared-01.json", + "mode": "shared", + "point_id": "maint-pilot-v2-relay-shared-01", + "sha256": "481cf19be4934744f5676829a3867814f1188447c4edd55833a6546f67c7e22b", + "status": "PREPARED_NOT_RUN", + "strategy": "read_rate_feedback_thermal_guard_v1", + "topology": "relay" + }, + { + "config": "maint-pilot-v2-dash-shared-01.json", + "mode": "shared", + "point_id": "maint-pilot-v2-dash-shared-01", + "sha256": "c3408dc9ed450d7f47256608397c8e9562acdc08f52d0810b7802928fd7a9caf", + "status": "PREPARED_NOT_RUN", + "strategy": "read_rate_feedback_thermal_guard_v1", + "topology": "dash" + }, + { + "config": "maint-pilot-v2-all_hbf_direct-shared-01.json", + "mode": "shared", + "point_id": "maint-pilot-v2-all_hbf_direct-shared-01", + "sha256": "bb0e9a6fd4ca6e0ab347195e13f94c27b03d1f164c7fbdbd7a8bbea46d3151d5", + "status": "PREPARED_NOT_RUN", + "strategy": "read_rate_feedback_thermal_guard_v1", + "topology": "all_hbf_direct" + } + ], + "run_authorization": "ROOT_SCHEDULES_AFTER_ACTIVE_BASE_AND_REVIEW", + "schema_version": "eq3-maintenance-four-topology-shared-pilot-v2", + "science_scope": "CONDITIONAL_AGGREGATE_RATE_SERVICE_MAINTENANCE_NOT_NATIVE_MQSIM", + "status": "PREPARED_NOT_RUN" +} diff --git a/experiments/eq3_system_thermal/maintenance_pilot_inputs_v2/MAIN_PROPOSAL.json b/experiments/eq3_system_thermal/maintenance_pilot_inputs_v2/MAIN_PROPOSAL.json new file mode 100644 index 0000000..8f5f3b5 --- /dev/null +++ b/experiments/eq3_system_thermal/maintenance_pilot_inputs_v2/MAIN_PROPOSAL.json @@ -0,0 +1,33 @@ +{ + "configuration_generation": "DEFERRED_UNTIL_PILOT_REVIEW", + "disabled_semantics": "FRESH_NO_MAINTENANCE_DIAGNOSTIC_NOT_AGE_MATCHED_CONTENTION_ARM", + "prerequisite": "maintenance_pilot_inputs_v2 all four shared points reviewed", + "proposed_arms": { + "contention_ablation": { + "comparison": "same enabled age/workload/control; only maintenance resource contention differs", + "modes": [ + "shared", + "ideal_independent" + ], + "strategy": "read_rate_feedback_thermal_guard_v1", + "topology": "mixed_direct" + }, + "strategy_matrix": { + "mode": "shared", + "strategies": [ + "guard_only", + "thermal_hysteresis_guard", + "read_rate_feedback_thermal_guard_v1" + ], + "topologies": [ + "mixed_direct", + "relay", + "dash", + "all_hbf_direct" + ] + } + }, + "schema_version": "eq3-maintenance-main-proposal-v1", + "status": "WAITING_PILOT_AND_ROOT_REVIEW", + "status_reason": "Do not freeze or run main maintenance matrix before shared pilot evidence" +} diff --git a/experiments/eq3_system_thermal/maintenance_pilot_inputs_v2/SOFTWARE_COST_PROJECTION.json b/experiments/eq3_system_thermal/maintenance_pilot_inputs_v2/SOFTWARE_COST_PROJECTION.json new file mode 100644 index 0000000..5dfb3b8 --- /dev/null +++ b/experiments/eq3_system_thermal/maintenance_pilot_inputs_v2/SOFTWARE_COST_PROJECTION.json @@ -0,0 +1,42 @@ +{ + "schema_version": "eq3-maintenance-fixed-software-cost-v1", + "classification": "FIXED_SOFTWARE_VALIDATION_NOT_THERMAL_OR_SCIENTIFIC_EXPERIMENT", + "input": { + "topology": "all_hbf_direct", + "hbf_stacks": 8, + "channels_per_stack": 16, + "aged_blocks_per_stack": 4096, + "total_aged_blocks": 32768, + "aged_subset_bytes_per_stack": 4294967296, + "windows_measured": 10, + "windows_projected": 600, + "forced_due_at_start_for_stress": true, + "thermal": "FIXED_FAKE_300K_NO_SOLVER", + "output_sink": "COUNTING_NULL_SINK" + }, + "root_cause": { + "classification": "CONFIRMED_BUG", + "location": "maintenance_driver.py::register_extents", + "before": "reconstructed all mapped source IDs for every inserted extent (quadratic)", + "fix": "construct once and update incrementally", + "pre_fix_10_window_wall_s": 23.379321462998632, + "post_fix_10_window_wall_s": 1.3981969419983216 + }, + "post_fix_observed": { + "max_rss_kib": 90804, + "serialized_bytes": 6578487, + "retained_driver_delta_rows_at_final": 0, + "terminal_operations": 256, + "terminal_extents": 4096 + }, + "linear_projection_diagnostic": { + "wall_s_600_windows": 83.8918165198993, + "serialized_bytes_600_windows": 394709220, + "warning": "Linear fixed-software projection only; thermal solver and workload-dependent maintenance timing are excluded." + }, + "limits": [ + "No heat solve was run.", + "No native MQSim backend was run.", + "The forced-due stress case is not the frozen day-minus-2s pilot scenario." + ] +} diff --git a/experiments/eq3_system_thermal/maintenance_pilot_inputs_v2/maint-pilot-v2-all_hbf_direct-shared-01.json b/experiments/eq3_system_thermal/maintenance_pilot_inputs_v2/maint-pilot-v2-all_hbf_direct-shared-01.json new file mode 100644 index 0000000..1372d20 --- /dev/null +++ b/experiments/eq3_system_thermal/maintenance_pilot_inputs_v2/maint-pilot-v2-all_hbf_direct-shared-01.json @@ -0,0 +1,382 @@ +{ + "energy": { + "ecc_j_per_byte": 0.0, + "ecc_scope": "ZERO_INCREMENT_BASELINE_NOT_MEASURED_ZERO; scenario overrides explicit", + "erase_array_j_per_operation": 5e-05, + "evidence": { + "erase": "DERIVED_ENGINEERING_PROXY_0.05W_TIMES_NATIVE_FIXTURE_1MS_NOT_HBF_MEASUREMENT", + "program": "DERIVED_ENGINEERING_PROXY_OLD_0.05W_0.01W_100US_4KIB; NOT_HBF_CALIBRATION", + "read": "USER_CONFIRMED_40_ARRAY_10_BASE_PJ_PER_B", + "relay_hbm": "REUSED_EQ3_MAINTENANCE_ENERGY_PROFILE_SCENARIO" + }, + "hbm_array_j_per_byte": 4e-11, + "hbm_base_j_per_byte": 2e-12, + "program_array_j_per_byte": 1.220703125e-09, + "program_base_j_per_byte": 2.4414062500000004e-10, + "read_array_j_per_byte": 4e-11, + "read_base_j_per_byte": 1e-11, + "relay_partner_receive_j_per_byte": 2e-12, + "relay_partner_send_j_per_byte": 2e-12 + }, + "gpu_external_w": 0, + "maintenance": { + "aged_blocks_per_stack": 4096, + "block_bytes": 1048576, + "ea_ev": 1.04, + "energy_evidence": { + "erase": "DERIVED_ENGINEERING_PROXY_0.05W_NATIVE_ADAPTER_1MS_NOT_HBF_CALIBRATION", + "program": "DERIVED_ENGINEERING_PROXY_0.05W_100US_NOT_HBF_CALIBRATION" + }, + "erase_energy_j_per_operation": 5e-05, + "initial_equivalent_age_ns": 86398000000000, + "initial_temperature_k": 300.0, + "initial_wall_age_ns": 0, + "max_blocks_per_cohort": 16, + "mode": "shared", + "pages_per_block": 256, + "program_energy_j_per_byte": 1.220703125e-09, + "refresh_trigger": "equivalent_age_or_wall", + "spares_per_channel": 16 + }, + "maintenance_scope": { + "aged_subset_bytes_per_hbf_stack": 4294967296, + "aged_subset_semantics": "EXPLICIT_SUBSET_NOT_WHOLE_MODEL_OR_PRODUCT_CAPACITY", + "backend": "AGGREGATE_RATE_SERVICE_NOT_NATIVE_MQSIM", + "initial_age_fairness": "SHARED_PILOT_DAY_MINUS_2S_EQUIVALENT_AGE;NOT_24H_TIME_COMPRESSION;DISABLED_NOT_PART_OF_PILOT", + "spare_ownership": "16_PER_EACH_OF_16_CHANNELS_NO_CROSS_CHANNEL_MOVE", + "spares_per_stack": 256 + }, + "point_id": "maint-pilot-v2-all_hbf_direct-shared-01", + "recovery_ns": 4000000000, + "resource_limits": { + "address_space_gib": 8, + "watchdog_s": 600 + }, + "schema_version": "eq3-system-point-v1", + "scope": "CONDITIONAL_SIMULATED", + "service": { + "channels": { + "hbf0": { + "0": 96000000000, + "1": 96000000000, + "10": 96000000000, + "11": 96000000000, + "12": 96000000000, + "13": 96000000000, + "14": 96000000000, + "15": 96000000000, + "2": 96000000000, + "3": 96000000000, + "4": 96000000000, + "5": 96000000000, + "6": 96000000000, + "7": 96000000000, + "8": 96000000000, + "9": 96000000000 + }, + "hbf1": { + "0": 96000000000, + "1": 96000000000, + "10": 96000000000, + "11": 96000000000, + "12": 96000000000, + "13": 96000000000, + "14": 96000000000, + "15": 96000000000, + "2": 96000000000, + "3": 96000000000, + "4": 96000000000, + "5": 96000000000, + "6": 96000000000, + "7": 96000000000, + "8": 96000000000, + "9": 96000000000 + }, + "hbf2": { + "0": 96000000000, + "1": 96000000000, + "10": 96000000000, + "11": 96000000000, + "12": 96000000000, + "13": 96000000000, + "14": 96000000000, + "15": 96000000000, + "2": 96000000000, + "3": 96000000000, + "4": 96000000000, + "5": 96000000000, + "6": 96000000000, + "7": 96000000000, + "8": 96000000000, + "9": 96000000000 + }, + "hbf3": { + "0": 96000000000, + "1": 96000000000, + "10": 96000000000, + "11": 96000000000, + "12": 96000000000, + "13": 96000000000, + "14": 96000000000, + "15": 96000000000, + "2": 96000000000, + "3": 96000000000, + "4": 96000000000, + "5": 96000000000, + "6": 96000000000, + "7": 96000000000, + "8": 96000000000, + "9": 96000000000 + }, + "hbf4": { + "0": 96000000000, + "1": 96000000000, + "10": 96000000000, + "11": 96000000000, + "12": 96000000000, + "13": 96000000000, + "14": 96000000000, + "15": 96000000000, + "2": 96000000000, + "3": 96000000000, + "4": 96000000000, + "5": 96000000000, + "6": 96000000000, + "7": 96000000000, + "8": 96000000000, + "9": 96000000000 + }, + "hbf5": { + "0": 96000000000, + "1": 96000000000, + "10": 96000000000, + "11": 96000000000, + "12": 96000000000, + "13": 96000000000, + "14": 96000000000, + "15": 96000000000, + "2": 96000000000, + "3": 96000000000, + "4": 96000000000, + "5": 96000000000, + "6": 96000000000, + "7": 96000000000, + "8": 96000000000, + "9": 96000000000 + }, + "hbf6": { + "0": 96000000000, + "1": 96000000000, + "10": 96000000000, + "11": 96000000000, + "12": 96000000000, + "13": 96000000000, + "14": 96000000000, + "15": 96000000000, + "2": 96000000000, + "3": 96000000000, + "4": 96000000000, + "5": 96000000000, + "6": 96000000000, + "7": 96000000000, + "8": 96000000000, + "9": 96000000000 + }, + "hbf7": { + "0": 96000000000, + "1": 96000000000, + "10": 96000000000, + "11": 96000000000, + "12": 96000000000, + "13": 96000000000, + "14": 96000000000, + "15": 96000000000, + "2": 96000000000, + "3": 96000000000, + "4": 96000000000, + "5": 96000000000, + "6": 96000000000, + "7": 96000000000, + "8": 96000000000, + "9": 96000000000 + } + }, + "dash_routes": {}, + "evidence": "SCENARIO_ASSUMPTION", + "fabric": { + "evidence": "SCENARIO_ASSUMPTION", + "hbf": { + "hbf0": { + "bank_capacity_bytes": 8388608, + "bank_count": 2, + "direct_link": { + "bandwidth_bytes_per_s": 2048000000000, + "latency_ns": 10 + }, + "fill": { + "bandwidth_bytes_per_s": 2048000000000, + "latency_ns": 10 + }, + "pair": null, + "relay_link": null + }, + "hbf1": { + "bank_capacity_bytes": 8388608, + "bank_count": 2, + "direct_link": { + "bandwidth_bytes_per_s": 2048000000000, + "latency_ns": 10 + }, + "fill": { + "bandwidth_bytes_per_s": 2048000000000, + "latency_ns": 10 + }, + "pair": null, + "relay_link": null + }, + "hbf2": { + "bank_capacity_bytes": 8388608, + "bank_count": 2, + "direct_link": { + "bandwidth_bytes_per_s": 2048000000000, + "latency_ns": 10 + }, + "fill": { + "bandwidth_bytes_per_s": 2048000000000, + "latency_ns": 10 + }, + "pair": null, + "relay_link": null + }, + "hbf3": { + "bank_capacity_bytes": 8388608, + "bank_count": 2, + "direct_link": { + "bandwidth_bytes_per_s": 2048000000000, + "latency_ns": 10 + }, + "fill": { + "bandwidth_bytes_per_s": 2048000000000, + "latency_ns": 10 + }, + "pair": null, + "relay_link": null + }, + "hbf4": { + "bank_capacity_bytes": 8388608, + "bank_count": 2, + "direct_link": { + "bandwidth_bytes_per_s": 2048000000000, + "latency_ns": 10 + }, + "fill": { + "bandwidth_bytes_per_s": 2048000000000, + "latency_ns": 10 + }, + "pair": null, + "relay_link": null + }, + "hbf5": { + "bank_capacity_bytes": 8388608, + "bank_count": 2, + "direct_link": { + "bandwidth_bytes_per_s": 2048000000000, + "latency_ns": 10 + }, + "fill": { + "bandwidth_bytes_per_s": 2048000000000, + "latency_ns": 10 + }, + "pair": null, + "relay_link": null + }, + "hbf6": { + "bank_capacity_bytes": 8388608, + "bank_count": 2, + "direct_link": { + "bandwidth_bytes_per_s": 2048000000000, + "latency_ns": 10 + }, + "fill": { + "bandwidth_bytes_per_s": 2048000000000, + "latency_ns": 10 + }, + "pair": null, + "relay_link": null + }, + "hbf7": { + "bank_capacity_bytes": 8388608, + "bank_count": 2, + "direct_link": { + "bandwidth_bytes_per_s": 2048000000000, + "latency_ns": 10 + }, + "fill": { + "bandwidth_bytes_per_s": 2048000000000, + "latency_ns": 10 + }, + "pair": null, + "relay_link": null + } + }, + "hbm": {} + }, + "maintenance_guard_policy": "LIGHT_SEVERE_EXEMPT_SHUTDOWN_BLOCKED", + "operation_cost_evidence": "PROXY_16_PLANES_4KIB_100US_PROGRAM_1MS_ERASE_256PAGES_PER_BLOCK", + "operation_media_cost": { + "erase": { + "denominator": 8192, + "numerator": 46875 + }, + "migration": { + "denominator": 1, + "numerator": 1 + }, + "program": { + "denominator": 64, + "numerator": 9375 + }, + "read": { + "denominator": 1, + "numerator": 1 + }, + "refresh_read": { + "denominator": 1, + "numerator": 1 + }, + "retry": { + "denominator": 1, + "numerator": 1 + } + }, + "schema_version": "eq3-topology-fluid-config-v1", + "topology": "all_hbf_direct", + "window_ns": 20000000 + }, + "strategy": "read_rate_feedback_thermal_guard_v1", + "thermal_limits_k": { + "gpu": [ + 363.15, + 373.15, + 383.15 + ], + "hbf": [ + 353.15, + 363.15, + 378.15 + ], + "hbm": [ + 353.15, + 363.15, + 378.15 + ] + }, + "topology": "all_hbf_direct", + "unknown_idle_power": "EXCLUDED_NOT_ZERO_MEASUREMENT", + "workload": { + "active_ns": 8000000000, + "channel_distribution": "uniform", + "pattern": "continuous", + "per_stack_Bps": 384000000000 + } +} diff --git a/experiments/eq3_system_thermal/maintenance_pilot_inputs_v2/maint-pilot-v2-dash-shared-01.json b/experiments/eq3_system_thermal/maintenance_pilot_inputs_v2/maint-pilot-v2-dash-shared-01.json new file mode 100644 index 0000000..1296622 --- /dev/null +++ b/experiments/eq3_system_thermal/maintenance_pilot_inputs_v2/maint-pilot-v2-dash-shared-01.json @@ -0,0 +1,444 @@ +{ + "energy": { + "ecc_j_per_byte": 0.0, + "ecc_scope": "ZERO_INCREMENT_BASELINE_NOT_MEASURED_ZERO; scenario overrides explicit", + "erase_array_j_per_operation": 5e-05, + "evidence": { + "erase": "DERIVED_ENGINEERING_PROXY_0.05W_TIMES_NATIVE_FIXTURE_1MS_NOT_HBF_MEASUREMENT", + "program": "DERIVED_ENGINEERING_PROXY_OLD_0.05W_0.01W_100US_4KIB; NOT_HBF_CALIBRATION", + "read": "USER_CONFIRMED_40_ARRAY_10_BASE_PJ_PER_B", + "relay_hbm": "REUSED_EQ3_MAINTENANCE_ENERGY_PROFILE_SCENARIO" + }, + "hbm_array_j_per_byte": 4e-11, + "hbm_base_j_per_byte": 2e-12, + "program_array_j_per_byte": 1.220703125e-09, + "program_base_j_per_byte": 2.4414062500000004e-10, + "read_array_j_per_byte": 4e-11, + "read_base_j_per_byte": 1e-11, + "relay_partner_receive_j_per_byte": 2e-12, + "relay_partner_send_j_per_byte": 2e-12 + }, + "gpu_external_w": 0, + "maintenance": { + "aged_blocks_per_stack": 4096, + "block_bytes": 1048576, + "ea_ev": 1.04, + "energy_evidence": { + "erase": "DERIVED_ENGINEERING_PROXY_0.05W_NATIVE_ADAPTER_1MS_NOT_HBF_CALIBRATION", + "program": "DERIVED_ENGINEERING_PROXY_0.05W_100US_NOT_HBF_CALIBRATION" + }, + "erase_energy_j_per_operation": 5e-05, + "initial_equivalent_age_ns": 86398000000000, + "initial_temperature_k": 300.0, + "initial_wall_age_ns": 0, + "max_blocks_per_cohort": 16, + "mode": "shared", + "pages_per_block": 256, + "program_energy_j_per_byte": 1.220703125e-09, + "refresh_trigger": "equivalent_age_or_wall", + "spares_per_channel": 16 + }, + "maintenance_scope": { + "aged_subset_bytes_per_hbf_stack": 4294967296, + "aged_subset_semantics": "EXPLICIT_SUBSET_NOT_WHOLE_MODEL_OR_PRODUCT_CAPACITY", + "backend": "AGGREGATE_RATE_SERVICE_NOT_NATIVE_MQSIM", + "initial_age_fairness": "SHARED_PILOT_DAY_MINUS_2S_EQUIVALENT_AGE;NOT_24H_TIME_COMPRESSION;DISABLED_NOT_PART_OF_PILOT", + "spare_ownership": "16_PER_EACH_OF_16_CHANNELS_NO_CROSS_CHANNEL_MOVE", + "spares_per_stack": 256 + }, + "point_id": "maint-pilot-v2-dash-shared-01", + "recovery_ns": 4000000000, + "resource_limits": { + "address_space_gib": 8, + "watchdog_s": 600 + }, + "schema_version": "eq3-system-point-v1", + "scope": "CONDITIONAL_SIMULATED", + "service": { + "channels": { + "hbf0": { + "0": 96000000000, + "1": 96000000000, + "10": 96000000000, + "11": 96000000000, + "12": 96000000000, + "13": 96000000000, + "14": 96000000000, + "15": 96000000000, + "2": 96000000000, + "3": 96000000000, + "4": 96000000000, + "5": 96000000000, + "6": 96000000000, + "7": 96000000000, + "8": 96000000000, + "9": 96000000000 + }, + "hbf1": { + "0": 96000000000, + "1": 96000000000, + "10": 96000000000, + "11": 96000000000, + "12": 96000000000, + "13": 96000000000, + "14": 96000000000, + "15": 96000000000, + "2": 96000000000, + "3": 96000000000, + "4": 96000000000, + "5": 96000000000, + "6": 96000000000, + "7": 96000000000, + "8": 96000000000, + "9": 96000000000 + }, + "hbf2": { + "0": 96000000000, + "1": 96000000000, + "10": 96000000000, + "11": 96000000000, + "12": 96000000000, + "13": 96000000000, + "14": 96000000000, + "15": 96000000000, + "2": 96000000000, + "3": 96000000000, + "4": 96000000000, + "5": 96000000000, + "6": 96000000000, + "7": 96000000000, + "8": 96000000000, + "9": 96000000000 + }, + "hbf3": { + "0": 96000000000, + "1": 96000000000, + "10": 96000000000, + "11": 96000000000, + "12": 96000000000, + "13": 96000000000, + "14": 96000000000, + "15": 96000000000, + "2": 96000000000, + "3": 96000000000, + "4": 96000000000, + "5": 96000000000, + "6": 96000000000, + "7": 96000000000, + "8": 96000000000, + "9": 96000000000 + }, + "hbm0": { + "0": 96000000000, + "1": 96000000000, + "10": 96000000000, + "11": 96000000000, + "12": 96000000000, + "13": 96000000000, + "14": 96000000000, + "15": 96000000000, + "2": 96000000000, + "3": 96000000000, + "4": 96000000000, + "5": 96000000000, + "6": 96000000000, + "7": 96000000000, + "8": 96000000000, + "9": 96000000000 + }, + "hbm1": { + "0": 96000000000, + "1": 96000000000, + "10": 96000000000, + "11": 96000000000, + "12": 96000000000, + "13": 96000000000, + "14": 96000000000, + "15": 96000000000, + "2": 96000000000, + "3": 96000000000, + "4": 96000000000, + "5": 96000000000, + "6": 96000000000, + "7": 96000000000, + "8": 96000000000, + "9": 96000000000 + }, + "hbm2": { + "0": 96000000000, + "1": 96000000000, + "10": 96000000000, + "11": 96000000000, + "12": 96000000000, + "13": 96000000000, + "14": 96000000000, + "15": 96000000000, + "2": 96000000000, + "3": 96000000000, + "4": 96000000000, + "5": 96000000000, + "6": 96000000000, + "7": 96000000000, + "8": 96000000000, + "9": 96000000000 + }, + "hbm3": { + "0": 96000000000, + "1": 96000000000, + "10": 96000000000, + "11": 96000000000, + "12": 96000000000, + "13": 96000000000, + "14": 96000000000, + "15": 96000000000, + "2": 96000000000, + "3": 96000000000, + "4": 96000000000, + "5": 96000000000, + "6": 96000000000, + "7": 96000000000, + "8": 96000000000, + "9": 96000000000 + } + }, + "dash_routes": { + "hbf0": { + "0": "direct", + "1": "relay", + "10": "direct", + "11": "relay", + "12": "direct", + "13": "relay", + "14": "direct", + "15": "relay", + "2": "direct", + "3": "relay", + "4": "direct", + "5": "relay", + "6": "direct", + "7": "relay", + "8": "direct", + "9": "relay" + }, + "hbf1": { + "0": "direct", + "1": "relay", + "10": "direct", + "11": "relay", + "12": "direct", + "13": "relay", + "14": "direct", + "15": "relay", + "2": "direct", + "3": "relay", + "4": "direct", + "5": "relay", + "6": "direct", + "7": "relay", + "8": "direct", + "9": "relay" + }, + "hbf2": { + "0": "direct", + "1": "relay", + "10": "direct", + "11": "relay", + "12": "direct", + "13": "relay", + "14": "direct", + "15": "relay", + "2": "direct", + "3": "relay", + "4": "direct", + "5": "relay", + "6": "direct", + "7": "relay", + "8": "direct", + "9": "relay" + }, + "hbf3": { + "0": "direct", + "1": "relay", + "10": "direct", + "11": "relay", + "12": "direct", + "13": "relay", + "14": "direct", + "15": "relay", + "2": "direct", + "3": "relay", + "4": "direct", + "5": "relay", + "6": "direct", + "7": "relay", + "8": "direct", + "9": "relay" + } + }, + "evidence": "SCENARIO_ASSUMPTION", + "fabric": { + "evidence": "SCENARIO_ASSUMPTION", + "hbf": { + "hbf0": { + "bank_capacity_bytes": 8388608, + "bank_count": 2, + "direct_link": { + "bandwidth_bytes_per_s": 2048000000000, + "latency_ns": 10 + }, + "fill": { + "bandwidth_bytes_per_s": 2048000000000, + "latency_ns": 10 + }, + "pair": "hbm0", + "relay_link": { + "bandwidth_bytes_per_s": 2048000000000, + "latency_ns": 10 + } + }, + "hbf1": { + "bank_capacity_bytes": 8388608, + "bank_count": 2, + "direct_link": { + "bandwidth_bytes_per_s": 2048000000000, + "latency_ns": 10 + }, + "fill": { + "bandwidth_bytes_per_s": 2048000000000, + "latency_ns": 10 + }, + "pair": "hbm1", + "relay_link": { + "bandwidth_bytes_per_s": 2048000000000, + "latency_ns": 10 + } + }, + "hbf2": { + "bank_capacity_bytes": 8388608, + "bank_count": 2, + "direct_link": { + "bandwidth_bytes_per_s": 2048000000000, + "latency_ns": 10 + }, + "fill": { + "bandwidth_bytes_per_s": 2048000000000, + "latency_ns": 10 + }, + "pair": "hbm2", + "relay_link": { + "bandwidth_bytes_per_s": 2048000000000, + "latency_ns": 10 + } + }, + "hbf3": { + "bank_capacity_bytes": 8388608, + "bank_count": 2, + "direct_link": { + "bandwidth_bytes_per_s": 2048000000000, + "latency_ns": 10 + }, + "fill": { + "bandwidth_bytes_per_s": 2048000000000, + "latency_ns": 10 + }, + "pair": "hbm3", + "relay_link": { + "bandwidth_bytes_per_s": 2048000000000, + "latency_ns": 10 + } + } + }, + "hbm": { + "hbm0": { + "bank_capacity_bytes": 4194304, + "bank_count": 2, + "gpu_link": { + "bandwidth_bytes_per_s": 2048000000000, + "latency_ns": 10 + } + }, + "hbm1": { + "bank_capacity_bytes": 4194304, + "bank_count": 2, + "gpu_link": { + "bandwidth_bytes_per_s": 2048000000000, + "latency_ns": 10 + } + }, + "hbm2": { + "bank_capacity_bytes": 4194304, + "bank_count": 2, + "gpu_link": { + "bandwidth_bytes_per_s": 2048000000000, + "latency_ns": 10 + } + }, + "hbm3": { + "bank_capacity_bytes": 4194304, + "bank_count": 2, + "gpu_link": { + "bandwidth_bytes_per_s": 2048000000000, + "latency_ns": 10 + } + } + } + }, + "maintenance_guard_policy": "LIGHT_SEVERE_EXEMPT_SHUTDOWN_BLOCKED", + "operation_cost_evidence": "PROXY_16_PLANES_4KIB_100US_PROGRAM_1MS_ERASE_256PAGES_PER_BLOCK", + "operation_media_cost": { + "erase": { + "denominator": 8192, + "numerator": 46875 + }, + "migration": { + "denominator": 1, + "numerator": 1 + }, + "program": { + "denominator": 64, + "numerator": 9375 + }, + "read": { + "denominator": 1, + "numerator": 1 + }, + "refresh_read": { + "denominator": 1, + "numerator": 1 + }, + "retry": { + "denominator": 1, + "numerator": 1 + } + }, + "schema_version": "eq3-topology-fluid-config-v1", + "topology": "dash", + "window_ns": 20000000 + }, + "strategy": "read_rate_feedback_thermal_guard_v1", + "thermal_limits_k": { + "gpu": [ + 363.15, + 373.15, + 383.15 + ], + "hbf": [ + 353.15, + 363.15, + 378.15 + ], + "hbm": [ + 353.15, + 363.15, + 378.15 + ] + }, + "topology": "dash", + "unknown_idle_power": "EXCLUDED_NOT_ZERO_MEASUREMENT", + "workload": { + "active_ns": 8000000000, + "channel_distribution": "uniform", + "pattern": "continuous", + "per_stack_Bps": 384000000000 + } +} diff --git a/experiments/eq3_system_thermal/maintenance_pilot_inputs_v2/maint-pilot-v2-mixed_direct-shared-01.json b/experiments/eq3_system_thermal/maintenance_pilot_inputs_v2/maint-pilot-v2-mixed_direct-shared-01.json new file mode 100644 index 0000000..8057584 --- /dev/null +++ b/experiments/eq3_system_thermal/maintenance_pilot_inputs_v2/maint-pilot-v2-mixed_direct-shared-01.json @@ -0,0 +1,359 @@ +{ + "energy": { + "ecc_j_per_byte": 0.0, + "ecc_scope": "ZERO_INCREMENT_BASELINE_NOT_MEASURED_ZERO; scenario overrides explicit", + "erase_array_j_per_operation": 5e-05, + "evidence": { + "erase": "DERIVED_ENGINEERING_PROXY_0.05W_TIMES_NATIVE_FIXTURE_1MS_NOT_HBF_MEASUREMENT", + "program": "DERIVED_ENGINEERING_PROXY_OLD_0.05W_0.01W_100US_4KIB; NOT_HBF_CALIBRATION", + "read": "USER_CONFIRMED_40_ARRAY_10_BASE_PJ_PER_B", + "relay_hbm": "REUSED_EQ3_MAINTENANCE_ENERGY_PROFILE_SCENARIO" + }, + "hbm_array_j_per_byte": 4e-11, + "hbm_base_j_per_byte": 2e-12, + "program_array_j_per_byte": 1.220703125e-09, + "program_base_j_per_byte": 2.4414062500000004e-10, + "read_array_j_per_byte": 4e-11, + "read_base_j_per_byte": 1e-11, + "relay_partner_receive_j_per_byte": 2e-12, + "relay_partner_send_j_per_byte": 2e-12 + }, + "gpu_external_w": 0, + "maintenance": { + "aged_blocks_per_stack": 4096, + "block_bytes": 1048576, + "ea_ev": 1.04, + "energy_evidence": { + "erase": "DERIVED_ENGINEERING_PROXY_0.05W_NATIVE_ADAPTER_1MS_NOT_HBF_CALIBRATION", + "program": "DERIVED_ENGINEERING_PROXY_0.05W_100US_NOT_HBF_CALIBRATION" + }, + "erase_energy_j_per_operation": 5e-05, + "initial_equivalent_age_ns": 86398000000000, + "initial_temperature_k": 300.0, + "initial_wall_age_ns": 0, + "max_blocks_per_cohort": 16, + "mode": "shared", + "pages_per_block": 256, + "program_energy_j_per_byte": 1.220703125e-09, + "refresh_trigger": "equivalent_age_or_wall", + "spares_per_channel": 16 + }, + "maintenance_scope": { + "aged_subset_bytes_per_hbf_stack": 4294967296, + "aged_subset_semantics": "EXPLICIT_SUBSET_NOT_WHOLE_MODEL_OR_PRODUCT_CAPACITY", + "backend": "AGGREGATE_RATE_SERVICE_NOT_NATIVE_MQSIM", + "initial_age_fairness": "SHARED_PILOT_DAY_MINUS_2S_EQUIVALENT_AGE;NOT_24H_TIME_COMPRESSION;DISABLED_NOT_PART_OF_PILOT", + "spare_ownership": "16_PER_EACH_OF_16_CHANNELS_NO_CROSS_CHANNEL_MOVE", + "spares_per_stack": 256 + }, + "point_id": "maint-pilot-v2-mixed_direct-shared-01", + "recovery_ns": 4000000000, + "resource_limits": { + "address_space_gib": 8, + "watchdog_s": 600 + }, + "schema_version": "eq3-system-point-v1", + "scope": "CONDITIONAL_SIMULATED", + "service": { + "channels": { + "hbf0": { + "0": 96000000000, + "1": 96000000000, + "10": 96000000000, + "11": 96000000000, + "12": 96000000000, + "13": 96000000000, + "14": 96000000000, + "15": 96000000000, + "2": 96000000000, + "3": 96000000000, + "4": 96000000000, + "5": 96000000000, + "6": 96000000000, + "7": 96000000000, + "8": 96000000000, + "9": 96000000000 + }, + "hbf1": { + "0": 96000000000, + "1": 96000000000, + "10": 96000000000, + "11": 96000000000, + "12": 96000000000, + "13": 96000000000, + "14": 96000000000, + "15": 96000000000, + "2": 96000000000, + "3": 96000000000, + "4": 96000000000, + "5": 96000000000, + "6": 96000000000, + "7": 96000000000, + "8": 96000000000, + "9": 96000000000 + }, + "hbf2": { + "0": 96000000000, + "1": 96000000000, + "10": 96000000000, + "11": 96000000000, + "12": 96000000000, + "13": 96000000000, + "14": 96000000000, + "15": 96000000000, + "2": 96000000000, + "3": 96000000000, + "4": 96000000000, + "5": 96000000000, + "6": 96000000000, + "7": 96000000000, + "8": 96000000000, + "9": 96000000000 + }, + "hbf3": { + "0": 96000000000, + "1": 96000000000, + "10": 96000000000, + "11": 96000000000, + "12": 96000000000, + "13": 96000000000, + "14": 96000000000, + "15": 96000000000, + "2": 96000000000, + "3": 96000000000, + "4": 96000000000, + "5": 96000000000, + "6": 96000000000, + "7": 96000000000, + "8": 96000000000, + "9": 96000000000 + }, + "hbm0": { + "0": 96000000000, + "1": 96000000000, + "10": 96000000000, + "11": 96000000000, + "12": 96000000000, + "13": 96000000000, + "14": 96000000000, + "15": 96000000000, + "2": 96000000000, + "3": 96000000000, + "4": 96000000000, + "5": 96000000000, + "6": 96000000000, + "7": 96000000000, + "8": 96000000000, + "9": 96000000000 + }, + "hbm1": { + "0": 96000000000, + "1": 96000000000, + "10": 96000000000, + "11": 96000000000, + "12": 96000000000, + "13": 96000000000, + "14": 96000000000, + "15": 96000000000, + "2": 96000000000, + "3": 96000000000, + "4": 96000000000, + "5": 96000000000, + "6": 96000000000, + "7": 96000000000, + "8": 96000000000, + "9": 96000000000 + }, + "hbm2": { + "0": 96000000000, + "1": 96000000000, + "10": 96000000000, + "11": 96000000000, + "12": 96000000000, + "13": 96000000000, + "14": 96000000000, + "15": 96000000000, + "2": 96000000000, + "3": 96000000000, + "4": 96000000000, + "5": 96000000000, + "6": 96000000000, + "7": 96000000000, + "8": 96000000000, + "9": 96000000000 + }, + "hbm3": { + "0": 96000000000, + "1": 96000000000, + "10": 96000000000, + "11": 96000000000, + "12": 96000000000, + "13": 96000000000, + "14": 96000000000, + "15": 96000000000, + "2": 96000000000, + "3": 96000000000, + "4": 96000000000, + "5": 96000000000, + "6": 96000000000, + "7": 96000000000, + "8": 96000000000, + "9": 96000000000 + } + }, + "dash_routes": {}, + "evidence": "SCENARIO_ASSUMPTION", + "fabric": { + "evidence": "SCENARIO_ASSUMPTION", + "hbf": { + "hbf0": { + "bank_capacity_bytes": 8388608, + "bank_count": 2, + "direct_link": { + "bandwidth_bytes_per_s": 2048000000000, + "latency_ns": 10 + }, + "fill": { + "bandwidth_bytes_per_s": 2048000000000, + "latency_ns": 10 + }, + "pair": null, + "relay_link": null + }, + "hbf1": { + "bank_capacity_bytes": 8388608, + "bank_count": 2, + "direct_link": { + "bandwidth_bytes_per_s": 2048000000000, + "latency_ns": 10 + }, + "fill": { + "bandwidth_bytes_per_s": 2048000000000, + "latency_ns": 10 + }, + "pair": null, + "relay_link": null + }, + "hbf2": { + "bank_capacity_bytes": 8388608, + "bank_count": 2, + "direct_link": { + "bandwidth_bytes_per_s": 2048000000000, + "latency_ns": 10 + }, + "fill": { + "bandwidth_bytes_per_s": 2048000000000, + "latency_ns": 10 + }, + "pair": null, + "relay_link": null + }, + "hbf3": { + "bank_capacity_bytes": 8388608, + "bank_count": 2, + "direct_link": { + "bandwidth_bytes_per_s": 2048000000000, + "latency_ns": 10 + }, + "fill": { + "bandwidth_bytes_per_s": 2048000000000, + "latency_ns": 10 + }, + "pair": null, + "relay_link": null + } + }, + "hbm": { + "hbm0": { + "bank_capacity_bytes": 4194304, + "bank_count": 2, + "gpu_link": { + "bandwidth_bytes_per_s": 2048000000000, + "latency_ns": 10 + } + }, + "hbm1": { + "bank_capacity_bytes": 4194304, + "bank_count": 2, + "gpu_link": { + "bandwidth_bytes_per_s": 2048000000000, + "latency_ns": 10 + } + }, + "hbm2": { + "bank_capacity_bytes": 4194304, + "bank_count": 2, + "gpu_link": { + "bandwidth_bytes_per_s": 2048000000000, + "latency_ns": 10 + } + }, + "hbm3": { + "bank_capacity_bytes": 4194304, + "bank_count": 2, + "gpu_link": { + "bandwidth_bytes_per_s": 2048000000000, + "latency_ns": 10 + } + } + } + }, + "maintenance_guard_policy": "LIGHT_SEVERE_EXEMPT_SHUTDOWN_BLOCKED", + "operation_cost_evidence": "PROXY_16_PLANES_4KIB_100US_PROGRAM_1MS_ERASE_256PAGES_PER_BLOCK", + "operation_media_cost": { + "erase": { + "denominator": 8192, + "numerator": 46875 + }, + "migration": { + "denominator": 1, + "numerator": 1 + }, + "program": { + "denominator": 64, + "numerator": 9375 + }, + "read": { + "denominator": 1, + "numerator": 1 + }, + "refresh_read": { + "denominator": 1, + "numerator": 1 + }, + "retry": { + "denominator": 1, + "numerator": 1 + } + }, + "schema_version": "eq3-topology-fluid-config-v1", + "topology": "mixed_direct", + "window_ns": 20000000 + }, + "strategy": "read_rate_feedback_thermal_guard_v1", + "thermal_limits_k": { + "gpu": [ + 363.15, + 373.15, + 383.15 + ], + "hbf": [ + 353.15, + 363.15, + 378.15 + ], + "hbm": [ + 353.15, + 363.15, + 378.15 + ] + }, + "topology": "mixed_direct", + "unknown_idle_power": "EXCLUDED_NOT_ZERO_MEASUREMENT", + "workload": { + "active_ns": 8000000000, + "channel_distribution": "uniform", + "pattern": "continuous", + "per_stack_Bps": 384000000000 + } +} diff --git a/experiments/eq3_system_thermal/maintenance_pilot_inputs_v2/maint-pilot-v2-relay-shared-01.json b/experiments/eq3_system_thermal/maintenance_pilot_inputs_v2/maint-pilot-v2-relay-shared-01.json new file mode 100644 index 0000000..d82ba0b --- /dev/null +++ b/experiments/eq3_system_thermal/maintenance_pilot_inputs_v2/maint-pilot-v2-relay-shared-01.json @@ -0,0 +1,371 @@ +{ + "energy": { + "ecc_j_per_byte": 0.0, + "ecc_scope": "ZERO_INCREMENT_BASELINE_NOT_MEASURED_ZERO; scenario overrides explicit", + "erase_array_j_per_operation": 5e-05, + "evidence": { + "erase": "DERIVED_ENGINEERING_PROXY_0.05W_TIMES_NATIVE_FIXTURE_1MS_NOT_HBF_MEASUREMENT", + "program": "DERIVED_ENGINEERING_PROXY_OLD_0.05W_0.01W_100US_4KIB; NOT_HBF_CALIBRATION", + "read": "USER_CONFIRMED_40_ARRAY_10_BASE_PJ_PER_B", + "relay_hbm": "REUSED_EQ3_MAINTENANCE_ENERGY_PROFILE_SCENARIO" + }, + "hbm_array_j_per_byte": 4e-11, + "hbm_base_j_per_byte": 2e-12, + "program_array_j_per_byte": 1.220703125e-09, + "program_base_j_per_byte": 2.4414062500000004e-10, + "read_array_j_per_byte": 4e-11, + "read_base_j_per_byte": 1e-11, + "relay_partner_receive_j_per_byte": 2e-12, + "relay_partner_send_j_per_byte": 2e-12 + }, + "gpu_external_w": 0, + "maintenance": { + "aged_blocks_per_stack": 4096, + "block_bytes": 1048576, + "ea_ev": 1.04, + "energy_evidence": { + "erase": "DERIVED_ENGINEERING_PROXY_0.05W_NATIVE_ADAPTER_1MS_NOT_HBF_CALIBRATION", + "program": "DERIVED_ENGINEERING_PROXY_0.05W_100US_NOT_HBF_CALIBRATION" + }, + "erase_energy_j_per_operation": 5e-05, + "initial_equivalent_age_ns": 86398000000000, + "initial_temperature_k": 300.0, + "initial_wall_age_ns": 0, + "max_blocks_per_cohort": 16, + "mode": "shared", + "pages_per_block": 256, + "program_energy_j_per_byte": 1.220703125e-09, + "refresh_trigger": "equivalent_age_or_wall", + "spares_per_channel": 16 + }, + "maintenance_scope": { + "aged_subset_bytes_per_hbf_stack": 4294967296, + "aged_subset_semantics": "EXPLICIT_SUBSET_NOT_WHOLE_MODEL_OR_PRODUCT_CAPACITY", + "backend": "AGGREGATE_RATE_SERVICE_NOT_NATIVE_MQSIM", + "initial_age_fairness": "SHARED_PILOT_DAY_MINUS_2S_EQUIVALENT_AGE;NOT_24H_TIME_COMPRESSION;DISABLED_NOT_PART_OF_PILOT", + "spare_ownership": "16_PER_EACH_OF_16_CHANNELS_NO_CROSS_CHANNEL_MOVE", + "spares_per_stack": 256 + }, + "point_id": "maint-pilot-v2-relay-shared-01", + "recovery_ns": 4000000000, + "resource_limits": { + "address_space_gib": 8, + "watchdog_s": 600 + }, + "schema_version": "eq3-system-point-v1", + "scope": "CONDITIONAL_SIMULATED", + "service": { + "channels": { + "hbf0": { + "0": 96000000000, + "1": 96000000000, + "10": 96000000000, + "11": 96000000000, + "12": 96000000000, + "13": 96000000000, + "14": 96000000000, + "15": 96000000000, + "2": 96000000000, + "3": 96000000000, + "4": 96000000000, + "5": 96000000000, + "6": 96000000000, + "7": 96000000000, + "8": 96000000000, + "9": 96000000000 + }, + "hbf1": { + "0": 96000000000, + "1": 96000000000, + "10": 96000000000, + "11": 96000000000, + "12": 96000000000, + "13": 96000000000, + "14": 96000000000, + "15": 96000000000, + "2": 96000000000, + "3": 96000000000, + "4": 96000000000, + "5": 96000000000, + "6": 96000000000, + "7": 96000000000, + "8": 96000000000, + "9": 96000000000 + }, + "hbf2": { + "0": 96000000000, + "1": 96000000000, + "10": 96000000000, + "11": 96000000000, + "12": 96000000000, + "13": 96000000000, + "14": 96000000000, + "15": 96000000000, + "2": 96000000000, + "3": 96000000000, + "4": 96000000000, + "5": 96000000000, + "6": 96000000000, + "7": 96000000000, + "8": 96000000000, + "9": 96000000000 + }, + "hbf3": { + "0": 96000000000, + "1": 96000000000, + "10": 96000000000, + "11": 96000000000, + "12": 96000000000, + "13": 96000000000, + "14": 96000000000, + "15": 96000000000, + "2": 96000000000, + "3": 96000000000, + "4": 96000000000, + "5": 96000000000, + "6": 96000000000, + "7": 96000000000, + "8": 96000000000, + "9": 96000000000 + }, + "hbm0": { + "0": 96000000000, + "1": 96000000000, + "10": 96000000000, + "11": 96000000000, + "12": 96000000000, + "13": 96000000000, + "14": 96000000000, + "15": 96000000000, + "2": 96000000000, + "3": 96000000000, + "4": 96000000000, + "5": 96000000000, + "6": 96000000000, + "7": 96000000000, + "8": 96000000000, + "9": 96000000000 + }, + "hbm1": { + "0": 96000000000, + "1": 96000000000, + "10": 96000000000, + "11": 96000000000, + "12": 96000000000, + "13": 96000000000, + "14": 96000000000, + "15": 96000000000, + "2": 96000000000, + "3": 96000000000, + "4": 96000000000, + "5": 96000000000, + "6": 96000000000, + "7": 96000000000, + "8": 96000000000, + "9": 96000000000 + }, + "hbm2": { + "0": 96000000000, + "1": 96000000000, + "10": 96000000000, + "11": 96000000000, + "12": 96000000000, + "13": 96000000000, + "14": 96000000000, + "15": 96000000000, + "2": 96000000000, + "3": 96000000000, + "4": 96000000000, + "5": 96000000000, + "6": 96000000000, + "7": 96000000000, + "8": 96000000000, + "9": 96000000000 + }, + "hbm3": { + "0": 96000000000, + "1": 96000000000, + "10": 96000000000, + "11": 96000000000, + "12": 96000000000, + "13": 96000000000, + "14": 96000000000, + "15": 96000000000, + "2": 96000000000, + "3": 96000000000, + "4": 96000000000, + "5": 96000000000, + "6": 96000000000, + "7": 96000000000, + "8": 96000000000, + "9": 96000000000 + } + }, + "dash_routes": {}, + "evidence": "SCENARIO_ASSUMPTION", + "fabric": { + "evidence": "SCENARIO_ASSUMPTION", + "hbf": { + "hbf0": { + "bank_capacity_bytes": 8388608, + "bank_count": 2, + "direct_link": { + "bandwidth_bytes_per_s": 2048000000000, + "latency_ns": 10 + }, + "fill": { + "bandwidth_bytes_per_s": 2048000000000, + "latency_ns": 10 + }, + "pair": "hbm0", + "relay_link": { + "bandwidth_bytes_per_s": 2048000000000, + "latency_ns": 10 + } + }, + "hbf1": { + "bank_capacity_bytes": 8388608, + "bank_count": 2, + "direct_link": { + "bandwidth_bytes_per_s": 2048000000000, + "latency_ns": 10 + }, + "fill": { + "bandwidth_bytes_per_s": 2048000000000, + "latency_ns": 10 + }, + "pair": "hbm1", + "relay_link": { + "bandwidth_bytes_per_s": 2048000000000, + "latency_ns": 10 + } + }, + "hbf2": { + "bank_capacity_bytes": 8388608, + "bank_count": 2, + "direct_link": { + "bandwidth_bytes_per_s": 2048000000000, + "latency_ns": 10 + }, + "fill": { + "bandwidth_bytes_per_s": 2048000000000, + "latency_ns": 10 + }, + "pair": "hbm2", + "relay_link": { + "bandwidth_bytes_per_s": 2048000000000, + "latency_ns": 10 + } + }, + "hbf3": { + "bank_capacity_bytes": 8388608, + "bank_count": 2, + "direct_link": { + "bandwidth_bytes_per_s": 2048000000000, + "latency_ns": 10 + }, + "fill": { + "bandwidth_bytes_per_s": 2048000000000, + "latency_ns": 10 + }, + "pair": "hbm3", + "relay_link": { + "bandwidth_bytes_per_s": 2048000000000, + "latency_ns": 10 + } + } + }, + "hbm": { + "hbm0": { + "bank_capacity_bytes": 4194304, + "bank_count": 2, + "gpu_link": { + "bandwidth_bytes_per_s": 2048000000000, + "latency_ns": 10 + } + }, + "hbm1": { + "bank_capacity_bytes": 4194304, + "bank_count": 2, + "gpu_link": { + "bandwidth_bytes_per_s": 2048000000000, + "latency_ns": 10 + } + }, + "hbm2": { + "bank_capacity_bytes": 4194304, + "bank_count": 2, + "gpu_link": { + "bandwidth_bytes_per_s": 2048000000000, + "latency_ns": 10 + } + }, + "hbm3": { + "bank_capacity_bytes": 4194304, + "bank_count": 2, + "gpu_link": { + "bandwidth_bytes_per_s": 2048000000000, + "latency_ns": 10 + } + } + } + }, + "maintenance_guard_policy": "LIGHT_SEVERE_EXEMPT_SHUTDOWN_BLOCKED", + "operation_cost_evidence": "PROXY_16_PLANES_4KIB_100US_PROGRAM_1MS_ERASE_256PAGES_PER_BLOCK", + "operation_media_cost": { + "erase": { + "denominator": 8192, + "numerator": 46875 + }, + "migration": { + "denominator": 1, + "numerator": 1 + }, + "program": { + "denominator": 64, + "numerator": 9375 + }, + "read": { + "denominator": 1, + "numerator": 1 + }, + "refresh_read": { + "denominator": 1, + "numerator": 1 + }, + "retry": { + "denominator": 1, + "numerator": 1 + } + }, + "schema_version": "eq3-topology-fluid-config-v1", + "topology": "relay", + "window_ns": 20000000 + }, + "strategy": "read_rate_feedback_thermal_guard_v1", + "thermal_limits_k": { + "gpu": [ + 363.15, + 373.15, + 383.15 + ], + "hbf": [ + 353.15, + 363.15, + 378.15 + ], + "hbm": [ + 353.15, + 363.15, + 378.15 + ] + }, + "topology": "relay", + "unknown_idle_power": "EXCLUDED_NOT_ZERO_MEASUREMENT", + "workload": { + "active_ns": 8000000000, + "channel_distribution": "uniform", + "pattern": "continuous", + "per_stack_Bps": 384000000000 + } +} diff --git a/experiments/eq3_system_thermal/maintenance_point.schema.json b/experiments/eq3_system_thermal/maintenance_point.schema.json new file mode 100644 index 0000000..a4074fe --- /dev/null +++ b/experiments/eq3_system_thermal/maintenance_point.schema.json @@ -0,0 +1,30 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "title": "EQ3 aggregate maintenance point extension", + "type": "object", + "required": ["mode", "ea_ev", "refresh_trigger", "initial_equivalent_age_ns", + "initial_wall_age_ns", "initial_temperature_k", "block_bytes", "pages_per_block", + "aged_blocks_per_stack", "spares_per_channel", "max_blocks_per_cohort", + "program_energy_j_per_byte", "erase_energy_j_per_operation", "energy_evidence"], + "additionalProperties": false, + "properties": { + "mode": {"enum": ["disabled", "shared", "ideal_independent"]}, + "ea_ev": {"enum": [1.01, 1.04, 1.08]}, + "refresh_trigger": {"enum": ["wall_only", "equivalent_age_or_wall"]}, + "initial_equivalent_age_ns": {"type": "integer", "minimum": 0}, + "initial_wall_age_ns": {"type": "integer", "minimum": 0}, + "initial_temperature_k": {"type": "number", "exclusiveMinimum": 0}, + "block_bytes": {"const": 1048576}, + "pages_per_block": {"const": 256}, + "aged_blocks_per_stack": {"const": 4096}, + "spares_per_channel": {"const": 16}, + "max_blocks_per_cohort": {"type": "integer", "minimum": 1, "maximum": 16}, + "program_energy_j_per_byte": {"type": "number", "minimum": 0}, + "erase_energy_j_per_operation": {"type": "number", "minimum": 0}, + "energy_evidence": { + "type": "object", "required": ["program", "erase"], + "additionalProperties": false, + "properties": {"program": {"type": "string"}, "erase": {"type": "string"}} + } + } +} diff --git a/experiments/eq3_system_thermal/model_variants.py b/experiments/eq3_system_thermal/model_variants.py new file mode 100755 index 0000000..571a790 --- /dev/null +++ b/experiments/eq3_system_thermal/model_variants.py @@ -0,0 +1,293 @@ +#!/usr/bin/env python3 +"""Build explicit, default-disconnected thermal-model sensitivity variants. + +This tool derives a model directory from an existing layered RC export. It +does not run a thermal solver and does not modify its input directory. +""" + +from __future__ import annotations + +import argparse +import hashlib +import json +import math +import os +import shutil +import tempfile +from collections import defaultdict +from pathlib import Path +from typing import Any + + +SCHEMA = "eq3-system-thermal-model-variant-v1" +AMBIENTS_K = (300.0, 310.0, 320.0) +EXTERNAL_RESISTANCE_SCALES = (0.5, 1.0, 1.5) +COUPLING_MODES = ("full", "no_cross_domain_lateral") +REQUIRED_FILES = ("model.txt", "normalized.json", "rc_grid.json", "rc_sensors.json") + + +def _number(value: float) -> str: + return format(value, ".17g") + + +def _sha256(path: Path) -> str: + digest = hashlib.sha256() + with path.open("rb") as stream: + for block in iter(lambda: stream.read(1024 * 1024), b""): + digest.update(block) + return digest.hexdigest() + + +def _parse_model(path: Path) -> tuple[list[str], dict[str, dict[str, Any]], list[dict[str, Any]]]: + header: list[str] = [] + nodes: dict[str, dict[str, Any]] = {} + edges: list[dict[str, Any]] = [] + for line_number, raw in enumerate(path.read_text().splitlines(), 1): + fields = raw.split() + if not fields: + continue + if fields[0] == "node": + if len(fields) != 11 or fields[1] in nodes: + raise ValueError(f"invalid/duplicate node at {path}:{line_number}") + nodes[fields[1]] = { + "id": fields[1], "physical": fields[2], "role": fields[3], + "component": fields[4], "die": fields[5], + "capacity_j_k": float(fields[6]), "initial_k": float(fields[7]), + "static_w": float(fields[8]), "boundary_g_w_k": float(fields[9]), + "boundary_k": float(fields[10]), + } + elif fields[0] == "edge": + if len(fields) != 5: + raise ValueError(f"invalid edge at {path}:{line_number}") + edges.append({"a": fields[1], "b": fields[2], "g_w_k": float(fields[3]), + "kind": fields[4]}) + else: + header.append(raw) + if not nodes or header[:2] != ["HBFSIM_EQ3_THERMAL_MODEL 1", "coupling on"]: + raise ValueError("unsupported thermal model format") + return header, nodes, edges + + +def _device_domains(normalized: dict[str, Any], cells: list[dict[str, Any]]) -> dict[str, str]: + devices = [] + for device in normalized.get("devices", []): + if device.get("external"): + continue + physical = str(device.get("physical_type", "")).upper() + if physical == "GPU" or physical.startswith("HBM") or physical == "HBF": + xy = device["xy_m"] + footprint = device["footprint_m"] + devices.append((str(device["id"]), xy[0] + footprint[0] / 2, + xy[1] + footprint[1] / 2)) + if not devices: + raise ValueError("no internal GPU/HBM/HBF device anchors") + devices.sort() + domains: dict[str, str] = {} + for cell in cells: + x, y = cell["center_m"][:2] + # Lexicographic device id is the deterministic tie breaker. + domains[cell["id"]] = min(devices, key=lambda d: ((x-d[1])**2 + (y-d[2])**2, d[0]))[0] + return domains + + +def _boundary_conductance(cell: dict[str, Any], normalized: dict[str, Any], + nz: int, scale: float) -> tuple[float, dict[str, float]]: + _x, _y, z = cell["xyz_index"] + area = cell["size_m"][0] * cell["size_m"][1] + half_cell_r = cell["size_m"][2] / (2 * cell["k_xyz_w_m_k"][2] * area) + parts: dict[str, float] = {} + for side, applies in (("bottom", z == 0), ("top", z == nz - 1)): + if not applies: + continue + h = float(normalized["boundaries"][side]["h_w_m2_k"]) + convection_r = math.inf if h == 0 else scale / (h * area) + parts[side] = 0.0 if math.isinf(convection_r) else 1.0 / (half_cell_r + convection_r) + return sum(parts.values()), parts + + +def _audit(nodes: dict[str, dict[str, Any]], edges: list[dict[str, Any]]) -> dict[str, Any]: + incident = defaultdict(float) + pairs: set[tuple[str, str]] = set() + for edge in edges: + a, b, g = edge["a"], edge["b"], edge["g_w_k"] + if a not in nodes or b not in nodes or a == b or not math.isfinite(g) or g <= 0: + raise ValueError("invalid retained thermal edge") + pair = tuple(sorted((a, b))) + if pair in pairs: + raise ValueError(f"duplicate thermal edge {pair}") + pairs.add(pair) + incident[a] += g + incident[b] += g + max_row_residual = 0.0 + diagonal_sum = 0.0 + for node_id, node in nodes.items(): + diagonal = incident[node_id] + node["boundary_g_w_k"] + diagonal_sum += diagonal + max_row_residual = max(max_row_residual, + abs(diagonal - incident[node_id] - node["boundary_g_w_k"])) + return { + "edge_count": len(edges), + "edge_conductance_sum_w_k": sum(e["g_w_k"] for e in edges), + "boundary_conductance_sum_w_k": sum(n["boundary_g_w_k"] for n in nodes.values()), + "laplacian_diagonal_sum_w_k": diagonal_sum, + "max_reconstructed_row_balance_residual_w_k": max_row_residual, + "symmetric_unique_edge_pairs": True, + "internal_edge_energy_conservation": "PASS", + } + + +def build_variant(source_model_dir: Path | str, output_dir: Path | str, *, ambient_k: float, + external_resistance_scale: float, coupling_mode: str) -> dict[str, Any]: + """Create one self-contained derived model directory and return its manifest.""" + source = Path(source_model_dir).resolve(strict=True) + output = Path(output_dir).resolve() + if output.exists(): + raise FileExistsError(output) + if ambient_k not in AMBIENTS_K or not math.isfinite(ambient_k): + raise ValueError(f"ambient_k must be one of {AMBIENTS_K}") + if external_resistance_scale not in EXTERNAL_RESISTANCE_SCALES: + raise ValueError(f"external_resistance_scale must be one of {EXTERNAL_RESISTANCE_SCALES}") + if coupling_mode not in COUPLING_MODES: + raise ValueError(f"coupling_mode must be one of {COUPLING_MODES}") + for name in REQUIRED_FILES: + if not (source / name).is_file(): + raise FileNotFoundError(source / name) + + normalized = json.loads((source / "normalized.json").read_text()) + grid = json.loads((source / "rc_grid.json").read_text()) + cells = grid["cells"] + if len(cells) != math.prod(grid["shape"]): + raise ValueError("grid shape/cell count mismatch") + header, nodes, edges = _parse_model(source / "model.txt") + index = {cell["id"]: int(cell["index"]) for cell in cells} + cell_by_id = {cell["id"]: cell for cell in cells} + if set(nodes) != set(index) or len(index) != len(cells): + raise ValueError("model/grid node identity mismatch") + + original_capacity = sum(node["capacity_j_k"] for node in nodes.values()) + original_static = sum(node["static_w"] for node in nodes.values()) + original_boundary_g = sum(node["boundary_g_w_k"] for node in nodes.values()) + boundary_parts = {"top": {"old_external_r_m2_k_w": 1 / float(normalized["boundaries"]["top"]["h_w_m2_k"]), + "new_external_r_m2_k_w": external_resistance_scale / float(normalized["boundaries"]["top"]["h_w_m2_k"])}, + "bottom": {"old_external_r_m2_k_w": 1 / float(normalized["boundaries"]["bottom"]["h_w_m2_k"]), + "new_external_r_m2_k_w": external_resistance_scale / float(normalized["boundaries"]["bottom"]["h_w_m2_k"])}} + boundary_side_g = defaultdict(float) + for node_id, node in nodes.items(): + g, parts = _boundary_conductance(cell_by_id[node_id], normalized, grid["shape"][2], + external_resistance_scale) + node["boundary_g_w_k"] = g + node["initial_k"] = ambient_k + node["boundary_k"] = ambient_k + for side, value in parts.items(): + boundary_side_g[side] += value + for side in ("top", "bottom"): + boundary_parts[side]["new_conductance_sum_w_k"] = boundary_side_g[side] + + domains = _device_domains(normalized, cells) + kept_edges: list[dict[str, Any]] = [] + removed_edges: list[dict[str, Any]] = [] + axis_counts = defaultdict(int) + for edge in edges: + ca, cb = cell_by_id.get(edge["a"]), cell_by_id.get(edge["b"]) + if ca is None or cb is None: + raise ValueError("edge references a node absent from grid") + delta = [abs(a-b) for a, b in zip(ca["xyz_index"], cb["xyz_index"])] + if sum(delta) != 1: + raise ValueError("non-neighbor grid edge") + axis = delta.index(1) + cut = (coupling_mode == "no_cross_domain_lateral" and axis in (0, 1) + and domains[edge["a"]] != domains[edge["b"]]) + (removed_edges if cut else kept_edges).append(edge) + axis_counts[("removed" if cut else "kept", "xyz"[axis])] += 1 + + audit = _audit(nodes, kept_edges) + if not math.isclose(original_capacity, sum(n["capacity_j_k"] for n in nodes.values()), + rel_tol=0, abs_tol=1e-12): + raise ValueError("capacity changed") + if not math.isclose(original_static, sum(n["static_w"] for n in nodes.values()), + rel_tol=0, abs_tol=1e-12): + raise ValueError("static power changed") + + normalized["boundaries"]["initial_temperature_k"] = ambient_k + normalized["boundaries"]["top"]["ambient_k"] = ambient_k + normalized["boundaries"]["bottom"]["ambient_k"] = ambient_k + + source_hashes = {name: _sha256(source / name) for name in REQUIRED_FILES} + manifest: dict[str, Any] = { + "schema_version": SCHEMA, + "status": "DERIVED_NOT_SOLVED", + "evidence_class": "CONDITIONAL_ABLATION" if coupling_mode != "full" else "SENSITIVITY_VARIANT", + "source_model_dir": str(source), + "source_hashes_sha256": source_hashes, + "ambient_k": ambient_k, + "external_resistance_scale": external_resistance_scale, + "external_resistance_semantics": "scales only 1/h on top and bottom; cell half-thickness conduction and material k are unchanged", + "boundary_resistance_parts": boundary_parts, + "coupling_mode": coupling_mode, + "coupling_semantics": ("full source network" if coupling_mode == "full" else + "VORONOI_THERMAL_DOMAIN_LATERAL_ABLATION: nearest device footprint-center XY domain; removes cross-domain x/y edges including substrate lateral paths and GPU-memory paths; preserves every z edge, within-domain x/y edge, and original per-node top/bottom shared cooling"), + "not_equivalent_to_independent_packages": coupling_mode != "full", + "node_count": len(nodes), + "retained_edge_count": len(kept_edges), + "removed_edge_count": len(removed_edges), + "removed_edge_conductance_sum_w_k": sum(e["g_w_k"] for e in removed_edges), + "edge_axis_counts": {f"{state}_{axis}": count for (state, axis), count in sorted(axis_counts.items())}, + "capacity_j_k_unchanged": original_capacity, + "static_power_w_unchanged": original_static, + "source_boundary_conductance_sum_w_k": original_boundary_g, + "derived_boundary_conductance_sum_w_k": sum(n["boundary_g_w_k"] for n in nodes.values()), + "matrix_audit": audit, + "solver_started": False, + "fast_thermal_rom": "UNAVAILABLE_FOR_CURRENT_GEOMETRY", + } + + output.parent.mkdir(parents=True, exist_ok=True) + temp = Path(tempfile.mkdtemp(prefix=output.name + ".tmp-", dir=output.parent)) + try: + for name in REQUIRED_FILES: + if name not in ("model.txt", "normalized.json"): + shutil.copy2(source / name, temp / name) + ordered_nodes = sorted(nodes.values(), key=lambda n: index[n["id"]]) + ordered_edges = sorted(kept_edges, key=lambda e: (min(index[e["a"]], index[e["b"]]), + max(index[e["a"]], index[e["b"]]))) + lines = list(header) + for n in ordered_nodes: + lines.append("node {} {} {} {} {} {} {} {} {} {}".format( + n["id"], n["physical"], n["role"], n["component"], n["die"], + _number(n["capacity_j_k"]), _number(n["initial_k"]), _number(n["static_w"]), + _number(n["boundary_g_w_k"]), _number(n["boundary_k"]))) + for edge in ordered_edges: + a, b = sorted((edge["a"], edge["b"]), key=index.__getitem__) + lines.append(f"edge {a} {b} {_number(edge['g_w_k'])} {edge['kind']}") + (temp / "model.txt").write_text("\n".join(lines) + "\n") + (temp / "normalized.json").write_text(json.dumps(normalized, indent=2, sort_keys=True) + "\n") + manifest["derived_hashes_sha256"] = { + name: _sha256(temp / name) for name in REQUIRED_FILES + } + (temp / "variant_manifest.json").write_text(json.dumps(manifest, indent=2, sort_keys=True) + "\n") + os.rename(temp, output) + except BaseException: + shutil.rmtree(temp, ignore_errors=True) + raise + return manifest + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--source-model-dir", type=Path, required=True) + parser.add_argument("--output-dir", type=Path, required=True) + parser.add_argument("--ambient-k", type=float, choices=AMBIENTS_K, required=True) + parser.add_argument("--external-resistance-scale", type=float, + choices=EXTERNAL_RESISTANCE_SCALES, required=True) + parser.add_argument("--coupling-mode", choices=COUPLING_MODES, required=True) + args = parser.parse_args() + manifest = build_variant(args.source_model_dir, args.output_dir, + ambient_k=args.ambient_k, + external_resistance_scale=args.external_resistance_scale, + coupling_mode=args.coupling_mode) + print(json.dumps(manifest, sort_keys=True)) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/experiments/eq3_system_thermal/prepare_causal_campaign.py b/experiments/eq3_system_thermal/prepare_causal_campaign.py new file mode 100644 index 0000000..1ad4ad8 --- /dev/null +++ b/experiments/eq3_system_thermal/prepare_causal_campaign.py @@ -0,0 +1,92 @@ +#!/usr/bin/env python3 +"""Prepare bounded model-derived causal pilots; never launch a solver.""" +import argparse +from copy import deepcopy +import hashlib +import json +import math +from pathlib import Path +from prepare_stage import config as base_config, TOPOLOGIES, STRATEGIES +from causal_workload import build_architecture_trace + +HERE=Path(__file__).resolve().parent + +def save(path,value): + path.parent.mkdir(parents=True,exist_ok=True) + path.write_text(json.dumps(value,indent=2,allow_nan=False)+"\n") + +def make_config(point_id,topology,strategy,model_id,active_s,recovery_s,rate=1_536_000_000_000): + base=base_config(point_id,topology,rate,strategy,active_s,recovery_s) + service=base['service'];groups={};targets=[] + for stack in service['fabric']['hbf']: + channels=['direct','relay'] if topology=='dash' else ['uniform'] + service['channels'][stack]={c:1_536_000_000_000//len(channels) for c in channels} + groups[stack]={c:{'resource_id':stack+':uniform16-media', + 'bandwidth_bytes_per_s':1_536_000_000_000} for c in channels} + for c in channels: + route=c if topology=='dash' else ('relay' if topology=='relay' else 'direct') + targets.append({'stack':stack,'channel':c,'route':route}) + if topology=='dash': + service['dash_routes']={s:{'direct':'direct','relay':'relay'} for s in service['fabric']['hbf']} + for stack, channels in service['channels'].items(): + if stack not in groups: + groups[stack]={c:{'resource_id':stack+':ch'+c, + 'bandwidth_bytes_per_s':rate} for c,rate in channels.items()} + service['causal_channel_groups']=groups + trace={'model_id':model_id,'batch_intervals':1,'batch_size':1,'batch_interval_ns':1, + 'prefetch_layers':0,'prefetch_mode':'on_demand','attention_compute_ns_per_token':1000, + 'mlp_compute_ns_per_token':1000,'output_compute_ns_per_token':1000, + 'embedding_access':'selected_token_rows','max_active_batches':4} + trace_path=HERE/'stage/traces/tiny_qwen2_cpu_trace_v1/trace.json' + trace.update(dependency_mode='tiny_cpu_template', + tiny_trace_path=str(trace_path.relative_to(HERE.parents[1])), + tiny_trace_sha256=json.loads(trace_path.read_text())['trace_sha256'], + projection_context_tokens=4096) + probe=build_architecture_trace(trace) + read_bytes=sum(t['tensor']['bytes'] for t in probe['batches'][0]['tasks'] if t['type']=='storage') + n=len(service['fabric']['hbf']) + trace['batch_interval_ns']=max(1,read_bytes*10**9//(n*rate)) + trace['total_batches']=(int(active_s*1e9)-1)//trace['batch_interval_ns']+1 + return {'schema_version':'eq3-causal-point-v1','point_id':point_id,'topology':topology, + 'strategy':strategy,'active_ns':int(active_s*1e9),'recovery_ns':int(recovery_s*1e9), + 'service':service,'causal_channel_granularity':{'mode':'uniform_stack_group_16ch', + 'physical_channels_per_stack':16,'semantics':'UNIFORM_EQUIVALENT_GROUP_WITH_SHARED_DASH_MEDIA_NOT_ONE_PHYSICAL_CHANNEL'}, + 'trace':trace,'executor':{'cache_mode':'disabled','cache_capacity_bytes':0, + 'coalescing_enabled':True,'prefetch_wait_mode':'wait_at_consumption','stripe_unit_bytes':4096, + 'migration_mode':'fixed','retry_count_per_source_read':0,'stripe_targets':targets}, + 'target_bytes_per_s_by_stack':{s:rate for s in service['channels']}, + 'gpu_compute_w':0,'gpu_external_w':0,'thermal_limits_k':base['thermal_limits_k'], + 'resource_limits':{'address_space_gib':8,'watchdog_s':900}, + 'maintenance':{'mode':'disabled'},'evidence':{ + 'read_pressure':'MODEL_REGENERATED_PAYLOAD_PER_BATCH_TIMES_EXOGENOUS_BATCH_ARRIVALS', + 'nominal_per_batch_read_bytes':read_bytes,'nominal_per_stack_input_Bps':rate, + 'hbm_policy_target':'NOMINAL_ENDPOINT_PROFILE_ONLY_NO_HBM_REQUEST_GENERATION', + 'coalescing':'ACTUAL_PHYSICAL_DEMAND_MAY_BE_LOWER_THAN_NOMINAL_LOGICAL_DEMAND', + 'compute':'1US_PER_BUNDLE_EXPLICIT_MEMORY_BOUND_SCENARIO_NOT_GPU_CALIBRATION', + 'gpu_power':'UNKNOWN_SELF_HEAT_EXCLUDED_NOT_MEASURED_ZERO', + 'grain':'UNIFORM_CHANNEL_GROUP_EQUIVALENCE_FIXED_TESTED; NONUNIFORM_USES_SEPARATE_16CH_RATE_POINTS', + 'token':'ONLY_FINAL_CAUSAL_COMPUTE_COMPLETION; NOT_REAL_MODEL_THROUGHPUT'}} + +def main(): + ap=argparse.ArgumentParser();ap.add_argument('--stage',type=Path,required=True) + ap.add_argument('--kind',choices=['pilot','main'],default='pilot') + a=ap.parse_args();stage=a.stage.resolve();dest=stage/'causal-v1' + index=json.loads((stage/'RUN_INDEX.json').read_text());templates={r['topology']:r for r in index['points']} + points=[] + specs=[(t,STRATEGIES[2],'Qwen/Qwen2.5-72B-Instruct',8,4) for t in TOPOLOGIES] if a.kind=='pilot' else [ + (t,s,'Qwen/Qwen2.5-72B-Instruct',20,10) for t in TOPOLOGIES for s in STRATEGIES] + for t,s,m,active,recovery in specs: + pid=f"causal-{a.kind}-{t}-{m.split(chr(47))[-1]}-{STRATEGIES.index(s)}-01" + cfg=make_config(pid,t,s,m,active,recovery) + path=dest/'inputs'/(pid+'.json') + if path.exists():raise FileExistsError(path) + save(path,cfg);row=deepcopy(templates[t]);row.update(point_id=pid,kind=a.kind, + config=str(path),config_sha256=hashlib.sha256(path.read_bytes()).hexdigest(), + output=str(dest/'points'/pid));points.append(row) + save(dest/(a.kind.upper()+'_INDEX_CANDIDATE.json'),{'status':'CANDIDATE_NOT_FROZEN', + 'authorization':'USER_EXPLICIT_FOUR_TOPOLOGY_PLAN','points':points, + 'resources':{'point_wall_s':900,'stage_wall_s':14400,'point_output_gib':1, + 'stage_output_gib':20,'parent_combined_output_gib':100,'host_ram_reserve_gib':32,'host_disk_reserve_gib':100}, + 'pending':'Bind actual tiny trace consumer and source locks before execution; pilots reviewed before main.'}) + +if __name__=='__main__':main() diff --git a/experiments/eq3_system_thermal/prepare_causal_main.py b/experiments/eq3_system_thermal/prepare_causal_main.py new file mode 100644 index 0000000..4164d8a --- /dev/null +++ b/experiments/eq3_system_thermal/prepare_causal_main.py @@ -0,0 +1,97 @@ +#!/usr/bin/env python3 +"""Generate bounded causal comparisons only after actual pilot receipt review.""" +import argparse +from copy import deepcopy +import json +import math +from pathlib import Path +from analyze_extensions import analyze_point +from prepare_causal_campaign import make_config +from prepare_stage import STRATEGIES, TOPOLOGIES +from freeze_extension import freeze, save, HERE + + +def specs(): + result=[] + for topology in TOPOLOGIES: + for strategy in STRATEGIES: + result.append((topology,strategy,'72B',1536,'baseline')) + for topology in ('mixed_direct','relay','dash'): + result += [(topology,STRATEGIES[2],'7B',384,'baseline'), + (topology,STRATEGIES[2],'72B',1536,'prefetch1'), + (topology,STRATEGIES[2],'72B',1536,'prefetch1_issue_stall'), + (topology,STRATEGIES[2],'72B',1536,'no_coalescing'), + (topology,STRATEGIES[2],'72B',1536,'hot_static'), + (topology,STRATEGIES[2],'72B',1536,'hot_adaptive'), + (topology,STRATEGIES[2],'7B',384,'cache16GiB')] + result += [('mixed_direct',STRATEGIES[2],'7B',384,'cache4GiB'), + ('mixed_direct',STRATEGIES[2],'72B',1536,'retry1')] + return result + + +def make_main(topology,strategy,model,rate,arm): + pid=f'causal-main-{topology}-{model}-{rate}-{STRATEGIES.index(strategy)}-{arm}-01' + c=make_config(pid,topology,strategy,f'Qwen/Qwen2.5-{model}-Instruct',20,10,rate*10**9) + c['resource_limits']['watchdog_s']=1800 + c['comparison_arm']=arm + if arm.startswith('prefetch1'): + c['trace'].update(prefetch_layers=1,prefetch_mode='layer_lookahead') + if arm.endswith('issue_stall'):c['executor']['prefetch_wait_mode']='stall_at_issue' + if arm=='no_coalescing':c['executor']['coalescing_enabled']=False + if arm in ('hot_static','hot_adaptive'): + original=deepcopy(c['executor']['stripe_targets']) + c['executor']['stripe_targets']=[{**r,'stack':'hbf0'} for r in original] + c['evidence']['placement_input']='ALL_INITIAL_SOURCE_WEIGHT_STRIPES_ON_HBF0; SAME_LOGICAL_MODEL_ARRIVALS_AS_BASELINE' + if arm=='hot_adaptive': + c['executor'].update(migration_mode='basic',migration_access_threshold=2, + migration_capacity_bytes=160*1024**3,fast_stripe_targets=original) + c['evidence']['migration_capacity']='FINITE_160GIB_DESTINATION_RESERVATION_ACROSS_EXISTING_HBF_STACKS; PROGRAM_ERASE_COST_COUNTED' + if arm.startswith('cache'): + capacity=int(arm[len('cache'):-len('GiB')])*1024**3 + targets=c['executor']['stripe_targets'] + for stack in c['service']['fabric']['hbm']: + capacity_Bps=sum(c['service']['channels'][stack].values()) + c['service']['channels'][stack]={'uniform':capacity_Bps} + c['service']['causal_channel_groups'][stack]={'uniform':{ + 'resource_id':stack+':uniform16-media','bandwidth_bytes_per_s':capacity_Bps}} + fast=[{'stack':f'hbm{i%4}','channel':'uniform','route':'direct'} for i,_ in enumerate(targets)] + c['executor'].update(cache_mode='external_hbm',cache_capacity_bytes=capacity, + fast_stripe_targets=fast) + c['evidence']['cache_capacity']='TOTAL_4_OR_16GIB_SCENARIO_ALLOCATION_WITHIN_FOUR_36GB_HBM_LABELS; ACTUAL_FILL_LRU_HIT_SERVICE' + if arm=='retry1':c['executor']['retry_count_per_source_read']=1 + c['evidence']['coalescing_scope']='IDENTICAL_WHOLE_TENSOR_READS_SHARE_ALL_THEIR_PAGES; ARBITRARY_PARTIAL_PAGE_OVERLAP_UNSUPPORTED' + return c + + +def main(): + p=argparse.ArgumentParser(description=__doc__);p.add_argument('--stage',type=Path,required=True) + p.add_argument('--pilot-index',type=Path,required=True);a=p.parse_args() + stage=a.stage.resolve();index=json.loads(a.pilot_index.read_text()) + if len(index['points'])!=4:raise ValueError('four actual topology pilots required') + costs=[] + for row in index['points']: + point=Path(row['output']);analysis=analyze_point(point) + if analysis['analysis_status']!='VALIDATED_COMPLETE_RECEIPTS':raise ValueError('pilot not validated') + done=json.loads((point/'DONE.json').read_text());costs.append(done['wall_s']) + # Six-hour extension is finite. Stop at review rather than repeatedly launch + # points whose pilot already predicts missing the whole-stage budget. + predicted=sum(costs)/len(costs)*2.5*len(specs()) + if predicted>21600:raise RuntimeError('measured pilot projects >6h causal extension; resource/identifiability review required') + dest=stage/'causal-main-v1' + if dest.exists():raise FileExistsError(dest) + dest.mkdir();paths=[] + for spec in specs(): + c=make_main(*spec);path=dest/'inputs'/(c['point_id']+'.json') + path.parent.mkdir(parents=True,exist_ok=True);save(path,c);paths.append(path) + frozen=freeze(stage,dest,paths,HERE/'run_causal_point.py',point_wall_s=1800, + dependency=a.pilot_index.parent/'DONE.json') + frozen['pilot_review']={'actual_wall_s':costs,'linear_30s_projection_s':predicted, + 'interpretation':'UNCERTAIN_LINEAR_PROJECTION; MAIN_WATCHDOGS_STILL_ENFORCED'} + frozen['design']={'core_72b_policy_topology_points':12,'total_points':len(specs()), + 'secondary_7b_rate_per_stack_TBps':.384,'primary_72b_rate_per_stack_TBps':1.536, + 'cross_model_comparison':'DIFFERENT_DEMAND_NOT_ISOLATED_MODEL_SIZE_EFFECT', + 'paired_arms':'same model/rate/topology/policy/age and initial state within each ablation', + 'unavailable':'QUALIFIED_THERMAL_FAST_PATH; REAL_GPU_TIMING; ARBITRARY_PARTIAL_PAGE_COALESCING'} + save(dest/'MAIN_INDEX.json',frozen) + +if __name__=='__main__':main() diff --git a/experiments/eq3_system_thermal/prepare_ecc_pilots.py b/experiments/eq3_system_thermal/prepare_ecc_pilots.py new file mode 100644 index 0000000..7746702 --- /dev/null +++ b/experiments/eq3_system_thermal/prepare_ecc_pilots.py @@ -0,0 +1,56 @@ +#!/usr/bin/env python3 +"""Prepare paired bounded software-integration thermal pilots; never launch.""" +import argparse +from copy import deepcopy +import json +from pathlib import Path +from prepare_causal_campaign import make_config +from prepare_stage import TOPOLOGIES +from freeze_extension import freeze,save + +HERE=Path(__file__).resolve().parent + +def prepare(stage,destination): + if destination.exists():raise FileExistsError(destination) + destination.mkdir(parents=True) + inputs=destination/'inputs';inputs.mkdir() + profile=json.loads((HERE/'ecc_proxy/profile_v1.json').read_text()) + configs=[] + for topology in TOPOLOGIES: + for strength in (0,.1): + name=f'ecc-pilot-{topology}-s{strength:g}-01' + config=make_config(name,topology,'guard_only','Qwen/Qwen2.5-72B-Instruct',1,1) + p=deepcopy(profile);p['transfer_strength']=strength + config['hbf_read_cost_proxy']={'mode':'conditional_nand_history_v1','profile':p, + 'initial_by_stack':{s:{'equivalent_age_days_30c':90,'pe_cycles':1000, + 'temperature_k':300} for s in config['service']['fabric']['hbf']}} + config['evidence']['purpose']='PAIRED_PROXY_SERVICE_ENERGY_DAG_INTEGRATION_NOT_CONTROLLER_BENEFIT' + path=inputs/(name+'.json');save(path,config);configs.append(path) + index=freeze(stage,destination,configs,HERE/'run_causal_point.py',point_wall_s=600, + dependency=stage/'maintenance-v1/DONE.json') + index['authorization']='USER_FOUR_TOPOLOGY_PLAN_PLUS_HBF_NAND_OCP_SANDISK_CONDITIONAL_PROXY_CLARIFICATION' + index['resources'].update(stage_wall_s=3600,sensitivity_output_gib=8) + save(destination/'PILOT_INDEX.json',index) + preflight={ + 'status':'FROZEN_WAITING_SERIAL_CPU_RELEASE','question':'Does optional HBF retry cost consume media/decoder resources and energy while preserving unique useful delivery and four-topology routing?', + 'classification':'CONDITIONAL_SIMULATED_ENGINEERING_PILOT_NOT_HBF_CALIBRATION', + 'environment_id':'eq3-thermal-cpu-v1','points':8,'repetitions':1, + 'duration':'1s input +1s drain; backlog may remain','nominal_demand_TBps_per_HBF':1.536, + 'model':'Qwen2.5-72B regenerated shapes using recorded tiny same-architecture dependency structure', + 'initial_age':'90 equivalent days at30C uniformly perHBF; NOT90 wall days at85C', + 'initial_PE':1000,'initial_temperature_K':300, + 'comparison':'same input, transfer factor0 versus0.1; bothguard_only, no static retries/maintenance', + 'expected_effect':'roughly1.9 media effort atinitial age inweak proxy; actual goodput limitedby allresources; lower useful work maylower ornotchange temperature', + 'confounders':['decoder throughput assumption','uniform stack age conservatively driven hottestarraydie','initialwearfixed','shortagecurve assumed','DAGcoalescingchangesphysicaldemand'], + 'minimum_validation':['resource service slows eligible HBF reads','physicalretryenergy present onlyenabled','usefuluniquecompletion','causal token status','no HBM NANDcost','nonnegative conserved energy','no future temperatures used'], + 'stop':['source/model/input mismatch','400K domain failure retains evidence','600s point or3600s stage','8GiB addressspace','2GiB pointoutput','host32GiB RAM/100GiB diskreserve'], + 'output':'immutable perwindowraw, source manifest, costdecisionreceipt, DONE/FAILED, pairedanalysis andtimepanels', + 'interpretation':'No controller benefit or realHBFerror prediction. Refresh→ECC coupling stillunsupported untilmatchingdataextentconsumer; do notreset whole stack.', + 'next_gate':'strict pilot audit and cost review before new policy matrix; existing base retained as noECCbaseline', + } + save(destination/'PREFLIGHT.json',preflight) + (destination/'PREFLIGHT.md').write_text('# ECC paired integration pilots\n\n'+json.dumps(preflight,indent=2,ensure_ascii=False)+'\n') + +if __name__=='__main__': + ap=argparse.ArgumentParser();ap.add_argument('--stage',type=Path,required=True);ap.add_argument('--destination',type=Path,required=True) + a=ap.parse_args();prepare(a.stage.resolve(),a.destination.resolve()) diff --git a/experiments/eq3_system_thermal/prepare_maintenance_main.py b/experiments/eq3_system_thermal/prepare_maintenance_main.py new file mode 100644 index 0000000..d5f9e2e --- /dev/null +++ b/experiments/eq3_system_thermal/prepare_maintenance_main.py @@ -0,0 +1,270 @@ +#!/usr/bin/env python3 +"""Prepare the reviewed 19-point maintenance candidate; never launch or freeze it.""" +from __future__ import annotations + +import argparse +import copy +import hashlib +import json +from pathlib import Path + + +TOPOLOGIES = ("mixed_direct", "relay", "dash", "all_hbf_direct") +STRATEGIES = ("guard_only", "thermal_hysteresis_guard", + "read_rate_feedback_thermal_guard_v1") +REFERENCE_RATE_BPS = 1_536_000_000_000 +ACTIVE_NS = 20_000_000_000 +RECOVERY_NS = 10_000_000_000 +INVARIANT_MAINTENANCE_KEYS = ( + "refresh_trigger", "initial_equivalent_age_ns", "initial_wall_age_ns", + "initial_temperature_k", "block_bytes", "pages_per_block", + "aged_blocks_per_stack", "spares_per_channel", "max_blocks_per_cohort", + "program_energy_j_per_byte", "erase_energy_j_per_operation", "energy_evidence", +) + + +def digest(path: Path) -> str: + return hashlib.sha256(Path(path).read_bytes()).hexdigest() + + +def load(path: Path) -> dict: + value = json.loads(Path(path).read_text(encoding="utf-8")) + if not isinstance(value, dict): + raise ValueError(f"{path} must contain a JSON object") + return value + + +def save(path: Path, value: dict) -> None: + Path(path).write_text(json.dumps(value, indent=2, sort_keys=True, + allow_nan=False) + "\n", encoding="utf-8") + + +def _pilot_gate(pilot_index_path: Path, analysis_root: Path, + review_path: Path) -> tuple[dict, dict, dict[str, dict]]: + index = load(pilot_index_path) + points = index.get("points", []) + if len(points) != 4 or {row.get("topology") for row in points} != set(TOPOLOGIES): + raise ValueError("pilot index must contain exactly one point for every topology") + analyses = {} + analysis_hashes = {} + for row in points: + point_id = row["point_id"] + output = Path(row["output"]) + done_path = output / "DONE.json" + if not done_path.is_file() or load(done_path).get("status") != "COMPLETED": + raise ValueError(f"pilot {point_id} lacks a COMPLETED DONE receipt") + directory = analysis_root / point_id + analysis_path, analysis_done = directory / "analysis.json", directory / "DONE.json" + if not analysis_path.is_file() or not analysis_done.is_file() \ + or load(analysis_done).get("status") != "COMPLETED": + raise ValueError(f"pilot {point_id} lacks completed strict analysis") + analysis = load(analysis_path) + if analysis.get("analysis_status") != "VALIDATED_COMPLETE_RECEIPTS" \ + or analysis.get("runner") != "maintenance" \ + or analysis.get("identity", {}).get("point_id") != point_id: + raise ValueError(f"pilot {point_id} strict analysis identity/status failed") + checks = analysis.get("checks", {}) + if any(checks.get(key) != "PASS" for key in + ("timeline", "byte_conservation", "energy_to_thermal")): + raise ValueError(f"pilot {point_id} strict receipt checks failed") + if checks.get("terminal_uniqueness") != "EXACT_JOB_IDS": + raise ValueError(f"pilot {point_id} lacks exact completion identity") + if analysis.get("maintenance", {}).get("terminal_operation_count", 0) <= 0: + raise ValueError(f"pilot {point_id} did not exercise terminal maintenance work") + analyses[point_id] = analysis + analysis_hashes[point_id] = digest(analysis_path) + review = load(review_path) + if review.get("status") != "APPROVED_FOR_MAIN_INPUT_GENERATION": + raise ValueError("pilot review has not approved main input generation") + if review.get("pilot_index_sha256") != digest(pilot_index_path) \ + or review.get("analysis_sha256") != analysis_hashes: + raise ValueError("pilot review is not bound to these index/analysis receipts") + resources = review.get("measured_resource_budget") + required = ("point_wall_s", "stage_wall_s", "point_output_gib", + "stage_output_gib", "address_space_gib", "cpu_threads", + "host_ram_reserve_gib", "host_disk_reserve_gib") + if not isinstance(resources, dict) or any( + isinstance(resources.get(key), bool) or not isinstance(resources.get(key), (int, float)) + or resources[key] <= 0 for key in required): + raise ValueError("pilot review lacks a positive measured resource budget") + return index, review, analyses + + +def _pilot_templates(index: dict) -> tuple[dict[str, dict], dict[str, dict]]: + templates, rows = {}, {} + for row in index["points"]: + topology = row["topology"] + config_path = Path(row["config"]) + if digest(config_path) != row.get("config_sha256"): + raise ValueError(f"pilot input identity mismatch for {topology}") + config = load(config_path) + if config.get("topology") != topology or config.get("maintenance", {}).get("mode") != "shared": + raise ValueError("pilot template topology/mode mismatch") + if config.get("workload", {}).get("per_stack_Bps") != REFERENCE_RATE_BPS: + raise ValueError("pilot template is not the 1.536 TB/s maintenance input") + if config.get("maintenance", {}).get("refresh_trigger") != "equivalent_age_or_wall": + raise ValueError("pilot template does not consume equivalent age") + templates[topology], rows[topology] = config, row + reference = templates[TOPOLOGIES[0]] + for topology, config in templates.items(): + if config.get("energy") != reference.get("energy"): + raise ValueError(f"pilot energy profile differs for {topology}") + for key in INVARIANT_MAINTENANCE_KEYS: + if config["maintenance"].get(key) != reference["maintenance"].get(key): + raise ValueError(f"pilot maintenance invariant {key} differs for {topology}") + if config.get("maintenance_scope") != reference.get("maintenance_scope"): + raise ValueError(f"pilot maintenance scope differs for {topology}") + return templates, rows + + +def _specs() -> list[dict]: + result = [{"topology": topology, "strategy": strategy, "mode": "shared", "ea_ev": 1.04, + "axis": "shared_strategy_matrix"} + for topology in TOPOLOGIES for strategy in STRATEGIES] + result += [{"topology": topology, "strategy": STRATEGIES[2], + "mode": "ideal_independent", "ea_ev": 1.04, + "axis": "matched_resource_contention"} + for topology in ("mixed_direct", "relay", "dash")] + result += [{"topology": "mixed_direct", "strategy": strategy, + "mode": "shared", "ea_ev": ea, "axis": "arrhenius_ea_sensitivity"} + for ea in (1.01, 1.08) for strategy in (STRATEGIES[0], STRATEGIES[2])] + return result + + +def _point_id(spec: dict) -> str: + return (f"maint-main-v1-{spec['topology']}-p{STRATEGIES.index(spec['strategy'])}-" + f"{spec['mode']}-ea{round(spec['ea_ev'] * 100):03d}-01") + + +def prepare(pilot_index_path: Path, analysis_root: Path, review_path: Path, + destination: Path) -> dict: + pilot_index_path, analysis_root, review_path, destination = map( + Path, (pilot_index_path, analysis_root, review_path, destination)) + if destination.name != "maintenance-main-v1": + raise ValueError("destination must be a distinct maintenance-main-v1 stage") + if destination.exists(): + raise FileExistsError(destination) + index, review, analyses = _pilot_gate(pilot_index_path, analysis_root, review_path) + templates, pilot_rows = _pilot_templates(index) + destination.mkdir(parents=True) + inputs = destination / "inputs" + inputs.mkdir() + points = [] + for spec in _specs(): + point_id = _point_id(spec) + config = copy.deepcopy(templates[spec["topology"]]) + config["point_id"] = point_id + config["strategy"] = spec["strategy"] + config["workload"]["active_ns"] = ACTIVE_NS + config["recovery_ns"] = RECOVERY_NS + config["maintenance"]["mode"] = spec["mode"] + config["maintenance"]["ea_ev"] = spec["ea_ev"] + config["maintenance_scope"]["matrix_axis"] = spec["axis"] + path = inputs / f"{point_id}.json" + save(path, config) + template = pilot_rows[spec["topology"]] + points.append({ + "point_id": point_id, **spec, "config": str(path.resolve()), + "config_sha256": digest(path), "model_dir": template["model_dir"], + "thermal_binary": template["thermal_binary"], + "artifact_root": template["artifact_root"], + "output": str((destination / "points" / point_id).resolve()), + }) + if len(points) != 19 or len({row["point_id"] for row in points}) != 19: + raise AssertionError("maintenance matrix must contain 19 unique points") + pilot_receipts = {point_id: { + "analysis_sha256": review["analysis_sha256"][point_id], + "analysis_status": analysis["analysis_status"], + "maintenance_terminal_count": analysis["maintenance"]["terminal_operation_count"], + } for point_id, analysis in analyses.items()} + candidate = { + "schema_version": "eq3-maintenance-main-candidate-v1", + "status": "PILOT_REVIEW_PASSED_PREPARED_NOT_FROZEN_NOT_LAUNCHABLE", + "point_count": 19, "points": points, + "pilot_gate": {"pilot_index": str(pilot_index_path.resolve()), + "pilot_index_sha256": digest(pilot_index_path), + "review": str(review_path.resolve()), "review_sha256": digest(review_path), + "receipts": pilot_receipts}, + "runtime_locks_to_rebind_at_freeze": { + "runner": index.get("runner"), "runtime_source_locks_sha256": index.get( + "runtime_source_locks_sha256"), "model_locks_sha256": index.get( + "model_locks_sha256"), "thermal_binary_sha256": index.get( + "thermal_binary_sha256")}, + "resources_candidate_from_pilot_review": review["measured_resource_budget"], + "gpu_count": 0, + "execution_gate": "REQUIRES_SEPARATE_FREEZE_AND_ROOT_SCHEDULING;THIS_INDEX_IS_NOT_RUNNABLE", + } + save(destination / "CANDIDATE_INDEX.json", candidate) + preflight = { + "schema_version": "eq3-maintenance-main-preflight-candidate-v1", + "status": "CANDIDATE_NOT_FROZEN_NOT_LAUNCHED", + "research_question": ("How topology and existing policy affect foreground/refresh " + "contention, and how Ea changes the consumed equivalent-age trigger."), + "evidence_class": "CONDITIONAL_SIMULATED_AGGREGATE_RATE_SERVICE", + "authority": "USER_EXPLICIT_FOUR_TOPOLOGY_PLAN", + "code_and_environment_candidate": { + "environment_id": "eq3-thermal-cpu-v1", + "runtime_locks_from_pilot": candidate["runtime_locks_to_rebind_at_freeze"], + "freeze_requirement": "REHASH_CURRENT_SOURCES_MODELS_BINARY_AND_INPUTS_BEFORE_RUN"}, + "matrix": {"shared_strategy_points": 12, "matched_ideal_points": 3, + "ea_endpoint_points": 4, "total": 19, + "active_ns": ACTIVE_NS, "recovery_ns": RECOVERY_NS, + "per_stack_Bps": REFERENCE_RATE_BPS, "replicates": 1, + "determinism": "FIXED_INPUT_SINGLE_DETERMINISTIC_RUN_PER_POINT"}, + "fixed_inputs": {"maintenance_age_and_subset": { + key: templates[TOPOLOGIES[0]]["maintenance"][key] + for key in INVARIANT_MAINTENANCE_KEYS}, + "maintenance_scope": templates[TOPOLOGIES[0]]["maintenance_scope"], + "energy": templates[TOPOLOGIES[0]]["energy"]}, + "controls": ["same topology/workload/age/energy within policy comparisons", + "shared versus ideal changes resource contention only", + "Ea endpoints consume equivalent-age policy; Ea1.04 reused from shared matrix"], + "metrics": ["foreground bytes/backlog/delay", "maintenance commit/failure/spares/age/wear", + "component energy and owner temperatures", "controller states and due bytes"], + "mechanism": { + "refresh_chain": "read -> program -> version commit -> old block erase", + "ea_consumer": "ReliabilityLedger equivalent_age_or_wall trigger", + "success_age_rule": "reset exact extents only after successful commit"}, + "pilot_gate": candidate["pilot_gate"], + "resources": review["measured_resource_budget"], + "outputs": {"stage": str(destination.resolve()), "raw": "points//", + "analysis": "new derived directory; never overwrite raw"}, + "acceptance": ["all raw terminal receipts retained", "strict identity/timeline/byte/energy checks", + "unique completion identities", "paired comparisons retain fixed inputs"], + "safety_stop": ["stop on identity or dependency mismatch", + "preserve and classify domain failures", + "stop and diagnose non-domain failure before remaining points", + "CPU-only and no parallel thermal jobs"], + "failure_contract": "PRESERVE_FAILED_RUNS;NO_AUTOMATIC_SCIENTIFIC_PASS", + "limits": ["not native MQSim NAND timing", "metadata validity only; no payload integrity", + "program/erase energy and speed are engineering proxies", + "P2 remains unqualified; external GDDR temperature unavailable"], + } + save(destination / "PREFLIGHT_CANDIDATE.json", preflight) + (destination / "PREFLIGHT_CANDIDATE.md").write_text( + "# Maintenance main candidate\n\n" + "This directory contains 19 generated inputs after all four maintenance pilots and their " + "strict receipt analyses were reviewed. It is **not frozen or runnable**. Root must bind " + "current runtime/model/binary hashes into a separate run index before scheduling.\n\n" + "The matrix contains 12 shared-resource topology/policy points, three policy-matched " + "ideal-independent contention points for mixed/relay/DASH, and four mixed-topology " + "Ea endpoint points. Ea=1.04 comparisons reuse the shared matrix. All points keep the " + "same 1.536 TB/s offered rate, 20 s active plus 10 s recovery, near-due age, 4 GiB/stack " + "aged subset, channel-owned spares, and energy/timing proxies.\n\n" + "Interpretation remains conditional aggregate-service evidence. No scientific PASS is " + "created by input generation.\n", encoding="utf-8") + return candidate + + +def main() -> None: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--pilot-index", type=Path, required=True) + parser.add_argument("--analysis-root", type=Path, required=True) + parser.add_argument("--pilot-review", type=Path, required=True) + parser.add_argument("--destination", type=Path, required=True) + args = parser.parse_args() + prepare(args.pilot_index, args.analysis_root, args.pilot_review, args.destination) + + +if __name__ == "__main__": + main() diff --git a/experiments/eq3_system_thermal/prepare_maintenance_rate_repair.py b/experiments/eq3_system_thermal/prepare_maintenance_rate_repair.py new file mode 100644 index 0000000..f9f9a65 --- /dev/null +++ b/experiments/eq3_system_thermal/prepare_maintenance_rate_repair.py @@ -0,0 +1,46 @@ +#!/usr/bin/env python3 +"""Correct the observed384GB/s pilot input to its registered1536GB/s value.""" +import argparse +from copy import deepcopy +import json +from pathlib import Path +from freeze_extension import freeze,save + +HERE=Path(__file__).resolve().parent + +def prepare(stage,destination): + old=json.loads((stage/'maintenance-v1/PILOT_INDEX.json').read_text()) + destination.mkdir(parents=True,exist_ok=False);(destination/'inputs').mkdir() + configs=[];changes=[] + for row in old['points']: + original=json.loads(Path(row['config']).read_text());fixed=deepcopy(original) + if original['workload']['per_stack_Bps']!=384_000_000_000:raise ValueError('unexpected original reproducer') + fixed['workload']['per_stack_Bps']=1_536_000_000_000 + fixed['point_id']=f"maint-rate-repair-{row['topology']}-01" + assert fixed['maintenance']==original['maintenance'] + path=destination/'inputs'/(fixed['point_id']+'.json');save(path,fixed);configs.append(path) + changes.append({'old_point':row['point_id'],'new_point':fixed['point_id'], + 'old_actual_per_stack_Bps':384_000_000_000,'registered_and_corrected_per_stack_Bps':1_536_000_000_000, + 'unchanged':'age,wear,physicalparameters,energy,guard,workloadtiming,topology,maintenancepolicy'}) + index=freeze(stage,destination,configs,HERE/'run_maintenance_point.py',point_wall_s=600, + dependency=stage/'ecc-pilot-v1/DONE.json') + index['resources'].update(stage_wall_s=3600,sensitivity_output_gib=8) + save(destination/'PILOT_INDEX.json',index) + preflight={'status':'FROZEN_WAITING_ECC_SERIAL_COMPLETION','classification':'CONFIRMED_INPUT_PREPARATION_BUG', + 'source_preflight':str(stage/'maintenance-v1/PREFLIGHT.md'), + 'source_failure_evidence':str(stage/'maintenance-v1/PILOT_AUDIT.json'), + 'root_cause':'input snapshot384GB/s contradicts registered1536GB/s; lowactualinputkeptagedextents belowdue', + 'fix':'correct workloadrate only; noage orphysics tuning', + 'forecast':'existing mixed1.536TB/s no-maintenance raw first8s yields3.868..6.235 equivalent seconds at85C; above original2s remaining margin', + 'forecast_limit':'other topology and maintenance interaction need actual pilot; no guaranteed benefit', + 'environment_id':'eq3-thermal-cpu-v1','changes':changes,'points':4,'duration':'8sinput+4sdrain', + 'resources':index['resources'],'approval_scope':'existing explicit1536GB/s fourtopology maintenancepilot scope', + 'pass':['conserved bytes/energy','nonzero actualdue/terminalcommit/program/erase','no falseage reset','validsource/destinationversions','sharedresourcecontentionreceipts'], + 'stop':['400K domainfailure retainsraw','600sperpoint3600sstage','source/input mismatch','hostresource reserves'], + 'limits':'NO_ECC_REFRESH_BENEFIT_CLAIM;ECC_READ_AGE_NOT_YET_MAPPED_TO_THESE_EXTENTS'} + save(destination/'PREFLIGHT.json',preflight) + (destination/'PREFLIGHT.md').write_text('# Maintenance rate preparation repair\n\n'+json.dumps(preflight,indent=2)+'\n') + +if __name__=='__main__': + ap=argparse.ArgumentParser();ap.add_argument('--stage',type=Path,required=True);ap.add_argument('--destination',type=Path,required=True) + a=ap.parse_args();prepare(a.stage.resolve(),a.destination.resolve()) diff --git a/experiments/eq3_system_thermal/prepare_rate_diagnostics.py b/experiments/eq3_system_thermal/prepare_rate_diagnostics.py new file mode 100644 index 0000000..41a6dd8 --- /dev/null +++ b/experiments/eq3_system_thermal/prepare_rate_diagnostics.py @@ -0,0 +1,217 @@ +#!/usr/bin/env python3 +"""Prepare nine bounded nonuniform/coupling diagnostics without launching them.""" + +from __future__ import annotations + +import argparse +import hashlib +import json +from pathlib import Path + +from prepare_stage import config as base_config + + +HERE = Path(__file__).resolve().parent +ROOT = HERE.parents[1] +RUNNER = HERE / "run_endpoint_guard_point.py" +TOPOLOGIES = ("mixed_direct", "relay", "dash", "all_hbf_direct") +STRATEGIES = ("guard_only", "thermal_hysteresis_guard", + "read_rate_feedback_thermal_guard_v1") +RATE = 1_536_000_000_000 + + +def digest(path: Path) -> str: + return hashlib.sha256(path.read_bytes()).hexdigest() + + +def save(path: Path, value: dict) -> None: + path.write_text(json.dumps(value, indent=2, sort_keys=True, allow_nan=False) + "\n") + + +def required_files(directory: Path) -> dict[str, str]: + names = ("model.txt", "normalized.json", "rc_grid.json", "rc_sensors.json") + if not all((directory / name).is_file() for name in names): + raise FileNotFoundError(directory) + return {name: digest(directory / name) for name in names} + + +def _models(manifest_path: Path) -> dict[str, str]: + data = json.loads(manifest_path.read_text()) + result = { + "mixed_full": data["source_models"]["mixed_full_2mm"]["model_dir"], + "all_hbf_full": data["source_models"]["all_hbf_full_2mm"]["model_dir"], + } + rows = [row for row in data["derived_models"] + if row["ambient_k"] == 300 and row["external_resistance_scale"] == 1 + and row["coupling_mode"] == "no_cross_domain_lateral"] + if len(rows) != 1: + raise ValueError("exactly one registered mixed no-cross-domain model is required") + result["mixed_no_cross_domain"] = rows[0]["model_dir"] + for path in result.values(): + required_files(Path(path)) + return result + + +def _base_points(index_path: Path) -> tuple[dict[tuple, dict], dict]: + index = json.loads(index_path.read_text()) + rows = {} + for entry in index["points"]: + if entry.get("kind") != "base": + continue + config_path = Path(entry["config"]) + if digest(config_path) != entry["config_sha256"]: + raise ValueError("base config hash differs from RUN_INDEX") + config = json.loads(config_path.read_text()) + key = (config["topology"], config["workload"]["per_stack_Bps"], config["strategy"]) + if key in rows: + raise ValueError(f"duplicate base identity {key}") + rows[key] = {"point_id": entry["point_id"], "config": str(config_path.resolve()), + "config_sha256": entry["config_sha256"], + "output": entry["output"]} + return rows, index + + +def prepare(output: Path, model_manifest: Path, base_index: Path, + thermal_binary: Path, artifact_root: Path) -> dict: + if output.exists(): + raise FileExistsError(output) + models = _models(model_manifest) + base, frozen_base_index = _base_points(base_index) + thermal_binary = thermal_binary.resolve(strict=True) + artifact_root = artifact_root.resolve(strict=True) + if not thermal_binary.is_relative_to(artifact_root): + raise ValueError("thermal binary must be within artifact_root") + inputs = output / "inputs"; inputs.mkdir(parents=True) + entries = [] + + def emit(point_id: str, topology: str, distribution: str, *, hot_stack=None, + model_key=None, diagnostic: str) -> None: + config = base_config(point_id, topology, RATE, + "read_rate_feedback_thermal_guard_v1", 20, 10) + config["workload"]["channel_distribution"] = distribution + if hot_stack is not None: + config["workload"]["hot_stack"] = hot_stack + config["rate_diagnostic"] = { + "schema_version": "eq3-system-rate-diagnostic-point-v1", + "diagnostic": diagnostic, + "offered_demand_semantics": "MODELLED_BYTES_NOT_LLM_TOKEN_OR_NATIVE_MQSIM_THROUGHPUT", + "total_offered_preserved_vs_uniform": True, + } + path = inputs / f"{point_id}.json"; save(path, config) + if model_key is None: + model_key = "all_hbf_full" if topology == "all_hbf_direct" else "mixed_full" + entries.append({"point_id": point_id, "kind": "rate_diagnostic", + "topology": topology, "diagnostic": diagnostic, + "config": str(path.resolve()), "config_sha256": digest(path), + "model_dir": models[model_key], "thermal_binary": str(thermal_binary), + "artifact_root": str(artifact_root), + "output": str((output / "points" / point_id).resolve())}) + + for topology in TOPOLOGIES: + emit(f"diag-{topology}-first-half-1536-feedback-01", topology, "first_half", + diagnostic="FIRST_HALF_CHANNELS_SAME_TOTAL_OFFERED") + emit(f"diag-{topology}-hot-hbf0-1536-feedback-01", topology, "uniform", + hot_stack="hbf0", diagnostic="HOT_STACK_HBF0_SAME_TOTAL_OFFERED") + emit("diag-mixed-no-cross-domain-1536-feedback-01", "mixed_direct", "uniform", + model_key="mixed_no_cross_domain", + diagnostic="NO_CROSS_DOMAIN_LATERAL_THERMAL_COUPLING_SAME_WORKLOAD") + if len(entries) != 9: + raise AssertionError("rate diagnostic design must contain nine points") + + def baseline(topology: str, rate: int, strategy: str) -> dict: + key = (topology, rate, strategy) + if key not in base: + raise ValueError(f"base comparison missing {key}") + return base[key] + + uniform_feedback = { + topology: baseline(topology, RATE, "read_rate_feedback_thermal_guard_v1") + for topology in TOPOLOGIES + } + comparisons = { + "nonuniform_vs_uniform": [ + {"diagnostic_point_id": row["point_id"], + "uniform_base": uniform_feedback[row["topology"]], + "identity": "SAME_TOPOLOGY_RATE_POLICY_DURATION_TOTAL_OFFERED"} + for row in entries if row["diagnostic"].startswith(("FIRST_HALF", "HOT_STACK")) + ], + "no_coupling_vs_full": [{ + "diagnostic_point_id": "diag-mixed-no-cross-domain-1536-feedback-01", + "full_coupling_base": uniform_feedback["mixed_direct"], + "identity": "SAME_CONFIG_AND_WORKLOAD_ONLY_THERMAL_MODEL_DIFFERS", + }], + "same_total_offered_existing_base": [ + {"strategy": strategy, + "mixed_4x1536": baseline("mixed_direct", RATE, strategy), + "all_hbf_8x768": baseline("all_hbf_direct", 768_000_000_000, strategy), + "identity": "6.144_TBPS_TOTAL_MODELLED_OFFERED_BOTH"} + for strategy in STRATEGIES + ], + "same_per_stack_existing_base": [ + {"strategy": strategy, + "mixed_4x1536": baseline("mixed_direct", RATE, strategy), + "all_hbf_8x1536": baseline("all_hbf_direct", RATE, strategy), + "identity": "1.536_TBPS_PER_STACK_DIFFERENT_STACK_COUNT_AND_TOTAL_OFFERED"} + for strategy in STRATEGIES + ], + } + + unique_models = sorted({row["model_dir"] for row in entries}) + runtime_sources = [RUNNER, HERE / "endpoint_policy.py", HERE / "run_system_point.py", HERE / "energy.py", + HERE / "rate_workload.py", HERE / "topology_service.py", + ROOT / "experiments/eq3_maintenance/thermal_client.py", + ROOT / "experiments/eq3_maintenance/read_rate_policy.py"] + index = { + "schema_version": "eq3-system-rate-diagnostics-index-v2", + "status": "PENDING_DEPENDENCIES_BASE_MATRIX", + "authorization": "USER_EXPLICIT_BOUNDED_NONUNIFORM_AND_COUPLING_DIAGNOSTICS", + "point_count": 9, "points": entries, "comparisons": comparisons, + "base_index": str(base_index.resolve()), "base_index_sha256": digest(base_index), + "base_source_locks": frozen_base_index.get("source_locks", {}), + "model_manifest": str(model_manifest.resolve()), + "model_manifest_sha256": digest(model_manifest), + "model_locks_sha256": {path: required_files(Path(path)) for path in unique_models}, + "runner": str(RUNNER.resolve()), + "runner_sha256": digest(RUNNER), + "endpoint_policy_semantics": ( + "HBF_LEGACY_READ_RATE_POLICY; HBM_SHARED_ENDPOINT_THERMAL_GUARD_" + "RESTORES_BASELINE_ON_NORMAL_WITHOUT_FOREGROUND_DEMAND_INFERENCE"), + "runtime_source_locks_sha256": { + str(path.resolve().relative_to(ROOT.resolve())): digest(path) for path in runtime_sources}, + "thermal_binary_sha256": digest(thermal_binary), + "dependencies": {"base_done": str((output.parent / "BASE_DONE.json").resolve()), + "required_status": "COMPLETED"}, + "resources": {"point_wall_s": 600, "point_output_gib": 1, + "address_space_gib": 8, "cpu_experiments": 1, "gpu": 0, + "new_output_gib": 4.5, "parent_combined_output_gib": 80, + "parent_accounting_gib": {"base_estimate": 30, + "sensitivity_allocation": 27, + "rate_diagnostics_allocation": 4.5, + "remaining_for_maintenance_causal": 18.5}, + "stage_wall_s": 21600, "host_ram_reserve_gib": 32, + "host_disk_reserve_gib": 100}, + "limitations": ["MODELLED_AGGREGATED_BYTE_PRESSURE_NOT_LLM_THROUGHPUT", + "NO_NEW_BASELINE_RUNS", "NO_TOKEN_PER_S_INFERENCE", + "CROSS_TOPOLOGY_GEOMETRY_AND_STACK_COUNT_REMAIN_CONFOUNDERS"], + } + save(output / "RATE_DIAGNOSTICS_INDEX.json", index) + return index + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--output", type=Path, required=True) + parser.add_argument("--model-manifest", type=Path, required=True) + parser.add_argument("--base-index", type=Path, required=True) + parser.add_argument("--thermal-binary", type=Path, required=True) + parser.add_argument("--artifact-root", type=Path, required=True) + args = parser.parse_args() + index = prepare(args.output.resolve(), args.model_manifest.resolve(strict=True), + args.base_index.resolve(strict=True), args.thermal_binary, + args.artifact_root) + print(json.dumps({"status": index["status"], "point_count": index["point_count"]})) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/experiments/eq3_system_thermal/prepare_sensitivity.py b/experiments/eq3_system_thermal/prepare_sensitivity.py new file mode 100644 index 0000000..432e8b5 --- /dev/null +++ b/experiments/eq3_system_thermal/prepare_sensitivity.py @@ -0,0 +1,276 @@ +#!/usr/bin/env python3 +"""Prepare, but never launch, the authorized bounded sensitivity design.""" + +from __future__ import annotations + +import argparse +import copy +import hashlib +import json +from pathlib import Path + +from prepare_stage import config as base_config + +HERE = Path(__file__).resolve().parent +ROOT = HERE.parents[1] +RUNNER = HERE / "run_endpoint_guard_point.py" + + +STRATEGIES = ("guard_only", "read_rate_feedback_thermal_guard_v1") +RATE_BPS = 1_536_000_000_000 +ACTIVE_S = 20 +RECOVERY_S = 10 +BASE_LIMITS_K = { + "hbf": [353.15, 363.15, 378.15], + "hbm": [353.15, 363.15, 378.15], + "gpu": [363.15, 373.15, 383.15], +} + + +def _save(path: Path, value: dict) -> None: + path.write_text(json.dumps(value, indent=2, sort_keys=True, allow_nan=False) + "\n") + + +def _digest(path: Path) -> str: + return hashlib.sha256(path.read_bytes()).hexdigest() + + +def _model_map(model_manifest: Path) -> dict[tuple[int, float], str]: + data = json.loads(model_manifest.read_text()) + mixed = data["source_models"]["mixed_full_2mm"]["model_dir"] + result = {(300, 1.0): mixed} + for row in data["derived_models"]: + if row["coupling_mode"] == "full": + result[(int(row["ambient_k"]), float(row["external_resistance_scale"]))] = row["model_dir"] + required = {(300, .5), (300, 1.0), (300, 1.5), (310, 1.0), (320, 1.0), + (320, 1.5)} + if set(result) != required: + raise ValueError(f"thermal model variants differ from frozen sensitivity set: {set(result)}") + for path in result.values(): + directory = Path(path) + if not all((directory / name).is_file() for name in + ("model.txt", "normalized.json", "rc_grid.json", "rc_sensors.json")): + raise FileNotFoundError(directory) + return result + + +def _levels() -> dict: + return { + "ambient_k": [300, 310, 320], + "external_resistance_scale": [.5, 1.0, 1.5], + "hbf_read_energy_scale": [.5, 1.0, 1.5], + "gpu_external_w": [0, 100, 200], + # Shutdown remains the registered 105 C hard envelope. + "hbf_guard_light_severe_offset_k": [-5, 0, 5], + "hbm_guard_light_severe_offset_k": [-5, 0, 5], + } + + +def physical_combinations() -> list[dict]: + nominal = { + "ambient_k": 300, + "external_resistance_scale": 1.0, + "hbf_read_energy_scale": 1.0, + "gpu_external_w": 0, + "hbf_guard_light_severe_offset_k": 0, + "hbm_guard_light_severe_offset_k": 0, + } + rows = [{"id": "baseline", "values": dict(nominal), "kind": "baseline"}] + for axis, values in _levels().items(): + center = nominal[axis] + for value in values: + if value == center: + continue + row = dict(nominal); row[axis] = value + value_label = str(value).replace("-", "m").replace(".", "p") + rows.append({"id": f"oat-{axis}-{value_label}", + "values": row, "kind": "oat", "axis": axis}) + interactions = ( + ("interaction-hot-ambient-poor-boundary", + {"ambient_k": 320, "external_resistance_scale": 1.5}), + ("interaction-high-memory-energy-high-gpu", + {"hbf_read_energy_scale": 1.5, "gpu_external_w": 200}), + ("interaction-conservative-hbf-hbm-guards", + {"hbf_guard_light_severe_offset_k": -5, + "hbm_guard_light_severe_offset_k": -5}), + ) + for identity, changes in interactions: + row = dict(nominal); row.update(changes) + rows.append({"id": identity, "values": row, "kind": "interaction", + "axes": sorted(changes)}) + if len(rows) != 16 or len({json.dumps(row["values"], sort_keys=True) for row in rows}) != 16: + raise AssertionError("sensitivity physical combinations are not the frozen 13 OAT + 3 interactions") + return rows + + +def _apply(config: dict, values: dict) -> None: + scale = values["hbf_read_energy_scale"] + config["energy"]["read_array_j_per_byte"] = 40e-12 * scale + config["energy"]["read_base_j_per_byte"] = 10e-12 * scale + config["gpu_external_w"] = values["gpu_external_w"] + config["thermal_limits_k"] = copy.deepcopy(BASE_LIMITS_K) + for device in ("hbf", "hbm"): + offset = values[f"{device}_guard_light_severe_offset_k"] + config["thermal_limits_k"][device][0] += offset + config["thermal_limits_k"][device][1] += offset + if config["thermal_limits_k"][device][2] != 378.15: + raise AssertionError("shutdown envelope changed") + + +def _point(point_id: str, topology: str, strategy: str, values: dict, + execution_mode: str, model_dir: str, mechanism_scope: str) -> tuple[dict, dict]: + config = base_config(point_id, topology, RATE_BPS, strategy, ACTIVE_S, RECOVERY_S) + _apply(config, values) + if execution_mode == "uncontrolled_first_constraint": + config["control_disabled"] = True + config["sensitivity"] = { + "schema_version": "eq3-system-sensitivity-point-v1", + "execution_mode": execution_mode, + "mechanism_scope": mechanism_scope, + "values": values, + "energy_evidence": "SCENARIO_ASSUMPTION_SCALE_OF_USER_CONFIRMED_40_ARRAY_10_BASE_PJ_PER_B", + "guard_evidence": "RESEARCH_POLICY_SENSITIVITY_NOT_PRODUCT_LIMIT", + "shutdown_envelope_k": {"hbf": 378.15, "hbm": 378.15, "gpu": 383.15}, + "domain_max_k_unchanged": 400, + } + return config, {"point_id": point_id, "kind": "sensitivity", "topology": topology, + "strategy": strategy, "execution_mode": execution_mode, + "mechanism_scope": mechanism_scope, "model_dir": model_dir, + "sensitivity_values": values} + + +def prepare(output: Path, model_manifest: Path, thermal_binary: Path, + artifact_root: Path) -> dict: + if output.exists(): + raise FileExistsError(output) + models = _model_map(model_manifest) + thermal_binary = thermal_binary.resolve(strict=True) + artifact_root = artifact_root.resolve(strict=True) + if not thermal_binary.is_relative_to(artifact_root): + raise ValueError("thermal binary must be within artifact_root") + inputs = output / "inputs" + inputs.mkdir(parents=True) + entries = [] + + def emit(spec: dict, topology: str, strategy: str, execution_mode: str, + mechanism_scope: str) -> None: + values = spec["values"] + point_id = (f"sens-{topology}-{spec['id']}-" + f"{'u' if execution_mode.startswith('uncontrolled') else STRATEGIES.index(strategy)}-01") + model_dir = models[(values["ambient_k"], values["external_resistance_scale"])] + config, entry = _point(point_id, topology, strategy, values, execution_mode, + model_dir, mechanism_scope) + path = inputs / f"{point_id}.json" + _save(path, config) + entry.update({"config": str(path.resolve()), "config_sha256": _digest(path), + "thermal_binary": str(thermal_binary), "artifact_root": str(artifact_root), + "output": str((output / "points" / point_id).resolve())}) + entries.append(entry) + + combinations = physical_combinations() + for spec in combinations: + scope = ("MIXED_DIRECT_HBM_LIMIT_OBSERVATIONAL_ONLY_NO_HBM_DEMAND_OR_RELAY_FEEDBACK" + if spec.get("axis") == "hbm_guard_light_severe_offset_k" else + "MIXED_DIRECT_CONTROLLED_OAT_OR_INTERACTION") + for strategy in STRATEGIES: + emit(spec, "mixed_direct", strategy, "controlled", scope) + emit(spec, "mixed_direct", "guard_only", "uncontrolled_first_constraint", scope) + + # Two HBM threshold endpoints are repeated on relay because only that route + # makes the paired HBM endpoint state causal for HBF delivery. + for spec in combinations: + if spec.get("axis") != "hbm_guard_light_severe_offset_k": + continue + for strategy in STRATEGIES: + emit(spec, "relay", strategy, "controlled", + "RELAY_HBM_ENDPOINT_ACTUAL_CONSUMER_EXTENSION") + emit(spec, "relay", "guard_only", "uncontrolled_first_constraint", + "RELAY_HBM_ENDPOINT_ACTUAL_CONSUMER_EXTENSION") + + if len(entries) != 54 or len({row["point_id"] for row in entries}) != 54: + raise AssertionError("active sensitivity design must contain exactly 54 unique points") + deferred = { + "schema_version": "eq3-system-sensitivity-deferred-ea-v1", + "status": "DEFERRED_CONSUMER_NOT_READY", + "point_count_when_ready": 6, + "ea_ev": [1.01, 1.04, 1.08], + "strategies": list(STRATEGIES), + "required_consumer": "ReliabilityLedger equivalent_age_or_wall plus actual maintenance_driver completion", + "initial_age_contract": "near-equivalent-age DAY minus approximately 2 s, exact value frozen with consumer", + "fairness_contract": "same 4 GiB per-stack aged subset; separate fresh-null comparator; no reuse of fresh rate baseline", + "claim_limit": "CONDITIONAL_EQUIVALENT_AGE_ONLY_NO_RBER_ECC_OR_LIFETIME", + "runnable": False, + } + _save(output / "DEFERRED_EA.json", deferred) + unique_models = sorted({row["model_dir"] for row in entries}) + model_locks = { + directory: {name: _digest(Path(directory) / name) for name in + ("model.txt", "normalized.json", "rc_grid.json", "rc_sensors.json")} + for directory in unique_models + } + runtime_sources = [ + RUNNER, HERE / "endpoint_policy.py", HERE / "run_system_point.py", + HERE / "energy.py", HERE / "rate_workload.py", + HERE / "topology_service.py", + ROOT / "experiments" / "eq3_maintenance" / "thermal_client.py", + ROOT / "experiments" / "eq3_maintenance" / "read_rate_policy.py", + ] + index = { + "schema_version": "eq3-system-sensitivity-index-v2", + "status": "PENDING_DEPENDENCIES_BASE_MATRIX", + "authorization": "USER_EXPLICIT_SEVEN_AXIS_BOUNDED_SENSITIVITY", + "point_count": 54, + "points": sorted(entries, key=lambda row: row["point_id"]), + "model_manifest": str(model_manifest.resolve()), + "model_manifest_sha256": _digest(model_manifest), + "model_locks_sha256": model_locks, + "runner": str(RUNNER.resolve()), + "runner_sha256": _digest(RUNNER), + "endpoint_policy_semantics": ( + "HBF_LEGACY_READ_RATE_POLICY; HBM_SHARED_ENDPOINT_THERMAL_GUARD_" + "RESTORES_BASELINE_ON_NORMAL_WITHOUT_FOREGROUND_DEMAND_INFERENCE"), + "runtime_source_locks_sha256": { + str(path.resolve().relative_to(ROOT.resolve())): _digest(path) + for path in runtime_sources + }, + "thermal_binary_sha256": _digest(thermal_binary), + "dependencies": { + "base_done": str((output.parent / "BASE_DONE.json").resolve()), + "required_status": "COMPLETED", + }, + "design": { + "representative": "Q1 mixed_direct W1 continuous 1.536 TB/s per HBF stack; relay only for HBM endpoint consumer extension", + "controlled_strategies": list(STRATEGIES), + "active_s": ACTIVE_S, "recovery_s": RECOVERY_S, + "oat_physical_combinations": 13, "interaction_combinations": 3, + "controlled_points": 32, "uncontrolled_first_constraint_points": 16, + "relay_hbm_consumer_extension_points": 6, + "total_active_points": 54, "deferred_ea_points": 6, + }, + "failure_contract": "400K_DOMAIN_FAILURE_RETAINED; continue independent points; no clamp or threshold relaxation", + "resources": {"point_wall_s": 600, "point_output_gib": 1, + "address_space_gib": 8, "cpu_experiments": 1, "gpu": 0, + "stage_wall_s": 21600, "sensitivity_output_gib": 27, + "parent_combined_output_gib": 80, + "host_ram_reserve_gib": 32, "host_disk_reserve_gib": 100, + "basis": "four pilots max 98.34 s projected 30 s point and 496,762,063 B; base projection 29,805,723,750 B"}, + } + _save(output / "SENSITIVITY_INDEX.json", index) + return index + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--output", type=Path, required=True) + parser.add_argument("--model-manifest", type=Path, required=True) + parser.add_argument("--thermal-binary", type=Path, required=True) + parser.add_argument("--artifact-root", type=Path, required=True) + args = parser.parse_args() + index = prepare(args.output.resolve(), args.model_manifest.resolve(strict=True), + args.thermal_binary, args.artifact_root) + print(json.dumps({"status": index["status"], "point_count": index["point_count"]})) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/experiments/eq3_system_thermal/prepare_stage.py b/experiments/eq3_system_thermal/prepare_stage.py new file mode 100644 index 0000000..ec5bea7 --- /dev/null +++ b/experiments/eq3_system_thermal/prepare_stage.py @@ -0,0 +1,78 @@ +#!/usr/bin/env python3 +"""Freeze explicitly authorized four-topology base matrix and its pilots.""" +import argparse +import hashlib +import json +from pathlib import Path +from energy import engineering_energy_profile +from topology_service import default_config, media_cost_from_service_rate + +HERE = Path(__file__).resolve().parent +STRATEGIES = ['guard_only','thermal_hysteresis_guard','read_rate_feedback_thermal_guard_v1'] +TOPOLOGIES = ['mixed_direct','relay','dash','all_hbf_direct'] + + +def save(path, obj): + path.write_text(json.dumps(obj,indent=2,allow_nan=False)+'\n') + + +def config(point_id, topology, rate, strategy, active_s=20, recovery_s=10): + service = default_config(topology) + # Existing basic fabric engineering parameters, separate from OCP media cap. + for row in service['fabric']['hbf'].values(): + for name in ('fill','direct_link','relay_link'): + if row[name] is not None: + row[name]['latency_ns'] = 10 + for row in service['fabric']['hbm'].values(): + row['gpu_link'] = {'latency_ns':10,'bandwidth_bytes_per_s':2_048_000_000_000} + # Explicit plane-parallel engineering proxy, not claimed OCP program speed. + program_Bps_per_channel = 16 * 4096 * 10_000 + service['operation_media_cost']['program'] = media_cost_from_service_rate(96_000_000_000,program_Bps_per_channel) + service['operation_media_cost']['erase'] = media_cost_from_service_rate(96_000_000_000,16*4096*256*1000) + service['operation_cost_evidence'] = 'PROXY_16_PLANES_4KIB_100US_PROGRAM_1MS_ERASE_256PAGES_PER_BLOCK' + profile = engineering_energy_profile() + profile['erase_array_j_per_operation'] = .05 * .001 + profile['evidence']['erase'] = 'DERIVED_ENGINEERING_PROXY_0.05W_TIMES_NATIVE_FIXTURE_1MS_NOT_HBF_MEASUREMENT' + return {'schema_version':'eq3-system-point-v1','point_id':point_id,'topology':topology, + 'strategy':strategy,'service':service,'energy':profile, + 'workload':{'active_ns':int(active_s*1e9),'per_stack_Bps':rate, + 'pattern':'continuous','channel_distribution':'uniform'}, + 'recovery_ns':int(recovery_s*1e9),'gpu_external_w':0, + 'thermal_limits_k':{'hbf':[353.15,363.15,378.15], + 'hbm':[353.15,363.15,378.15], + 'gpu':[363.15,373.15,383.15]}, + 'resource_limits':{'address_space_gib':8,'watchdog_s':600}, + 'scope':'CONDITIONAL_SIMULATED','unknown_idle_power':'EXCLUDED_NOT_ZERO_MEASUREMENT'} + + +def main(): + p=argparse.ArgumentParser(description=__doc__) + for name in ('stage','mixed-model','all-hbf-model','thermal-binary','artifact-root'): + p.add_argument('--'+name,type=Path,required=True) + a=p.parse_args() + directory=a.stage/'inputs';directory.mkdir(exist_ok=False) + points=[] + for topology in TOPOLOGIES: + specs=[('pilot',1_536_000_000_000,STRATEGIES[2],8,4)] + specs += [('base',rate,strategy,20,10) for rate in + (384_000_000_000,768_000_000_000,1_152_000_000_000,1_536_000_000_000,1_920_000_000_000) + for strategy in STRATEGIES] + for kind,rate,strategy,active,recovery in specs: + point_id=f'{kind}-{topology}-{rate//10**9}-{STRATEGIES.index(strategy)}-01' + path=directory/(point_id+'.json') + save(path,config(point_id,topology,rate,strategy,active,recovery)) + points.append({'point_id':point_id,'kind':kind,'topology':topology, + 'config':str(path.resolve()),'config_sha256':hashlib.sha256(path.read_bytes()).hexdigest(), + 'model_dir':str((a.all_hbf_model if topology=='all_hbf_direct' else a.mixed_model).resolve()), + 'thermal_binary':str(a.thermal_binary.resolve()),'artifact_root':str(a.artifact_root.resolve()), + 'output':str((a.stage/'points'/point_id).resolve())}) + points.sort(key=lambda row:(row['kind']!='pilot',row['point_id'])) + save(a.stage/'RUN_INDEX.json',{'schema_version':'eq3-system-run-index-v1', + 'authorization':'USER_EXPLICIT_FOUR_TOPOLOGY_PLAN','points':points, + 'source_locks':{name:hashlib.sha256((HERE/name).read_bytes()).hexdigest() for name in + ('run_system_point.py','energy.py','rate_workload.py','topology_service.py','launch_stage.py')}, + 'resources':{'stage_wall_s':21600,'point_wall_s':600,'point_output_gib':1, + 'stage_output_gib':80,'host_ram_reserve_gib':32,'host_disk_reserve_gib':100}}) + + +if __name__=='__main__':main() diff --git a/experiments/eq3_system_thermal/rate_workload.py b/experiments/eq3_system_thermal/rate_workload.py new file mode 100644 index 0000000..c7a81a7 --- /dev/null +++ b/experiments/eq3_system_thermal/rate_workload.py @@ -0,0 +1,64 @@ +"""Exact offered traffic, independent of controller and service outcomes.""" +from __future__ import annotations + + +class RateWorkload: + def __init__(self, channels, config): + self.channels = channels + self.config = config + self.active_ns = config["active_ns"] + self.rate = config["per_stack_Bps"] + self.pattern = config.get("pattern", "continuous") + if self.pattern not in {"continuous", "burst_equal_mean"}: + raise ValueError("unknown rate pattern") + if type(self.rate) is not int or self.rate < 0: + raise ValueError("rate must be nonnegative integer B/s") + self.remainder = {s: 0 for s in channels} + self.cursor = {s: 0 for s in channels} + self.now = 0 + self.total = 0 + + def advance(self, start_ns, end_ns): + if start_ns != self.now or end_ns <= start_ns: + raise ValueError("noncontiguous workload horizon") + duration = max(0, min(end_ns, self.active_ns) - start_ns) + if self.pattern == "burst_equal_mean": + period = self.config.get("burst_period_ns", 200_000_000) + on = period // 2 + if period <= 0 or period % (end_ns - start_ns) or on % (end_ns - start_ns): + raise ValueError("burst edges must align with thermal observation windows") + duration *= 2 if start_ns % period < on else 0 + result = {} + hot = self.config.get("hot_stack") + distribution = self.config.get("channel_distribution", "uniform") + for stack, channel_mapping in sorted(self.channels.items()): + ids = sorted(channel_mapping, key=int) + if distribution == "first_quarter": + active = ids[:max(1, len(ids) // 4)] + elif distribution == "first_half": + active = ids[:max(1, len(ids) // 2)] + elif distribution == "uniform": + active = ids + else: + raise ValueError("unknown channel distribution") + # Hot-stack experiment keeps aggregate offered bytes fixed: half + # to one stack, half uniformly to the other stacks. + if hot is not None: + if hot not in self.channels or len(self.channels) < 2: + raise ValueError("invalid hot-stack identity") + numerator = len(self.channels) + denominator = 2 if stack == hot else 2 * (len(self.channels) - 1) + else: + numerator, denominator = 1, 1 + divisor = 1_000_000_000 * denominator + offered, self.remainder[stack] = divmod( + self.rate * duration * numerator + self.remainder[stack], divisor) + base, tail = divmod(offered, len(active)) + row = {c: (base if c in active else 0) for c in ids} + for offset in range(tail): + row[active[(self.cursor[stack] + offset) % len(active)]] += 1 + self.cursor[stack] = (self.cursor[stack] + tail) % len(active) + result[stack] = row + self.total += offered + self.now = end_ns + return result diff --git a/experiments/eq3_system_thermal/reliability.py b/experiments/eq3_system_thermal/reliability.py new file mode 100644 index 0000000..ec51d8b --- /dev/null +++ b/experiments/eq3_system_thermal/reliability.py @@ -0,0 +1,232 @@ +"""Conditional retention-age accounting; never an RBER or lifetime model.""" + +from __future__ import annotations + +import math +import hashlib +from copy import deepcopy +from typing import Any + + +ALLOWED_EA_EV = (1.01, 1.04, 1.08) +DEFAULT_TREF_K = 358.15 +KB_EV_PER_K = 8.62e-5 +DAY_NS = 86_400_000_000_000 + + +def arrhenius_acceleration(temperature_k: float, ea_ev: float, tref_k: float) -> float: + """Return equivalent-reference-age rate at ``temperature_k``.""" + for value, name in ((temperature_k, "temperature_k"), (ea_ev, "ea_ev"), + (tref_k, "tref_k")): + if not isinstance(value, (int, float)) or not math.isfinite(value) or value <= 0: + raise ValueError(f"{name} must be finite and positive") + return math.exp(ea_ev / KB_EV_PER_K * (1.0 / tref_k - 1.0 / temperature_k)) + + +class ReliabilityLedger: + """Keep per-block conditional age and observed program/erase facts. + + The 24-hour value is a wall-clock scenario cadence. It is deliberately + independent of equivalent age and is not a failure threshold. + """ + + def __init__(self, config: dict[str, Any]): + ea_ev = float(config.get("ea_ev", 1.04)) + if ea_ev not in ALLOWED_EA_EV: + raise ValueError(f"ea_ev must be one of {ALLOWED_EA_EV}") + self.ea_ev = ea_ev + self.tref_k = float(config.get("tref_k", DEFAULT_TREF_K)) + if self.tref_k != DEFAULT_TREF_K: + raise ValueError("this scenario contract fixes tref_k at 358.15 K") + self.period_ns = int(config.get("maintenance_period_ns", DAY_NS)) + if self.period_ns != DAY_NS: + raise ValueError("this scenario contract fixes maintenance cadence at 24 h") + self.initial_age_ns = int(config.get("initial_equivalent_age_ns", 0)) + self.initial_wall_age_ns = int(config.get("initial_wall_age_ns", 0)) + self.epoch_ns = int(config.get("maintenance_epoch_ns", 0)) + self.refresh_trigger = config.get("refresh_trigger", "wall_only") + if self.refresh_trigger not in {"wall_only", "equivalent_age_or_wall"}: + raise ValueError("unsupported refresh_trigger") + if min(self.initial_age_ns, self.initial_wall_age_ns, self.epoch_ns) < 0: + raise ValueError("initial age and epoch must be non-negative") + if self.initial_wall_age_ns > self.period_ns: + raise ValueError("initial wall age cannot exceed the 24 h cadence") + self._blocks: dict[str, dict[str, Any]] = {} + self.events: list[dict[str, Any]] = [] + self._drained_event_counts: dict[str, int] = {} + + def _block(self, block_id: str) -> dict[str, Any]: + if not isinstance(block_id, str) or not block_id: + raise ValueError("block_id must be a non-empty physical identity") + return self._blocks.setdefault(block_id, { + "block_id": block_id, + "equivalent_age_ns": float(self.initial_age_ns), + "last_update_ns": self.epoch_ns, + "program_phase_started_count": 0, + "program_completed_count": 0, + "erase_phase_started_count": 0, + "erase_completed_count": 0, + "last_refresh_commit_ns": None, + "next_wall_due_ns": self.epoch_ns + self.period_ns - self.initial_wall_age_ns, + "retry_scenario_count": 0, + "retry_scenario_latency_ns": 0, + "retry_scenario_energy_j": 0.0, + }) + + def due_reasons(self, block_id: str, at_ns: int) -> list[str]: + """Return conservative refresh-policy triggers, never failure facts.""" + state = self._block(block_id) + if not isinstance(at_ns, int) or at_ns < state["last_update_ns"]: + raise ValueError("due query precedes accounted block time") + reasons = [] + if at_ns >= state["next_wall_due_ns"]: + reasons.append("WALL_24H_SCENARIO_CADENCE") + if (self.refresh_trigger == "equivalent_age_or_wall" + and state["equivalent_age_ns"] >= self.period_ns): + reasons.append("EQUIVALENT_AGE_24H_CONSERVATIVE_POLICY") + return reasons + + def accounted_through_ns(self, block_id: str) -> int: + """Expose the exact age-integration frontier for online consumers.""" + return self._block(block_id)["last_update_ns"] + + def advance_temperature(self, block_id: str, start_ns: int, end_ns: int, + temperature_k: float) -> float: + state = self._block(block_id) + if not all(isinstance(v, int) and not isinstance(v, bool) for v in (start_ns, end_ns)): + raise ValueError("temperature interval times must be integer ns") + if start_ns != state["last_update_ns"] or end_ns < start_ns: + raise ValueError("temperature intervals must be contiguous and non-negative") + rate = arrhenius_acceleration(float(temperature_k), self.ea_ev, self.tref_k) + delta = (end_ns - start_ns) * rate + state["equivalent_age_ns"] += delta + state["last_update_ns"] = end_ns + self.events.append({"kind": "temperature_age", "block_id": block_id, + "start_ns": start_ns, "end_ns": end_ns, + "temperature_k": float(temperature_k), + "acceleration": rate, "equivalent_age_delta_ns": delta}) + return delta + + def advance_temperature_many(self, block_ids: list[str], start_ns: int, end_ns: int, + temperature_k: float) -> float: + """Advance a same-temperature cohort with one compact audit event.""" + if not block_ids or len(set(block_ids)) != len(block_ids): + raise ValueError("block_ids must be a nonempty unique list") + if not all(isinstance(item, str) and item for item in block_ids): + raise ValueError("block IDs must be nonempty strings") + if not all(isinstance(v, int) and not isinstance(v, bool) for v in (start_ns, end_ns)): + raise ValueError("temperature interval times must be integer ns") + if end_ns < start_ns: + raise ValueError("temperature interval must be non-negative") + rate = arrhenius_acceleration(float(temperature_k), self.ea_ev, self.tref_k) + delta = (end_ns - start_ns) * rate + for block_id in block_ids: + state = self._block(block_id) + if state["last_update_ns"] != start_ns: + raise ValueError("temperature cohort intervals must be contiguous") + state["equivalent_age_ns"] += delta + state["last_update_ns"] = end_ns + identities = "\n".join(sorted(block_ids)).encode() + self.events.append({"kind": "temperature_age_cohort", + "block_count": len(block_ids), + "block_ids_sha256": hashlib.sha256(identities).hexdigest(), + "first_block_id": min(block_ids), "last_block_id": max(block_ids), + "start_ns": start_ns, "end_ns": end_ns, + "temperature_k": float(temperature_k), + "acceleration": rate, "equivalent_age_delta_ns_each": delta}) + return delta + + def _media_terminal(self, kind: str, block_id: str, completion_ns: int, + succeeded: bool, operation_id: str, phase_started: bool) -> None: + if kind not in {"program", "erase"}: + raise ValueError("unsupported media operation") + state = self._block(block_id) + if not isinstance(completion_ns, int) or completion_ns < state["last_update_ns"]: + raise ValueError("media completion precedes accounted block time") + if succeeded and not phase_started: + raise ValueError("successful media operation must have started") + if phase_started: + state[f"{kind}_phase_started_count"] += 1 + if succeeded: + state[f"{kind}_completed_count"] += 1 + self.events.append({"kind": kind, "block_id": block_id, + "operation_id": str(operation_id), + "completion_ns": completion_ns, + "phase_started": bool(phase_started), + "succeeded": bool(succeeded), + "completed_counted": bool(succeeded), + "wear_exposure_semantics": "PHASE_START_COUNT_ONLY_NO_DAMAGE_CURVE"}) + + def record_program(self, block_id: str, completion_ns: int, succeeded: bool, + operation_id: str, *, phase_started: bool = True) -> None: + self._media_terminal("program", block_id, completion_ns, succeeded, operation_id, + phase_started) + + def record_erase(self, block_id: str, completion_ns: int, succeeded: bool, + operation_id: str, *, phase_started: bool = True) -> None: + self._media_terminal("erase", block_id, completion_ns, succeeded, operation_id, + phase_started) + + def record_refresh_terminal(self, block_id: str, completion_ns: int, committed: bool, + maintenance_request_id: str) -> None: + state = self._block(block_id) + if not isinstance(completion_ns, int) or completion_ns != state["last_update_ns"]: + raise ValueError("refresh terminal must coincide with accounted block time") + age_before = state["equivalent_age_ns"] + if committed: + state["equivalent_age_ns"] = 0.0 + state["last_refresh_commit_ns"] = completion_ns + state["next_wall_due_ns"] = completion_ns + self.period_ns + self.events.append({ + "kind": "refresh_terminal", "block_id": block_id, + "maintenance_request_id": str(maintenance_request_id), + "completion_ns": completion_ns, "committed": bool(committed), + "age_before_ns": age_before, + "age_after_ns": state["equivalent_age_ns"], + }) + + def record_retry_scenario(self, block_id: str, at_ns: int, count: int, + extra_latency_ns: int, extra_energy_j: float, + scenario_id: str) -> None: + state = self._block(block_id) + if not all(isinstance(v, int) and not isinstance(v, bool) and v >= 0 + for v in (at_ns, count, extra_latency_ns)): + raise ValueError("retry fact counts/times must be non-negative integer values") + if at_ns < state["last_update_ns"] or not math.isfinite(extra_energy_j) or extra_energy_j < 0: + raise ValueError("invalid retry scenario fact") + state["retry_scenario_count"] += count + state["retry_scenario_latency_ns"] += extra_latency_ns + state["retry_scenario_energy_j"] += extra_energy_j + self.events.append({ + "kind": "retry_scenario", "block_id": block_id, "at_ns": at_ns, + "count": count, "extra_latency_ns": extra_latency_ns, + "extra_energy_j": extra_energy_j, "scenario_id": str(scenario_id), + "evidence_class": "SCENARIO_ASSUMPTION_NO_RBER_CLAIM", + }) + + def drain_events(self) -> list[dict[str, Any]]: + """Move the current audit delta to the caller and release its memory.""" + result = self.events + self.events = [] + for row in result: + kind = row["kind"] + self._drained_event_counts[kind] = self._drained_event_counts.get(kind, 0) + 1 + return result + + def snapshot(self) -> dict[str, Any]: + return { + "schema_version": "eq3-conditional-reliability-v1", + "evidence_class": "CONDITIONAL_SIMULATED_HEATWATCH_PROXY", + "ea_ev": self.ea_ev, "tref_k": self.tref_k, + "maintenance_period_ns": self.period_ns, + "maintenance_period_semantics": "SCENARIO_CADENCE_NOT_FAILURE_THRESHOLD", + "refresh_trigger": self.refresh_trigger, + "equivalent_age_trigger_semantics": "CONSERVATIVE_REFRESH_POLICY_NOT_FAILURE_THRESHOLD", + "initial_age_semantics": "WALL_AND_EQUIVALENT_AGE_ARE_INDEPENDENT_AND_UNSCALED", + "blocks": deepcopy(self._blocks), "retained_events": deepcopy(self.events), + "drained_event_counts": dict(sorted(self._drained_event_counts.items())), + "unavailable": ["RBER", "ECC_STRENGTH", "FAILURE_PROBABILITY", "LIFETIME"], + } + + +__all__ = ["ReliabilityLedger", "arrhenius_acceleration"] diff --git a/experiments/eq3_system_thermal/run_causal_point.py b/experiments/eq3_system_thermal/run_causal_point.py new file mode 100644 index 0000000..fa537fe --- /dev/null +++ b/experiments/eq3_system_thermal/run_causal_point.py @@ -0,0 +1,741 @@ +#!/usr/bin/env python3 +"""Run one isolated causal storage/control/thermal engineering point.""" +from __future__ import annotations + +import argparse +from collections import defaultdict +from dataclasses import asdict +from datetime import datetime, timezone +import hashlib +import json +import math +import os +from pathlib import Path +import platform +import resource +import subprocess +import sys +import time + +HERE = Path(__file__).resolve().parent +ROOT = HERE.parents[1] +MAINTENANCE = ROOT / "experiments" / "eq3_maintenance" +sys.path.insert(0, str(MAINTENANCE)) + +from causal_service import CausalTopologyService +from causal_workload import CausalExecutor, build_architecture_trace +from causal_maintenance_age import CausalMaintenanceAgeAdapter +from endpoint_policy import EndpointAwarePolicy +from maintenance_driver import MaintenanceDriver +from read_rate_policy import EngineeringProfile, StackWindowFacts, WindowFacts +from reliability import ReliabilityLedger +from thermal_client import ThermalService + +WINDOW_NS = 20_000_000 + + +def save(path, value): + Path(path).write_text(json.dumps(value, indent=2, allow_nan=False) + "\n") + + +def digest(path): + return hashlib.sha256(Path(path).read_bytes()).hexdigest() + + +def percentile(samples, pct): + total = sum(size for _, size in samples) + if not total: + return None + threshold = (total * pct + 99) // 100 + cursor = 0 + for delay, size in sorted(samples): + cursor += size + if cursor >= threshold: + return delay + raise AssertionError("percentile accumulation failed") + + +class CausalEnergyAdapter: + """Map disjoint causal activity phases to thermal component energy.""" + + def __init__(self, normalized, service_config, profile, granularity=None): + self.components = {row["id"] for row in normalized["components"]} + self.profile = profile + self.hbf_channels = {} + self.hbm_dies = defaultdict(list) + granularity = granularity or {"mode": "physical_channels"} + if granularity.get("mode") not in {"physical_channels", "uniform_stack_group_16ch"}: + raise ValueError("unknown causal channel granularity") + self.granularity = dict(granularity) + for stack in service_config["fabric"]["hbf"]: + dies = sorted((row for row in normalized["components"] + if row.get("device_id") == stack and row.get("role") == "array_die"), + key=lambda row: row["die_index"]) + channels = sorted(service_config["channels"][stack], + key=(int if granularity["mode"] == "physical_channels" else str)) + if granularity["mode"] == "physical_channels": + if len(dies) != len(channels): + raise ValueError("causal HBF channel/die mapping must be explicit one-to-one") + self.hbf_channels[stack] = {channel: [die["id"]] + for channel, die in zip(channels, dies)} + else: + if (granularity.get("physical_channels_per_stack") != 16 + or len(dies) != 16 or "causal_channel_groups" not in service_config): + raise ValueError("uniform group requires explicit 16-die/channel evidence") + self.hbf_channels[stack] = {channel: [die["id"] for die in dies] + for channel in channels} + for row in normalized["components"]: + if (str(row.get("physical_type", "")).startswith("HBM") + and row.get("role") == "array_die"): + self.hbm_dies[row["device_id"]].append(row["id"]) + + def map(self, activities, *, gpu_compute_j=0.0, gpu_external_j=0.0): + component = defaultdict(float) + scope = defaultdict(float) + + def add(owner, joules, name): + if owner not in self.components: + raise ValueError(f"energy owner absent from thermal model: {owner}") + if not math.isfinite(joules) or joules < 0: + raise ValueError("energy must be finite and nonnegative") + component[owner] += joules + scope[name] += joules + + def hbm(stack, size, name, array_j_per_byte, base_j_per_byte): + dies = self.hbm_dies.get(stack, ()) + if not dies: + raise ValueError(f"HBM dies unavailable for {stack}") + for die in dies: + add(die, size * array_j_per_byte / len(dies), name + ":array_uniform") + add(stack + ".base", size * base_j_per_byte, name + ":base") + + for row in activities: + phase, operation = row["phase"], row["operation"] + stack, size = row["stack"], row["bytes"] + if phase == "media_read": + if stack.startswith("hbm"): + hbm(stack, size, operation, + self.profile["hbm_array_j_per_byte"], + self.profile["hbm_base_j_per_byte"]) + else: + dies = self.hbf_channels[stack][str(row["channel"])] + array_scope = (operation + ":array" if len(dies) == 1 + else operation + ":array_uniform_group") + for die in dies: + add(die, size * self.profile["read_array_j_per_byte"] / len(dies), + array_scope) + add(stack + ".base", size * self.profile["read_base_j_per_byte"], operation + ":base") + elif phase == "media_program": + dies = self.hbf_channels[stack][str(row["channel"])] + array_scope = (operation + ":array" if len(dies) == 1 + else operation + ":array_uniform_group") + for die in dies: + add(die, size * self.profile["program_array_j_per_byte"] / len(dies), + array_scope) + add(stack + ".base", size * self.profile["program_base_j_per_byte"], operation + ":base") + elif phase == "media_erase": + blocks = size / self.profile["erase_block_bytes"] + dies = self.hbf_channels[stack][str(row["channel"])] + erase_scope = ("erase:array_prorated" if len(dies) == 1 + else "erase:array_prorated_uniform_group") + for die in dies: + add(die, blocks * self.profile["erase_j_per_block"] / len(dies), + erase_scope) + elif phase == "media_fill": + hbm(stack, size, "hbm_fill", + self.profile["hbm_fill_array_j_per_byte"], + self.profile["hbm_fill_base_j_per_byte"]) + elif phase in {"relay_receive", "partner_gpu_drain"}: + partner = row.get("partner", stack) + coefficient = (self.profile["relay_receive_j_per_byte"] + if phase == "relay_receive" + else self.profile["relay_send_j_per_byte"]) + add(partner + ".base", size * coefficient, "relay:" + phase) + elif phase in {"reverse_relay", "partner_gpu_receive"}: + partner = row.get("partner") + owner = stack if phase == "reverse_relay" else partner + coefficient = (self.profile["relay_send_j_per_byte"] + if phase == "reverse_relay" + else self.profile["relay_receive_j_per_byte"]) + add(owner + ".base", size * coefficient, "migration:" + phase) + add("gpu", gpu_compute_j, "gpu:causal_compute") + add("gpu", gpu_external_j, "gpu:independent_external") + return {"component_energy_j": dict(component), "scope_energy_j": dict(scope), + "total_j": sum(component.values()), + "evidence": "CONDITIONAL_CAUSAL_ACTIVITY_ENGINEERING_PROXY_NOT_CALIBRATED"} + + +def _default_energy_profile(raw): + defaults = { + "read_array_j_per_byte": 40e-12, "read_base_j_per_byte": 10e-12, + "program_array_j_per_byte": 0.05 * 100e-6 / 4096, + "program_base_j_per_byte": 0.01 * 100e-6 / 4096, + "hbm_array_j_per_byte": 40e-12, "hbm_base_j_per_byte": 2e-12, + "hbm_fill_array_j_per_byte": 40e-12, "hbm_fill_base_j_per_byte": 2e-12, + "relay_receive_j_per_byte": 2e-12, "relay_send_j_per_byte": 2e-12, + "erase_j_per_block": 50e-6, "erase_block_bytes": 1_048_576, + } + result = {**defaults, **raw} + if any(not isinstance(value, (int, float)) or not math.isfinite(value) or value < 0 + for value in result.values()): + raise ValueError("invalid causal energy profile") + return result + + +def _changed_rows(rows, prior): + changed = [] + fields = ("cumulative_served_bytes", "remaining_bytes", "unadmitted_bytes", + "inflight_transferred_bytes", "state", "completion_ns") + for row in rows: + signature = tuple(row.get(field) for field in fields) + if prior.get(row["job_id"]) != signature: + changed.append(row) + prior[row["job_id"]] = signature + for row in changed: + if row["state"] == "DONE": + prior.pop(row["job_id"], None) + return changed + + +def _identity_hash(values): + return hashlib.sha256("\n".join(sorted(values)).encode()).hexdigest() + + +def _aggregate_activities(rows): + grouped = {} + keys = ("operation", "stack", "channel", "route", "partner", "phase", "resource") + for row in rows: + identity = tuple(row.get(key) for key in keys) + item = grouped.setdefault(identity, {key: row.get(key) for key in keys}) + item["activity_count"] = item.get("activity_count", 0) + 1 + item["bytes"] = item.get("bytes", 0) + row["bytes"] + item["first_start_ns"] = min(item.get("first_start_ns", row["start_ns"]), row["start_ns"]) + item["last_end_ns"] = max(item.get("last_end_ns", row["end_ns"]), row["end_ns"]) + return list(grouped.values()) + + +def _aggregate_progress(rows): + grouped = {} + for row in rows: + identity = (row["operation"], row["stack"], row["route"], row["state"]) + item = grouped.setdefault(identity, { + "operation": row["operation"], "stack": row["stack"], + "route": row["route"], "state": row["state"], "job_count": 0, + "total_bytes": 0, "cumulative_served_bytes": 0, "remaining_bytes": 0, + "unadmitted_bytes": 0, "inflight_transferred_bytes": 0, + }) + item["job_count"] += 1 + for key in ("total_bytes", "cumulative_served_bytes", "remaining_bytes", + "unadmitted_bytes", "inflight_transferred_bytes"): + item[key] += row[key] + return list(grouped.values()) + + +def _aggregate_completions(rows): + grouped = {} + for row in rows: + identity = (row["operation"], row["stack"]) + item = grouped.setdefault(identity, {"operation": row["operation"], + "stack": row["stack"], "count": 0, + "bytes": 0, "first_completion_ns": None, + "last_completion_ns": None, "ids": []}) + item["count"] += 1; item["bytes"] += row["bytes"] + item["ids"].append(row["job_id"]) + item["first_completion_ns"] = (row["completion_ns"] if item["first_completion_ns"] is None + else min(item["first_completion_ns"], row["completion_ns"])) + item["last_completion_ns"] = (row["completion_ns"] if item["last_completion_ns"] is None + else max(item["last_completion_ns"], row["completion_ns"])) + result = [] + for item in grouped.values(): + item["job_ids_sha256"] = _identity_hash(item.pop("ids")); result.append(item) + return result + + +def _aggregate_observations(events, submissions): + event_groups = {} + for row in events: + kind = row["kind"] + item = event_groups.setdefault(kind, {"kind": kind, "count": 0, + "completed_bytes": 0, "retry_count": 0}) + item["count"] += 1 + item["completed_bytes"] += int(row.get("completed_bytes", row.get("bytes", 0))) + item["retry_count"] += int(row.get("retry_count", 0)) + submission_groups = {} + for row in submissions: + identity = (row["operation"], row["stack"], row.get("route")) + item = submission_groups.setdefault(identity, { + "operation": row["operation"], "stack": row["stack"], + "route": row.get("route"), "count": 0, "bytes": 0, "ids": []}) + item["count"] += 1; item["bytes"] += row["bytes"]; item["ids"].append(row["job_id"]) + compact_submissions = [] + for item in submission_groups.values(): + item["job_ids_sha256"] = _identity_hash(item.pop("ids")); compact_submissions.append(item) + return list(event_groups.values()), compact_submissions + + +def _pending_trace_bytes_by_stack(trace, executor_config): + """Exact source-placement bytes for arrived but not instantiated batches.""" + targets = executor_config.get("stripe_targets") + if not targets: + target = executor_config.get("default_placement") + targets = [target] if target else None + if not targets: + raise ValueError("pending backlog requires explicit stripe targets") + stripe = int(executor_config["stripe_unit_bytes"]) + result = defaultdict(int) + for batch in trace["batches"]: + for task in batch["tasks"]: + if task["type"] != "storage": + continue + size = int(task["tensor"]["bytes"]) + full, tail = divmod(size, stripe) + base, extra = divmod(full, len(targets)) + for index, target in enumerate(targets): + child = (base + (index < extra)) * stripe + if tail and index == full % len(targets): + child += tail + result[target["stack"]] += child + return dict(result) + + +def _useful_backlog(offered_total, delivered_total, pending_bytes): + # A dependency's payload is useful only when all children/retries finish. + # Include already-finished siblings until that unique useful completion. + result = offered_total - delivered_total + pending_bytes + if result < 0: + raise AssertionError("effective delivery exceeds offered payload") + return result + + +def _build_maintenance(config, hbf_channels): + raw = config.get("maintenance", {"mode": "disabled"}) + if raw.get("mode") == "disabled": + return None, None, None + if raw.get("mode") != "shared": + raise ValueError("causal runner supports disabled or shared maintenance") + blocks = int(raw["aged_blocks_per_stack"]) + spares = int(raw["spares_per_channel"]) + ledger = ReliabilityLedger({ + "ea_ev": raw["ea_ev"], "refresh_trigger": raw["refresh_trigger"], + "initial_equivalent_age_ns": raw["initial_equivalent_age_ns"], + "initial_wall_age_ns": raw["initial_wall_age_ns"], + }) + pools, extents, temperatures = {}, [], {} + for stack, channels in sorted(hbf_channels.items()): + ordered = sorted(channels, key=int) + if blocks <= 0 or blocks % len(ordered): + raise ValueError("aged_blocks_per_stack must be positive and divide HBF channels") + pools[stack], temperatures[stack] = {}, {} + for channel in ordered: + pools[stack][channel] = [f"{stack}:ch{channel}:spare{i}" for i in range(spares)] + temperatures[stack][channel] = float(raw["initial_temperature_k"]) + for index in range(blocks // len(ordered)): + extents.append({"extent_id": f"{stack}:ch{channel}:extent{index}", + "stack": stack, "channel": channel, + "source_block_id": f"{stack}:ch{channel}:source{index}", + "version": 0}) + driver = MaintenanceDriver(ledger, { + "block_bytes": int(raw["block_bytes"]), + "pages_per_block": int(raw["pages_per_block"]), + "max_blocks_per_cohort": int(raw["max_blocks_per_cohort"]), + "spare_block_ids_by_stack_channel": pools, + "program_energy_j_per_byte": raw.get("program_energy_j_per_byte"), + "erase_energy_j_per_operation": raw.get("erase_energy_j_per_operation"), + "energy_evidence": raw["energy_evidence"], + }) + driver.register_extents(extents) + return ledger, driver, CausalMaintenanceAgeAdapter(ledger, driver, temperatures) + + +def _observed_channel_temperatures(heat, hbf_channels): + entities = heat.get("entity_temperatures_k") + if not isinstance(entities, dict): + raise ValueError("maintenance requires per-entity thermal temperatures") + result = {} + for stack, channels in hbf_channels.items(): + result[stack] = {} + for channel, components in channels.items(): + values = [] + for component in components: + row = entities.get(component) + if not isinstance(row, dict) or "hotspot_k" not in row: + raise ValueError(f"missing thermal hotspot for {component}") + values.append(float(row["hotspot_k"])) + result[stack][channel] = max(values) + return result + + +def execute(config, normalized, thermal, sink, *, initial_trace=None, trace_factory=None): + window_ns = int(config.get("window_ns", WINDOW_NS)) + if window_ns <= 0: + raise ValueError("window_ns must be positive") + end_ns = int(config["active_ns"]) + int(config["recovery_ns"]) + if end_ns % window_ns: + raise ValueError("duration must align to control windows") + trace_config = dict(config["trace"]) + total_batches = int(trace_config["total_batches"]) + max_active = int(trace_config.get("max_active_batches", 4)) + interval_ns = int(trace_config["batch_interval_ns"]) + if total_batches <= 0 or max_active <= 0 or interval_ns <= 0: + raise ValueError("invalid streaming trace bounds") + if trace_factory is None: + trace_factory = lambda index: build_architecture_trace({ + **trace_config, "batch_intervals": 1, "first_interval": index}) + seed = initial_trace or trace_factory(0) + executor = CausalExecutor(seed, config["executor"]) + next_batch = 1 + reliability_config = config.get("hbf_read_cost_proxy", {"mode": "disabled"}) + reliability_provider = None + if reliability_config["mode"] == "disabled": + service = CausalTopologyService(config["service"]) + elif reliability_config["mode"] == "conditional_nand_history_v1": + from ecc_cost_proxy import ReadCostProxy + from ecc_service_adapter import ReliabilityCausalService + if int(config["executor"].get("retry_count_per_source_read", 0)): + raise ValueError("static retry and history retry cannot be silently combined") + if config.get("maintenance", {"mode": "disabled"})["mode"] != "disabled": + raise ValueError("UNSUPPORTED_COMPOSITION: per-stack read age lacks refreshed-extent identity") + initial = reliability_config["initial_by_stack"] + if set(initial) != set(config["service"]["fabric"]["hbf"]): + raise ValueError("HBF reliability state must cover exactly the actual HBF stacks") + reliability_provider = ReadCostProxy(reliability_config["profile"], initial) + service = ReliabilityCausalService(config["service"], reliability_provider) + else: + raise ValueError("unsupported HBF read-cost proxy mode") + granularity = config.get("causal_channel_granularity", {"mode": "physical_channels"}) + energy = CausalEnergyAdapter(normalized, config["service"], + _default_energy_profile(config.get("energy", {})), + granularity) + if granularity.get("mode") != "physical_channels" \ + and config.get("maintenance", {"mode": "disabled"}).get("mode") != "disabled": + raise ValueError("grouped causal channels do not claim physical maintenance extents") + ledger, maintenance_driver, maintenance_age = _build_maintenance( + config, energy.hbf_channels) + stacks = sorted(config["service"]["channels"]) + targets = config["target_bytes_per_s_by_stack"] + if set(targets) != set(stacks): + raise ValueError("target_bytes_per_s_by_stack must exactly cover service stacks") + baseline = {stack: sum(config["service"]["channels"][stack].values()) * window_ns // 10**9 + for stack in stacks} + budgets = dict(baseline) + states = {stack: "normal" for stack in stacks} + policies = {} + for stack in stacks: + profile = EngineeringProfile( + profile_id=f"{config['point_id']}:{stack}", enabled=True, + strategy=config["strategy"], window_ns=window_ns, + target_bytes_per_s=int(targets[stack]), + step_bytes=max(1, baseline[stack] // 20), + minimum_budget_bytes=max(1, baseline[stack] // 10), + maximum_budget_bytes=baseline[stack], severe_budget_bytes=0, + light_fraction=float(config.get("light_fraction", .5))) + policies[stack] = EndpointAwarePolicy(profile) + known_jobs = {} + progress_signatures = {} + cumulative_tokens = 0 + total_energy = 0.0 + compute_intervals = [] + peak = {} + offered_total = defaultdict(int) + delivered_total = defaultdict(int) + group_stack_bytes = defaultdict(lambda: defaultdict(int)) + group_logical_issue = {} + pending_traces = [] + deferred_maintenance_jobs = [] + + for start in range(0, end_ns, window_ns): + stop = start + window_ns + service.begin_window(start, stop, budgets, states) + now = start + maintenance_jobs = list(deferred_maintenance_jobs) + deferred_maintenance_jobs = [] + if maintenance_age is not None: + maintenance_jobs.extend(maintenance_age.start_window(start)) + activities, changed_by_id, completions, executor_events, submissions = [], {}, [], [], [] + maintenance_deltas = [] + token_window = 0 + offered_window, delivered_window = defaultdict(int), defaultdict(int) + retry_window = defaultdict(int) + delays = defaultdict(list) + + def consume_observations(observed): + executor_events.extend(observed["events"]) + submissions.extend(observed["submissions"]) + for event in observed["events"]: + if event["kind"] == "compute_start": + compute_intervals.append((event["start_ns"], event["scheduled_end_ns"])) + elif event["kind"] == "storage_complete": + group_id = event["group_id"] + by_stack = group_stack_bytes.pop(group_id, {}) + logical = group_logical_issue.pop(group_id, event["completion_ns"]) + for stack, byte_count in by_stack.items(): + delivered_window[stack] += byte_count + delivered_total[stack] += byte_count + delays[stack].append((event["completion_ns"] - logical, byte_count)) + while now < stop: + retired = executor.retire_completed_batches(now) + newly_retired = sum(row["token_count"] for row in retired) + token_window += newly_retired + cumulative_tokens += newly_retired + while pending_traces and len(executor.trace["batches"]) < max_active: + executor.append_trace(pending_traces.pop(0)) + while next_batch < total_batches and next_batch * interval_ns <= now: + candidate_trace = trace_factory(next_batch) + if len(executor.trace["batches"]) < max_active and not pending_traces: + executor.append_trace(candidate_trace) + else: + pending_traces.append(candidate_trace) + next_batch += 1 + jobs = executor.poll(now) + jobs.extend(executor.offer_migrations(now)) + jobs.extend(maintenance_jobs) + maintenance_jobs = [] + physical_jobs = [] + for raw in jobs: + job = dict(raw) + metadata = dict(job.get("metadata", {})) + if job["arrival_ns"] < now: + metadata.setdefault("logical_issue_ns", job["arrival_ns"]) + metadata["external_wait_before_submit_ns"] = now - job["arrival_ns"] + job["arrival_ns"] = now + job["metadata"] = metadata + physical_jobs.append(job) + known_jobs[job["job_id"]] = job + if job["operation"] == "read" and "maintenance_id" not in job: + offered_window[job["stack"]] += job["bytes"] + offered_total[job["stack"]] += job["bytes"] + group_id = metadata.get("parent_group_id", job["job_id"]) + group_stack_bytes[group_id][job["stack"]] += job["bytes"] + logical = metadata.get("logical_issue_ns", job["arrival_ns"]) + group_logical_issue[group_id] = min( + group_logical_issue.get(group_id, logical), logical) + elif job["operation"] == "retry" and "maintenance_id" not in job: + retry_window[job["stack"]] += 1 + if physical_jobs: + service.submit_jobs(physical_jobs) + observed = executor.drain_observations() + consume_observations(observed) + candidates = [stop] + for value in (service.next_event_ns(), executor.next_internal_event_ns()): + if value is not None and value > now: + candidates.append(value) + arrival = next_batch * interval_ns if next_batch < total_batches else None + if arrival is not None and arrival > now: + candidates.append(arrival) + horizon = min(candidates) + receipt = service.advance_to(horizon) + activities.extend(receipt["activities"]) + for progress in _changed_rows(receipt["job_progress"], progress_signatures): + changed_by_id[progress["job_id"]] = progress + rows = {row["job_id"]: row for row in receipt["job_progress"]} + if maintenance_age is not None: + maintenance_delta = maintenance_age.consume_receipt(receipt) + next_phases = maintenance_delta.pop("next_phase_jobs") + maintenance_deltas.append(maintenance_delta) + if receipt["end_ns"] < stop: + maintenance_jobs.extend(next_phases) + else: + deferred_maintenance_jobs.extend(next_phases) + for job_id in receipt["completion_ids"]: + row = rows[job_id] + job = known_jobs.pop(job_id) + if job_id in executor.job_bytes: + executor.complete(job_id, row["completion_ns"], row["total_bytes"]) + completions.append({"job_id": job_id, "completion_ns": row["completion_ns"], + "bytes": row["total_bytes"], "operation": job["operation"], + "stack": job["stack"]}) + now = horizon + observed = executor.drain_observations() + consume_observations(observed) + retired = executor.retire_completed_batches(stop) + newly_retired = sum(row["token_count"] for row in retired) + token_window += newly_retired + cumulative_tokens += newly_retired + compute_ns = sum(max(0, min(stop, end) - max(start, begin)) + for begin, end in compute_intervals) + compute_intervals = [(begin, end) for begin, end in compute_intervals if end > stop] + mapped = energy.map( + activities, + gpu_compute_j=float(config.get("gpu_compute_w", 0)) * compute_ns / 1e9, + gpu_external_j=float(config.get("gpu_external_w", 0)) * window_ns / 1e9, + ) + total_energy += mapped["total_j"] + heat = thermal.advance(start, stop, mapped["component_energy_j"]) + reliability_window = None + if reliability_provider is not None: + channel_temperatures = _observed_channel_temperatures(heat, energy.hbf_channels) + # Conservative uniform-age stack proxy; not a claim about page history. + stack_temperatures = {stack: max(values.values()) + for stack, values in channel_temperatures.items()} + reliability_provider.observe(start, stop, stack_temperatures) + reliability_window = { + "mode": reliability_config["mode"], + "temperature_mapping": "HOTTEST_ARRAY_DIE_PER_STACK_CONSERVATIVE_UNIFORM_AGE_PROXY", + "state": reliability_provider.snapshot(), + "admission_cost_decisions": service.drain_cost_decisions(), + } + maintenance_finish = None + if maintenance_age is not None: + maintenance_finish = maintenance_age.finish_window( + stop, _observed_channel_temperatures(heat, energy.hbf_channels)) + for owner, value in heat["temperatures"].items(): + peak[owner] = max(peak.get(owner, value), value) + observed_states = {stack: heat["stack_states"][stack] for stack in stacks} + next_budgets, decisions, stack_facts = {}, {}, {} + for stack in stacks: + backlog_jobs = [job for job in known_jobs.values() + if job["stack"] == stack and job["operation"] == "read" + and "maintenance_id" not in job] + pending_bytes = sum(_pending_trace_bytes_by_stack(item, config["executor"]).get(stack, 0) + for item in pending_traces) + backlog = _useful_backlog(offered_total[stack], delivered_total[stack], pending_bytes) + oldest = max( + [stop - job.get("metadata", {}).get("logical_issue_ns", job["arrival_ns"]) + for job in backlog_jobs] + + [stop - item["batches"][0]["arrival_ns"] for item in pending_traces + if _pending_trace_bytes_by_stack(item, config["executor"]).get(stack, 0)], + default=0) + facts = StackWindowFacts( + stack_id=stack, offered_bytes=offered_window[stack], + delivered_bytes=delivered_window[stack], backlog_bytes=backlog, + oldest_wait_ns=oldest, latency_p95_ns=percentile(delays[stack], 95), + censored_requests=(len(backlog_jobs) + sum( + task["type"] == "storage" for item in pending_traces + for batch in item["batches"] for task in batch["tasks"])), + gate_limited=(bool(backlog_jobs) + and receipt["endpoint_quota_remaining_scaled"][stack] == 0), + backend_busy_fraction=None, resource_busy=None, + retry_count=(retry_window[stack] if reliability_provider is None else None)) + decision = policies[stack].evaluate(WindowFacts( + start_ns=start, end_ns=stop, guard_state=observed_states[stack], + stacks=(facts,), current_budget_bytes={stack: budgets[stack]}, + guard_states={stack: observed_states[stack]}, + hysteresis_budget_bytes={stack: heat["hysteresis_budget_bytes"][stack]})) + next_budgets[stack] = decision.stack_decisions[0].budget_bytes + decisions[stack] = asdict(decision) + stack_facts[stack] = asdict(facts) + grouped_output = granularity.get("mode") == "uniform_stack_group_16ch" + if grouped_output: + output_activities = _aggregate_activities(activities) + output_progress = _aggregate_progress(list(changed_by_id.values())) + output_completions = _aggregate_completions(completions) + output_events, output_submissions = _aggregate_observations( + executor_events, submissions) + else: + output_activities = activities + output_progress = list(changed_by_id.values()) + output_completions = completions + output_events, output_submissions = executor_events, submissions + row = { + "start_ns": start, "end_ns": stop, + "service": {"activities": output_activities, + "changed_job_progress": output_progress, + "completions": output_completions, + "endpoint_quota_remaining_scaled": receipt["endpoint_quota_remaining_scaled"], + "buffer_occupancy_bounds": receipt["buffer_occupancy_bounds"]}, + "executor": {"events": output_events, "submissions": output_submissions, + "completed_tokens": token_window, + "cumulative_completed_tokens": cumulative_tokens, + "active_batches": len(executor.trace["batches"]), + "uninstantiated_arrived_batches": len(pending_traces), + "token_rate_semantics": "EXACT_DAG_COMPLETIONS_NOT_CALIBRATED_TOKENS_PER_SECOND"}, + "output_granularity": ("UNIFORM_16_PHYSICAL_CHANNEL_GROUP_WINDOW_AGGREGATES" + if grouped_output else "PHYSICAL_CHANNEL_JOB_DELTAS"), + "energy": mapped, "thermal": heat, + "maintenance": ({"mode": "disabled"} if maintenance_age is None else { + "mode": "shared_exact_service", "receipt_deltas": maintenance_deltas, + "window_age_finish": maintenance_finish, + "driver": maintenance_driver.drain_delta(stop), + }), + "control": {"budgets": budgets, "next_budgets": next_budgets, + "observed_states": observed_states, "decisions": decisions, + "stack_facts": stack_facts, + "backend_latency": "UNKNOWN_FLUID_MODEL", + "backlog_semantics": "SUBMITTED_MINUS_FINAL_USEFUL_DELIVERY_PLUS_ARRIVED_UNINSTANTIATED_PAYLOAD;ACTIVE_FUTURE_UNISSUED_TASKS_EXCLUDED"}, + } + if reliability_window is not None: + row["hbf_read_cost_proxy"] = reliability_window + sink.write(json.dumps(row, separators=(",", ":"), allow_nan=False) + "\n") + budgets, states = next_budgets, observed_states + final = executor.result(end_ns) + return { + "hbf_read_cost_proxy": (None if reliability_provider is None else reliability_provider.snapshot()), + "trace_origin": seed["trace_origin"], + "structure_provenance": seed.get("structure_provenance", "SYNTHETIC_METADATA_ONLY"), + "completed_tokens": cumulative_tokens, + "incomplete_active_tokens": sum(not token["complete"] for token in final["tokens"]), + "uninstantiated_batches": len(pending_traces) + total_batches - next_batch, + "offered_useful_bytes_by_stack": dict(offered_total), + "delivered_useful_bytes_by_stack": dict(delivered_total), + "pending_external_jobs": len(final["pending_external_jobs"]), + "maintenance": ({"mode": "disabled"} if maintenance_driver is None else { + "mode": "shared_exact_service", + "driver": maintenance_driver.snapshot(), + "reliability": ledger.snapshot(), + "deferred_phase_jobs": len(deferred_maintenance_jobs), + }), + "energy_j": total_energy, "peak_k_by_owner": peak, + "service_facts": service.immutable_facts(), + "scope": "CONDITIONAL_CAUSAL_FLUID_NOT_NATIVE_NAND_OR_CALIBRATED_TOKEN_RATE", + } + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + for name in ("config", "model-dir", "thermal-binary", "artifact-root", "output"): + parser.add_argument("--" + name, type=Path, required=True) + args = parser.parse_args() + args.output.mkdir(parents=True, exist_ok=False) + started = time.monotonic() + try: + for key in ("OMP_NUM_THREADS", "OPENBLAS_NUM_THREADS", "MKL_NUM_THREADS", "NUMEXPR_NUM_THREADS"): + os.environ[key] = "1" + os.environ["CUDA_VISIBLE_DEVICES"] = "" + config = json.loads(args.config.read_text()) + save(args.output / "config.json", config) + sources = [HERE / name for name in ("run_causal_point.py", "causal_service.py", + "causal_workload.py", "endpoint_policy.py", "topology_service.py", + "maintenance_driver.py", "reliability.py", + "causal_maintenance_age.py", "tiny_cpu_trace.py")] + if config.get("hbf_read_cost_proxy", {}).get("mode", "disabled") != "disabled": + sources += [HERE / "ecc_cost_proxy.py", HERE / "ecc_service_adapter.py"] + sources += [MAINTENANCE / "thermal_client.py", MAINTENANCE / "read_rate_policy.py"] + sources += [ROOT / "tools" / "eq3_basic_fabric.py"] + if config["trace"].get("dependency_mode") == "tiny_cpu_template": + sources += [ROOT / config["trace"]["tiny_trace_path"]] + manifest = { + "started_utc": datetime.now(timezone.utc).isoformat(), + "environment_id": "eq3-thermal-cpu-v1", "python": sys.version, + "platform": platform.platform(), "input_sha256": digest(args.config), + "source_sha256": {str(path.relative_to(ROOT)): digest(path) for path in sources}, + "thermal_binary_sha256": digest(args.thermal_binary), + "source_revision": subprocess.check_output( + ["git", "rev-parse", "HEAD"], cwd=ROOT, text=True).strip(), + "model_dir": str(args.model_dir.resolve()), "cpu_threads": 1, "gpu_count": 0, + "resource_limits": config["resource_limits"], + } + save(args.output / "manifest.json", manifest) + limit = config["resource_limits"]["address_space_gib"] * 1024**3 + resource.setrlimit(resource.RLIMIT_AS, (limit, limit)); resource.setrlimit(resource.RLIMIT_CORE, (0, 0)) + normalized = json.loads((args.model_dir / "normalized.json").read_text()) + baseline = {stack: sum(channels.values()) * int(config.get("window_ns", WINDOW_NS)) // 10**9 + for stack, channels in config["service"]["channels"].items()} + with ThermalService(args.thermal_binary, args.model_dir, args.output / "thermal-process", + artifact_root=args.artifact_root, baseline_budgets=baseline, + limits=config.get("thermal_limits_k")) as thermal: + manifest.update(thermal_model_lock=thermal.lock, thermal_header=thermal.header) + save(args.output / "manifest.json", manifest) + with (args.output / "windows.jsonl").open("w") as sink: + summary = execute(config, normalized, thermal, sink) + save(args.output / "DONE.json", {"status": "COMPLETED", "summary": summary, + "wall_s": time.monotonic() - started, + "child_peak_rss_kib": resource.getrusage(resource.RUSAGE_CHILDREN).ru_maxrss}) + except BaseException as exc: + save(args.output / "FAILED.json", {"status": "FAILED", "error": repr(exc), + "wall_s": time.monotonic() - started, + "raw_preserved": True}) + raise + + +if __name__ == "__main__": + main() diff --git a/experiments/eq3_system_thermal/run_endpoint_guard_point.py b/experiments/eq3_system_thermal/run_endpoint_guard_point.py new file mode 100644 index 0000000..f3d9772 --- /dev/null +++ b/experiments/eq3_system_thermal/run_endpoint_guard_point.py @@ -0,0 +1,61 @@ +#!/usr/bin/env python3 +"""Run one system point with the default-off shared-HBM endpoint guard adapter.""" + +from __future__ import annotations + +import hashlib +import json +from pathlib import Path +import sys + +# This entry point must be executable directly by the serial launcher. Import +# the existing maintenance policy from its repository location before loading +# the endpoint adapter; do not rely on a caller-provided PYTHONPATH. +HERE = Path(__file__).resolve().parent +ROOT = HERE.parents[1] +sys.path.insert(0, str(ROOT / "experiments" / "eq3_maintenance")) + +import endpoint_policy +import run_system_point + + +def _digest(path: Path) -> str: + return hashlib.sha256(path.read_bytes()).hexdigest() + + +def _output_argument() -> Path | None: + try: + return Path(sys.argv[sys.argv.index("--output") + 1]) + except (ValueError, IndexError): + return None + + +def _record_adapter(output: Path | None) -> None: + if output is None or not (output / "manifest.json").is_file(): + return + path = output / "manifest.json" + value = json.loads(path.read_text()) + sources = (Path(__file__).resolve(), Path(endpoint_policy.__file__).resolve()) + value["entrypoint_adapter"] = { + "schema_version": "eq3-shared-hbm-endpoint-policy-adapter-v1", + "capability": ("HBF_LEGACY_READ_RATE_POLICY_PLUS_HBM_SHARED_ENDPOINT_" + "THERMAL_GUARD_WITH_NORMAL_BASELINE_RESTORE"), + "source_sha256": {str(source): _digest(source) for source in sources}, + "default_connected": False, + } + run_system_point.save(path, value) + + +def main() -> None: + # The isolated process is the opt-in boundary. The base runner and policy + # module remain unchanged for frozen historical points. + run_system_point.ReadRatePolicy = endpoint_policy.EndpointAwarePolicy + output = _output_argument() + try: + run_system_point.main() + finally: + _record_adapter(output) + + +if __name__ == "__main__": + main() diff --git a/experiments/eq3_system_thermal/run_maintenance_point.py b/experiments/eq3_system_thermal/run_maintenance_point.py new file mode 100644 index 0000000..83e0e5d --- /dev/null +++ b/experiments/eq3_system_thermal/run_maintenance_point.py @@ -0,0 +1,358 @@ +#!/usr/bin/env python3 +"""Separate aggregate maintenance/rate/thermal point runner. + +This does not modify or replace the frozen base runner. Maintenance is an +aggregate rate-service scenario, not native MQSim NAND execution. +""" +from __future__ import annotations + +import argparse +from dataclasses import asdict +from datetime import datetime, timezone +import json +import os +from pathlib import Path +import platform +import resource +import subprocess +import sys +import time + +HERE = Path(__file__).resolve().parent +ROOT = HERE.parents[1] +MAINTENANCE = ROOT / "experiments" / "eq3_maintenance" +sys.path.insert(0, str(MAINTENANCE)) + +from thermal_client import ThermalService +from read_rate_policy import EngineeringProfile, StackWindowFacts, WindowFacts +from endpoint_policy import EndpointAwarePolicy +from energy import EnergyMapper +from maintenance_driver import MaintenanceDriver +from rate_workload import RateWorkload +from reliability import ReliabilityLedger +from run_system_point import channel_map, digest, energy_activities, percentile, save +from topology_service import TopologyService + + +WINDOW_NS = 20_000_000 +MODES = {"disabled", "shared", "ideal_independent"} + + +def _maintenance(config, mapping): + raw = config["maintenance"] + required = { + "mode", "ea_ev", "refresh_trigger", "initial_equivalent_age_ns", + "initial_wall_age_ns", "initial_temperature_k", "block_bytes", + "pages_per_block", "aged_blocks_per_stack", "spares_per_channel", + "max_blocks_per_cohort", "program_energy_j_per_byte", + "erase_energy_j_per_operation", "energy_evidence", + } + if not isinstance(raw, dict) or set(raw) != required: + raise ValueError("maintenance config has missing or unknown fields") + if raw["mode"] not in MODES: + raise ValueError("unknown maintenance mode") + if raw["mode"] == "disabled": + return None, None, {} + blocks = raw["aged_blocks_per_stack"] + spares = raw["spares_per_channel"] + if any(isinstance(value, bool) or not isinstance(value, int) or value <= 0 + for value in (blocks, spares, raw["block_bytes"], raw["pages_per_block"], + raw["max_blocks_per_cohort"])): + raise ValueError("maintenance geometry/count fields must be positive integers") + ledger = ReliabilityLedger({"ea_ev": raw["ea_ev"], + "refresh_trigger": raw["refresh_trigger"], + "initial_equivalent_age_ns": raw["initial_equivalent_age_ns"], + "initial_wall_age_ns": raw["initial_wall_age_ns"]}) + pools, rows, by_channel = {}, [], {} + for stack, channels in sorted(mapping.items()): + ordered = sorted(channels, key=int) + if blocks % len(ordered): + raise ValueError("aged_blocks_per_stack must divide configured channels") + per_channel = blocks // len(ordered) + pools[stack] = {} + for channel in ordered: + pools[stack][channel] = [f"{stack}:ch{channel}:spare{i}" for i in range(spares)] + identities = [] + for index in range(per_channel): + extent = f"{stack}:ch{channel}:extent{index}" + identities.append(extent) + rows.append({"extent_id": extent, "stack": stack, "channel": channel, + "source_block_id": f"{stack}:ch{channel}:source{index}", + "version": 0}) + by_channel[(stack, channel)] = identities + driver = MaintenanceDriver(ledger, { + "block_bytes": raw["block_bytes"], "pages_per_block": raw["pages_per_block"], + "max_blocks_per_cohort": raw["max_blocks_per_cohort"], + "spare_block_ids_by_stack_channel": pools, + "program_energy_j_per_byte": raw["program_energy_j_per_byte"], + "erase_energy_j_per_operation": raw["erase_energy_j_per_operation"], + "energy_evidence": raw["energy_evidence"], + }) + driver.register_extents(rows) + return ledger, driver, by_channel + + +def _erase_energy(receipts, mapping, block_bytes, coefficient): + result = defaultdict_float() + for receipt in receipts: + for row in receipt["activities"]: + if row["phase"] != "media_erase": + continue + component = mapping[row["stack"]][str(row["channel"])] + result[component] += row["bytes"] / block_bytes * coefficient + return dict(result) + + +def defaultdict_float(): + from collections import defaultdict + return defaultdict(float) + + +def _compact_reliability(ledger, driver, now_ns): + if ledger is None: + return {"mode": "disabled", "status": "FRESH_BASELINE_NO_MAINTENANCE"} + states = ledger._blocks # bounded internal read; avoids a full deepcopy every window + return { + "mode": "enabled", "block_count": len(states), + "max_equivalent_age_ns": max((row["equivalent_age_ns"] for row in states.values()), + default=0), + "due_block_count": sum(bool(ledger.due_reasons(block, now_ns)) + for block in states), + "active_extent_count": len(driver.active_extents), + "outstanding_job_count": len(driver.outstanding), + "free_spare_count": sum(len(pool) for pool in driver.free_spares.values()), + "quarantined_block_count": len(driver.quarantined_blocks), + } + + +def _maintenance_pressure(ledger, extent_ids, now_ns, block_bytes): + if ledger is None: + return 0, None + due = 0 + deadlines = [] + for extent_id in extent_ids: + state = ledger._blocks[extent_id] + deadlines.append(state["next_wall_due_ns"]) + if ledger.due_reasons(extent_id, now_ns): + due += block_bytes + deadlines.append(now_ns) + return due, min(deadlines) if deadlines else None + + +def _receipt_output_delta(receipt): + """Project external progress to changed/current-terminal rows for raw output. + + The full receipt is consumed by energy, maintenance, and control first. A + stalled external job remains represented by ``blocked`` and by the + driver's outstanding summary; repeating its unchanged cumulative progress + in every 20 ms raw row adds no observation. + """ + result = dict(receipt) + end_ns = receipt["end_ns"] + result["job_progress"] = [row for row in receipt.get("job_progress", []) + if row["admitted_this_window_bytes"] > 0 + or row["served_this_window_bytes"] > 0 + or row["completion_ns"] == end_ns] + result["output_projection"] = ( + "CHANGED_OR_CURRENT_TERMINAL_EXTERNAL_PROGRESS;BLOCKED_ROWS_RETAINED" + ) + return result + + +def execute(config, normalized, thermal, sink): + service = TopologyService(config["service"]) + mapping = channel_map(normalized, config["service"]) + workload = RateWorkload(mapping, config["workload"]) + energy = EnergyMapper(normalized, mapping, config["energy"]) + ledger, driver, extents_by_channel = _maintenance(config, mapping) + extents_by_stack = {stack: [] for stack in config["service"]["channels"]} + if driver: + for extent_id, extent in driver.extents.items(): + extents_by_stack[extent["stack"]].append(extent_id) + maintenance_mode = config["maintenance"]["mode"] + independent = (TopologyService(config["service"]) + if maintenance_mode == "ideal_independent" else None) + stacks = sorted(config["service"]["channels"]) + hbf = sorted(mapping) + baseline = {s: sum(config["service"]["channels"][s].values()) * WINDOW_NS // 10**9 + for s in stacks} + budgets, states = dict(baseline), {s: "normal" for s in stacks} + profiles = {s: EngineeringProfile( + profile_id=config["point_id"] + ":" + s, enabled=True, + strategy=config["strategy"], window_ns=WINDOW_NS, + target_bytes_per_s=min(config["workload"]["per_stack_Bps"], + sum(config["service"]["channels"][s].values()) * 4 // 5), + step_bytes=baseline[s] // 20, minimum_budget_bytes=baseline[s] // 10, + maximum_budget_bytes=baseline[s], severe_budget_bytes=0, light_fraction=.5) + for s in stacks} + policies = {s: EndpointAwarePolicy(profile) for s, profile in profiles.items()} + previous_die_k = {component: float(config["maintenance"]["initial_temperature_k"]) + for channels in mapping.values() for component in channels.values()} + total_energy, peak = 0.0, {} + end = config["workload"]["active_ns"] + config["recovery_ns"] + if end % WINDOW_NS: + raise ValueError("run duration must align with thermal window") + final_receipt = None + for start in range(0, end, WINDOW_NS): + stop = start + WINDOW_NS + offered = workload.advance(start, stop) + jobs = driver.poll(start) if driver else [] + if maintenance_mode == "ideal_independent": + receipt = service.advance(start, stop, offered, budgets, states) + maintenance_receipt = independent.advance( + start, stop, {}, baseline, states, jobs) + receipts = [receipt, maintenance_receipt] + else: + receipt = service.advance(start, stop, offered, budgets, states, jobs) + maintenance_receipt = None + receipts = [receipt] + activities = [] + for item in receipts: + activities.extend(row for row in energy_activities(item) + if row["operation"] != "erase") + mapped = energy.map(activities, + gpu_external_j=config.get("gpu_external_w", 0) * WINDOW_NS / 1e9) + if driver: + erase = _erase_energy(receipts, mapping, config["maintenance"]["block_bytes"], + config["maintenance"]["erase_energy_j_per_operation"]) + for component, joules in erase.items(): + mapped["component_energy_j"][component] = ( + mapped["component_energy_j"].get(component, 0.0) + joules) + mapped["scope_energy_j"]["erase:array"] = ( + mapped["scope_energy_j"].get("erase:array", 0.0) + joules) + mapped["total_j"] += joules + total_energy += mapped["total_j"] + heat = thermal.advance(start, stop, mapped["component_energy_j"]) + maintenance_delta = None + if driver: + entities = heat["entity_temperatures_k"] + for key, extent_ids in extents_by_channel.items(): + component = mapping[key[0]][key[1]] + current = float(entities[component]["hotspot_k"]) + midpoint = (previous_die_k[component] + current) / 2 + ledger.advance_temperature_many(extent_ids, start, stop, midpoint) + previous_die_k[component] = current + maintenance_delta = driver.consume_receipt( + maintenance_receipt if maintenance_receipt is not None else receipt) + maintenance_delta["reliability_events"] = ledger.drain_events() + for owner, temperature in heat["temperatures"].items(): + peak[owner] = max(peak.get(owner, temperature), temperature) + observed = {s: heat["stack_states"][s] for s in stacks} + next_budgets, decisions = {}, {} + for stack in stacks: + row = receipt["stacks"][stack] + delivered, backlog = row["delivered_effective_bytes"], row["backlog_effective_bytes"] + utilization = min(1.0, delivered / baseline[stack]) + due_bytes, due_deadline = _maintenance_pressure( + ledger, extents_by_stack.get(stack, ()), stop, + config["maintenance"]["block_bytes"]) + facts = StackWindowFacts( + stack_id=stack, offered_bytes=row["offered_effective_bytes"], + delivered_bytes=delivered, backlog_bytes=backlog, + oldest_wait_ns=row["oldest_wait_ns"] or 0, + latency_p95_ns=percentile(row["delivered_delay_histogram_bytes"]), + censored_requests=0, gate_limited=backlog > 0 and delivered >= budgets[stack], + backend_busy_fraction=utilization, resource_busy=utilization >= 1, + maintenance_due_bytes=due_bytes, + maintenance_earliest_deadline_ns=due_deadline) + decision = policies[stack].evaluate(WindowFacts( + start_ns=start, end_ns=stop, guard_state=observed[stack], stacks=(facts,), + current_budget_bytes={stack: budgets[stack]}, guard_states={stack: observed[stack]}, + hysteresis_budget_bytes={stack: heat["hysteresis_budget_bytes"][stack]})) + next_budgets[stack] = decision.stack_decisions[0].budget_bytes + decisions[stack] = asdict(decision) + if config.get("control_disabled", False): + next_budgets, next_states = dict(baseline), {s: "normal" for s in stacks} + else: + next_states = observed + sink.write(json.dumps({ + "start_ns": start, "end_ns": stop, + "service": _receipt_output_delta(receipt), + "independent_maintenance_service": ( + None if maintenance_receipt is None else + _receipt_output_delta(maintenance_receipt)), + "maintenance_delta": maintenance_delta, + "reliability": _compact_reliability(ledger, driver, stop), + "energy": mapped, "thermal": heat, + "control": {"budgets": budgets, "next_budgets": next_budgets, + "observed_states": observed, "decisions": decisions, + "facts": "MODELLED_FLUID_NOT_NATIVE_BUSY_OR_BACKEND_LATENCY"}, + }, separators=(",", ":"), allow_nan=False) + "\n") + budgets, states = next_budgets, next_states + final_receipt = receipt + delivered = sum(final_receipt["stacks"][s]["cumulative_delivered_effective_bytes"] + for s in hbf) + backlog = sum(final_receipt["stacks"][s]["backlog_effective_bytes"] for s in hbf) + if workload.total != delivered + backlog: + raise AssertionError("foreground effective-byte conservation failed") + thermal_energy = heat["energy_j"]["cumulative"]["total_input_j"] + if abs(thermal_energy - total_energy) > 1e-9 * max(1, total_energy): + raise AssertionError("activity-to-thermal energy mismatch") + final = None if driver is None else { + "driver": driver.snapshot(), "reliability": ledger.snapshot(), + "mode": maintenance_mode, + "independent_semantics": ("IDEAL_INDEPENDENT_AGGREGATE_RESOURCE_PROXY_NOT_SECOND_NATIVE_ENGINE" + if independent else None), + } + return ({"offered_bytes": workload.total, "delivered_bytes": delivered, + "backlog_bytes": backlog, "energy_j": total_energy, + "peak_k_by_stack": peak, "service_facts": service.immutable_facts(), + "maintenance_mode": maintenance_mode, + "maintenance_scope": "AGGREGATE_RATE_SERVICE_NOT_NATIVE_NAND", + "token_throughput": "UNAVAILABLE_RATE_WORKLOAD_HAS_NO_TOKEN_DEPENDENCY_DAG"}, final) + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + for name in ("config", "model-dir", "thermal-binary", "artifact-root", "output"): + parser.add_argument("--" + name, type=Path, required=True) + args = parser.parse_args() + args.output.mkdir(parents=True, exist_ok=False) + started = time.monotonic() + try: + for key in ("OMP_NUM_THREADS", "OPENBLAS_NUM_THREADS", "MKL_NUM_THREADS", + "NUMEXPR_NUM_THREADS"): + os.environ[key] = "1" + os.environ["CUDA_VISIBLE_DEVICES"] = "" + config = json.loads(args.config.read_text()) + save(args.output / "config.json", config) + manifest = {"started_utc": datetime.now(timezone.utc).isoformat(), + "environment_id": "eq3-thermal-cpu-v1", "python": sys.version, + "platform": platform.platform(), "input_sha256": digest(args.config), + "source_revision": subprocess.check_output( + ["git", "rev-parse", "HEAD"], cwd=ROOT, text=True).strip(), + "source_sha256": {str(path.relative_to(ROOT)): digest(path) for path in + (HERE / "run_maintenance_point.py", + HERE / "maintenance_driver.py", HERE / "reliability.py", + HERE / "topology_service.py", HERE / "energy.py")}, + "thermal_binary_sha256": digest(args.thermal_binary), + "model_dir": str(args.model_dir.resolve()), "cpu_threads": 1, + "gpu_count": 0, "resource_limits": config["resource_limits"]} + save(args.output / "manifest.json", manifest) + limit = config["resource_limits"]["address_space_gib"] * 1024**3 + resource.setrlimit(resource.RLIMIT_AS, (limit, limit)) + resource.setrlimit(resource.RLIMIT_CORE, (0, 0)) + normalized = json.loads((args.model_dir / "normalized.json").read_text()) + baseline = {s: sum(c.values()) * WINDOW_NS // 10**9 + for s, c in config["service"]["channels"].items()} + with ThermalService(args.thermal_binary, args.model_dir, args.output / "thermal-process", + artifact_root=args.artifact_root, baseline_budgets=baseline, + limits=config.get("thermal_limits_k")) as thermal: + manifest.update(thermal_model_lock=thermal.lock, thermal_header=thermal.header) + save(args.output / "manifest.json", manifest) + with (args.output / "windows.jsonl").open("w") as sink: + summary, final = execute(config, normalized, thermal, sink) + if final is not None: + save(args.output / "maintenance-final.json", final) + summary["maintenance_final_sha256"] = digest(args.output / "maintenance-final.json") + save(args.output / "DONE.json", {"status": "COMPLETED", "summary": summary, + "wall_s": time.monotonic() - started, + "child_peak_rss_kib": resource.getrusage(resource.RUSAGE_CHILDREN).ru_maxrss}) + except BaseException as error: + save(args.output / "FAILED.json", {"status": "FAILED", "error": repr(error), + "wall_s": time.monotonic() - started, "raw_preserved": True}) + raise + + +if __name__ == "__main__": + main() diff --git a/experiments/eq3_system_thermal/run_system_point.py b/experiments/eq3_system_thermal/run_system_point.py new file mode 100644 index 0000000..47a3ad5 --- /dev/null +++ b/experiments/eq3_system_thermal/run_system_point.py @@ -0,0 +1,218 @@ +#!/usr/bin/env python3 +"""Isolated conditional four-topology flow/energy/thermal feedback runner.""" +from __future__ import annotations +import argparse +from dataclasses import asdict +from datetime import datetime, timezone +import hashlib +import json +import os +from pathlib import Path +import platform +import resource +import subprocess +import sys +import time + +HERE = Path(__file__).resolve().parent +ROOT = HERE.parents[1] +MAINTENANCE = ROOT / 'experiments' / 'eq3_maintenance' +sys.path.insert(0, str(MAINTENANCE)) +from thermal_client import ThermalService +from read_rate_policy import EngineeringProfile, ReadRatePolicy, StackWindowFacts, WindowFacts +from energy import EnergyMapper +from rate_workload import RateWorkload +from topology_service import TopologyService + +WINDOW_NS = 20_000_000 + + +def save(path, obj): + Path(path).write_text(json.dumps(obj, indent=2, allow_nan=False) + '\n') + + +def digest(path): + return hashlib.sha256(Path(path).read_bytes()).hexdigest() + + +def energy_activities(service): + """Select disjoint scopes; fabric stage observations are not extra reads.""" + result = [] + for raw in service['activities']: + row = dict(raw) + phase = row['phase'] + if phase.startswith('media_'): + if row['stack'].startswith('hbm') and row['operation'] == 'read': + row['operation'] = 'hbm_read' + result.append(row) + elif phase == 'relay_receive': + row['operation'] = 'relay_receive' + result.append(row) + elif phase == 'partner_gpu_drain': + row['operation'] = 'relay_send' + result.append(row) + return result + + +def channel_map(normalized, config): + result = {} + for stack in config['fabric']['hbf']: + dies = sorted((r for r in normalized['components'] + if r['device_id'] == stack and r['role'] == 'array_die'), + key=lambda r: r['die_index']) + channels = sorted(config['channels'][stack], key=int) + if len(dies) != len(channels): + raise ValueError('explicit one-channel/one-die scenario requires matching geometry') + result[stack] = {c: d['id'] for c, d in zip(channels, dies)} + return result + + +def percentile(histogram, pct=95): + total = sum(r['bytes'] for r in histogram) + cursor = 0 + for row in sorted(histogram, key=lambda r: r['delay_ns']): + cursor += row['bytes'] + if cursor >= (total * pct + 99) // 100: + return row['delay_ns'] + return None + + +def execute(config, normalized, thermal, sink): + service = TopologyService(config['service']) + mapping = channel_map(normalized, config['service']) + workload = RateWorkload(mapping, config['workload']) + energy = EnergyMapper(normalized, mapping, config['energy']) + stacks = sorted(config['service']['channels']) + hbf = sorted(mapping) + baseline = {s: sum(config['service']['channels'][s].values()) * WINDOW_NS // 10**9 + for s in stacks} + budgets = dict(baseline) + states = {s: 'normal' for s in stacks} + profiles = {s: EngineeringProfile( + profile_id=config['point_id'] + ':' + s, enabled=True, + strategy=config['strategy'], window_ns=WINDOW_NS, + target_bytes_per_s=min(config['workload']['per_stack_Bps'], + sum(config['service']['channels'][s].values()) * 4 // 5), + step_bytes=baseline[s] // 20, minimum_budget_bytes=baseline[s] // 10, + maximum_budget_bytes=baseline[s], severe_budget_bytes=0, light_fraction=.5) + for s in stacks} + policies = {s: ReadRatePolicy(p) for s, p in profiles.items()} + total_energy = 0.0 + peak = {} + first = {} + state_duration = {s: {v: 0 for v in ('normal','light','severe','shutdown')} for s in stacks} + end = config['workload']['active_ns'] + config['recovery_ns'] + if end % WINDOW_NS: + raise ValueError('run duration must align with thermal window') + for start in range(0, end, WINDOW_NS): + stop = start + WINDOW_NS + offered = workload.advance(start, stop) + receipt = service.advance(start, stop, offered, budgets, states) + mapped = energy.map(energy_activities(receipt), + gpu_external_j=config.get('gpu_external_w', 0) * WINDOW_NS / 1e9) + total_energy += mapped['total_j'] + heat = thermal.advance(start, stop, mapped['component_energy_j']) + for owner, temperature in heat['temperatures'].items(): + peak[owner] = max(peak.get(owner, temperature), temperature) + thresholds = thermal.limits['gpu' if owner == 'gpu' else owner[:3]] + for label, limit in zip(('light','severe','shutdown'), thresholds): + if temperature >= limit: + first.setdefault(owner + ':' + label, stop) + observed = {s: heat['stack_states'][s] for s in stacks} + next_budgets = {} + decisions = {} + for s in stacks: + r = receipt['stacks'][s] + delivered = r['delivered_effective_bytes'] + backlog = r['backlog_effective_bytes'] + utilization = min(1.0, delivered / baseline[s]) + facts = StackWindowFacts( + stack_id=s, offered_bytes=r['offered_effective_bytes'], delivered_bytes=delivered, + backlog_bytes=backlog, oldest_wait_ns=r['oldest_wait_ns'] or 0, + latency_p95_ns=percentile(r['delivered_delay_histogram_bytes']), + censored_requests=0, gate_limited=backlog > 0 and delivered >= budgets[s], + backend_busy_fraction=utilization, resource_busy=utilization >= 1) + decision = policies[s].evaluate(WindowFacts( + start_ns=start, end_ns=stop, guard_state=observed[s], stacks=(facts,), + current_budget_bytes={s: budgets[s]}, guard_states={s: observed[s]}, + hysteresis_budget_bytes={s: heat['hysteresis_budget_bytes'][s]})) + next_budgets[s] = decision.stack_decisions[0].budget_bytes + decisions[s] = asdict(decision) + state_duration[s][observed[s]] += WINDOW_NS + if config.get('control_disabled', False): + next_budgets = dict(baseline) + next_states = {s: 'normal' for s in stacks} + else: + next_states = observed + row = {'start_ns': start, 'end_ns': stop, 'service': receipt, 'energy': mapped, + 'thermal': heat, 'control': {'budgets': budgets, 'next_budgets': next_budgets, + 'observed_states': observed, 'decisions': decisions, + 'facts': 'MODELLED_FLUID_NOT_NATIVE_BUSY_OR_BACKEND_LATENCY'}, + 'reliability': {'status': 'NO_MAINTENANCE_DEMAND_IN_BASE_RATE_WORKLOAD'}, + 'causal': None} + sink.write(json.dumps(row, separators=(',', ':'), allow_nan=False) + '\n') + budgets, states = next_budgets, next_states + delivered = sum(receipt['stacks'][s]['cumulative_delivered_effective_bytes'] for s in hbf) + backlog = sum(receipt['stacks'][s]['backlog_effective_bytes'] for s in hbf) + if workload.total != delivered + backlog: + raise AssertionError('end-to-end effective-byte conservation failed') + thermal_energy = heat['energy_j']['cumulative']['total_input_j'] + if abs(thermal_energy-total_energy) > 1e-9 * max(1, total_energy): + raise AssertionError('activity-to-thermal energy mismatch') + return {'offered_bytes': workload.total, 'delivered_bytes': delivered, + 'backlog_bytes': backlog, 'energy_j': total_energy, 'peak_k_by_stack': peak, + 'first_threshold_ns': first, 'state_duration_ns': state_duration, + 'final_stacks': receipt['stacks'], 'thermal_energy_receipt': heat['energy_j']['cumulative'], + 'service_facts': service.immutable_facts(), + 'token_throughput': 'UNAVAILABLE_RATE_WORKLOAD_HAS_NO_TOKEN_DEPENDENCY_DAG', + 'external_gddr_temperature': 'UNAVAILABLE_OUTSIDE_PACKAGE_DOMAIN', + 'scope': 'CONDITIONAL_SIMULATED_NOT_MQSIM_TBPS_OR_CALIBRATED_HARDWARE'} + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + for name in ('config', 'model-dir', 'thermal-binary', 'artifact-root', 'output'): + parser.add_argument('--' + name, type=Path, required=True) + args = parser.parse_args() + args.output.mkdir(parents=True, exist_ok=False) + started = time.monotonic() + try: + for key in ('OMP_NUM_THREADS','OPENBLAS_NUM_THREADS','MKL_NUM_THREADS','NUMEXPR_NUM_THREADS'): + os.environ[key] = '1' + os.environ['CUDA_VISIBLE_DEVICES'] = '' + config = json.loads(args.config.read_text()) + save(args.output / 'config.json', config) + manifest = {'started_utc': datetime.now(timezone.utc).isoformat(), + 'environment_id': 'eq3-thermal-cpu-v1', 'python':sys.version, + 'platform':platform.platform(), 'input_sha256':digest(args.config), + 'source_sha256':{str(p.relative_to(ROOT)):digest(p) for p in + [HERE/name for name in ('run_system_point.py','energy.py','rate_workload.py','topology_service.py')] + + [MAINTENANCE/'thermal_client.py',MAINTENANCE/'read_rate_policy.py',ROOT/'tools'/'eq3_basic_fabric.py']}, + 'thermal_binary_sha256':digest(args.thermal_binary), + 'source_revision':subprocess.check_output(['git','rev-parse','HEAD'],cwd=ROOT,text=True).strip(), + 'model_dir':str(args.model_dir.resolve()), 'cpu_threads':1,'gpu_count':0, + 'resource_limits':config['resource_limits']} + save(args.output / 'manifest.json', manifest) + limit = config['resource_limits']['address_space_gib'] * 1024**3 + resource.setrlimit(resource.RLIMIT_AS, (limit,limit)) + resource.setrlimit(resource.RLIMIT_CORE, (0,0)) + normalized = json.loads((args.model_dir/'normalized.json').read_text()) + baseline = {s:sum(c.values())*WINDOW_NS//10**9 for s,c in config['service']['channels'].items()} + with ThermalService(args.thermal_binary,args.model_dir,args.output/'thermal-process', + artifact_root=args.artifact_root,baseline_budgets=baseline, + limits=config.get('thermal_limits_k')) as thermal: + manifest.update(thermal_model_lock=thermal.lock,thermal_header=thermal.header) + save(args.output/'manifest.json',manifest) + with (args.output/'windows.jsonl').open('w') as sink: + summary = execute(config,normalized,thermal,sink) + save(args.output/'DONE.json',{'status':'COMPLETED','summary':summary, + 'wall_s':time.monotonic()-started, + 'child_peak_rss_kib':resource.getrusage(resource.RUSAGE_CHILDREN).ru_maxrss}) + except BaseException as exc: + save(args.output/'FAILED.json',{'status':'FAILED','error':repr(exc), + 'wall_s':time.monotonic()-started,'raw_preserved':True}) + raise + + +if __name__ == '__main__': + main() diff --git a/experiments/eq3_system_thermal/stage/traces/tiny_qwen2_cpu_trace_v1/CONSUMER_EVIDENCE.json b/experiments/eq3_system_thermal/stage/traces/tiny_qwen2_cpu_trace_v1/CONSUMER_EVIDENCE.json new file mode 100644 index 0000000..75a433e --- /dev/null +++ b/experiments/eq3_system_thermal/stage/traces/tiny_qwen2_cpu_trace_v1/CONSUMER_EVIDENCE.json @@ -0,0 +1,56 @@ +{ + "compute_timing_semantics": "EXPLICIT_SCENARIO_NS_NOT_TINY_CPU_RUNTIME", + "consumer": "causal_workload.build_architecture_trace(dependency_mode=tiny_cpu_template)", + "fixed_tests": { + "count": 24, + "result": "PASS", + "thermal_or_native_runs": 0 + }, + "limitations": [ + "STRUCTURE_AND_WEIGHT_ACCESS_ORDER_TEMPLATE_ONLY", + "SYNTHETIC_MODE_REMAINS_AVAILABLE", + "NO_PRETRAINED_QUALITY", + "NO_GPU_TIMING", + "NO_TOKEN_PERFORMANCE_CALIBRATION" + ], + "models": { + "Qwen/Qwen2.5-72B-Instruct": { + "analytical_output_head_macs_per_token": 1245708288, + "analytical_total_macs_per_token_at_context": 76827066368, + "analytical_transformer_macs_per_token_at_context": 75581358080, + "first_attention_template_op_ids": [ + "prefill:op1:layer0.input_rmsnorm", + "prefill:op2:layer0.q_projection", + "prefill:op3:layer0.k_projection", + "prefill:op4:layer0.v_projection", + "prefill:op5:layer0.rope_gqa_causal_attention", + "prefill:op6:layer0.o_projection" + ], + "layer_count": 80, + "logical_region_count": 163, + "task_count_per_batch": 324, + "tensor_payload_bytes": 145412407296 + }, + "Qwen/Qwen2.5-7B-Instruct": { + "analytical_output_head_macs_per_token": 544997376, + "analytical_total_macs_per_token_at_context": 7892369408, + "analytical_transformer_macs_per_token_at_context": 7347372032, + "first_attention_template_op_ids": [ + "prefill:op1:layer0.input_rmsnorm", + "prefill:op2:layer0.q_projection", + "prefill:op3:layer0.k_projection", + "prefill:op4:layer0.v_projection", + "prefill:op5:layer0.rope_gqa_causal_attention", + "prefill:op6:layer0.o_projection" + ], + "layer_count": 28, + "logical_region_count": 59, + "task_count_per_batch": 116, + "tensor_payload_bytes": 15231233024 + } + }, + "projection_context_tokens": 4096, + "schema_version": "eq3-tiny-trace-causal-consumer-evidence-v1", + "status": "FIXED_TEST_PASS", + "tiny_trace_sha256": "feea4657abc5e7b981208b23b8d2fff9836194694c5c3dec69c6bf59bbe627a7" +} diff --git a/experiments/eq3_system_thermal/stage/traces/tiny_qwen2_cpu_trace_v1/README.md b/experiments/eq3_system_thermal/stage/traces/tiny_qwen2_cpu_trace_v1/README.md new file mode 100644 index 0000000..8be49c6 --- /dev/null +++ b/experiments/eq3_system_thermal/stage/traces/tiny_qwen2_cpu_trace_v1/README.md @@ -0,0 +1,43 @@ +# Tiny Qwen2-style CPU trace + +This artifact executes a deterministic NumPy forward with random tiny weights: +hidden size 56, 28 query heads, 4 KV heads, head dimension 2, intermediate +size 128, 28 layers, four-token prefill, and two single-token decode steps. + +It records actual CPU array access names, order, shapes and bytes, plus explicit +operation dependency IDs. It is a structural trace derived from a tiny +Qwen2-style decoder. It is not a pretrained model, native GPU/framework trace, +GPU timing result, or token-performance calibration. + +The same output separately regenerates Qwen2.5-7B/72B logical region bytes, +1 MiB scenario addresses, and analytical dense MAC counts from the registered +official metadata. Those projections do not inherit the tiny CPU runtime. + +Reproduce from `experiments/eq3_system_thermal` with BLAS thread counts set to +one: + +```text +python3 tiny_cpu_trace.py \ + --config stage/traces/tiny_qwen2_cpu_trace_v1/config.json \ + --output stage/traces/tiny_qwen2_cpu_trace_v1/trace.json +``` + +## Optional causal consumer + +`causal_workload.build_architecture_trace` keeps +`dependency_mode=synthetic_metadata_dag` as the existing mode. The optional +`dependency_mode=tiny_cpu_template` additionally requires this artifact's +repository-relative `tiny_trace_path`, exact `tiny_trace_sha256`, and a positive +`projection_context_tokens`. + +The consumer recomputes and validates the artifact checksum, causal dependency +order, actual float32 weight-access shapes/bytes, attention-to-MLP-to-head +pattern, and the selected target model's layer count, BF16 tensor shapes, +1 MiB scenario addresses, payload bytes and analytical MAC count. Generated +storage and compute tasks carry the observed template operation IDs. Their +`duration_ns` values still come only from explicit scenario inputs; the tiny +CPU runtime is never transferred to 7B/72B timing. + +`CONSUMER_EVIDENCE.json` records the fixed-test consumer output for both +registered Qwen2.5 targets. It is software evidence, not a thermal or native +backend result. diff --git a/experiments/eq3_system_thermal/stage/traces/tiny_qwen2_cpu_trace_v1/RUN_RECEIPT.json b/experiments/eq3_system_thermal/stage/traces/tiny_qwen2_cpu_trace_v1/RUN_RECEIPT.json new file mode 100644 index 0000000..c695ab3 --- /dev/null +++ b/experiments/eq3_system_thermal/stage/traces/tiny_qwen2_cpu_trace_v1/RUN_RECEIPT.json @@ -0,0 +1,18 @@ +{ + "classification": "FIXED_STRUCTURAL_CPU_TRACE_NOT_EXPERIMENT", + "command": "OMP/OPENBLAS/MKL/BLIS/VECLIB/NUMEXPR=1 python3 tiny_cpu_trace.py --config .../config.json --output .../trace.json", + "config_file_sha256": "24ae28153b10a8fb36492a1f4a641ff0664941307f2d1c812162f7c17a141bed", + "cpu_threads_requested": 1, + "limitations": [ + "RANDOM_TINY_WEIGHTS", + "NO_PRETRAINED_QUALITY", + "NO_GPU_TIMING", + "NO_TOKEN_PERFORMANCE_CALIBRATION" + ], + "max_rss_kib": 46460, + "schema_version": "eq3-tiny-qwen2-cpu-trace-run-receipt-v1", + "status": "PASS", + "trace_file_sha256": "c3019ec27d1f35ca74b7844451fef9b4f8f97094ef14f673ddabed9dbdbe0592", + "trace_internal_sha256": "feea4657abc5e7b981208b23b8d2fff9836194694c5c3dec69c6bf59bbe627a7", + "wall_seconds": 0.11 +} diff --git a/experiments/eq3_system_thermal/stage/traces/tiny_qwen2_cpu_trace_v1/config.json b/experiments/eq3_system_thermal/stage/traces/tiny_qwen2_cpu_trace_v1/config.json new file mode 100644 index 0000000..8bdc6a8 --- /dev/null +++ b/experiments/eq3_system_thermal/stage/traces/tiny_qwen2_cpu_trace_v1/config.json @@ -0,0 +1,15 @@ +{ + "decode_tokens": 2, + "hidden_size": 56, + "intermediate_size": 128, + "layers": 28, + "num_attention_heads": 28, + "num_key_value_heads": 4, + "projection_context_tokens": 4096, + "prompt_token_ids": [1, 7, 11, 19], + "rms_norm_eps": 1e-06, + "rope_theta": 10000.0, + "schema_version": "eq3-tiny-qwen2-cpu-trace-config-v1", + "seed": 20260920, + "vocab_size": 128 +} diff --git a/experiments/eq3_system_thermal/stage/traces/tiny_qwen2_cpu_trace_v1/trace.json b/experiments/eq3_system_thermal/stage/traces/tiny_qwen2_cpu_trace_v1/trace.json new file mode 100644 index 0000000..ee0da48 --- /dev/null +++ b/experiments/eq3_system_thermal/stage/traces/tiny_qwen2_cpu_trace_v1/trace.json @@ -0,0 +1,28517 @@ +{ + "actual_access_bytes": 9783648, + "actual_cpu_weight_storage_bytes": 3289440, + "blas_threads_requested": 1, + "claims_excluded": [ + "PRETRAINED_MODEL_QUALITY", + "NATIVE_GPU_TRACE", + "GPU_TIMING", + "TOKEN_PERFORMANCE_CALIBRATION" + ], + "classification": "TRACE_DERIVED_TINY_RANDOM_WEIGHT_CPU_FORWARD", + "config": { + "decode_tokens": 2, + "head_dim": 2, + "hidden_size": 56, + "intermediate_size": 128, + "layers": 28, + "num_attention_heads": 28, + "num_key_value_heads": 4, + "projection_context_tokens": 4096, + "prompt_token_ids": [ + 1, + 7, + 11, + 19 + ], + "rms_norm_eps": 1e-06, + "rope_theta": 10000.0, + "schema_version": "eq3-tiny-qwen2-cpu-trace-config-v1", + "seed": 20260920, + "vocab_size": 128 + }, + "config_sha256": "831f6f8ab9ffb1ff3253970b64977720da1b1b949ff7a1c0ca8de230bfe2b0b0", + "forwards": [ + { + "context_tokens_after": 4, + "forward_id": "prefill", + "input_tokens": [ + 1, + 7, + 11, + 19, + 90, + 83 + ], + "logits_sha256": "8b5fe0ea0ba7d60eeb7a1e58478e82bda178c8c8ed3a2fe8c3c19ea4733ee296", + "terminal_op_id": "prefill:op282:lm_head" + }, + { + "context_tokens_after": 5, + "forward_id": "decode0", + "input_tokens": [ + 90 + ], + "logits_sha256": "d620feb3d3ecf30f0454b6dd398bb028837507e4534e4c604604b5622298873d", + "terminal_op_id": "decode0:op565:lm_head" + }, + { + "context_tokens_after": 6, + "forward_id": "decode1", + "input_tokens": [ + 83 + ], + "logits_sha256": "ff371970394020fbe422de6bbbd4e74cce2dbada968cbce4c489c9fd849ca721", + "terminal_op_id": "decode1:op848:lm_head" + } + ], + "generated_tokens": [ + 1, + 7, + 11, + 19, + 90, + 83 + ], + "numpy_version": "2.3.5", + "operations": [ + { + "depends_on": [ + "token_ids" + ], + "details": { + "token_count": 4 + }, + "forward_id": "prefill", + "name": "embedding_lookup", + "op_id": "prefill:op0:embedding_lookup", + "sequence": 0 + }, + { + "depends_on": [ + "prefill:op0:embedding_lookup" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer0.input_rmsnorm", + "op_id": "prefill:op1:layer0.input_rmsnorm", + "sequence": 1 + }, + { + "depends_on": [ + "prefill:op1:layer0.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer0.q_projection", + "op_id": "prefill:op2:layer0.q_projection", + "sequence": 2 + }, + { + "depends_on": [ + "prefill:op1:layer0.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer0.k_projection", + "op_id": "prefill:op3:layer0.k_projection", + "sequence": 3 + }, + { + "depends_on": [ + "prefill:op1:layer0.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer0.v_projection", + "op_id": "prefill:op4:layer0.v_projection", + "sequence": 4 + }, + { + "depends_on": [ + "prefill:op2:layer0.q_projection", + "prefill:op3:layer0.k_projection", + "prefill:op4:layer0.v_projection" + ], + "details": { + "context_tokens": 4, + "kv_repeat_groups": 7, + "kv_shape": [ + 4, + 4, + 2 + ], + "q_shape": [ + 4, + 28, + 2 + ] + }, + "forward_id": "prefill", + "name": "layer0.rope_gqa_causal_attention", + "op_id": "prefill:op5:layer0.rope_gqa_causal_attention", + "sequence": 5 + }, + { + "depends_on": [ + "prefill:op5:layer0.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer0.o_projection", + "op_id": "prefill:op6:layer0.o_projection", + "sequence": 6 + }, + { + "depends_on": [ + "prefill:op6:layer0.o_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer0.post_attention_rmsnorm", + "op_id": "prefill:op7:layer0.post_attention_rmsnorm", + "sequence": 7 + }, + { + "depends_on": [ + "prefill:op7:layer0.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer0.gate_projection", + "op_id": "prefill:op8:layer0.gate_projection", + "sequence": 8 + }, + { + "depends_on": [ + "prefill:op7:layer0.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer0.up_projection", + "op_id": "prefill:op9:layer0.up_projection", + "sequence": 9 + }, + { + "depends_on": [ + "prefill:op8:layer0.gate_projection", + "prefill:op9:layer0.up_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer0.swiglu_down_projection", + "op_id": "prefill:op10:layer0.swiglu_down_projection", + "sequence": 10 + }, + { + "depends_on": [ + "prefill:op10:layer0.swiglu_down_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer1.input_rmsnorm", + "op_id": "prefill:op11:layer1.input_rmsnorm", + "sequence": 11 + }, + { + "depends_on": [ + "prefill:op11:layer1.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer1.q_projection", + "op_id": "prefill:op12:layer1.q_projection", + "sequence": 12 + }, + { + "depends_on": [ + "prefill:op11:layer1.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer1.k_projection", + "op_id": "prefill:op13:layer1.k_projection", + "sequence": 13 + }, + { + "depends_on": [ + "prefill:op11:layer1.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer1.v_projection", + "op_id": "prefill:op14:layer1.v_projection", + "sequence": 14 + }, + { + "depends_on": [ + "prefill:op12:layer1.q_projection", + "prefill:op13:layer1.k_projection", + "prefill:op14:layer1.v_projection" + ], + "details": { + "context_tokens": 4, + "kv_repeat_groups": 7, + "kv_shape": [ + 4, + 4, + 2 + ], + "q_shape": [ + 4, + 28, + 2 + ] + }, + "forward_id": "prefill", + "name": "layer1.rope_gqa_causal_attention", + "op_id": "prefill:op15:layer1.rope_gqa_causal_attention", + "sequence": 15 + }, + { + "depends_on": [ + "prefill:op15:layer1.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer1.o_projection", + "op_id": "prefill:op16:layer1.o_projection", + "sequence": 16 + }, + { + "depends_on": [ + "prefill:op16:layer1.o_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer1.post_attention_rmsnorm", + "op_id": "prefill:op17:layer1.post_attention_rmsnorm", + "sequence": 17 + }, + { + "depends_on": [ + "prefill:op17:layer1.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer1.gate_projection", + "op_id": "prefill:op18:layer1.gate_projection", + "sequence": 18 + }, + { + "depends_on": [ + "prefill:op17:layer1.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer1.up_projection", + "op_id": "prefill:op19:layer1.up_projection", + "sequence": 19 + }, + { + "depends_on": [ + "prefill:op18:layer1.gate_projection", + "prefill:op19:layer1.up_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer1.swiglu_down_projection", + "op_id": "prefill:op20:layer1.swiglu_down_projection", + "sequence": 20 + }, + { + "depends_on": [ + "prefill:op20:layer1.swiglu_down_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer2.input_rmsnorm", + "op_id": "prefill:op21:layer2.input_rmsnorm", + "sequence": 21 + }, + { + "depends_on": [ + "prefill:op21:layer2.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer2.q_projection", + "op_id": "prefill:op22:layer2.q_projection", + "sequence": 22 + }, + { + "depends_on": [ + "prefill:op21:layer2.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer2.k_projection", + "op_id": "prefill:op23:layer2.k_projection", + "sequence": 23 + }, + { + "depends_on": [ + "prefill:op21:layer2.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer2.v_projection", + "op_id": "prefill:op24:layer2.v_projection", + "sequence": 24 + }, + { + "depends_on": [ + "prefill:op22:layer2.q_projection", + "prefill:op23:layer2.k_projection", + "prefill:op24:layer2.v_projection" + ], + "details": { + "context_tokens": 4, + "kv_repeat_groups": 7, + "kv_shape": [ + 4, + 4, + 2 + ], + "q_shape": [ + 4, + 28, + 2 + ] + }, + "forward_id": "prefill", + "name": "layer2.rope_gqa_causal_attention", + "op_id": "prefill:op25:layer2.rope_gqa_causal_attention", + "sequence": 25 + }, + { + "depends_on": [ + "prefill:op25:layer2.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer2.o_projection", + "op_id": "prefill:op26:layer2.o_projection", + "sequence": 26 + }, + { + "depends_on": [ + "prefill:op26:layer2.o_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer2.post_attention_rmsnorm", + "op_id": "prefill:op27:layer2.post_attention_rmsnorm", + "sequence": 27 + }, + { + "depends_on": [ + "prefill:op27:layer2.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer2.gate_projection", + "op_id": "prefill:op28:layer2.gate_projection", + "sequence": 28 + }, + { + "depends_on": [ + "prefill:op27:layer2.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer2.up_projection", + "op_id": "prefill:op29:layer2.up_projection", + "sequence": 29 + }, + { + "depends_on": [ + "prefill:op28:layer2.gate_projection", + "prefill:op29:layer2.up_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer2.swiglu_down_projection", + "op_id": "prefill:op30:layer2.swiglu_down_projection", + "sequence": 30 + }, + { + "depends_on": [ + "prefill:op30:layer2.swiglu_down_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer3.input_rmsnorm", + "op_id": "prefill:op31:layer3.input_rmsnorm", + "sequence": 31 + }, + { + "depends_on": [ + "prefill:op31:layer3.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer3.q_projection", + "op_id": "prefill:op32:layer3.q_projection", + "sequence": 32 + }, + { + "depends_on": [ + "prefill:op31:layer3.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer3.k_projection", + "op_id": "prefill:op33:layer3.k_projection", + "sequence": 33 + }, + { + "depends_on": [ + "prefill:op31:layer3.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer3.v_projection", + "op_id": "prefill:op34:layer3.v_projection", + "sequence": 34 + }, + { + "depends_on": [ + "prefill:op32:layer3.q_projection", + "prefill:op33:layer3.k_projection", + "prefill:op34:layer3.v_projection" + ], + "details": { + "context_tokens": 4, + "kv_repeat_groups": 7, + "kv_shape": [ + 4, + 4, + 2 + ], + "q_shape": [ + 4, + 28, + 2 + ] + }, + "forward_id": "prefill", + "name": "layer3.rope_gqa_causal_attention", + "op_id": "prefill:op35:layer3.rope_gqa_causal_attention", + "sequence": 35 + }, + { + "depends_on": [ + "prefill:op35:layer3.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer3.o_projection", + "op_id": "prefill:op36:layer3.o_projection", + "sequence": 36 + }, + { + "depends_on": [ + "prefill:op36:layer3.o_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer3.post_attention_rmsnorm", + "op_id": "prefill:op37:layer3.post_attention_rmsnorm", + "sequence": 37 + }, + { + "depends_on": [ + "prefill:op37:layer3.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer3.gate_projection", + "op_id": "prefill:op38:layer3.gate_projection", + "sequence": 38 + }, + { + "depends_on": [ + "prefill:op37:layer3.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer3.up_projection", + "op_id": "prefill:op39:layer3.up_projection", + "sequence": 39 + }, + { + "depends_on": [ + "prefill:op38:layer3.gate_projection", + "prefill:op39:layer3.up_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer3.swiglu_down_projection", + "op_id": "prefill:op40:layer3.swiglu_down_projection", + "sequence": 40 + }, + { + "depends_on": [ + "prefill:op40:layer3.swiglu_down_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer4.input_rmsnorm", + "op_id": "prefill:op41:layer4.input_rmsnorm", + "sequence": 41 + }, + { + "depends_on": [ + "prefill:op41:layer4.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer4.q_projection", + "op_id": "prefill:op42:layer4.q_projection", + "sequence": 42 + }, + { + "depends_on": [ + "prefill:op41:layer4.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer4.k_projection", + "op_id": "prefill:op43:layer4.k_projection", + "sequence": 43 + }, + { + "depends_on": [ + "prefill:op41:layer4.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer4.v_projection", + "op_id": "prefill:op44:layer4.v_projection", + "sequence": 44 + }, + { + "depends_on": [ + "prefill:op42:layer4.q_projection", + "prefill:op43:layer4.k_projection", + "prefill:op44:layer4.v_projection" + ], + "details": { + "context_tokens": 4, + "kv_repeat_groups": 7, + "kv_shape": [ + 4, + 4, + 2 + ], + "q_shape": [ + 4, + 28, + 2 + ] + }, + "forward_id": "prefill", + "name": "layer4.rope_gqa_causal_attention", + "op_id": "prefill:op45:layer4.rope_gqa_causal_attention", + "sequence": 45 + }, + { + "depends_on": [ + "prefill:op45:layer4.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer4.o_projection", + "op_id": "prefill:op46:layer4.o_projection", + "sequence": 46 + }, + { + "depends_on": [ + "prefill:op46:layer4.o_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer4.post_attention_rmsnorm", + "op_id": "prefill:op47:layer4.post_attention_rmsnorm", + "sequence": 47 + }, + { + "depends_on": [ + "prefill:op47:layer4.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer4.gate_projection", + "op_id": "prefill:op48:layer4.gate_projection", + "sequence": 48 + }, + { + "depends_on": [ + "prefill:op47:layer4.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer4.up_projection", + "op_id": "prefill:op49:layer4.up_projection", + "sequence": 49 + }, + { + "depends_on": [ + "prefill:op48:layer4.gate_projection", + "prefill:op49:layer4.up_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer4.swiglu_down_projection", + "op_id": "prefill:op50:layer4.swiglu_down_projection", + "sequence": 50 + }, + { + "depends_on": [ + "prefill:op50:layer4.swiglu_down_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer5.input_rmsnorm", + "op_id": "prefill:op51:layer5.input_rmsnorm", + "sequence": 51 + }, + { + "depends_on": [ + "prefill:op51:layer5.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer5.q_projection", + "op_id": "prefill:op52:layer5.q_projection", + "sequence": 52 + }, + { + "depends_on": [ + "prefill:op51:layer5.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer5.k_projection", + "op_id": "prefill:op53:layer5.k_projection", + "sequence": 53 + }, + { + "depends_on": [ + "prefill:op51:layer5.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer5.v_projection", + "op_id": "prefill:op54:layer5.v_projection", + "sequence": 54 + }, + { + "depends_on": [ + "prefill:op52:layer5.q_projection", + "prefill:op53:layer5.k_projection", + "prefill:op54:layer5.v_projection" + ], + "details": { + "context_tokens": 4, + "kv_repeat_groups": 7, + "kv_shape": [ + 4, + 4, + 2 + ], + "q_shape": [ + 4, + 28, + 2 + ] + }, + "forward_id": "prefill", + "name": "layer5.rope_gqa_causal_attention", + "op_id": "prefill:op55:layer5.rope_gqa_causal_attention", + "sequence": 55 + }, + { + "depends_on": [ + "prefill:op55:layer5.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer5.o_projection", + "op_id": "prefill:op56:layer5.o_projection", + "sequence": 56 + }, + { + "depends_on": [ + "prefill:op56:layer5.o_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer5.post_attention_rmsnorm", + "op_id": "prefill:op57:layer5.post_attention_rmsnorm", + "sequence": 57 + }, + { + "depends_on": [ + "prefill:op57:layer5.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer5.gate_projection", + "op_id": "prefill:op58:layer5.gate_projection", + "sequence": 58 + }, + { + "depends_on": [ + "prefill:op57:layer5.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer5.up_projection", + "op_id": "prefill:op59:layer5.up_projection", + "sequence": 59 + }, + { + "depends_on": [ + "prefill:op58:layer5.gate_projection", + "prefill:op59:layer5.up_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer5.swiglu_down_projection", + "op_id": "prefill:op60:layer5.swiglu_down_projection", + "sequence": 60 + }, + { + "depends_on": [ + "prefill:op60:layer5.swiglu_down_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer6.input_rmsnorm", + "op_id": "prefill:op61:layer6.input_rmsnorm", + "sequence": 61 + }, + { + "depends_on": [ + "prefill:op61:layer6.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer6.q_projection", + "op_id": "prefill:op62:layer6.q_projection", + "sequence": 62 + }, + { + "depends_on": [ + "prefill:op61:layer6.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer6.k_projection", + "op_id": "prefill:op63:layer6.k_projection", + "sequence": 63 + }, + { + "depends_on": [ + "prefill:op61:layer6.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer6.v_projection", + "op_id": "prefill:op64:layer6.v_projection", + "sequence": 64 + }, + { + "depends_on": [ + "prefill:op62:layer6.q_projection", + "prefill:op63:layer6.k_projection", + "prefill:op64:layer6.v_projection" + ], + "details": { + "context_tokens": 4, + "kv_repeat_groups": 7, + "kv_shape": [ + 4, + 4, + 2 + ], + "q_shape": [ + 4, + 28, + 2 + ] + }, + "forward_id": "prefill", + "name": "layer6.rope_gqa_causal_attention", + "op_id": "prefill:op65:layer6.rope_gqa_causal_attention", + "sequence": 65 + }, + { + "depends_on": [ + "prefill:op65:layer6.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer6.o_projection", + "op_id": "prefill:op66:layer6.o_projection", + "sequence": 66 + }, + { + "depends_on": [ + "prefill:op66:layer6.o_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer6.post_attention_rmsnorm", + "op_id": "prefill:op67:layer6.post_attention_rmsnorm", + "sequence": 67 + }, + { + "depends_on": [ + "prefill:op67:layer6.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer6.gate_projection", + "op_id": "prefill:op68:layer6.gate_projection", + "sequence": 68 + }, + { + "depends_on": [ + "prefill:op67:layer6.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer6.up_projection", + "op_id": "prefill:op69:layer6.up_projection", + "sequence": 69 + }, + { + "depends_on": [ + "prefill:op68:layer6.gate_projection", + "prefill:op69:layer6.up_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer6.swiglu_down_projection", + "op_id": "prefill:op70:layer6.swiglu_down_projection", + "sequence": 70 + }, + { + "depends_on": [ + "prefill:op70:layer6.swiglu_down_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer7.input_rmsnorm", + "op_id": "prefill:op71:layer7.input_rmsnorm", + "sequence": 71 + }, + { + "depends_on": [ + "prefill:op71:layer7.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer7.q_projection", + "op_id": "prefill:op72:layer7.q_projection", + "sequence": 72 + }, + { + "depends_on": [ + "prefill:op71:layer7.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer7.k_projection", + "op_id": "prefill:op73:layer7.k_projection", + "sequence": 73 + }, + { + "depends_on": [ + "prefill:op71:layer7.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer7.v_projection", + "op_id": "prefill:op74:layer7.v_projection", + "sequence": 74 + }, + { + "depends_on": [ + "prefill:op72:layer7.q_projection", + "prefill:op73:layer7.k_projection", + "prefill:op74:layer7.v_projection" + ], + "details": { + "context_tokens": 4, + "kv_repeat_groups": 7, + "kv_shape": [ + 4, + 4, + 2 + ], + "q_shape": [ + 4, + 28, + 2 + ] + }, + "forward_id": "prefill", + "name": "layer7.rope_gqa_causal_attention", + "op_id": "prefill:op75:layer7.rope_gqa_causal_attention", + "sequence": 75 + }, + { + "depends_on": [ + "prefill:op75:layer7.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer7.o_projection", + "op_id": "prefill:op76:layer7.o_projection", + "sequence": 76 + }, + { + "depends_on": [ + "prefill:op76:layer7.o_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer7.post_attention_rmsnorm", + "op_id": "prefill:op77:layer7.post_attention_rmsnorm", + "sequence": 77 + }, + { + "depends_on": [ + "prefill:op77:layer7.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer7.gate_projection", + "op_id": "prefill:op78:layer7.gate_projection", + "sequence": 78 + }, + { + "depends_on": [ + "prefill:op77:layer7.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer7.up_projection", + "op_id": "prefill:op79:layer7.up_projection", + "sequence": 79 + }, + { + "depends_on": [ + "prefill:op78:layer7.gate_projection", + "prefill:op79:layer7.up_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer7.swiglu_down_projection", + "op_id": "prefill:op80:layer7.swiglu_down_projection", + "sequence": 80 + }, + { + "depends_on": [ + "prefill:op80:layer7.swiglu_down_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer8.input_rmsnorm", + "op_id": "prefill:op81:layer8.input_rmsnorm", + "sequence": 81 + }, + { + "depends_on": [ + "prefill:op81:layer8.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer8.q_projection", + "op_id": "prefill:op82:layer8.q_projection", + "sequence": 82 + }, + { + "depends_on": [ + "prefill:op81:layer8.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer8.k_projection", + "op_id": "prefill:op83:layer8.k_projection", + "sequence": 83 + }, + { + "depends_on": [ + "prefill:op81:layer8.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer8.v_projection", + "op_id": "prefill:op84:layer8.v_projection", + "sequence": 84 + }, + { + "depends_on": [ + "prefill:op82:layer8.q_projection", + "prefill:op83:layer8.k_projection", + "prefill:op84:layer8.v_projection" + ], + "details": { + "context_tokens": 4, + "kv_repeat_groups": 7, + "kv_shape": [ + 4, + 4, + 2 + ], + "q_shape": [ + 4, + 28, + 2 + ] + }, + "forward_id": "prefill", + "name": "layer8.rope_gqa_causal_attention", + "op_id": "prefill:op85:layer8.rope_gqa_causal_attention", + "sequence": 85 + }, + { + "depends_on": [ + "prefill:op85:layer8.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer8.o_projection", + "op_id": "prefill:op86:layer8.o_projection", + "sequence": 86 + }, + { + "depends_on": [ + "prefill:op86:layer8.o_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer8.post_attention_rmsnorm", + "op_id": "prefill:op87:layer8.post_attention_rmsnorm", + "sequence": 87 + }, + { + "depends_on": [ + "prefill:op87:layer8.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer8.gate_projection", + "op_id": "prefill:op88:layer8.gate_projection", + "sequence": 88 + }, + { + "depends_on": [ + "prefill:op87:layer8.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer8.up_projection", + "op_id": "prefill:op89:layer8.up_projection", + "sequence": 89 + }, + { + "depends_on": [ + "prefill:op88:layer8.gate_projection", + "prefill:op89:layer8.up_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer8.swiglu_down_projection", + "op_id": "prefill:op90:layer8.swiglu_down_projection", + "sequence": 90 + }, + { + "depends_on": [ + "prefill:op90:layer8.swiglu_down_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer9.input_rmsnorm", + "op_id": "prefill:op91:layer9.input_rmsnorm", + "sequence": 91 + }, + { + "depends_on": [ + "prefill:op91:layer9.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer9.q_projection", + "op_id": "prefill:op92:layer9.q_projection", + "sequence": 92 + }, + { + "depends_on": [ + "prefill:op91:layer9.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer9.k_projection", + "op_id": "prefill:op93:layer9.k_projection", + "sequence": 93 + }, + { + "depends_on": [ + "prefill:op91:layer9.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer9.v_projection", + "op_id": "prefill:op94:layer9.v_projection", + "sequence": 94 + }, + { + "depends_on": [ + "prefill:op92:layer9.q_projection", + "prefill:op93:layer9.k_projection", + "prefill:op94:layer9.v_projection" + ], + "details": { + "context_tokens": 4, + "kv_repeat_groups": 7, + "kv_shape": [ + 4, + 4, + 2 + ], + "q_shape": [ + 4, + 28, + 2 + ] + }, + "forward_id": "prefill", + "name": "layer9.rope_gqa_causal_attention", + "op_id": "prefill:op95:layer9.rope_gqa_causal_attention", + "sequence": 95 + }, + { + "depends_on": [ + "prefill:op95:layer9.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer9.o_projection", + "op_id": "prefill:op96:layer9.o_projection", + "sequence": 96 + }, + { + "depends_on": [ + "prefill:op96:layer9.o_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer9.post_attention_rmsnorm", + "op_id": "prefill:op97:layer9.post_attention_rmsnorm", + "sequence": 97 + }, + { + "depends_on": [ + "prefill:op97:layer9.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer9.gate_projection", + "op_id": "prefill:op98:layer9.gate_projection", + "sequence": 98 + }, + { + "depends_on": [ + "prefill:op97:layer9.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer9.up_projection", + "op_id": "prefill:op99:layer9.up_projection", + "sequence": 99 + }, + { + "depends_on": [ + "prefill:op98:layer9.gate_projection", + "prefill:op99:layer9.up_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer9.swiglu_down_projection", + "op_id": "prefill:op100:layer9.swiglu_down_projection", + "sequence": 100 + }, + { + "depends_on": [ + "prefill:op100:layer9.swiglu_down_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer10.input_rmsnorm", + "op_id": "prefill:op101:layer10.input_rmsnorm", + "sequence": 101 + }, + { + "depends_on": [ + "prefill:op101:layer10.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer10.q_projection", + "op_id": "prefill:op102:layer10.q_projection", + "sequence": 102 + }, + { + "depends_on": [ + "prefill:op101:layer10.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer10.k_projection", + "op_id": "prefill:op103:layer10.k_projection", + "sequence": 103 + }, + { + "depends_on": [ + "prefill:op101:layer10.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer10.v_projection", + "op_id": "prefill:op104:layer10.v_projection", + "sequence": 104 + }, + { + "depends_on": [ + "prefill:op102:layer10.q_projection", + "prefill:op103:layer10.k_projection", + "prefill:op104:layer10.v_projection" + ], + "details": { + "context_tokens": 4, + "kv_repeat_groups": 7, + "kv_shape": [ + 4, + 4, + 2 + ], + "q_shape": [ + 4, + 28, + 2 + ] + }, + "forward_id": "prefill", + "name": "layer10.rope_gqa_causal_attention", + "op_id": "prefill:op105:layer10.rope_gqa_causal_attention", + "sequence": 105 + }, + { + "depends_on": [ + "prefill:op105:layer10.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer10.o_projection", + "op_id": "prefill:op106:layer10.o_projection", + "sequence": 106 + }, + { + "depends_on": [ + "prefill:op106:layer10.o_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer10.post_attention_rmsnorm", + "op_id": "prefill:op107:layer10.post_attention_rmsnorm", + "sequence": 107 + }, + { + "depends_on": [ + "prefill:op107:layer10.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer10.gate_projection", + "op_id": "prefill:op108:layer10.gate_projection", + "sequence": 108 + }, + { + "depends_on": [ + "prefill:op107:layer10.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer10.up_projection", + "op_id": "prefill:op109:layer10.up_projection", + "sequence": 109 + }, + { + "depends_on": [ + "prefill:op108:layer10.gate_projection", + "prefill:op109:layer10.up_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer10.swiglu_down_projection", + "op_id": "prefill:op110:layer10.swiglu_down_projection", + "sequence": 110 + }, + { + "depends_on": [ + "prefill:op110:layer10.swiglu_down_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer11.input_rmsnorm", + "op_id": "prefill:op111:layer11.input_rmsnorm", + "sequence": 111 + }, + { + "depends_on": [ + "prefill:op111:layer11.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer11.q_projection", + "op_id": "prefill:op112:layer11.q_projection", + "sequence": 112 + }, + { + "depends_on": [ + "prefill:op111:layer11.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer11.k_projection", + "op_id": "prefill:op113:layer11.k_projection", + "sequence": 113 + }, + { + "depends_on": [ + "prefill:op111:layer11.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer11.v_projection", + "op_id": "prefill:op114:layer11.v_projection", + "sequence": 114 + }, + { + "depends_on": [ + "prefill:op112:layer11.q_projection", + "prefill:op113:layer11.k_projection", + "prefill:op114:layer11.v_projection" + ], + "details": { + "context_tokens": 4, + "kv_repeat_groups": 7, + "kv_shape": [ + 4, + 4, + 2 + ], + "q_shape": [ + 4, + 28, + 2 + ] + }, + "forward_id": "prefill", + "name": "layer11.rope_gqa_causal_attention", + "op_id": "prefill:op115:layer11.rope_gqa_causal_attention", + "sequence": 115 + }, + { + "depends_on": [ + "prefill:op115:layer11.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer11.o_projection", + "op_id": "prefill:op116:layer11.o_projection", + "sequence": 116 + }, + { + "depends_on": [ + "prefill:op116:layer11.o_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer11.post_attention_rmsnorm", + "op_id": "prefill:op117:layer11.post_attention_rmsnorm", + "sequence": 117 + }, + { + "depends_on": [ + "prefill:op117:layer11.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer11.gate_projection", + "op_id": "prefill:op118:layer11.gate_projection", + "sequence": 118 + }, + { + "depends_on": [ + "prefill:op117:layer11.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer11.up_projection", + "op_id": "prefill:op119:layer11.up_projection", + "sequence": 119 + }, + { + "depends_on": [ + "prefill:op118:layer11.gate_projection", + "prefill:op119:layer11.up_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer11.swiglu_down_projection", + "op_id": "prefill:op120:layer11.swiglu_down_projection", + "sequence": 120 + }, + { + "depends_on": [ + "prefill:op120:layer11.swiglu_down_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer12.input_rmsnorm", + "op_id": "prefill:op121:layer12.input_rmsnorm", + "sequence": 121 + }, + { + "depends_on": [ + "prefill:op121:layer12.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer12.q_projection", + "op_id": "prefill:op122:layer12.q_projection", + "sequence": 122 + }, + { + "depends_on": [ + "prefill:op121:layer12.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer12.k_projection", + "op_id": "prefill:op123:layer12.k_projection", + "sequence": 123 + }, + { + "depends_on": [ + "prefill:op121:layer12.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer12.v_projection", + "op_id": "prefill:op124:layer12.v_projection", + "sequence": 124 + }, + { + "depends_on": [ + "prefill:op122:layer12.q_projection", + "prefill:op123:layer12.k_projection", + "prefill:op124:layer12.v_projection" + ], + "details": { + "context_tokens": 4, + "kv_repeat_groups": 7, + "kv_shape": [ + 4, + 4, + 2 + ], + "q_shape": [ + 4, + 28, + 2 + ] + }, + "forward_id": "prefill", + "name": "layer12.rope_gqa_causal_attention", + "op_id": "prefill:op125:layer12.rope_gqa_causal_attention", + "sequence": 125 + }, + { + "depends_on": [ + "prefill:op125:layer12.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer12.o_projection", + "op_id": "prefill:op126:layer12.o_projection", + "sequence": 126 + }, + { + "depends_on": [ + "prefill:op126:layer12.o_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer12.post_attention_rmsnorm", + "op_id": "prefill:op127:layer12.post_attention_rmsnorm", + "sequence": 127 + }, + { + "depends_on": [ + "prefill:op127:layer12.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer12.gate_projection", + "op_id": "prefill:op128:layer12.gate_projection", + "sequence": 128 + }, + { + "depends_on": [ + "prefill:op127:layer12.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer12.up_projection", + "op_id": "prefill:op129:layer12.up_projection", + "sequence": 129 + }, + { + "depends_on": [ + "prefill:op128:layer12.gate_projection", + "prefill:op129:layer12.up_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer12.swiglu_down_projection", + "op_id": "prefill:op130:layer12.swiglu_down_projection", + "sequence": 130 + }, + { + "depends_on": [ + "prefill:op130:layer12.swiglu_down_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer13.input_rmsnorm", + "op_id": "prefill:op131:layer13.input_rmsnorm", + "sequence": 131 + }, + { + "depends_on": [ + "prefill:op131:layer13.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer13.q_projection", + "op_id": "prefill:op132:layer13.q_projection", + "sequence": 132 + }, + { + "depends_on": [ + "prefill:op131:layer13.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer13.k_projection", + "op_id": "prefill:op133:layer13.k_projection", + "sequence": 133 + }, + { + "depends_on": [ + "prefill:op131:layer13.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer13.v_projection", + "op_id": "prefill:op134:layer13.v_projection", + "sequence": 134 + }, + { + "depends_on": [ + "prefill:op132:layer13.q_projection", + "prefill:op133:layer13.k_projection", + "prefill:op134:layer13.v_projection" + ], + "details": { + "context_tokens": 4, + "kv_repeat_groups": 7, + "kv_shape": [ + 4, + 4, + 2 + ], + "q_shape": [ + 4, + 28, + 2 + ] + }, + "forward_id": "prefill", + "name": "layer13.rope_gqa_causal_attention", + "op_id": "prefill:op135:layer13.rope_gqa_causal_attention", + "sequence": 135 + }, + { + "depends_on": [ + "prefill:op135:layer13.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer13.o_projection", + "op_id": "prefill:op136:layer13.o_projection", + "sequence": 136 + }, + { + "depends_on": [ + "prefill:op136:layer13.o_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer13.post_attention_rmsnorm", + "op_id": "prefill:op137:layer13.post_attention_rmsnorm", + "sequence": 137 + }, + { + "depends_on": [ + "prefill:op137:layer13.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer13.gate_projection", + "op_id": "prefill:op138:layer13.gate_projection", + "sequence": 138 + }, + { + "depends_on": [ + "prefill:op137:layer13.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer13.up_projection", + "op_id": "prefill:op139:layer13.up_projection", + "sequence": 139 + }, + { + "depends_on": [ + "prefill:op138:layer13.gate_projection", + "prefill:op139:layer13.up_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer13.swiglu_down_projection", + "op_id": "prefill:op140:layer13.swiglu_down_projection", + "sequence": 140 + }, + { + "depends_on": [ + "prefill:op140:layer13.swiglu_down_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer14.input_rmsnorm", + "op_id": "prefill:op141:layer14.input_rmsnorm", + "sequence": 141 + }, + { + "depends_on": [ + "prefill:op141:layer14.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer14.q_projection", + "op_id": "prefill:op142:layer14.q_projection", + "sequence": 142 + }, + { + "depends_on": [ + "prefill:op141:layer14.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer14.k_projection", + "op_id": "prefill:op143:layer14.k_projection", + "sequence": 143 + }, + { + "depends_on": [ + "prefill:op141:layer14.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer14.v_projection", + "op_id": "prefill:op144:layer14.v_projection", + "sequence": 144 + }, + { + "depends_on": [ + "prefill:op142:layer14.q_projection", + "prefill:op143:layer14.k_projection", + "prefill:op144:layer14.v_projection" + ], + "details": { + "context_tokens": 4, + "kv_repeat_groups": 7, + "kv_shape": [ + 4, + 4, + 2 + ], + "q_shape": [ + 4, + 28, + 2 + ] + }, + "forward_id": "prefill", + "name": "layer14.rope_gqa_causal_attention", + "op_id": "prefill:op145:layer14.rope_gqa_causal_attention", + "sequence": 145 + }, + { + "depends_on": [ + "prefill:op145:layer14.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer14.o_projection", + "op_id": "prefill:op146:layer14.o_projection", + "sequence": 146 + }, + { + "depends_on": [ + "prefill:op146:layer14.o_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer14.post_attention_rmsnorm", + "op_id": "prefill:op147:layer14.post_attention_rmsnorm", + "sequence": 147 + }, + { + "depends_on": [ + "prefill:op147:layer14.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer14.gate_projection", + "op_id": "prefill:op148:layer14.gate_projection", + "sequence": 148 + }, + { + "depends_on": [ + "prefill:op147:layer14.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer14.up_projection", + "op_id": "prefill:op149:layer14.up_projection", + "sequence": 149 + }, + { + "depends_on": [ + "prefill:op148:layer14.gate_projection", + "prefill:op149:layer14.up_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer14.swiglu_down_projection", + "op_id": "prefill:op150:layer14.swiglu_down_projection", + "sequence": 150 + }, + { + "depends_on": [ + "prefill:op150:layer14.swiglu_down_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer15.input_rmsnorm", + "op_id": "prefill:op151:layer15.input_rmsnorm", + "sequence": 151 + }, + { + "depends_on": [ + "prefill:op151:layer15.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer15.q_projection", + "op_id": "prefill:op152:layer15.q_projection", + "sequence": 152 + }, + { + "depends_on": [ + "prefill:op151:layer15.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer15.k_projection", + "op_id": "prefill:op153:layer15.k_projection", + "sequence": 153 + }, + { + "depends_on": [ + "prefill:op151:layer15.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer15.v_projection", + "op_id": "prefill:op154:layer15.v_projection", + "sequence": 154 + }, + { + "depends_on": [ + "prefill:op152:layer15.q_projection", + "prefill:op153:layer15.k_projection", + "prefill:op154:layer15.v_projection" + ], + "details": { + "context_tokens": 4, + "kv_repeat_groups": 7, + "kv_shape": [ + 4, + 4, + 2 + ], + "q_shape": [ + 4, + 28, + 2 + ] + }, + "forward_id": "prefill", + "name": "layer15.rope_gqa_causal_attention", + "op_id": "prefill:op155:layer15.rope_gqa_causal_attention", + "sequence": 155 + }, + { + "depends_on": [ + "prefill:op155:layer15.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer15.o_projection", + "op_id": "prefill:op156:layer15.o_projection", + "sequence": 156 + }, + { + "depends_on": [ + "prefill:op156:layer15.o_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer15.post_attention_rmsnorm", + "op_id": "prefill:op157:layer15.post_attention_rmsnorm", + "sequence": 157 + }, + { + "depends_on": [ + "prefill:op157:layer15.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer15.gate_projection", + "op_id": "prefill:op158:layer15.gate_projection", + "sequence": 158 + }, + { + "depends_on": [ + "prefill:op157:layer15.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer15.up_projection", + "op_id": "prefill:op159:layer15.up_projection", + "sequence": 159 + }, + { + "depends_on": [ + "prefill:op158:layer15.gate_projection", + "prefill:op159:layer15.up_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer15.swiglu_down_projection", + "op_id": "prefill:op160:layer15.swiglu_down_projection", + "sequence": 160 + }, + { + "depends_on": [ + "prefill:op160:layer15.swiglu_down_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer16.input_rmsnorm", + "op_id": "prefill:op161:layer16.input_rmsnorm", + "sequence": 161 + }, + { + "depends_on": [ + "prefill:op161:layer16.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer16.q_projection", + "op_id": "prefill:op162:layer16.q_projection", + "sequence": 162 + }, + { + "depends_on": [ + "prefill:op161:layer16.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer16.k_projection", + "op_id": "prefill:op163:layer16.k_projection", + "sequence": 163 + }, + { + "depends_on": [ + "prefill:op161:layer16.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer16.v_projection", + "op_id": "prefill:op164:layer16.v_projection", + "sequence": 164 + }, + { + "depends_on": [ + "prefill:op162:layer16.q_projection", + "prefill:op163:layer16.k_projection", + "prefill:op164:layer16.v_projection" + ], + "details": { + "context_tokens": 4, + "kv_repeat_groups": 7, + "kv_shape": [ + 4, + 4, + 2 + ], + "q_shape": [ + 4, + 28, + 2 + ] + }, + "forward_id": "prefill", + "name": "layer16.rope_gqa_causal_attention", + "op_id": "prefill:op165:layer16.rope_gqa_causal_attention", + "sequence": 165 + }, + { + "depends_on": [ + "prefill:op165:layer16.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer16.o_projection", + "op_id": "prefill:op166:layer16.o_projection", + "sequence": 166 + }, + { + "depends_on": [ + "prefill:op166:layer16.o_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer16.post_attention_rmsnorm", + "op_id": "prefill:op167:layer16.post_attention_rmsnorm", + "sequence": 167 + }, + { + "depends_on": [ + "prefill:op167:layer16.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer16.gate_projection", + "op_id": "prefill:op168:layer16.gate_projection", + "sequence": 168 + }, + { + "depends_on": [ + "prefill:op167:layer16.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer16.up_projection", + "op_id": "prefill:op169:layer16.up_projection", + "sequence": 169 + }, + { + "depends_on": [ + "prefill:op168:layer16.gate_projection", + "prefill:op169:layer16.up_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer16.swiglu_down_projection", + "op_id": "prefill:op170:layer16.swiglu_down_projection", + "sequence": 170 + }, + { + "depends_on": [ + "prefill:op170:layer16.swiglu_down_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer17.input_rmsnorm", + "op_id": "prefill:op171:layer17.input_rmsnorm", + "sequence": 171 + }, + { + "depends_on": [ + "prefill:op171:layer17.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer17.q_projection", + "op_id": "prefill:op172:layer17.q_projection", + "sequence": 172 + }, + { + "depends_on": [ + "prefill:op171:layer17.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer17.k_projection", + "op_id": "prefill:op173:layer17.k_projection", + "sequence": 173 + }, + { + "depends_on": [ + "prefill:op171:layer17.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer17.v_projection", + "op_id": "prefill:op174:layer17.v_projection", + "sequence": 174 + }, + { + "depends_on": [ + "prefill:op172:layer17.q_projection", + "prefill:op173:layer17.k_projection", + "prefill:op174:layer17.v_projection" + ], + "details": { + "context_tokens": 4, + "kv_repeat_groups": 7, + "kv_shape": [ + 4, + 4, + 2 + ], + "q_shape": [ + 4, + 28, + 2 + ] + }, + "forward_id": "prefill", + "name": "layer17.rope_gqa_causal_attention", + "op_id": "prefill:op175:layer17.rope_gqa_causal_attention", + "sequence": 175 + }, + { + "depends_on": [ + "prefill:op175:layer17.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer17.o_projection", + "op_id": "prefill:op176:layer17.o_projection", + "sequence": 176 + }, + { + "depends_on": [ + "prefill:op176:layer17.o_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer17.post_attention_rmsnorm", + "op_id": "prefill:op177:layer17.post_attention_rmsnorm", + "sequence": 177 + }, + { + "depends_on": [ + "prefill:op177:layer17.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer17.gate_projection", + "op_id": "prefill:op178:layer17.gate_projection", + "sequence": 178 + }, + { + "depends_on": [ + "prefill:op177:layer17.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer17.up_projection", + "op_id": "prefill:op179:layer17.up_projection", + "sequence": 179 + }, + { + "depends_on": [ + "prefill:op178:layer17.gate_projection", + "prefill:op179:layer17.up_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer17.swiglu_down_projection", + "op_id": "prefill:op180:layer17.swiglu_down_projection", + "sequence": 180 + }, + { + "depends_on": [ + "prefill:op180:layer17.swiglu_down_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer18.input_rmsnorm", + "op_id": "prefill:op181:layer18.input_rmsnorm", + "sequence": 181 + }, + { + "depends_on": [ + "prefill:op181:layer18.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer18.q_projection", + "op_id": "prefill:op182:layer18.q_projection", + "sequence": 182 + }, + { + "depends_on": [ + "prefill:op181:layer18.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer18.k_projection", + "op_id": "prefill:op183:layer18.k_projection", + "sequence": 183 + }, + { + "depends_on": [ + "prefill:op181:layer18.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer18.v_projection", + "op_id": "prefill:op184:layer18.v_projection", + "sequence": 184 + }, + { + "depends_on": [ + "prefill:op182:layer18.q_projection", + "prefill:op183:layer18.k_projection", + "prefill:op184:layer18.v_projection" + ], + "details": { + "context_tokens": 4, + "kv_repeat_groups": 7, + "kv_shape": [ + 4, + 4, + 2 + ], + "q_shape": [ + 4, + 28, + 2 + ] + }, + "forward_id": "prefill", + "name": "layer18.rope_gqa_causal_attention", + "op_id": "prefill:op185:layer18.rope_gqa_causal_attention", + "sequence": 185 + }, + { + "depends_on": [ + "prefill:op185:layer18.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer18.o_projection", + "op_id": "prefill:op186:layer18.o_projection", + "sequence": 186 + }, + { + "depends_on": [ + "prefill:op186:layer18.o_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer18.post_attention_rmsnorm", + "op_id": "prefill:op187:layer18.post_attention_rmsnorm", + "sequence": 187 + }, + { + "depends_on": [ + "prefill:op187:layer18.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer18.gate_projection", + "op_id": "prefill:op188:layer18.gate_projection", + "sequence": 188 + }, + { + "depends_on": [ + "prefill:op187:layer18.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer18.up_projection", + "op_id": "prefill:op189:layer18.up_projection", + "sequence": 189 + }, + { + "depends_on": [ + "prefill:op188:layer18.gate_projection", + "prefill:op189:layer18.up_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer18.swiglu_down_projection", + "op_id": "prefill:op190:layer18.swiglu_down_projection", + "sequence": 190 + }, + { + "depends_on": [ + "prefill:op190:layer18.swiglu_down_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer19.input_rmsnorm", + "op_id": "prefill:op191:layer19.input_rmsnorm", + "sequence": 191 + }, + { + "depends_on": [ + "prefill:op191:layer19.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer19.q_projection", + "op_id": "prefill:op192:layer19.q_projection", + "sequence": 192 + }, + { + "depends_on": [ + "prefill:op191:layer19.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer19.k_projection", + "op_id": "prefill:op193:layer19.k_projection", + "sequence": 193 + }, + { + "depends_on": [ + "prefill:op191:layer19.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer19.v_projection", + "op_id": "prefill:op194:layer19.v_projection", + "sequence": 194 + }, + { + "depends_on": [ + "prefill:op192:layer19.q_projection", + "prefill:op193:layer19.k_projection", + "prefill:op194:layer19.v_projection" + ], + "details": { + "context_tokens": 4, + "kv_repeat_groups": 7, + "kv_shape": [ + 4, + 4, + 2 + ], + "q_shape": [ + 4, + 28, + 2 + ] + }, + "forward_id": "prefill", + "name": "layer19.rope_gqa_causal_attention", + "op_id": "prefill:op195:layer19.rope_gqa_causal_attention", + "sequence": 195 + }, + { + "depends_on": [ + "prefill:op195:layer19.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer19.o_projection", + "op_id": "prefill:op196:layer19.o_projection", + "sequence": 196 + }, + { + "depends_on": [ + "prefill:op196:layer19.o_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer19.post_attention_rmsnorm", + "op_id": "prefill:op197:layer19.post_attention_rmsnorm", + "sequence": 197 + }, + { + "depends_on": [ + "prefill:op197:layer19.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer19.gate_projection", + "op_id": "prefill:op198:layer19.gate_projection", + "sequence": 198 + }, + { + "depends_on": [ + "prefill:op197:layer19.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer19.up_projection", + "op_id": "prefill:op199:layer19.up_projection", + "sequence": 199 + }, + { + "depends_on": [ + "prefill:op198:layer19.gate_projection", + "prefill:op199:layer19.up_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer19.swiglu_down_projection", + "op_id": "prefill:op200:layer19.swiglu_down_projection", + "sequence": 200 + }, + { + "depends_on": [ + "prefill:op200:layer19.swiglu_down_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer20.input_rmsnorm", + "op_id": "prefill:op201:layer20.input_rmsnorm", + "sequence": 201 + }, + { + "depends_on": [ + "prefill:op201:layer20.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer20.q_projection", + "op_id": "prefill:op202:layer20.q_projection", + "sequence": 202 + }, + { + "depends_on": [ + "prefill:op201:layer20.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer20.k_projection", + "op_id": "prefill:op203:layer20.k_projection", + "sequence": 203 + }, + { + "depends_on": [ + "prefill:op201:layer20.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer20.v_projection", + "op_id": "prefill:op204:layer20.v_projection", + "sequence": 204 + }, + { + "depends_on": [ + "prefill:op202:layer20.q_projection", + "prefill:op203:layer20.k_projection", + "prefill:op204:layer20.v_projection" + ], + "details": { + "context_tokens": 4, + "kv_repeat_groups": 7, + "kv_shape": [ + 4, + 4, + 2 + ], + "q_shape": [ + 4, + 28, + 2 + ] + }, + "forward_id": "prefill", + "name": "layer20.rope_gqa_causal_attention", + "op_id": "prefill:op205:layer20.rope_gqa_causal_attention", + "sequence": 205 + }, + { + "depends_on": [ + "prefill:op205:layer20.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer20.o_projection", + "op_id": "prefill:op206:layer20.o_projection", + "sequence": 206 + }, + { + "depends_on": [ + "prefill:op206:layer20.o_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer20.post_attention_rmsnorm", + "op_id": "prefill:op207:layer20.post_attention_rmsnorm", + "sequence": 207 + }, + { + "depends_on": [ + "prefill:op207:layer20.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer20.gate_projection", + "op_id": "prefill:op208:layer20.gate_projection", + "sequence": 208 + }, + { + "depends_on": [ + "prefill:op207:layer20.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer20.up_projection", + "op_id": "prefill:op209:layer20.up_projection", + "sequence": 209 + }, + { + "depends_on": [ + "prefill:op208:layer20.gate_projection", + "prefill:op209:layer20.up_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer20.swiglu_down_projection", + "op_id": "prefill:op210:layer20.swiglu_down_projection", + "sequence": 210 + }, + { + "depends_on": [ + "prefill:op210:layer20.swiglu_down_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer21.input_rmsnorm", + "op_id": "prefill:op211:layer21.input_rmsnorm", + "sequence": 211 + }, + { + "depends_on": [ + "prefill:op211:layer21.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer21.q_projection", + "op_id": "prefill:op212:layer21.q_projection", + "sequence": 212 + }, + { + "depends_on": [ + "prefill:op211:layer21.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer21.k_projection", + "op_id": "prefill:op213:layer21.k_projection", + "sequence": 213 + }, + { + "depends_on": [ + "prefill:op211:layer21.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer21.v_projection", + "op_id": "prefill:op214:layer21.v_projection", + "sequence": 214 + }, + { + "depends_on": [ + "prefill:op212:layer21.q_projection", + "prefill:op213:layer21.k_projection", + "prefill:op214:layer21.v_projection" + ], + "details": { + "context_tokens": 4, + "kv_repeat_groups": 7, + "kv_shape": [ + 4, + 4, + 2 + ], + "q_shape": [ + 4, + 28, + 2 + ] + }, + "forward_id": "prefill", + "name": "layer21.rope_gqa_causal_attention", + "op_id": "prefill:op215:layer21.rope_gqa_causal_attention", + "sequence": 215 + }, + { + "depends_on": [ + "prefill:op215:layer21.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer21.o_projection", + "op_id": "prefill:op216:layer21.o_projection", + "sequence": 216 + }, + { + "depends_on": [ + "prefill:op216:layer21.o_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer21.post_attention_rmsnorm", + "op_id": "prefill:op217:layer21.post_attention_rmsnorm", + "sequence": 217 + }, + { + "depends_on": [ + "prefill:op217:layer21.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer21.gate_projection", + "op_id": "prefill:op218:layer21.gate_projection", + "sequence": 218 + }, + { + "depends_on": [ + "prefill:op217:layer21.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer21.up_projection", + "op_id": "prefill:op219:layer21.up_projection", + "sequence": 219 + }, + { + "depends_on": [ + "prefill:op218:layer21.gate_projection", + "prefill:op219:layer21.up_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer21.swiglu_down_projection", + "op_id": "prefill:op220:layer21.swiglu_down_projection", + "sequence": 220 + }, + { + "depends_on": [ + "prefill:op220:layer21.swiglu_down_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer22.input_rmsnorm", + "op_id": "prefill:op221:layer22.input_rmsnorm", + "sequence": 221 + }, + { + "depends_on": [ + "prefill:op221:layer22.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer22.q_projection", + "op_id": "prefill:op222:layer22.q_projection", + "sequence": 222 + }, + { + "depends_on": [ + "prefill:op221:layer22.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer22.k_projection", + "op_id": "prefill:op223:layer22.k_projection", + "sequence": 223 + }, + { + "depends_on": [ + "prefill:op221:layer22.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer22.v_projection", + "op_id": "prefill:op224:layer22.v_projection", + "sequence": 224 + }, + { + "depends_on": [ + "prefill:op222:layer22.q_projection", + "prefill:op223:layer22.k_projection", + "prefill:op224:layer22.v_projection" + ], + "details": { + "context_tokens": 4, + "kv_repeat_groups": 7, + "kv_shape": [ + 4, + 4, + 2 + ], + "q_shape": [ + 4, + 28, + 2 + ] + }, + "forward_id": "prefill", + "name": "layer22.rope_gqa_causal_attention", + "op_id": "prefill:op225:layer22.rope_gqa_causal_attention", + "sequence": 225 + }, + { + "depends_on": [ + "prefill:op225:layer22.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer22.o_projection", + "op_id": "prefill:op226:layer22.o_projection", + "sequence": 226 + }, + { + "depends_on": [ + "prefill:op226:layer22.o_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer22.post_attention_rmsnorm", + "op_id": "prefill:op227:layer22.post_attention_rmsnorm", + "sequence": 227 + }, + { + "depends_on": [ + "prefill:op227:layer22.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer22.gate_projection", + "op_id": "prefill:op228:layer22.gate_projection", + "sequence": 228 + }, + { + "depends_on": [ + "prefill:op227:layer22.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer22.up_projection", + "op_id": "prefill:op229:layer22.up_projection", + "sequence": 229 + }, + { + "depends_on": [ + "prefill:op228:layer22.gate_projection", + "prefill:op229:layer22.up_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer22.swiglu_down_projection", + "op_id": "prefill:op230:layer22.swiglu_down_projection", + "sequence": 230 + }, + { + "depends_on": [ + "prefill:op230:layer22.swiglu_down_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer23.input_rmsnorm", + "op_id": "prefill:op231:layer23.input_rmsnorm", + "sequence": 231 + }, + { + "depends_on": [ + "prefill:op231:layer23.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer23.q_projection", + "op_id": "prefill:op232:layer23.q_projection", + "sequence": 232 + }, + { + "depends_on": [ + "prefill:op231:layer23.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer23.k_projection", + "op_id": "prefill:op233:layer23.k_projection", + "sequence": 233 + }, + { + "depends_on": [ + "prefill:op231:layer23.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer23.v_projection", + "op_id": "prefill:op234:layer23.v_projection", + "sequence": 234 + }, + { + "depends_on": [ + "prefill:op232:layer23.q_projection", + "prefill:op233:layer23.k_projection", + "prefill:op234:layer23.v_projection" + ], + "details": { + "context_tokens": 4, + "kv_repeat_groups": 7, + "kv_shape": [ + 4, + 4, + 2 + ], + "q_shape": [ + 4, + 28, + 2 + ] + }, + "forward_id": "prefill", + "name": "layer23.rope_gqa_causal_attention", + "op_id": "prefill:op235:layer23.rope_gqa_causal_attention", + "sequence": 235 + }, + { + "depends_on": [ + "prefill:op235:layer23.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer23.o_projection", + "op_id": "prefill:op236:layer23.o_projection", + "sequence": 236 + }, + { + "depends_on": [ + "prefill:op236:layer23.o_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer23.post_attention_rmsnorm", + "op_id": "prefill:op237:layer23.post_attention_rmsnorm", + "sequence": 237 + }, + { + "depends_on": [ + "prefill:op237:layer23.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer23.gate_projection", + "op_id": "prefill:op238:layer23.gate_projection", + "sequence": 238 + }, + { + "depends_on": [ + "prefill:op237:layer23.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer23.up_projection", + "op_id": "prefill:op239:layer23.up_projection", + "sequence": 239 + }, + { + "depends_on": [ + "prefill:op238:layer23.gate_projection", + "prefill:op239:layer23.up_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer23.swiglu_down_projection", + "op_id": "prefill:op240:layer23.swiglu_down_projection", + "sequence": 240 + }, + { + "depends_on": [ + "prefill:op240:layer23.swiglu_down_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer24.input_rmsnorm", + "op_id": "prefill:op241:layer24.input_rmsnorm", + "sequence": 241 + }, + { + "depends_on": [ + "prefill:op241:layer24.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer24.q_projection", + "op_id": "prefill:op242:layer24.q_projection", + "sequence": 242 + }, + { + "depends_on": [ + "prefill:op241:layer24.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer24.k_projection", + "op_id": "prefill:op243:layer24.k_projection", + "sequence": 243 + }, + { + "depends_on": [ + "prefill:op241:layer24.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer24.v_projection", + "op_id": "prefill:op244:layer24.v_projection", + "sequence": 244 + }, + { + "depends_on": [ + "prefill:op242:layer24.q_projection", + "prefill:op243:layer24.k_projection", + "prefill:op244:layer24.v_projection" + ], + "details": { + "context_tokens": 4, + "kv_repeat_groups": 7, + "kv_shape": [ + 4, + 4, + 2 + ], + "q_shape": [ + 4, + 28, + 2 + ] + }, + "forward_id": "prefill", + "name": "layer24.rope_gqa_causal_attention", + "op_id": "prefill:op245:layer24.rope_gqa_causal_attention", + "sequence": 245 + }, + { + "depends_on": [ + "prefill:op245:layer24.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer24.o_projection", + "op_id": "prefill:op246:layer24.o_projection", + "sequence": 246 + }, + { + "depends_on": [ + "prefill:op246:layer24.o_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer24.post_attention_rmsnorm", + "op_id": "prefill:op247:layer24.post_attention_rmsnorm", + "sequence": 247 + }, + { + "depends_on": [ + "prefill:op247:layer24.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer24.gate_projection", + "op_id": "prefill:op248:layer24.gate_projection", + "sequence": 248 + }, + { + "depends_on": [ + "prefill:op247:layer24.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer24.up_projection", + "op_id": "prefill:op249:layer24.up_projection", + "sequence": 249 + }, + { + "depends_on": [ + "prefill:op248:layer24.gate_projection", + "prefill:op249:layer24.up_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer24.swiglu_down_projection", + "op_id": "prefill:op250:layer24.swiglu_down_projection", + "sequence": 250 + }, + { + "depends_on": [ + "prefill:op250:layer24.swiglu_down_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer25.input_rmsnorm", + "op_id": "prefill:op251:layer25.input_rmsnorm", + "sequence": 251 + }, + { + "depends_on": [ + "prefill:op251:layer25.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer25.q_projection", + "op_id": "prefill:op252:layer25.q_projection", + "sequence": 252 + }, + { + "depends_on": [ + "prefill:op251:layer25.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer25.k_projection", + "op_id": "prefill:op253:layer25.k_projection", + "sequence": 253 + }, + { + "depends_on": [ + "prefill:op251:layer25.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer25.v_projection", + "op_id": "prefill:op254:layer25.v_projection", + "sequence": 254 + }, + { + "depends_on": [ + "prefill:op252:layer25.q_projection", + "prefill:op253:layer25.k_projection", + "prefill:op254:layer25.v_projection" + ], + "details": { + "context_tokens": 4, + "kv_repeat_groups": 7, + "kv_shape": [ + 4, + 4, + 2 + ], + "q_shape": [ + 4, + 28, + 2 + ] + }, + "forward_id": "prefill", + "name": "layer25.rope_gqa_causal_attention", + "op_id": "prefill:op255:layer25.rope_gqa_causal_attention", + "sequence": 255 + }, + { + "depends_on": [ + "prefill:op255:layer25.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer25.o_projection", + "op_id": "prefill:op256:layer25.o_projection", + "sequence": 256 + }, + { + "depends_on": [ + "prefill:op256:layer25.o_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer25.post_attention_rmsnorm", + "op_id": "prefill:op257:layer25.post_attention_rmsnorm", + "sequence": 257 + }, + { + "depends_on": [ + "prefill:op257:layer25.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer25.gate_projection", + "op_id": "prefill:op258:layer25.gate_projection", + "sequence": 258 + }, + { + "depends_on": [ + "prefill:op257:layer25.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer25.up_projection", + "op_id": "prefill:op259:layer25.up_projection", + "sequence": 259 + }, + { + "depends_on": [ + "prefill:op258:layer25.gate_projection", + "prefill:op259:layer25.up_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer25.swiglu_down_projection", + "op_id": "prefill:op260:layer25.swiglu_down_projection", + "sequence": 260 + }, + { + "depends_on": [ + "prefill:op260:layer25.swiglu_down_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer26.input_rmsnorm", + "op_id": "prefill:op261:layer26.input_rmsnorm", + "sequence": 261 + }, + { + "depends_on": [ + "prefill:op261:layer26.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer26.q_projection", + "op_id": "prefill:op262:layer26.q_projection", + "sequence": 262 + }, + { + "depends_on": [ + "prefill:op261:layer26.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer26.k_projection", + "op_id": "prefill:op263:layer26.k_projection", + "sequence": 263 + }, + { + "depends_on": [ + "prefill:op261:layer26.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer26.v_projection", + "op_id": "prefill:op264:layer26.v_projection", + "sequence": 264 + }, + { + "depends_on": [ + "prefill:op262:layer26.q_projection", + "prefill:op263:layer26.k_projection", + "prefill:op264:layer26.v_projection" + ], + "details": { + "context_tokens": 4, + "kv_repeat_groups": 7, + "kv_shape": [ + 4, + 4, + 2 + ], + "q_shape": [ + 4, + 28, + 2 + ] + }, + "forward_id": "prefill", + "name": "layer26.rope_gqa_causal_attention", + "op_id": "prefill:op265:layer26.rope_gqa_causal_attention", + "sequence": 265 + }, + { + "depends_on": [ + "prefill:op265:layer26.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer26.o_projection", + "op_id": "prefill:op266:layer26.o_projection", + "sequence": 266 + }, + { + "depends_on": [ + "prefill:op266:layer26.o_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer26.post_attention_rmsnorm", + "op_id": "prefill:op267:layer26.post_attention_rmsnorm", + "sequence": 267 + }, + { + "depends_on": [ + "prefill:op267:layer26.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer26.gate_projection", + "op_id": "prefill:op268:layer26.gate_projection", + "sequence": 268 + }, + { + "depends_on": [ + "prefill:op267:layer26.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer26.up_projection", + "op_id": "prefill:op269:layer26.up_projection", + "sequence": 269 + }, + { + "depends_on": [ + "prefill:op268:layer26.gate_projection", + "prefill:op269:layer26.up_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer26.swiglu_down_projection", + "op_id": "prefill:op270:layer26.swiglu_down_projection", + "sequence": 270 + }, + { + "depends_on": [ + "prefill:op270:layer26.swiglu_down_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer27.input_rmsnorm", + "op_id": "prefill:op271:layer27.input_rmsnorm", + "sequence": 271 + }, + { + "depends_on": [ + "prefill:op271:layer27.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer27.q_projection", + "op_id": "prefill:op272:layer27.q_projection", + "sequence": 272 + }, + { + "depends_on": [ + "prefill:op271:layer27.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer27.k_projection", + "op_id": "prefill:op273:layer27.k_projection", + "sequence": 273 + }, + { + "depends_on": [ + "prefill:op271:layer27.input_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer27.v_projection", + "op_id": "prefill:op274:layer27.v_projection", + "sequence": 274 + }, + { + "depends_on": [ + "prefill:op272:layer27.q_projection", + "prefill:op273:layer27.k_projection", + "prefill:op274:layer27.v_projection" + ], + "details": { + "context_tokens": 4, + "kv_repeat_groups": 7, + "kv_shape": [ + 4, + 4, + 2 + ], + "q_shape": [ + 4, + 28, + 2 + ] + }, + "forward_id": "prefill", + "name": "layer27.rope_gqa_causal_attention", + "op_id": "prefill:op275:layer27.rope_gqa_causal_attention", + "sequence": 275 + }, + { + "depends_on": [ + "prefill:op275:layer27.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer27.o_projection", + "op_id": "prefill:op276:layer27.o_projection", + "sequence": 276 + }, + { + "depends_on": [ + "prefill:op276:layer27.o_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer27.post_attention_rmsnorm", + "op_id": "prefill:op277:layer27.post_attention_rmsnorm", + "sequence": 277 + }, + { + "depends_on": [ + "prefill:op277:layer27.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer27.gate_projection", + "op_id": "prefill:op278:layer27.gate_projection", + "sequence": 278 + }, + { + "depends_on": [ + "prefill:op277:layer27.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer27.up_projection", + "op_id": "prefill:op279:layer27.up_projection", + "sequence": 279 + }, + { + "depends_on": [ + "prefill:op278:layer27.gate_projection", + "prefill:op279:layer27.up_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "layer27.swiglu_down_projection", + "op_id": "prefill:op280:layer27.swiglu_down_projection", + "sequence": 280 + }, + { + "depends_on": [ + "prefill:op280:layer27.swiglu_down_projection" + ], + "details": {}, + "forward_id": "prefill", + "name": "final_rmsnorm", + "op_id": "prefill:op281:final_rmsnorm", + "sequence": 281 + }, + { + "depends_on": [ + "prefill:op281:final_rmsnorm" + ], + "details": {}, + "forward_id": "prefill", + "name": "lm_head", + "op_id": "prefill:op282:lm_head", + "sequence": 282 + }, + { + "depends_on": [ + "token_ids" + ], + "details": { + "token_count": 1 + }, + "forward_id": "decode0", + "name": "embedding_lookup", + "op_id": "decode0:op283:embedding_lookup", + "sequence": 283 + }, + { + "depends_on": [ + "decode0:op283:embedding_lookup" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer0.input_rmsnorm", + "op_id": "decode0:op284:layer0.input_rmsnorm", + "sequence": 284 + }, + { + "depends_on": [ + "decode0:op284:layer0.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer0.q_projection", + "op_id": "decode0:op285:layer0.q_projection", + "sequence": 285 + }, + { + "depends_on": [ + "decode0:op284:layer0.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer0.k_projection", + "op_id": "decode0:op286:layer0.k_projection", + "sequence": 286 + }, + { + "depends_on": [ + "decode0:op284:layer0.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer0.v_projection", + "op_id": "decode0:op287:layer0.v_projection", + "sequence": 287 + }, + { + "depends_on": [ + "decode0:op285:layer0.q_projection", + "decode0:op286:layer0.k_projection", + "decode0:op287:layer0.v_projection" + ], + "details": { + "context_tokens": 5, + "kv_repeat_groups": 7, + "kv_shape": [ + 1, + 4, + 2 + ], + "q_shape": [ + 1, + 28, + 2 + ] + }, + "forward_id": "decode0", + "name": "layer0.rope_gqa_causal_attention", + "op_id": "decode0:op288:layer0.rope_gqa_causal_attention", + "sequence": 288 + }, + { + "depends_on": [ + "decode0:op288:layer0.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer0.o_projection", + "op_id": "decode0:op289:layer0.o_projection", + "sequence": 289 + }, + { + "depends_on": [ + "decode0:op289:layer0.o_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer0.post_attention_rmsnorm", + "op_id": "decode0:op290:layer0.post_attention_rmsnorm", + "sequence": 290 + }, + { + "depends_on": [ + "decode0:op290:layer0.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer0.gate_projection", + "op_id": "decode0:op291:layer0.gate_projection", + "sequence": 291 + }, + { + "depends_on": [ + "decode0:op290:layer0.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer0.up_projection", + "op_id": "decode0:op292:layer0.up_projection", + "sequence": 292 + }, + { + "depends_on": [ + "decode0:op291:layer0.gate_projection", + "decode0:op292:layer0.up_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer0.swiglu_down_projection", + "op_id": "decode0:op293:layer0.swiglu_down_projection", + "sequence": 293 + }, + { + "depends_on": [ + "decode0:op293:layer0.swiglu_down_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer1.input_rmsnorm", + "op_id": "decode0:op294:layer1.input_rmsnorm", + "sequence": 294 + }, + { + "depends_on": [ + "decode0:op294:layer1.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer1.q_projection", + "op_id": "decode0:op295:layer1.q_projection", + "sequence": 295 + }, + { + "depends_on": [ + "decode0:op294:layer1.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer1.k_projection", + "op_id": "decode0:op296:layer1.k_projection", + "sequence": 296 + }, + { + "depends_on": [ + "decode0:op294:layer1.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer1.v_projection", + "op_id": "decode0:op297:layer1.v_projection", + "sequence": 297 + }, + { + "depends_on": [ + "decode0:op295:layer1.q_projection", + "decode0:op296:layer1.k_projection", + "decode0:op297:layer1.v_projection" + ], + "details": { + "context_tokens": 5, + "kv_repeat_groups": 7, + "kv_shape": [ + 1, + 4, + 2 + ], + "q_shape": [ + 1, + 28, + 2 + ] + }, + "forward_id": "decode0", + "name": "layer1.rope_gqa_causal_attention", + "op_id": "decode0:op298:layer1.rope_gqa_causal_attention", + "sequence": 298 + }, + { + "depends_on": [ + "decode0:op298:layer1.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer1.o_projection", + "op_id": "decode0:op299:layer1.o_projection", + "sequence": 299 + }, + { + "depends_on": [ + "decode0:op299:layer1.o_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer1.post_attention_rmsnorm", + "op_id": "decode0:op300:layer1.post_attention_rmsnorm", + "sequence": 300 + }, + { + "depends_on": [ + "decode0:op300:layer1.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer1.gate_projection", + "op_id": "decode0:op301:layer1.gate_projection", + "sequence": 301 + }, + { + "depends_on": [ + "decode0:op300:layer1.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer1.up_projection", + "op_id": "decode0:op302:layer1.up_projection", + "sequence": 302 + }, + { + "depends_on": [ + "decode0:op301:layer1.gate_projection", + "decode0:op302:layer1.up_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer1.swiglu_down_projection", + "op_id": "decode0:op303:layer1.swiglu_down_projection", + "sequence": 303 + }, + { + "depends_on": [ + "decode0:op303:layer1.swiglu_down_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer2.input_rmsnorm", + "op_id": "decode0:op304:layer2.input_rmsnorm", + "sequence": 304 + }, + { + "depends_on": [ + "decode0:op304:layer2.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer2.q_projection", + "op_id": "decode0:op305:layer2.q_projection", + "sequence": 305 + }, + { + "depends_on": [ + "decode0:op304:layer2.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer2.k_projection", + "op_id": "decode0:op306:layer2.k_projection", + "sequence": 306 + }, + { + "depends_on": [ + "decode0:op304:layer2.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer2.v_projection", + "op_id": "decode0:op307:layer2.v_projection", + "sequence": 307 + }, + { + "depends_on": [ + "decode0:op305:layer2.q_projection", + "decode0:op306:layer2.k_projection", + "decode0:op307:layer2.v_projection" + ], + "details": { + "context_tokens": 5, + "kv_repeat_groups": 7, + "kv_shape": [ + 1, + 4, + 2 + ], + "q_shape": [ + 1, + 28, + 2 + ] + }, + "forward_id": "decode0", + "name": "layer2.rope_gqa_causal_attention", + "op_id": "decode0:op308:layer2.rope_gqa_causal_attention", + "sequence": 308 + }, + { + "depends_on": [ + "decode0:op308:layer2.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer2.o_projection", + "op_id": "decode0:op309:layer2.o_projection", + "sequence": 309 + }, + { + "depends_on": [ + "decode0:op309:layer2.o_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer2.post_attention_rmsnorm", + "op_id": "decode0:op310:layer2.post_attention_rmsnorm", + "sequence": 310 + }, + { + "depends_on": [ + "decode0:op310:layer2.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer2.gate_projection", + "op_id": "decode0:op311:layer2.gate_projection", + "sequence": 311 + }, + { + "depends_on": [ + "decode0:op310:layer2.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer2.up_projection", + "op_id": "decode0:op312:layer2.up_projection", + "sequence": 312 + }, + { + "depends_on": [ + "decode0:op311:layer2.gate_projection", + "decode0:op312:layer2.up_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer2.swiglu_down_projection", + "op_id": "decode0:op313:layer2.swiglu_down_projection", + "sequence": 313 + }, + { + "depends_on": [ + "decode0:op313:layer2.swiglu_down_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer3.input_rmsnorm", + "op_id": "decode0:op314:layer3.input_rmsnorm", + "sequence": 314 + }, + { + "depends_on": [ + "decode0:op314:layer3.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer3.q_projection", + "op_id": "decode0:op315:layer3.q_projection", + "sequence": 315 + }, + { + "depends_on": [ + "decode0:op314:layer3.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer3.k_projection", + "op_id": "decode0:op316:layer3.k_projection", + "sequence": 316 + }, + { + "depends_on": [ + "decode0:op314:layer3.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer3.v_projection", + "op_id": "decode0:op317:layer3.v_projection", + "sequence": 317 + }, + { + "depends_on": [ + "decode0:op315:layer3.q_projection", + "decode0:op316:layer3.k_projection", + "decode0:op317:layer3.v_projection" + ], + "details": { + "context_tokens": 5, + "kv_repeat_groups": 7, + "kv_shape": [ + 1, + 4, + 2 + ], + "q_shape": [ + 1, + 28, + 2 + ] + }, + "forward_id": "decode0", + "name": "layer3.rope_gqa_causal_attention", + "op_id": "decode0:op318:layer3.rope_gqa_causal_attention", + "sequence": 318 + }, + { + "depends_on": [ + "decode0:op318:layer3.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer3.o_projection", + "op_id": "decode0:op319:layer3.o_projection", + "sequence": 319 + }, + { + "depends_on": [ + "decode0:op319:layer3.o_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer3.post_attention_rmsnorm", + "op_id": "decode0:op320:layer3.post_attention_rmsnorm", + "sequence": 320 + }, + { + "depends_on": [ + "decode0:op320:layer3.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer3.gate_projection", + "op_id": "decode0:op321:layer3.gate_projection", + "sequence": 321 + }, + { + "depends_on": [ + "decode0:op320:layer3.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer3.up_projection", + "op_id": "decode0:op322:layer3.up_projection", + "sequence": 322 + }, + { + "depends_on": [ + "decode0:op321:layer3.gate_projection", + "decode0:op322:layer3.up_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer3.swiglu_down_projection", + "op_id": "decode0:op323:layer3.swiglu_down_projection", + "sequence": 323 + }, + { + "depends_on": [ + "decode0:op323:layer3.swiglu_down_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer4.input_rmsnorm", + "op_id": "decode0:op324:layer4.input_rmsnorm", + "sequence": 324 + }, + { + "depends_on": [ + "decode0:op324:layer4.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer4.q_projection", + "op_id": "decode0:op325:layer4.q_projection", + "sequence": 325 + }, + { + "depends_on": [ + "decode0:op324:layer4.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer4.k_projection", + "op_id": "decode0:op326:layer4.k_projection", + "sequence": 326 + }, + { + "depends_on": [ + "decode0:op324:layer4.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer4.v_projection", + "op_id": "decode0:op327:layer4.v_projection", + "sequence": 327 + }, + { + "depends_on": [ + "decode0:op325:layer4.q_projection", + "decode0:op326:layer4.k_projection", + "decode0:op327:layer4.v_projection" + ], + "details": { + "context_tokens": 5, + "kv_repeat_groups": 7, + "kv_shape": [ + 1, + 4, + 2 + ], + "q_shape": [ + 1, + 28, + 2 + ] + }, + "forward_id": "decode0", + "name": "layer4.rope_gqa_causal_attention", + "op_id": "decode0:op328:layer4.rope_gqa_causal_attention", + "sequence": 328 + }, + { + "depends_on": [ + "decode0:op328:layer4.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer4.o_projection", + "op_id": "decode0:op329:layer4.o_projection", + "sequence": 329 + }, + { + "depends_on": [ + "decode0:op329:layer4.o_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer4.post_attention_rmsnorm", + "op_id": "decode0:op330:layer4.post_attention_rmsnorm", + "sequence": 330 + }, + { + "depends_on": [ + "decode0:op330:layer4.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer4.gate_projection", + "op_id": "decode0:op331:layer4.gate_projection", + "sequence": 331 + }, + { + "depends_on": [ + "decode0:op330:layer4.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer4.up_projection", + "op_id": "decode0:op332:layer4.up_projection", + "sequence": 332 + }, + { + "depends_on": [ + "decode0:op331:layer4.gate_projection", + "decode0:op332:layer4.up_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer4.swiglu_down_projection", + "op_id": "decode0:op333:layer4.swiglu_down_projection", + "sequence": 333 + }, + { + "depends_on": [ + "decode0:op333:layer4.swiglu_down_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer5.input_rmsnorm", + "op_id": "decode0:op334:layer5.input_rmsnorm", + "sequence": 334 + }, + { + "depends_on": [ + "decode0:op334:layer5.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer5.q_projection", + "op_id": "decode0:op335:layer5.q_projection", + "sequence": 335 + }, + { + "depends_on": [ + "decode0:op334:layer5.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer5.k_projection", + "op_id": "decode0:op336:layer5.k_projection", + "sequence": 336 + }, + { + "depends_on": [ + "decode0:op334:layer5.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer5.v_projection", + "op_id": "decode0:op337:layer5.v_projection", + "sequence": 337 + }, + { + "depends_on": [ + "decode0:op335:layer5.q_projection", + "decode0:op336:layer5.k_projection", + "decode0:op337:layer5.v_projection" + ], + "details": { + "context_tokens": 5, + "kv_repeat_groups": 7, + "kv_shape": [ + 1, + 4, + 2 + ], + "q_shape": [ + 1, + 28, + 2 + ] + }, + "forward_id": "decode0", + "name": "layer5.rope_gqa_causal_attention", + "op_id": "decode0:op338:layer5.rope_gqa_causal_attention", + "sequence": 338 + }, + { + "depends_on": [ + "decode0:op338:layer5.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer5.o_projection", + "op_id": "decode0:op339:layer5.o_projection", + "sequence": 339 + }, + { + "depends_on": [ + "decode0:op339:layer5.o_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer5.post_attention_rmsnorm", + "op_id": "decode0:op340:layer5.post_attention_rmsnorm", + "sequence": 340 + }, + { + "depends_on": [ + "decode0:op340:layer5.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer5.gate_projection", + "op_id": "decode0:op341:layer5.gate_projection", + "sequence": 341 + }, + { + "depends_on": [ + "decode0:op340:layer5.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer5.up_projection", + "op_id": "decode0:op342:layer5.up_projection", + "sequence": 342 + }, + { + "depends_on": [ + "decode0:op341:layer5.gate_projection", + "decode0:op342:layer5.up_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer5.swiglu_down_projection", + "op_id": "decode0:op343:layer5.swiglu_down_projection", + "sequence": 343 + }, + { + "depends_on": [ + "decode0:op343:layer5.swiglu_down_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer6.input_rmsnorm", + "op_id": "decode0:op344:layer6.input_rmsnorm", + "sequence": 344 + }, + { + "depends_on": [ + "decode0:op344:layer6.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer6.q_projection", + "op_id": "decode0:op345:layer6.q_projection", + "sequence": 345 + }, + { + "depends_on": [ + "decode0:op344:layer6.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer6.k_projection", + "op_id": "decode0:op346:layer6.k_projection", + "sequence": 346 + }, + { + "depends_on": [ + "decode0:op344:layer6.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer6.v_projection", + "op_id": "decode0:op347:layer6.v_projection", + "sequence": 347 + }, + { + "depends_on": [ + "decode0:op345:layer6.q_projection", + "decode0:op346:layer6.k_projection", + "decode0:op347:layer6.v_projection" + ], + "details": { + "context_tokens": 5, + "kv_repeat_groups": 7, + "kv_shape": [ + 1, + 4, + 2 + ], + "q_shape": [ + 1, + 28, + 2 + ] + }, + "forward_id": "decode0", + "name": "layer6.rope_gqa_causal_attention", + "op_id": "decode0:op348:layer6.rope_gqa_causal_attention", + "sequence": 348 + }, + { + "depends_on": [ + "decode0:op348:layer6.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer6.o_projection", + "op_id": "decode0:op349:layer6.o_projection", + "sequence": 349 + }, + { + "depends_on": [ + "decode0:op349:layer6.o_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer6.post_attention_rmsnorm", + "op_id": "decode0:op350:layer6.post_attention_rmsnorm", + "sequence": 350 + }, + { + "depends_on": [ + "decode0:op350:layer6.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer6.gate_projection", + "op_id": "decode0:op351:layer6.gate_projection", + "sequence": 351 + }, + { + "depends_on": [ + "decode0:op350:layer6.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer6.up_projection", + "op_id": "decode0:op352:layer6.up_projection", + "sequence": 352 + }, + { + "depends_on": [ + "decode0:op351:layer6.gate_projection", + "decode0:op352:layer6.up_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer6.swiglu_down_projection", + "op_id": "decode0:op353:layer6.swiglu_down_projection", + "sequence": 353 + }, + { + "depends_on": [ + "decode0:op353:layer6.swiglu_down_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer7.input_rmsnorm", + "op_id": "decode0:op354:layer7.input_rmsnorm", + "sequence": 354 + }, + { + "depends_on": [ + "decode0:op354:layer7.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer7.q_projection", + "op_id": "decode0:op355:layer7.q_projection", + "sequence": 355 + }, + { + "depends_on": [ + "decode0:op354:layer7.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer7.k_projection", + "op_id": "decode0:op356:layer7.k_projection", + "sequence": 356 + }, + { + "depends_on": [ + "decode0:op354:layer7.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer7.v_projection", + "op_id": "decode0:op357:layer7.v_projection", + "sequence": 357 + }, + { + "depends_on": [ + "decode0:op355:layer7.q_projection", + "decode0:op356:layer7.k_projection", + "decode0:op357:layer7.v_projection" + ], + "details": { + "context_tokens": 5, + "kv_repeat_groups": 7, + "kv_shape": [ + 1, + 4, + 2 + ], + "q_shape": [ + 1, + 28, + 2 + ] + }, + "forward_id": "decode0", + "name": "layer7.rope_gqa_causal_attention", + "op_id": "decode0:op358:layer7.rope_gqa_causal_attention", + "sequence": 358 + }, + { + "depends_on": [ + "decode0:op358:layer7.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer7.o_projection", + "op_id": "decode0:op359:layer7.o_projection", + "sequence": 359 + }, + { + "depends_on": [ + "decode0:op359:layer7.o_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer7.post_attention_rmsnorm", + "op_id": "decode0:op360:layer7.post_attention_rmsnorm", + "sequence": 360 + }, + { + "depends_on": [ + "decode0:op360:layer7.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer7.gate_projection", + "op_id": "decode0:op361:layer7.gate_projection", + "sequence": 361 + }, + { + "depends_on": [ + "decode0:op360:layer7.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer7.up_projection", + "op_id": "decode0:op362:layer7.up_projection", + "sequence": 362 + }, + { + "depends_on": [ + "decode0:op361:layer7.gate_projection", + "decode0:op362:layer7.up_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer7.swiglu_down_projection", + "op_id": "decode0:op363:layer7.swiglu_down_projection", + "sequence": 363 + }, + { + "depends_on": [ + "decode0:op363:layer7.swiglu_down_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer8.input_rmsnorm", + "op_id": "decode0:op364:layer8.input_rmsnorm", + "sequence": 364 + }, + { + "depends_on": [ + "decode0:op364:layer8.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer8.q_projection", + "op_id": "decode0:op365:layer8.q_projection", + "sequence": 365 + }, + { + "depends_on": [ + "decode0:op364:layer8.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer8.k_projection", + "op_id": "decode0:op366:layer8.k_projection", + "sequence": 366 + }, + { + "depends_on": [ + "decode0:op364:layer8.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer8.v_projection", + "op_id": "decode0:op367:layer8.v_projection", + "sequence": 367 + }, + { + "depends_on": [ + "decode0:op365:layer8.q_projection", + "decode0:op366:layer8.k_projection", + "decode0:op367:layer8.v_projection" + ], + "details": { + "context_tokens": 5, + "kv_repeat_groups": 7, + "kv_shape": [ + 1, + 4, + 2 + ], + "q_shape": [ + 1, + 28, + 2 + ] + }, + "forward_id": "decode0", + "name": "layer8.rope_gqa_causal_attention", + "op_id": "decode0:op368:layer8.rope_gqa_causal_attention", + "sequence": 368 + }, + { + "depends_on": [ + "decode0:op368:layer8.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer8.o_projection", + "op_id": "decode0:op369:layer8.o_projection", + "sequence": 369 + }, + { + "depends_on": [ + "decode0:op369:layer8.o_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer8.post_attention_rmsnorm", + "op_id": "decode0:op370:layer8.post_attention_rmsnorm", + "sequence": 370 + }, + { + "depends_on": [ + "decode0:op370:layer8.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer8.gate_projection", + "op_id": "decode0:op371:layer8.gate_projection", + "sequence": 371 + }, + { + "depends_on": [ + "decode0:op370:layer8.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer8.up_projection", + "op_id": "decode0:op372:layer8.up_projection", + "sequence": 372 + }, + { + "depends_on": [ + "decode0:op371:layer8.gate_projection", + "decode0:op372:layer8.up_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer8.swiglu_down_projection", + "op_id": "decode0:op373:layer8.swiglu_down_projection", + "sequence": 373 + }, + { + "depends_on": [ + "decode0:op373:layer8.swiglu_down_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer9.input_rmsnorm", + "op_id": "decode0:op374:layer9.input_rmsnorm", + "sequence": 374 + }, + { + "depends_on": [ + "decode0:op374:layer9.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer9.q_projection", + "op_id": "decode0:op375:layer9.q_projection", + "sequence": 375 + }, + { + "depends_on": [ + "decode0:op374:layer9.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer9.k_projection", + "op_id": "decode0:op376:layer9.k_projection", + "sequence": 376 + }, + { + "depends_on": [ + "decode0:op374:layer9.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer9.v_projection", + "op_id": "decode0:op377:layer9.v_projection", + "sequence": 377 + }, + { + "depends_on": [ + "decode0:op375:layer9.q_projection", + "decode0:op376:layer9.k_projection", + "decode0:op377:layer9.v_projection" + ], + "details": { + "context_tokens": 5, + "kv_repeat_groups": 7, + "kv_shape": [ + 1, + 4, + 2 + ], + "q_shape": [ + 1, + 28, + 2 + ] + }, + "forward_id": "decode0", + "name": "layer9.rope_gqa_causal_attention", + "op_id": "decode0:op378:layer9.rope_gqa_causal_attention", + "sequence": 378 + }, + { + "depends_on": [ + "decode0:op378:layer9.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer9.o_projection", + "op_id": "decode0:op379:layer9.o_projection", + "sequence": 379 + }, + { + "depends_on": [ + "decode0:op379:layer9.o_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer9.post_attention_rmsnorm", + "op_id": "decode0:op380:layer9.post_attention_rmsnorm", + "sequence": 380 + }, + { + "depends_on": [ + "decode0:op380:layer9.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer9.gate_projection", + "op_id": "decode0:op381:layer9.gate_projection", + "sequence": 381 + }, + { + "depends_on": [ + "decode0:op380:layer9.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer9.up_projection", + "op_id": "decode0:op382:layer9.up_projection", + "sequence": 382 + }, + { + "depends_on": [ + "decode0:op381:layer9.gate_projection", + "decode0:op382:layer9.up_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer9.swiglu_down_projection", + "op_id": "decode0:op383:layer9.swiglu_down_projection", + "sequence": 383 + }, + { + "depends_on": [ + "decode0:op383:layer9.swiglu_down_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer10.input_rmsnorm", + "op_id": "decode0:op384:layer10.input_rmsnorm", + "sequence": 384 + }, + { + "depends_on": [ + "decode0:op384:layer10.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer10.q_projection", + "op_id": "decode0:op385:layer10.q_projection", + "sequence": 385 + }, + { + "depends_on": [ + "decode0:op384:layer10.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer10.k_projection", + "op_id": "decode0:op386:layer10.k_projection", + "sequence": 386 + }, + { + "depends_on": [ + "decode0:op384:layer10.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer10.v_projection", + "op_id": "decode0:op387:layer10.v_projection", + "sequence": 387 + }, + { + "depends_on": [ + "decode0:op385:layer10.q_projection", + "decode0:op386:layer10.k_projection", + "decode0:op387:layer10.v_projection" + ], + "details": { + "context_tokens": 5, + "kv_repeat_groups": 7, + "kv_shape": [ + 1, + 4, + 2 + ], + "q_shape": [ + 1, + 28, + 2 + ] + }, + "forward_id": "decode0", + "name": "layer10.rope_gqa_causal_attention", + "op_id": "decode0:op388:layer10.rope_gqa_causal_attention", + "sequence": 388 + }, + { + "depends_on": [ + "decode0:op388:layer10.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer10.o_projection", + "op_id": "decode0:op389:layer10.o_projection", + "sequence": 389 + }, + { + "depends_on": [ + "decode0:op389:layer10.o_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer10.post_attention_rmsnorm", + "op_id": "decode0:op390:layer10.post_attention_rmsnorm", + "sequence": 390 + }, + { + "depends_on": [ + "decode0:op390:layer10.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer10.gate_projection", + "op_id": "decode0:op391:layer10.gate_projection", + "sequence": 391 + }, + { + "depends_on": [ + "decode0:op390:layer10.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer10.up_projection", + "op_id": "decode0:op392:layer10.up_projection", + "sequence": 392 + }, + { + "depends_on": [ + "decode0:op391:layer10.gate_projection", + "decode0:op392:layer10.up_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer10.swiglu_down_projection", + "op_id": "decode0:op393:layer10.swiglu_down_projection", + "sequence": 393 + }, + { + "depends_on": [ + "decode0:op393:layer10.swiglu_down_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer11.input_rmsnorm", + "op_id": "decode0:op394:layer11.input_rmsnorm", + "sequence": 394 + }, + { + "depends_on": [ + "decode0:op394:layer11.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer11.q_projection", + "op_id": "decode0:op395:layer11.q_projection", + "sequence": 395 + }, + { + "depends_on": [ + "decode0:op394:layer11.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer11.k_projection", + "op_id": "decode0:op396:layer11.k_projection", + "sequence": 396 + }, + { + "depends_on": [ + "decode0:op394:layer11.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer11.v_projection", + "op_id": "decode0:op397:layer11.v_projection", + "sequence": 397 + }, + { + "depends_on": [ + "decode0:op395:layer11.q_projection", + "decode0:op396:layer11.k_projection", + "decode0:op397:layer11.v_projection" + ], + "details": { + "context_tokens": 5, + "kv_repeat_groups": 7, + "kv_shape": [ + 1, + 4, + 2 + ], + "q_shape": [ + 1, + 28, + 2 + ] + }, + "forward_id": "decode0", + "name": "layer11.rope_gqa_causal_attention", + "op_id": "decode0:op398:layer11.rope_gqa_causal_attention", + "sequence": 398 + }, + { + "depends_on": [ + "decode0:op398:layer11.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer11.o_projection", + "op_id": "decode0:op399:layer11.o_projection", + "sequence": 399 + }, + { + "depends_on": [ + "decode0:op399:layer11.o_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer11.post_attention_rmsnorm", + "op_id": "decode0:op400:layer11.post_attention_rmsnorm", + "sequence": 400 + }, + { + "depends_on": [ + "decode0:op400:layer11.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer11.gate_projection", + "op_id": "decode0:op401:layer11.gate_projection", + "sequence": 401 + }, + { + "depends_on": [ + "decode0:op400:layer11.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer11.up_projection", + "op_id": "decode0:op402:layer11.up_projection", + "sequence": 402 + }, + { + "depends_on": [ + "decode0:op401:layer11.gate_projection", + "decode0:op402:layer11.up_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer11.swiglu_down_projection", + "op_id": "decode0:op403:layer11.swiglu_down_projection", + "sequence": 403 + }, + { + "depends_on": [ + "decode0:op403:layer11.swiglu_down_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer12.input_rmsnorm", + "op_id": "decode0:op404:layer12.input_rmsnorm", + "sequence": 404 + }, + { + "depends_on": [ + "decode0:op404:layer12.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer12.q_projection", + "op_id": "decode0:op405:layer12.q_projection", + "sequence": 405 + }, + { + "depends_on": [ + "decode0:op404:layer12.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer12.k_projection", + "op_id": "decode0:op406:layer12.k_projection", + "sequence": 406 + }, + { + "depends_on": [ + "decode0:op404:layer12.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer12.v_projection", + "op_id": "decode0:op407:layer12.v_projection", + "sequence": 407 + }, + { + "depends_on": [ + "decode0:op405:layer12.q_projection", + "decode0:op406:layer12.k_projection", + "decode0:op407:layer12.v_projection" + ], + "details": { + "context_tokens": 5, + "kv_repeat_groups": 7, + "kv_shape": [ + 1, + 4, + 2 + ], + "q_shape": [ + 1, + 28, + 2 + ] + }, + "forward_id": "decode0", + "name": "layer12.rope_gqa_causal_attention", + "op_id": "decode0:op408:layer12.rope_gqa_causal_attention", + "sequence": 408 + }, + { + "depends_on": [ + "decode0:op408:layer12.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer12.o_projection", + "op_id": "decode0:op409:layer12.o_projection", + "sequence": 409 + }, + { + "depends_on": [ + "decode0:op409:layer12.o_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer12.post_attention_rmsnorm", + "op_id": "decode0:op410:layer12.post_attention_rmsnorm", + "sequence": 410 + }, + { + "depends_on": [ + "decode0:op410:layer12.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer12.gate_projection", + "op_id": "decode0:op411:layer12.gate_projection", + "sequence": 411 + }, + { + "depends_on": [ + "decode0:op410:layer12.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer12.up_projection", + "op_id": "decode0:op412:layer12.up_projection", + "sequence": 412 + }, + { + "depends_on": [ + "decode0:op411:layer12.gate_projection", + "decode0:op412:layer12.up_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer12.swiglu_down_projection", + "op_id": "decode0:op413:layer12.swiglu_down_projection", + "sequence": 413 + }, + { + "depends_on": [ + "decode0:op413:layer12.swiglu_down_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer13.input_rmsnorm", + "op_id": "decode0:op414:layer13.input_rmsnorm", + "sequence": 414 + }, + { + "depends_on": [ + "decode0:op414:layer13.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer13.q_projection", + "op_id": "decode0:op415:layer13.q_projection", + "sequence": 415 + }, + { + "depends_on": [ + "decode0:op414:layer13.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer13.k_projection", + "op_id": "decode0:op416:layer13.k_projection", + "sequence": 416 + }, + { + "depends_on": [ + "decode0:op414:layer13.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer13.v_projection", + "op_id": "decode0:op417:layer13.v_projection", + "sequence": 417 + }, + { + "depends_on": [ + "decode0:op415:layer13.q_projection", + "decode0:op416:layer13.k_projection", + "decode0:op417:layer13.v_projection" + ], + "details": { + "context_tokens": 5, + "kv_repeat_groups": 7, + "kv_shape": [ + 1, + 4, + 2 + ], + "q_shape": [ + 1, + 28, + 2 + ] + }, + "forward_id": "decode0", + "name": "layer13.rope_gqa_causal_attention", + "op_id": "decode0:op418:layer13.rope_gqa_causal_attention", + "sequence": 418 + }, + { + "depends_on": [ + "decode0:op418:layer13.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer13.o_projection", + "op_id": "decode0:op419:layer13.o_projection", + "sequence": 419 + }, + { + "depends_on": [ + "decode0:op419:layer13.o_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer13.post_attention_rmsnorm", + "op_id": "decode0:op420:layer13.post_attention_rmsnorm", + "sequence": 420 + }, + { + "depends_on": [ + "decode0:op420:layer13.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer13.gate_projection", + "op_id": "decode0:op421:layer13.gate_projection", + "sequence": 421 + }, + { + "depends_on": [ + "decode0:op420:layer13.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer13.up_projection", + "op_id": "decode0:op422:layer13.up_projection", + "sequence": 422 + }, + { + "depends_on": [ + "decode0:op421:layer13.gate_projection", + "decode0:op422:layer13.up_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer13.swiglu_down_projection", + "op_id": "decode0:op423:layer13.swiglu_down_projection", + "sequence": 423 + }, + { + "depends_on": [ + "decode0:op423:layer13.swiglu_down_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer14.input_rmsnorm", + "op_id": "decode0:op424:layer14.input_rmsnorm", + "sequence": 424 + }, + { + "depends_on": [ + "decode0:op424:layer14.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer14.q_projection", + "op_id": "decode0:op425:layer14.q_projection", + "sequence": 425 + }, + { + "depends_on": [ + "decode0:op424:layer14.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer14.k_projection", + "op_id": "decode0:op426:layer14.k_projection", + "sequence": 426 + }, + { + "depends_on": [ + "decode0:op424:layer14.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer14.v_projection", + "op_id": "decode0:op427:layer14.v_projection", + "sequence": 427 + }, + { + "depends_on": [ + "decode0:op425:layer14.q_projection", + "decode0:op426:layer14.k_projection", + "decode0:op427:layer14.v_projection" + ], + "details": { + "context_tokens": 5, + "kv_repeat_groups": 7, + "kv_shape": [ + 1, + 4, + 2 + ], + "q_shape": [ + 1, + 28, + 2 + ] + }, + "forward_id": "decode0", + "name": "layer14.rope_gqa_causal_attention", + "op_id": "decode0:op428:layer14.rope_gqa_causal_attention", + "sequence": 428 + }, + { + "depends_on": [ + "decode0:op428:layer14.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer14.o_projection", + "op_id": "decode0:op429:layer14.o_projection", + "sequence": 429 + }, + { + "depends_on": [ + "decode0:op429:layer14.o_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer14.post_attention_rmsnorm", + "op_id": "decode0:op430:layer14.post_attention_rmsnorm", + "sequence": 430 + }, + { + "depends_on": [ + "decode0:op430:layer14.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer14.gate_projection", + "op_id": "decode0:op431:layer14.gate_projection", + "sequence": 431 + }, + { + "depends_on": [ + "decode0:op430:layer14.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer14.up_projection", + "op_id": "decode0:op432:layer14.up_projection", + "sequence": 432 + }, + { + "depends_on": [ + "decode0:op431:layer14.gate_projection", + "decode0:op432:layer14.up_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer14.swiglu_down_projection", + "op_id": "decode0:op433:layer14.swiglu_down_projection", + "sequence": 433 + }, + { + "depends_on": [ + "decode0:op433:layer14.swiglu_down_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer15.input_rmsnorm", + "op_id": "decode0:op434:layer15.input_rmsnorm", + "sequence": 434 + }, + { + "depends_on": [ + "decode0:op434:layer15.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer15.q_projection", + "op_id": "decode0:op435:layer15.q_projection", + "sequence": 435 + }, + { + "depends_on": [ + "decode0:op434:layer15.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer15.k_projection", + "op_id": "decode0:op436:layer15.k_projection", + "sequence": 436 + }, + { + "depends_on": [ + "decode0:op434:layer15.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer15.v_projection", + "op_id": "decode0:op437:layer15.v_projection", + "sequence": 437 + }, + { + "depends_on": [ + "decode0:op435:layer15.q_projection", + "decode0:op436:layer15.k_projection", + "decode0:op437:layer15.v_projection" + ], + "details": { + "context_tokens": 5, + "kv_repeat_groups": 7, + "kv_shape": [ + 1, + 4, + 2 + ], + "q_shape": [ + 1, + 28, + 2 + ] + }, + "forward_id": "decode0", + "name": "layer15.rope_gqa_causal_attention", + "op_id": "decode0:op438:layer15.rope_gqa_causal_attention", + "sequence": 438 + }, + { + "depends_on": [ + "decode0:op438:layer15.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer15.o_projection", + "op_id": "decode0:op439:layer15.o_projection", + "sequence": 439 + }, + { + "depends_on": [ + "decode0:op439:layer15.o_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer15.post_attention_rmsnorm", + "op_id": "decode0:op440:layer15.post_attention_rmsnorm", + "sequence": 440 + }, + { + "depends_on": [ + "decode0:op440:layer15.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer15.gate_projection", + "op_id": "decode0:op441:layer15.gate_projection", + "sequence": 441 + }, + { + "depends_on": [ + "decode0:op440:layer15.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer15.up_projection", + "op_id": "decode0:op442:layer15.up_projection", + "sequence": 442 + }, + { + "depends_on": [ + "decode0:op441:layer15.gate_projection", + "decode0:op442:layer15.up_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer15.swiglu_down_projection", + "op_id": "decode0:op443:layer15.swiglu_down_projection", + "sequence": 443 + }, + { + "depends_on": [ + "decode0:op443:layer15.swiglu_down_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer16.input_rmsnorm", + "op_id": "decode0:op444:layer16.input_rmsnorm", + "sequence": 444 + }, + { + "depends_on": [ + "decode0:op444:layer16.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer16.q_projection", + "op_id": "decode0:op445:layer16.q_projection", + "sequence": 445 + }, + { + "depends_on": [ + "decode0:op444:layer16.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer16.k_projection", + "op_id": "decode0:op446:layer16.k_projection", + "sequence": 446 + }, + { + "depends_on": [ + "decode0:op444:layer16.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer16.v_projection", + "op_id": "decode0:op447:layer16.v_projection", + "sequence": 447 + }, + { + "depends_on": [ + "decode0:op445:layer16.q_projection", + "decode0:op446:layer16.k_projection", + "decode0:op447:layer16.v_projection" + ], + "details": { + "context_tokens": 5, + "kv_repeat_groups": 7, + "kv_shape": [ + 1, + 4, + 2 + ], + "q_shape": [ + 1, + 28, + 2 + ] + }, + "forward_id": "decode0", + "name": "layer16.rope_gqa_causal_attention", + "op_id": "decode0:op448:layer16.rope_gqa_causal_attention", + "sequence": 448 + }, + { + "depends_on": [ + "decode0:op448:layer16.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer16.o_projection", + "op_id": "decode0:op449:layer16.o_projection", + "sequence": 449 + }, + { + "depends_on": [ + "decode0:op449:layer16.o_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer16.post_attention_rmsnorm", + "op_id": "decode0:op450:layer16.post_attention_rmsnorm", + "sequence": 450 + }, + { + "depends_on": [ + "decode0:op450:layer16.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer16.gate_projection", + "op_id": "decode0:op451:layer16.gate_projection", + "sequence": 451 + }, + { + "depends_on": [ + "decode0:op450:layer16.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer16.up_projection", + "op_id": "decode0:op452:layer16.up_projection", + "sequence": 452 + }, + { + "depends_on": [ + "decode0:op451:layer16.gate_projection", + "decode0:op452:layer16.up_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer16.swiglu_down_projection", + "op_id": "decode0:op453:layer16.swiglu_down_projection", + "sequence": 453 + }, + { + "depends_on": [ + "decode0:op453:layer16.swiglu_down_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer17.input_rmsnorm", + "op_id": "decode0:op454:layer17.input_rmsnorm", + "sequence": 454 + }, + { + "depends_on": [ + "decode0:op454:layer17.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer17.q_projection", + "op_id": "decode0:op455:layer17.q_projection", + "sequence": 455 + }, + { + "depends_on": [ + "decode0:op454:layer17.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer17.k_projection", + "op_id": "decode0:op456:layer17.k_projection", + "sequence": 456 + }, + { + "depends_on": [ + "decode0:op454:layer17.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer17.v_projection", + "op_id": "decode0:op457:layer17.v_projection", + "sequence": 457 + }, + { + "depends_on": [ + "decode0:op455:layer17.q_projection", + "decode0:op456:layer17.k_projection", + "decode0:op457:layer17.v_projection" + ], + "details": { + "context_tokens": 5, + "kv_repeat_groups": 7, + "kv_shape": [ + 1, + 4, + 2 + ], + "q_shape": [ + 1, + 28, + 2 + ] + }, + "forward_id": "decode0", + "name": "layer17.rope_gqa_causal_attention", + "op_id": "decode0:op458:layer17.rope_gqa_causal_attention", + "sequence": 458 + }, + { + "depends_on": [ + "decode0:op458:layer17.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer17.o_projection", + "op_id": "decode0:op459:layer17.o_projection", + "sequence": 459 + }, + { + "depends_on": [ + "decode0:op459:layer17.o_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer17.post_attention_rmsnorm", + "op_id": "decode0:op460:layer17.post_attention_rmsnorm", + "sequence": 460 + }, + { + "depends_on": [ + "decode0:op460:layer17.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer17.gate_projection", + "op_id": "decode0:op461:layer17.gate_projection", + "sequence": 461 + }, + { + "depends_on": [ + "decode0:op460:layer17.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer17.up_projection", + "op_id": "decode0:op462:layer17.up_projection", + "sequence": 462 + }, + { + "depends_on": [ + "decode0:op461:layer17.gate_projection", + "decode0:op462:layer17.up_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer17.swiglu_down_projection", + "op_id": "decode0:op463:layer17.swiglu_down_projection", + "sequence": 463 + }, + { + "depends_on": [ + "decode0:op463:layer17.swiglu_down_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer18.input_rmsnorm", + "op_id": "decode0:op464:layer18.input_rmsnorm", + "sequence": 464 + }, + { + "depends_on": [ + "decode0:op464:layer18.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer18.q_projection", + "op_id": "decode0:op465:layer18.q_projection", + "sequence": 465 + }, + { + "depends_on": [ + "decode0:op464:layer18.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer18.k_projection", + "op_id": "decode0:op466:layer18.k_projection", + "sequence": 466 + }, + { + "depends_on": [ + "decode0:op464:layer18.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer18.v_projection", + "op_id": "decode0:op467:layer18.v_projection", + "sequence": 467 + }, + { + "depends_on": [ + "decode0:op465:layer18.q_projection", + "decode0:op466:layer18.k_projection", + "decode0:op467:layer18.v_projection" + ], + "details": { + "context_tokens": 5, + "kv_repeat_groups": 7, + "kv_shape": [ + 1, + 4, + 2 + ], + "q_shape": [ + 1, + 28, + 2 + ] + }, + "forward_id": "decode0", + "name": "layer18.rope_gqa_causal_attention", + "op_id": "decode0:op468:layer18.rope_gqa_causal_attention", + "sequence": 468 + }, + { + "depends_on": [ + "decode0:op468:layer18.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer18.o_projection", + "op_id": "decode0:op469:layer18.o_projection", + "sequence": 469 + }, + { + "depends_on": [ + "decode0:op469:layer18.o_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer18.post_attention_rmsnorm", + "op_id": "decode0:op470:layer18.post_attention_rmsnorm", + "sequence": 470 + }, + { + "depends_on": [ + "decode0:op470:layer18.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer18.gate_projection", + "op_id": "decode0:op471:layer18.gate_projection", + "sequence": 471 + }, + { + "depends_on": [ + "decode0:op470:layer18.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer18.up_projection", + "op_id": "decode0:op472:layer18.up_projection", + "sequence": 472 + }, + { + "depends_on": [ + "decode0:op471:layer18.gate_projection", + "decode0:op472:layer18.up_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer18.swiglu_down_projection", + "op_id": "decode0:op473:layer18.swiglu_down_projection", + "sequence": 473 + }, + { + "depends_on": [ + "decode0:op473:layer18.swiglu_down_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer19.input_rmsnorm", + "op_id": "decode0:op474:layer19.input_rmsnorm", + "sequence": 474 + }, + { + "depends_on": [ + "decode0:op474:layer19.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer19.q_projection", + "op_id": "decode0:op475:layer19.q_projection", + "sequence": 475 + }, + { + "depends_on": [ + "decode0:op474:layer19.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer19.k_projection", + "op_id": "decode0:op476:layer19.k_projection", + "sequence": 476 + }, + { + "depends_on": [ + "decode0:op474:layer19.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer19.v_projection", + "op_id": "decode0:op477:layer19.v_projection", + "sequence": 477 + }, + { + "depends_on": [ + "decode0:op475:layer19.q_projection", + "decode0:op476:layer19.k_projection", + "decode0:op477:layer19.v_projection" + ], + "details": { + "context_tokens": 5, + "kv_repeat_groups": 7, + "kv_shape": [ + 1, + 4, + 2 + ], + "q_shape": [ + 1, + 28, + 2 + ] + }, + "forward_id": "decode0", + "name": "layer19.rope_gqa_causal_attention", + "op_id": "decode0:op478:layer19.rope_gqa_causal_attention", + "sequence": 478 + }, + { + "depends_on": [ + "decode0:op478:layer19.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer19.o_projection", + "op_id": "decode0:op479:layer19.o_projection", + "sequence": 479 + }, + { + "depends_on": [ + "decode0:op479:layer19.o_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer19.post_attention_rmsnorm", + "op_id": "decode0:op480:layer19.post_attention_rmsnorm", + "sequence": 480 + }, + { + "depends_on": [ + "decode0:op480:layer19.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer19.gate_projection", + "op_id": "decode0:op481:layer19.gate_projection", + "sequence": 481 + }, + { + "depends_on": [ + "decode0:op480:layer19.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer19.up_projection", + "op_id": "decode0:op482:layer19.up_projection", + "sequence": 482 + }, + { + "depends_on": [ + "decode0:op481:layer19.gate_projection", + "decode0:op482:layer19.up_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer19.swiglu_down_projection", + "op_id": "decode0:op483:layer19.swiglu_down_projection", + "sequence": 483 + }, + { + "depends_on": [ + "decode0:op483:layer19.swiglu_down_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer20.input_rmsnorm", + "op_id": "decode0:op484:layer20.input_rmsnorm", + "sequence": 484 + }, + { + "depends_on": [ + "decode0:op484:layer20.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer20.q_projection", + "op_id": "decode0:op485:layer20.q_projection", + "sequence": 485 + }, + { + "depends_on": [ + "decode0:op484:layer20.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer20.k_projection", + "op_id": "decode0:op486:layer20.k_projection", + "sequence": 486 + }, + { + "depends_on": [ + "decode0:op484:layer20.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer20.v_projection", + "op_id": "decode0:op487:layer20.v_projection", + "sequence": 487 + }, + { + "depends_on": [ + "decode0:op485:layer20.q_projection", + "decode0:op486:layer20.k_projection", + "decode0:op487:layer20.v_projection" + ], + "details": { + "context_tokens": 5, + "kv_repeat_groups": 7, + "kv_shape": [ + 1, + 4, + 2 + ], + "q_shape": [ + 1, + 28, + 2 + ] + }, + "forward_id": "decode0", + "name": "layer20.rope_gqa_causal_attention", + "op_id": "decode0:op488:layer20.rope_gqa_causal_attention", + "sequence": 488 + }, + { + "depends_on": [ + "decode0:op488:layer20.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer20.o_projection", + "op_id": "decode0:op489:layer20.o_projection", + "sequence": 489 + }, + { + "depends_on": [ + "decode0:op489:layer20.o_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer20.post_attention_rmsnorm", + "op_id": "decode0:op490:layer20.post_attention_rmsnorm", + "sequence": 490 + }, + { + "depends_on": [ + "decode0:op490:layer20.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer20.gate_projection", + "op_id": "decode0:op491:layer20.gate_projection", + "sequence": 491 + }, + { + "depends_on": [ + "decode0:op490:layer20.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer20.up_projection", + "op_id": "decode0:op492:layer20.up_projection", + "sequence": 492 + }, + { + "depends_on": [ + "decode0:op491:layer20.gate_projection", + "decode0:op492:layer20.up_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer20.swiglu_down_projection", + "op_id": "decode0:op493:layer20.swiglu_down_projection", + "sequence": 493 + }, + { + "depends_on": [ + "decode0:op493:layer20.swiglu_down_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer21.input_rmsnorm", + "op_id": "decode0:op494:layer21.input_rmsnorm", + "sequence": 494 + }, + { + "depends_on": [ + "decode0:op494:layer21.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer21.q_projection", + "op_id": "decode0:op495:layer21.q_projection", + "sequence": 495 + }, + { + "depends_on": [ + "decode0:op494:layer21.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer21.k_projection", + "op_id": "decode0:op496:layer21.k_projection", + "sequence": 496 + }, + { + "depends_on": [ + "decode0:op494:layer21.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer21.v_projection", + "op_id": "decode0:op497:layer21.v_projection", + "sequence": 497 + }, + { + "depends_on": [ + "decode0:op495:layer21.q_projection", + "decode0:op496:layer21.k_projection", + "decode0:op497:layer21.v_projection" + ], + "details": { + "context_tokens": 5, + "kv_repeat_groups": 7, + "kv_shape": [ + 1, + 4, + 2 + ], + "q_shape": [ + 1, + 28, + 2 + ] + }, + "forward_id": "decode0", + "name": "layer21.rope_gqa_causal_attention", + "op_id": "decode0:op498:layer21.rope_gqa_causal_attention", + "sequence": 498 + }, + { + "depends_on": [ + "decode0:op498:layer21.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer21.o_projection", + "op_id": "decode0:op499:layer21.o_projection", + "sequence": 499 + }, + { + "depends_on": [ + "decode0:op499:layer21.o_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer21.post_attention_rmsnorm", + "op_id": "decode0:op500:layer21.post_attention_rmsnorm", + "sequence": 500 + }, + { + "depends_on": [ + "decode0:op500:layer21.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer21.gate_projection", + "op_id": "decode0:op501:layer21.gate_projection", + "sequence": 501 + }, + { + "depends_on": [ + "decode0:op500:layer21.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer21.up_projection", + "op_id": "decode0:op502:layer21.up_projection", + "sequence": 502 + }, + { + "depends_on": [ + "decode0:op501:layer21.gate_projection", + "decode0:op502:layer21.up_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer21.swiglu_down_projection", + "op_id": "decode0:op503:layer21.swiglu_down_projection", + "sequence": 503 + }, + { + "depends_on": [ + "decode0:op503:layer21.swiglu_down_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer22.input_rmsnorm", + "op_id": "decode0:op504:layer22.input_rmsnorm", + "sequence": 504 + }, + { + "depends_on": [ + "decode0:op504:layer22.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer22.q_projection", + "op_id": "decode0:op505:layer22.q_projection", + "sequence": 505 + }, + { + "depends_on": [ + "decode0:op504:layer22.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer22.k_projection", + "op_id": "decode0:op506:layer22.k_projection", + "sequence": 506 + }, + { + "depends_on": [ + "decode0:op504:layer22.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer22.v_projection", + "op_id": "decode0:op507:layer22.v_projection", + "sequence": 507 + }, + { + "depends_on": [ + "decode0:op505:layer22.q_projection", + "decode0:op506:layer22.k_projection", + "decode0:op507:layer22.v_projection" + ], + "details": { + "context_tokens": 5, + "kv_repeat_groups": 7, + "kv_shape": [ + 1, + 4, + 2 + ], + "q_shape": [ + 1, + 28, + 2 + ] + }, + "forward_id": "decode0", + "name": "layer22.rope_gqa_causal_attention", + "op_id": "decode0:op508:layer22.rope_gqa_causal_attention", + "sequence": 508 + }, + { + "depends_on": [ + "decode0:op508:layer22.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer22.o_projection", + "op_id": "decode0:op509:layer22.o_projection", + "sequence": 509 + }, + { + "depends_on": [ + "decode0:op509:layer22.o_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer22.post_attention_rmsnorm", + "op_id": "decode0:op510:layer22.post_attention_rmsnorm", + "sequence": 510 + }, + { + "depends_on": [ + "decode0:op510:layer22.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer22.gate_projection", + "op_id": "decode0:op511:layer22.gate_projection", + "sequence": 511 + }, + { + "depends_on": [ + "decode0:op510:layer22.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer22.up_projection", + "op_id": "decode0:op512:layer22.up_projection", + "sequence": 512 + }, + { + "depends_on": [ + "decode0:op511:layer22.gate_projection", + "decode0:op512:layer22.up_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer22.swiglu_down_projection", + "op_id": "decode0:op513:layer22.swiglu_down_projection", + "sequence": 513 + }, + { + "depends_on": [ + "decode0:op513:layer22.swiglu_down_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer23.input_rmsnorm", + "op_id": "decode0:op514:layer23.input_rmsnorm", + "sequence": 514 + }, + { + "depends_on": [ + "decode0:op514:layer23.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer23.q_projection", + "op_id": "decode0:op515:layer23.q_projection", + "sequence": 515 + }, + { + "depends_on": [ + "decode0:op514:layer23.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer23.k_projection", + "op_id": "decode0:op516:layer23.k_projection", + "sequence": 516 + }, + { + "depends_on": [ + "decode0:op514:layer23.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer23.v_projection", + "op_id": "decode0:op517:layer23.v_projection", + "sequence": 517 + }, + { + "depends_on": [ + "decode0:op515:layer23.q_projection", + "decode0:op516:layer23.k_projection", + "decode0:op517:layer23.v_projection" + ], + "details": { + "context_tokens": 5, + "kv_repeat_groups": 7, + "kv_shape": [ + 1, + 4, + 2 + ], + "q_shape": [ + 1, + 28, + 2 + ] + }, + "forward_id": "decode0", + "name": "layer23.rope_gqa_causal_attention", + "op_id": "decode0:op518:layer23.rope_gqa_causal_attention", + "sequence": 518 + }, + { + "depends_on": [ + "decode0:op518:layer23.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer23.o_projection", + "op_id": "decode0:op519:layer23.o_projection", + "sequence": 519 + }, + { + "depends_on": [ + "decode0:op519:layer23.o_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer23.post_attention_rmsnorm", + "op_id": "decode0:op520:layer23.post_attention_rmsnorm", + "sequence": 520 + }, + { + "depends_on": [ + "decode0:op520:layer23.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer23.gate_projection", + "op_id": "decode0:op521:layer23.gate_projection", + "sequence": 521 + }, + { + "depends_on": [ + "decode0:op520:layer23.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer23.up_projection", + "op_id": "decode0:op522:layer23.up_projection", + "sequence": 522 + }, + { + "depends_on": [ + "decode0:op521:layer23.gate_projection", + "decode0:op522:layer23.up_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer23.swiglu_down_projection", + "op_id": "decode0:op523:layer23.swiglu_down_projection", + "sequence": 523 + }, + { + "depends_on": [ + "decode0:op523:layer23.swiglu_down_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer24.input_rmsnorm", + "op_id": "decode0:op524:layer24.input_rmsnorm", + "sequence": 524 + }, + { + "depends_on": [ + "decode0:op524:layer24.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer24.q_projection", + "op_id": "decode0:op525:layer24.q_projection", + "sequence": 525 + }, + { + "depends_on": [ + "decode0:op524:layer24.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer24.k_projection", + "op_id": "decode0:op526:layer24.k_projection", + "sequence": 526 + }, + { + "depends_on": [ + "decode0:op524:layer24.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer24.v_projection", + "op_id": "decode0:op527:layer24.v_projection", + "sequence": 527 + }, + { + "depends_on": [ + "decode0:op525:layer24.q_projection", + "decode0:op526:layer24.k_projection", + "decode0:op527:layer24.v_projection" + ], + "details": { + "context_tokens": 5, + "kv_repeat_groups": 7, + "kv_shape": [ + 1, + 4, + 2 + ], + "q_shape": [ + 1, + 28, + 2 + ] + }, + "forward_id": "decode0", + "name": "layer24.rope_gqa_causal_attention", + "op_id": "decode0:op528:layer24.rope_gqa_causal_attention", + "sequence": 528 + }, + { + "depends_on": [ + "decode0:op528:layer24.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer24.o_projection", + "op_id": "decode0:op529:layer24.o_projection", + "sequence": 529 + }, + { + "depends_on": [ + "decode0:op529:layer24.o_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer24.post_attention_rmsnorm", + "op_id": "decode0:op530:layer24.post_attention_rmsnorm", + "sequence": 530 + }, + { + "depends_on": [ + "decode0:op530:layer24.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer24.gate_projection", + "op_id": "decode0:op531:layer24.gate_projection", + "sequence": 531 + }, + { + "depends_on": [ + "decode0:op530:layer24.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer24.up_projection", + "op_id": "decode0:op532:layer24.up_projection", + "sequence": 532 + }, + { + "depends_on": [ + "decode0:op531:layer24.gate_projection", + "decode0:op532:layer24.up_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer24.swiglu_down_projection", + "op_id": "decode0:op533:layer24.swiglu_down_projection", + "sequence": 533 + }, + { + "depends_on": [ + "decode0:op533:layer24.swiglu_down_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer25.input_rmsnorm", + "op_id": "decode0:op534:layer25.input_rmsnorm", + "sequence": 534 + }, + { + "depends_on": [ + "decode0:op534:layer25.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer25.q_projection", + "op_id": "decode0:op535:layer25.q_projection", + "sequence": 535 + }, + { + "depends_on": [ + "decode0:op534:layer25.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer25.k_projection", + "op_id": "decode0:op536:layer25.k_projection", + "sequence": 536 + }, + { + "depends_on": [ + "decode0:op534:layer25.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer25.v_projection", + "op_id": "decode0:op537:layer25.v_projection", + "sequence": 537 + }, + { + "depends_on": [ + "decode0:op535:layer25.q_projection", + "decode0:op536:layer25.k_projection", + "decode0:op537:layer25.v_projection" + ], + "details": { + "context_tokens": 5, + "kv_repeat_groups": 7, + "kv_shape": [ + 1, + 4, + 2 + ], + "q_shape": [ + 1, + 28, + 2 + ] + }, + "forward_id": "decode0", + "name": "layer25.rope_gqa_causal_attention", + "op_id": "decode0:op538:layer25.rope_gqa_causal_attention", + "sequence": 538 + }, + { + "depends_on": [ + "decode0:op538:layer25.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer25.o_projection", + "op_id": "decode0:op539:layer25.o_projection", + "sequence": 539 + }, + { + "depends_on": [ + "decode0:op539:layer25.o_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer25.post_attention_rmsnorm", + "op_id": "decode0:op540:layer25.post_attention_rmsnorm", + "sequence": 540 + }, + { + "depends_on": [ + "decode0:op540:layer25.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer25.gate_projection", + "op_id": "decode0:op541:layer25.gate_projection", + "sequence": 541 + }, + { + "depends_on": [ + "decode0:op540:layer25.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer25.up_projection", + "op_id": "decode0:op542:layer25.up_projection", + "sequence": 542 + }, + { + "depends_on": [ + "decode0:op541:layer25.gate_projection", + "decode0:op542:layer25.up_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer25.swiglu_down_projection", + "op_id": "decode0:op543:layer25.swiglu_down_projection", + "sequence": 543 + }, + { + "depends_on": [ + "decode0:op543:layer25.swiglu_down_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer26.input_rmsnorm", + "op_id": "decode0:op544:layer26.input_rmsnorm", + "sequence": 544 + }, + { + "depends_on": [ + "decode0:op544:layer26.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer26.q_projection", + "op_id": "decode0:op545:layer26.q_projection", + "sequence": 545 + }, + { + "depends_on": [ + "decode0:op544:layer26.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer26.k_projection", + "op_id": "decode0:op546:layer26.k_projection", + "sequence": 546 + }, + { + "depends_on": [ + "decode0:op544:layer26.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer26.v_projection", + "op_id": "decode0:op547:layer26.v_projection", + "sequence": 547 + }, + { + "depends_on": [ + "decode0:op545:layer26.q_projection", + "decode0:op546:layer26.k_projection", + "decode0:op547:layer26.v_projection" + ], + "details": { + "context_tokens": 5, + "kv_repeat_groups": 7, + "kv_shape": [ + 1, + 4, + 2 + ], + "q_shape": [ + 1, + 28, + 2 + ] + }, + "forward_id": "decode0", + "name": "layer26.rope_gqa_causal_attention", + "op_id": "decode0:op548:layer26.rope_gqa_causal_attention", + "sequence": 548 + }, + { + "depends_on": [ + "decode0:op548:layer26.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer26.o_projection", + "op_id": "decode0:op549:layer26.o_projection", + "sequence": 549 + }, + { + "depends_on": [ + "decode0:op549:layer26.o_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer26.post_attention_rmsnorm", + "op_id": "decode0:op550:layer26.post_attention_rmsnorm", + "sequence": 550 + }, + { + "depends_on": [ + "decode0:op550:layer26.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer26.gate_projection", + "op_id": "decode0:op551:layer26.gate_projection", + "sequence": 551 + }, + { + "depends_on": [ + "decode0:op550:layer26.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer26.up_projection", + "op_id": "decode0:op552:layer26.up_projection", + "sequence": 552 + }, + { + "depends_on": [ + "decode0:op551:layer26.gate_projection", + "decode0:op552:layer26.up_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer26.swiglu_down_projection", + "op_id": "decode0:op553:layer26.swiglu_down_projection", + "sequence": 553 + }, + { + "depends_on": [ + "decode0:op553:layer26.swiglu_down_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer27.input_rmsnorm", + "op_id": "decode0:op554:layer27.input_rmsnorm", + "sequence": 554 + }, + { + "depends_on": [ + "decode0:op554:layer27.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer27.q_projection", + "op_id": "decode0:op555:layer27.q_projection", + "sequence": 555 + }, + { + "depends_on": [ + "decode0:op554:layer27.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer27.k_projection", + "op_id": "decode0:op556:layer27.k_projection", + "sequence": 556 + }, + { + "depends_on": [ + "decode0:op554:layer27.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer27.v_projection", + "op_id": "decode0:op557:layer27.v_projection", + "sequence": 557 + }, + { + "depends_on": [ + "decode0:op555:layer27.q_projection", + "decode0:op556:layer27.k_projection", + "decode0:op557:layer27.v_projection" + ], + "details": { + "context_tokens": 5, + "kv_repeat_groups": 7, + "kv_shape": [ + 1, + 4, + 2 + ], + "q_shape": [ + 1, + 28, + 2 + ] + }, + "forward_id": "decode0", + "name": "layer27.rope_gqa_causal_attention", + "op_id": "decode0:op558:layer27.rope_gqa_causal_attention", + "sequence": 558 + }, + { + "depends_on": [ + "decode0:op558:layer27.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer27.o_projection", + "op_id": "decode0:op559:layer27.o_projection", + "sequence": 559 + }, + { + "depends_on": [ + "decode0:op559:layer27.o_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer27.post_attention_rmsnorm", + "op_id": "decode0:op560:layer27.post_attention_rmsnorm", + "sequence": 560 + }, + { + "depends_on": [ + "decode0:op560:layer27.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer27.gate_projection", + "op_id": "decode0:op561:layer27.gate_projection", + "sequence": 561 + }, + { + "depends_on": [ + "decode0:op560:layer27.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer27.up_projection", + "op_id": "decode0:op562:layer27.up_projection", + "sequence": 562 + }, + { + "depends_on": [ + "decode0:op561:layer27.gate_projection", + "decode0:op562:layer27.up_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "layer27.swiglu_down_projection", + "op_id": "decode0:op563:layer27.swiglu_down_projection", + "sequence": 563 + }, + { + "depends_on": [ + "decode0:op563:layer27.swiglu_down_projection" + ], + "details": {}, + "forward_id": "decode0", + "name": "final_rmsnorm", + "op_id": "decode0:op564:final_rmsnorm", + "sequence": 564 + }, + { + "depends_on": [ + "decode0:op564:final_rmsnorm" + ], + "details": {}, + "forward_id": "decode0", + "name": "lm_head", + "op_id": "decode0:op565:lm_head", + "sequence": 565 + }, + { + "depends_on": [ + "token_ids" + ], + "details": { + "token_count": 1 + }, + "forward_id": "decode1", + "name": "embedding_lookup", + "op_id": "decode1:op566:embedding_lookup", + "sequence": 566 + }, + { + "depends_on": [ + "decode1:op566:embedding_lookup" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer0.input_rmsnorm", + "op_id": "decode1:op567:layer0.input_rmsnorm", + "sequence": 567 + }, + { + "depends_on": [ + "decode1:op567:layer0.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer0.q_projection", + "op_id": "decode1:op568:layer0.q_projection", + "sequence": 568 + }, + { + "depends_on": [ + "decode1:op567:layer0.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer0.k_projection", + "op_id": "decode1:op569:layer0.k_projection", + "sequence": 569 + }, + { + "depends_on": [ + "decode1:op567:layer0.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer0.v_projection", + "op_id": "decode1:op570:layer0.v_projection", + "sequence": 570 + }, + { + "depends_on": [ + "decode1:op568:layer0.q_projection", + "decode1:op569:layer0.k_projection", + "decode1:op570:layer0.v_projection" + ], + "details": { + "context_tokens": 6, + "kv_repeat_groups": 7, + "kv_shape": [ + 1, + 4, + 2 + ], + "q_shape": [ + 1, + 28, + 2 + ] + }, + "forward_id": "decode1", + "name": "layer0.rope_gqa_causal_attention", + "op_id": "decode1:op571:layer0.rope_gqa_causal_attention", + "sequence": 571 + }, + { + "depends_on": [ + "decode1:op571:layer0.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer0.o_projection", + "op_id": "decode1:op572:layer0.o_projection", + "sequence": 572 + }, + { + "depends_on": [ + "decode1:op572:layer0.o_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer0.post_attention_rmsnorm", + "op_id": "decode1:op573:layer0.post_attention_rmsnorm", + "sequence": 573 + }, + { + "depends_on": [ + "decode1:op573:layer0.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer0.gate_projection", + "op_id": "decode1:op574:layer0.gate_projection", + "sequence": 574 + }, + { + "depends_on": [ + "decode1:op573:layer0.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer0.up_projection", + "op_id": "decode1:op575:layer0.up_projection", + "sequence": 575 + }, + { + "depends_on": [ + "decode1:op574:layer0.gate_projection", + "decode1:op575:layer0.up_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer0.swiglu_down_projection", + "op_id": "decode1:op576:layer0.swiglu_down_projection", + "sequence": 576 + }, + { + "depends_on": [ + "decode1:op576:layer0.swiglu_down_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer1.input_rmsnorm", + "op_id": "decode1:op577:layer1.input_rmsnorm", + "sequence": 577 + }, + { + "depends_on": [ + "decode1:op577:layer1.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer1.q_projection", + "op_id": "decode1:op578:layer1.q_projection", + "sequence": 578 + }, + { + "depends_on": [ + "decode1:op577:layer1.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer1.k_projection", + "op_id": "decode1:op579:layer1.k_projection", + "sequence": 579 + }, + { + "depends_on": [ + "decode1:op577:layer1.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer1.v_projection", + "op_id": "decode1:op580:layer1.v_projection", + "sequence": 580 + }, + { + "depends_on": [ + "decode1:op578:layer1.q_projection", + "decode1:op579:layer1.k_projection", + "decode1:op580:layer1.v_projection" + ], + "details": { + "context_tokens": 6, + "kv_repeat_groups": 7, + "kv_shape": [ + 1, + 4, + 2 + ], + "q_shape": [ + 1, + 28, + 2 + ] + }, + "forward_id": "decode1", + "name": "layer1.rope_gqa_causal_attention", + "op_id": "decode1:op581:layer1.rope_gqa_causal_attention", + "sequence": 581 + }, + { + "depends_on": [ + "decode1:op581:layer1.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer1.o_projection", + "op_id": "decode1:op582:layer1.o_projection", + "sequence": 582 + }, + { + "depends_on": [ + "decode1:op582:layer1.o_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer1.post_attention_rmsnorm", + "op_id": "decode1:op583:layer1.post_attention_rmsnorm", + "sequence": 583 + }, + { + "depends_on": [ + "decode1:op583:layer1.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer1.gate_projection", + "op_id": "decode1:op584:layer1.gate_projection", + "sequence": 584 + }, + { + "depends_on": [ + "decode1:op583:layer1.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer1.up_projection", + "op_id": "decode1:op585:layer1.up_projection", + "sequence": 585 + }, + { + "depends_on": [ + "decode1:op584:layer1.gate_projection", + "decode1:op585:layer1.up_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer1.swiglu_down_projection", + "op_id": "decode1:op586:layer1.swiglu_down_projection", + "sequence": 586 + }, + { + "depends_on": [ + "decode1:op586:layer1.swiglu_down_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer2.input_rmsnorm", + "op_id": "decode1:op587:layer2.input_rmsnorm", + "sequence": 587 + }, + { + "depends_on": [ + "decode1:op587:layer2.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer2.q_projection", + "op_id": "decode1:op588:layer2.q_projection", + "sequence": 588 + }, + { + "depends_on": [ + "decode1:op587:layer2.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer2.k_projection", + "op_id": "decode1:op589:layer2.k_projection", + "sequence": 589 + }, + { + "depends_on": [ + "decode1:op587:layer2.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer2.v_projection", + "op_id": "decode1:op590:layer2.v_projection", + "sequence": 590 + }, + { + "depends_on": [ + "decode1:op588:layer2.q_projection", + "decode1:op589:layer2.k_projection", + "decode1:op590:layer2.v_projection" + ], + "details": { + "context_tokens": 6, + "kv_repeat_groups": 7, + "kv_shape": [ + 1, + 4, + 2 + ], + "q_shape": [ + 1, + 28, + 2 + ] + }, + "forward_id": "decode1", + "name": "layer2.rope_gqa_causal_attention", + "op_id": "decode1:op591:layer2.rope_gqa_causal_attention", + "sequence": 591 + }, + { + "depends_on": [ + "decode1:op591:layer2.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer2.o_projection", + "op_id": "decode1:op592:layer2.o_projection", + "sequence": 592 + }, + { + "depends_on": [ + "decode1:op592:layer2.o_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer2.post_attention_rmsnorm", + "op_id": "decode1:op593:layer2.post_attention_rmsnorm", + "sequence": 593 + }, + { + "depends_on": [ + "decode1:op593:layer2.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer2.gate_projection", + "op_id": "decode1:op594:layer2.gate_projection", + "sequence": 594 + }, + { + "depends_on": [ + "decode1:op593:layer2.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer2.up_projection", + "op_id": "decode1:op595:layer2.up_projection", + "sequence": 595 + }, + { + "depends_on": [ + "decode1:op594:layer2.gate_projection", + "decode1:op595:layer2.up_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer2.swiglu_down_projection", + "op_id": "decode1:op596:layer2.swiglu_down_projection", + "sequence": 596 + }, + { + "depends_on": [ + "decode1:op596:layer2.swiglu_down_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer3.input_rmsnorm", + "op_id": "decode1:op597:layer3.input_rmsnorm", + "sequence": 597 + }, + { + "depends_on": [ + "decode1:op597:layer3.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer3.q_projection", + "op_id": "decode1:op598:layer3.q_projection", + "sequence": 598 + }, + { + "depends_on": [ + "decode1:op597:layer3.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer3.k_projection", + "op_id": "decode1:op599:layer3.k_projection", + "sequence": 599 + }, + { + "depends_on": [ + "decode1:op597:layer3.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer3.v_projection", + "op_id": "decode1:op600:layer3.v_projection", + "sequence": 600 + }, + { + "depends_on": [ + "decode1:op598:layer3.q_projection", + "decode1:op599:layer3.k_projection", + "decode1:op600:layer3.v_projection" + ], + "details": { + "context_tokens": 6, + "kv_repeat_groups": 7, + "kv_shape": [ + 1, + 4, + 2 + ], + "q_shape": [ + 1, + 28, + 2 + ] + }, + "forward_id": "decode1", + "name": "layer3.rope_gqa_causal_attention", + "op_id": "decode1:op601:layer3.rope_gqa_causal_attention", + "sequence": 601 + }, + { + "depends_on": [ + "decode1:op601:layer3.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer3.o_projection", + "op_id": "decode1:op602:layer3.o_projection", + "sequence": 602 + }, + { + "depends_on": [ + "decode1:op602:layer3.o_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer3.post_attention_rmsnorm", + "op_id": "decode1:op603:layer3.post_attention_rmsnorm", + "sequence": 603 + }, + { + "depends_on": [ + "decode1:op603:layer3.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer3.gate_projection", + "op_id": "decode1:op604:layer3.gate_projection", + "sequence": 604 + }, + { + "depends_on": [ + "decode1:op603:layer3.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer3.up_projection", + "op_id": "decode1:op605:layer3.up_projection", + "sequence": 605 + }, + { + "depends_on": [ + "decode1:op604:layer3.gate_projection", + "decode1:op605:layer3.up_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer3.swiglu_down_projection", + "op_id": "decode1:op606:layer3.swiglu_down_projection", + "sequence": 606 + }, + { + "depends_on": [ + "decode1:op606:layer3.swiglu_down_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer4.input_rmsnorm", + "op_id": "decode1:op607:layer4.input_rmsnorm", + "sequence": 607 + }, + { + "depends_on": [ + "decode1:op607:layer4.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer4.q_projection", + "op_id": "decode1:op608:layer4.q_projection", + "sequence": 608 + }, + { + "depends_on": [ + "decode1:op607:layer4.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer4.k_projection", + "op_id": "decode1:op609:layer4.k_projection", + "sequence": 609 + }, + { + "depends_on": [ + "decode1:op607:layer4.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer4.v_projection", + "op_id": "decode1:op610:layer4.v_projection", + "sequence": 610 + }, + { + "depends_on": [ + "decode1:op608:layer4.q_projection", + "decode1:op609:layer4.k_projection", + "decode1:op610:layer4.v_projection" + ], + "details": { + "context_tokens": 6, + "kv_repeat_groups": 7, + "kv_shape": [ + 1, + 4, + 2 + ], + "q_shape": [ + 1, + 28, + 2 + ] + }, + "forward_id": "decode1", + "name": "layer4.rope_gqa_causal_attention", + "op_id": "decode1:op611:layer4.rope_gqa_causal_attention", + "sequence": 611 + }, + { + "depends_on": [ + "decode1:op611:layer4.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer4.o_projection", + "op_id": "decode1:op612:layer4.o_projection", + "sequence": 612 + }, + { + "depends_on": [ + "decode1:op612:layer4.o_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer4.post_attention_rmsnorm", + "op_id": "decode1:op613:layer4.post_attention_rmsnorm", + "sequence": 613 + }, + { + "depends_on": [ + "decode1:op613:layer4.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer4.gate_projection", + "op_id": "decode1:op614:layer4.gate_projection", + "sequence": 614 + }, + { + "depends_on": [ + "decode1:op613:layer4.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer4.up_projection", + "op_id": "decode1:op615:layer4.up_projection", + "sequence": 615 + }, + { + "depends_on": [ + "decode1:op614:layer4.gate_projection", + "decode1:op615:layer4.up_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer4.swiglu_down_projection", + "op_id": "decode1:op616:layer4.swiglu_down_projection", + "sequence": 616 + }, + { + "depends_on": [ + "decode1:op616:layer4.swiglu_down_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer5.input_rmsnorm", + "op_id": "decode1:op617:layer5.input_rmsnorm", + "sequence": 617 + }, + { + "depends_on": [ + "decode1:op617:layer5.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer5.q_projection", + "op_id": "decode1:op618:layer5.q_projection", + "sequence": 618 + }, + { + "depends_on": [ + "decode1:op617:layer5.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer5.k_projection", + "op_id": "decode1:op619:layer5.k_projection", + "sequence": 619 + }, + { + "depends_on": [ + "decode1:op617:layer5.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer5.v_projection", + "op_id": "decode1:op620:layer5.v_projection", + "sequence": 620 + }, + { + "depends_on": [ + "decode1:op618:layer5.q_projection", + "decode1:op619:layer5.k_projection", + "decode1:op620:layer5.v_projection" + ], + "details": { + "context_tokens": 6, + "kv_repeat_groups": 7, + "kv_shape": [ + 1, + 4, + 2 + ], + "q_shape": [ + 1, + 28, + 2 + ] + }, + "forward_id": "decode1", + "name": "layer5.rope_gqa_causal_attention", + "op_id": "decode1:op621:layer5.rope_gqa_causal_attention", + "sequence": 621 + }, + { + "depends_on": [ + "decode1:op621:layer5.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer5.o_projection", + "op_id": "decode1:op622:layer5.o_projection", + "sequence": 622 + }, + { + "depends_on": [ + "decode1:op622:layer5.o_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer5.post_attention_rmsnorm", + "op_id": "decode1:op623:layer5.post_attention_rmsnorm", + "sequence": 623 + }, + { + "depends_on": [ + "decode1:op623:layer5.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer5.gate_projection", + "op_id": "decode1:op624:layer5.gate_projection", + "sequence": 624 + }, + { + "depends_on": [ + "decode1:op623:layer5.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer5.up_projection", + "op_id": "decode1:op625:layer5.up_projection", + "sequence": 625 + }, + { + "depends_on": [ + "decode1:op624:layer5.gate_projection", + "decode1:op625:layer5.up_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer5.swiglu_down_projection", + "op_id": "decode1:op626:layer5.swiglu_down_projection", + "sequence": 626 + }, + { + "depends_on": [ + "decode1:op626:layer5.swiglu_down_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer6.input_rmsnorm", + "op_id": "decode1:op627:layer6.input_rmsnorm", + "sequence": 627 + }, + { + "depends_on": [ + "decode1:op627:layer6.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer6.q_projection", + "op_id": "decode1:op628:layer6.q_projection", + "sequence": 628 + }, + { + "depends_on": [ + "decode1:op627:layer6.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer6.k_projection", + "op_id": "decode1:op629:layer6.k_projection", + "sequence": 629 + }, + { + "depends_on": [ + "decode1:op627:layer6.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer6.v_projection", + "op_id": "decode1:op630:layer6.v_projection", + "sequence": 630 + }, + { + "depends_on": [ + "decode1:op628:layer6.q_projection", + "decode1:op629:layer6.k_projection", + "decode1:op630:layer6.v_projection" + ], + "details": { + "context_tokens": 6, + "kv_repeat_groups": 7, + "kv_shape": [ + 1, + 4, + 2 + ], + "q_shape": [ + 1, + 28, + 2 + ] + }, + "forward_id": "decode1", + "name": "layer6.rope_gqa_causal_attention", + "op_id": "decode1:op631:layer6.rope_gqa_causal_attention", + "sequence": 631 + }, + { + "depends_on": [ + "decode1:op631:layer6.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer6.o_projection", + "op_id": "decode1:op632:layer6.o_projection", + "sequence": 632 + }, + { + "depends_on": [ + "decode1:op632:layer6.o_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer6.post_attention_rmsnorm", + "op_id": "decode1:op633:layer6.post_attention_rmsnorm", + "sequence": 633 + }, + { + "depends_on": [ + "decode1:op633:layer6.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer6.gate_projection", + "op_id": "decode1:op634:layer6.gate_projection", + "sequence": 634 + }, + { + "depends_on": [ + "decode1:op633:layer6.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer6.up_projection", + "op_id": "decode1:op635:layer6.up_projection", + "sequence": 635 + }, + { + "depends_on": [ + "decode1:op634:layer6.gate_projection", + "decode1:op635:layer6.up_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer6.swiglu_down_projection", + "op_id": "decode1:op636:layer6.swiglu_down_projection", + "sequence": 636 + }, + { + "depends_on": [ + "decode1:op636:layer6.swiglu_down_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer7.input_rmsnorm", + "op_id": "decode1:op637:layer7.input_rmsnorm", + "sequence": 637 + }, + { + "depends_on": [ + "decode1:op637:layer7.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer7.q_projection", + "op_id": "decode1:op638:layer7.q_projection", + "sequence": 638 + }, + { + "depends_on": [ + "decode1:op637:layer7.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer7.k_projection", + "op_id": "decode1:op639:layer7.k_projection", + "sequence": 639 + }, + { + "depends_on": [ + "decode1:op637:layer7.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer7.v_projection", + "op_id": "decode1:op640:layer7.v_projection", + "sequence": 640 + }, + { + "depends_on": [ + "decode1:op638:layer7.q_projection", + "decode1:op639:layer7.k_projection", + "decode1:op640:layer7.v_projection" + ], + "details": { + "context_tokens": 6, + "kv_repeat_groups": 7, + "kv_shape": [ + 1, + 4, + 2 + ], + "q_shape": [ + 1, + 28, + 2 + ] + }, + "forward_id": "decode1", + "name": "layer7.rope_gqa_causal_attention", + "op_id": "decode1:op641:layer7.rope_gqa_causal_attention", + "sequence": 641 + }, + { + "depends_on": [ + "decode1:op641:layer7.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer7.o_projection", + "op_id": "decode1:op642:layer7.o_projection", + "sequence": 642 + }, + { + "depends_on": [ + "decode1:op642:layer7.o_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer7.post_attention_rmsnorm", + "op_id": "decode1:op643:layer7.post_attention_rmsnorm", + "sequence": 643 + }, + { + "depends_on": [ + "decode1:op643:layer7.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer7.gate_projection", + "op_id": "decode1:op644:layer7.gate_projection", + "sequence": 644 + }, + { + "depends_on": [ + "decode1:op643:layer7.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer7.up_projection", + "op_id": "decode1:op645:layer7.up_projection", + "sequence": 645 + }, + { + "depends_on": [ + "decode1:op644:layer7.gate_projection", + "decode1:op645:layer7.up_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer7.swiglu_down_projection", + "op_id": "decode1:op646:layer7.swiglu_down_projection", + "sequence": 646 + }, + { + "depends_on": [ + "decode1:op646:layer7.swiglu_down_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer8.input_rmsnorm", + "op_id": "decode1:op647:layer8.input_rmsnorm", + "sequence": 647 + }, + { + "depends_on": [ + "decode1:op647:layer8.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer8.q_projection", + "op_id": "decode1:op648:layer8.q_projection", + "sequence": 648 + }, + { + "depends_on": [ + "decode1:op647:layer8.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer8.k_projection", + "op_id": "decode1:op649:layer8.k_projection", + "sequence": 649 + }, + { + "depends_on": [ + "decode1:op647:layer8.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer8.v_projection", + "op_id": "decode1:op650:layer8.v_projection", + "sequence": 650 + }, + { + "depends_on": [ + "decode1:op648:layer8.q_projection", + "decode1:op649:layer8.k_projection", + "decode1:op650:layer8.v_projection" + ], + "details": { + "context_tokens": 6, + "kv_repeat_groups": 7, + "kv_shape": [ + 1, + 4, + 2 + ], + "q_shape": [ + 1, + 28, + 2 + ] + }, + "forward_id": "decode1", + "name": "layer8.rope_gqa_causal_attention", + "op_id": "decode1:op651:layer8.rope_gqa_causal_attention", + "sequence": 651 + }, + { + "depends_on": [ + "decode1:op651:layer8.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer8.o_projection", + "op_id": "decode1:op652:layer8.o_projection", + "sequence": 652 + }, + { + "depends_on": [ + "decode1:op652:layer8.o_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer8.post_attention_rmsnorm", + "op_id": "decode1:op653:layer8.post_attention_rmsnorm", + "sequence": 653 + }, + { + "depends_on": [ + "decode1:op653:layer8.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer8.gate_projection", + "op_id": "decode1:op654:layer8.gate_projection", + "sequence": 654 + }, + { + "depends_on": [ + "decode1:op653:layer8.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer8.up_projection", + "op_id": "decode1:op655:layer8.up_projection", + "sequence": 655 + }, + { + "depends_on": [ + "decode1:op654:layer8.gate_projection", + "decode1:op655:layer8.up_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer8.swiglu_down_projection", + "op_id": "decode1:op656:layer8.swiglu_down_projection", + "sequence": 656 + }, + { + "depends_on": [ + "decode1:op656:layer8.swiglu_down_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer9.input_rmsnorm", + "op_id": "decode1:op657:layer9.input_rmsnorm", + "sequence": 657 + }, + { + "depends_on": [ + "decode1:op657:layer9.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer9.q_projection", + "op_id": "decode1:op658:layer9.q_projection", + "sequence": 658 + }, + { + "depends_on": [ + "decode1:op657:layer9.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer9.k_projection", + "op_id": "decode1:op659:layer9.k_projection", + "sequence": 659 + }, + { + "depends_on": [ + "decode1:op657:layer9.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer9.v_projection", + "op_id": "decode1:op660:layer9.v_projection", + "sequence": 660 + }, + { + "depends_on": [ + "decode1:op658:layer9.q_projection", + "decode1:op659:layer9.k_projection", + "decode1:op660:layer9.v_projection" + ], + "details": { + "context_tokens": 6, + "kv_repeat_groups": 7, + "kv_shape": [ + 1, + 4, + 2 + ], + "q_shape": [ + 1, + 28, + 2 + ] + }, + "forward_id": "decode1", + "name": "layer9.rope_gqa_causal_attention", + "op_id": "decode1:op661:layer9.rope_gqa_causal_attention", + "sequence": 661 + }, + { + "depends_on": [ + "decode1:op661:layer9.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer9.o_projection", + "op_id": "decode1:op662:layer9.o_projection", + "sequence": 662 + }, + { + "depends_on": [ + "decode1:op662:layer9.o_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer9.post_attention_rmsnorm", + "op_id": "decode1:op663:layer9.post_attention_rmsnorm", + "sequence": 663 + }, + { + "depends_on": [ + "decode1:op663:layer9.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer9.gate_projection", + "op_id": "decode1:op664:layer9.gate_projection", + "sequence": 664 + }, + { + "depends_on": [ + "decode1:op663:layer9.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer9.up_projection", + "op_id": "decode1:op665:layer9.up_projection", + "sequence": 665 + }, + { + "depends_on": [ + "decode1:op664:layer9.gate_projection", + "decode1:op665:layer9.up_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer9.swiglu_down_projection", + "op_id": "decode1:op666:layer9.swiglu_down_projection", + "sequence": 666 + }, + { + "depends_on": [ + "decode1:op666:layer9.swiglu_down_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer10.input_rmsnorm", + "op_id": "decode1:op667:layer10.input_rmsnorm", + "sequence": 667 + }, + { + "depends_on": [ + "decode1:op667:layer10.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer10.q_projection", + "op_id": "decode1:op668:layer10.q_projection", + "sequence": 668 + }, + { + "depends_on": [ + "decode1:op667:layer10.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer10.k_projection", + "op_id": "decode1:op669:layer10.k_projection", + "sequence": 669 + }, + { + "depends_on": [ + "decode1:op667:layer10.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer10.v_projection", + "op_id": "decode1:op670:layer10.v_projection", + "sequence": 670 + }, + { + "depends_on": [ + "decode1:op668:layer10.q_projection", + "decode1:op669:layer10.k_projection", + "decode1:op670:layer10.v_projection" + ], + "details": { + "context_tokens": 6, + "kv_repeat_groups": 7, + "kv_shape": [ + 1, + 4, + 2 + ], + "q_shape": [ + 1, + 28, + 2 + ] + }, + "forward_id": "decode1", + "name": "layer10.rope_gqa_causal_attention", + "op_id": "decode1:op671:layer10.rope_gqa_causal_attention", + "sequence": 671 + }, + { + "depends_on": [ + "decode1:op671:layer10.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer10.o_projection", + "op_id": "decode1:op672:layer10.o_projection", + "sequence": 672 + }, + { + "depends_on": [ + "decode1:op672:layer10.o_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer10.post_attention_rmsnorm", + "op_id": "decode1:op673:layer10.post_attention_rmsnorm", + "sequence": 673 + }, + { + "depends_on": [ + "decode1:op673:layer10.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer10.gate_projection", + "op_id": "decode1:op674:layer10.gate_projection", + "sequence": 674 + }, + { + "depends_on": [ + "decode1:op673:layer10.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer10.up_projection", + "op_id": "decode1:op675:layer10.up_projection", + "sequence": 675 + }, + { + "depends_on": [ + "decode1:op674:layer10.gate_projection", + "decode1:op675:layer10.up_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer10.swiglu_down_projection", + "op_id": "decode1:op676:layer10.swiglu_down_projection", + "sequence": 676 + }, + { + "depends_on": [ + "decode1:op676:layer10.swiglu_down_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer11.input_rmsnorm", + "op_id": "decode1:op677:layer11.input_rmsnorm", + "sequence": 677 + }, + { + "depends_on": [ + "decode1:op677:layer11.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer11.q_projection", + "op_id": "decode1:op678:layer11.q_projection", + "sequence": 678 + }, + { + "depends_on": [ + "decode1:op677:layer11.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer11.k_projection", + "op_id": "decode1:op679:layer11.k_projection", + "sequence": 679 + }, + { + "depends_on": [ + "decode1:op677:layer11.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer11.v_projection", + "op_id": "decode1:op680:layer11.v_projection", + "sequence": 680 + }, + { + "depends_on": [ + "decode1:op678:layer11.q_projection", + "decode1:op679:layer11.k_projection", + "decode1:op680:layer11.v_projection" + ], + "details": { + "context_tokens": 6, + "kv_repeat_groups": 7, + "kv_shape": [ + 1, + 4, + 2 + ], + "q_shape": [ + 1, + 28, + 2 + ] + }, + "forward_id": "decode1", + "name": "layer11.rope_gqa_causal_attention", + "op_id": "decode1:op681:layer11.rope_gqa_causal_attention", + "sequence": 681 + }, + { + "depends_on": [ + "decode1:op681:layer11.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer11.o_projection", + "op_id": "decode1:op682:layer11.o_projection", + "sequence": 682 + }, + { + "depends_on": [ + "decode1:op682:layer11.o_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer11.post_attention_rmsnorm", + "op_id": "decode1:op683:layer11.post_attention_rmsnorm", + "sequence": 683 + }, + { + "depends_on": [ + "decode1:op683:layer11.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer11.gate_projection", + "op_id": "decode1:op684:layer11.gate_projection", + "sequence": 684 + }, + { + "depends_on": [ + "decode1:op683:layer11.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer11.up_projection", + "op_id": "decode1:op685:layer11.up_projection", + "sequence": 685 + }, + { + "depends_on": [ + "decode1:op684:layer11.gate_projection", + "decode1:op685:layer11.up_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer11.swiglu_down_projection", + "op_id": "decode1:op686:layer11.swiglu_down_projection", + "sequence": 686 + }, + { + "depends_on": [ + "decode1:op686:layer11.swiglu_down_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer12.input_rmsnorm", + "op_id": "decode1:op687:layer12.input_rmsnorm", + "sequence": 687 + }, + { + "depends_on": [ + "decode1:op687:layer12.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer12.q_projection", + "op_id": "decode1:op688:layer12.q_projection", + "sequence": 688 + }, + { + "depends_on": [ + "decode1:op687:layer12.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer12.k_projection", + "op_id": "decode1:op689:layer12.k_projection", + "sequence": 689 + }, + { + "depends_on": [ + "decode1:op687:layer12.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer12.v_projection", + "op_id": "decode1:op690:layer12.v_projection", + "sequence": 690 + }, + { + "depends_on": [ + "decode1:op688:layer12.q_projection", + "decode1:op689:layer12.k_projection", + "decode1:op690:layer12.v_projection" + ], + "details": { + "context_tokens": 6, + "kv_repeat_groups": 7, + "kv_shape": [ + 1, + 4, + 2 + ], + "q_shape": [ + 1, + 28, + 2 + ] + }, + "forward_id": "decode1", + "name": "layer12.rope_gqa_causal_attention", + "op_id": "decode1:op691:layer12.rope_gqa_causal_attention", + "sequence": 691 + }, + { + "depends_on": [ + "decode1:op691:layer12.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer12.o_projection", + "op_id": "decode1:op692:layer12.o_projection", + "sequence": 692 + }, + { + "depends_on": [ + "decode1:op692:layer12.o_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer12.post_attention_rmsnorm", + "op_id": "decode1:op693:layer12.post_attention_rmsnorm", + "sequence": 693 + }, + { + "depends_on": [ + "decode1:op693:layer12.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer12.gate_projection", + "op_id": "decode1:op694:layer12.gate_projection", + "sequence": 694 + }, + { + "depends_on": [ + "decode1:op693:layer12.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer12.up_projection", + "op_id": "decode1:op695:layer12.up_projection", + "sequence": 695 + }, + { + "depends_on": [ + "decode1:op694:layer12.gate_projection", + "decode1:op695:layer12.up_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer12.swiglu_down_projection", + "op_id": "decode1:op696:layer12.swiglu_down_projection", + "sequence": 696 + }, + { + "depends_on": [ + "decode1:op696:layer12.swiglu_down_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer13.input_rmsnorm", + "op_id": "decode1:op697:layer13.input_rmsnorm", + "sequence": 697 + }, + { + "depends_on": [ + "decode1:op697:layer13.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer13.q_projection", + "op_id": "decode1:op698:layer13.q_projection", + "sequence": 698 + }, + { + "depends_on": [ + "decode1:op697:layer13.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer13.k_projection", + "op_id": "decode1:op699:layer13.k_projection", + "sequence": 699 + }, + { + "depends_on": [ + "decode1:op697:layer13.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer13.v_projection", + "op_id": "decode1:op700:layer13.v_projection", + "sequence": 700 + }, + { + "depends_on": [ + "decode1:op698:layer13.q_projection", + "decode1:op699:layer13.k_projection", + "decode1:op700:layer13.v_projection" + ], + "details": { + "context_tokens": 6, + "kv_repeat_groups": 7, + "kv_shape": [ + 1, + 4, + 2 + ], + "q_shape": [ + 1, + 28, + 2 + ] + }, + "forward_id": "decode1", + "name": "layer13.rope_gqa_causal_attention", + "op_id": "decode1:op701:layer13.rope_gqa_causal_attention", + "sequence": 701 + }, + { + "depends_on": [ + "decode1:op701:layer13.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer13.o_projection", + "op_id": "decode1:op702:layer13.o_projection", + "sequence": 702 + }, + { + "depends_on": [ + "decode1:op702:layer13.o_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer13.post_attention_rmsnorm", + "op_id": "decode1:op703:layer13.post_attention_rmsnorm", + "sequence": 703 + }, + { + "depends_on": [ + "decode1:op703:layer13.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer13.gate_projection", + "op_id": "decode1:op704:layer13.gate_projection", + "sequence": 704 + }, + { + "depends_on": [ + "decode1:op703:layer13.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer13.up_projection", + "op_id": "decode1:op705:layer13.up_projection", + "sequence": 705 + }, + { + "depends_on": [ + "decode1:op704:layer13.gate_projection", + "decode1:op705:layer13.up_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer13.swiglu_down_projection", + "op_id": "decode1:op706:layer13.swiglu_down_projection", + "sequence": 706 + }, + { + "depends_on": [ + "decode1:op706:layer13.swiglu_down_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer14.input_rmsnorm", + "op_id": "decode1:op707:layer14.input_rmsnorm", + "sequence": 707 + }, + { + "depends_on": [ + "decode1:op707:layer14.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer14.q_projection", + "op_id": "decode1:op708:layer14.q_projection", + "sequence": 708 + }, + { + "depends_on": [ + "decode1:op707:layer14.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer14.k_projection", + "op_id": "decode1:op709:layer14.k_projection", + "sequence": 709 + }, + { + "depends_on": [ + "decode1:op707:layer14.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer14.v_projection", + "op_id": "decode1:op710:layer14.v_projection", + "sequence": 710 + }, + { + "depends_on": [ + "decode1:op708:layer14.q_projection", + "decode1:op709:layer14.k_projection", + "decode1:op710:layer14.v_projection" + ], + "details": { + "context_tokens": 6, + "kv_repeat_groups": 7, + "kv_shape": [ + 1, + 4, + 2 + ], + "q_shape": [ + 1, + 28, + 2 + ] + }, + "forward_id": "decode1", + "name": "layer14.rope_gqa_causal_attention", + "op_id": "decode1:op711:layer14.rope_gqa_causal_attention", + "sequence": 711 + }, + { + "depends_on": [ + "decode1:op711:layer14.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer14.o_projection", + "op_id": "decode1:op712:layer14.o_projection", + "sequence": 712 + }, + { + "depends_on": [ + "decode1:op712:layer14.o_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer14.post_attention_rmsnorm", + "op_id": "decode1:op713:layer14.post_attention_rmsnorm", + "sequence": 713 + }, + { + "depends_on": [ + "decode1:op713:layer14.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer14.gate_projection", + "op_id": "decode1:op714:layer14.gate_projection", + "sequence": 714 + }, + { + "depends_on": [ + "decode1:op713:layer14.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer14.up_projection", + "op_id": "decode1:op715:layer14.up_projection", + "sequence": 715 + }, + { + "depends_on": [ + "decode1:op714:layer14.gate_projection", + "decode1:op715:layer14.up_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer14.swiglu_down_projection", + "op_id": "decode1:op716:layer14.swiglu_down_projection", + "sequence": 716 + }, + { + "depends_on": [ + "decode1:op716:layer14.swiglu_down_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer15.input_rmsnorm", + "op_id": "decode1:op717:layer15.input_rmsnorm", + "sequence": 717 + }, + { + "depends_on": [ + "decode1:op717:layer15.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer15.q_projection", + "op_id": "decode1:op718:layer15.q_projection", + "sequence": 718 + }, + { + "depends_on": [ + "decode1:op717:layer15.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer15.k_projection", + "op_id": "decode1:op719:layer15.k_projection", + "sequence": 719 + }, + { + "depends_on": [ + "decode1:op717:layer15.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer15.v_projection", + "op_id": "decode1:op720:layer15.v_projection", + "sequence": 720 + }, + { + "depends_on": [ + "decode1:op718:layer15.q_projection", + "decode1:op719:layer15.k_projection", + "decode1:op720:layer15.v_projection" + ], + "details": { + "context_tokens": 6, + "kv_repeat_groups": 7, + "kv_shape": [ + 1, + 4, + 2 + ], + "q_shape": [ + 1, + 28, + 2 + ] + }, + "forward_id": "decode1", + "name": "layer15.rope_gqa_causal_attention", + "op_id": "decode1:op721:layer15.rope_gqa_causal_attention", + "sequence": 721 + }, + { + "depends_on": [ + "decode1:op721:layer15.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer15.o_projection", + "op_id": "decode1:op722:layer15.o_projection", + "sequence": 722 + }, + { + "depends_on": [ + "decode1:op722:layer15.o_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer15.post_attention_rmsnorm", + "op_id": "decode1:op723:layer15.post_attention_rmsnorm", + "sequence": 723 + }, + { + "depends_on": [ + "decode1:op723:layer15.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer15.gate_projection", + "op_id": "decode1:op724:layer15.gate_projection", + "sequence": 724 + }, + { + "depends_on": [ + "decode1:op723:layer15.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer15.up_projection", + "op_id": "decode1:op725:layer15.up_projection", + "sequence": 725 + }, + { + "depends_on": [ + "decode1:op724:layer15.gate_projection", + "decode1:op725:layer15.up_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer15.swiglu_down_projection", + "op_id": "decode1:op726:layer15.swiglu_down_projection", + "sequence": 726 + }, + { + "depends_on": [ + "decode1:op726:layer15.swiglu_down_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer16.input_rmsnorm", + "op_id": "decode1:op727:layer16.input_rmsnorm", + "sequence": 727 + }, + { + "depends_on": [ + "decode1:op727:layer16.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer16.q_projection", + "op_id": "decode1:op728:layer16.q_projection", + "sequence": 728 + }, + { + "depends_on": [ + "decode1:op727:layer16.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer16.k_projection", + "op_id": "decode1:op729:layer16.k_projection", + "sequence": 729 + }, + { + "depends_on": [ + "decode1:op727:layer16.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer16.v_projection", + "op_id": "decode1:op730:layer16.v_projection", + "sequence": 730 + }, + { + "depends_on": [ + "decode1:op728:layer16.q_projection", + "decode1:op729:layer16.k_projection", + "decode1:op730:layer16.v_projection" + ], + "details": { + "context_tokens": 6, + "kv_repeat_groups": 7, + "kv_shape": [ + 1, + 4, + 2 + ], + "q_shape": [ + 1, + 28, + 2 + ] + }, + "forward_id": "decode1", + "name": "layer16.rope_gqa_causal_attention", + "op_id": "decode1:op731:layer16.rope_gqa_causal_attention", + "sequence": 731 + }, + { + "depends_on": [ + "decode1:op731:layer16.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer16.o_projection", + "op_id": "decode1:op732:layer16.o_projection", + "sequence": 732 + }, + { + "depends_on": [ + "decode1:op732:layer16.o_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer16.post_attention_rmsnorm", + "op_id": "decode1:op733:layer16.post_attention_rmsnorm", + "sequence": 733 + }, + { + "depends_on": [ + "decode1:op733:layer16.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer16.gate_projection", + "op_id": "decode1:op734:layer16.gate_projection", + "sequence": 734 + }, + { + "depends_on": [ + "decode1:op733:layer16.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer16.up_projection", + "op_id": "decode1:op735:layer16.up_projection", + "sequence": 735 + }, + { + "depends_on": [ + "decode1:op734:layer16.gate_projection", + "decode1:op735:layer16.up_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer16.swiglu_down_projection", + "op_id": "decode1:op736:layer16.swiglu_down_projection", + "sequence": 736 + }, + { + "depends_on": [ + "decode1:op736:layer16.swiglu_down_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer17.input_rmsnorm", + "op_id": "decode1:op737:layer17.input_rmsnorm", + "sequence": 737 + }, + { + "depends_on": [ + "decode1:op737:layer17.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer17.q_projection", + "op_id": "decode1:op738:layer17.q_projection", + "sequence": 738 + }, + { + "depends_on": [ + "decode1:op737:layer17.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer17.k_projection", + "op_id": "decode1:op739:layer17.k_projection", + "sequence": 739 + }, + { + "depends_on": [ + "decode1:op737:layer17.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer17.v_projection", + "op_id": "decode1:op740:layer17.v_projection", + "sequence": 740 + }, + { + "depends_on": [ + "decode1:op738:layer17.q_projection", + "decode1:op739:layer17.k_projection", + "decode1:op740:layer17.v_projection" + ], + "details": { + "context_tokens": 6, + "kv_repeat_groups": 7, + "kv_shape": [ + 1, + 4, + 2 + ], + "q_shape": [ + 1, + 28, + 2 + ] + }, + "forward_id": "decode1", + "name": "layer17.rope_gqa_causal_attention", + "op_id": "decode1:op741:layer17.rope_gqa_causal_attention", + "sequence": 741 + }, + { + "depends_on": [ + "decode1:op741:layer17.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer17.o_projection", + "op_id": "decode1:op742:layer17.o_projection", + "sequence": 742 + }, + { + "depends_on": [ + "decode1:op742:layer17.o_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer17.post_attention_rmsnorm", + "op_id": "decode1:op743:layer17.post_attention_rmsnorm", + "sequence": 743 + }, + { + "depends_on": [ + "decode1:op743:layer17.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer17.gate_projection", + "op_id": "decode1:op744:layer17.gate_projection", + "sequence": 744 + }, + { + "depends_on": [ + "decode1:op743:layer17.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer17.up_projection", + "op_id": "decode1:op745:layer17.up_projection", + "sequence": 745 + }, + { + "depends_on": [ + "decode1:op744:layer17.gate_projection", + "decode1:op745:layer17.up_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer17.swiglu_down_projection", + "op_id": "decode1:op746:layer17.swiglu_down_projection", + "sequence": 746 + }, + { + "depends_on": [ + "decode1:op746:layer17.swiglu_down_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer18.input_rmsnorm", + "op_id": "decode1:op747:layer18.input_rmsnorm", + "sequence": 747 + }, + { + "depends_on": [ + "decode1:op747:layer18.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer18.q_projection", + "op_id": "decode1:op748:layer18.q_projection", + "sequence": 748 + }, + { + "depends_on": [ + "decode1:op747:layer18.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer18.k_projection", + "op_id": "decode1:op749:layer18.k_projection", + "sequence": 749 + }, + { + "depends_on": [ + "decode1:op747:layer18.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer18.v_projection", + "op_id": "decode1:op750:layer18.v_projection", + "sequence": 750 + }, + { + "depends_on": [ + "decode1:op748:layer18.q_projection", + "decode1:op749:layer18.k_projection", + "decode1:op750:layer18.v_projection" + ], + "details": { + "context_tokens": 6, + "kv_repeat_groups": 7, + "kv_shape": [ + 1, + 4, + 2 + ], + "q_shape": [ + 1, + 28, + 2 + ] + }, + "forward_id": "decode1", + "name": "layer18.rope_gqa_causal_attention", + "op_id": "decode1:op751:layer18.rope_gqa_causal_attention", + "sequence": 751 + }, + { + "depends_on": [ + "decode1:op751:layer18.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer18.o_projection", + "op_id": "decode1:op752:layer18.o_projection", + "sequence": 752 + }, + { + "depends_on": [ + "decode1:op752:layer18.o_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer18.post_attention_rmsnorm", + "op_id": "decode1:op753:layer18.post_attention_rmsnorm", + "sequence": 753 + }, + { + "depends_on": [ + "decode1:op753:layer18.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer18.gate_projection", + "op_id": "decode1:op754:layer18.gate_projection", + "sequence": 754 + }, + { + "depends_on": [ + "decode1:op753:layer18.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer18.up_projection", + "op_id": "decode1:op755:layer18.up_projection", + "sequence": 755 + }, + { + "depends_on": [ + "decode1:op754:layer18.gate_projection", + "decode1:op755:layer18.up_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer18.swiglu_down_projection", + "op_id": "decode1:op756:layer18.swiglu_down_projection", + "sequence": 756 + }, + { + "depends_on": [ + "decode1:op756:layer18.swiglu_down_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer19.input_rmsnorm", + "op_id": "decode1:op757:layer19.input_rmsnorm", + "sequence": 757 + }, + { + "depends_on": [ + "decode1:op757:layer19.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer19.q_projection", + "op_id": "decode1:op758:layer19.q_projection", + "sequence": 758 + }, + { + "depends_on": [ + "decode1:op757:layer19.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer19.k_projection", + "op_id": "decode1:op759:layer19.k_projection", + "sequence": 759 + }, + { + "depends_on": [ + "decode1:op757:layer19.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer19.v_projection", + "op_id": "decode1:op760:layer19.v_projection", + "sequence": 760 + }, + { + "depends_on": [ + "decode1:op758:layer19.q_projection", + "decode1:op759:layer19.k_projection", + "decode1:op760:layer19.v_projection" + ], + "details": { + "context_tokens": 6, + "kv_repeat_groups": 7, + "kv_shape": [ + 1, + 4, + 2 + ], + "q_shape": [ + 1, + 28, + 2 + ] + }, + "forward_id": "decode1", + "name": "layer19.rope_gqa_causal_attention", + "op_id": "decode1:op761:layer19.rope_gqa_causal_attention", + "sequence": 761 + }, + { + "depends_on": [ + "decode1:op761:layer19.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer19.o_projection", + "op_id": "decode1:op762:layer19.o_projection", + "sequence": 762 + }, + { + "depends_on": [ + "decode1:op762:layer19.o_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer19.post_attention_rmsnorm", + "op_id": "decode1:op763:layer19.post_attention_rmsnorm", + "sequence": 763 + }, + { + "depends_on": [ + "decode1:op763:layer19.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer19.gate_projection", + "op_id": "decode1:op764:layer19.gate_projection", + "sequence": 764 + }, + { + "depends_on": [ + "decode1:op763:layer19.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer19.up_projection", + "op_id": "decode1:op765:layer19.up_projection", + "sequence": 765 + }, + { + "depends_on": [ + "decode1:op764:layer19.gate_projection", + "decode1:op765:layer19.up_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer19.swiglu_down_projection", + "op_id": "decode1:op766:layer19.swiglu_down_projection", + "sequence": 766 + }, + { + "depends_on": [ + "decode1:op766:layer19.swiglu_down_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer20.input_rmsnorm", + "op_id": "decode1:op767:layer20.input_rmsnorm", + "sequence": 767 + }, + { + "depends_on": [ + "decode1:op767:layer20.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer20.q_projection", + "op_id": "decode1:op768:layer20.q_projection", + "sequence": 768 + }, + { + "depends_on": [ + "decode1:op767:layer20.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer20.k_projection", + "op_id": "decode1:op769:layer20.k_projection", + "sequence": 769 + }, + { + "depends_on": [ + "decode1:op767:layer20.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer20.v_projection", + "op_id": "decode1:op770:layer20.v_projection", + "sequence": 770 + }, + { + "depends_on": [ + "decode1:op768:layer20.q_projection", + "decode1:op769:layer20.k_projection", + "decode1:op770:layer20.v_projection" + ], + "details": { + "context_tokens": 6, + "kv_repeat_groups": 7, + "kv_shape": [ + 1, + 4, + 2 + ], + "q_shape": [ + 1, + 28, + 2 + ] + }, + "forward_id": "decode1", + "name": "layer20.rope_gqa_causal_attention", + "op_id": "decode1:op771:layer20.rope_gqa_causal_attention", + "sequence": 771 + }, + { + "depends_on": [ + "decode1:op771:layer20.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer20.o_projection", + "op_id": "decode1:op772:layer20.o_projection", + "sequence": 772 + }, + { + "depends_on": [ + "decode1:op772:layer20.o_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer20.post_attention_rmsnorm", + "op_id": "decode1:op773:layer20.post_attention_rmsnorm", + "sequence": 773 + }, + { + "depends_on": [ + "decode1:op773:layer20.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer20.gate_projection", + "op_id": "decode1:op774:layer20.gate_projection", + "sequence": 774 + }, + { + "depends_on": [ + "decode1:op773:layer20.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer20.up_projection", + "op_id": "decode1:op775:layer20.up_projection", + "sequence": 775 + }, + { + "depends_on": [ + "decode1:op774:layer20.gate_projection", + "decode1:op775:layer20.up_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer20.swiglu_down_projection", + "op_id": "decode1:op776:layer20.swiglu_down_projection", + "sequence": 776 + }, + { + "depends_on": [ + "decode1:op776:layer20.swiglu_down_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer21.input_rmsnorm", + "op_id": "decode1:op777:layer21.input_rmsnorm", + "sequence": 777 + }, + { + "depends_on": [ + "decode1:op777:layer21.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer21.q_projection", + "op_id": "decode1:op778:layer21.q_projection", + "sequence": 778 + }, + { + "depends_on": [ + "decode1:op777:layer21.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer21.k_projection", + "op_id": "decode1:op779:layer21.k_projection", + "sequence": 779 + }, + { + "depends_on": [ + "decode1:op777:layer21.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer21.v_projection", + "op_id": "decode1:op780:layer21.v_projection", + "sequence": 780 + }, + { + "depends_on": [ + "decode1:op778:layer21.q_projection", + "decode1:op779:layer21.k_projection", + "decode1:op780:layer21.v_projection" + ], + "details": { + "context_tokens": 6, + "kv_repeat_groups": 7, + "kv_shape": [ + 1, + 4, + 2 + ], + "q_shape": [ + 1, + 28, + 2 + ] + }, + "forward_id": "decode1", + "name": "layer21.rope_gqa_causal_attention", + "op_id": "decode1:op781:layer21.rope_gqa_causal_attention", + "sequence": 781 + }, + { + "depends_on": [ + "decode1:op781:layer21.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer21.o_projection", + "op_id": "decode1:op782:layer21.o_projection", + "sequence": 782 + }, + { + "depends_on": [ + "decode1:op782:layer21.o_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer21.post_attention_rmsnorm", + "op_id": "decode1:op783:layer21.post_attention_rmsnorm", + "sequence": 783 + }, + { + "depends_on": [ + "decode1:op783:layer21.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer21.gate_projection", + "op_id": "decode1:op784:layer21.gate_projection", + "sequence": 784 + }, + { + "depends_on": [ + "decode1:op783:layer21.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer21.up_projection", + "op_id": "decode1:op785:layer21.up_projection", + "sequence": 785 + }, + { + "depends_on": [ + "decode1:op784:layer21.gate_projection", + "decode1:op785:layer21.up_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer21.swiglu_down_projection", + "op_id": "decode1:op786:layer21.swiglu_down_projection", + "sequence": 786 + }, + { + "depends_on": [ + "decode1:op786:layer21.swiglu_down_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer22.input_rmsnorm", + "op_id": "decode1:op787:layer22.input_rmsnorm", + "sequence": 787 + }, + { + "depends_on": [ + "decode1:op787:layer22.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer22.q_projection", + "op_id": "decode1:op788:layer22.q_projection", + "sequence": 788 + }, + { + "depends_on": [ + "decode1:op787:layer22.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer22.k_projection", + "op_id": "decode1:op789:layer22.k_projection", + "sequence": 789 + }, + { + "depends_on": [ + "decode1:op787:layer22.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer22.v_projection", + "op_id": "decode1:op790:layer22.v_projection", + "sequence": 790 + }, + { + "depends_on": [ + "decode1:op788:layer22.q_projection", + "decode1:op789:layer22.k_projection", + "decode1:op790:layer22.v_projection" + ], + "details": { + "context_tokens": 6, + "kv_repeat_groups": 7, + "kv_shape": [ + 1, + 4, + 2 + ], + "q_shape": [ + 1, + 28, + 2 + ] + }, + "forward_id": "decode1", + "name": "layer22.rope_gqa_causal_attention", + "op_id": "decode1:op791:layer22.rope_gqa_causal_attention", + "sequence": 791 + }, + { + "depends_on": [ + "decode1:op791:layer22.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer22.o_projection", + "op_id": "decode1:op792:layer22.o_projection", + "sequence": 792 + }, + { + "depends_on": [ + "decode1:op792:layer22.o_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer22.post_attention_rmsnorm", + "op_id": "decode1:op793:layer22.post_attention_rmsnorm", + "sequence": 793 + }, + { + "depends_on": [ + "decode1:op793:layer22.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer22.gate_projection", + "op_id": "decode1:op794:layer22.gate_projection", + "sequence": 794 + }, + { + "depends_on": [ + "decode1:op793:layer22.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer22.up_projection", + "op_id": "decode1:op795:layer22.up_projection", + "sequence": 795 + }, + { + "depends_on": [ + "decode1:op794:layer22.gate_projection", + "decode1:op795:layer22.up_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer22.swiglu_down_projection", + "op_id": "decode1:op796:layer22.swiglu_down_projection", + "sequence": 796 + }, + { + "depends_on": [ + "decode1:op796:layer22.swiglu_down_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer23.input_rmsnorm", + "op_id": "decode1:op797:layer23.input_rmsnorm", + "sequence": 797 + }, + { + "depends_on": [ + "decode1:op797:layer23.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer23.q_projection", + "op_id": "decode1:op798:layer23.q_projection", + "sequence": 798 + }, + { + "depends_on": [ + "decode1:op797:layer23.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer23.k_projection", + "op_id": "decode1:op799:layer23.k_projection", + "sequence": 799 + }, + { + "depends_on": [ + "decode1:op797:layer23.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer23.v_projection", + "op_id": "decode1:op800:layer23.v_projection", + "sequence": 800 + }, + { + "depends_on": [ + "decode1:op798:layer23.q_projection", + "decode1:op799:layer23.k_projection", + "decode1:op800:layer23.v_projection" + ], + "details": { + "context_tokens": 6, + "kv_repeat_groups": 7, + "kv_shape": [ + 1, + 4, + 2 + ], + "q_shape": [ + 1, + 28, + 2 + ] + }, + "forward_id": "decode1", + "name": "layer23.rope_gqa_causal_attention", + "op_id": "decode1:op801:layer23.rope_gqa_causal_attention", + "sequence": 801 + }, + { + "depends_on": [ + "decode1:op801:layer23.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer23.o_projection", + "op_id": "decode1:op802:layer23.o_projection", + "sequence": 802 + }, + { + "depends_on": [ + "decode1:op802:layer23.o_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer23.post_attention_rmsnorm", + "op_id": "decode1:op803:layer23.post_attention_rmsnorm", + "sequence": 803 + }, + { + "depends_on": [ + "decode1:op803:layer23.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer23.gate_projection", + "op_id": "decode1:op804:layer23.gate_projection", + "sequence": 804 + }, + { + "depends_on": [ + "decode1:op803:layer23.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer23.up_projection", + "op_id": "decode1:op805:layer23.up_projection", + "sequence": 805 + }, + { + "depends_on": [ + "decode1:op804:layer23.gate_projection", + "decode1:op805:layer23.up_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer23.swiglu_down_projection", + "op_id": "decode1:op806:layer23.swiglu_down_projection", + "sequence": 806 + }, + { + "depends_on": [ + "decode1:op806:layer23.swiglu_down_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer24.input_rmsnorm", + "op_id": "decode1:op807:layer24.input_rmsnorm", + "sequence": 807 + }, + { + "depends_on": [ + "decode1:op807:layer24.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer24.q_projection", + "op_id": "decode1:op808:layer24.q_projection", + "sequence": 808 + }, + { + "depends_on": [ + "decode1:op807:layer24.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer24.k_projection", + "op_id": "decode1:op809:layer24.k_projection", + "sequence": 809 + }, + { + "depends_on": [ + "decode1:op807:layer24.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer24.v_projection", + "op_id": "decode1:op810:layer24.v_projection", + "sequence": 810 + }, + { + "depends_on": [ + "decode1:op808:layer24.q_projection", + "decode1:op809:layer24.k_projection", + "decode1:op810:layer24.v_projection" + ], + "details": { + "context_tokens": 6, + "kv_repeat_groups": 7, + "kv_shape": [ + 1, + 4, + 2 + ], + "q_shape": [ + 1, + 28, + 2 + ] + }, + "forward_id": "decode1", + "name": "layer24.rope_gqa_causal_attention", + "op_id": "decode1:op811:layer24.rope_gqa_causal_attention", + "sequence": 811 + }, + { + "depends_on": [ + "decode1:op811:layer24.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer24.o_projection", + "op_id": "decode1:op812:layer24.o_projection", + "sequence": 812 + }, + { + "depends_on": [ + "decode1:op812:layer24.o_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer24.post_attention_rmsnorm", + "op_id": "decode1:op813:layer24.post_attention_rmsnorm", + "sequence": 813 + }, + { + "depends_on": [ + "decode1:op813:layer24.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer24.gate_projection", + "op_id": "decode1:op814:layer24.gate_projection", + "sequence": 814 + }, + { + "depends_on": [ + "decode1:op813:layer24.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer24.up_projection", + "op_id": "decode1:op815:layer24.up_projection", + "sequence": 815 + }, + { + "depends_on": [ + "decode1:op814:layer24.gate_projection", + "decode1:op815:layer24.up_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer24.swiglu_down_projection", + "op_id": "decode1:op816:layer24.swiglu_down_projection", + "sequence": 816 + }, + { + "depends_on": [ + "decode1:op816:layer24.swiglu_down_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer25.input_rmsnorm", + "op_id": "decode1:op817:layer25.input_rmsnorm", + "sequence": 817 + }, + { + "depends_on": [ + "decode1:op817:layer25.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer25.q_projection", + "op_id": "decode1:op818:layer25.q_projection", + "sequence": 818 + }, + { + "depends_on": [ + "decode1:op817:layer25.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer25.k_projection", + "op_id": "decode1:op819:layer25.k_projection", + "sequence": 819 + }, + { + "depends_on": [ + "decode1:op817:layer25.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer25.v_projection", + "op_id": "decode1:op820:layer25.v_projection", + "sequence": 820 + }, + { + "depends_on": [ + "decode1:op818:layer25.q_projection", + "decode1:op819:layer25.k_projection", + "decode1:op820:layer25.v_projection" + ], + "details": { + "context_tokens": 6, + "kv_repeat_groups": 7, + "kv_shape": [ + 1, + 4, + 2 + ], + "q_shape": [ + 1, + 28, + 2 + ] + }, + "forward_id": "decode1", + "name": "layer25.rope_gqa_causal_attention", + "op_id": "decode1:op821:layer25.rope_gqa_causal_attention", + "sequence": 821 + }, + { + "depends_on": [ + "decode1:op821:layer25.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer25.o_projection", + "op_id": "decode1:op822:layer25.o_projection", + "sequence": 822 + }, + { + "depends_on": [ + "decode1:op822:layer25.o_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer25.post_attention_rmsnorm", + "op_id": "decode1:op823:layer25.post_attention_rmsnorm", + "sequence": 823 + }, + { + "depends_on": [ + "decode1:op823:layer25.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer25.gate_projection", + "op_id": "decode1:op824:layer25.gate_projection", + "sequence": 824 + }, + { + "depends_on": [ + "decode1:op823:layer25.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer25.up_projection", + "op_id": "decode1:op825:layer25.up_projection", + "sequence": 825 + }, + { + "depends_on": [ + "decode1:op824:layer25.gate_projection", + "decode1:op825:layer25.up_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer25.swiglu_down_projection", + "op_id": "decode1:op826:layer25.swiglu_down_projection", + "sequence": 826 + }, + { + "depends_on": [ + "decode1:op826:layer25.swiglu_down_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer26.input_rmsnorm", + "op_id": "decode1:op827:layer26.input_rmsnorm", + "sequence": 827 + }, + { + "depends_on": [ + "decode1:op827:layer26.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer26.q_projection", + "op_id": "decode1:op828:layer26.q_projection", + "sequence": 828 + }, + { + "depends_on": [ + "decode1:op827:layer26.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer26.k_projection", + "op_id": "decode1:op829:layer26.k_projection", + "sequence": 829 + }, + { + "depends_on": [ + "decode1:op827:layer26.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer26.v_projection", + "op_id": "decode1:op830:layer26.v_projection", + "sequence": 830 + }, + { + "depends_on": [ + "decode1:op828:layer26.q_projection", + "decode1:op829:layer26.k_projection", + "decode1:op830:layer26.v_projection" + ], + "details": { + "context_tokens": 6, + "kv_repeat_groups": 7, + "kv_shape": [ + 1, + 4, + 2 + ], + "q_shape": [ + 1, + 28, + 2 + ] + }, + "forward_id": "decode1", + "name": "layer26.rope_gqa_causal_attention", + "op_id": "decode1:op831:layer26.rope_gqa_causal_attention", + "sequence": 831 + }, + { + "depends_on": [ + "decode1:op831:layer26.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer26.o_projection", + "op_id": "decode1:op832:layer26.o_projection", + "sequence": 832 + }, + { + "depends_on": [ + "decode1:op832:layer26.o_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer26.post_attention_rmsnorm", + "op_id": "decode1:op833:layer26.post_attention_rmsnorm", + "sequence": 833 + }, + { + "depends_on": [ + "decode1:op833:layer26.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer26.gate_projection", + "op_id": "decode1:op834:layer26.gate_projection", + "sequence": 834 + }, + { + "depends_on": [ + "decode1:op833:layer26.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer26.up_projection", + "op_id": "decode1:op835:layer26.up_projection", + "sequence": 835 + }, + { + "depends_on": [ + "decode1:op834:layer26.gate_projection", + "decode1:op835:layer26.up_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer26.swiglu_down_projection", + "op_id": "decode1:op836:layer26.swiglu_down_projection", + "sequence": 836 + }, + { + "depends_on": [ + "decode1:op836:layer26.swiglu_down_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer27.input_rmsnorm", + "op_id": "decode1:op837:layer27.input_rmsnorm", + "sequence": 837 + }, + { + "depends_on": [ + "decode1:op837:layer27.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer27.q_projection", + "op_id": "decode1:op838:layer27.q_projection", + "sequence": 838 + }, + { + "depends_on": [ + "decode1:op837:layer27.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer27.k_projection", + "op_id": "decode1:op839:layer27.k_projection", + "sequence": 839 + }, + { + "depends_on": [ + "decode1:op837:layer27.input_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer27.v_projection", + "op_id": "decode1:op840:layer27.v_projection", + "sequence": 840 + }, + { + "depends_on": [ + "decode1:op838:layer27.q_projection", + "decode1:op839:layer27.k_projection", + "decode1:op840:layer27.v_projection" + ], + "details": { + "context_tokens": 6, + "kv_repeat_groups": 7, + "kv_shape": [ + 1, + 4, + 2 + ], + "q_shape": [ + 1, + 28, + 2 + ] + }, + "forward_id": "decode1", + "name": "layer27.rope_gqa_causal_attention", + "op_id": "decode1:op841:layer27.rope_gqa_causal_attention", + "sequence": 841 + }, + { + "depends_on": [ + "decode1:op841:layer27.rope_gqa_causal_attention" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer27.o_projection", + "op_id": "decode1:op842:layer27.o_projection", + "sequence": 842 + }, + { + "depends_on": [ + "decode1:op842:layer27.o_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer27.post_attention_rmsnorm", + "op_id": "decode1:op843:layer27.post_attention_rmsnorm", + "sequence": 843 + }, + { + "depends_on": [ + "decode1:op843:layer27.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer27.gate_projection", + "op_id": "decode1:op844:layer27.gate_projection", + "sequence": 844 + }, + { + "depends_on": [ + "decode1:op843:layer27.post_attention_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer27.up_projection", + "op_id": "decode1:op845:layer27.up_projection", + "sequence": 845 + }, + { + "depends_on": [ + "decode1:op844:layer27.gate_projection", + "decode1:op845:layer27.up_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "layer27.swiglu_down_projection", + "op_id": "decode1:op846:layer27.swiglu_down_projection", + "sequence": 846 + }, + { + "depends_on": [ + "decode1:op846:layer27.swiglu_down_projection" + ], + "details": {}, + "forward_id": "decode1", + "name": "final_rmsnorm", + "op_id": "decode1:op847:final_rmsnorm", + "sequence": 847 + }, + { + "depends_on": [ + "decode1:op847:final_rmsnorm" + ], + "details": {}, + "forward_id": "decode1", + "name": "lm_head", + "op_id": "decode1:op848:lm_head", + "sequence": 848 + } + ], + "schema_version": "eq3-tiny-qwen2-cpu-trace-v1", + "source_sha256": "8fc4329659d74a23d305d35137cc473805e4e005c8a0fa0bd0f1ee429cc91e30", + "target_projection": { + "Qwen/Qwen2.5-72B-Instruct": { + "analytical_macs_per_token_at_context": 75581358080, + "compute_cost_semantics": "ANALYTICAL_DENSE_MAC_COUNT_EXCLUDES_NORMS_ROPE_SOFTMAX_AND_RUNTIME", + "context_tokens": 4096, + "logical_region_address_semantics": "DERIVED_1MIB_ALIGNED_SCENARIO_NOT_SAFETENSORS_FILE_OFFSETS", + "logical_regions": [ + { + "logical_address_bytes": 0, + "name": "model.embed_tokens", + "payload_bytes": 2491416576 + }, + { + "logical_address_bytes": 2491416576, + "name": "model.layers.0.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 2794455040, + "name": "model.layers.0.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 4248829952, + "name": "model.layers.1.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 4551868416, + "name": "model.layers.1.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 6006243328, + "name": "model.layers.2.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 6309281792, + "name": "model.layers.2.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 7763656704, + "name": "model.layers.3.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 8066695168, + "name": "model.layers.3.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 9521070080, + "name": "model.layers.4.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 9824108544, + "name": "model.layers.4.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 11278483456, + "name": "model.layers.5.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 11581521920, + "name": "model.layers.5.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 13035896832, + "name": "model.layers.6.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 13338935296, + "name": "model.layers.6.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 14793310208, + "name": "model.layers.7.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 15096348672, + "name": "model.layers.7.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 16550723584, + "name": "model.layers.8.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 16853762048, + "name": "model.layers.8.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 18308136960, + "name": "model.layers.9.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 18611175424, + "name": "model.layers.9.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 20065550336, + "name": "model.layers.10.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 20368588800, + "name": "model.layers.10.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 21822963712, + "name": "model.layers.11.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 22126002176, + "name": "model.layers.11.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 23580377088, + "name": "model.layers.12.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 23883415552, + "name": "model.layers.12.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 25337790464, + "name": "model.layers.13.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 25640828928, + "name": "model.layers.13.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 27095203840, + "name": "model.layers.14.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 27398242304, + "name": "model.layers.14.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 28852617216, + "name": "model.layers.15.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 29155655680, + "name": "model.layers.15.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 30610030592, + "name": "model.layers.16.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 30913069056, + "name": "model.layers.16.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 32367443968, + "name": "model.layers.17.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 32670482432, + "name": "model.layers.17.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 34124857344, + "name": "model.layers.18.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 34427895808, + "name": "model.layers.18.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 35882270720, + "name": "model.layers.19.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 36185309184, + "name": "model.layers.19.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 37639684096, + "name": "model.layers.20.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 37942722560, + "name": "model.layers.20.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 39397097472, + "name": "model.layers.21.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 39700135936, + "name": "model.layers.21.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 41154510848, + "name": "model.layers.22.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 41457549312, + "name": "model.layers.22.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 42911924224, + "name": "model.layers.23.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 43214962688, + "name": "model.layers.23.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 44669337600, + "name": "model.layers.24.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 44972376064, + "name": "model.layers.24.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 46426750976, + "name": "model.layers.25.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 46729789440, + "name": "model.layers.25.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 48184164352, + "name": "model.layers.26.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 48487202816, + "name": "model.layers.26.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 49941577728, + "name": "model.layers.27.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 50244616192, + "name": "model.layers.27.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 51698991104, + "name": "model.layers.28.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 52002029568, + "name": "model.layers.28.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 53456404480, + "name": "model.layers.29.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 53759442944, + "name": "model.layers.29.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 55213817856, + "name": "model.layers.30.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 55516856320, + "name": "model.layers.30.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 56971231232, + "name": "model.layers.31.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 57274269696, + "name": "model.layers.31.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 58728644608, + "name": "model.layers.32.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 59031683072, + "name": "model.layers.32.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 60486057984, + "name": "model.layers.33.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 60789096448, + "name": "model.layers.33.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 62243471360, + "name": "model.layers.34.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 62546509824, + "name": "model.layers.34.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 64000884736, + "name": "model.layers.35.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 64303923200, + "name": "model.layers.35.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 65758298112, + "name": "model.layers.36.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 66061336576, + "name": "model.layers.36.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 67515711488, + "name": "model.layers.37.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 67818749952, + "name": "model.layers.37.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 69273124864, + "name": "model.layers.38.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 69576163328, + "name": "model.layers.38.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 71030538240, + "name": "model.layers.39.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 71333576704, + "name": "model.layers.39.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 72787951616, + "name": "model.layers.40.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 73090990080, + "name": "model.layers.40.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 74545364992, + "name": "model.layers.41.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 74848403456, + "name": "model.layers.41.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 76302778368, + "name": "model.layers.42.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 76605816832, + "name": "model.layers.42.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 78060191744, + "name": "model.layers.43.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 78363230208, + "name": "model.layers.43.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 79817605120, + "name": "model.layers.44.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 80120643584, + "name": "model.layers.44.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 81575018496, + "name": "model.layers.45.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 81878056960, + "name": "model.layers.45.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 83332431872, + "name": "model.layers.46.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 83635470336, + "name": "model.layers.46.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 85089845248, + "name": "model.layers.47.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 85392883712, + "name": "model.layers.47.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 86847258624, + "name": "model.layers.48.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 87150297088, + "name": "model.layers.48.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 88604672000, + "name": "model.layers.49.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 88907710464, + "name": "model.layers.49.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 90362085376, + "name": "model.layers.50.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 90665123840, + "name": "model.layers.50.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 92119498752, + "name": "model.layers.51.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 92422537216, + "name": "model.layers.51.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 93876912128, + "name": "model.layers.52.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 94179950592, + "name": "model.layers.52.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 95634325504, + "name": "model.layers.53.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 95937363968, + "name": "model.layers.53.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 97391738880, + "name": "model.layers.54.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 97694777344, + "name": "model.layers.54.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 99149152256, + "name": "model.layers.55.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 99452190720, + "name": "model.layers.55.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 100906565632, + "name": "model.layers.56.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 101209604096, + "name": "model.layers.56.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 102663979008, + "name": "model.layers.57.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 102967017472, + "name": "model.layers.57.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 104421392384, + "name": "model.layers.58.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 104724430848, + "name": "model.layers.58.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 106178805760, + "name": "model.layers.59.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 106481844224, + "name": "model.layers.59.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 107936219136, + "name": "model.layers.60.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 108239257600, + "name": "model.layers.60.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 109693632512, + "name": "model.layers.61.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 109996670976, + "name": "model.layers.61.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 111451045888, + "name": "model.layers.62.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 111754084352, + "name": "model.layers.62.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 113208459264, + "name": "model.layers.63.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 113511497728, + "name": "model.layers.63.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 114965872640, + "name": "model.layers.64.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 115268911104, + "name": "model.layers.64.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 116723286016, + "name": "model.layers.65.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 117026324480, + "name": "model.layers.65.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 118480699392, + "name": "model.layers.66.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 118783737856, + "name": "model.layers.66.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 120238112768, + "name": "model.layers.67.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 120541151232, + "name": "model.layers.67.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 121995526144, + "name": "model.layers.68.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 122298564608, + "name": "model.layers.68.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 123752939520, + "name": "model.layers.69.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 124055977984, + "name": "model.layers.69.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 125510352896, + "name": "model.layers.70.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 125813391360, + "name": "model.layers.70.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 127267766272, + "name": "model.layers.71.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 127570804736, + "name": "model.layers.71.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 129025179648, + "name": "model.layers.72.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 129328218112, + "name": "model.layers.72.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 130782593024, + "name": "model.layers.73.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 131085631488, + "name": "model.layers.73.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 132540006400, + "name": "model.layers.74.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 132843044864, + "name": "model.layers.74.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 134297419776, + "name": "model.layers.75.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 134600458240, + "name": "model.layers.75.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 136054833152, + "name": "model.layers.76.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 136357871616, + "name": "model.layers.76.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 137812246528, + "name": "model.layers.77.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 138115284992, + "name": "model.layers.77.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 139569659904, + "name": "model.layers.78.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 139872698368, + "name": "model.layers.78.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 141327073280, + "name": "model.layers.79.attention_bundle", + "payload_bytes": 302026752 + }, + { + "logical_address_bytes": 141630111744, + "name": "model.layers.79.mlp_bundle", + "payload_bytes": 1453342720 + }, + { + "logical_address_bytes": 143084486656, + "name": "model.norm", + "payload_bytes": 16384 + }, + { + "logical_address_bytes": 143085535232, + "name": "lm_head", + "payload_bytes": 2491416576 + } + ], + "source_resolved_commit": "UNKNOWN", + "tensor_payload_bytes": 145412407296 + }, + "Qwen/Qwen2.5-7B-Instruct": { + "analytical_macs_per_token_at_context": 7347372032, + "compute_cost_semantics": "ANALYTICAL_DENSE_MAC_COUNT_EXCLUDES_NORMS_ROPE_SOFTMAX_AND_RUNTIME", + "context_tokens": 4096, + "logical_region_address_semantics": "DERIVED_1MIB_ALIGNED_SCENARIO_NOT_SAFETENSORS_FILE_OFFSETS", + "logical_regions": [ + { + "logical_address_bytes": 0, + "name": "model.embed_tokens", + "payload_bytes": 1089994752 + }, + { + "logical_address_bytes": 1090519040, + "name": "model.layers.0.attention_bundle", + "payload_bytes": 58736640 + }, + { + "logical_address_bytes": 1150287872, + "name": "model.layers.0.mlp_bundle", + "payload_bytes": 407378944 + }, + { + "logical_address_bytes": 1558183936, + "name": "model.layers.1.attention_bundle", + "payload_bytes": 58736640 + }, + { + "logical_address_bytes": 1617952768, + "name": "model.layers.1.mlp_bundle", + "payload_bytes": 407378944 + }, + { + "logical_address_bytes": 2025848832, + "name": "model.layers.2.attention_bundle", + "payload_bytes": 58736640 + }, + { + "logical_address_bytes": 2085617664, + "name": "model.layers.2.mlp_bundle", + "payload_bytes": 407378944 + }, + { + "logical_address_bytes": 2493513728, + "name": "model.layers.3.attention_bundle", + "payload_bytes": 58736640 + }, + { + "logical_address_bytes": 2553282560, + "name": "model.layers.3.mlp_bundle", + "payload_bytes": 407378944 + }, + { + "logical_address_bytes": 2961178624, + "name": "model.layers.4.attention_bundle", + "payload_bytes": 58736640 + }, + { + "logical_address_bytes": 3020947456, + "name": "model.layers.4.mlp_bundle", + "payload_bytes": 407378944 + }, + { + "logical_address_bytes": 3428843520, + "name": "model.layers.5.attention_bundle", + "payload_bytes": 58736640 + }, + { + "logical_address_bytes": 3488612352, + "name": "model.layers.5.mlp_bundle", + "payload_bytes": 407378944 + }, + { + "logical_address_bytes": 3896508416, + "name": "model.layers.6.attention_bundle", + "payload_bytes": 58736640 + }, + { + "logical_address_bytes": 3956277248, + "name": "model.layers.6.mlp_bundle", + "payload_bytes": 407378944 + }, + { + "logical_address_bytes": 4364173312, + "name": "model.layers.7.attention_bundle", + "payload_bytes": 58736640 + }, + { + "logical_address_bytes": 4423942144, + "name": "model.layers.7.mlp_bundle", + "payload_bytes": 407378944 + }, + { + "logical_address_bytes": 4831838208, + "name": "model.layers.8.attention_bundle", + "payload_bytes": 58736640 + }, + { + "logical_address_bytes": 4891607040, + "name": "model.layers.8.mlp_bundle", + "payload_bytes": 407378944 + }, + { + "logical_address_bytes": 5299503104, + "name": "model.layers.9.attention_bundle", + "payload_bytes": 58736640 + }, + { + "logical_address_bytes": 5359271936, + "name": "model.layers.9.mlp_bundle", + "payload_bytes": 407378944 + }, + { + "logical_address_bytes": 5767168000, + "name": "model.layers.10.attention_bundle", + "payload_bytes": 58736640 + }, + { + "logical_address_bytes": 5826936832, + "name": "model.layers.10.mlp_bundle", + "payload_bytes": 407378944 + }, + { + "logical_address_bytes": 6234832896, + "name": "model.layers.11.attention_bundle", + "payload_bytes": 58736640 + }, + { + "logical_address_bytes": 6294601728, + "name": "model.layers.11.mlp_bundle", + "payload_bytes": 407378944 + }, + { + "logical_address_bytes": 6702497792, + "name": "model.layers.12.attention_bundle", + "payload_bytes": 58736640 + }, + { + "logical_address_bytes": 6762266624, + "name": "model.layers.12.mlp_bundle", + "payload_bytes": 407378944 + }, + { + "logical_address_bytes": 7170162688, + "name": "model.layers.13.attention_bundle", + "payload_bytes": 58736640 + }, + { + "logical_address_bytes": 7229931520, + "name": "model.layers.13.mlp_bundle", + "payload_bytes": 407378944 + }, + { + "logical_address_bytes": 7637827584, + "name": "model.layers.14.attention_bundle", + "payload_bytes": 58736640 + }, + { + "logical_address_bytes": 7697596416, + "name": "model.layers.14.mlp_bundle", + "payload_bytes": 407378944 + }, + { + "logical_address_bytes": 8105492480, + "name": "model.layers.15.attention_bundle", + "payload_bytes": 58736640 + }, + { + "logical_address_bytes": 8165261312, + "name": "model.layers.15.mlp_bundle", + "payload_bytes": 407378944 + }, + { + "logical_address_bytes": 8573157376, + "name": "model.layers.16.attention_bundle", + "payload_bytes": 58736640 + }, + { + "logical_address_bytes": 8632926208, + "name": "model.layers.16.mlp_bundle", + "payload_bytes": 407378944 + }, + { + "logical_address_bytes": 9040822272, + "name": "model.layers.17.attention_bundle", + "payload_bytes": 58736640 + }, + { + "logical_address_bytes": 9100591104, + "name": "model.layers.17.mlp_bundle", + "payload_bytes": 407378944 + }, + { + "logical_address_bytes": 9508487168, + "name": "model.layers.18.attention_bundle", + "payload_bytes": 58736640 + }, + { + "logical_address_bytes": 9568256000, + "name": "model.layers.18.mlp_bundle", + "payload_bytes": 407378944 + }, + { + "logical_address_bytes": 9976152064, + "name": "model.layers.19.attention_bundle", + "payload_bytes": 58736640 + }, + { + "logical_address_bytes": 10035920896, + "name": "model.layers.19.mlp_bundle", + "payload_bytes": 407378944 + }, + { + "logical_address_bytes": 10443816960, + "name": "model.layers.20.attention_bundle", + "payload_bytes": 58736640 + }, + { + "logical_address_bytes": 10503585792, + "name": "model.layers.20.mlp_bundle", + "payload_bytes": 407378944 + }, + { + "logical_address_bytes": 10911481856, + "name": "model.layers.21.attention_bundle", + "payload_bytes": 58736640 + }, + { + "logical_address_bytes": 10971250688, + "name": "model.layers.21.mlp_bundle", + "payload_bytes": 407378944 + }, + { + "logical_address_bytes": 11379146752, + "name": "model.layers.22.attention_bundle", + "payload_bytes": 58736640 + }, + { + "logical_address_bytes": 11438915584, + "name": "model.layers.22.mlp_bundle", + "payload_bytes": 407378944 + }, + { + "logical_address_bytes": 11846811648, + "name": "model.layers.23.attention_bundle", + "payload_bytes": 58736640 + }, + { + "logical_address_bytes": 11906580480, + "name": "model.layers.23.mlp_bundle", + "payload_bytes": 407378944 + }, + { + "logical_address_bytes": 12314476544, + "name": "model.layers.24.attention_bundle", + "payload_bytes": 58736640 + }, + { + "logical_address_bytes": 12374245376, + "name": "model.layers.24.mlp_bundle", + "payload_bytes": 407378944 + }, + { + "logical_address_bytes": 12782141440, + "name": "model.layers.25.attention_bundle", + "payload_bytes": 58736640 + }, + { + "logical_address_bytes": 12841910272, + "name": "model.layers.25.mlp_bundle", + "payload_bytes": 407378944 + }, + { + "logical_address_bytes": 13249806336, + "name": "model.layers.26.attention_bundle", + "payload_bytes": 58736640 + }, + { + "logical_address_bytes": 13309575168, + "name": "model.layers.26.mlp_bundle", + "payload_bytes": 407378944 + }, + { + "logical_address_bytes": 13717471232, + "name": "model.layers.27.attention_bundle", + "payload_bytes": 58736640 + }, + { + "logical_address_bytes": 13777240064, + "name": "model.layers.27.mlp_bundle", + "payload_bytes": 407378944 + }, + { + "logical_address_bytes": 14185136128, + "name": "model.norm", + "payload_bytes": 7168 + }, + { + "logical_address_bytes": 14186184704, + "name": "lm_head", + "payload_bytes": 1089994752 + } + ], + "source_resolved_commit": "UNKNOWN", + "tensor_payload_bytes": 15231233024 + } + }, + "target_projection_source": "experiments/eq3_maintenance/sources/qwen2_5_weight_models.json", + "trace_sha256": "feea4657abc5e7b981208b23b8d2fff9836194694c5c3dec69c6bf59bbe627a7", + "weight_accesses": [ + { + "access_bytes": 896, + "access_shape": [ + 4, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op0:embedding_lookup", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 0, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.embed_tokens.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op1:layer0.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 1, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.0.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op2:layer0.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 2, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.0.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op2:layer0.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 3, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.0.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op3:layer0.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 4, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.0.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "prefill", + "op_id": "prefill:op3:layer0.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 5, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.0.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op4:layer0.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 6, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.0.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "prefill", + "op_id": "prefill:op4:layer0.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 7, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.0.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op6:layer0.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 8, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.0.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op7:layer0.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 9, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.0.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op8:layer0.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 10, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.0.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op9:layer0.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 11, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.0.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "prefill", + "op_id": "prefill:op10:layer0.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 12, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.0.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op11:layer1.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 13, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.1.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op12:layer1.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 14, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.1.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op12:layer1.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 15, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.1.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op13:layer1.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 16, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.1.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "prefill", + "op_id": "prefill:op13:layer1.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 17, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.1.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op14:layer1.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 18, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.1.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "prefill", + "op_id": "prefill:op14:layer1.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 19, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.1.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op16:layer1.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 20, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.1.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op17:layer1.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 21, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.1.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op18:layer1.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 22, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.1.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op19:layer1.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 23, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.1.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "prefill", + "op_id": "prefill:op20:layer1.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 24, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.1.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op21:layer2.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 25, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.2.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op22:layer2.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 26, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.2.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op22:layer2.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 27, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.2.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op23:layer2.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 28, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.2.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "prefill", + "op_id": "prefill:op23:layer2.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 29, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.2.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op24:layer2.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 30, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.2.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "prefill", + "op_id": "prefill:op24:layer2.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 31, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.2.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op26:layer2.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 32, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.2.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op27:layer2.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 33, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.2.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op28:layer2.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 34, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.2.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op29:layer2.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 35, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.2.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "prefill", + "op_id": "prefill:op30:layer2.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 36, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.2.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op31:layer3.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 37, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.3.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op32:layer3.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 38, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.3.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op32:layer3.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 39, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.3.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op33:layer3.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 40, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.3.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "prefill", + "op_id": "prefill:op33:layer3.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 41, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.3.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op34:layer3.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 42, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.3.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "prefill", + "op_id": "prefill:op34:layer3.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 43, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.3.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op36:layer3.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 44, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.3.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op37:layer3.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 45, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.3.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op38:layer3.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 46, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.3.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op39:layer3.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 47, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.3.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "prefill", + "op_id": "prefill:op40:layer3.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 48, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.3.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op41:layer4.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 49, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.4.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op42:layer4.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 50, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.4.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op42:layer4.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 51, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.4.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op43:layer4.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 52, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.4.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "prefill", + "op_id": "prefill:op43:layer4.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 53, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.4.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op44:layer4.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 54, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.4.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "prefill", + "op_id": "prefill:op44:layer4.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 55, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.4.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op46:layer4.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 56, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.4.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op47:layer4.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 57, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.4.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op48:layer4.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 58, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.4.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op49:layer4.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 59, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.4.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "prefill", + "op_id": "prefill:op50:layer4.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 60, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.4.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op51:layer5.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 61, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.5.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op52:layer5.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 62, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.5.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op52:layer5.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 63, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.5.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op53:layer5.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 64, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.5.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "prefill", + "op_id": "prefill:op53:layer5.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 65, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.5.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op54:layer5.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 66, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.5.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "prefill", + "op_id": "prefill:op54:layer5.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 67, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.5.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op56:layer5.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 68, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.5.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op57:layer5.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 69, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.5.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op58:layer5.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 70, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.5.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op59:layer5.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 71, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.5.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "prefill", + "op_id": "prefill:op60:layer5.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 72, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.5.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op61:layer6.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 73, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.6.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op62:layer6.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 74, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.6.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op62:layer6.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 75, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.6.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op63:layer6.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 76, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.6.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "prefill", + "op_id": "prefill:op63:layer6.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 77, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.6.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op64:layer6.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 78, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.6.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "prefill", + "op_id": "prefill:op64:layer6.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 79, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.6.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op66:layer6.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 80, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.6.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op67:layer6.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 81, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.6.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op68:layer6.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 82, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.6.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op69:layer6.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 83, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.6.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "prefill", + "op_id": "prefill:op70:layer6.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 84, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.6.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op71:layer7.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 85, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.7.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op72:layer7.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 86, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.7.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op72:layer7.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 87, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.7.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op73:layer7.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 88, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.7.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "prefill", + "op_id": "prefill:op73:layer7.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 89, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.7.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op74:layer7.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 90, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.7.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "prefill", + "op_id": "prefill:op74:layer7.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 91, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.7.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op76:layer7.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 92, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.7.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op77:layer7.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 93, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.7.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op78:layer7.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 94, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.7.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op79:layer7.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 95, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.7.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "prefill", + "op_id": "prefill:op80:layer7.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 96, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.7.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op81:layer8.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 97, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.8.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op82:layer8.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 98, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.8.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op82:layer8.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 99, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.8.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op83:layer8.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 100, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.8.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "prefill", + "op_id": "prefill:op83:layer8.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 101, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.8.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op84:layer8.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 102, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.8.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "prefill", + "op_id": "prefill:op84:layer8.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 103, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.8.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op86:layer8.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 104, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.8.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op87:layer8.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 105, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.8.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op88:layer8.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 106, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.8.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op89:layer8.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 107, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.8.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "prefill", + "op_id": "prefill:op90:layer8.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 108, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.8.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op91:layer9.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 109, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.9.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op92:layer9.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 110, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.9.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op92:layer9.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 111, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.9.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op93:layer9.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 112, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.9.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "prefill", + "op_id": "prefill:op93:layer9.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 113, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.9.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op94:layer9.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 114, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.9.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "prefill", + "op_id": "prefill:op94:layer9.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 115, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.9.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op96:layer9.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 116, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.9.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op97:layer9.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 117, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.9.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op98:layer9.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 118, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.9.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op99:layer9.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 119, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.9.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "prefill", + "op_id": "prefill:op100:layer9.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 120, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.9.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op101:layer10.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 121, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.10.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op102:layer10.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 122, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.10.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op102:layer10.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 123, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.10.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op103:layer10.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 124, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.10.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "prefill", + "op_id": "prefill:op103:layer10.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 125, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.10.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op104:layer10.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 126, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.10.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "prefill", + "op_id": "prefill:op104:layer10.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 127, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.10.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op106:layer10.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 128, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.10.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op107:layer10.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 129, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.10.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op108:layer10.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 130, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.10.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op109:layer10.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 131, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.10.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "prefill", + "op_id": "prefill:op110:layer10.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 132, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.10.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op111:layer11.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 133, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.11.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op112:layer11.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 134, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.11.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op112:layer11.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 135, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.11.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op113:layer11.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 136, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.11.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "prefill", + "op_id": "prefill:op113:layer11.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 137, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.11.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op114:layer11.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 138, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.11.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "prefill", + "op_id": "prefill:op114:layer11.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 139, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.11.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op116:layer11.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 140, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.11.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op117:layer11.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 141, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.11.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op118:layer11.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 142, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.11.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op119:layer11.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 143, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.11.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "prefill", + "op_id": "prefill:op120:layer11.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 144, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.11.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op121:layer12.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 145, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.12.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op122:layer12.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 146, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.12.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op122:layer12.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 147, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.12.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op123:layer12.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 148, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.12.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "prefill", + "op_id": "prefill:op123:layer12.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 149, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.12.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op124:layer12.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 150, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.12.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "prefill", + "op_id": "prefill:op124:layer12.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 151, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.12.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op126:layer12.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 152, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.12.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op127:layer12.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 153, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.12.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op128:layer12.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 154, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.12.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op129:layer12.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 155, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.12.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "prefill", + "op_id": "prefill:op130:layer12.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 156, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.12.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op131:layer13.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 157, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.13.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op132:layer13.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 158, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.13.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op132:layer13.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 159, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.13.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op133:layer13.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 160, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.13.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "prefill", + "op_id": "prefill:op133:layer13.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 161, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.13.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op134:layer13.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 162, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.13.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "prefill", + "op_id": "prefill:op134:layer13.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 163, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.13.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op136:layer13.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 164, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.13.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op137:layer13.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 165, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.13.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op138:layer13.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 166, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.13.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op139:layer13.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 167, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.13.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "prefill", + "op_id": "prefill:op140:layer13.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 168, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.13.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op141:layer14.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 169, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.14.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op142:layer14.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 170, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.14.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op142:layer14.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 171, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.14.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op143:layer14.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 172, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.14.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "prefill", + "op_id": "prefill:op143:layer14.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 173, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.14.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op144:layer14.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 174, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.14.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "prefill", + "op_id": "prefill:op144:layer14.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 175, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.14.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op146:layer14.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 176, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.14.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op147:layer14.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 177, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.14.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op148:layer14.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 178, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.14.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op149:layer14.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 179, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.14.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "prefill", + "op_id": "prefill:op150:layer14.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 180, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.14.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op151:layer15.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 181, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.15.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op152:layer15.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 182, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.15.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op152:layer15.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 183, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.15.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op153:layer15.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 184, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.15.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "prefill", + "op_id": "prefill:op153:layer15.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 185, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.15.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op154:layer15.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 186, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.15.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "prefill", + "op_id": "prefill:op154:layer15.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 187, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.15.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op156:layer15.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 188, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.15.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op157:layer15.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 189, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.15.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op158:layer15.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 190, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.15.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op159:layer15.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 191, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.15.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "prefill", + "op_id": "prefill:op160:layer15.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 192, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.15.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op161:layer16.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 193, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.16.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op162:layer16.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 194, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.16.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op162:layer16.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 195, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.16.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op163:layer16.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 196, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.16.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "prefill", + "op_id": "prefill:op163:layer16.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 197, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.16.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op164:layer16.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 198, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.16.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "prefill", + "op_id": "prefill:op164:layer16.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 199, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.16.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op166:layer16.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 200, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.16.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op167:layer16.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 201, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.16.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op168:layer16.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 202, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.16.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op169:layer16.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 203, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.16.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "prefill", + "op_id": "prefill:op170:layer16.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 204, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.16.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op171:layer17.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 205, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.17.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op172:layer17.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 206, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.17.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op172:layer17.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 207, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.17.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op173:layer17.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 208, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.17.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "prefill", + "op_id": "prefill:op173:layer17.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 209, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.17.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op174:layer17.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 210, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.17.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "prefill", + "op_id": "prefill:op174:layer17.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 211, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.17.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op176:layer17.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 212, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.17.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op177:layer17.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 213, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.17.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op178:layer17.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 214, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.17.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op179:layer17.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 215, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.17.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "prefill", + "op_id": "prefill:op180:layer17.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 216, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.17.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op181:layer18.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 217, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.18.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op182:layer18.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 218, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.18.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op182:layer18.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 219, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.18.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op183:layer18.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 220, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.18.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "prefill", + "op_id": "prefill:op183:layer18.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 221, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.18.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op184:layer18.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 222, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.18.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "prefill", + "op_id": "prefill:op184:layer18.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 223, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.18.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op186:layer18.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 224, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.18.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op187:layer18.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 225, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.18.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op188:layer18.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 226, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.18.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op189:layer18.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 227, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.18.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "prefill", + "op_id": "prefill:op190:layer18.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 228, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.18.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op191:layer19.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 229, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.19.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op192:layer19.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 230, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.19.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op192:layer19.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 231, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.19.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op193:layer19.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 232, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.19.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "prefill", + "op_id": "prefill:op193:layer19.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 233, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.19.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op194:layer19.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 234, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.19.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "prefill", + "op_id": "prefill:op194:layer19.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 235, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.19.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op196:layer19.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 236, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.19.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op197:layer19.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 237, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.19.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op198:layer19.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 238, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.19.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op199:layer19.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 239, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.19.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "prefill", + "op_id": "prefill:op200:layer19.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 240, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.19.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op201:layer20.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 241, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.20.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op202:layer20.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 242, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.20.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op202:layer20.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 243, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.20.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op203:layer20.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 244, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.20.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "prefill", + "op_id": "prefill:op203:layer20.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 245, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.20.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op204:layer20.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 246, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.20.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "prefill", + "op_id": "prefill:op204:layer20.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 247, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.20.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op206:layer20.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 248, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.20.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op207:layer20.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 249, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.20.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op208:layer20.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 250, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.20.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op209:layer20.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 251, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.20.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "prefill", + "op_id": "prefill:op210:layer20.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 252, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.20.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op211:layer21.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 253, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.21.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op212:layer21.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 254, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.21.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op212:layer21.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 255, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.21.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op213:layer21.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 256, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.21.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "prefill", + "op_id": "prefill:op213:layer21.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 257, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.21.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op214:layer21.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 258, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.21.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "prefill", + "op_id": "prefill:op214:layer21.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 259, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.21.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op216:layer21.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 260, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.21.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op217:layer21.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 261, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.21.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op218:layer21.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 262, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.21.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op219:layer21.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 263, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.21.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "prefill", + "op_id": "prefill:op220:layer21.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 264, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.21.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op221:layer22.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 265, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.22.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op222:layer22.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 266, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.22.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op222:layer22.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 267, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.22.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op223:layer22.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 268, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.22.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "prefill", + "op_id": "prefill:op223:layer22.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 269, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.22.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op224:layer22.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 270, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.22.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "prefill", + "op_id": "prefill:op224:layer22.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 271, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.22.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op226:layer22.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 272, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.22.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op227:layer22.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 273, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.22.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op228:layer22.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 274, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.22.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op229:layer22.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 275, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.22.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "prefill", + "op_id": "prefill:op230:layer22.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 276, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.22.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op231:layer23.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 277, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.23.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op232:layer23.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 278, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.23.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op232:layer23.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 279, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.23.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op233:layer23.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 280, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.23.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "prefill", + "op_id": "prefill:op233:layer23.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 281, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.23.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op234:layer23.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 282, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.23.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "prefill", + "op_id": "prefill:op234:layer23.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 283, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.23.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op236:layer23.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 284, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.23.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op237:layer23.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 285, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.23.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op238:layer23.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 286, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.23.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op239:layer23.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 287, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.23.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "prefill", + "op_id": "prefill:op240:layer23.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 288, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.23.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op241:layer24.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 289, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.24.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op242:layer24.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 290, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.24.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op242:layer24.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 291, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.24.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op243:layer24.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 292, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.24.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "prefill", + "op_id": "prefill:op243:layer24.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 293, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.24.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op244:layer24.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 294, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.24.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "prefill", + "op_id": "prefill:op244:layer24.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 295, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.24.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op246:layer24.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 296, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.24.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op247:layer24.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 297, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.24.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op248:layer24.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 298, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.24.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op249:layer24.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 299, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.24.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "prefill", + "op_id": "prefill:op250:layer24.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 300, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.24.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op251:layer25.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 301, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.25.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op252:layer25.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 302, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.25.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op252:layer25.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 303, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.25.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op253:layer25.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 304, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.25.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "prefill", + "op_id": "prefill:op253:layer25.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 305, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.25.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op254:layer25.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 306, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.25.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "prefill", + "op_id": "prefill:op254:layer25.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 307, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.25.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op256:layer25.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 308, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.25.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op257:layer25.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 309, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.25.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op258:layer25.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 310, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.25.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op259:layer25.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 311, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.25.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "prefill", + "op_id": "prefill:op260:layer25.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 312, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.25.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op261:layer26.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 313, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.26.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op262:layer26.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 314, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.26.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op262:layer26.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 315, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.26.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op263:layer26.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 316, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.26.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "prefill", + "op_id": "prefill:op263:layer26.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 317, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.26.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op264:layer26.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 318, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.26.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "prefill", + "op_id": "prefill:op264:layer26.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 319, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.26.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op266:layer26.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 320, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.26.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op267:layer26.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 321, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.26.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op268:layer26.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 322, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.26.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op269:layer26.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 323, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.26.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "prefill", + "op_id": "prefill:op270:layer26.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 324, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.26.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op271:layer27.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 325, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.27.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op272:layer27.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 326, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.27.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op272:layer27.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 327, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.27.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op273:layer27.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 328, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.27.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "prefill", + "op_id": "prefill:op273:layer27.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 329, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.27.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op274:layer27.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 330, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.27.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "prefill", + "op_id": "prefill:op274:layer27.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 331, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.27.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op276:layer27.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 332, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.27.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op277:layer27.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 333, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.27.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op278:layer27.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 334, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.27.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op279:layer27.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 335, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.27.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "prefill", + "op_id": "prefill:op280:layer27.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 336, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.27.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op281:final_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 337, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.norm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "prefill", + "op_id": "prefill:op282:lm_head", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 338, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "lm_head.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 1, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op283:embedding_lookup", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 339, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.embed_tokens.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op284:layer0.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 340, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.0.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op285:layer0.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 341, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.0.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op285:layer0.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 342, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.0.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op286:layer0.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 343, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.0.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode0", + "op_id": "decode0:op286:layer0.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 344, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.0.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op287:layer0.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 345, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.0.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode0", + "op_id": "decode0:op287:layer0.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 346, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.0.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op289:layer0.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 347, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.0.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op290:layer0.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 348, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.0.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op291:layer0.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 349, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.0.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op292:layer0.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 350, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.0.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "decode0", + "op_id": "decode0:op293:layer0.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 351, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.0.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op294:layer1.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 352, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.1.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op295:layer1.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 353, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.1.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op295:layer1.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 354, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.1.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op296:layer1.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 355, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.1.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode0", + "op_id": "decode0:op296:layer1.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 356, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.1.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op297:layer1.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 357, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.1.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode0", + "op_id": "decode0:op297:layer1.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 358, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.1.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op299:layer1.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 359, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.1.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op300:layer1.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 360, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.1.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op301:layer1.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 361, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.1.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op302:layer1.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 362, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.1.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "decode0", + "op_id": "decode0:op303:layer1.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 363, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.1.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op304:layer2.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 364, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.2.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op305:layer2.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 365, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.2.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op305:layer2.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 366, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.2.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op306:layer2.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 367, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.2.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode0", + "op_id": "decode0:op306:layer2.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 368, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.2.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op307:layer2.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 369, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.2.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode0", + "op_id": "decode0:op307:layer2.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 370, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.2.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op309:layer2.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 371, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.2.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op310:layer2.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 372, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.2.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op311:layer2.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 373, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.2.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op312:layer2.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 374, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.2.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "decode0", + "op_id": "decode0:op313:layer2.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 375, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.2.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op314:layer3.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 376, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.3.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op315:layer3.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 377, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.3.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op315:layer3.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 378, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.3.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op316:layer3.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 379, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.3.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode0", + "op_id": "decode0:op316:layer3.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 380, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.3.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op317:layer3.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 381, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.3.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode0", + "op_id": "decode0:op317:layer3.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 382, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.3.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op319:layer3.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 383, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.3.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op320:layer3.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 384, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.3.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op321:layer3.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 385, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.3.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op322:layer3.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 386, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.3.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "decode0", + "op_id": "decode0:op323:layer3.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 387, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.3.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op324:layer4.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 388, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.4.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op325:layer4.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 389, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.4.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op325:layer4.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 390, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.4.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op326:layer4.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 391, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.4.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode0", + "op_id": "decode0:op326:layer4.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 392, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.4.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op327:layer4.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 393, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.4.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode0", + "op_id": "decode0:op327:layer4.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 394, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.4.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op329:layer4.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 395, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.4.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op330:layer4.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 396, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.4.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op331:layer4.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 397, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.4.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op332:layer4.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 398, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.4.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "decode0", + "op_id": "decode0:op333:layer4.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 399, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.4.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op334:layer5.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 400, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.5.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op335:layer5.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 401, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.5.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op335:layer5.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 402, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.5.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op336:layer5.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 403, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.5.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode0", + "op_id": "decode0:op336:layer5.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 404, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.5.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op337:layer5.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 405, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.5.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode0", + "op_id": "decode0:op337:layer5.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 406, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.5.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op339:layer5.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 407, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.5.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op340:layer5.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 408, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.5.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op341:layer5.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 409, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.5.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op342:layer5.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 410, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.5.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "decode0", + "op_id": "decode0:op343:layer5.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 411, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.5.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op344:layer6.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 412, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.6.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op345:layer6.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 413, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.6.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op345:layer6.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 414, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.6.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op346:layer6.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 415, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.6.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode0", + "op_id": "decode0:op346:layer6.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 416, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.6.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op347:layer6.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 417, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.6.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode0", + "op_id": "decode0:op347:layer6.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 418, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.6.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op349:layer6.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 419, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.6.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op350:layer6.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 420, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.6.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op351:layer6.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 421, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.6.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op352:layer6.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 422, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.6.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "decode0", + "op_id": "decode0:op353:layer6.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 423, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.6.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op354:layer7.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 424, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.7.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op355:layer7.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 425, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.7.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op355:layer7.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 426, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.7.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op356:layer7.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 427, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.7.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode0", + "op_id": "decode0:op356:layer7.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 428, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.7.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op357:layer7.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 429, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.7.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode0", + "op_id": "decode0:op357:layer7.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 430, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.7.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op359:layer7.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 431, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.7.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op360:layer7.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 432, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.7.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op361:layer7.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 433, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.7.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op362:layer7.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 434, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.7.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "decode0", + "op_id": "decode0:op363:layer7.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 435, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.7.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op364:layer8.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 436, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.8.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op365:layer8.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 437, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.8.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op365:layer8.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 438, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.8.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op366:layer8.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 439, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.8.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode0", + "op_id": "decode0:op366:layer8.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 440, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.8.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op367:layer8.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 441, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.8.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode0", + "op_id": "decode0:op367:layer8.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 442, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.8.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op369:layer8.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 443, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.8.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op370:layer8.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 444, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.8.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op371:layer8.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 445, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.8.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op372:layer8.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 446, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.8.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "decode0", + "op_id": "decode0:op373:layer8.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 447, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.8.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op374:layer9.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 448, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.9.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op375:layer9.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 449, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.9.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op375:layer9.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 450, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.9.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op376:layer9.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 451, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.9.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode0", + "op_id": "decode0:op376:layer9.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 452, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.9.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op377:layer9.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 453, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.9.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode0", + "op_id": "decode0:op377:layer9.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 454, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.9.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op379:layer9.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 455, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.9.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op380:layer9.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 456, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.9.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op381:layer9.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 457, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.9.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op382:layer9.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 458, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.9.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "decode0", + "op_id": "decode0:op383:layer9.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 459, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.9.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op384:layer10.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 460, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.10.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op385:layer10.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 461, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.10.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op385:layer10.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 462, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.10.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op386:layer10.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 463, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.10.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode0", + "op_id": "decode0:op386:layer10.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 464, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.10.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op387:layer10.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 465, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.10.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode0", + "op_id": "decode0:op387:layer10.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 466, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.10.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op389:layer10.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 467, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.10.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op390:layer10.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 468, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.10.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op391:layer10.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 469, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.10.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op392:layer10.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 470, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.10.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "decode0", + "op_id": "decode0:op393:layer10.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 471, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.10.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op394:layer11.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 472, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.11.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op395:layer11.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 473, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.11.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op395:layer11.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 474, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.11.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op396:layer11.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 475, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.11.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode0", + "op_id": "decode0:op396:layer11.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 476, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.11.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op397:layer11.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 477, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.11.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode0", + "op_id": "decode0:op397:layer11.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 478, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.11.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op399:layer11.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 479, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.11.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op400:layer11.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 480, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.11.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op401:layer11.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 481, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.11.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op402:layer11.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 482, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.11.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "decode0", + "op_id": "decode0:op403:layer11.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 483, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.11.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op404:layer12.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 484, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.12.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op405:layer12.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 485, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.12.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op405:layer12.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 486, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.12.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op406:layer12.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 487, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.12.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode0", + "op_id": "decode0:op406:layer12.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 488, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.12.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op407:layer12.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 489, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.12.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode0", + "op_id": "decode0:op407:layer12.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 490, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.12.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op409:layer12.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 491, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.12.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op410:layer12.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 492, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.12.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op411:layer12.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 493, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.12.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op412:layer12.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 494, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.12.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "decode0", + "op_id": "decode0:op413:layer12.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 495, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.12.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op414:layer13.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 496, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.13.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op415:layer13.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 497, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.13.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op415:layer13.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 498, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.13.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op416:layer13.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 499, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.13.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode0", + "op_id": "decode0:op416:layer13.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 500, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.13.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op417:layer13.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 501, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.13.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode0", + "op_id": "decode0:op417:layer13.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 502, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.13.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op419:layer13.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 503, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.13.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op420:layer13.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 504, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.13.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op421:layer13.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 505, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.13.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op422:layer13.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 506, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.13.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "decode0", + "op_id": "decode0:op423:layer13.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 507, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.13.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op424:layer14.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 508, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.14.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op425:layer14.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 509, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.14.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op425:layer14.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 510, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.14.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op426:layer14.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 511, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.14.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode0", + "op_id": "decode0:op426:layer14.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 512, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.14.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op427:layer14.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 513, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.14.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode0", + "op_id": "decode0:op427:layer14.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 514, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.14.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op429:layer14.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 515, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.14.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op430:layer14.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 516, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.14.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op431:layer14.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 517, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.14.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op432:layer14.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 518, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.14.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "decode0", + "op_id": "decode0:op433:layer14.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 519, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.14.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op434:layer15.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 520, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.15.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op435:layer15.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 521, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.15.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op435:layer15.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 522, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.15.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op436:layer15.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 523, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.15.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode0", + "op_id": "decode0:op436:layer15.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 524, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.15.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op437:layer15.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 525, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.15.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode0", + "op_id": "decode0:op437:layer15.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 526, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.15.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op439:layer15.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 527, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.15.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op440:layer15.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 528, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.15.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op441:layer15.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 529, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.15.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op442:layer15.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 530, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.15.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "decode0", + "op_id": "decode0:op443:layer15.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 531, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.15.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op444:layer16.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 532, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.16.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op445:layer16.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 533, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.16.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op445:layer16.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 534, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.16.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op446:layer16.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 535, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.16.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode0", + "op_id": "decode0:op446:layer16.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 536, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.16.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op447:layer16.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 537, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.16.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode0", + "op_id": "decode0:op447:layer16.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 538, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.16.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op449:layer16.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 539, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.16.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op450:layer16.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 540, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.16.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op451:layer16.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 541, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.16.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op452:layer16.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 542, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.16.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "decode0", + "op_id": "decode0:op453:layer16.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 543, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.16.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op454:layer17.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 544, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.17.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op455:layer17.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 545, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.17.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op455:layer17.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 546, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.17.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op456:layer17.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 547, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.17.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode0", + "op_id": "decode0:op456:layer17.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 548, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.17.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op457:layer17.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 549, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.17.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode0", + "op_id": "decode0:op457:layer17.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 550, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.17.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op459:layer17.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 551, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.17.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op460:layer17.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 552, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.17.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op461:layer17.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 553, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.17.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op462:layer17.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 554, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.17.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "decode0", + "op_id": "decode0:op463:layer17.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 555, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.17.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op464:layer18.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 556, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.18.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op465:layer18.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 557, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.18.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op465:layer18.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 558, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.18.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op466:layer18.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 559, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.18.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode0", + "op_id": "decode0:op466:layer18.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 560, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.18.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op467:layer18.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 561, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.18.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode0", + "op_id": "decode0:op467:layer18.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 562, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.18.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op469:layer18.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 563, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.18.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op470:layer18.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 564, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.18.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op471:layer18.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 565, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.18.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op472:layer18.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 566, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.18.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "decode0", + "op_id": "decode0:op473:layer18.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 567, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.18.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op474:layer19.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 568, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.19.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op475:layer19.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 569, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.19.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op475:layer19.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 570, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.19.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op476:layer19.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 571, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.19.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode0", + "op_id": "decode0:op476:layer19.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 572, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.19.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op477:layer19.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 573, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.19.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode0", + "op_id": "decode0:op477:layer19.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 574, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.19.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op479:layer19.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 575, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.19.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op480:layer19.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 576, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.19.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op481:layer19.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 577, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.19.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op482:layer19.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 578, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.19.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "decode0", + "op_id": "decode0:op483:layer19.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 579, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.19.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op484:layer20.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 580, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.20.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op485:layer20.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 581, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.20.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op485:layer20.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 582, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.20.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op486:layer20.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 583, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.20.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode0", + "op_id": "decode0:op486:layer20.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 584, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.20.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op487:layer20.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 585, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.20.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode0", + "op_id": "decode0:op487:layer20.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 586, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.20.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op489:layer20.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 587, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.20.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op490:layer20.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 588, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.20.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op491:layer20.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 589, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.20.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op492:layer20.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 590, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.20.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "decode0", + "op_id": "decode0:op493:layer20.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 591, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.20.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op494:layer21.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 592, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.21.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op495:layer21.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 593, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.21.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op495:layer21.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 594, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.21.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op496:layer21.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 595, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.21.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode0", + "op_id": "decode0:op496:layer21.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 596, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.21.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op497:layer21.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 597, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.21.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode0", + "op_id": "decode0:op497:layer21.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 598, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.21.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op499:layer21.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 599, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.21.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op500:layer21.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 600, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.21.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op501:layer21.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 601, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.21.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op502:layer21.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 602, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.21.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "decode0", + "op_id": "decode0:op503:layer21.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 603, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.21.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op504:layer22.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 604, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.22.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op505:layer22.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 605, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.22.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op505:layer22.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 606, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.22.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op506:layer22.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 607, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.22.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode0", + "op_id": "decode0:op506:layer22.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 608, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.22.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op507:layer22.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 609, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.22.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode0", + "op_id": "decode0:op507:layer22.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 610, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.22.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op509:layer22.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 611, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.22.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op510:layer22.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 612, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.22.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op511:layer22.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 613, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.22.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op512:layer22.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 614, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.22.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "decode0", + "op_id": "decode0:op513:layer22.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 615, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.22.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op514:layer23.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 616, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.23.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op515:layer23.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 617, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.23.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op515:layer23.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 618, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.23.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op516:layer23.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 619, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.23.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode0", + "op_id": "decode0:op516:layer23.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 620, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.23.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op517:layer23.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 621, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.23.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode0", + "op_id": "decode0:op517:layer23.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 622, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.23.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op519:layer23.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 623, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.23.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op520:layer23.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 624, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.23.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op521:layer23.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 625, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.23.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op522:layer23.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 626, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.23.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "decode0", + "op_id": "decode0:op523:layer23.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 627, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.23.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op524:layer24.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 628, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.24.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op525:layer24.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 629, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.24.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op525:layer24.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 630, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.24.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op526:layer24.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 631, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.24.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode0", + "op_id": "decode0:op526:layer24.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 632, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.24.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op527:layer24.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 633, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.24.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode0", + "op_id": "decode0:op527:layer24.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 634, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.24.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op529:layer24.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 635, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.24.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op530:layer24.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 636, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.24.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op531:layer24.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 637, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.24.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op532:layer24.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 638, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.24.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "decode0", + "op_id": "decode0:op533:layer24.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 639, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.24.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op534:layer25.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 640, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.25.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op535:layer25.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 641, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.25.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op535:layer25.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 642, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.25.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op536:layer25.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 643, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.25.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode0", + "op_id": "decode0:op536:layer25.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 644, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.25.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op537:layer25.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 645, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.25.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode0", + "op_id": "decode0:op537:layer25.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 646, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.25.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op539:layer25.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 647, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.25.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op540:layer25.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 648, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.25.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op541:layer25.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 649, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.25.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op542:layer25.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 650, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.25.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "decode0", + "op_id": "decode0:op543:layer25.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 651, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.25.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op544:layer26.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 652, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.26.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op545:layer26.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 653, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.26.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op545:layer26.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 654, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.26.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op546:layer26.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 655, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.26.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode0", + "op_id": "decode0:op546:layer26.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 656, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.26.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op547:layer26.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 657, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.26.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode0", + "op_id": "decode0:op547:layer26.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 658, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.26.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op549:layer26.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 659, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.26.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op550:layer26.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 660, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.26.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op551:layer26.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 661, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.26.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op552:layer26.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 662, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.26.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "decode0", + "op_id": "decode0:op553:layer26.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 663, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.26.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op554:layer27.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 664, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.27.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op555:layer27.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 665, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.27.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op555:layer27.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 666, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.27.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op556:layer27.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 667, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.27.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode0", + "op_id": "decode0:op556:layer27.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 668, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.27.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op557:layer27.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 669, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.27.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode0", + "op_id": "decode0:op557:layer27.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 670, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.27.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op559:layer27.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 671, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.27.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op560:layer27.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 672, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.27.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op561:layer27.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 673, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.27.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op562:layer27.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 674, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.27.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "decode0", + "op_id": "decode0:op563:layer27.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 675, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.27.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op564:final_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 676, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.norm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode0", + "op_id": "decode0:op565:lm_head", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 677, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "lm_head.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 1, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op566:embedding_lookup", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 678, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.embed_tokens.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op567:layer0.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 679, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.0.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op568:layer0.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 680, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.0.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op568:layer0.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 681, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.0.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op569:layer0.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 682, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.0.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode1", + "op_id": "decode1:op569:layer0.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 683, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.0.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op570:layer0.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 684, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.0.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode1", + "op_id": "decode1:op570:layer0.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 685, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.0.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op572:layer0.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 686, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.0.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op573:layer0.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 687, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.0.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op574:layer0.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 688, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.0.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op575:layer0.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 689, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.0.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "decode1", + "op_id": "decode1:op576:layer0.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 690, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.0.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op577:layer1.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 691, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.1.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op578:layer1.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 692, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.1.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op578:layer1.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 693, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.1.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op579:layer1.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 694, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.1.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode1", + "op_id": "decode1:op579:layer1.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 695, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.1.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op580:layer1.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 696, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.1.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode1", + "op_id": "decode1:op580:layer1.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 697, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.1.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op582:layer1.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 698, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.1.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op583:layer1.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 699, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.1.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op584:layer1.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 700, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.1.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op585:layer1.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 701, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.1.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "decode1", + "op_id": "decode1:op586:layer1.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 702, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.1.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op587:layer2.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 703, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.2.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op588:layer2.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 704, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.2.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op588:layer2.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 705, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.2.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op589:layer2.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 706, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.2.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode1", + "op_id": "decode1:op589:layer2.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 707, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.2.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op590:layer2.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 708, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.2.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode1", + "op_id": "decode1:op590:layer2.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 709, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.2.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op592:layer2.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 710, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.2.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op593:layer2.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 711, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.2.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op594:layer2.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 712, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.2.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op595:layer2.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 713, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.2.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "decode1", + "op_id": "decode1:op596:layer2.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 714, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.2.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op597:layer3.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 715, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.3.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op598:layer3.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 716, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.3.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op598:layer3.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 717, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.3.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op599:layer3.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 718, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.3.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode1", + "op_id": "decode1:op599:layer3.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 719, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.3.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op600:layer3.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 720, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.3.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode1", + "op_id": "decode1:op600:layer3.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 721, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.3.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op602:layer3.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 722, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.3.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op603:layer3.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 723, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.3.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op604:layer3.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 724, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.3.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op605:layer3.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 725, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.3.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "decode1", + "op_id": "decode1:op606:layer3.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 726, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.3.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op607:layer4.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 727, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.4.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op608:layer4.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 728, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.4.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op608:layer4.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 729, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.4.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op609:layer4.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 730, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.4.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode1", + "op_id": "decode1:op609:layer4.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 731, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.4.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op610:layer4.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 732, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.4.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode1", + "op_id": "decode1:op610:layer4.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 733, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.4.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op612:layer4.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 734, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.4.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op613:layer4.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 735, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.4.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op614:layer4.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 736, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.4.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op615:layer4.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 737, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.4.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "decode1", + "op_id": "decode1:op616:layer4.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 738, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.4.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op617:layer5.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 739, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.5.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op618:layer5.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 740, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.5.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op618:layer5.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 741, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.5.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op619:layer5.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 742, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.5.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode1", + "op_id": "decode1:op619:layer5.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 743, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.5.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op620:layer5.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 744, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.5.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode1", + "op_id": "decode1:op620:layer5.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 745, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.5.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op622:layer5.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 746, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.5.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op623:layer5.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 747, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.5.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op624:layer5.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 748, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.5.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op625:layer5.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 749, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.5.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "decode1", + "op_id": "decode1:op626:layer5.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 750, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.5.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op627:layer6.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 751, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.6.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op628:layer6.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 752, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.6.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op628:layer6.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 753, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.6.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op629:layer6.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 754, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.6.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode1", + "op_id": "decode1:op629:layer6.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 755, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.6.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op630:layer6.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 756, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.6.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode1", + "op_id": "decode1:op630:layer6.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 757, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.6.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op632:layer6.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 758, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.6.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op633:layer6.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 759, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.6.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op634:layer6.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 760, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.6.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op635:layer6.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 761, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.6.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "decode1", + "op_id": "decode1:op636:layer6.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 762, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.6.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op637:layer7.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 763, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.7.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op638:layer7.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 764, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.7.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op638:layer7.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 765, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.7.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op639:layer7.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 766, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.7.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode1", + "op_id": "decode1:op639:layer7.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 767, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.7.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op640:layer7.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 768, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.7.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode1", + "op_id": "decode1:op640:layer7.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 769, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.7.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op642:layer7.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 770, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.7.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op643:layer7.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 771, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.7.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op644:layer7.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 772, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.7.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op645:layer7.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 773, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.7.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "decode1", + "op_id": "decode1:op646:layer7.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 774, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.7.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op647:layer8.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 775, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.8.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op648:layer8.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 776, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.8.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op648:layer8.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 777, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.8.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op649:layer8.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 778, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.8.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode1", + "op_id": "decode1:op649:layer8.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 779, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.8.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op650:layer8.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 780, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.8.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode1", + "op_id": "decode1:op650:layer8.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 781, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.8.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op652:layer8.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 782, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.8.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op653:layer8.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 783, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.8.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op654:layer8.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 784, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.8.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op655:layer8.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 785, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.8.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "decode1", + "op_id": "decode1:op656:layer8.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 786, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.8.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op657:layer9.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 787, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.9.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op658:layer9.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 788, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.9.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op658:layer9.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 789, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.9.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op659:layer9.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 790, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.9.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode1", + "op_id": "decode1:op659:layer9.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 791, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.9.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op660:layer9.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 792, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.9.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode1", + "op_id": "decode1:op660:layer9.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 793, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.9.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op662:layer9.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 794, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.9.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op663:layer9.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 795, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.9.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op664:layer9.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 796, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.9.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op665:layer9.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 797, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.9.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "decode1", + "op_id": "decode1:op666:layer9.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 798, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.9.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op667:layer10.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 799, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.10.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op668:layer10.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 800, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.10.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op668:layer10.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 801, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.10.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op669:layer10.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 802, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.10.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode1", + "op_id": "decode1:op669:layer10.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 803, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.10.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op670:layer10.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 804, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.10.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode1", + "op_id": "decode1:op670:layer10.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 805, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.10.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op672:layer10.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 806, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.10.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op673:layer10.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 807, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.10.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op674:layer10.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 808, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.10.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op675:layer10.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 809, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.10.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "decode1", + "op_id": "decode1:op676:layer10.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 810, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.10.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op677:layer11.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 811, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.11.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op678:layer11.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 812, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.11.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op678:layer11.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 813, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.11.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op679:layer11.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 814, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.11.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode1", + "op_id": "decode1:op679:layer11.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 815, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.11.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op680:layer11.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 816, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.11.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode1", + "op_id": "decode1:op680:layer11.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 817, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.11.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op682:layer11.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 818, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.11.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op683:layer11.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 819, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.11.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op684:layer11.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 820, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.11.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op685:layer11.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 821, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.11.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "decode1", + "op_id": "decode1:op686:layer11.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 822, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.11.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op687:layer12.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 823, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.12.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op688:layer12.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 824, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.12.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op688:layer12.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 825, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.12.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op689:layer12.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 826, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.12.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode1", + "op_id": "decode1:op689:layer12.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 827, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.12.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op690:layer12.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 828, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.12.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode1", + "op_id": "decode1:op690:layer12.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 829, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.12.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op692:layer12.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 830, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.12.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op693:layer12.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 831, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.12.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op694:layer12.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 832, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.12.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op695:layer12.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 833, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.12.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "decode1", + "op_id": "decode1:op696:layer12.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 834, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.12.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op697:layer13.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 835, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.13.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op698:layer13.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 836, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.13.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op698:layer13.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 837, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.13.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op699:layer13.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 838, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.13.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode1", + "op_id": "decode1:op699:layer13.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 839, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.13.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op700:layer13.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 840, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.13.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode1", + "op_id": "decode1:op700:layer13.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 841, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.13.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op702:layer13.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 842, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.13.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op703:layer13.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 843, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.13.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op704:layer13.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 844, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.13.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op705:layer13.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 845, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.13.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "decode1", + "op_id": "decode1:op706:layer13.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 846, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.13.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op707:layer14.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 847, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.14.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op708:layer14.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 848, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.14.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op708:layer14.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 849, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.14.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op709:layer14.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 850, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.14.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode1", + "op_id": "decode1:op709:layer14.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 851, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.14.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op710:layer14.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 852, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.14.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode1", + "op_id": "decode1:op710:layer14.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 853, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.14.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op712:layer14.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 854, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.14.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op713:layer14.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 855, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.14.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op714:layer14.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 856, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.14.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op715:layer14.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 857, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.14.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "decode1", + "op_id": "decode1:op716:layer14.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 858, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.14.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op717:layer15.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 859, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.15.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op718:layer15.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 860, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.15.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op718:layer15.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 861, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.15.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op719:layer15.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 862, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.15.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode1", + "op_id": "decode1:op719:layer15.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 863, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.15.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op720:layer15.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 864, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.15.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode1", + "op_id": "decode1:op720:layer15.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 865, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.15.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op722:layer15.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 866, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.15.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op723:layer15.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 867, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.15.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op724:layer15.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 868, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.15.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op725:layer15.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 869, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.15.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "decode1", + "op_id": "decode1:op726:layer15.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 870, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.15.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op727:layer16.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 871, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.16.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op728:layer16.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 872, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.16.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op728:layer16.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 873, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.16.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op729:layer16.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 874, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.16.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode1", + "op_id": "decode1:op729:layer16.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 875, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.16.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op730:layer16.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 876, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.16.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode1", + "op_id": "decode1:op730:layer16.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 877, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.16.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op732:layer16.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 878, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.16.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op733:layer16.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 879, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.16.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op734:layer16.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 880, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.16.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op735:layer16.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 881, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.16.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "decode1", + "op_id": "decode1:op736:layer16.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 882, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.16.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op737:layer17.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 883, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.17.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op738:layer17.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 884, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.17.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op738:layer17.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 885, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.17.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op739:layer17.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 886, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.17.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode1", + "op_id": "decode1:op739:layer17.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 887, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.17.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op740:layer17.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 888, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.17.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode1", + "op_id": "decode1:op740:layer17.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 889, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.17.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op742:layer17.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 890, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.17.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op743:layer17.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 891, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.17.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op744:layer17.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 892, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.17.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op745:layer17.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 893, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.17.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "decode1", + "op_id": "decode1:op746:layer17.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 894, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.17.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op747:layer18.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 895, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.18.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op748:layer18.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 896, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.18.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op748:layer18.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 897, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.18.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op749:layer18.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 898, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.18.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode1", + "op_id": "decode1:op749:layer18.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 899, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.18.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op750:layer18.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 900, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.18.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode1", + "op_id": "decode1:op750:layer18.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 901, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.18.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op752:layer18.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 902, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.18.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op753:layer18.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 903, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.18.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op754:layer18.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 904, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.18.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op755:layer18.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 905, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.18.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "decode1", + "op_id": "decode1:op756:layer18.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 906, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.18.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op757:layer19.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 907, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.19.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op758:layer19.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 908, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.19.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op758:layer19.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 909, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.19.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op759:layer19.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 910, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.19.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode1", + "op_id": "decode1:op759:layer19.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 911, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.19.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op760:layer19.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 912, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.19.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode1", + "op_id": "decode1:op760:layer19.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 913, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.19.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op762:layer19.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 914, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.19.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op763:layer19.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 915, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.19.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op764:layer19.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 916, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.19.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op765:layer19.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 917, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.19.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "decode1", + "op_id": "decode1:op766:layer19.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 918, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.19.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op767:layer20.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 919, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.20.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op768:layer20.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 920, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.20.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op768:layer20.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 921, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.20.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op769:layer20.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 922, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.20.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode1", + "op_id": "decode1:op769:layer20.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 923, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.20.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op770:layer20.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 924, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.20.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode1", + "op_id": "decode1:op770:layer20.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 925, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.20.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op772:layer20.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 926, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.20.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op773:layer20.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 927, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.20.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op774:layer20.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 928, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.20.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op775:layer20.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 929, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.20.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "decode1", + "op_id": "decode1:op776:layer20.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 930, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.20.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op777:layer21.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 931, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.21.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op778:layer21.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 932, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.21.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op778:layer21.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 933, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.21.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op779:layer21.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 934, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.21.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode1", + "op_id": "decode1:op779:layer21.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 935, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.21.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op780:layer21.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 936, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.21.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode1", + "op_id": "decode1:op780:layer21.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 937, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.21.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op782:layer21.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 938, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.21.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op783:layer21.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 939, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.21.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op784:layer21.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 940, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.21.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op785:layer21.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 941, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.21.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "decode1", + "op_id": "decode1:op786:layer21.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 942, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.21.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op787:layer22.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 943, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.22.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op788:layer22.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 944, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.22.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op788:layer22.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 945, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.22.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op789:layer22.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 946, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.22.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode1", + "op_id": "decode1:op789:layer22.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 947, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.22.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op790:layer22.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 948, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.22.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode1", + "op_id": "decode1:op790:layer22.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 949, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.22.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op792:layer22.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 950, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.22.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op793:layer22.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 951, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.22.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op794:layer22.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 952, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.22.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op795:layer22.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 953, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.22.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "decode1", + "op_id": "decode1:op796:layer22.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 954, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.22.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op797:layer23.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 955, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.23.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op798:layer23.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 956, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.23.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op798:layer23.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 957, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.23.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op799:layer23.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 958, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.23.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode1", + "op_id": "decode1:op799:layer23.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 959, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.23.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op800:layer23.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 960, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.23.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode1", + "op_id": "decode1:op800:layer23.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 961, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.23.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op802:layer23.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 962, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.23.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op803:layer23.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 963, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.23.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op804:layer23.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 964, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.23.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op805:layer23.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 965, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.23.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "decode1", + "op_id": "decode1:op806:layer23.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 966, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.23.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op807:layer24.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 967, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.24.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op808:layer24.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 968, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.24.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op808:layer24.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 969, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.24.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op809:layer24.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 970, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.24.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode1", + "op_id": "decode1:op809:layer24.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 971, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.24.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op810:layer24.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 972, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.24.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode1", + "op_id": "decode1:op810:layer24.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 973, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.24.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op812:layer24.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 974, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.24.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op813:layer24.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 975, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.24.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op814:layer24.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 976, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.24.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op815:layer24.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 977, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.24.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "decode1", + "op_id": "decode1:op816:layer24.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 978, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.24.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op817:layer25.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 979, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.25.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op818:layer25.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 980, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.25.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op818:layer25.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 981, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.25.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op819:layer25.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 982, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.25.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode1", + "op_id": "decode1:op819:layer25.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 983, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.25.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op820:layer25.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 984, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.25.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode1", + "op_id": "decode1:op820:layer25.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 985, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.25.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op822:layer25.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 986, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.25.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op823:layer25.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 987, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.25.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op824:layer25.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 988, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.25.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op825:layer25.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 989, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.25.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "decode1", + "op_id": "decode1:op826:layer25.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 990, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.25.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op827:layer26.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 991, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.26.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op828:layer26.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 992, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.26.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op828:layer26.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 993, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.26.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op829:layer26.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 994, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.26.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode1", + "op_id": "decode1:op829:layer26.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 995, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.26.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op830:layer26.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 996, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.26.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode1", + "op_id": "decode1:op830:layer26.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 997, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.26.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op832:layer26.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 998, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.26.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op833:layer26.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 999, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.26.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op834:layer26.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 1000, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.26.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op835:layer26.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 1001, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.26.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "decode1", + "op_id": "decode1:op836:layer26.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 1002, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.26.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op837:layer27.input_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 1003, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.27.input_layernorm.weight" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op838:layer27.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 1004, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.27.self_attn.q_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op838:layer27.q_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 1005, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.27.self_attn.q_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op839:layer27.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 1006, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.27.self_attn.k_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode1", + "op_id": "decode1:op839:layer27.k_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 1007, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.27.self_attn.k_proj.bias" + }, + { + "access_bytes": 1792, + "access_shape": [ + 8, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op840:layer27.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 1008, + "storage_dtype": "float32", + "storage_shape": [ + 8, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.27.self_attn.v_proj.weight" + }, + { + "access_bytes": 32, + "access_shape": [ + 8 + ], + "forward_id": "decode1", + "op_id": "decode1:op840:layer27.v_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 1009, + "storage_dtype": "float32", + "storage_shape": [ + 8 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.27.self_attn.v_proj.bias" + }, + { + "access_bytes": 12544, + "access_shape": [ + 56, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op842:layer27.o_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 1010, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.27.self_attn.o_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op843:layer27.post_attention_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 1011, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.27.post_attention_layernorm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op844:layer27.gate_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 1012, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.27.mlp.gate_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op845:layer27.up_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 1013, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.27.mlp.up_proj.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 56, + 128 + ], + "forward_id": "decode1", + "op_id": "decode1:op846:layer27.swiglu_down_projection", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 1014, + "storage_dtype": "float32", + "storage_shape": [ + 56, + 128 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.layers.27.mlp.down_proj.weight" + }, + { + "access_bytes": 224, + "access_shape": [ + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op847:final_rmsnorm", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 1015, + "storage_dtype": "float32", + "storage_shape": [ + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "model.norm.weight" + }, + { + "access_bytes": 28672, + "access_shape": [ + 128, + 56 + ], + "forward_id": "decode1", + "op_id": "decode1:op848:lm_head", + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + "sequence": 1016, + "storage_dtype": "float32", + "storage_shape": [ + 128, + 56 + ], + "target_projection_dtype_bytes": 2, + "weight_name": "lm_head.weight" + } + ] +} diff --git a/experiments/eq3_system_thermal/test_aggregate_maintenance_campaign.py b/experiments/eq3_system_thermal/test_aggregate_maintenance_campaign.py new file mode 100644 index 0000000..1c89afd --- /dev/null +++ b/experiments/eq3_system_thermal/test_aggregate_maintenance_campaign.py @@ -0,0 +1,13 @@ +import unittest + +from aggregate_maintenance_campaign import observed_max + + +class MaintenanceAggregateTests(unittest.TestCase): + def test_unavailable_owner_remains_unknown(self): + self.assertIsNone(observed_max([None, None])) + self.assertEqual(observed_max([None, 301.0, 302.0]), 302.0) + + +if __name__ == "__main__": + unittest.main() diff --git a/experiments/eq3_system_thermal/test_analyze_campaign.py b/experiments/eq3_system_thermal/test_analyze_campaign.py new file mode 100644 index 0000000..d842c55 --- /dev/null +++ b/experiments/eq3_system_thermal/test_analyze_campaign.py @@ -0,0 +1,103 @@ +import hashlib +import json +import os +from pathlib import Path +import tempfile +import unittest + +from analyze_campaign import analyze_campaign, analyze_point, write_outputs + + +def save(path: Path, value): + path.write_text(json.dumps(value, allow_nan=False) + "\n") + + +def make_point(root: Path, *, corrupt_backlog=False) -> Path: + point = root / "base-mixed_direct-384-0-01" + point.mkdir(parents=True) + config = {"point_id": point.name, "topology": "mixed_direct", "strategy": "guard_only", + "workload": {"active_ns": 20, "per_stack_Bps": 1_536_000_000_000}, + "recovery_ns": 20} + save(point / "config.json", config) + save(point / "manifest.json", { + "input_sha256": hashlib.sha256((point / "config.json").read_bytes()).hexdigest(), + "model_dir": "/models/mixed-full-2mm", + }) + save(point / "DONE.json", {"status": "COMPLETED", "summary": {}}) + rows = [] + for index, (offered, delivered, backlog) in enumerate(((100, 60, 40), (0, 40, 0))): + if corrupt_backlog and index == 1: + backlog = 1 + total_j = delivered * 50e-12 + rows.append({ + "start_ns": index * 20, "end_ns": (index + 1) * 20, + "service": { + "start_ns": index * 20, "end_ns": (index + 1) * 20, + "stacks": { + "hbf0": {"offered_effective_bytes": offered, + "delivered_effective_bytes": delivered, + "backlog_effective_bytes": backlog, + "media_activity_bytes": delivered, + "oldest_wait_ns": 20 if backlog else None}, + "hbm0": {"offered_effective_bytes": 0, + "delivered_effective_bytes": 0, + "backlog_effective_bytes": 0, + "media_activity_bytes": delivered, + "oldest_wait_ns": None}, + }, + "job_progress": [], "maintenance_completion_ids": [], + }, + "energy": {"component_energy_j": {"hbf0.base": total_j}, + "scope_energy_j": {"read:base": total_j}, "total_j": total_j}, + "thermal": {"temperatures": {"gpu": 301, "hbf0": 302 + index, "hbm0": 303}, + "stack_states": {"hbf0": "normal", "hbm0": "light" if index == 0 else "normal"}, + "entity_temperatures": {}, + "energy_j": {"window": {"activity_input_j": total_j, + "total_input_j": total_j}, + "cumulative": {"total_input_j": (60 if index == 0 else 100) * 50e-12}}}, + "control": {"budgets": {}, "next_budgets": {}}, + "reliability": {"status": "NO_MAINTENANCE_DEMAND_IN_BASE_RATE_WORKLOAD"}, + "causal": None, + }) + (point / "windows.jsonl").write_text("".join(json.dumps(row) + "\n" for row in rows)) + return point + + +class AnalyzerTest(unittest.TestCase): + def test_point_strict_conservation_and_unavailable_labels(self): + with tempfile.TemporaryDirectory() as td: + point = make_point(Path(td)) + result = analyze_point(point) + self.assertEqual(result["totals"]["offered_effective_bytes"], 100) + self.assertEqual(result["totals"]["delivered_effective_bytes"], 100) + self.assertEqual(result["totals"]["byte_conservation_error"], 0) + self.assertEqual(result["per_stack"]["hbm0"]["media_activity_bytes"], 100) + self.assertEqual(result["per_stack"]["hbm0"]["state_time_ns"]["light"], 20) + self.assertEqual(result["token_metric"]["status"], "UNAVAILABLE") + self.assertEqual(result["maintenance_metric"]["status"], "NOT_EXERCISED_NO_DEMAND") + self.assertEqual(set(result["_trace"]["temperatures_k_by_owner"]), + {"gpu", "hbf0", "hbm0"}) + + def test_detects_byte_failure(self): + with tempfile.TemporaryDirectory() as td: + point = make_point(Path(td), corrupt_backlog=True) + with self.assertRaisesRegex(ValueError, "byte conservation"): + analyze_point(point) + + def test_partial_campaign_outputs_six_panel_and_csv(self): + with tempfile.TemporaryDirectory() as td: + root = Path(td) + make_point(root) + analysis = analyze_campaign(root, require_complete=False) + self.assertEqual(analysis["completed_point_count"], 1) + output = root / "derived" + os.environ["MPLCONFIGDIR"] = str(root / "mpl-cache") + write_outputs(analysis, output, plots=True) + self.assertTrue((output / "per-stack-summary.csv").is_file()) + self.assertTrue((output / "six-panel-mixed-direct.png").is_file()) + self.assertTrue((output / "owner-temperatures-mixed-direct-guard-only-1536.png").is_file()) + self.assertIn("Token/s remains unavailable", (output / "SYSTEM_THERMAL_CAMPAIGN_ANALYSIS.md").read_text()) + + +if __name__ == "__main__": + unittest.main() diff --git a/experiments/eq3_system_thermal/test_analyze_extensions.py b/experiments/eq3_system_thermal/test_analyze_extensions.py new file mode 100644 index 0000000..eb670db --- /dev/null +++ b/experiments/eq3_system_thermal/test_analyze_extensions.py @@ -0,0 +1,202 @@ +#!/usr/bin/env python3 +import hashlib +import json +from pathlib import Path +import tempfile +import unittest + +from analyze_extensions import analyze_point, plot_panels + + +def save(path, value): + path.write_text(json.dumps(value, indent=2, allow_nan=False) + "\n") + + +def jsonl(path, values): + path.write_text("".join(json.dumps(value, separators=(",", ":")) + "\n" + for value in values)) + + +def thermal(start, end, cumulative): + return { + "start_ns": start, "end_ns": end, + "temperatures": {"gpu": 301.0, "hbf0": 302.0, "hbm0": 303.0}, + "stack_states": {"hbf0": "normal", "hbm0": "light"}, + "energy_j": {"cumulative": {"total_input_j": cumulative}}, + } + + +def identity(point, config): + save(point / "config.json", config) + config_hash = hashlib.sha256((point / "config.json").read_bytes()).hexdigest() + save(point / "manifest.json", { + "input_sha256": config_hash, "source_revision": "fixed-test-revision", + "source_sha256": {"runner.py": "b" * 64}, + "thermal_binary_sha256": "c" * 64, + }) + + +def service_stacks(window, cumulative): + return { + "hbf0": {"offered_effective_bytes": 50, "delivered_effective_bytes": 50, + "backlog_effective_bytes": 0, + "cumulative_offered_effective_bytes": cumulative, + "cumulative_delivered_effective_bytes": cumulative}, + "hbm0": {"offered_effective_bytes": 0, "delivered_effective_bytes": 0, + "backlog_effective_bytes": 0, + "cumulative_offered_effective_bytes": 0, + "cumulative_delivered_effective_bytes": 0}, + } + + +class AnalyzeExtensionTests(unittest.TestCase): + def setUp(self): + self.temp = tempfile.TemporaryDirectory() + self.root = Path(self.temp.name) + + def tearDown(self): + self.temp.cleanup() + + def maintenance_point(self): + point = self.root / "maintenance" + point.mkdir() + config = {"point_id": "maintenance-fixed", + "workload": {"active_ns": 10}, "recovery_ns": 10, + "maintenance": {"mode": "shared"}} + identity(point, config) + rows = [] + for index, (start, end) in enumerate(((0, 10), (10, 20))): + terminal = ([] if index == 0 else [{ + "operation_id": "m0", "status": "COMMITTED", "start_ns": 0, + "end_ns": 20, "extent_ids": ["e0"], "committed_extents": ["e0"], + "conflicted_extents": [], "failures": [], + "age_reset_extent_ids": ["e0"]}]) + rows.append({ + "start_ns": start, "end_ns": end, + "service": {"stacks": service_stacks(index, 50 * (index + 1)), + "activities": [{"operation": "read", "phase": "media_read", + "stack": "hbf0", "bytes": 50}], + "completion_ids": [f"f{index}"], "job_progress": []}, + "independent_maintenance_service": None, + "maintenance_delta": {"terminal_results": terminal}, + "energy": {"component_energy_j": {"hbf0.die0": 1.0}, "total_j": 1.0}, + "thermal": thermal(start, end, index + 1.0), + "control": {"observed_states": {"hbf0": "normal", "hbm0": "light"}}, + }) + jsonl(point / "windows.jsonl", rows) + final = { + "mode": "shared", + "driver": {"terminal_summary": {"operation_count": 1, "extent_count": 1, + "status_counts": {"COMMITTED": 1}}, + "free_spares_by_stack_channel": {"hbf0": {"0": ["s0"]}}, + "quarantined_blocks": [], + "physical_wear": {"src": {"block_program_work_started": 1, + "block_program_work_completed": 1, + "nand_page_programs_started": 256, + "nand_page_programs_completed": 256, + "erase_phase_started": 1, "erase_completed": 1}}, + "limitations": ["METADATA_VERSION_VALIDITY_NO_PAYLOAD_INTEGRITY"]}, + "reliability": {"blocks": {"e0": {"equivalent_age_ns": 7, + "last_refresh_commit_ns": 20}}}, + } + save(point / "maintenance-final.json", final) + final_hash = hashlib.sha256((point / "maintenance-final.json").read_bytes()).hexdigest() + save(point / "DONE.json", {"status": "COMPLETED", "summary": { + "offered_bytes": 100, "delivered_bytes": 100, "backlog_bytes": 0, + "energy_j": 2.0, "maintenance_final_sha256": final_hash}}) + return point + + def causal_point(self, duplicate_completion=False): + point = self.root / "causal" + point.mkdir() + config = {"point_id": "causal-fixed", "active_ns": 10, "recovery_ns": 10, + "trace": {"dependency_mode": "synthetic_metadata_dag"}} + identity(point, config) + rows = [] + for index, (start, end) in enumerate(((0, 10), (10, 20))): + completion = "j0" if duplicate_completion else f"j{index}" + events = ([{"kind": "cache_hit"}] if index == 0 else [ + {"kind": "prefetch_issue"}, {"kind": "migration_commit", "bytes": 8}, + {"kind": "retry_complete", "retry_count": 1}]) + rows.append({ + "start_ns": start, "end_ns": end, + "service": {"activities": [{"operation": "read", "phase": "media_read", + "stack": "hbf0", "bytes": 50}], + "changed_job_progress": [{"remaining_bytes": 0}], + "completions": [{"job_id": completion}]}, + "executor": {"events": events, "completed_tokens": 1, + "cumulative_completed_tokens": index + 1}, + "maintenance": {"mode": "disabled"}, + "energy": {"component_energy_j": {"hbf0.die0": 1.0}, "total_j": 1.0}, + "thermal": thermal(start, end, index + 1.0), + "control": {"observed_states": {"hbf0": "normal", "hbm0": "light"}, + "stack_facts": { + "hbf0": {"offered_bytes": 50, "delivered_bytes": 50, + "backlog_bytes": 0}, + "hbm0": {"offered_bytes": 0, "delivered_bytes": 0, + "backlog_bytes": 0}}}, + "output_granularity": "PHYSICAL_CHANNEL_JOB_DELTAS", + }) + jsonl(point / "windows.jsonl", rows) + save(point / "DONE.json", {"status": "COMPLETED", "summary": { + "trace_origin": "SYNTHETIC_ARCHITECTURE_METADATA_DAG", + "structure_provenance": "SYNTHETIC_METADATA_ONLY", + "completed_tokens": 2, "uninstantiated_batches": 0, + "pending_external_jobs": 0, + "offered_useful_bytes_by_stack": {"hbf0": 100}, + "delivered_useful_bytes_by_stack": {"hbf0": 100}, + "maintenance": {"mode": "disabled"}, "energy_j": 2.0}}) + return point + + def test_complete_maintenance_receipts_and_wear(self): + result = analyze_point(self.maintenance_point()) + self.assertEqual(result["analysis_status"], "VALIDATED_COMPLETE_RECEIPTS") + self.assertEqual(result["maintenance"]["terminal_status_counts"], {"COMMITTED": 1}) + self.assertEqual(result["maintenance"]["wear"]["nand_page_programs_completed"], 256) + self.assertEqual(result["causal_tokens"]["availability"], + "UNAVAILABLE_RATE_WORKLOAD_HAS_NO_TOKEN_DEPENDENCY_DAG") + + def test_complete_causal_receipts_consumers_and_plot(self): + result = analyze_point(self.causal_point()) + self.assertEqual(result["causal_tokens"]["completed_tokens"], 2) + self.assertEqual(result["causal_consumers"]["cache"], {"cache_hit": 1}) + self.assertEqual(result["causal_consumers"]["retry"]["observed_retry_count"], 1) + output = self.root / "panels.png" + plot_panels(result, output) + self.assertGreater(output.stat().st_size, 0) + + def test_expected_retry_proxy_does_not_claim_integer_count(self): + point = self.causal_point() + config = json.loads((point / "config.json").read_text()) + config["hbf_read_cost_proxy"] = {"mode": "conditional_nand_history_v1"} + identity(point, config) + result = analyze_point(point) + retry = result["causal_consumers"]["retry"] + self.assertIsNone(retry["observed_retry_count"]) + self.assertEqual(retry["count_semantics"], + "UNKNOWN_INTEGER_COUNT_EXPECTED_WORK_PROXY") + + def test_failed_run_is_preserved_without_claim(self): + point = self.root / "failed" + point.mkdir() + save(point / "FAILED.json", {"status": "FAILED", "error": "fixed failure"}) + result = analyze_point(point) + self.assertEqual(result["analysis_status"], "FAILED_RUN_PRESERVED_NOT_ANALYZED") + self.assertEqual(result["scientific_pass"], "NOT_ASSESSED") + self.assertIsNone(result["panels"]) + + def test_duplicate_causal_terminal_is_rejected(self): + with self.assertRaisesRegex(ValueError, "duplicate causal completion"): + analyze_point(self.causal_point(duplicate_completion=True)) + + def test_energy_timeline_mismatch_is_rejected(self): + point = self.maintenance_point() + rows = list(map(json.loads, (point / "windows.jsonl").read_text().splitlines())) + rows[1]["thermal"]["energy_j"]["cumulative"]["total_input_j"] = 3.0 + jsonl(point / "windows.jsonl", rows) + with self.assertRaisesRegex(ValueError, "thermal cumulative energy"): + analyze_point(point) + + +if __name__ == "__main__": + unittest.main() diff --git a/experiments/eq3_system_thermal/test_analyze_uncontrolled_first_constraint.py b/experiments/eq3_system_thermal/test_analyze_uncontrolled_first_constraint.py new file mode 100644 index 0000000..e9ebb97 --- /dev/null +++ b/experiments/eq3_system_thermal/test_analyze_uncontrolled_first_constraint.py @@ -0,0 +1,34 @@ +from pathlib import Path +import tempfile +import unittest + +from analyze_uncontrolled_first_constraint import first_crossings, last_error_response + + +class FirstCrossingTests(unittest.TestCase): + def test_last_error_skips_trailing_quit_command(self): + with tempfile.TemporaryDirectory() as temporary: + path = Path(temporary) / "transcript.jsonl" + path.write_text('{"response":{"type":"ERROR","status":"DOMAIN_FAILURE"}}\n' + '{"command":"QUIT"}\n') + self.assertEqual(last_error_response(path)["status"], "DOMAIN_FAILURE") + + def test_ties_and_twenty_ms_interval(self): + rows = [ + {"end_ns": 20_000_000, "thermal": {"temperatures": { + "hbf0": 352.0, "hbf1": 352.0, "gpu": 350.0}}}, + {"end_ns": 40_000_000, "thermal": {"temperatures": { + "hbf0": 354.0, "hbf1": 354.0, "gpu": 350.0}}}, + ] + result = first_crossings(rows, { + "hbf": [353.15, 363.15, 378.15], "hbm": [353.15, 363.15, 378.15], + "gpu": [363.15, 373.15, 383.15]}) + self.assertEqual(result["light"]["end_ns"], 40_000_000) + self.assertEqual(result["light"]["interval_ns"], { + "lower_exclusive": 20_000_000, "upper_inclusive": 40_000_000}) + self.assertEqual(result["light"]["tied_owners"], ["hbf0", "hbf1"]) + self.assertIsNone(result["severe"]) + + +if __name__ == "__main__": + unittest.main() diff --git a/experiments/eq3_system_thermal/test_causal_maintenance_age.py b/experiments/eq3_system_thermal/test_causal_maintenance_age.py new file mode 100644 index 0000000..58c4144 --- /dev/null +++ b/experiments/eq3_system_thermal/test_causal_maintenance_age.py @@ -0,0 +1,67 @@ +import unittest + +from causal_maintenance_age import CausalMaintenanceAgeAdapter +from causal_service import CausalTopologyService +from maintenance_driver import MaintenanceDriver +from reliability import DAY_NS, ReliabilityLedger +from topology_service import default_config + + +W = 20_000_000 + + +class CausalMaintenanceAgeTests(unittest.TestCase): + def test_exact_completion_commit_then_window_tail_age(self): + config = default_config("mixed_direct") + service = CausalTopologyService(config) + ledger = ReliabilityLedger({"ea_ev": 1.04, + "refresh_trigger": "equivalent_age_or_wall", + "initial_equivalent_age_ns": DAY_NS}) + driver = MaintenanceDriver(ledger, { + "block_bytes": 4096, "pages_per_block": 1, "max_blocks_per_cohort": 1, + "spare_block_ids_by_stack_channel": {"hbf0": {"0": ["spare0"]}}, + "program_energy_j_per_byte": 0.0, "erase_energy_j_per_operation": 0.0, + "energy_evidence": {"program": "FIXED_TEST", "erase": "FIXED_TEST"}, + }) + driver.register_extents([{ + "extent_id": "extent0", "stack": "hbf0", "channel": "0", + "source_block_id": "source0", "version": 0, + }]) + temperatures = {"hbf0": {"0": 358.15}} + age = CausalMaintenanceAgeAdapter(ledger, driver, temperatures) + budgets = {stack: 10**12 for stack in config["channels"]} + states = {stack: "normal" for stack in config["channels"]} + service.begin_window(0, W, budgets, states) + service.submit_jobs(age.start_window(0)) + while service.now_ns < W: + horizon = service.next_event_ns() + receipt = service.advance_to(horizon) + delta = age.consume_receipt(receipt) + if delta["next_phase_jobs"] and service.now_ns < W: + service.submit_jobs(delta["next_phase_jobs"]) + age.finish_window(W, {"hbf0": {"0": 360.0}}) + state = ledger.snapshot()["blocks"]["extent0"] + self.assertIsNotNone(state["last_refresh_commit_ns"]) + self.assertLess(state["last_refresh_commit_ns"], W) + self.assertGreater(state["equivalent_age_ns"], 0) + self.assertEqual(state["last_update_ns"], W) + self.assertEqual(driver.snapshot()["terminal_summary"]["status_counts"], + {"COMMITTED": 1}) + + def test_rejects_missing_temperature_identity(self): + ledger = ReliabilityLedger({"ea_ev": 1.04}) + driver = MaintenanceDriver(ledger, { + "block_bytes": 4096, "pages_per_block": 1, "max_blocks_per_cohort": 1, + "spare_block_ids_by_stack_channel": {"hbf0": {"0": ["spare0"]}}, + "program_energy_j_per_byte": None, "erase_energy_j_per_operation": None, + "energy_evidence": {"program": "UNKNOWN", "erase": "UNKNOWN"}, + }) + driver.register_extents([{"extent_id": "extent0", "stack": "hbf0", + "channel": "0", "source_block_id": "source0", + "version": 0}]) + with self.assertRaisesRegex(ValueError, "cover"): + CausalMaintenanceAgeAdapter(ledger, driver, {}) + + +if __name__ == "__main__": + unittest.main() diff --git a/experiments/eq3_system_thermal/test_causal_service.py b/experiments/eq3_system_thermal/test_causal_service.py new file mode 100644 index 0000000..8f7eaa6 --- /dev/null +++ b/experiments/eq3_system_thermal/test_causal_service.py @@ -0,0 +1,345 @@ +#!/usr/bin/env python3 +import unittest + +from causal_service import CausalTopologyService +from topology_service import default_config, media_cost_from_service_rate + + +W = 20_000_000 + + +def budgets(config, value=10**12): + return {stack: value for stack in config["channels"]} + + +def states(config, value="normal"): + return {stack: value for stack in config["channels"]} + + +def advance_next_completion(service): + while True: + horizon = service.next_completion_ns() + if horizon is None: + horizon = service.next_event_ns() + receipt = service.advance_to(horizon) + if receipt["completion_ids"]: + return receipt + if receipt.get("window_complete"): + raise AssertionError("no completion within control window") + + +class CausalServiceTests(unittest.TestCase): + def test_completion_and_dependency_chain_are_not_window_quantized(self): + config = default_config("all_hbf_direct") + service = CausalTopologyService(config) + service.begin_window(0, W, budgets(config), states(config)) + service.submit_jobs([{ + "job_id": "layer0", "stack": "hbf0", "channel": "0", "route": "direct", + "operation": "read", "bytes": 4096, "arrival_ns": 0, + "metadata": {"tensor_id": "t0"}, + }]) + first = advance_next_completion(service) + self.assertEqual(first["completion_ids"], ["layer0"]) + completion = next(row for row in first["job_progress"] + if row["job_id"] == "layer0")["completion_ns"] + service.submit_jobs([{ + "job_id": "layer1", "stack": "hbf0", "channel": "1", "route": "direct", + "operation": "read", "bytes": 4096, "arrival_ns": completion, + "metadata": {"tensor_id": "t1"}, + }]) + second = advance_next_completion(service) + self.assertEqual(second["completion_ids"], ["layer1"]) + + def test_hbm_local_and_relay_share_exact_bandwidth(self): + config = default_config("relay") + config["fabric"]["hbm"]["hbm0"]["gpu_link"] = { + "latency_ns": 0, "bandwidth_bytes_per_s": 1_000, + } + service = CausalTopologyService(config) + service.begin_window(0, W, budgets(config), states(config)) + service.submit_jobs([ + {"job_id": "relay", "stack": "hbf0", "channel": "0", "route": "relay", + "operation": "read", "bytes": 100, "arrival_ns": 0}, + {"job_id": "local", "stack": "hbm0", "channel": "0", "route": "direct", + "operation": "read", "bytes": 100, "arrival_ns": 0}, + ]) + receipt = service.advance_to(W) + progress = {row["job_id"]: row for row in receipt["job_progress"]} + transferred = sum(progress[job]["inflight_transferred_bytes"] + for job in ("relay", "local")) + self.assertLessEqual(transferred, 20) + self.assertEqual(progress["relay"]["state"], "ACTIVE") + self.assertEqual(progress["local"]["state"], "ACTIVE") + + def test_two_bank_analytical_occupancy_is_bounded_with_many_flows(self): + config = default_config("dash") + service = CausalTopologyService(config) + service.begin_window(0, W, budgets(config), states(config)) + service.submit_jobs([ + {"job_id": "a", "stack": "hbf0", "channel": "0", "route": "direct", + "operation": "read", "bytes": 10**9, "arrival_ns": 0}, + {"job_id": "b", "stack": "hbf0", "channel": "1", "route": "relay", + "operation": "read", "bytes": 10**9, "arrival_ns": 0}, + {"job_id": "c", "stack": "hbf0", "channel": "2", "route": "direct", + "operation": "read", "bytes": 10**9, "arrival_ns": 0}, + ]) + receipt = service.advance_to(1) + bound = receipt["buffer_occupancy_bounds"]["hbf0"] + self.assertLessEqual( + bound["maximum_occupancy_bytes"], + bound["bank_count"] * bound["bank_capacity_bytes"], + ) + progress = {row["job_id"]: row for row in receipt["job_progress"]} + self.assertEqual(sum(row["state"] == "ACTIVE" for row in progress.values()), 3) + + def test_future_quota_reserved_once_and_new_shutdown_work_blocked(self): + config = default_config("mixed_direct") + service = CausalTopologyService(config) + b = budgets(config) + b["hbf0"] = 100 + light = states(config) + light["hbf0"] = "light" + service.begin_window(0, W, b, light) + service.submit_jobs([ + {"job_id": "a", "stack": "hbf0", "channel": "0", "operation": "read", + "route": "direct", "bytes": 100, "arrival_ns": 0}, + {"job_id": "b", "stack": "hbf0", "channel": "1", "operation": "read", + "route": "direct", "bytes": 100, "arrival_ns": 0}, + ]) + receipt = service.advance_to(W) + self.assertEqual(receipt["endpoint_quota_remaining_scaled"]["hbf0"], 0) + admitted = {row["job_id"]: row["unadmitted_bytes"] for row in receipt["job_progress"]} + self.assertEqual(sum(100 - value for value in admitted.values()), 100) + + shutdown = states(config) + shutdown["hbf0"] = "shutdown" + service.begin_window(W, 2 * W, b, shutdown) + service.submit_jobs([{ + "job_id": "new", "stack": "hbf0", "channel": "2", "operation": "read", + "route": "direct", "bytes": 10, "arrival_ns": W, + }]) + second = service.advance_to(2 * W) + row = next(row for row in second["job_progress"] if row["job_id"] == "new") + self.assertEqual(row["state"], "QUEUED") + self.assertEqual(row["remaining_bytes"], 10) + + def test_maintenance_severe_allowed_shutdown_waits(self): + config = default_config("mixed_direct") + service = CausalTopologyService(config) + severe = states(config) + severe["hbf0"] = "severe" + service.begin_window(0, W, budgets(config), severe) + service.submit_jobs([{ + "job_id": "program", "maintenance_id": "program", "stack": "hbf0", + "channel": "0", "operation": "program", "bytes": 4096, "arrival_ns": 0, + }]) + receipt = service.advance_to(W) + self.assertEqual(receipt["maintenance_completion_ids"], ["program"]) + + def test_program_service_cost_reduces_rate_on_shared_media(self): + config = default_config("mixed_direct") + config["operation_media_cost"]["program"] = media_cost_from_service_rate( + 96_000_000_000, 16 * 40_960_000 + ) + progress = {} + for operation in ("read", "program"): + service = CausalTopologyService(config) + service.begin_window(0, W, budgets(config), states(config)) + job = {"job_id": operation, "stack": "hbf0", "channel": "0", + "operation": operation, "bytes": 10**9, "arrival_ns": 0} + if operation == "read": + job["route"] = "direct" + else: + job["maintenance_id"] = operation + service.submit_jobs([job]) + receipt = service.advance_to(1_000_000) + row = next(row for row in receipt["job_progress"] if row["job_id"] == operation) + progress[operation] = row["inflight_transferred_bytes"] + read_progress = progress["read"] + program_progress = progress["program"] + self.assertGreater(read_progress, program_progress) + + def test_subwindow_advance_does_not_reset_endpoint_quota(self): + config = default_config("mixed_direct") + service = CausalTopologyService(config) + limited = budgets(config) + limited["hbf0"] = 100 + service.begin_window(0, W, limited, states(config)) + service.submit_jobs([{ + "job_id": "large", "stack": "hbf0", "channel": "0", "operation": "read", + "route": "direct", "bytes": 1000, "arrival_ns": 0, + }]) + first = service.advance_to(1) + second = service.advance_to(2) + self.assertEqual(first["endpoint_quota_remaining_scaled"]["hbf0"], 0) + self.assertEqual(second["endpoint_quota_remaining_scaled"]["hbf0"], 0) + row = next(row for row in second["job_progress"] if row["job_id"] == "large") + self.assertEqual(row["unadmitted_bytes"], 900) + + def test_inflight_slice_drains_after_new_shutdown_state(self): + config = default_config("all_hbf_direct") + config["fabric"]["hbf"]["hbf0"]["direct_link"]["latency_ns"] = W + 10 + service = CausalTopologyService(config) + service.begin_window(0, W, budgets(config), states(config)) + service.submit_jobs([{ + "job_id": "old", "stack": "hbf0", "channel": "0", "operation": "read", + "route": "direct", "bytes": 1, "arrival_ns": 0, + }]) + service.advance_to(W) + shutdown = states(config) + shutdown["hbf0"] = "shutdown" + service.begin_window(W, 2 * W, budgets(config), shutdown) + receipt = advance_next_completion(service) + self.assertEqual(receipt["completion_ids"], ["old"]) + + def test_many_channels_share_resources_without_bank_per_flow_cap(self): + config = default_config("all_hbf_direct") + service = CausalTopologyService(config) + service.begin_window(0, W, budgets(config), states(config)) + service.submit_jobs([ + {"job_id": name, "stack": "hbf0", "channel": str(index), + "operation": "read", "route": "direct", "bytes": 1, "arrival_ns": 0} + for index, name in enumerate(("a", "b", "c")) + ]) + first = service.advance_to(1) + progress = {row["job_id"]: row for row in first["job_progress"]} + self.assertNotIn("QUEUED", {row["state"] for row in progress.values()}) + + def test_shutdown_maintenance_waits_then_resumes(self): + config = default_config("mixed_direct") + service = CausalTopologyService(config) + shutdown = states(config) + shutdown["hbf0"] = "shutdown" + service.begin_window(0, W, budgets(config), shutdown) + service.submit_jobs([{ + "job_id": "m", "maintenance_id": "m", "stack": "hbf0", "channel": "0", + "operation": "erase", "bytes": 1, "arrival_ns": 0, + }]) + blocked = service.advance_to(W) + self.assertEqual(blocked["completion_ids"], []) + service.begin_window(W, 2 * W, budgets(config), states(config)) + resumed = advance_next_completion(service) + self.assertEqual(resumed["maintenance_completion_ids"], ["m"]) + + def test_boundary_does_not_admit_with_expired_window_state(self): + config = default_config("mixed_direct") + service = CausalTopologyService(config) + zero = budgets(config) + zero["hbf0"] = 0 + service.begin_window(0, W, zero, states(config)) + service.submit_jobs([{ + "job_id": "boundary", "stack": "hbf0", "channel": "0", + "operation": "read", "route": "direct", "bytes": 1, "arrival_ns": 0, + }]) + first = service.advance_to(W) + row = next(row for row in first["job_progress"] if row["job_id"] == "boundary") + self.assertEqual(row["state"], "QUEUED") + service.begin_window(W, 2 * W, budgets(config), states(config)) + second = advance_next_completion(service) + self.assertEqual(second["completion_ids"], ["boundary"]) + + def test_bank_turnover_does_not_repeat_pipeline_latency(self): + config = default_config("all_hbf_direct") + row = config["fabric"]["hbf"]["hbf0"] + row["bank_capacity_bytes"] = 1 + row["fill"] = {"latency_ns": 10, "bandwidth_bytes_per_s": 1_000_000_000} + row["direct_link"] = {"latency_ns": 10, "bandwidth_bytes_per_s": 1_000_000_000} + config["channels"]["hbf0"]["0"] = 1_000_000_000 + service = CausalTopologyService(config) + service.begin_window(0, W, budgets(config), states(config)) + service.submit_jobs([{ + "job_id": "turnover", "stack": "hbf0", "channel": "0", + "operation": "read", "route": "direct", "bytes": 3, "arrival_ns": 0, + }]) + receipt = advance_next_completion(service) + row = next(row for row in receipt["job_progress"] if row["job_id"] == "turnover") + self.assertEqual(row["completion_ns"], 23) + + def test_completion_id_emitted_once(self): + config = default_config("all_hbf_direct") + service = CausalTopologyService(config) + service.begin_window(0, W, budgets(config), states(config)) + service.submit_jobs([{ + "job_id": "once", "stack": "hbf0", "channel": "0", "operation": "read", + "route": "direct", "bytes": 1, "arrival_ns": 0, + }]) + first = advance_next_completion(service) + self.assertEqual(first["completion_ids"], ["once"]) + second = service.advance_to(W) + self.assertEqual(second["completion_ids"], []) + + def test_hbm_fill_uses_shared_half_duplex_gpu_link_and_is_not_useful_read(self): + config = default_config("relay") + config["fabric"]["hbm"]["hbm0"]["gpu_link"] = { + "latency_ns": 0, "bandwidth_bytes_per_s": 1_000, + } + service = CausalTopologyService(config) + service.begin_window(0, W, budgets(config), states(config)) + service.submit_jobs([ + {"job_id": "fill", "stack": "hbm0", "channel": "0", + "operation": "hbm_fill", "bytes": 100, "arrival_ns": 0}, + {"job_id": "read", "stack": "hbm0", "channel": "1", "route": "direct", + "operation": "read", "bytes": 100, "arrival_ns": 0}, + ]) + receipt = service.advance_to(W) + progress = {row["job_id"]: row for row in receipt["job_progress"]} + transferred = sum(progress[job]["inflight_transferred_bytes"] + for job in ("fill", "read")) + self.assertLessEqual(transferred, 20) + fill_phases = {row["phase"] for row in receipt["activities"] + if row["operation"] == "hbm_fill"} + self.assertEqual(fill_phases, {"media_fill", "hbm_base_fill", "gpu_link_fill"}) + + def test_migration_program_uses_program_media_and_reverse_relay_path(self): + config = default_config("relay") + config["operation_media_cost"]["program"] = media_cost_from_service_rate( + 96_000_000_000, 16 * 40_960_000 + ) + service = CausalTopologyService(config) + service.begin_window(0, W, budgets(config), states(config)) + service.submit_jobs([{ + "job_id": "move:program", "stack": "hbf0", "channel": "0", + "operation": "migration_program", "route": "relay", + "bytes": 4096, "arrival_ns": 0, + "metadata": {"source_job_id": "move:read"}, + }]) + receipt = service.advance_to(W) + phases = {row["phase"] for row in receipt["activities"]} + self.assertEqual(phases, { + "media_program", "destination_base", "destination_fill", + "reverse_relay", "partner_gpu_receive", + }) + self.assertEqual(receipt["completion_ids"], ["move:program"]) + + def test_dash_route_children_share_one_explicit_media_group(self): + config = default_config("dash") + groups = {} + for stack in config["fabric"]["hbf"]: + config["channels"][stack] = {"direct": 1_000, "relay": 1_000} + config["dash_routes"][stack] = {"direct": "direct", "relay": "relay"} + row = {"resource_id": f"{stack}:uniform-media-group", + "bandwidth_bytes_per_s": 1_000} + groups[stack] = {"direct": dict(row), "relay": dict(row)} + for stack in config["fabric"]["hbm"]: + config["channels"][stack] = {"local": 1_000} + groups[stack] = {"local": { + "resource_id": f"{stack}:uniform-media-group", + "bandwidth_bytes_per_s": 1_000}} + config["causal_channel_groups"] = groups + service = CausalTopologyService(config) + service.begin_window(0, W, budgets(config), states(config)) + service.submit_jobs([ + {"job_id": "direct", "stack": "hbf0", "channel": "direct", + "operation": "read", "route": "direct", "bytes": 100, "arrival_ns": 0}, + {"job_id": "relay", "stack": "hbf0", "channel": "relay", + "operation": "read", "route": "relay", "bytes": 100, "arrival_ns": 0}, + ]) + receipt = service.advance_to(W) + progress = {row["job_id"]: row for row in receipt["job_progress"]} + transferred = sum(row["inflight_transferred_bytes"] for row in progress.values()) + self.assertLessEqual(transferred, 20) + self.assertIn("hbf0:uniform-media-group", receipt["resource_work_units_scaled"]) + + +if __name__ == "__main__": + unittest.main() diff --git a/experiments/eq3_system_thermal/test_ecc_cost_proxy.py b/experiments/eq3_system_thermal/test_ecc_cost_proxy.py new file mode 100644 index 0000000..a13a7c4 --- /dev/null +++ b/experiments/eq3_system_thermal/test_ecc_cost_proxy.py @@ -0,0 +1,33 @@ +import json +from copy import deepcopy +from pathlib import Path +import unittest +from ecc_cost_proxy import ReadCostProxy,ProxyDomainError,DAY_NS +PROFILE=json.loads((Path(__file__).parent/'ecc_proxy/profile_v1.json').read_text()) + +def proxy(age=0,pe=0,temp=303.15,strength=1): + p=deepcopy(PROFILE);p['transfer_strength']=strength + return ReadCostProxy(p,{'hbf0':{'equivalent_age_days_30c':age,'pe_cycles':pe,'temperature_k':temp}}) + +class CostTests(unittest.TestCase): + def test_published_anchor_and_separate_scenario_bounds(self): + self.assertEqual(proxy().cost('hbf0',0)['expected_retry_steps'],0) + self.assertAlmostEqual(proxy(365,2000).cost('hbf0',0)['expected_retry_steps'],19.9) + self.assertGreater(proxy(90,0).cost('hbf0',0)['expected_retry_steps'],3) + self.assertGreaterEqual(proxy(90,1000).cost('hbf0',0)['expected_retry_steps'],8) + self.assertGreaterEqual(proxy(180,0).cost('hbf0',0)['expected_retry_steps'],.544*7) + def test_history_hotter_accumulates_more_without_changing_wall_time(self): + cold,hot=proxy(temp=303.15),proxy(temp=358.15) + for p in (cold,hot):p.observe(0,20_000_000_000,{'hbf0':303.15}) + self.assertGreater(hot.cost('hbf0',20_000_000_000)['expected_retry_steps'],cold.cost('hbf0',20_000_000_000)['expected_retry_steps']) + self.assertEqual(hot.states['hbf0']['last_ns'],20_000_000_000) + def test_cooling_never_clears_previous_damage_or_instantly_increases_cost(self): + p=proxy(age=10,temp=358.15);before=p.cost('hbf0',0) + p.observe(0,0,{'hbf0':303.15});after=p.cost('hbf0',0) + self.assertEqual(before['expected_retry_steps'],after['expected_retry_steps']) + def test_null_transfer_and_domain_failures(self): + self.assertEqual(proxy(365,2000,strength=0).cost('hbf0',0)['attempt_work_milli'],1000) + with self.assertRaises(ProxyDomainError):proxy(366) + with self.assertRaises(ProxyDomainError):proxy(pe=2001) + with self.assertRaises(ValueError):ReadCostProxy(PROFILE,{'hbm0':{'equivalent_age_days_30c':0,'pe_cycles':0,'temperature_k':300}}) +if __name__=='__main__':unittest.main() diff --git a/experiments/eq3_system_thermal/test_ecc_runner.py b/experiments/eq3_system_thermal/test_ecc_runner.py new file mode 100644 index 0000000..4c02f23 --- /dev/null +++ b/experiments/eq3_system_thermal/test_ecc_runner.py @@ -0,0 +1,78 @@ +"""Fixed integration checks, no numerical thermal solve or research matrix.""" +import io +import json +from copy import deepcopy +from pathlib import Path +import unittest +from run_causal_point import execute +from topology_service import default_config +from test_run_causal_point import normalized, FakeThermal, trace, energy_profile + +PROFILE=json.loads((Path(__file__).parent/'ecc_proxy/profile_v1.json').read_text()) + +class RunnerIntegrationTests(unittest.TestCase): + def run_case(self, strength, *, history_probe=False, hot=False): + service=default_config('mixed_direct') + baseline={s:10**12 for s in service['channels']} + p=deepcopy(PROFILE);p['transfer_strength']=strength + config={ + 'point_id':'ecc-fixed','active_ns':20_000_000,'recovery_ns':20_000_000, + 'window_ns':20_000_000,'strategy':'guard_only','service':service, + 'energy':energy_profile(),'target_bytes_per_s_by_stack':{s:1 for s in baseline}, + 'trace':{'total_batches':1,'max_active_batches':1,'batch_interval_ns':100}, + 'executor':{'cache_mode':'disabled','cache_capacity_bytes':0, + 'coalescing_enabled':True,'prefetch_wait_mode':'wait_at_consumption', + 'stripe_unit_bytes':4096,'migration_mode':'fixed','retry_count_per_source_read':0, + 'stripe_targets':[{'stack':'hbf0','channel':'0','route':'direct'}]}, + 'hbf_read_cost_proxy':{'mode':'conditional_nand_history_v1','profile':p, + 'initial_by_stack':{s:{'equivalent_age_days_30c':(0 if history_probe else 90),'pe_cycles':(0 if history_probe else 1000), + 'temperature_k':300} for s in service['fabric']['hbf']}}} + def history_trace(index): + result=trace(index) + result['batches'][0]['arrival_ns']=index*20_000_000 + return result + class ProbeThermal(FakeThermal): + def advance(self,start,end,energy): + result=super().advance(start,end,energy) + if hot: + for value in result['entity_temperatures_k'].values(): + value.update(hotspot_k=358.15,mean_k=358.15) + return result + factory=history_trace if history_probe else trace + if history_probe: + config['active_ns']=60_000_000;config['recovery_ns']=0 + config['trace'].update(total_batches=3,batch_interval_ns=20_000_000) + sink=io.StringIO() + result=execute(config,normalized(service),ProbeThermal(sorted(baseline),baseline),sink, + initial_trace=factory(0),trace_factory=factory) + return result,[json.loads(x) for x in sink.getvalue().splitlines()] + + def test_real_consumer_changes_physical_work_energy_not_useful_payload(self): + base,brows=self.run_case(0);cost,rows=self.run_case(.1) + self.assertEqual(base['completed_tokens'],cost['completed_tokens']) + self.assertEqual(base['delivered_useful_bytes_by_stack'],cost['delivered_useful_bytes_by_stack']) + self.assertGreater(cost['energy_j'],base['energy_j']) + activities=[a for w in rows for a in w['service']['activities']] + media=sum(a['bytes'] for a in activities if a['phase']=='media_read') + self.assertGreater(media,4096) + self.assertAlmostEqual(cost['energy_j'],media*50e-12) + decisions=rows[0]['hbf_read_cost_proxy']['admission_cost_decisions'] + self.assertEqual(len(decisions),1) + self.assertEqual(decisions[0]['attempt_work_milli'],1900) + self.assertGreater(rows[0]['service']['completions'][0]['completion_ns'], + brows[0]['service']['completions'][0]['completion_ns']) + self.assertEqual(rows[-1]['hbf_read_cost_proxy']['state']['states']['hbf0']['last_ns'],40_000_000) + self.assertEqual(rows[-1]['hbf_read_cost_proxy']['admission_cost_decisions'],[]) + + def test_previous_thermal_observation_changes_only_later_admission_effort(self): + _,cold=self.run_case(.1,history_probe=True) + _,hot=self.run_case(.1,history_probe=True,hot=True) + costs=lambda rows:[d['expected_retry_steps'] for row in rows + for d in row['hbf_read_cost_proxy']['admission_cost_decisions']] + c,h=costs(cold),costs(hot) + self.assertEqual(len(c),3) + self.assertEqual(c[0],h[0]) + self.assertEqual(c[1],h[1]) # same already-elapsed first-window temperature + self.assertGreater(h[2],c[2]) + +if __name__=='__main__':unittest.main() diff --git a/experiments/eq3_system_thermal/test_ecc_service_adapter.py b/experiments/eq3_system_thermal/test_ecc_service_adapter.py new file mode 100644 index 0000000..bb87ad9 --- /dev/null +++ b/experiments/eq3_system_thermal/test_ecc_service_adapter.py @@ -0,0 +1,45 @@ +from copy import deepcopy +import unittest +from causal_service import CausalTopologyService +from ecc_service_adapter import ReliabilityCausalService +from topology_service import default_config + +class FixedProvider: + def __init__(self,attempts): + self.states={'hbf0':{}};self.attempts=attempts + self.profile={'ecc_decoder_headroom_over_fresh_media':2} + def cost(self,stack,now):return {'attempt_work_milli':self.attempts,'expected_retry_steps':self.attempts/1000-1} + +def run(attempts,topology='mixed_direct'): + c=default_config(topology);p=FixedProvider(attempts);s=ReliabilityCausalService(c,p) + s.begin_window(0,20_000_000,{k:10**12 for k in c['channels']},{k:'normal' for k in c['channels']}) + s.submit_jobs([{'job_id':'read1','stack':'hbf0','channel':'0','route':'relay' if topology=='relay' else 'direct', + 'operation':'read','bytes':96_000_000,'arrival_ns':0}]) + return s,s.advance_to(20_000_000) + +class AdapterTests(unittest.TestCase): + def test_retry_consumes_media_and_delays_unique_completion(self): + _,base=run(1000);_,retry=run(2000) + self.assertEqual(retry['completion_ids'],['read1']) + self.assertGreater(retry['job_progress'][0]['completion_ns'],base['job_progress'][0]['completion_ns']) + self.assertEqual(sum(x['bytes'] for x in retry['activities'] if x['phase']=='media_read'),192_000_000) + self.assertEqual(sum(x['bytes'] for x in retry['activities'] if x['phase']=='direct_gpu_link'),96_000_000) + def test_relay_success_payload_is_not_retransmitted_for_internal_retry(self): + _,receipt=run(2500,'relay') + self.assertEqual(sum(x['bytes'] for x in receipt['activities'] if x['phase']=='media_read'),240_000_000) + self.assertEqual(sum(x['bytes'] for x in receipt['activities'] if x['phase']=='relay_send'),96_000_000) + self.assertEqual(sum(x['bytes'] for x in receipt['activities'] if x['phase']=='partner_gpu_drain'),96_000_000) + def test_null_effort_matches_unmodified_service(self): + wrapped,receipt=run(1000);c=default_config('mixed_direct');s=CausalTopologyService(c) + s.begin_window(0,20_000_000,{k:10**12 for k in c['channels']},{k:'normal' for k in c['channels']}) + s.submit_jobs([{'job_id':'read1','stack':'hbf0','channel':'0','route':'direct','operation':'read','bytes':96_000_000,'arrival_ns':0}]) + original=s.advance_to(20_000_000) + self.assertEqual(receipt['job_progress'][0]['completion_ns'],original['job_progress'][0]['completion_ns']) + self.assertEqual(receipt['completion_ids'],original['completion_ids']) + def test_active_cost_frozen_when_temperature_feedback_changes_provider(self): + c=default_config('mixed_direct');p=FixedProvider(2000);s=ReliabilityCausalService(c,p) + s.begin_window(0,20_000_000,{k:10**12 for k in c['channels']},{k:'normal' for k in c['channels']}) + s.submit_jobs([{'job_id':'r','stack':'hbf0','channel':'0','route':'direct','operation':'read','bytes':96_000_000,'arrival_ns':0}]) + s.advance_to(500_000);p.attempts=1000;r=s.advance_to(20_000_000) + self.assertEqual(r['job_progress'][0]['metadata']['reliability_cost_proxy']['attempt_work_milli'],2000) +if __name__=='__main__':unittest.main() diff --git a/experiments/eq3_system_thermal/test_endpoint_policy.py b/experiments/eq3_system_thermal/test_endpoint_policy.py new file mode 100644 index 0000000..63e5d58 --- /dev/null +++ b/experiments/eq3_system_thermal/test_endpoint_policy.py @@ -0,0 +1,74 @@ +import os +from pathlib import Path +import subprocess +import sys +import unittest + +from endpoint_policy import EndpointAwarePolicy +from read_rate_policy import EngineeringProfile, ReadRatePolicy, StackWindowFacts, WindowFacts + + +WINDOW = 20_000_000 +BASELINE = 1000 + + +def profile(stack): + return EngineeringProfile( + profile_id="fixed:" + stack, enabled=True, + strategy="read_rate_feedback_thermal_guard_v1", window_ns=WINDOW, + target_bytes_per_s=50_000, step_bytes=50, minimum_budget_bytes=100, + maximum_budget_bytes=BASELINE, severe_budget_bytes=0, light_fraction=.5) + + +def facts(stack, state, budget, *, offered=0, delivered=0, backlog=0): + row = StackWindowFacts( + stack_id=stack, offered_bytes=offered, delivered_bytes=delivered, + backlog_bytes=backlog, oldest_wait_ns=0, latency_p95_ns=None, + censored_requests=0, gate_limited=False, + backend_busy_fraction=0, resource_busy=False) + return WindowFacts(0, WINDOW, state, (row,), {stack: budget}, + guard_states={stack: state}, + hysteresis_budget_bytes={stack: BASELINE // 2}) + + +class EndpointPolicyTests(unittest.TestCase): + def test_isolated_entrypoint_does_not_require_pythonpath(self): + entry = Path(__file__).with_name("run_endpoint_guard_point.py") + environment = dict(os.environ) + environment.pop("PYTHONPATH", None) + result = subprocess.run( + [sys.executable, str(entry), "--help"], cwd="/tmp", env=environment, + text=True, capture_output=True, check=False) + self.assertEqual(result.returncode, 0, result.stderr) + self.assertIn("--thermal-binary", result.stdout) + + def test_reproduces_legacy_hbm_half_cap_hold_without_local_demand(self): + policy = ReadRatePolicy(profile("hbm0")) + decision = policy.evaluate(facts("hbm0", "normal", BASELINE // 2)) + self.assertEqual(decision.stack_decisions[0].budget_bytes, BASELINE // 2) + self.assertEqual(decision.stack_decisions[0].outcome, "INSUFFICIENT_DEMAND") + + def test_hbm_endpoint_light_caps_and_normal_restores_without_fake_demand(self): + policy = EndpointAwarePolicy(profile("hbm0")) + light = policy.evaluate(facts("hbm0", "light", BASELINE)) + self.assertEqual(light.stack_decisions[0].budget_bytes, BASELINE // 2) + normal = policy.evaluate(facts("hbm0", "normal", BASELINE // 2)) + self.assertEqual(normal.stack_decisions[0].budget_bytes, BASELINE) + self.assertEqual(normal.stack_decisions[0].reasons, + ("THERMAL_NORMAL_RESTORE_BASELINE",)) + self.assertIn("no HBM foreground delivery", normal.fact_semantics) + + def test_hbm_severe_shutdown_and_hbf_legacy_equivalence(self): + hbm = EndpointAwarePolicy(profile("hbm0")) + for state in ("severe", "shutdown"): + self.assertEqual(hbm.evaluate(facts("hbm0", state, BASELINE)). + stack_decisions[0].budget_bytes, 0) + left = EndpointAwarePolicy(profile("hbf0")) + right = ReadRatePolicy(profile("hbf0")) + observed = facts("hbf0", "normal", BASELINE, offered=2000, + delivered=900, backlog=1100) + self.assertEqual(left.evaluate(observed), right.evaluate(observed)) + + +if __name__ == "__main__": + unittest.main() diff --git a/experiments/eq3_system_thermal/test_energy.py b/experiments/eq3_system_thermal/test_energy.py new file mode 100644 index 0000000..eb10f44 --- /dev/null +++ b/experiments/eq3_system_thermal/test_energy.py @@ -0,0 +1,39 @@ +import unittest + +from energy import EnergyMapper, engineering_energy_profile + + +class EnergyTests(unittest.TestCase): + def mapper(self): + rows = [{"id": name} for name in ("hbf0.die0", "hbf0.base", "hbm0.base", "gpu")] + rows.append({"id": "hbm0.die0", "physical_type": "HBM4", "role": "array_die", "device_id": "hbm0"}) + return EnergyMapper({"components": rows}, {"hbf0": {"0": "hbf0.die0"}}, engineering_energy_profile()) + + def test_read_relay_disjoint_and_no_hbm_array(self): + result = self.mapper().map([ + dict(operation="read", stack="hbf0", channel="0", bytes=4096), + dict(operation="relay_receive", stack="hbm0", bytes=4096), + dict(operation="relay_send", stack="hbm0", bytes=4096), + ]) + self.assertAlmostEqual(result["total_j"], 4096 * 54e-12, places=18) + self.assertNotIn("hbm0.die0", result["component_energy_j"]) + + def test_retry_has_heat_without_effective_delivery_requirement(self): + result = self.mapper().map([dict(operation="retry", stack="hbf0", channel="0", bytes=4096)]) + self.assertAlmostEqual(result["total_j"], 4096 * 50e-12, places=18) + self.assertEqual(self.mapper().map([])["total_j"], 0) + + def test_actual_hbm4_geometry_label(self): + result = self.mapper().map([dict(operation="hbm_read", stack="hbm0", bytes=4096)]) + self.assertAlmostEqual(result["total_j"], 4096 * 42e-12, places=18) + + def test_unknown_erase_and_duplicate_facts_fail(self): + with self.assertRaisesRegex(ValueError, "UNKNOWN"): + self.mapper().map([dict(operation="erase", stack="hbf0", channel="0", bytes=0, operations=1)]) + row = dict(operation="read", stack="hbf0", channel="0", bytes=4096, activity_id="one") + with self.assertRaisesRegex(ValueError, "duplicate"): + self.mapper().map([row, row]) + + +if __name__ == "__main__": + unittest.main() diff --git a/experiments/eq3_system_thermal/test_maintenance_driver.py b/experiments/eq3_system_thermal/test_maintenance_driver.py new file mode 100644 index 0000000..c7bc5d0 --- /dev/null +++ b/experiments/eq3_system_thermal/test_maintenance_driver.py @@ -0,0 +1,151 @@ +import unittest + +from maintenance_driver import MaintenanceDriver +from reliability import DAY_NS, ReliabilityLedger +from topology_service import TopologyService, default_config + + +W = 20_000_000 + + +def driver_config(spares): + return { + "block_bytes": 4096, + "pages_per_block": 256, + "max_blocks_per_cohort": 8, + "spare_block_ids_by_stack_channel": {"hbf0": {"0": spares}}, + "program_energy_j_per_byte": 0.05 * 100e-6 / 4096, + "erase_energy_j_per_operation": 0.05 * 1e-3, + "energy_evidence": { + "program": "DERIVED_ENGINEERING_PROXY_0.05W_100US_NOT_HBF_CALIBRATION", + "erase": "DERIVED_ENGINEERING_PROXY_0.05W_NATIVE_ADAPTER_1MS_NOT_HBF_CALIBRATION", + }, + } + + +def budgets(config): + return {stack: 10**15 for stack in config["channels"]} + + +def states(config): + return {stack: "normal" for stack in config["channels"]} + + +class Harness: + def __init__(self, extent_count=1, spare_count=2): + self.config = default_config("mixed_direct") + self.service = TopologyService(self.config) + self.ledger = ReliabilityLedger({"ea_ev": 1.04, + "refresh_trigger": "equivalent_age_or_wall", + "initial_equivalent_age_ns": DAY_NS}) + self.driver = MaintenanceDriver( + self.ledger, driver_config([f"spare{i}" for i in range(spare_count)])) + self.extents = [f"extent{i}" for i in range(extent_count)] + self.driver.register_extents([ + {"extent_id": extent, "stack": "hbf0", "channel": "0", + "source_block_id": f"source{i}", "version": 0} + for i, extent in enumerate(self.extents) + ]) + self.now = 0 + + def window(self, jobs, failed=()): + start, end = self.now, self.now + W + receipt = self.service.advance(start, end, {}, budgets(self.config), + states(self.config), jobs) + for extent in self.extents: + self.ledger.advance_temperature(extent, start, end, 358.15) + delta = self.driver.consume_receipt(receipt, failed_job_ids=failed) + self.now = end + return receipt, delta + + +class MaintenanceDriverTests(unittest.TestCase): + def test_equal_near_due_initial_age_reaches_policy_without_time_compression(self): + ledger = ReliabilityLedger({"ea_ev": 1.04, + "refresh_trigger": "equivalent_age_or_wall", + "initial_equivalent_age_ns": DAY_NS - W}) + driver = MaintenanceDriver(ledger, driver_config(["spare0", "spare1"])) + driver.register_extents([ + {"extent_id": f"extent{i}", "stack": "hbf0", "channel": "0", + "source_block_id": f"source{i}", "version": 0} + for i in range(2) + ]) + self.assertEqual(driver.poll(0), []) + for extent in ("extent0", "extent1"): + ledger.advance_temperature(extent, 0, W, 358.15) + jobs = driver.poll(W) + self.assertEqual(jobs[0]["metadata"]["extent_ids"], ["extent0", "extent1"]) + self.assertEqual(jobs[0]["metadata"]["block_count"], 2) + + def test_actual_service_read_program_commit_erase_and_energy(self): + h = Harness() + read = h.driver.poll(0) + self.assertEqual(read[0]["operation"], "refresh_read") + h.window(read) + program = h.driver.poll(h.now) + self.assertEqual(program[0]["operation"], "program") + _, program_delta = h.window(program) + self.assertEqual(program_delta["summary"]["outstanding_job_count"], 0) + self.assertTrue(program_delta["energy_facts"]) + # Age resets at program/version commit, before old-source erase. + self.assertEqual(h.ledger.snapshot()["blocks"]["extent0"]["equivalent_age_ns"], 0) + erase = h.driver.poll(h.now) + self.assertEqual(erase[0]["operation"], "erase") + self.assertEqual(erase[0]["metadata"]["physical_block_ids"], ["source0"]) + _, erase_delta = h.window(erase) + self.assertEqual(erase_delta["summary"]["committed_extent_count"], 1) + self.assertEqual(h.driver.drain_delta()["events"], []) + + result = h.driver.snapshot() + self.assertEqual(result["terminal_summary"]["status_counts"], {"COMMITTED": 1}) + self.assertEqual(result["extents"]["extent0"]["physical_block_id"], "spare0") + self.assertEqual(result["physical_wear"]["spare0"]["block_program_work_completed"], 1) + self.assertEqual(result["physical_wear"]["spare0"]["nand_page_programs_completed"], 256) + self.assertEqual(result["physical_wear"]["source0"]["erase_completed"], 1) + self.assertIn("source0", result["free_spares_by_stack_channel"]["hbf0"]["0"]) + energies = {row["phase"]: row["energy_j"] + for row in program_delta["energy_facts"] + erase_delta["energy_facts"]} + self.assertAlmostEqual(energies["program"], 5e-6) + self.assertAlmostEqual(energies["erase_old"], 50e-6) + + def test_version_conflict_keeps_age_and_reclaims_destination(self): + h = Harness() + h.window(h.driver.poll(0)) + h.driver.record_foreground_write("extent0", h.now) + h.window(h.driver.poll(h.now)) + age_after_program = h.ledger.snapshot()["blocks"]["extent0"]["equivalent_age_ns"] + self.assertGreater(age_after_program, DAY_NS) + cleanup = h.driver.poll(h.now) + self.assertEqual(cleanup[0]["metadata"]["phase"], "erase_cleanup") + self.assertEqual(cleanup[0]["metadata"]["physical_block_ids"], ["spare0"]) + h.window(cleanup) + result = h.driver.snapshot() + self.assertEqual(result["terminal_summary"]["status_counts"], + {"FAILED_OR_VERSION_CONFLICT": 1}) + self.assertEqual(result["extents"]["extent0"]["physical_block_id"], "source0") + self.assertIn("spare0", result["free_spares_by_stack_channel"]["hbf0"]["0"]) + + def test_program_failure_retains_age_and_spare_is_bounded(self): + h = Harness(extent_count=2, spare_count=1) + read = h.driver.poll(0) + self.assertEqual(read[0]["metadata"]["block_count"], 1) + self.assertEqual(len(h.driver.active_extents), 1) + h.window(read) + program = h.driver.poll(h.now) + h.window(program, failed={program[0]["job_id"]}) + self.assertGreater(h.ledger.snapshot()["blocks"]["extent0"]["equivalent_age_ns"], DAY_NS) + cleanup = h.driver.poll(h.now) + h.window(cleanup) + result = h.driver.snapshot() + self.assertEqual(result["terminal_summary"]["status_counts"], + {"FAILED_OR_VERSION_CONFLICT": 1}) + self.assertEqual(result["physical_wear"]["spare0"]["nand_page_programs_started"], 256) + self.assertEqual(result["physical_wear"]["spare0"]["nand_page_programs_completed"], 0) + self.assertIn("spare0", result["free_spares_by_stack_channel"]["hbf0"]["0"]) + # The second due extent can now claim the single recovered spare. + next_read = h.driver.poll(h.now) + self.assertEqual(next_read[0]["metadata"]["extent_ids"], ["extent1"]) + + +if __name__ == "__main__": + unittest.main() diff --git a/experiments/eq3_system_thermal/test_model_variants.py b/experiments/eq3_system_thermal/test_model_variants.py new file mode 100755 index 0000000..5a4677e --- /dev/null +++ b/experiments/eq3_system_thermal/test_model_variants.py @@ -0,0 +1,144 @@ +#!/usr/bin/env python3 +import json +import math +import tempfile +import unittest +from pathlib import Path + +from model_variants import build_variant + + +def make_source(root: Path, reverse: bool = False) -> Path: + root.mkdir() + axes = [[0.0, 1.0, 2.0], [0.0, 1.0, 2.0], [0.0, 0.5, 1.0]] + cells = [] + for z in range(2): + for y in range(2): + for x in range(2): + i = z * 4 + y * 2 + x + cells.append({"id": f"n{z}_{y}_{x}", "index": i, + "xyz_index": [x, y, z], "component": "package", + "center_m": [x + .5, y + .5, z * .5 + .25], + "size_m": [1.0, 1.0, .5], + "k_xyz_w_m_k": [2.0, 2.0, 2.0]}) + grid = {"axes_m": axes, "shape": [2, 2, 2], "cells": cells, + "component_cells": {"package": list(range(8))}} + normalized = { + "boundaries": {"initial_temperature_k": 300.0, + "top": {"ambient_k": 300.0, "h_w_m2_k": 4.0}, + "bottom": {"ambient_k": 300.0, "h_w_m2_k": 2.0}}, + "devices": [ + {"id": "gpu", "physical_type": "GPU", "external": False, + "xy_m": [0.0, 0.0], "footprint_m": [1.0, 2.0]}, + {"id": "hbf0", "physical_type": "HBF", "external": False, + "xy_m": [1.0, 0.0], "footprint_m": [1.0, 2.0]}, + ], + } + # At scale=1: half-cell R is .125 K/W; convection R is .5 bottom/.25 top. + bottom_g, top_g = 1 / .625, 1 / .375 + nodes = [] + for c in cells: + g = bottom_g if c["xyz_index"][2] == 0 else top_g + nodes.append(f"node {c['id']} other package package -1 3 300 0 {g:.17g} 300") + edges = [] + for c in cells: + x, y, z = c["xyz_index"] + for dx, dy, dz in ((1, 0, 0), (0, 1, 0), (0, 0, 1)): + q = (x + dx, y + dy, z + dz) + if all(q[i] < (2, 2, 2)[i] for i in range(3)): + edges.append(f"edge {c['id']} n{q[2]}_{q[1]}_{q[0]} 2 component") + if reverse: + nodes.reverse(); edges.reverse() + (root / "model.txt").write_text("HBFSIM_EQ3_THERMAL_MODEL 1\ncoupling on\n" + + "\n".join(nodes + edges) + "\n") + (root / "normalized.json").write_text(json.dumps(normalized)) + (root / "rc_grid.json").write_text(json.dumps(grid)) + (root / "rc_sensors.json").write_text(json.dumps({"sensors": []})) + return root + + +def parse_model(path: Path): + nodes, edges = {}, [] + for line in path.read_text().splitlines(): + f = line.split() + if f and f[0] == "node": + nodes[f[1]] = {"capacity": float(f[6]), "initial": float(f[7]), + "static": float(f[8]), "g": float(f[9]), + "boundary": float(f[10])} + elif f and f[0] == "edge": + edges.append((f[1], f[2], float(f[3]))) + return nodes, edges + + +class ModelVariantTest(unittest.TestCase): + def test_ambient_and_external_resistance_formula_preserve_c(self): + with tempfile.TemporaryDirectory() as td: + source = make_source(Path(td) / "source") + output = Path(td) / "derived" + manifest = build_variant(source, output, ambient_k=310.0, + external_resistance_scale=1.5, + coupling_mode="full") + nodes, edges = parse_model(output / "model.txt") + # half cell=.125; scaled top external=.375, bottom=.75 K/W. + self.assertAlmostEqual(nodes["n1_0_0"]["g"], 2.0) + self.assertAlmostEqual(nodes["n0_0_0"]["g"], 1 / .875) + self.assertTrue(all(n["initial"] == 310 and n["boundary"] == 310 + and n["capacity"] == 3 for n in nodes.values())) + self.assertEqual(len(edges), 12) + self.assertEqual(manifest["capacity_j_k_unchanged"], 24) + normalized = json.loads((output / "normalized.json").read_text()) + self.assertEqual(normalized["boundaries"]["initial_temperature_k"], 310) + self.assertEqual(normalized["boundaries"]["top"]["ambient_k"], 310) + self.assertEqual(normalized["boundaries"]["bottom"]["ambient_k"], 310) + + def test_domain_cut_preserves_vertical_and_reconstructs_conservative_laplacian(self): + with tempfile.TemporaryDirectory() as td: + source = make_source(Path(td) / "source") + output = Path(td) / "derived" + manifest = build_variant(source, output, ambient_k=300.0, + external_resistance_scale=1.0, + coupling_mode="no_cross_domain_lateral") + _nodes, edges = parse_model(output / "model.txt") + self.assertEqual(manifest["removed_edge_count"], 4) + self.assertEqual(manifest["edge_axis_counts"]["removed_x"], 4) + self.assertEqual(manifest["edge_axis_counts"]["kept_z"], 4) + self.assertEqual(len(edges), 8) + self.assertEqual(manifest["matrix_audit"]["internal_edge_energy_conservation"], "PASS") + self.assertLess(manifest["matrix_audit"]["max_reconstructed_row_balance_residual_w_k"], 1e-12) + # Independently assemble the internal Laplacian and check every row sum. + ids = sorted({p for e in edges for p in e[:2]}) + at = {node_id: i for i, node_id in enumerate(ids)} + lap = [[0.0] * len(ids) for _ in ids] + for a, b, g in edges: + ia, ib = at[a], at[b] + lap[ia][ia] += g; lap[ib][ib] += g + lap[ia][ib] -= g; lap[ib][ia] -= g + self.assertTrue(all(math.isclose(sum(row), 0, abs_tol=1e-12) for row in lap)) + + def test_node_and_edge_reordering_has_identical_derived_model(self): + with tempfile.TemporaryDirectory() as td: + root = Path(td) + source_a = make_source(root / "a") + source_b = make_source(root / "b", reverse=True) + build_variant(source_a, root / "out_a", ambient_k=320.0, + external_resistance_scale=.5, coupling_mode="no_cross_domain_lateral") + build_variant(source_b, root / "out_b", ambient_k=320.0, + external_resistance_scale=.5, coupling_mode="no_cross_domain_lateral") + self.assertEqual((root / "out_a/model.txt").read_bytes(), + (root / "out_b/model.txt").read_bytes()) + + def test_rejects_nonregistered_variant_and_existing_output(self): + with tempfile.TemporaryDirectory() as td: + root = Path(td) + source = make_source(root / "source") + with self.assertRaises(ValueError): + build_variant(source, root / "bad", ambient_k=305.0, + external_resistance_scale=1.0, coupling_mode="full") + (root / "exists").mkdir() + with self.assertRaises(FileExistsError): + build_variant(source, root / "exists", ambient_k=300.0, + external_resistance_scale=1.0, coupling_mode="full") + + +if __name__ == "__main__": + unittest.main() diff --git a/experiments/eq3_system_thermal/test_prepare_maintenance_main.py b/experiments/eq3_system_thermal/test_prepare_maintenance_main.py new file mode 100644 index 0000000..c765414 --- /dev/null +++ b/experiments/eq3_system_thermal/test_prepare_maintenance_main.py @@ -0,0 +1,115 @@ +#!/usr/bin/env python3 +import hashlib +import json +from pathlib import Path +import tempfile +import unittest + +from maintenance_inputs import add_maintenance +from prepare_maintenance_main import prepare, STRATEGIES, TOPOLOGIES +from prepare_stage import config as base_config + + +def save(path, value): + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(json.dumps(value, indent=2, sort_keys=True, + allow_nan=False) + "\n") + + +def digest(path): + return hashlib.sha256(path.read_bytes()).hexdigest() + + +class MaintenanceMainPreparationTests(unittest.TestCase): + def setUp(self): + self.temp = tempfile.TemporaryDirectory() + self.root = Path(self.temp.name) + self.pilot_index = self.root / "PILOT_INDEX.json" + self.analysis_root = self.root / "analysis" + points = [] + for topology in TOPOLOGIES: + point_id = f"pilot-{topology}" + config = add_maintenance(base_config( + point_id, topology, 1_536_000_000_000, STRATEGIES[2], 8, 4), "shared") + config["point_id"] = point_id + config_path = self.root / "pilot-inputs" / f"{point_id}.json" + save(config_path, config) + output = self.root / "pilot-points" / point_id + save(output / "DONE.json", {"status": "COMPLETED"}) + analysis = { + "analysis_status": "VALIDATED_COMPLETE_RECEIPTS", "runner": "maintenance", + "identity": {"point_id": point_id}, + "checks": {"timeline": "PASS", "byte_conservation": "PASS", + "energy_to_thermal": "PASS", + "terminal_uniqueness": "EXACT_JOB_IDS"}, + "maintenance": {"terminal_operation_count": 4}, + } + save(self.analysis_root / point_id / "analysis.json", analysis) + save(self.analysis_root / point_id / "DONE.json", {"status": "COMPLETED"}) + points.append({ + "point_id": point_id, "topology": topology, + "config": str(config_path), "config_sha256": digest(config_path), + "output": str(output), "model_dir": f"model-{topology}", + "thermal_binary": "thermal-service", "artifact_root": "artifacts", + }) + save(self.pilot_index, { + "points": points, "runner": "run_maintenance_point.py", + "runtime_source_locks_sha256": {"runner": "a" * 64}, + "model_locks_sha256": {"model": {"normalized.json": "b" * 64}}, + "thermal_binary_sha256": "c" * 64, + }) + analysis_hashes = {row["point_id"]: digest( + self.analysis_root / row["point_id"] / "analysis.json") for row in points} + self.review = self.root / "PILOT_REVIEW.json" + save(self.review, { + "status": "APPROVED_FOR_MAIN_INPUT_GENERATION", + "pilot_index_sha256": digest(self.pilot_index), + "analysis_sha256": analysis_hashes, + "measured_resource_budget": {"point_wall_s": 600, "stage_wall_s": 7200, + "point_output_gib": 2, "stage_output_gib": 40, "address_space_gib": 8, + "cpu_threads": 1, "host_ram_reserve_gib": 32, + "host_disk_reserve_gib": 100}, + }) + + def tearDown(self): + self.temp.cleanup() + + def test_generates_exact_reviewed_19_point_candidate(self): + destination = self.root / "maintenance-main-v1" + result = prepare(self.pilot_index, self.analysis_root, self.review, destination) + self.assertEqual(result["point_count"], 19) + points = [json.loads(Path(row["config"]).read_text()) for row in result["points"]] + self.assertEqual(sum(row["maintenance"]["mode"] == "shared" for row in points), 16) + self.assertEqual(sum(row["maintenance"]["mode"] == "ideal_independent" + for row in points), 3) + self.assertEqual({row["maintenance"]["ea_ev"] for row in points}, {1.01, 1.04, 1.08}) + self.assertEqual(sum(row["maintenance"]["ea_ev"] != 1.04 for row in points), 4) + self.assertTrue(all(row["workload"]["active_ns"] == 20_000_000_000 + and row["recovery_ns"] == 10_000_000_000 for row in points)) + self.assertTrue(all(row["maintenance"]["aged_blocks_per_stack"] * + row["maintenance"]["block_bytes"] == 4 * 1024**3 + for row in points)) + self.assertFalse((destination / "RUN_INDEX.json").exists()) + self.assertEqual(json.loads((destination / "CANDIDATE_INDEX.json").read_text())["status"], + "PILOT_REVIEW_PASSED_PREPARED_NOT_FROZEN_NOT_LAUNCHABLE") + + def test_requires_bound_review_and_all_completed_pilots(self): + review = json.loads(self.review.read_text()) + review["analysis_sha256"]["pilot-dash"] = "0" * 64 + save(self.review, review) + with self.assertRaisesRegex(ValueError, "not bound"): + prepare(self.pilot_index, self.analysis_root, self.review, + self.root / "maintenance-main-v1") + + def test_refuses_pilot_stage_or_existing_destination(self): + with self.assertRaisesRegex(ValueError, "maintenance-main-v1"): + prepare(self.pilot_index, self.analysis_root, self.review, + self.root / "maintenance-v1") + destination = self.root / "maintenance-main-v1" + destination.mkdir() + with self.assertRaises(FileExistsError): + prepare(self.pilot_index, self.analysis_root, self.review, destination) + + +if __name__ == "__main__": + unittest.main() diff --git a/experiments/eq3_system_thermal/test_prepare_rate_diagnostics.py b/experiments/eq3_system_thermal/test_prepare_rate_diagnostics.py new file mode 100644 index 0000000..e57b21e --- /dev/null +++ b/experiments/eq3_system_thermal/test_prepare_rate_diagnostics.py @@ -0,0 +1,78 @@ +import json +from pathlib import Path +import tempfile +import unittest + +from prepare_rate_diagnostics import RATE, STRATEGIES, TOPOLOGIES, digest, prepare +from prepare_stage import config as base_config + + +def model(root, name): + path = root / name + path.mkdir() + for filename in ("model.txt", "normalized.json", "rc_grid.json", "rc_sensors.json"): + (path / filename).write_text(filename + "\n") + return path + + +class RateDiagnosticPreparationTests(unittest.TestCase): + def test_nine_points_and_frozen_existing_base_comparisons(self): + with tempfile.TemporaryDirectory() as temporary: + root = Path(temporary) + mixed = model(root, "mixed") + all_hbf = model(root, "all-hbf") + no_cross = model(root, "no-cross") + model_manifest = root / "models.json" + model_manifest.write_text(json.dumps({ + "source_models": { + "mixed_full_2mm": {"model_dir": str(mixed)}, + "all_hbf_full_2mm": {"model_dir": str(all_hbf)}, + }, + "derived_models": [{"ambient_k": 300, "external_resistance_scale": 1, + "coupling_mode": "no_cross_domain_lateral", "model_dir": str(no_cross)}], + }) + "\n") + base_inputs = root / "base-inputs"; base_inputs.mkdir() + base_points = [] + identities = [(topology, RATE, strategy) for topology in TOPOLOGIES + for strategy in STRATEGIES] + identities += [("all_hbf_direct", 768_000_000_000, strategy) + for strategy in STRATEGIES] + for index, (topology, rate, strategy) in enumerate(identities): + value = base_config(f"base-{index}", topology, rate, strategy, 20, 10) + path = base_inputs / f"{index}.json" + path.write_text(json.dumps(value) + "\n") + base_points.append({"point_id": value["point_id"], "kind": "base", + "config": str(path), "config_sha256": digest(path), + "output": str(root / "base-points" / value["point_id"])}) + base_index = root / "RUN_INDEX.json" + base_index.write_text(json.dumps({"points": base_points, + "source_locks": {"fixture": "locked"}}) + "\n") + artifacts = root / "artifacts"; artifacts.mkdir() + binary = artifacts / "thermal"; binary.write_text("fixture\n") + output = root / "prepared" + result = prepare(output, model_manifest, base_index, binary, artifacts) + + self.assertEqual(result["point_count"], 9) + self.assertEqual(result["schema_version"], "eq3-system-rate-diagnostics-index-v2") + self.assertTrue(result["runner"].endswith("run_endpoint_guard_point.py")) + self.assertEqual(result["resources"]["new_output_gib"], 4.5) + self.assertEqual(result["resources"]["parent_accounting_gib"] + ["remaining_for_maintenance_causal"], 18.5) + self.assertEqual(len(result["comparisons"]["nonuniform_vs_uniform"]), 8) + self.assertEqual(len(result["comparisons"]["no_coupling_vs_full"]), 1) + self.assertEqual(len(result["comparisons"]["same_total_offered_existing_base"]), 3) + self.assertEqual(len(result["comparisons"]["same_per_stack_existing_base"]), 3) + for row in result["points"]: + value = json.loads(Path(row["config"]).read_text()) + self.assertEqual(value["strategy"], "read_rate_feedback_thermal_guard_v1") + self.assertEqual(value["workload"]["active_ns"], 20_000_000_000) + self.assertEqual(value["recovery_ns"], 10_000_000_000) + self.assertEqual(value["workload"]["per_stack_Bps"], RATE) + hot = [json.loads(Path(row["config"]).read_text()) for row in result["points"] + if row["diagnostic"].startswith("HOT_STACK")] + self.assertEqual(len(hot), 4) + self.assertTrue(all(row["workload"]["hot_stack"] == "hbf0" for row in hot)) + + +if __name__ == "__main__": + unittest.main() diff --git a/experiments/eq3_system_thermal/test_prepare_sensitivity.py b/experiments/eq3_system_thermal/test_prepare_sensitivity.py new file mode 100644 index 0000000..1c484ba --- /dev/null +++ b/experiments/eq3_system_thermal/test_prepare_sensitivity.py @@ -0,0 +1,80 @@ +import json +from pathlib import Path +import tempfile +import unittest + +from prepare_sensitivity import prepare +from launch_sensitivity import domain_failure + + +def model(root: Path, name: str) -> Path: + path = root / name + path.mkdir() + for filename in ("model.txt", "normalized.json", "rc_grid.json", "rc_sensors.json"): + (path / filename).write_text(filename + "\n") + return path + + +class SensitivityPreparationTest(unittest.TestCase): + def test_frozen_counts_consumers_and_shutdown_envelope(self): + with tempfile.TemporaryDirectory() as td: + root = Path(td) + mixed = model(root, "mixed") + variants = [] + for name, ambient, resistance in ( + ("a310", 310, 1.0), ("a320", 320, 1.0), + ("r05", 300, .5), ("r15", 300, 1.5), ("a320r15", 320, 1.5)): + path = model(root, name) + variants.append({"ambient_k": ambient, "external_resistance_scale": resistance, + "coupling_mode": "full", "model_dir": str(path)}) + manifest = root / "models.json" + manifest.write_text(json.dumps({"source_models": { + "mixed_full_2mm": {"model_dir": str(mixed)}}, + "derived_models": variants}) + "\n") + artifacts = root / "artifacts"; artifacts.mkdir() + binary = artifacts / "thermal"; binary.write_text("fixture\n") + output = root / "prepared" + index = prepare(output, manifest, binary, artifacts) + self.assertEqual(index["point_count"], 54) + self.assertEqual(index["schema_version"], "eq3-system-sensitivity-index-v2") + self.assertTrue(index["runner"].endswith("run_endpoint_guard_point.py")) + self.assertEqual(index["status"], "PENDING_DEPENDENCIES_BASE_MATRIX") + self.assertEqual(index["resources"]["sensitivity_output_gib"], 27) + self.assertEqual(index["resources"]["parent_combined_output_gib"], 80) + self.assertEqual(len({row["point_id"] for row in index["points"]}), 54) + self.assertEqual(sum(row["topology"] == "mixed_direct" for row in index["points"]), 48) + self.assertEqual(sum(row["topology"] == "relay" for row in index["points"]), 6) + self.assertEqual(sum(row["execution_mode"] == "uncontrolled_first_constraint" + for row in index["points"]), 18) + relay = [row for row in index["points"] if row["topology"] == "relay"] + self.assertTrue(all(row["mechanism_scope"] == + "RELAY_HBM_ENDPOINT_ACTUAL_CONSUMER_EXTENSION" for row in relay)) + for row in index["points"]: + config = json.loads(Path(row["config"]).read_text()) + self.assertEqual(config["thermal_limits_k"]["hbf"][2], 378.15) + self.assertEqual(config["thermal_limits_k"]["hbm"][2], 378.15) + self.assertEqual(config["thermal_limits_k"]["gpu"], [363.15, 373.15, 383.15]) + self.assertEqual(config["sensitivity"]["domain_max_k_unchanged"], 400) + if row["execution_mode"] == "uncontrolled_first_constraint": + self.assertTrue(config["control_disabled"]) + deferred = json.loads((output / "DEFERRED_EA.json").read_text()) + self.assertFalse(deferred["runnable"]) + self.assertEqual(deferred["point_count_when_ready"], 6) + self.assertEqual(deferred["ea_ev"], [1.01, 1.04, 1.08]) + + def test_launcher_classifies_only_explicit_thermal_domain_failure(self): + with tempfile.TemporaryDirectory() as td: + output = Path(td) + process = output / "thermal-process"; process.mkdir() + save = lambda path, value: path.write_text(json.dumps(value) + "\n") + save(output / "FAILED.json", {"status": "FAILED"}) + transcript = process / "thermal-transcript.jsonl" + save(transcript, {"response": {"type": "ERROR", "status": "NUMERICAL_FAILURE"}}) + self.assertFalse(domain_failure(output)) + transcript.write_text(json.dumps({"response": { + "type": "ERROR", "status": "DOMAIN_FAILURE"}}) + "\n") + self.assertTrue(domain_failure(output)) + + +if __name__ == "__main__": + unittest.main() diff --git a/experiments/eq3_system_thermal/test_rate_workload.py b/experiments/eq3_system_thermal/test_rate_workload.py new file mode 100644 index 0000000..4c11ae8 --- /dev/null +++ b/experiments/eq3_system_thermal/test_rate_workload.py @@ -0,0 +1,32 @@ +import unittest +from rate_workload import RateWorkload + + +class RateWorkloadTests(unittest.TestCase): + channels = {f"hbf{s}": {str(c): f"hbf{s}.die{c}" for c in range(16)} for s in range(4)} + + def total(self, pattern="continuous", **extra): + source = RateWorkload(self.channels, dict(active_ns=1_000_000_000, + per_stack_Bps=1_920_000_000_000, pattern=pattern, **extra)) + by_stack = {s: 0 for s in self.channels} + for start in range(0, 1_200_000_000, 20_000_000): + rows = source.advance(start, start + 20_000_000) + if start >= 1_000_000_000: + self.assertEqual(sum(map(sum, (r.values() for r in rows.values()))), 0) + for stack, values in rows.items(): + by_stack[stack] += sum(values.values()) + return source.total, by_stack + + def test_burst_preserves_mean_and_over_capacity_demand(self): + self.assertEqual(self.total()[0], 4 * 1_920_000_000_000) + self.assertEqual(self.total()[0], self.total("burst_equal_mean")[0]) + + def test_concentration_preserves_package_demand(self): + total, rows = self.total(hot_stack="hbf0") + self.assertEqual(total, self.total()[0]) + self.assertEqual(rows["hbf0"], total // 2) + self.assertEqual(total, self.total(channel_distribution="first_quarter")[0]) + + +if __name__ == "__main__": + unittest.main() diff --git a/experiments/eq3_system_thermal/test_run_causal_point.py b/experiments/eq3_system_thermal/test_run_causal_point.py new file mode 100644 index 0000000..cece1eb --- /dev/null +++ b/experiments/eq3_system_thermal/test_run_causal_point.py @@ -0,0 +1,235 @@ +#!/usr/bin/env python3 +import io +import json +import unittest +from copy import deepcopy + +from causal_service import CausalTopologyService +from causal_workload import TRACE_ORIGIN +from run_causal_point import CausalEnergyAdapter, execute, _useful_backlog +from topology_service import default_config + + +def normalized(config): + rows = [{"id": "gpu", "role": "gpu", "device_id": "gpu"}] + for stack in config["channels"]: + rows.append({"id": stack + ".base", "role": "base", "device_id": stack}) + physical = "HBM3_PROXY" if stack.startswith("hbm") else "HBF_PROXY" + die_count = 4 if stack.startswith("hbm") else 16 + for index in range(die_count): + rows.append({"id": f"{stack}.die{index}", "role": "array_die", + "device_id": stack, "die_index": index, + "physical_type": physical}) + return {"components": rows} + + +def trace(index): + arrival = index * 100 + return { + "trace_origin": TRACE_ORIGIN, "model_id": "tiny", "embedding_access": "tiny", + "prefetch_layers": 0, "batches": [{ + "interval_id": index, "arrival_ns": arrival, + "terminal_task_id": f"c{index}", "token_ids": [f"t{index}"], + "tasks": [ + {"task_id": f"r{index}", "type": "storage", + "tensor": {"tensor_id": "weight", "bytes": 4096}, + "issue_after": [], "consume_after": [], "consumer_count": 1, + "batch_interval_id": index}, + {"task_id": f"c{index}", "type": "compute", "duration_ns": 5, + "depends_on": [f"r{index}"], "is_token_terminal": True, + "batch_interval_id": index}, + ], + }], + } + + +class FakeThermal: + def __init__(self, stacks, baseline): + self.stacks = stacks + self.baseline = baseline + self.total = 0.0 + + def advance(self, start, end, energy): + self.total += sum(energy.values()) + entities = { + f"{stack}.die{index}": {"hotspot_k": 300.0, "mean_k": 300.0} + for stack in self.stacks if stack.startswith("hbf") for index in range(16) + } + return { + "start_ns": start, "end_ns": end, + "temperatures": {**{stack: 300.0 for stack in self.stacks}, "gpu": 300.0}, + "stack_states": {stack: "normal" for stack in self.stacks}, + "hysteresis_budget_bytes": dict(self.baseline), + "energy_j": {"cumulative": {"total_input_j": self.total}}, + "entity_temperatures_k": entities, + } + + +def energy_profile(): + return { + "read_array_j_per_byte": 40e-12, "read_base_j_per_byte": 10e-12, + "program_array_j_per_byte": 1e-9, "program_base_j_per_byte": 2e-9, + "hbm_array_j_per_byte": 40e-12, "hbm_base_j_per_byte": 2e-12, + "hbm_fill_array_j_per_byte": 40e-12, "hbm_fill_base_j_per_byte": 2e-12, + "relay_receive_j_per_byte": 2e-12, "relay_send_j_per_byte": 2e-12, + "erase_j_per_block": 50e-6, "erase_block_bytes": 1_048_576, + } + + +class RunnerTests(unittest.TestCase): + def test_effective_backlog_keeps_finished_sibling_until_group_retry_completes(self): + self.assertEqual(_useful_backlog(8192, 0, 4096), 12288) + self.assertEqual(_useful_backlog(8192, 8192, 4096), 4096) + with self.assertRaises(AssertionError): + _useful_backlog(4096, 8192, 0) + + def test_two_batches_close_exact_service_compute_energy_and_control_loop(self): + service = default_config("all_hbf_direct") + baseline = {stack: sum(channels.values()) * 20_000_000 // 10**9 + for stack, channels in service["channels"].items()} + config = { + "point_id": "tiny", "active_ns": 20_000_000, "recovery_ns": 0, + "window_ns": 20_000_000, "strategy": "guard_only", + "service": service, "energy": energy_profile(), + "gpu_compute_w": 100.0, "gpu_external_w": 10.0, + "target_bytes_per_s_by_stack": {stack: 1 for stack in service["channels"]}, + "trace": {"total_batches": 2, "max_active_batches": 2, + "batch_interval_ns": 100}, + "executor": {"cache_mode": "disabled", "cache_capacity_bytes": 0, + "coalescing_enabled": True, + "prefetch_wait_mode": "wait_at_consumption", + "stripe_unit_bytes": 4096, "migration_mode": "fixed", + "retry_count_per_source_read": 1, + "stripe_targets": [{"stack": "hbf0", "channel": "0", + "route": "direct"}]}, + } + sink = io.StringIO() + summary = execute(config, normalized(service), + FakeThermal(sorted(service["channels"]), baseline), sink, + initial_trace=trace(0), trace_factory=trace) + self.assertEqual(summary["completed_tokens"], 2) + self.assertEqual(summary["pending_external_jobs"], 0) + self.assertEqual(summary["uninstantiated_batches"], 0) + row = json.loads(sink.getvalue()) + self.assertEqual(row["executor"]["completed_tokens"], 2) + self.assertGreater(row["energy"]["scope_energy_j"]["gpu:causal_compute"], 0) + # Two logical consumers coalesce into one source read; its configured + # retry is physical traffic and does not create useful bytes. + self.assertEqual(len(row["service"]["completions"]), 2) + self.assertEqual(summary["offered_useful_bytes_by_stack"]["hbf0"], 4096) + self.assertEqual(summary["delivered_useful_bytes_by_stack"]["hbf0"], 4096) + media_bytes = sum(item["bytes"] for item in row["service"]["activities"] + if item["phase"] == "media_read") + self.assertEqual(media_bytes, 8192) + ids = [item["job_id"] for item in row["service"]["changed_job_progress"]] + self.assertEqual(len(ids), len(set(ids))) + + def test_energy_adapter_counts_only_disjoint_phases_and_prorates_erase(self): + service = default_config("relay") + adapter = CausalEnergyAdapter(normalized(service), service, energy_profile()) + rows = [ + {"operation": "hbm_fill", "phase": "media_fill", "stack": "hbm0", + "channel": "0", "bytes": 100}, + {"operation": "hbm_fill", "phase": "hbm_base_fill", "stack": "hbm0", + "channel": "0", "bytes": 100}, + {"operation": "migration_program", "phase": "reverse_relay", "stack": "hbf0", + "partner": "hbm0", "channel": "0", "bytes": 100}, + {"operation": "migration_program", "phase": "partner_gpu_receive", "stack": "hbf0", + "partner": "hbm0", "channel": "0", "bytes": 100}, + {"operation": "erase", "phase": "media_erase", "stack": "hbf0", + "channel": "0", "bytes": 524_288}, + ] + result = adapter.map(rows) + self.assertAlmostEqual(result["scope_energy_j"]["hbm_fill:array_uniform"], 4e-9) + self.assertAlmostEqual(result["scope_energy_j"]["hbm_fill:base"], .2e-9) + self.assertAlmostEqual(result["scope_energy_j"]["migration:reverse_relay"], .2e-9) + self.assertAlmostEqual(result["scope_energy_j"]["migration:partner_gpu_receive"], .2e-9) + self.assertAlmostEqual(result["scope_energy_j"]["erase:array_prorated"], 25e-6) + + def test_foreground_and_due_maintenance_share_one_exact_service(self): + service = default_config("mixed_direct") + baseline = {stack: sum(channels.values()) * 20_000_000 // 10**9 + for stack, channels in service["channels"].items()} + config = { + "point_id": "shared-maint", "active_ns": 20_000_000, "recovery_ns": 0, + "window_ns": 20_000_000, "strategy": "guard_only", + "service": service, "energy": energy_profile(), + "gpu_compute_w": 0.0, "gpu_external_w": 0.0, + "target_bytes_per_s_by_stack": {stack: 1 for stack in service["channels"]}, + "trace": {"total_batches": 1, "max_active_batches": 1, + "batch_interval_ns": 100}, + "executor": {"cache_mode": "disabled", "cache_capacity_bytes": 0, + "coalescing_enabled": True, + "prefetch_wait_mode": "wait_at_consumption", + "stripe_unit_bytes": 4096, "migration_mode": "fixed", + "stripe_targets": [{"stack": "hbf0", "channel": "0", + "route": "direct"}]}, + "maintenance": { + "mode": "shared", "ea_ev": 1.04, "refresh_trigger": "wall_only", + "initial_equivalent_age_ns": 0, + "initial_wall_age_ns": 86_400_000_000_000, + "initial_temperature_k": 300.0, "block_bytes": 4096, + "pages_per_block": 1, "aged_blocks_per_stack": 16, + "spares_per_channel": 1, "max_blocks_per_cohort": 1, + "program_energy_j_per_byte": None, + "erase_energy_j_per_operation": None, + "energy_evidence": {"program": "TEST_PROXY", "erase": "TEST_PROXY"}, + }, + } + sink = io.StringIO() + summary = execute(config, normalized(service), + FakeThermal(sorted(service["channels"]), baseline), sink, + initial_trace=trace(0), trace_factory=trace) + terminal = summary["maintenance"]["driver"]["terminal_summary"] + self.assertEqual(summary["completed_tokens"], 1) + self.assertEqual(terminal["extent_count"], 64) + self.assertEqual(terminal["status_counts"], {"COMMITTED": 64}) + row = json.loads(sink.getvalue()) + operations = {item["operation"] for item in row["service"]["completions"]} + self.assertTrue({"read", "refresh_read", "program", "erase"}.issubset(operations)) + self.assertEqual(row["maintenance"]["window_age_finish"]["temperature_semantics"], + "PREVIOUS_KNOWN_WINDOW_TEMPERATURE_PIECEWISE_CONSTANT_NO_RETROACTIVE_REAGE") + + def test_uniform_16_channel_group_matches_physical_bytes_capacity_and_energy(self): + physical = default_config("all_hbf_direct") + grouped = deepcopy(physical) + groups = {} + for stack in grouped["channels"]: + grouped["channels"][stack] = {"group": 16 * 96_000_000_000} + groups[stack] = {"group": { + "resource_id": f"{stack}:uniform-16ch-media", + "bandwidth_bytes_per_s": 16 * 96_000_000_000}} + grouped["causal_channel_groups"] = groups + receipts = [] + for config, jobs in ( + (physical, [{"job_id": f"p{i}", "stack": "hbf0", "channel": str(i), + "operation": "read", "route": "direct", "bytes": 1000, + "arrival_ns": 0} for i in range(16)]), + (grouped, [{"job_id": "g", "stack": "hbf0", "channel": "group", + "operation": "read", "route": "direct", "bytes": 16000, + "arrival_ns": 0}]), + ): + service = CausalTopologyService(config) + service.begin_window(0, 20_000_000, + {stack: 10**9 for stack in config["channels"]}, + {stack: "normal" for stack in config["channels"]}) + service.submit_jobs(jobs) + receipts.append(service.advance_to(20_000_000)) + self.assertEqual([sum(row["bytes"] for row in receipt["activities"] + if row["phase"] == "media_read") for receipt in receipts], + [16000, 16000]) + profile = energy_profile() + physical_energy = CausalEnergyAdapter( + normalized(physical), physical, profile).map(receipts[0]["activities"]) + group_energy = CausalEnergyAdapter( + normalized(grouped), grouped, profile, + {"mode": "uniform_stack_group_16ch", "physical_channels_per_stack": 16} + ).map(receipts[1]["activities"]) + self.assertEqual(set(physical_energy["component_energy_j"]), + set(group_energy["component_energy_j"])) + for owner, value in physical_energy["component_energy_j"].items(): + self.assertAlmostEqual(value, group_energy["component_energy_j"][owner]) + + +if __name__ == "__main__": + unittest.main() diff --git a/experiments/eq3_system_thermal/test_run_maintenance_point.py b/experiments/eq3_system_thermal/test_run_maintenance_point.py new file mode 100644 index 0000000..786d35f --- /dev/null +++ b/experiments/eq3_system_thermal/test_run_maintenance_point.py @@ -0,0 +1,174 @@ +import io +import json +import unittest + +from energy import engineering_energy_profile +from maintenance_inputs import add_maintenance +from run_maintenance_point import execute, WINDOW_NS +from topology_service import default_config, media_cost_from_service_rate +from reliability import DAY_NS + + +def normalized(service): + components = [{"id": "gpu", "device_id": "gpu", "role": "compute"}] + for stack in service["fabric"]["hbf"]: + components.append({"id": stack + ".base", "device_id": stack, "role": "base"}) + for channel in sorted(service["channels"][stack], key=int): + components.append({"id": f"{stack}.die{channel}", "device_id": stack, + "role": "array_die", "die_index": int(channel)}) + return {"components": components} + + +class FakeThermal: + def __init__(self, normalized_data, service, *, temperature_k=300.0, guard_state="normal"): + self.components = [row["id"] for row in normalized_data["components"]] + self.stacks = sorted(service["channels"]) + self.total = 0.0 + self.temperature_k = temperature_k + self.guard_state = guard_state + self.limits = {"hbf": [353.15, 363.15, 378.15], + "hbm": [353.15, 363.15, 378.15], + "gpu": [363.15, 373.15, 383.15]} + + def advance(self, start, end, component_energy): + self.total += sum(component_energy.values()) + entity = {component: {"hotspot_k": self.temperature_k + self.total * 1e-6} + for component in self.components} + return {"entity_temperatures_k": entity, + "temperatures": {**{stack: self.temperature_k + self.total * 1e-6 + for stack in self.stacks}, "gpu": self.temperature_k}, + "stack_states": {stack: self.guard_state for stack in self.stacks}, + "hysteresis_budget_bytes": {stack: 10**18 for stack in self.stacks}, + "energy_j": {"cumulative": {"total_input_j": self.total}}} + + +class SequenceThermal(FakeThermal): + def __init__(self, normalized_data, service, states): + super().__init__(normalized_data, service) + self.sequence = iter(states) + + def advance(self, start, end, component_energy): + result = super().advance(start, end, component_energy) + state = next(self.sequence) + result["stack_states"] = {stack: state for stack in self.stacks} + return result + + +def config(mode): + service = default_config("all_hbf_direct") + service["operation_media_cost"]["program"] = media_cost_from_service_rate( + 96_000_000_000, 655_360_000) + service["operation_media_cost"]["erase"] = media_cost_from_service_rate( + 96_000_000_000, 16 * 4096 * 256 * 1000) + service["operation_cost_evidence"] = "FIXED_TEST_ENGINEERING_PROXY" + energy = engineering_energy_profile() + energy["erase_array_j_per_operation"] = 50e-6 + return {"point_id": "maintenance-fixed", "strategy": "guard_only", + "service": service, "energy": energy, + "workload": {"active_ns": WINDOW_NS, "per_stack_Bps": 1, + "pattern": "continuous", "channel_distribution": "uniform"}, + "recovery_ns": 2 * WINDOW_NS, "gpu_external_w": 0, + "control_disabled": True, + "maintenance": {"mode": mode, "ea_ev": 1.04, + "refresh_trigger": "equivalent_age_or_wall", + "initial_equivalent_age_ns": DAY_NS, "initial_wall_age_ns": 0, + "initial_temperature_k": 300.0, "block_bytes": 4096 * 256, + "pages_per_block": 256, "aged_blocks_per_stack": 16, + "spares_per_channel": 1, "max_blocks_per_cohort": 1, + "program_energy_j_per_byte": 0.05 * 100e-6 / 4096, + "erase_energy_j_per_operation": 50e-6, + "energy_evidence": {"program": "FIXED_PROXY", "erase": "FIXED_PROXY"}}} + + +class MaintenanceRunnerTests(unittest.TestCase): + def test_frozen_bounded_input_geometry(self): + base = config("disabled") + base.pop("maintenance") + frozen = add_maintenance(base, "shared") + self.assertEqual(frozen["maintenance"]["aged_blocks_per_stack"] * + frozen["maintenance"]["block_bytes"], 4 * 1024**3) + self.assertEqual(frozen["maintenance"]["spares_per_channel"], 16) + self.assertEqual(frozen["maintenance_scope"]["spares_per_stack"], 256) + + def test_shared_and_ideal_complete_same_exact_extents(self): + for mode in ("shared", "ideal_independent"): + cfg = config(mode) + norm = normalized(cfg["service"]) + sink = io.StringIO() + summary, final = execute(cfg, norm, FakeThermal(norm, cfg["service"]), sink) + rows = [json.loads(line) for line in sink.getvalue().splitlines()] + self.assertEqual(len(rows), 3) + self.assertEqual(summary["maintenance_mode"], mode) + self.assertEqual(summary["offered_bytes"], 0) + self.assertEqual(final["driver"]["terminal_summary"]["extent_count"], 8 * 16) + self.assertEqual(final["driver"]["terminal_summary"]["status_counts"], + {"COMMITTED": 8 * 16}) + self.assertEqual(len(final["driver"]["physical_wear"]), 2 * 8 * 16) + self.assertIsNotNone(rows[0]["independent_maintenance_service"] + if mode == "ideal_independent" else rows[0]["service"]) + self.assertLess(len(json.dumps(rows[0]["maintenance_delta"])), 200_000) + self.assertTrue(all(not row["maintenance_delta"]["events"] + or row["maintenance_delta"]["reliability_events"] + for row in rows)) + + def test_disabled_is_fresh_baseline(self): + cfg = config("disabled") + norm = normalized(cfg["service"]) + sink = io.StringIO() + summary, final = execute(cfg, norm, FakeThermal(norm, cfg["service"]), sink) + self.assertIsNone(final) + self.assertEqual(summary["maintenance_mode"], "disabled") + rows = [json.loads(line) for line in sink.getvalue().splitlines()] + self.assertTrue(all(row["reliability"]["status"] == + "FRESH_BASELINE_NO_MAINTENANCE" for row in rows)) + + def test_maintenance_due_facts_reach_feedback_policy(self): + cfg = config("shared") + cfg["strategy"] = "read_rate_feedback_thermal_guard_v1" + cfg["workload"]["per_stack_Bps"] = 10**12 + for row in cfg["service"]["fabric"]["hbf"].values(): + row["direct_link"]["bandwidth_bytes_per_s"] = 100_000_000_000 + norm = normalized(cfg["service"]) + sink = io.StringIO() + execute(cfg, norm, FakeThermal(norm, cfg["service"]), sink) + first = json.loads(sink.getvalue().splitlines()[0]) + for decision in first["control"]["decisions"].values(): + reasons = decision["stack_decisions"][0]["reasons"] + self.assertIn("MAINTENANCE_DUE_VISIBLE", reasons) + + def test_ideal_independent_preserves_shutdown_admission(self): + cfg = config("ideal_independent") + cfg["control_disabled"] = False + cfg["maintenance"]["initial_equivalent_age_ns"] = DAY_NS - WINDOW_NS + cfg["maintenance"]["initial_temperature_k"] = 358.15 + norm = normalized(cfg["service"]) + sink = io.StringIO() + _, final = execute( + cfg, norm, + FakeThermal(norm, cfg["service"], temperature_k=358.15, + guard_state="shutdown"), sink) + rows = [json.loads(line) for line in sink.getvalue().splitlines()] + blocked = rows[1]["independent_maintenance_service"]["blocked"] + self.assertTrue(blocked) + self.assertTrue(all("shutdown" in reason + for row in blocked for reason in row["reasons"])) + self.assertEqual(final["driver"]["terminal_summary"]["operation_count"], 0) + + def test_hbm_shared_endpoint_recovers_full_cap_after_light(self): + cfg = config("disabled") + cfg["service"] = default_config("mixed_direct") + cfg["control_disabled"] = False + cfg["strategy"] = "read_rate_feedback_thermal_guard_v1" + norm = normalized(cfg["service"]) + sink = io.StringIO() + execute(cfg, norm, SequenceThermal(norm, cfg["service"], + ["light", "normal", "normal"]), sink) + rows = [json.loads(line) for line in sink.getvalue().splitlines()] + hbm = next(stack for stack in cfg["service"]["channels"] if stack.startswith("hbm")) + baseline = sum(cfg["service"]["channels"][hbm].values()) * WINDOW_NS // 10**9 + self.assertEqual(rows[0]["control"]["next_budgets"][hbm], baseline // 2) + self.assertEqual(rows[1]["control"]["next_budgets"][hbm], baseline) + + +if __name__ == "__main__": + unittest.main() diff --git a/experiments/eq3_system_thermal/test_run_system_point.py b/experiments/eq3_system_thermal/test_run_system_point.py new file mode 100644 index 0000000..997e8e2 --- /dev/null +++ b/experiments/eq3_system_thermal/test_run_system_point.py @@ -0,0 +1,37 @@ +import io +import unittest +from prepare_stage import config +from run_system_point import execute + + +class ThermalFixture: + limits={'hbf':(353.15,363.15,378.15),'hbm':(353.15,363.15,378.15),'gpu':(363.15,373.15,383.15)} + def __init__(self, stacks): + self.stacks=stacks;self.total=0 + def advance(self,start,end,energy): + self.total+=sum(energy.values()) + return {'temperatures':{s:300.01 for s in self.stacks}, + 'stack_states':{s:'normal' for s in self.stacks}, + 'hysteresis_budget_bytes':{s:1_536_000_000_000*20_000_000//10**9 for s in self.stacks}, + 'energy_j':{'cumulative':{'total_input_j':self.total}}} + + +class RunnerTests(unittest.TestCase): + def test_four_topology_energy_and_recovery_drain(self): + for topology in ('mixed_direct','all_hbf_direct','relay','dash'): + c=config('fixed',topology,1_920_000_000_000,'guard_only',.04,.04) + rows=[{'id':'gpu','device_id':'gpu','role':'compute_die'}] + for stack in c['service']['channels']: + rows.append({'id':stack+'.base','device_id':stack,'role':'base'}) + for i in range(16): + rows.append({'id':f'{stack}.die{i}','device_id':stack,'role':'array_die', + 'die_index':i,'physical_type':stack[:3].upper()}) + result=execute(c,{'components':rows},ThermalFixture(c['service']['channels']),io.StringIO()) + self.assertEqual(result['offered_bytes'],result['delivered_bytes']) + self.assertEqual(result['backlog_bytes'],0) + coefficient={'mixed_direct':50e-12,'all_hbf_direct':50e-12, + 'relay':54e-12,'dash':52e-12}[topology] + self.assertAlmostEqual(result['energy_j'],result['delivered_bytes']*coefficient,places=8) + + +if __name__=='__main__':unittest.main() diff --git a/experiments/eq3_system_thermal/test_system_models.py b/experiments/eq3_system_thermal/test_system_models.py new file mode 100644 index 0000000..4154da5 --- /dev/null +++ b/experiments/eq3_system_thermal/test_system_models.py @@ -0,0 +1,283 @@ +import math +import unittest + +from causal_workload import CausalExecutor, build_architecture_trace, TRACE_ORIGIN +from reliability import DAY_NS, ReliabilityLedger, arrhenius_acceleration + + +class ReliabilityTests(unittest.TestCase): + def test_age_cadence_wear_and_commit_rules(self): + ledger = ReliabilityLedger({"ea_ev": 1.04, "tref_k": 358.15, + "initial_equivalent_age_ns": DAY_NS * 9}) + self.assertEqual(arrhenius_acceleration(358.15, 1.04, 358.15), 1.0) + ledger.advance_temperature("hbf0/ch0/d0/p0/b7", 0, 100, 358.15) + ledger.record_program("hbf0/ch0/d0/p0/b7", 100, True, "p0") + ledger.record_program("hbf0/ch0/d0/p0/b7", 100, False, "p1") + ledger.record_erase("hbf0/ch0/d0/p0/b7", 100, True, "e0") + ledger.record_refresh_terminal("hbf0/ch0/d0/p0/b7", 100, False, "m0") + before = ledger.snapshot()["blocks"]["hbf0/ch0/d0/p0/b7"] + self.assertEqual(before["program_phase_started_count"], 2) + self.assertEqual(before["program_completed_count"], 1) + self.assertEqual(before["erase_phase_started_count"], 1) + self.assertEqual(before["erase_completed_count"], 1) + self.assertEqual(before["equivalent_age_ns"], DAY_NS * 9 + 100) + self.assertEqual(before["next_wall_due_ns"], DAY_NS) # equivalent age does not compress wall cadence + ledger.record_refresh_terminal("hbf0/ch0/d0/p0/b7", 100, True, "m1") + after = ledger.snapshot()["blocks"]["hbf0/ch0/d0/p0/b7"] + self.assertEqual(after["equivalent_age_ns"], 0) + self.assertEqual(after["next_wall_due_ns"], DAY_NS + 100) + + def test_ea_has_explicit_conservative_trigger_consumer(self): + hot = ReliabilityLedger({"ea_ev": 1.08, "refresh_trigger": "equivalent_age_or_wall", + "initial_equivalent_age_ns": DAY_NS - 10}) + hot.advance_temperature("b0", 0, 1, 400.0) + self.assertIn("EQUIVALENT_AGE_24H_CONSERVATIVE_POLICY", hot.due_reasons("b0", 1)) + wall = ReliabilityLedger({"ea_ev": 1.08, "refresh_trigger": "wall_only", + "initial_equivalent_age_ns": DAY_NS * 2}) + wall.advance_temperature("b0", 0, 1, 400.0) + self.assertEqual(wall.due_reasons("b0", 1), []) + + def test_retry_is_explicit_scenario_not_rber(self): + ledger = ReliabilityLedger({"ea_ev": 1.01}) + ledger.record_retry_scenario("b0", 0, 2, 17, 1e-9, "fixed-ecc-arm") + row = ledger.drain_events()[-1] + self.assertEqual(row["evidence_class"], "SCENARIO_ASSUMPTION_NO_RBER_CLAIM") + + +class CausalTests(unittest.TestCase): + @staticmethod + def small_trace(): + return {'trace_origin':TRACE_ORIGIN,'batches':[ + {'arrival_ns':at,'terminal_task_id':f'c{i}','token_ids':[str(i)],'tasks':[ + {'task_id':f'r{i}','type':'storage','tensor':{'tensor_id':'weight','bytes':4096}, + 'issue_after':[],'consume_after':[],'consumer_count':1,'batch_interval_id':i}, + {'task_id':f'c{i}','type':'compute','duration_ns':5,'depends_on':[f'r{i}']}]} + for i,at in enumerate((0,100))]} + + def test_hbm_cache_fill_and_hits_require_actual_service(self): + executor=CausalExecutor(self.small_trace(),{ + 'cache_mode':'external_hbm','cache_capacity_bytes':4096,'coalescing_enabled':True, + 'prefetch_wait_mode':'wait_at_consumption','stripe_unit_bytes':4096, + 'stripe_targets':[{'stack':'hbf0','channel':'0','route':'direct'}], + 'fast_stripe_targets':[{'stack':'hbm0','channel':'0','route':'direct'}]}) + job=executor.poll(0)[0] + executor.complete(job['job_id'],10,4096) + self.assertNotIn('weight',executor.cache) + fill=executor.poll(10)[0] + self.assertEqual(fill['operation'],'hbm_fill') + executor.complete(fill['job_id'],20,4096) + self.assertIn('weight',executor.cache) + second=executor.poll(100)[0] + self.assertEqual(second['stack'],'hbm0') + self.assertNotIn('r1',executor.done) + executor.complete(second['job_id'],110,4096) + executor.poll(110) + self.assertNotIn('c1',executor.done) + executor.poll(115) + self.assertEqual(executor.done['c1'],115) + + def test_retry_consumes_work_before_unique_logical_success(self): + x=CausalExecutor(self.small_trace(),{'cache_mode':'disabled','cache_capacity_bytes':0, + 'coalescing_enabled':True,'prefetch_wait_mode':'wait_at_consumption','stripe_unit_bytes':4096, + 'retry_count_per_source_read':1, + 'stripe_targets':[{'stack':'hbf0','channel':'0','route':'direct'}]}) + read=x.poll(0)[0];x.complete(read['job_id'],10,4096) + self.assertNotIn('r0',x.done) + retry=x.poll(10)[0];self.assertEqual(retry['operation'],'retry') + x.complete(retry['job_id'],20,4096) + self.assertEqual(x.done['r0'],20) + success=[r for r in x.events if r['kind']=='storage_complete'] + self.assertEqual(len(success),1) + self.assertEqual(success[0]['retry_count'],1) + + def test_compute_is_a_shared_resource(self): + trace=self.small_trace();trace['batches'][1]['arrival_ns']=0 + executor=CausalExecutor(trace,{'cache_mode':'disabled','cache_capacity_bytes':0, + 'coalescing_enabled':False,'prefetch_wait_mode':'wait_at_consumption','stripe_unit_bytes':4096, + 'stripe_targets':[{'stack':'hbf0','channel':'0','route':'direct'}]}) + for job in executor.poll(0):executor.complete(job['job_id'],10,job['bytes']) + executor.poll(10);executor.poll(15);executor.poll(20) + intervals=[(r['start_ns'],r['completion_ns']) for r in executor.events if r['kind']=='compute_complete'] + self.assertEqual(intervals,[(10,15),(15,20)]) + + def test_issue_stall_blocks_compute_until_prefetch_delivery(self): + trace={'trace_origin':TRACE_ORIGIN,'batches':[{'arrival_ns':0,'terminal_task_id':'end', + 'token_ids':['t'],'tasks':[ + {'task_id':'read','type':'storage','tensor':{'tensor_id':'a','bytes':4096}, + 'issue_after':[],'consume_after':[],'consumer_count':1,'batch_interval_id':0}, + {'task_id':'compute','type':'compute','depends_on':['read'],'duration_ns':100}, + {'task_id':'prefetch','type':'storage','tensor':{'tensor_id':'b','bytes':4096}, + 'issue_after':['read'],'consume_after':['compute'],'consumer_count':1,'batch_interval_id':0}, + {'task_id':'end','type':'compute','depends_on':['prefetch'],'duration_ns':1}]}]} + starts={} + for mode in ('stall_at_issue','wait_at_consumption'): + x=CausalExecutor(trace,{'cache_mode':'disabled','cache_capacity_bytes':0, + 'coalescing_enabled':True,'prefetch_wait_mode':mode,'stripe_unit_bytes':4096, + 'stripe_targets':[{'stack':'hbf0','channel':'0','route':'direct'}]}) + read=x.poll(0)[0];x.complete(read['job_id'],10,4096) + prefetch=x.poll(10)[0] + x.complete(prefetch['job_id'],100,4096);x.poll(100) + starts[mode]=next(e['start_ns'] for e in x.events if e['kind']=='compute_start') + self.assertEqual(starts,{'stall_at_issue':100,'wait_at_consumption':10}) + + def test_streaming_retirement_preserves_facts_and_accepts_next_batch(self): + cfg=self.config('Qwen/Qwen2.5-7B-Instruct') + trace=build_architecture_trace(cfg) + x=CausalExecutor(trace,{'cache_mode':'disabled','cache_capacity_bytes':0, + 'coalescing_enabled':True,'prefetch_wait_mode':'wait_at_consumption','stripe_unit_bytes':4096, + 'stripe_targets':[{'stack':'hbf0','channel':'0','route':'direct'}]}) + now=0 + for _ in range(1000): + jobs=x.poll(now) + if jobs: + now+=1 + for job in jobs:x.complete(job['job_id'],now,job['bytes']) + elif (event:=x.next_internal_event_ns()) is not None:now=event + else:break + retired=x.retire_completed_batches(now) + self.assertEqual(retired[0]['token_count'],3) + self.assertEqual(x.tasks,{}) + next_trace=build_architecture_trace({**cfg,'first_interval':1}) + x.append_trace(next_trace) + self.assertTrue(x.poll(now)) + with self.assertRaisesRegex(ValueError,'duplicate'):x.append_trace(next_trace) + + @staticmethod + def config(model): + return {"model_id": model, "batch_intervals": 1, "batch_size": 3, + "batch_interval_ns": 13, "prefetch_layers": 2, + "attention_compute_ns_per_token": 3, + "mlp_compute_ns_per_token": 5, + "output_compute_ns_per_token": 11, + "embedding_access": "full_weight_stress"} + + def test_official_sizes_and_external_completion_gate(self): + for model, expected in (("Qwen/Qwen2.5-7B-Instruct", 15231233024), + ("Qwen/Qwen2.5-72B-Instruct", 145412407296)): + trace = build_architecture_trace(self.config(model)) + tensors = {} + for task in trace["batches"][0]["tasks"]: + if task["type"] == "storage": + tensors[task["tensor"]["tensor_id"]] = task["tensor"]["bytes"] + self.assertEqual(sum(tensors.values()), expected) + + trace = build_architecture_trace(self.config("Qwen/Qwen2.5-7B-Instruct")) + executor = CausalExecutor(trace, {"cache_mode":"disabled", "coalescing_enabled":True, "prefetch_wait_mode":"wait_at_consumption", + "cache_capacity_bytes": 0, "migration_mode": "fixed", + "stripe_unit_bytes": 4096, + "default_placement": {"stack": "hbf0", "channel": "0", "route": "direct"}, + }) + first = executor.poll(0) + self.assertTrue(first) + self.assertFalse(any(x["complete"] for x in executor.result(1_000_000)["tokens"])) + # Actual completion timestamps, not model bytes, unlock the dependency DAG. + now = 7 + for job in first: + executor.complete(job["job_id"], now, job["bytes"]) + for _ in range(1000): + jobs = executor.poll(now) + if jobs: + now += 7 + for job in jobs: + executor.complete(job["job_id"], now, job["bytes"]) + continue + event = executor.next_internal_event_ns() + if event is not None: + now = event + continue + if all(x["complete"] for x in executor.result(now)["tokens"]): + break + self.fail("causal executor stalled") + result = executor.result(now) + self.assertTrue(all(x["complete"] for x in result["tokens"])) + self.assertEqual(len({x["completion_ns"] for x in result["tokens"]}), 1) + self.assertNotEqual(result["tokens"][0]["completion_ns"] % 20_000_000, 0) + self.assertTrue(all(job["metadata"]["batch_consumer_count"] == 3 + for job in result["jobs"] if job["operation"] == "read")) + consumed = [x for x in result["events"] if x["kind"] == "storage_consumed"] + self.assertTrue(any(x["consume_ns"] > x["ready_ns"] for x in consumed)) + self.assertTrue(all("batch_interval_id" in job["metadata"] + for job in result["jobs"] if job["operation"] == "read")) + + def test_migration_read_program_commit_and_erase_are_distinct(self): + executor=CausalExecutor(self.small_trace(),{ + 'cache_mode':'disabled','cache_capacity_bytes':0,'coalescing_enabled':True, + 'prefetch_wait_mode':'wait_at_consumption','stripe_unit_bytes':4096, + 'migration_mode':'basic','migration_capacity_bytes':1048576, + 'migration_access_threshold':1, + 'stripe_targets':[{'stack':'hbf0','channel':'0','route':'direct'}], + 'fast_stripe_targets':[{'stack':'hbf1','channel':'0','route':'direct'}]}) + job=executor.poll(0)[0];executor.complete(job['job_id'],10,4096) + copy=executor.offer_migrations(10)[0] + self.assertEqual(copy['stack'],'hbf0');self.assertEqual(copy['operation'],'read') + executor.complete(copy['job_id'],20,copy['bytes']) + program=executor.poll(20)[0] + self.assertEqual(program['stack'],'hbf1');self.assertEqual(program['operation'],'migration_program') + self.assertNotEqual(executor.tensor_tier.get('weight'),'fast') + executor.complete(program['job_id'],30,program['bytes']) + self.assertEqual(executor.tensor_tier['weight'],'fast') + erase=executor.poll(30)[0] + self.assertEqual(erase['stack'],'hbf0');self.assertEqual(erase['operation'],'erase') + self.assertEqual(erase['bytes'],1048576) + executor.complete(erase['job_id'],40,erase['bytes']) + self.assertEqual(executor.pending_migrations,{}) + next_read=executor.poll(100)[0] + self.assertEqual(next_read['stack'],'hbf1') + + def test_full_capacity_lru_serves_later_interval_without_external_reads(self): + trace = build_architecture_trace({**self.config("Qwen/Qwen2.5-7B-Instruct"), + "batch_intervals": 2, + "batch_interval_ns": 10_000}) + executor = CausalExecutor(trace, {"cache_mode":"disabled", "coalescing_enabled":True, "prefetch_wait_mode":"wait_at_consumption", + "cache_mode":"ideal_metadata_only", "cache_capacity_bytes": trace["official_metadata"]["tensor_payload_bytes"], + "migration_mode": "fixed", + "stripe_unit_bytes": 4096, + "default_placement": {"stack": "hbf0", "channel": "0", "route": "direct"}, + }) + now = 0 + for _ in range(2000): + jobs = executor.poll(now) + if jobs: + now += 1 + for job in jobs: + executor.complete(job["job_id"], now, job["bytes"]) + elif (event := executor.next_internal_event_ns()) is not None: + now = event + elif all(x["complete"] for x in executor.result(now)["tokens"]): + break + else: + self.fail("two-interval cache execution stalled") + result = executor.result(now) + self.assertTrue(all(x["complete"] for x in result["tokens"])) + self.assertTrue(any(x["kind"] == "cache_hit" for x in result["events"])) + unique_tensors = {job["metadata"]["tensor_id"] for job in result["jobs"]} + self.assertEqual(len(result["jobs"]), len(unique_tensors)) + + def test_large_tensor_stripes_all_targets_and_waits_for_last_child(self): + config = {**self.config("Qwen/Qwen2.5-7B-Instruct"), + "embedding_access": "selected_token_rows"} + trace = build_architecture_trace(config) + embed = trace["batches"][0]["tasks"][0]["tensor"] + self.assertEqual(embed["bytes"], 3584 * 2 * 3) + self.assertEqual(embed["access_semantics"], "SELECTED_TOKEN_ROWS_SYNTHETIC_IDENTITIES") + targets = [{"stack": f"hbf{i}", "channel": str(i), "route": "direct"} + for i in range(4)] + executor = CausalExecutor(trace, {"cache_mode":"disabled", "coalescing_enabled":True, "prefetch_wait_mode":"wait_at_consumption", "cache_capacity_bytes": 0, + "migration_mode": "fixed", + "stripe_unit_bytes": 4096, + "stripe_targets": targets}) + jobs = executor.poll(0) + group = "causal:interval0:l0:attn_read" + children = [job for job in jobs if job["metadata"]["parent_group_id"] == group] + self.assertEqual({job["stack"] for job in children}, {"hbf0", "hbf1", "hbf2", "hbf3"}) + self.assertEqual(sum(job["bytes"] for job in children), + executor.tasks["interval0:l0:attn_read"]["tensor"]["bytes"]) + for child in children[:-1]: + executor.complete(child["job_id"], 5, child["bytes"]) + self.assertNotIn("external_ready_ns", executor.tasks["interval0:l0:attn_read"]) + executor.complete(children[-1]["job_id"], 9, children[-1]["bytes"]) + self.assertEqual(executor.tasks["interval0:l0:attn_read"]["external_ready_ns"], 9) + + +if __name__ == "__main__": + unittest.main() diff --git a/experiments/eq3_system_thermal/test_tiny_cpu_trace.py b/experiments/eq3_system_thermal/test_tiny_cpu_trace.py new file mode 100644 index 0000000..b76891b --- /dev/null +++ b/experiments/eq3_system_thermal/test_tiny_cpu_trace.py @@ -0,0 +1,69 @@ +import copy +import unittest + +from tiny_cpu_trace import build_trace + + +CONFIG = { + "schema_version": "eq3-tiny-qwen2-cpu-trace-config-v1", + "seed": 20260920, + "vocab_size": 128, + "hidden_size": 56, + "layers": 28, + "num_attention_heads": 28, + "num_key_value_heads": 4, + "intermediate_size": 128, + "rms_norm_eps": 1e-6, + "rope_theta": 10000.0, + "prompt_token_ids": [1, 7, 11, 19], + "decode_tokens": 2, + "projection_context_tokens": 4096, +} + + +class TinyCpuTraceTests(unittest.TestCase): + @classmethod + def setUpClass(cls): + cls.trace = build_trace(copy.deepcopy(CONFIG)) + + def test_forward_structure_and_gqa(self): + self.assertEqual([x["forward_id"] for x in self.trace["forwards"]], + ["prefill", "decode0", "decode1"]) + self.assertEqual([x["context_tokens_after"] for x in self.trace["forwards"]], + [4, 5, 6]) + attention = [x for x in self.trace["operations"] + if x["name"].endswith("rope_gqa_causal_attention")] + self.assertEqual(len(attention), 28 * 3) + self.assertEqual(attention[0]["details"]["q_shape"], [4, 28, 2]) + self.assertEqual(attention[0]["details"]["kv_shape"], [4, 4, 2]) + self.assertEqual(attention[-1]["details"]["context_tokens"], 6) + + def test_accesses_are_real_arrays_and_dependencies_are_ordered(self): + op_sequence = {row["op_id"]: row["sequence"] for row in self.trace["operations"]} + for row in self.trace["operations"]: + for dependency in row["depends_on"]: + if dependency != "token_ids": + self.assertLess(op_sequence[dependency], row["sequence"]) + self.assertTrue(self.trace["weight_accesses"]) + self.assertTrue(all(row["access_bytes"] > 0 for row in self.trace["weight_accesses"])) + self.assertEqual(self.trace["weight_accesses"][0]["weight_name"], + "model.embed_tokens.weight") + + def test_official_target_projection_is_independent_and_exact(self): + expected = {"Qwen/Qwen2.5-7B-Instruct": 15_231_233_024, + "Qwen/Qwen2.5-72B-Instruct": 145_412_407_296} + for model, byte_count in expected.items(): + row = self.trace["target_projection"][model] + self.assertEqual(row["tensor_payload_bytes"], byte_count) + self.assertEqual(sum(x["payload_bytes"] for x in row["logical_regions"]), byte_count) + self.assertGreater(row["analytical_macs_per_token_at_context"], 0) + addresses = [x["logical_address_bytes"] for x in row["logical_regions"]] + self.assertTrue(all(value % (1024 * 1024) == 0 for value in addresses)) + + def test_deterministic_checksum(self): + again = build_trace(copy.deepcopy(CONFIG)) + self.assertEqual(self.trace["trace_sha256"], again["trace_sha256"]) + + +if __name__ == "__main__": + unittest.main() diff --git a/experiments/eq3_system_thermal/test_tiny_trace_consumer.py b/experiments/eq3_system_thermal/test_tiny_trace_consumer.py new file mode 100644 index 0000000..df944f2 --- /dev/null +++ b/experiments/eq3_system_thermal/test_tiny_trace_consumer.py @@ -0,0 +1,125 @@ +import copy +import hashlib +import json +from pathlib import Path +import tempfile +import unittest + +from causal_workload import (CausalExecutor, TINY_TRACE_ORIGIN, TRACE_ORIGIN, + build_architecture_trace) + + +ARTIFACT = Path(__file__).parent / "stage/traces/tiny_qwen2_cpu_trace_v1/trace.json" +TRACE_SHA = "feea4657abc5e7b981208b23b8d2fff9836194694c5c3dec69c6bf59bbe627a7" + + +def config(model="Qwen/Qwen2.5-7B-Instruct", mode="tiny_cpu_template"): + result = { + "model_id": model, "batch_intervals": 1, "batch_size": 2, + "batch_interval_ns": 100, "prefetch_layers": 2, + "attention_compute_ns_per_token": 3, + "mlp_compute_ns_per_token": 5, + "output_compute_ns_per_token": 7, + "embedding_access": "full_weight_stress", + "dependency_mode": mode, + } + if mode == "tiny_cpu_template": + result.update({"tiny_trace_path": str(ARTIFACT), + "tiny_trace_sha256": TRACE_SHA, + "projection_context_tokens": 4096}) + return result + + +class TinyTraceConsumerTests(unittest.TestCase): + def test_on_demand_has_no_hidden_within_layer_prefetch(self): + cfg=config();cfg.update(prefetch_layers=0,prefetch_mode="on_demand") + trace=build_architecture_trace(cfg) + tasks={t['task_id']:t for t in trace['batches'][0]['tasks']} + self.assertEqual(tasks['interval0:l0:attn_read']['issue_after'],['interval0:embed_read']) + self.assertEqual(tasks['interval0:l0:mlp_read']['issue_after'],['interval0:l0:attn_compute']) + self.assertEqual(tasks['interval0:l1:attn_read']['issue_after'],['interval0:l0:mlp_compute']) + cfg['prefetch_layers']=1 + with self.assertRaises(ValueError):build_architecture_trace(cfg) + + def test_validated_template_drives_target_dependencies_and_shapes(self): + for model, layers, payload in ( + ("Qwen/Qwen2.5-7B-Instruct", 28, 15_231_233_024), + ("Qwen/Qwen2.5-72B-Instruct", 80, 145_412_407_296)): + trace = build_architecture_trace(config(model)) + self.assertEqual(trace["trace_origin"], TINY_TRACE_ORIGIN) + provenance = trace["structure_provenance"] + self.assertEqual(provenance["trace_sha256"], TRACE_SHA) + projection = provenance["target_projection"] + self.assertEqual(projection["layer_count"], layers) + self.assertEqual(projection["tensor_payload_bytes"], payload) + self.assertGreater(projection["analytical_total_macs_per_token_at_context"], 0) + for region in projection["logical_regions"]: + elements = sum(self._product(shape) for shape in region["weight_shapes"]) + self.assertEqual(elements * 2, region["bytes"]) + self.assertEqual(region["logical_address_bytes"] % (1024 * 1024), 0) + tasks = trace["batches"][0]["tasks"] + for layer in range(layers): + attention = next(row for row in tasks + if row["task_id"] == f"interval0:l{layer}:attn_compute") + mlp = next(row for row in tasks + if row["task_id"] == f"interval0:l{layer}:mlp_compute") + self.assertEqual(attention["structure_role"], + "ATTENTION_AFTER_INPUT_AND_WEIGHT_READ") + self.assertEqual(mlp["depends_on"], + [f"interval0:l{layer}:attn_compute", + f"interval0:l{layer}:mlp_read"]) + head = next(row for row in tasks if row["task_id"] == "interval0:head_read") + self.assertEqual(head["structure_template_op_ids"], + ["prefill:op282:lm_head"]) + + @staticmethod + def _product(shape): + result = 1 + for value in shape: + result *= value + return result + + def test_executor_consumes_template_trace_without_changing_compute_cost(self): + trace = build_architecture_trace(config()) + compute = [row for row in trace["batches"][0]["tasks"] if row["type"] == "compute"] + self.assertEqual(next(row for row in compute if row["task_id"].endswith("l0:attn_compute"))[ + "duration_ns"], 6) + executor = CausalExecutor(trace, { + "cache_mode": "disabled", "cache_capacity_bytes": 0, + "coalescing_enabled": True, "prefetch_wait_mode": "wait_at_consumption", + "migration_mode": "fixed", "stripe_unit_bytes": 4096, + "default_placement": {"stack": "hbf0", "channel": "0", "route": "direct"}, + }) + self.assertTrue(executor.poll(0)) + + def test_existing_synthetic_mode_remains_explicit(self): + trace = build_architecture_trace(config(mode="synthetic_metadata_dag")) + self.assertEqual(trace["trace_origin"], TRACE_ORIGIN) + self.assertEqual(trace["dependency_mode"], "synthetic_metadata_dag") + self.assertIsNone(trace["structure_provenance"]) + self.assertFalse(any("structure_template_op_ids" in row + for row in trace["batches"][0]["tasks"])) + + def test_checksum_and_causal_dependency_tampering_are_rejected(self): + with tempfile.TemporaryDirectory() as directory: + path = Path(directory) / "trace.json" + document = json.loads(ARTIFACT.read_text()) + operation = next(row for row in document["operations"] + if row["name"] == "layer0.rope_gqa_causal_attention") + operation["depends_on"] = ["missing"] + payload = dict(document) + payload.pop("trace_sha256") + checksum = hashlib.sha256(json.dumps( + payload, sort_keys=True, separators=(",", ":"), allow_nan=False + ).encode()).hexdigest() + document["trace_sha256"] = checksum + path.write_text(json.dumps(document)) + bad = config() + bad["tiny_trace_path"] = str(path) + bad["tiny_trace_sha256"] = checksum + with self.assertRaisesRegex(ValueError, "dependency"): + build_architecture_trace(bad) + + +if __name__ == "__main__": + unittest.main() diff --git a/experiments/eq3_system_thermal/test_topology_service.py b/experiments/eq3_system_thermal/test_topology_service.py new file mode 100644 index 0000000..f62bc8c --- /dev/null +++ b/experiments/eq3_system_thermal/test_topology_service.py @@ -0,0 +1,252 @@ +#!/usr/bin/env python3 +import copy +import unittest + +from topology_service import TopologyService, default_config, media_cost_from_service_rate +from eq3_basic_fabric import BasicFabric + + +W = 20_000_000 + + +def budgets(config, value=10**15): + return {stack: value for stack in config["channels"]} + + +def states(config, value="normal"): + return {stack: value for stack in config["channels"]} + + +class TopologyServiceTests(unittest.TestCase): + def test_default_factory_covers_four_topologies(self): + for topology in ("mixed_direct", "all_hbf_direct", "relay", "dash"): + config = default_config(topology) + service = TopologyService(config) + facts = service.immutable_facts() + self.assertEqual(facts["config"]["topology"], topology) + self.assertEqual(facts["buffer_semantics"], + "FINITE_TWO_BANK_CONTINUOUS_TURNOVER_NOT_WINDOW_CAPACITY") + self.assertTrue(all(len(row) == 16 for row in config["channels"].values())) + + def test_direct_conservation_and_large_continuous_bank_turnover(self): + config = default_config("all_hbf_direct") + service = TopologyService(config) + amount = 40 * 1024 * 1024 + result = service.advance(0, W, {"hbf0": {"0": amount}}, + budgets(config), states(config)) + self.assertEqual(result["served_by_stack_channel"]["hbf0"]["0"], amount) + self.assertTrue(result["stacks"]["hbf0"]["cumulative_conserved"]) + buffer = result["buffers"][0] + self.assertGreater(buffer["turnovers"], 1) + self.assertLessEqual(buffer["maximum_occupancy_bytes"], + buffer["bank_count"] * buffer["bank_capacity_bytes"]) + + def test_relay_joint_endpoint_gate_and_light_budget_applied_once(self): + config = default_config("relay") + service = TopologyService(config) + b = budgets(config) + b["hbf0"] = 1_000 + b["hbm0"] = 600 + s = states(config) + s["hbf0"] = "light" + s["hbm0"] = "light" + first = service.advance(0, W, {"hbf0": {"0": 2_000}}, b, s) + self.assertEqual(first["served_by_stack_channel"]["hbf0"]["0"], 600) + self.assertEqual(first["resources"]["endpoint:hbm0"]["used_work_units_scaled"], 600) + + s["hbm0"] = "shutdown" + second = service.advance(W, 2 * W, {"hbf0": {"1": 100}}, b, s) + self.assertEqual(second["served_by_stack_channel"]["hbf0"]["1"], 0) + self.assertTrue(any("hbm0:shutdown" in row["reasons"] for row in second["blocked"])) + + def test_hbm_local_and_relay_share_gpu_link(self): + config = default_config("relay") + config["fabric"]["hbm"]["hbm0"]["gpu_link"]["bandwidth_bytes_per_s"] = 1_000 + service = TopologyService(config) + result = service.advance( + 0, W, + {"hbf0": {"0": 100}, "hbm0": {"0": 100}}, + budgets(config), states(config), + ) + # 20 bytes fit on the shared 1 kB/s HBM GPU link in 20 ms. + self.assertEqual( + result["resources"]["hbm0:gpu-link"]["used_work_units_scaled"], 19 + ) + total = (result["served_by_stack_channel"]["hbf0"]["0"] + + result["served_by_stack_channel"]["hbm0"]["0"]) + self.assertEqual(total, 19) + self.assertGreater(result["served_by_stack_channel"]["hbf0"]["0"], 0) + self.assertGreater(result["served_by_stack_channel"]["hbm0"]["0"], 0) + + def test_dash_same_channel_two_ready_blocks_use_unique_routes(self): + config = default_config("dash") + service = TopologyService(config) + jobs = [ + {"job_id": "d", "stack": "hbf0", "channel": "0", "operation": "read", + "route": "direct", "bytes": 1000, "arrival_ns": 0}, + {"job_id": "r", "stack": "hbf0", "channel": "0", "operation": "read", + "route": "relay", "bytes": 1000, "arrival_ns": 0}, + ] + result = service.advance(0, W, {}, budgets(config), states(config), jobs) + self.assertEqual(set(result["completion_ids"]), {"d", "r"}) + phases = {(row["operation"], row["phase"]) for row in result["activities"]} + self.assertIn(("read", "direct_gpu_link"), phases) + self.assertIn(("read", "relay_send"), phases) + progress = {row["job_id"]: row for row in result["job_progress"]} + self.assertEqual(progress["d"]["remaining_bytes"], 0) + self.assertEqual(progress["r"]["remaining_bytes"], 0) + + def test_partial_job_completion_id_is_unique(self): + config = default_config("mixed_direct") + config["channels"]["hbf0"]["0"] = 1_000 + service = TopologyService(config) + job = {"job_id": "causal:1", "stack": "hbf0", "channel": "0", + "operation": "read", "route": "direct", "bytes": 30, "arrival_ns": 0} + first = service.advance(0, W, {}, budgets(config), states(config), [job]) + self.assertNotIn("causal:1", first["completion_ids"]) + row = next(row for row in first["job_progress"] if row["job_id"] == "causal:1") + self.assertGreater(row["remaining_bytes"], 0) + second = service.advance(W, 2 * W, {}, budgets(config), states(config)) + self.assertIn("causal:1", second["completion_ids"]) + third = service.advance(2 * W, 3 * W, {}, budgets(config), states(config)) + self.assertNotIn("causal:1", third["completion_ids"]) + + def test_maintenance_severe_exempt_shutdown_blocked_then_recovers(self): + config = default_config("mixed_direct") + service = TopologyService(config) + severe = states(config) + severe["hbf0"] = "severe" + jobs = [ + {"job_id": "fg", "stack": "hbf0", "channel": "0", "operation": "read", + "route": "direct", "bytes": 10, "arrival_ns": 0}, + {"job_id": "m", "maintenance_id": "maint:m", "stack": "hbf0", "channel": "1", + "operation": "program", "bytes": 10, "arrival_ns": 0}, + ] + first = service.advance(0, W, {}, budgets(config), severe, jobs) + self.assertEqual(next(row for row in first["job_progress"] if row["job_id"] == "fg")["admitted_this_window_bytes"], 0) + self.assertIn("maint:m", first["maintenance_completion_ids"]) + + shutdown = states(config) + shutdown["hbf0"] = "shutdown" + blocked_job = {"job_id": "blocked-maint", "maintenance_id": "maint:blocked", + "stack": "hbf0", "channel": "2", "operation": "erase", + "bytes": 10, "arrival_ns": W} + second = service.advance(W, 2 * W, {}, budgets(config), shutdown, [blocked_job]) + self.assertNotIn("maint:blocked", second["maintenance_completion_ids"]) + normal = states(config) + third = service.advance(2 * W, 3 * W, {}, budgets(config), normal) + self.assertIn("maint:blocked", third["maintenance_completion_ids"]) + + def test_blocked_foreground_does_not_head_of_line_block_maintenance(self): + config = default_config("mixed_direct") + service = TopologyService(config) + severe = states(config) + severe["hbf0"] = "severe" + jobs = [ + {"job_id": "fg", "stack": "hbf0", "channel": "0", "operation": "read", + "route": "direct", "bytes": 10, "arrival_ns": 0}, + {"job_id": "maint", "maintenance_id": "maint:hol", "stack": "hbf0", + "channel": "0", "operation": "program", "bytes": 10, "arrival_ns": 0}, + ] + result = service.advance(0, W, {}, budgets(config), severe, jobs) + progress = {row["job_id"]: row for row in result["job_progress"]} + self.assertEqual(progress["fg"]["admitted_this_window_bytes"], 0) + self.assertEqual(progress["maint"]["remaining_bytes"], 0) + self.assertIn("maint:hol", result["maintenance_completion_ids"]) + + def test_dash_blocked_relay_does_not_block_direct_same_channel(self): + config = default_config("dash") + service = TopologyService(config) + endpoint_states = states(config) + endpoint_states["hbm0"] = "shutdown" + jobs = [ + {"job_id": "relay", "stack": "hbf0", "channel": "0", "operation": "read", + "route": "relay", "bytes": 10, "arrival_ns": 0}, + {"job_id": "direct", "stack": "hbf0", "channel": "0", "operation": "read", + "route": "direct", "bytes": 10, "arrival_ns": 0}, + ] + result = service.advance(0, W, {}, budgets(config), endpoint_states, jobs) + self.assertIn("direct", result["completion_ids"]) + self.assertNotIn("relay", result["completion_ids"]) + self.assertTrue(any(row["job_id"] == "relay" for row in result["blocked"])) + + def test_operation_media_cost_separates_payload_from_work(self): + config = default_config("mixed_direct") + config["channels"]["hbf0"]["0"] = 1_000 + config["channels"]["hbf0"]["1"] = 1_000 + config["operation_media_cost"]["program"] = {"numerator": 10, "denominator": 1} + service = TopologyService(config) + jobs = [ + {"job_id": "read", "stack": "hbf0", "channel": "0", "operation": "read", + "route": "direct", "bytes": 100, "arrival_ns": 0}, + {"job_id": "program", "maintenance_id": "maint:p", "stack": "hbf0", + "channel": "1", "operation": "program", "bytes": 100, "arrival_ns": 0}, + ] + result = service.advance(0, W, {}, budgets(config), states(config), jobs) + progress = {row["job_id"]: row for row in result["job_progress"]} + self.assertGreater(progress["read"]["admitted_this_window_bytes"], + progress["program"]["admitted_this_window_bytes"]) + resource = result["resources"]["hbf0:channel:0:media"] + self.assertLessEqual(resource["used_work_units_scaled"], + resource["capacity_work_units_scaled"]) + self.assertTrue(any(row["phase"] == "media_program" for row in result["activities"])) + + def test_retry_consumes_physical_resources_but_not_useful_delivery(self): + config = default_config("mixed_direct") + service = TopologyService(config) + retry = {"job_id": "retry", "stack": "hbf0", "channel": "0", + "operation": "retry", "route": "direct", "bytes": 4096, + "arrival_ns": 0} + result = service.advance(0, W, {}, budgets(config), states(config), [retry]) + self.assertIn("retry", result["completion_ids"]) + self.assertEqual(result["served_by_stack_channel"]["hbf0"]["0"], 0) + self.assertEqual(result["stacks"]["hbf0"]["delivered_effective_bytes"], 0) + self.assertTrue(any(row["operation"] == "retry" and row["phase"] == "media_read" + for row in result["activities"])) + self.assertTrue(any(row["operation"] == "retry" and row["phase"] == "direct_gpu_link" + for row in result["activities"])) + + def test_program_proxy_ratio_and_single_chunk_basic_fabric_crosscheck(self): + self.assertEqual( + media_cost_from_service_rate(96_000_000_000, 16 * 40_960_000), + {"numerator": 9375, "denominator": 64}, + ) + config = default_config("relay") + service = TopologyService(config) + job = {"job_id": "relay", "stack": "hbf0", "channel": "0", + "operation": "read", "route": "relay", "bytes": 100, + "arrival_ns": 0, "metadata": {"tensor_id": "t0"}} + receipt = service.advance(0, W, {}, budgets(config), states(config), [job]) + progress = next(row for row in receipt["job_progress"] if row["job_id"] == "relay") + self.assertEqual(progress["metadata"], {"tensor_id": "t0"}) + + fabric = BasicFabric(config["fabric"]) + fabric.enqueue("relay", "hbf0", "relay", 100, 0) + fabric.advance(100) + exact = fabric.completions()[0]["completion_ns"] + # The topology service adds the explicit media phase; its remaining + # one-chunk route agrees with BasicFabric's fill/relay/drain timing. + media_ns = (100 * 1_000_000_000 + 96_000_000_000 - 1) // 96_000_000_000 + diagnostic = next(row for row in receipt["buffers"] if row["job_id"] == "relay") + self.assertEqual(diagnostic["pipeline_estimated_completion_ns"], exact + media_ns) + self.assertEqual(progress["completion_ns"], W) + + def test_completed_automatic_cohorts_are_retired(self): + config = default_config("all_hbf_direct") + service = TopologyService(config) + for index in range(20): + service.advance(index * W, (index + 1) * W, + {"hbf0": {"0": 1024}}, budgets(config), states(config)) + self.assertEqual(service._jobs, {}) + + def test_invalid_or_duplicate_identity_rejected(self): + config = default_config("dash") + service = TopologyService(config) + bad = {"job_id": "x", "stack": "hbf0", "channel": "0", "operation": "read", + "route": "invalid", "bytes": 1, "arrival_ns": 0} + with self.assertRaises(ValueError): + service.advance(0, W, {}, budgets(config), states(config), [bad]) + + +if __name__ == "__main__": + unittest.main() diff --git a/experiments/eq3_system_thermal/tiny_cpu_trace.py b/experiments/eq3_system_thermal/tiny_cpu_trace.py new file mode 100644 index 0000000..3d8b764 --- /dev/null +++ b/experiments/eq3_system_thermal/tiny_cpu_trace.py @@ -0,0 +1,305 @@ +#!/usr/bin/env python3 +"""Deterministic tiny Qwen2-style CPU forward trace; not a performance model.""" +from __future__ import annotations + +import argparse +import hashlib +import json +import os +from pathlib import Path + +for _name in ("OMP_NUM_THREADS", "OPENBLAS_NUM_THREADS", "MKL_NUM_THREADS", + "BLIS_NUM_THREADS", "VECLIB_MAXIMUM_THREADS", "NUMEXPR_NUM_THREADS"): + os.environ.setdefault(_name, "1") + +import numpy as np + + +HERE = Path(__file__).resolve().parent +CATALOG = HERE.parents[0] / "eq3_maintenance" / "sources" / "qwen2_5_weight_models.json" +ALIGNMENT = 1024 * 1024 + + +def _canonical(value) -> bytes: + return json.dumps(value, sort_keys=True, separators=(",", ":"), + allow_nan=False).encode("utf-8") + + +def _sha256(value) -> str: + data = value if isinstance(value, bytes) else _canonical(value) + return hashlib.sha256(data).hexdigest() + + +def _target_regions(meta: dict) -> list[dict]: + """Independently derive logical regions from registered official metadata.""" + a, scalar = meta["architecture"], meta["bytes_per_tensor_element"] + h, inter = a["hidden_size"], a["intermediate_size"] + heads, kv = a["num_attention_heads"], a["num_key_value_heads"] + if h % heads: + raise ValueError("target hidden size must divide attention heads") + hd, vocab = h // heads, a["vocab_size"] + regions = [("model.embed_tokens", vocab * h * scalar)] + for layer in range(a["num_hidden_layers"]): + attention = ((2 * h * h + h) + 2 * (h * kv * hd + kv * hd) + h) * scalar + mlp = (3 * h * inter + h) * scalar + regions.extend([(f"model.layers.{layer}.attention_bundle", attention), + (f"model.layers.{layer}.mlp_bundle", mlp)]) + regions.extend([("model.norm", h * scalar), ("lm_head", vocab * h * scalar)]) + address, result = 0, [] + for name, byte_count in regions: + result.append({"name": name, "logical_address_bytes": address, + "payload_bytes": byte_count}) + address = ((address + byte_count + ALIGNMENT - 1) // ALIGNMENT) * ALIGNMENT + if sum(row["payload_bytes"] for row in result) != meta["tensor_payload_bytes"]: + raise ValueError("independent region derivation disagrees with registered payload") + return result + + +def target_projection(catalog_path: Path, context_tokens: int) -> dict: + catalog = json.loads(catalog_path.read_text(encoding="utf-8")) + result = {} + for model_id, meta in catalog["models"].items(): + a = meta["architecture"] + h, heads, kv = a["hidden_size"], a["num_attention_heads"], a["num_key_value_heads"] + hd, inter, layers = h // heads, a["intermediate_size"], a["num_hidden_layers"] + linear_macs_layer = 2 * h * h + 2 * h * kv * hd + 3 * h * inter + attention_macs_layer = 2 * h * context_tokens + regions = _target_regions(meta) + result[model_id] = { + "source_resolved_commit": meta["resolved_commit"], + "tensor_payload_bytes": meta["tensor_payload_bytes"], + "logical_regions": regions, + "logical_region_address_semantics": ( + "DERIVED_1MIB_ALIGNED_SCENARIO_NOT_SAFETENSORS_FILE_OFFSETS" + ), + "analytical_macs_per_token_at_context": ( + layers * (linear_macs_layer + attention_macs_layer) + ), + "compute_cost_semantics": ( + "ANALYTICAL_DENSE_MAC_COUNT_EXCLUDES_NORMS_ROPE_SOFTMAX_AND_RUNTIME" + ), + "context_tokens": context_tokens, + } + return result + + +class TinyDecoder: + def __init__(self, config: dict): + self.config = config + self.rng = np.random.default_rng(config["seed"]) + self.dtype = np.float32 + self.weights = {} + self.operations = [] + self.accesses = [] + self._op_sequence = 0 + self._access_sequence = 0 + self.k_cache = [[] for _ in range(config["layers"])] + self.v_cache = [[] for _ in range(config["layers"])] + self._build_weights() + + def _weight(self, name, shape, scale=.02): + value = self.rng.normal(0, scale, shape).astype(self.dtype) + self.weights[name] = value + return value + + def _build_weights(self): + c, h, kvh, hd, inter = (self.config, self.config["hidden_size"], + self.config["num_key_value_heads"], + self.config["head_dim"], self.config["intermediate_size"]) + self._weight("model.embed_tokens.weight", (c["vocab_size"], h)) + for layer in range(c["layers"]): + p = f"model.layers.{layer}" + self.weights[p + ".input_layernorm.weight"] = np.ones(h, dtype=self.dtype) + self._weight(p + ".self_attn.q_proj.weight", (h, h)) + self._weight(p + ".self_attn.q_proj.bias", (h,)) + self._weight(p + ".self_attn.k_proj.weight", (kvh * hd, h)) + self._weight(p + ".self_attn.k_proj.bias", (kvh * hd,)) + self._weight(p + ".self_attn.v_proj.weight", (kvh * hd, h)) + self._weight(p + ".self_attn.v_proj.bias", (kvh * hd,)) + self._weight(p + ".self_attn.o_proj.weight", (h, h)) + self.weights[p + ".post_attention_layernorm.weight"] = np.ones(h, dtype=self.dtype) + self._weight(p + ".mlp.gate_proj.weight", (inter, h)) + self._weight(p + ".mlp.up_proj.weight", (inter, h)) + self._weight(p + ".mlp.down_proj.weight", (h, inter)) + self.weights["model.norm.weight"] = np.ones(h, dtype=self.dtype) + self._weight("lm_head.weight", (c["vocab_size"], h)) + + def _operation(self, forward_id, name, depends_on, details=None): + op_id = f"{forward_id}:op{self._op_sequence}:{name}" + self._op_sequence += 1 + self.operations.append({"sequence": len(self.operations), "op_id": op_id, + "forward_id": forward_id, "name": name, + "depends_on": list(depends_on), "details": details or {}}) + return op_id + + def _access(self, forward_id, op_id, name, access_shape=None): + value = self.weights[name] + shape = tuple(value.shape if access_shape is None else access_shape) + byte_count = int(np.prod(shape, dtype=np.int64)) * value.dtype.itemsize + self.accesses.append({ + "sequence": self._access_sequence, "forward_id": forward_id, + "op_id": op_id, "weight_name": name, + "storage_shape": list(value.shape), "access_shape": list(shape), + "access_bytes": byte_count, "storage_dtype": str(value.dtype), + "target_projection_dtype_bytes": 2, + "semantics": "ACTUAL_NUMPY_ARRAY_ACCESS_BY_TINY_FORWARD", + }) + self._access_sequence += 1 + return value + + def _rmsnorm(self, x, weight, eps): + return x * np.reciprocal(np.sqrt(np.mean(x * x, axis=-1, keepdims=True) + eps)) * weight + + @staticmethod + def _silu(x): + return x / (1.0 + np.exp(-x)) + + def _rope(self, x, positions): + hd = x.shape[-1] + inv = 1.0 / (self.config["rope_theta"] ** (np.arange(0, hd, 2) / hd)) + angles = positions[:, None] * inv[None, :] + cos, sin = np.cos(angles)[:, None, :], np.sin(angles)[:, None, :] + even, odd = x[..., 0::2], x[..., 1::2] + result = np.empty_like(x) + result[..., 0::2] = even * cos - odd * sin + result[..., 1::2] = even * sin + odd * cos + return result + + def forward(self, token_ids, start_position: int, forward_id: str): + c, ids = self.config, np.asarray(token_ids, dtype=np.int64) + embedding_op = self._operation(forward_id, "embedding_lookup", ["token_ids"], + {"token_count": int(ids.size)}) + table = self._access(forward_id, embedding_op, "model.embed_tokens.weight", + (ids.size, c["hidden_size"])) + x = table[ids].copy() + previous = embedding_op + positions = np.arange(start_position, start_position + ids.size) + for layer in range(c["layers"]): + p = f"model.layers.{layer}" + norm_op = self._operation(forward_id, f"layer{layer}.input_rmsnorm", [previous]) + norm = self._access(forward_id, norm_op, p + ".input_layernorm.weight") + n = self._rmsnorm(x, norm, c["rms_norm_eps"]) + projections = [] + for kind in ("q", "k", "v"): + op = self._operation(forward_id, f"layer{layer}.{kind}_projection", [norm_op]) + weight = self._access(forward_id, op, p + f".self_attn.{kind}_proj.weight") + bias = self._access(forward_id, op, p + f".self_attn.{kind}_proj.bias") + projections.append((n @ weight.T + bias, op)) + q_raw, q_op = projections[0] + k_raw, k_op = projections[1] + v_raw, v_op = projections[2] + q = self._rope(q_raw.reshape(ids.size, c["num_attention_heads"], c["head_dim"]), + positions) + k = self._rope(k_raw.reshape(ids.size, c["num_key_value_heads"], c["head_dim"]), + positions) + v = v_raw.reshape(ids.size, c["num_key_value_heads"], c["head_dim"]) + self.k_cache[layer].append(k) + self.v_cache[layer].append(v) + all_k = np.concatenate(self.k_cache[layer], axis=0) + all_v = np.concatenate(self.v_cache[layer], axis=0) + repeats = c["num_attention_heads"] // c["num_key_value_heads"] + expanded_k = np.repeat(all_k, repeats, axis=1) + expanded_v = np.repeat(all_v, repeats, axis=1) + attention_op = self._operation( + forward_id, f"layer{layer}.rope_gqa_causal_attention", [q_op, k_op, v_op], + {"q_shape": list(q.shape), "kv_shape": list(k.shape), + "context_tokens": int(all_k.shape[0]), "kv_repeat_groups": repeats}) + scores = np.einsum("thd,shd->ths", q, expanded_k) / np.sqrt(c["head_dim"]) + absolute_q = positions[:, None] + absolute_k = np.arange(all_k.shape[0])[None, :] + scores = np.where(absolute_k <= absolute_q[:, :, None], scores, -1e30) + scores -= np.max(scores, axis=-1, keepdims=True) + probs = np.exp(scores) + probs /= np.sum(probs, axis=-1, keepdims=True) + context = np.einsum("ths,shd->thd", probs, expanded_v).reshape(ids.size, -1) + out_op = self._operation(forward_id, f"layer{layer}.o_projection", [attention_op]) + out_w = self._access(forward_id, out_op, p + ".self_attn.o_proj.weight") + x = x + context @ out_w.T + post_op = self._operation(forward_id, f"layer{layer}.post_attention_rmsnorm", [out_op]) + post_w = self._access(forward_id, post_op, p + ".post_attention_layernorm.weight") + post = self._rmsnorm(x, post_w, c["rms_norm_eps"]) + gate_op = self._operation(forward_id, f"layer{layer}.gate_projection", [post_op]) + gate_w = self._access(forward_id, gate_op, p + ".mlp.gate_proj.weight") + up_op = self._operation(forward_id, f"layer{layer}.up_projection", [post_op]) + up_w = self._access(forward_id, up_op, p + ".mlp.up_proj.weight") + down_op = self._operation(forward_id, f"layer{layer}.swiglu_down_projection", + [gate_op, up_op]) + down_w = self._access(forward_id, down_op, p + ".mlp.down_proj.weight") + x = x + (self._silu(post @ gate_w.T) * (post @ up_w.T)) @ down_w.T + previous = down_op + norm_op = self._operation(forward_id, "final_rmsnorm", [previous]) + norm = self._access(forward_id, norm_op, "model.norm.weight") + x = self._rmsnorm(x, norm, c["rms_norm_eps"]) + head_op = self._operation(forward_id, "lm_head", [norm_op]) + head = self._access(forward_id, head_op, "lm_head.weight") + logits = x @ head.T + return logits, head_op + + +def build_trace(config: dict, source_path: Path = Path(__file__)) -> dict: + required = {"schema_version", "seed", "vocab_size", "hidden_size", "layers", + "num_attention_heads", "num_key_value_heads", "intermediate_size", + "rms_norm_eps", "rope_theta", "prompt_token_ids", "decode_tokens", + "projection_context_tokens"} + if set(config) != required: + raise ValueError("tiny trace config has missing or unknown fields") + if config["hidden_size"] % config["num_attention_heads"]: + raise ValueError("hidden size must divide attention heads") + if config["num_attention_heads"] % config["num_key_value_heads"]: + raise ValueError("query heads must divide KV heads") + config = dict(config) + config["head_dim"] = config["hidden_size"] // config["num_attention_heads"] + if config["head_dim"] % 2: + raise ValueError("RoPE head dimension must be even") + decoder = TinyDecoder(config) + forwards, tokens = [], list(config["prompt_token_ids"]) + logits, terminal = decoder.forward(tokens, 0, "prefill") + forwards.append({"forward_id": "prefill", "input_tokens": tokens, + "context_tokens_after": len(tokens), "terminal_op_id": terminal, + "logits_sha256": hashlib.sha256(logits.tobytes()).hexdigest()}) + next_token = int(np.argmax(logits[-1])) + for index in range(config["decode_tokens"]): + forward_id = f"decode{index}" + logits, terminal = decoder.forward([next_token], len(tokens), forward_id) + tokens.append(next_token) + forwards.append({"forward_id": forward_id, "input_tokens": [next_token], + "context_tokens_after": len(tokens), "terminal_op_id": terminal, + "logits_sha256": hashlib.sha256(logits.tobytes()).hexdigest()}) + next_token = int(np.argmax(logits[-1])) + source_bytes = source_path.read_bytes() + result = { + "schema_version": "eq3-tiny-qwen2-cpu-trace-v1", + "classification": "TRACE_DERIVED_TINY_RANDOM_WEIGHT_CPU_FORWARD", + "claims_excluded": ["PRETRAINED_MODEL_QUALITY", "NATIVE_GPU_TRACE", + "GPU_TIMING", "TOKEN_PERFORMANCE_CALIBRATION"], + "numpy_version": np.__version__, "blas_threads_requested": 1, + "config": config, "config_sha256": _sha256(config), + "source_sha256": hashlib.sha256(source_bytes).hexdigest(), + "forwards": forwards, "generated_tokens": tokens, + "operations": decoder.operations, "weight_accesses": decoder.accesses, + "actual_cpu_weight_storage_bytes": sum(x.nbytes for x in decoder.weights.values()), + "actual_access_bytes": sum(x["access_bytes"] for x in decoder.accesses), + "target_projection": target_projection(CATALOG, config["projection_context_tokens"]), + "target_projection_source": str(CATALOG.relative_to(HERE.parents[1])), + } + result["trace_sha256"] = _sha256(result) + return result + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--config", type=Path, required=True) + parser.add_argument("--output", type=Path, required=True) + args = parser.parse_args() + config = json.loads(args.config.read_text(encoding="utf-8")) + result = build_trace(config) + args.output.parent.mkdir(parents=True, exist_ok=True) + args.output.write_text(json.dumps(result, indent=2, sort_keys=True) + "\n", + encoding="utf-8") + print(json.dumps({"output": str(args.output), "trace_sha256": result["trace_sha256"], + "operation_count": len(result["operations"]), + "weight_access_count": len(result["weight_accesses"])})) + + +if __name__ == "__main__": + main() diff --git a/experiments/eq3_system_thermal/topology_model_map.json b/experiments/eq3_system_thermal/topology_model_map.json new file mode 100644 index 0000000..68ab3cc --- /dev/null +++ b/experiments/eq3_system_thermal/topology_model_map.json @@ -0,0 +1,16 @@ +{ + "schema_version": "eq3-system-thermal-topology-map-v1", + "fast_thermal_rom": "UNAVAILABLE_FOR_CURRENT_GEOMETRY", + "layouts": { + "mixed_full_2mm": { + "source_model_dir_argument": "MIXED_FULL_2MM_MODEL_DIR", + "topologies": ["Q1_mixed_direct", "Q2_4plus4_relay", "Q3_four_pair_DASH"], + "note": "same physical layout; route-specific power facts remain a separate runtime input" + }, + "all_hbf_full_2mm": { + "source_model_dir_argument": "ALL_HBF_FULL_2MM_MODEL_DIR", + "topologies": ["Q4_8HBF_direct_external_GDDR"], + "note": "external GDDR identity retained outside the package thermal domain" + } + } +} diff --git a/experiments/eq3_system_thermal/topology_service.py b/experiments/eq3_system_thermal/topology_service.py new file mode 100644 index 0000000..6d3e923 --- /dev/null +++ b/experiments/eq3_system_thermal/topology_service.py @@ -0,0 +1,810 @@ +#!/usr/bin/env python3 +"""Aggregated four-topology byte service for coupled system/thermal studies. + +The service reuses the BasicFabric route, pair, link and two-bank configuration +shape, but it does not create a request per page. It allocates integer byte +cohorts over fixed windows and reports activity facts for a separate energy +mapper. Parameters are engineering scenario assumptions, not calibrated +device timing. +""" + +from __future__ import annotations + +from collections import deque +import copy +from dataclasses import dataclass +from fractions import Fraction +import heapq +import math +from pathlib import Path +import sys +from typing import Deque, Dict, Mapping, Optional, Sequence, Tuple + + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "tools")) +from eq3_basic_fabric import BasicFabric + + +NANOSECONDS_PER_SECOND = 1_000_000_000 +WINDOW_NS = 20_000_000 +TOPOLOGIES = {"mixed_direct", "all_hbf_direct", "relay", "dash"} +OPERATIONS = {"read", "retry", "refresh_read", "program", "erase", "migration"} +READ_OPERATIONS = {"read", "retry"} +MAINTENANCE_OPERATIONS = {"refresh_read", "program", "erase", "migration"} +STATES = {"normal", "light", "severe", "shutdown"} +MAINTENANCE_POLICY = "LIGHT_SEVERE_EXEMPT_SHUTDOWN_BLOCKED" + + +def _integer(value: object, label: str, *, positive: bool = False) -> int: + if isinstance(value, bool) or not isinstance(value, int): + raise ValueError(f"{label} must be an integer") + if value < (1 if positive else 0): + raise ValueError(f"{label} must be {'positive' if positive else 'non-negative'}") + return value + + +def _exact(value: object, keys: Sequence[str], label: str) -> dict: + if not isinstance(value, dict) or set(value) != set(keys): + raise ValueError(f"{label} must contain exactly {sorted(keys)}") + return value + + +def _stage(latency_ns: int, bandwidth_Bps: int) -> dict: + return {"latency_ns": latency_ns, "bandwidth_bytes_per_s": bandwidth_Bps} + + +def media_cost_from_service_rate(read_equivalent_Bps: int, operation_payload_Bps: int) -> dict: + """Return an exact read-equivalent work ratio for an operation service rate.""" + + read_rate = _integer(read_equivalent_Bps, "read_equivalent_Bps", positive=True) + operation_rate = _integer(operation_payload_Bps, "operation_payload_Bps", positive=True) + ratio = Fraction(read_rate, operation_rate) + return {"numerator": ratio.numerator, "denominator": ratio.denominator} + + +def default_config(topology: str, stack_count: Optional[int] = None) -> dict: + """Return an explicit, uncalibrated 16-channel engineering configuration.""" + + if topology not in TOPOLOGIES: + raise ValueError("unknown topology") + expected = 8 if topology == "all_hbf_direct" else 4 + if stack_count is None: + stack_count = expected + if stack_count != expected: + raise ValueError(f"{topology} requires {expected} HBF stacks in the v1 fixture") + paired = topology in {"relay", "dash"} + hbm_count = 0 if topology == "all_hbf_direct" else 4 + link_Bps = 2_048_000_000_000 + link_latency_ns = 10 + fabric = { + "evidence": "SCENARIO_ASSUMPTION", + "hbf": { + f"hbf{i}": { + "pair": f"hbm{i}" if paired else None, + "bank_count": 2, + "bank_capacity_bytes": 8 * 1024 * 1024, + "fill": _stage(link_latency_ns, link_Bps), + "direct_link": _stage(link_latency_ns, link_Bps), + "relay_link": _stage(link_latency_ns, link_Bps) if paired else None, + } + for i in range(stack_count) + }, + "hbm": { + f"hbm{i}": { + "bank_count": 2, + "bank_capacity_bytes": 4 * 1024 * 1024, + "gpu_link": _stage(link_latency_ns, link_Bps), + } + for i in range(hbm_count) + }, + } + all_stacks = sorted(set(fabric["hbf"]) | set(fabric["hbm"])) + channels = { + stack: {str(channel): 96_000_000_000 for channel in range(16)} + for stack in all_stacks + } + dash_routes = ( + { + stack: {str(channel): ("direct" if channel % 2 == 0 else "relay") + for channel in range(16)} + for stack in fabric["hbf"] + } + if topology == "dash" else {} + ) + unit_cost = {"numerator": 1, "denominator": 1} + return { + "schema_version": "eq3-topology-fluid-config-v1", + "evidence": "SCENARIO_ASSUMPTION", + "topology": topology, + "window_ns": WINDOW_NS, + "fabric": fabric, + "channels": channels, + "dash_routes": dash_routes, + "operation_media_cost": {operation: dict(unit_cost) for operation in sorted(OPERATIONS)}, + "operation_cost_evidence": "ENGINEERING_FIXTURE_DEFAULT_ONE_FORMAL_PROFILE_MUST_OVERRIDE", + "maintenance_guard_policy": MAINTENANCE_POLICY, + } + + +def _lcm(left: int, right: int) -> int: + return left * right // math.gcd(left, right) + + +@dataclass +class _Job: + job_id: str + stack: str + channel: str + operation: str + route: Optional[str] + arrival_ns: int + total_bytes: int + admission_remaining_bytes: int + completed_bytes: int + sequence: int + foreground: bool + external: bool + maintenance_id: Optional[str] + metadata: dict + first_service_ns: Optional[int] = None + last_service_ns: Optional[int] = None + completion_ns: Optional[int] = None + pending_batches: int = 0 + + +def _normalize(config: dict) -> dict: + keys = ( + "schema_version", "evidence", "topology", "window_ns", "fabric", "channels", + "dash_routes", "operation_media_cost", "operation_cost_evidence", + "maintenance_guard_policy", + ) + config = _exact(copy.deepcopy(config), keys, "config") + if config["schema_version"] != "eq3-topology-fluid-config-v1": + raise ValueError("unsupported topology fluid config schema") + if config["evidence"] != "SCENARIO_ASSUMPTION": + raise ValueError("topology fluid parameters must remain SCENARIO_ASSUMPTION") + if config["topology"] not in TOPOLOGIES: + raise ValueError("unsupported topology") + if _integer(config["window_ns"], "window_ns", positive=True) != WINDOW_NS: + raise ValueError("v1 requires a 20 ms window") + if config["maintenance_guard_policy"] != MAINTENANCE_POLICY: + raise ValueError("maintenance guard policy differs from the preserved contract") + fabric = BasicFabric(config["fabric"]).immutable_facts()["config"] + hbf, hbm = fabric["hbf"], fabric["hbm"] + stacks = set(hbf) | set(hbm) + topology = config["topology"] + if topology == "all_hbf_direct" and hbm: + raise ValueError("all_hbf_direct cannot contain in-package HBM") + if topology in {"relay", "dash"} and (not hbm or any(row["pair"] is None for row in hbf.values())): + raise ValueError("relay and DASH require explicit HBF/HBM pairs") + if topology in {"mixed_direct", "all_hbf_direct"} and any(row["pair"] is not None for row in hbf.values()): + raise ValueError("direct topologies cannot silently configure relay pairs") + if not isinstance(config["channels"], dict) or set(config["channels"]) != stacks: + raise ValueError("channels must exactly cover configured HBF and HBM stacks") + channels = {} + for stack in sorted(stacks): + raw = config["channels"][stack] + if not isinstance(raw, dict) or not raw: + raise ValueError(f"channels.{stack} must be nonempty") + channels[stack] = {} + for channel, rate in sorted(raw.items()): + if not isinstance(channel, str) or not channel: + raise ValueError("channel IDs must be nonempty strings") + channels[stack][channel] = _integer(rate, f"channels.{stack}.{channel}", positive=True) + dash_routes = config["dash_routes"] + if topology == "dash": + if not isinstance(dash_routes, dict) or set(dash_routes) != set(hbf): + raise ValueError("DASH routes must exactly cover HBF stacks") + for stack in hbf: + if set(dash_routes[stack]) != set(channels[stack]): + raise ValueError("DASH routes must exactly cover HBF channels") + if set(dash_routes[stack].values()) - {"direct", "relay"}: + raise ValueError("DASH channel route must be direct or relay") + elif dash_routes != {}: + raise ValueError("dash_routes must be empty outside DASH topology") + costs = config["operation_media_cost"] + if not isinstance(costs, dict) or set(costs) != OPERATIONS: + raise ValueError("operation_media_cost must exactly cover supported operations") + normalized_costs = {} + scale = 1 + for operation in sorted(OPERATIONS): + row = _exact(costs[operation], ("numerator", "denominator"), f"cost.{operation}") + numerator = _integer(row["numerator"], f"cost.{operation}.numerator", positive=True) + denominator = _integer(row["denominator"], f"cost.{operation}.denominator", positive=True) + value = Fraction(numerator, denominator) + normalized_costs[operation] = value + scale = _lcm(scale, value.denominator) + if scale > 1_000_000: + raise ValueError("operation media-cost denominator scale is too large") + config["fabric"] = fabric + config["channels"] = channels + config["operation_media_cost_fraction"] = normalized_costs + config["work_scale"] = scale + return config + + +class TopologyService: + """Persistent, window-quantized aggregate topology service.""" + + schema_version = "eq3-topology-fluid-receipt-v1" + + def __init__(self, config: dict): + self._config = _normalize(config) + self._now = 0 + self._sequence = 0 + self._auto_id = 0 + self._jobs: Dict[str, _Job] = {} + self._known_ids = set() + self._external_active = set() + self._queues: Dict[Tuple[str, str], Deque[str]] = { + (stack, channel): deque() + for stack, channels in self._config["channels"].items() + for channel in channels + } + self._inflight: list[Tuple[int, int, str, int]] = [] + self._completion_ids_seen = set() + self._maintenance_ids_seen = set() + self._cumulative_offered = {key: 0 for key in self._queues} + self._cumulative_completed = {key: 0 for key in self._queues} + self._foreground_backlog = {key: 0 for key in self._queues} + + @property + def now_ns(self) -> int: + return self._now + + def immutable_facts(self) -> dict: + public = copy.deepcopy(self._config) + public.pop("operation_media_cost_fraction") + return { + "schema_version": "eq3-topology-fluid-facts-v1", + "enabled_by_default": False, + "config": public, + "service_semantics": "WINDOW_QUANTIZED_AGGREGATED_BYTE_COHORTS", + "buffer_semantics": "FINITE_TWO_BANK_CONTINUOUS_TURNOVER_NOT_WINDOW_CAPACITY", + "pipeline_semantics": "FIRST_CHUNK_STARTUP_PLUS_BOTTLENECK_TURNOVER_APPROXIMATION", + "latency_semantics": "AGGREGATED_PIPELINE_ESTIMATE_WINDOW_COMPLETION_FLOOR", + "backend_latency": "UNKNOWN", + "maintenance_guard_policy": MAINTENANCE_POLICY, + "light_budget_semantics": "CALLER_ALREADY_CAPPED_APPLIED_EXACTLY_ONCE", + } + + def _default_route(self, stack: str, channel: str) -> Optional[str]: + topology = self._config["topology"] + if stack in self._config["fabric"]["hbm"]: + return "direct" + if topology in {"mixed_direct", "all_hbf_direct"}: + return "direct" + if topology == "relay": + return "relay" + return self._config["dash_routes"][stack][channel] + + def _validate_route(self, stack: str, channel: str, route: Optional[str]) -> Optional[str]: + expected = self._default_route(stack, channel) + if stack in self._config["fabric"]["hbm"] and route != "direct": + raise ValueError("HBM local work must use direct route") + topology = self._config["topology"] + if stack in self._config["fabric"]["hbf"]: + if topology == "dash": + if route not in {"direct", "relay"}: + raise ValueError("DASH HBF work must use direct or relay route") + elif route != expected: + raise ValueError("route differs from topology") + return route + + def _add_job(self, *, job_id: str, stack: str, channel: str, operation: str, + route: Optional[str], byte_count: int, arrival_ns: int, + foreground: bool, external: bool, maintenance_id: Optional[str], + metadata: Optional[dict] = None) -> None: + if not isinstance(job_id, str) or not job_id or job_id in self._known_ids: + raise ValueError("job_id must be a new nonempty string") + if stack not in self._config["channels"] or channel not in self._config["channels"][stack]: + raise ValueError("job names an unknown stack/channel") + if operation not in OPERATIONS: + raise ValueError("unsupported operation") + route = self._validate_route(stack, channel, route) + if operation not in READ_OPERATIONS and route is not None: + # Internal operations have no package delivery route. + route = None + if maintenance_id is not None and (not isinstance(maintenance_id, str) or not maintenance_id): + raise ValueError("maintenance_id must be a nonempty string") + if maintenance_id is not None and maintenance_id in self._maintenance_ids_seen: + raise ValueError("maintenance_id was already completed") + if metadata is None: + metadata = {} + if not isinstance(metadata, dict): + raise ValueError("metadata must be an object") + job = _Job( + job_id=job_id, stack=stack, channel=channel, operation=operation, route=route, + arrival_ns=arrival_ns, total_bytes=byte_count, + admission_remaining_bytes=byte_count, completed_bytes=0, + sequence=self._sequence, foreground=foreground, external=external, + maintenance_id=maintenance_id, + metadata=copy.deepcopy(metadata), + ) + self._sequence += 1 + self._jobs[job_id] = job + self._known_ids.add(job_id) + if external: + self._external_active.add(job_id) + self._queues[(stack, channel)].append(job_id) + if foreground: + self._cumulative_offered[(stack, channel)] += byte_count + self._foreground_backlog[(stack, channel)] += byte_count + + def _endpoints(self, job: _Job) -> Tuple[str, ...]: + if job.route != "relay": + return (job.stack,) + pair = self._config["fabric"]["hbf"][job.stack]["pair"] + return tuple(dict.fromkeys((job.stack, pair))) + + def _phase_resources(self, job: _Job) -> list[Tuple[str, dict, int]]: + scale = self._config["work_scale"] + media_rate = self._config["channels"][job.stack][job.channel] + media_cost = self._config["operation_media_cost_fraction"][job.operation] + media_coeff = media_cost.numerator * (scale // media_cost.denominator) + phases = [(f"{job.stack}:channel:{job.channel}:media", _stage(0, media_rate), media_coeff)] + if job.operation not in READ_OPERATIONS: + return phases + fabric = self._config["fabric"] + if job.stack in fabric["hbm"]: + phases.append((f"{job.stack}:gpu-link", fabric["hbm"][job.stack]["gpu_link"], scale)) + return phases + row = fabric["hbf"][job.stack] + phases.append((f"{job.stack}:fill", row["fill"], scale)) + if job.route == "direct": + phases.append((f"{job.stack}:gpu-link", row["direct_link"], scale)) + else: + pair = row["pair"] + phases.append((f"{job.stack}->{pair}:relay-link", row["relay_link"], scale)) + phases.append((f"{pair}:gpu-link", fabric["hbm"][pair]["gpu_link"], scale)) + return phases + + def _buffer_capacity(self, job: _Job) -> int: + fabric = self._config["fabric"] + capacities = [] + if job.stack in fabric["hbf"]: + capacities.append(fabric["hbf"][job.stack]["bank_capacity_bytes"]) + if job.route == "relay": + pair = fabric["hbf"][job.stack]["pair"] + capacities.append(fabric["hbm"][pair]["bank_capacity_bytes"]) + elif job.stack in fabric["hbm"]: + capacities.append(fabric["hbm"][job.stack]["bank_capacity_bytes"]) + return min(capacities) if capacities else job.total_bytes + + def _buffer_specs(self, job: _Job) -> list[Tuple[str, int, int]]: + fabric = self._config["fabric"] + if job.stack in fabric["hbf"]: + row = fabric["hbf"][job.stack] + result = [(job.stack, row["bank_count"], row["bank_capacity_bytes"])] + if job.route == "relay": + partner = row["pair"] + partner_row = fabric["hbm"][partner] + result.append((partner, partner_row["bank_count"], + partner_row["bank_capacity_bytes"])) + return result + row = fabric["hbm"][job.stack] + return [(job.stack, row["bank_count"], row["bank_capacity_bytes"])] + + def _pipeline_completion(self, job: _Job, byte_count: int, service_start_ns: int) -> int: + phases = self._phase_resources(job) + scale = self._config["work_scale"] + chunk = min(byte_count, self._buffer_capacity(job)) + + def duration(stage: dict, coefficient: int, amount: int) -> int: + work = amount * coefficient + transfer = (work * NANOSECONDS_PER_SECOND + stage["bandwidth_bytes_per_s"] * scale - 1) // ( + stage["bandwidth_bytes_per_s"] * scale + ) + return stage["latency_ns"] + transfer + + first_chunk = [duration(stage, coefficient, chunk) for _, stage, coefficient in phases] + if byte_count <= chunk: + return service_start_ns + sum(first_chunk) + turns = (byte_count + chunk - 1) // chunk + # Two-bank ping-pong permits adjacent phases to overlap after the first + # chunk. The slowest phase sets continuous turnover; latency is not + # unconditionally added for every byte or every phase turn. + return service_start_ns + sum(first_chunk) + (turns - 1) * max(first_chunk) + + def _activity_rows(self, job: _Job, byte_count: int, start_ns: int, end_ns: int) -> list[dict]: + base = { + "operation": job.operation, "stack": job.stack, "channel": job.channel, + "bytes": byte_count, "route": job.route, "start_ns": start_ns, "end_ns": end_ns, + } + rows = [] + + def add(phase: str, resource: str, *, stack: Optional[str] = None, + partner: Optional[str] = None, operation: Optional[str] = None) -> None: + row = dict(base) + row.update(phase=phase, resource=resource) + if stack is not None: + row["stack"] = stack + if partner is not None: + row["partner"] = partner + if operation is not None: + row["operation"] = operation + rows.append(row) + + media = f"{job.stack}:channel:{job.channel}:media" + if job.operation == "migration": + add("media_read", media, operation="migration_read") + add("media_program", media, operation="migration_program") + elif job.operation in {"read", "retry", "refresh_read"}: + add("media_read", media) + elif job.operation == "program": + add("media_program", media) + else: + add("media_erase", media) + add("source_base", f"{job.stack}:base") + if job.operation not in READ_OPERATIONS: + return rows + fabric = self._config["fabric"] + if job.stack in fabric["hbm"]: + add("gpu_drain", f"{job.stack}:gpu-link") + elif job.route == "direct": + add("direct_gpu_link", f"{job.stack}:gpu-link") + else: + pair = fabric["hbf"][job.stack]["pair"] + add("relay_send", f"{job.stack}->{pair}:relay-link", partner=pair) + add("relay_receive", f"{pair}:base", stack=pair, partner=pair) + add("partner_gpu_drain", f"{pair}:gpu-link", stack=pair, partner=pair) + return rows + + def _complete_batches(self, horizon_ns: int, completed_by_channel: dict, + completed_by_job: dict, delay_samples: dict, + completion_ids: list[str], maintenance_ids: list[str]) -> None: + while self._inflight and self._inflight[0][0] <= horizon_ns: + completion_ns, _, job_id, byte_count = heapq.heappop(self._inflight) + job = self._jobs[job_id] + job.pending_batches -= 1 + job.completed_bytes += byte_count + completed_by_job[job_id] = completed_by_job.get(job_id, 0) + byte_count + if job.foreground: + key = (job.stack, job.channel) + completed_by_channel[key] += byte_count + self._cumulative_completed[key] += byte_count + self._foreground_backlog[key] -= byte_count + delay_samples[job.stack].append((completion_ns - job.arrival_ns, byte_count)) + if job.admission_remaining_bytes == 0 and job.pending_batches == 0: + job.completion_ns = completion_ns + if job.external and job.job_id not in self._completion_ids_seen: + completion_ids.append(job.job_id) + self._completion_ids_seen.add(job.job_id) + if job.maintenance_id is not None and job.maintenance_id not in self._maintenance_ids_seen: + maintenance_ids.append(job.maintenance_id) + self._maintenance_ids_seen.add(job.maintenance_id) + + @staticmethod + def _histogram(samples: list[Tuple[int, int]]) -> list[dict]: + totals = {} + for delay, byte_count in samples: + totals[delay] = totals.get(delay, 0) + byte_count + return [{"delay_ns": delay, "bytes": totals[delay]} for delay in sorted(totals)] + + def advance(self, start_ns: int, end_ns: int, + offered_by_stack_channel: Mapping[str, Mapping[str, int]], + budgets: Mapping[str, int], endpoint_states: Mapping[str, str], + extra_jobs: Sequence[dict] = ()) -> dict: + start_ns = _integer(start_ns, "start_ns") + end_ns = _integer(end_ns, "end_ns") + if start_ns != self._now or end_ns - start_ns != self._config["window_ns"]: + raise ValueError("advance requires the next contiguous fixed window") + stacks = set(self._config["channels"]) + if set(budgets) != stacks or set(endpoint_states) != stacks: + raise ValueError("budgets and endpoint_states must exactly cover configured stacks") + checked_budgets = {stack: _integer(budgets[stack], f"budgets.{stack}") for stack in stacks} + if any(endpoint_states[stack] not in STATES for stack in stacks): + raise ValueError("unknown endpoint state") + unknown_stacks = set(offered_by_stack_channel) - stacks + if unknown_stacks: + raise ValueError(f"unknown offered stacks: {sorted(unknown_stacks)}") + + offered_this_window = {key: 0 for key in self._queues} + for stack, raw_channels in offered_by_stack_channel.items(): + unknown_channels = set(raw_channels) - set(self._config["channels"][stack]) + if unknown_channels: + raise ValueError(f"unknown channels for {stack}: {sorted(unknown_channels)}") + for channel, raw_bytes in raw_channels.items(): + byte_count = _integer(raw_bytes, f"offered.{stack}.{channel}") + if byte_count: + job_id = f"foreground:{self._auto_id}" + self._auto_id += 1 + self._add_job( + job_id=job_id, stack=stack, channel=channel, operation="read", + route=self._default_route(stack, channel), byte_count=byte_count, + arrival_ns=start_ns, foreground=True, external=False, maintenance_id=None, + metadata={}, + ) + offered_this_window[(stack, channel)] += byte_count + + for raw in extra_jobs: + required = {"job_id", "stack", "channel", "operation", "bytes", "arrival_ns"} + optional = {"route", "maintenance_id", "metadata"} + if not isinstance(raw, dict) or not required.issubset(raw) or set(raw) - required - optional: + raise ValueError("extra job has missing or unknown fields") + arrival = _integer(raw["arrival_ns"], "extra_job.arrival_ns") + if arrival < start_ns or arrival >= end_ns: + raise ValueError("extra job arrival must lie in the current half-open window") + operation = raw["operation"] + route = raw.get("route", self._default_route(raw["stack"], raw["channel"])) + maintenance_id = raw.get("maintenance_id") + # Retry traffic consumes the same media/link resources but is not + # a new useful byte delivery. A later identified successful read + # must carry the useful completion explicitly. + foreground = operation == "read" and maintenance_id is None + byte_count = _integer(raw["bytes"], "extra_job.bytes", positive=True) + self._add_job( + job_id=raw["job_id"], stack=raw["stack"], channel=raw["channel"], + operation=operation, route=route, byte_count=byte_count, + arrival_ns=arrival, foreground=foreground, external=True, + maintenance_id=maintenance_id, + metadata=raw.get("metadata", {}), + ) + if foreground: + offered_this_window[(raw["stack"], raw["channel"])] += byte_count + + scale = self._config["work_scale"] + resource_specs = {} + for stack, channels in self._config["channels"].items(): + for channel, rate in channels.items(): + resource_specs[f"{stack}:channel:{channel}:media"] = _stage(0, rate) + fabric = self._config["fabric"] + for stack, row in fabric["hbf"].items(): + resource_specs[f"{stack}:fill"] = row["fill"] + resource_specs[f"{stack}:gpu-link"] = row["direct_link"] + if row["pair"] is not None: + resource_specs[f"{stack}->{row['pair']}:relay-link"] = row["relay_link"] + for stack, row in fabric["hbm"].items(): + resource_specs[f"{stack}:gpu-link"] = row["gpu_link"] + residual = { + resource: max(0, (end_ns - start_ns - spec["latency_ns"])) + * spec["bandwidth_bytes_per_s"] * scale // NANOSECONDS_PER_SECOND + for resource, spec in resource_specs.items() + } + for stack in stacks: + residual[f"endpoint:{stack}"] = checked_budgets[stack] * scale + initial_capacity = dict(residual) + allocations: Dict[str, int] = {} + blocked = [] + blocked_seen = set() + + def scheduling_class(job: _Job) -> str: + maintenance = job.operation in MAINTENANCE_OPERATIONS or job.maintenance_id is not None + if maintenance: + return f"maintenance:{job.operation}" + return f"foreground:{job.route}" + + def candidate() -> list[Tuple[_Job, Dict[str, int]]]: + result = [] + for key in sorted(self._queues): + queue = self._queues[key] + if queue: + self._queues[key] = queue = deque( + job_id for job_id in queue + if self._jobs[job_id].admission_remaining_bytes > 0 + ) + if not queue: + continue + first_by_class = {} + for job_id in queue: + job = self._jobs[job_id] + first_by_class.setdefault(scheduling_class(job), job) + for job in sorted(first_by_class.values(), key=lambda row: row.sequence): + if job.arrival_ns >= end_ns: + continue + endpoints = self._endpoints(job) + maintenance = job.operation in MAINTENANCE_OPERATIONS or job.maintenance_id is not None + states = {endpoint: endpoint_states[endpoint] for endpoint in endpoints} + reasons = [] + if maintenance: + reasons = [f"{endpoint}:shutdown" for endpoint, state in states.items() + if state == "shutdown"] + else: + reasons = [f"{endpoint}:{state}" for endpoint, state in states.items() + if state in {"severe", "shutdown"}] + if reasons: + if job.job_id not in blocked_seen: + blocked.append({"job_id": job.job_id, "bytes": job.admission_remaining_bytes, + "reasons": reasons, "maintenance": maintenance}) + blocked_seen.add(job.job_id) + continue + coefficients = {resource: coefficient for resource, _, coefficient + in self._phase_resources(job)} + if not maintenance: + for endpoint in endpoints: + coefficients[f"endpoint:{endpoint}"] = scale + result.append((job, coefficients)) + return result + + while True: + active = [] + for job, coefficients in candidate(): + if all(residual.get(resource, 0) >= coefficient + for resource, coefficient in coefficients.items()): + active.append((job, coefficients)) + if not active: + break + bounds = [job.admission_remaining_bytes for job, _ in active] + for resource in residual: + users = sum(coefficients.get(resource, 0) for _, coefficients in active) + if users: + bounds.append(residual[resource] // users) + share = min(bounds) + if share <= 0: + progress = False + for job, coefficients in active: + if all(residual[resource] >= coefficient + for resource, coefficient in coefficients.items()): + for resource, coefficient in coefficients.items(): + residual[resource] -= coefficient + job.admission_remaining_bytes -= 1 + allocations[job.job_id] = allocations.get(job.job_id, 0) + 1 + progress = True + if not progress: + break + continue + for job, coefficients in active: + amount = min(share, job.admission_remaining_bytes) + for resource, coefficient in coefficients.items(): + residual[resource] -= amount * coefficient + job.admission_remaining_bytes -= amount + allocations[job.job_id] = allocations.get(job.job_id, 0) + amount + + activities = [] + buffer_rows = [] + for job_id, amount in sorted(allocations.items(), key=lambda item: self._jobs[item[0]].sequence): + job = self._jobs[job_id] + service_start = max(start_ns, job.arrival_ns) + job.first_service_ns = service_start if job.first_service_ns is None else job.first_service_ns + job.last_service_ns = end_ns + pipeline_estimated_completion_ns = self._pipeline_completion( + job, amount, service_start + ) + # Rate-only v1 settles admitted aggregate bytes at the common + # thermal-window boundary. Shared resource capacities above are + # authoritative; the pipeline estimate is diagnostic and must not + # be used as an exact online dependency completion. + completion_ns = end_ns + job.pending_batches += 1 + heapq.heappush(self._inflight, (completion_ns, self._sequence, job_id, amount)) + self._sequence += 1 + activities.extend(self._activity_rows(job, amount, start_ns, end_ns)) + if job.operation in READ_OPERATIONS: + for buffer_stack, bank_count, bank_capacity in self._buffer_specs(job): + buffer_rows.append({ + "job_id": job.job_id, "source_stack": job.stack, + "buffer_stack": buffer_stack, "route": job.route, + "bank_count": bank_count, "bank_capacity_bytes": bank_capacity, + "maximum_occupancy_bytes": min(amount, bank_count * bank_capacity), + "turnovers": (amount + bank_capacity - 1) // bank_capacity, + "bytes": amount, + "pipeline_estimated_completion_ns": pipeline_estimated_completion_ns, + "settled_completion_ns": completion_ns, + }) + + completed_by_channel = {key: 0 for key in self._queues} + completed_by_job = {} + delay_samples = {stack: [] for stack in stacks} + completion_ids: list[str] = [] + maintenance_ids: list[str] = [] + self._complete_batches( + end_ns, completed_by_channel, completed_by_job, delay_samples, + completion_ids, maintenance_ids, + ) + + served = { + stack: {channel: completed_by_channel[(stack, channel)] for channel in channels} + for stack, channels in self._config["channels"].items() + } + stack_rows = {} + for stack, channels in self._config["channels"].items(): + backlog = sum(self._foreground_backlog[(stack, channel)] for channel in channels) + waiting = [] + for channel in channels: + queue = self._queues[(stack, channel)] + if queue: + waiting.append(end_ns - self._jobs[queue[0]].arrival_ns) + stack_activities = [row for row in activities if row["stack"] == stack] + stack_rows[stack] = { + "offered_effective_bytes": sum(offered_this_window[(stack, channel)] for channel in channels), + "delivered_effective_bytes": sum(served[stack].values()), + "backlog_effective_bytes": backlog, + "oldest_wait_ns": max(waiting) if waiting else None, + "delivered_delay_histogram_bytes": self._histogram(delay_samples[stack]), + "media_activity_bytes": sum(row["bytes"] for row in stack_activities + if row["phase"].startswith("media_")), + "link_bytes_by_phase": { + phase: sum(row["bytes"] for row in stack_activities if row["phase"] == phase) + for phase in sorted({row["phase"] for row in stack_activities + if row["phase"] in { + "direct_gpu_link", "relay_send", "relay_receive", + "gpu_drain", "partner_gpu_drain", + }}) + }, + } + cumulative_offered = sum(self._cumulative_offered[(stack, channel)] for channel in channels) + cumulative_completed = sum(self._cumulative_completed[(stack, channel)] for channel in channels) + if cumulative_offered != cumulative_completed + backlog: + raise AssertionError(f"foreground byte conservation failed for {stack}") + stack_rows[stack]["cumulative_offered_effective_bytes"] = cumulative_offered + stack_rows[stack]["cumulative_delivered_effective_bytes"] = cumulative_completed + stack_rows[stack]["cumulative_conserved"] = True + + resource_rows = {} + for resource, capacity in initial_capacity.items(): + used_work = capacity - residual[resource] + resource_rows[resource] = { + "capacity_work_units_scaled": capacity, + "used_work_units_scaled": used_work, + "remaining_work_units_scaled": residual[resource], + "work_scale": scale, + "capacity_semantics": ( + "CALLER_FUTURE_ENDPOINT_BUDGET_APPLIED_ONCE" + if resource.startswith("endpoint:") + else "WINDOW_BANDWIDTH_MINUS_ONE_STARTUP_LATENCY" + ), + } + + progress = [] + progress_ids = set(self._external_active) | set(completion_ids) + for job in sorted((self._jobs[job_id] for job_id in progress_ids), key=lambda row: row.sequence): + progress.append({ + "job_id": job.job_id, "maintenance_id": job.maintenance_id, + "metadata": copy.deepcopy(job.metadata), + "operation": job.operation, "stack": job.stack, "channel": job.channel, + "route": job.route, "arrival_ns": job.arrival_ns, + "total_bytes": job.total_bytes, + "admitted_this_window_bytes": allocations.get(job.job_id, 0), + "served_this_window_bytes": completed_by_job.get(job.job_id, 0), + "cumulative_served_bytes": job.completed_bytes, + "remaining_bytes": job.total_bytes - job.completed_bytes, + "admission_remaining_bytes": job.admission_remaining_bytes, + "first_service_ns": job.first_service_ns, + "last_service_ns": job.last_service_ns, + "completion_ns": job.completion_ns, + }) + + route_rows = {} + for job_id, amount in allocations.items(): + job = self._jobs[job_id] + key = f"{job.stack}:{job.route or 'internal'}:{job.operation}" + route_rows[key] = route_rows.get(key, 0) + amount + + self._now = end_ns + receipt = { + "schema_version": self.schema_version, + "start_ns": start_ns, "end_ns": end_ns, + "topology": self._config["topology"], + "served_by_stack_channel": served, + "stacks": stack_rows, + "activities": activities, + "route_phase_bytes": route_rows, + "resources": resource_rows, + "buffers": buffer_rows, + "blocked": blocked, + "job_progress": progress, + "completion_ids": completion_ids, + "maintenance_completion_ids": maintenance_ids, + "semantics": { + "service": "AGGREGATED_FLUID_NOT_MQSIM_TRANSACTION", + "buffer": "FINITE_TWO_BANK_CONTINUOUS_TURNOVER_NOT_WINDOW_CAPACITY", + "buffer_arbitration": "AGGREGATE_TURNOVER_SHARED_LINK_IS_AUTHORITATIVE_BANK_EVENT_ORDER_APPROXIMATE", + "pipeline": "FIRST_CHUNK_STARTUP_PLUS_BOTTLENECK_TURNOVER_APPROXIMATION", + "completion": "RATE_ONLY_COMMON_THERMAL_WINDOW_END_AFTER_SHARED_RESOURCE_ALLOCATION", + "latency_error_bound": "ONE_WINDOW_PLUS_AGGREGATE_CHUNK_PIPELINE_APPROXIMATION", + "causal_subwindow_api": "NOT_AVAILABLE_IN_V1_ADVANCE_CALLER_MUST_NOT_CHAIN_AT_WINDOW_END", + "backend_latency": "UNKNOWN", + "light_quota": "CALLER_BUDGET_APPLIED_EXACTLY_ONCE_NO_INTERNAL_HALF", + "inflight_state_change": "NO_CROSS_WINDOW_INFLIGHT_RATE_SETTLEMENT", + "maintenance_guard_policy": MAINTENANCE_POLICY, + "energy": "ACTIVITY_FACTS_ONLY_NO_UNSOURCED_JOULES", + }, + } + completed_job_ids = [ + job_id for job_id, job in self._jobs.items() if job.completion_ns is not None + ] + for job_id in completed_job_ids: + self._external_active.discard(job_id) + del self._jobs[job_id] + return receipt diff --git a/include/hbfsim/eq3_thermal/mqsim_observer.hpp b/include/hbfsim/eq3_thermal/mqsim_observer.hpp index 9b9f85f..e693174 100644 --- a/include/hbfsim/eq3_thermal/mqsim_observer.hpp +++ b/include/hbfsim/eq3_thermal/mqsim_observer.hpp @@ -1,7 +1,13 @@ #pragma once #include #include +#include +#include +#include +#include #include +#include +#include namespace hbfsim::eq3_thermal { // Existing MQSim API exposes request occupancy, NOT NAND command start/die/plane. @@ -48,4 +54,101 @@ class MqsimObserverAdapter { ActivityObserver& observer_; std::map bindings_; }; + +// Optional pre-submit composition boundary. The decision callback is pure with +// respect to MQSim: it must not submit requests, advance the engine, or drain +// observations. In particular, this adapter owns no event loop and buffers no +// completions. A caller handles DEFER by advancing through the existing +// run_next_completion_until() API, draining observations, and trying again. +enum class MqsimGateMode { Off, Enabled }; +enum class MqsimGateDisposition { Allow, Defer, Blocked, Unsupported }; + +struct MqsimGateDecision { + MqsimGateDisposition disposition{MqsimGateDisposition::Allow}; + std::optional target_time_ns; + std::string reason; +}; + +struct MqsimGateAttempt { + MqsimGateDecision decision; + bool submitted{}; + std::uint64_t original_arrival_ns{}; + std::uint64_t evaluated_ns{}; + std::optional backend_arrival_ns; + std::optional external_wait_ns; +}; + +enum class MqsimBackendOperation { Demand, DieLevelMaintenance }; +struct MqsimBackendCapability { + bool supported{}; + std::string status; + std::string detail; +}; + +using MqsimAdmissionGate = + std::function; + +class MqsimSubmissionGateAdapter { + public: + MqsimSubmissionGateAdapter(MqsimOnlineEngine& engine, + MqsimGateMode mode=MqsimGateMode::Off, + MqsimAdmissionGate gate={}) + :engine_(engine),mode_(mode),gate_(std::move(gate)) { + if(mode_==MqsimGateMode::Enabled&&!gate_) + throw std::invalid_argument("enabled MQSim submission gate needs a decision callback"); + } + + MqsimGateAttempt try_submit(const HbfRequest& request) { + const auto now=engine_.current_time_ns(); + if(mode_==MqsimGateMode::Off) { + engine_.submit(request); + return {{MqsimGateDisposition::Allow,std::nullopt,"gate disabled"},true, + request.arrival_ns,now,request.arrival_ns,0}; + } + if(submitted_.contains(request.request_id)) + throw std::logic_error("MQSim gate request was already submitted"); + // Do not make a control decision before this request exists in target + // simulation time: the controlled state may change before its arrival. + if(request.arrival_ns>now) + return {{MqsimGateDisposition::Defer,request.arrival_ns, + "request has not reached its target arrival time"},false, + request.arrival_ns,now,std::nullopt,std::nullopt}; + + auto decision=gate_(request,now); + if(decision.disposition==MqsimGateDisposition::Allow) { + if(decision.target_time_ns) + throw std::invalid_argument("ALLOW decision cannot carry a target time"); + auto backend=request; + // MQSim rejects arrivals before its current clock. Preserve that external + // wait in the attempt ledger and leave the backend completion untouched. + backend.arrival_ns=std::max(request.arrival_ns,now); + engine_.submit(backend); + submitted_.insert(request.request_id); + return {std::move(decision),true,request.arrival_ns,now, + backend.arrival_ns,backend.arrival_ns-request.arrival_ns}; + } + if(decision.reason.empty()) + throw std::invalid_argument("non-ALLOW MQSim gate decision needs a reason"); + if(decision.disposition==MqsimGateDisposition::Defer) { + if(!decision.target_time_ns||*decision.target_time_ns<=now) + throw std::invalid_argument("DEFER target must follow current target time"); + } else if(decision.target_time_ns) { + throw std::invalid_argument("BLOCKED/UNSUPPORTED decision cannot carry a target time"); + } + return {std::move(decision),false,request.arrival_ns,now,std::nullopt,std::nullopt}; + } + + static MqsimBackendCapability capability(MqsimBackendOperation operation) { + if(operation==MqsimBackendOperation::Demand) + return {true,"SUPPORTED","optional pre-submit demand gate"}; + return {false,"UNSUPPORTED_CAPABILITY", + "MQSimOnlineEngine exposes no die-level maintenance operation or completion"}; + } + + private: + MqsimOnlineEngine& engine_; + MqsimGateMode mode_; + MqsimAdmissionGate gate_; + std::set submitted_; +}; } // namespace hbfsim::eq3_thermal diff --git a/include/hbfsim/eq3_thermal/mqsim_stack_map.hpp b/include/hbfsim/eq3_thermal/mqsim_stack_map.hpp new file mode 100644 index 0000000..4a1ee5c --- /dev/null +++ b/include/hbfsim/eq3_thermal/mqsim_stack_map.hpp @@ -0,0 +1,141 @@ +#pragma once + +#include +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include + +namespace hbfsim::eq3_thermal { + +struct MqsimStackChannelGroup { + std::string stack_id; + std::vector channels; + std::uint32_t declared_dies{}; +}; + +struct MqsimStackPlacement { + HbfRequest backend_request{}; + std::uint64_t external_page{}; + std::uint64_t backend_page{}; + std::optional stack_id; + std::optional expected_channel; +}; + +// Optional one-page address-layout composition for same-kind HBF stacks. The +// transform is a persistent bijection, not dynamic polling or rebalancing. It +// does not submit, retry, reserve, or own work. Enabled mappings must be used +// exclusively for the lifetime of their MQSim engine because mapped and +// unmapped LPAs are distinct backend namespaces. +class MqsimStackMapAdapter { + public: + MqsimStackMapAdapter() = default; + + MqsimStackMapAdapter(const Profile& profile, + std::vector groups) + : page_bytes_(profile.page_bytes), capacity_bytes_(profile.capacity_bytes), + channels_(profile.channels), + dies_per_channel_(profile.dies_per_channel), groups_(std::move(groups)) { + if(profile.plane_allocation_scheme!=PlaneAllocationScheme::Cwdp) + throw std::invalid_argument("MQSim stack map requires CWDP allocation"); + if(!page_bytes_||!channels_||groups_.size()<2||channels_%groups_.size()) + throw std::invalid_argument("invalid MQSim stack-map geometry"); + channels_per_stack_=channels_/groups_.size(); + channel_to_stack_.resize(channels_); + for(std::size_t s=0;s=channels_||channel_to_stack_[channel]) + throw std::invalid_argument("overlapping or invalid MQSim channel group"); + channel_to_stack_[channel]=s; + } + } + for(const auto& owner:channel_to_stack_)if(!owner) + throw std::invalid_argument("MQSim stack map must cover every channel"); + enabled_=true; + } + + [[nodiscard]] bool enabled()const noexcept{return enabled_;} + + [[nodiscard]] MqsimStackPlacement map( + const HbfRequest& request, + std::optional requested_stack=std::nullopt)const { + if(!enabled_) + return {request,request.logical_address/page_bytes_, + request.logical_address/page_bytes_,std::nullopt,std::nullopt}; + if(request.bytes!=page_bytes_||request.logical_address%page_bytes_) + throw std::invalid_argument("MQSim stack map supports one aligned profile page"); + const auto page=request.logical_address/page_bytes_; + if(page>=capacity_bytes_/page_bytes_) + throw std::out_of_range("MQSim stack-map request exceeds capacity"); + const auto s=static_cast(page%groups_.size()); + const auto q=page/groups_.size(); + const auto k=static_cast(q%channels_per_stack_); + const auto r=q/channels_per_stack_; + const auto channel=groups_[s].channels[k]; + if(requested_stack&&*requested_stack!=groups_[s].stack_id) + throw std::invalid_argument("requested HBF stack conflicts with configured address stripe"); + if(r>(std::numeric_limits::max()-channel)/channels_) + throw std::overflow_error("MQSim backend page mapping overflows"); + const auto backend_page=r*channels_+channel; + if(backend_page>std::numeric_limits::max()/page_bytes_) + throw std::overflow_error("MQSim backend byte address overflows"); + auto backend=request; + backend.logical_address=backend_page*page_bytes_; + return {backend,page,backend_page,groups_[s].stack_id,channel}; + } + + // Convert a persistent page ordinal inside one explicitly named HBF stack + // to the global striped namespace, then apply the same bijection as map(). + [[nodiscard]] MqsimStackPlacement map_stack_page( + const HbfRequest& request,std::string_view requested_stack, + std::uint64_t stack_local_page)const { + if(!enabled_)throw std::logic_error("MQSim stack-local placement requires enabled mapping"); + std::size_t stack_index=groups_.size(); + for(std::size_t s=0;s(std::numeric_limits::max()-stack_index)/groups_.size()) + throw std::overflow_error("MQSim external page mapping overflows"); + const auto external_page=stack_local_page*groups_.size()+stack_index; + if(external_page>std::numeric_limits::max()/page_bytes_) + throw std::overflow_error("MQSim external byte address overflows"); + auto external=request; + external.logical_address=external_page*page_bytes_; + return map(external,requested_stack); + } + + [[nodiscard]] std::optional stack_for_channel( + std::uint32_t channel)const { + if(!enabled_||channel>=channel_to_stack_.size()||!channel_to_stack_[channel]) + return std::nullopt; + return groups_[*channel_to_stack_[channel]].stack_id; + } + + private: + bool enabled_{}; + std::uint64_t page_bytes_{1}; + std::uint64_t capacity_bytes_{}; + std::size_t channels_{}; + std::size_t dies_per_channel_{}; + std::size_t channels_per_stack_{}; + std::vector groups_; + std::vector> channel_to_stack_; +}; + +} // namespace hbfsim::eq3_thermal diff --git a/patches/mqsim/0003-hbf-command-observer.patch b/patches/mqsim/0003-hbf-command-observer.patch new file mode 100644 index 0000000..8edebc3 --- /dev/null +++ b/patches/mqsim/0003-hbf-command-observer.patch @@ -0,0 +1,254 @@ +diff --git a/src/ssd/NVM_PHY_ONFI_NVDDR2.cpp b/src/ssd/NVM_PHY_ONFI_NVDDR2.cpp +index 0b659d9..2715a50 100644 +--- a/src/ssd/NVM_PHY_ONFI_NVDDR2.cpp ++++ b/src/ssd/NVM_PHY_ONFI_NVDDR2.cpp +@@ -1,9 +1,62 @@ + #include ++#include + #include "../sim/Engine.h" + #include "NVM_PHY_ONFI_NVDDR2.h" + #include "Stats.h" + + namespace SSD_Components { ++ namespace { ++ HBF_Command_Observation_Sink hbf_command_observation_sink; ++ bool hbf_command_observation_failed = false; ++ std::uint64_t next_hbf_command_id = 1, next_hbf_transaction_id = 1; ++ HBF_Transaction_Observation transaction_observation(NVM_Transaction_Flash* tr) ++ { ++ if (tr->HBF_Observation_ID == 0) tr->HBF_Observation_ID = next_hbf_transaction_id++; ++ return {tr->HBF_Observation_ID, ++ tr->UserIORequest == NULL ? 0 : tr->UserIORequest->HBF_External_Request_ID, ++ static_cast(tr->Source), static_cast(tr->Type), ++ tr->LPA, tr->LPA != NO_LPA, tr->Data_and_metadata_size_in_byte, tr->Address.ChannelID, ++ tr->Address.ChipID, tr->Address.DieID, tr->Address.PlaneID, ++ tr->Address.BlockID, tr->Address.PageID}; ++ } ++ } ++ ++ void NVM_PHY_ONFI_NVDDR2::Set_hbf_command_observation_sink(HBF_Command_Observation_Sink sink) ++ { ++ hbf_command_observation_sink = std::move(sink); ++ hbf_command_observation_failed = false; ++ next_hbf_command_id = next_hbf_transaction_id = 1; ++ } ++ void NVM_PHY_ONFI_NVDDR2::Clear_hbf_command_observation_sink() ++ { ++ hbf_command_observation_sink = {}; ++ hbf_command_observation_failed = false; ++ } ++ bool NVM_PHY_ONFI_NVDDR2::Hbf_command_observation_failed() ++ { ++ return hbf_command_observation_failed; ++ } ++ void NVM_PHY_ONFI_NVDDR2::observe_command(DieBookKeepingEntry* die, ++ HBF_Command_Observation_Phase phase) ++ { ++ if (!hbf_command_observation_sink) return; ++ if (die->HBF_Observation_Command_ID == 0) die->HBF_Observation_Command_ID = next_hbf_command_id++; ++ HBF_Command_Observation value{die->HBF_Observation_Command_ID, phase, ++ Simulator->Time(), die->ActiveCommand->CommandCode, {}}; ++ for (auto* tr : die->ActiveTransactions) value.transactions.push_back(transaction_observation(tr)); ++ try { hbf_command_observation_sink(value); } ++ catch (...) { hbf_command_observation_failed = true; } ++ } ++ void NVM_PHY_ONFI_NVDDR2::observe_transfer(DieBookKeepingEntry* die, ++ NVM_Transaction_Flash* tr, HBF_Command_Observation_Phase phase) ++ { ++ if (!hbf_command_observation_sink) return; ++ HBF_Command_Observation value{die->HBF_Observation_Command_ID, phase, ++ Simulator->Time(), die->ActiveCommand->CommandCode, {transaction_observation(tr)}}; ++ try { hbf_command_observation_sink(value); } ++ catch (...) { hbf_command_observation_failed = true; } ++ } ++ + /*hack: using this style to emulate event/delegate*/ + NVM_PHY_ONFI_NVDDR2* NVM_PHY_ONFI_NVDDR2::_my_instance; + +@@ -314,6 +367,7 @@ namespace SSD_Components { + throw std::invalid_argument("NVM_PHY_ONFI_NVDDR2: Unhandled event specified!"); + } + ++ observe_command(dieBKE, HBF_Command_Observation_Phase::COMMAND_ISSUED); + target_channel->SetStatus(BusChannelStatus::BUSY, targetChip); + } + +@@ -346,6 +400,7 @@ namespace SSD_Components { + switch ((NVDDR2_SimEventType)ev->Type) { + case NVDDR2_SimEventType::READ_CMD_ADDR_TRANSFERRED: + //DEBUG2("Chip " << targetChip->ChannelID << ", " << targetChip->ChipID << ", " << dieBKE->ActiveTransactions.front()->Address.DieID << ": READ_CMD_ADDR_TRANSFERRED ") ++ observe_command(dieBKE, HBF_Command_Observation_Phase::MEDIA_BEGIN); + targetChip->EndCMDXfer(dieBKE->ActiveCommand); + for (auto tr : dieBKE->ActiveTransactions) { + tr->STAT_execution_time = dieBKE->Expected_finish_time - Simulator->Time(); +@@ -362,6 +417,7 @@ namespace SSD_Components { + break; + case NVDDR2_SimEventType::ERASE_SETUP_COMPLETED: + //DEBUG2("Chip " << targetChip->ChannelID << ", " << targetChip->ChipID << ", " << dieBKE->ActiveTransactions.front()->Address.DieID << ": ERASE_SETUP_COMPLETED ") ++ observe_command(dieBKE, HBF_Command_Observation_Phase::MEDIA_BEGIN); + targetChip->EndCMDXfer(dieBKE->ActiveCommand); + for (auto &tr : dieBKE->ActiveTransactions) { + tr->STAT_execution_time = dieBKE->Expected_finish_time - Simulator->Time(); +@@ -379,6 +435,7 @@ namespace SSD_Components { + case NVDDR2_SimEventType::PROGRAM_CMD_ADDR_DATA_TRANSFERRED: + case NVDDR2_SimEventType::PROGRAM_COPYBACK_CMD_ADDR_TRANSFERRED: + //DEBUG2("Chip " << targetChip->ChannelID << ", " << targetChip->ChipID << ", " << dieBKE->ActiveTransactions.front()->Address.DieID << ": PROGRAM_CMD_ADDR_DATA_TRANSFERRED " ) ++ observe_command(dieBKE, HBF_Command_Observation_Phase::MEDIA_BEGIN); + targetChip->EndCMDDataInXfer(dieBKE->ActiveCommand); + for (auto &tr : dieBKE->ActiveTransactions) { + tr->STAT_execution_time = dieBKE->Expected_finish_time - Simulator->Time(); +@@ -396,6 +453,8 @@ namespace SSD_Components { + break; + case NVDDR2_SimEventType::READ_DATA_TRANSFERRED: + //DEBUG2("Chip " << targetChip->ChannelID << ", " << targetChip->ChipID << ", " << dieBKE->ActiveTransactions.front()->Address.DieID << ": READ_DATA_TRANSFERRED ") ++ observe_transfer(dieBKE, dieBKE->ActiveTransfer, ++ HBF_Command_Observation_Phase::DATA_OUT_END); + targetChip->EndDataOutXfer(dieBKE->ActiveCommand); + copy_read_data_to_transaction((NVM_Transaction_Flash_RD*)dieBKE->ActiveTransfer, dieBKE->ActiveCommand); + #if 0 +@@ -452,6 +511,8 @@ namespace SSD_Components { + Simulator->Register_sim_event(Simulator->Time() + this->channels[channel_id]->ProgramCommandTime[waitingBKE->ActiveTransactions.size()], + this, waitingBKE, (int)NVDDR2_SimEventType::PROGRAM_COPYBACK_CMD_ADDR_TRANSFERRED); + waitingChipBKE->OngoingDieCMDTransfers.push(waitingBKE); ++ waitingBKE->HBF_Observation_Command_ID = 0; ++ observe_command(waitingBKE, HBF_Command_Observation_Phase::COMMAND_ISSUED); + + waitingBKE->Expected_finish_time = Simulator->Time() + this->channels[channel_id]->ProgramCommandTime[waitingBKE->ActiveTransactions.size()] + + targetChip->Get_command_execution_latency(waitingBKE->ActiveCommand->CommandCode, waitingBKE->ActiveCommand->Address[0].PageID); +@@ -493,6 +554,7 @@ namespace SSD_Components { + { + ChipBookKeepingEntry *chipBKE = &_my_instance->bookKeepingTable[chip->ChannelID][chip->ChipID]; + DieBookKeepingEntry *dieBKE = &(chipBKE->Die_book_keeping_records[command->Address[0].DieID]); ++ _my_instance->observe_command(dieBKE, HBF_Command_Observation_Phase::MEDIA_END); + + switch (command->CommandCode) + { +@@ -555,6 +617,9 @@ namespace SSD_Components { + Simulator->Register_sim_event(Simulator->Time() + _my_instance->channels[chip->ChannelID]->ProgramCommandTime[dieBKE->ActiveTransactions.size()], + _my_instance, dieBKE, (int)NVDDR2_SimEventType::PROGRAM_COPYBACK_CMD_ADDR_TRANSFERRED); + chipBKE->OngoingDieCMDTransfers.push(dieBKE); ++ dieBKE->HBF_Observation_Command_ID = 0; ++ _my_instance->observe_command(dieBKE, ++ HBF_Command_Observation_Phase::COMMAND_ISSUED); + _my_instance->channels[chip->ChannelID]->SetStatus(BusChannelStatus::BUSY, chip); + + dieBKE->Expected_finish_time = Simulator->Time() + _my_instance->channels[chip->ChannelID]->ProgramCommandTime[dieBKE->ActiveTransactions.size()] +@@ -626,6 +691,7 @@ namespace SSD_Components { + { + //DEBUG2("Chip " << tr->Address.ChannelID << ", " << tr->Address.ChipID << ": transfer read data started for LPA: " << tr->LPA) + dieBKE->ActiveTransfer = tr; ++ observe_transfer(dieBKE, tr, HBF_Command_Observation_Phase::DATA_OUT_BEGIN); + channels[tr->Address.ChannelID]->Chips[tr->Address.ChipID]->StartDataOutXfer(); + chipBKE->Status = ChipStatus::DATA_OUT; + Simulator->Register_sim_event(Simulator->Time() + NVDDR2DataOutTransferTime(tr->Data_and_metadata_size_in_byte, channels[tr->Address.ChannelID]), +diff --git a/src/ssd/NVM_PHY_ONFI_NVDDR2.h b/src/ssd/NVM_PHY_ONFI_NVDDR2.h +index be51ac2..197b160 100644 +--- a/src/ssd/NVM_PHY_ONFI_NVDDR2.h ++++ b/src/ssd/NVM_PHY_ONFI_NVDDR2.h +@@ -3,6 +3,9 @@ + + #include + #include ++#include ++#include ++#include + #include "../sim/Sim_Defs.h" + #include "../nvm_chip/flash_memory/FlashTypes.h" + #include "../nvm_chip/flash_memory/Flash_Command.h" +@@ -12,6 +15,27 @@ + + namespace SSD_Components + { ++ enum class HBF_Command_Observation_Phase { COMMAND_ISSUED, MEDIA_BEGIN, MEDIA_END, DATA_OUT_BEGIN, DATA_OUT_END }; ++ struct HBF_Transaction_Observation { ++ std::uint64_t transaction_id, external_request_id; ++ unsigned int source, type; ++ std::uint64_t logical_page; bool logical_page_known; ++ std::uint32_t bytes; ++ flash_channel_ID_type channel; ++ flash_chip_ID_type chip; ++ flash_die_ID_type die; ++ flash_plane_ID_type plane; ++ flash_block_ID_type block; ++ flash_page_ID_type page; ++ }; ++ struct HBF_Command_Observation { ++ std::uint64_t command_id; ++ HBF_Command_Observation_Phase phase; ++ sim_time_type time; ++ command_code_type command_code; ++ std::vector transactions; ++ }; ++ using HBF_Command_Observation_Sink = std::function; + enum class NVDDR2_SimEventType + { + READ_DATA_TRANSFERRED, READ_CMD_ADDR_TRANSFERRED, +@@ -37,6 +61,7 @@ namespace SSD_Components + sim_time_type Expected_finish_time; + sim_time_type RemainingExecTime; + sim_time_type DieInterleavedTime;//If the command transfer is done in die-interleaved mode, the transfer time is recorded in this temporary variable ++ std::uint64_t HBF_Observation_Command_ID{0}; + + void PrepareSuspend() + { +@@ -65,6 +90,7 @@ namespace SSD_Components + delete ActiveCommand; + ActiveCommand = NULL; + ActiveTransactions.clear(); ++ HBF_Observation_Command_ID = 0; + Free = true; + } + }; +@@ -108,7 +134,12 @@ namespace SSD_Components + NVM_Transaction_Flash* Is_chip_busy_with_stream(NVM_Transaction_Flash* transaction); + bool Is_chip_busy(NVM_Transaction_Flash* transaction); + void Change_memory_status_preconditioning(const NVM::NVM_Memory_Address* address, const void* status_info); ++ static void Set_hbf_command_observation_sink(HBF_Command_Observation_Sink sink); ++ static void Clear_hbf_command_observation_sink(); ++ static bool Hbf_command_observation_failed(); + private: ++ static void observe_command(DieBookKeepingEntry*, HBF_Command_Observation_Phase); ++ static void observe_transfer(DieBookKeepingEntry*, NVM_Transaction_Flash*, HBF_Command_Observation_Phase); + void transfer_read_data_from_chip(ChipBookKeepingEntry* chipBKE, DieBookKeepingEntry* dieBKE, NVM_Transaction_Flash* tr); + void perform_interleaved_cmd_data_transfer(NVM::FlashMemory::Flash_Chip* chip, DieBookKeepingEntry* bookKeepingEntry); + void send_resume_command_to_chip(NVM::FlashMemory::Flash_Chip* chip, ChipBookKeepingEntry* chipBKE); +diff --git a/src/ssd/NVM_Transaction.h b/src/ssd/NVM_Transaction.h +index 6a86e55..4cd5702 100644 +--- a/src/ssd/NVM_Transaction.h ++++ b/src/ssd/NVM_Transaction.h +@@ -2,6 +2,7 @@ + #define NVM_TRANSACTION_H + + #include ++#include + #include "../sim/Sim_Defs.h" + #include "../sim/Engine.h" + #include "User_Request.h" +@@ -29,6 +30,7 @@ namespace SSD_Components + /* Used to calculate service time and transfer time for a normal read/program operation used to respond to the host IORequests. + In other words, these variables are not important if FlashTransactions is used for garbage collection.*/ + sim_time_type STAT_execution_time, STAT_transfer_time; ++ std::uint64_t HBF_Observation_ID{0}; + }; + } + +diff --git a/src/ssd/User_Request.h b/src/ssd/User_Request.h +index e3711c9..f5811c9 100644 +--- a/src/ssd/User_Request.h ++++ b/src/ssd/User_Request.h +@@ -3,6 +3,7 @@ + + #include + #include ++#include + #include "SSD_Defs.h" + #include "../sim/Sim_Defs.h" + #include "Host_Interface_Defs.h" +@@ -32,6 +33,7 @@ namespace SSD_Components + bool ToBeIgnored; + void* IO_command_info;//used to store host I/O command info + void* Data; ++ std::uint64_t HBF_External_Request_ID{0}; + private: + static unsigned int lastId; + }; diff --git a/scripts/build/prepare_cuda_glibc_overlay.py b/scripts/build/prepare_cuda_glibc_overlay.py new file mode 100644 index 0000000..56e4aa9 --- /dev/null +++ b/scripts/build/prepare_cuda_glibc_overlay.py @@ -0,0 +1,36 @@ +#!/usr/bin/env python3 +"""Copy CUDA includes and align only rsqrt declarations with recent glibc. + +Optional build-local workaround for CUDA13.1/new glibc. Never modifies the +installed toolkit. Use returned directory with NVCC_PREPEND_FLAGS=-I. +""" +import argparse +import hashlib +import json +from pathlib import Path +import re +import shutil + +p = argparse.ArgumentParser(description=__doc__) +p.add_argument('--include', type=Path, required=True) +p.add_argument('--output', type=Path, required=True) +a = p.parse_args() +source = (a.include / 'crt/math_functions.h').read_bytes() +text = source.decode() +for function, signature in [('rsqrt', 'double x'), ('rsqrtf', 'float x')]: + pattern = rf'(?m)^(extern __DEVICE_FUNCTIONS_DECL__ __device_builtin__[^\n]+\b{function}\({signature}\));$' + text, count = re.subn(pattern, r'\1 noexcept(true);', text) + if count != 1: + raise SystemExit(f'unsupported header: expected one unqualified {function} declaration, got {count}') +if a.output.exists(): + raise SystemExit('output must be new; preserve existing build evidence') +shutil.copytree(a.include, a.output) +(a.output / 'crt/math_functions.h').write_text(text) +receipt = {'source': str(a.include.resolve()), + 'original_sha256': hashlib.sha256(source).hexdigest(), + 'patched_sha256': hashlib.sha256(text.encode()).hexdigest(), + 'change': 'Only rsqrt/rsqrtf declarations receive noexcept(true), matching this glibc', + 'scope': 'BUILD_LOCAL_WORKAROUND_NOT_INSTALLED_TOOLKIT_CHANGE', + 'reference': 'https://forums.developer.nvidia.com/t/cuda-headers-in-crt-math-functions-h-still-broken-in-debian-13-repo/362708'} +(a.output.parent / (a.output.name + '.json')).write_text(json.dumps(receipt, indent=2)+'\n') +print(a.output.resolve()) diff --git a/scripts/eval/mqsim_service.py b/scripts/eval/mqsim_service.py index e97d83e..9a8065a 100644 --- a/scripts/eval/mqsim_service.py +++ b/scripts/eval/mqsim_service.py @@ -20,11 +20,13 @@ class MqsimService: source='MQSIM_SIMULATED' - def __init__(self, binary, profile, directory, *, timeout=30, parallel_units=None): + def __init__(self, binary, profile, directory, *, timeout=30, parallel_units=None, + artifact_root=None, stack_map=None, native_observations=False): self.binary=Path(binary).resolve(strict=True) self.profile=Path(profile).resolve(strict=True) self.directory=Path(directory).resolve() - if not self.binary.is_relative_to(ROOT) or not self.directory.is_relative_to(ROOT): + artifact_root = Path(artifact_root).resolve() if artifact_root is not None else ROOT + if not self.binary.is_relative_to(artifact_root) or not self.directory.is_relative_to(artifact_root): raise ValueError('service binary and artifacts must remain in experiment checkout') if not math.isfinite(timeout) or timeout<=0: raise ValueError('service timeout must be positive') @@ -34,6 +36,7 @@ def __init__(self, binary, profile, directory, *, timeout=30, parallel_units=Non self.requests={} self.completions={} self.observations=[] + self.native_observations=[] self.finished=False self.process=None self.directory.mkdir(parents=True,exist_ok=False) @@ -42,6 +45,10 @@ def __init__(self, binary, profile, directory, *, timeout=30, parallel_units=Non self.argv=[str(self.binary),'--profile',str(self.profile)] if parallel_units is not None: self.argv+=['--parallel-units',str(parallel_units)] + if stack_map is not None: + self.argv+=['--stack-map',str(Path(stack_map).resolve(strict=True))] + if native_observations: + self.argv+=['--native-command-observations','on'] try: self.process=subprocess.Popen(self.argv,stdin=subprocess.PIPE,stdout=subprocess.PIPE, stderr=self.stderr,bufsize=0) @@ -80,6 +87,7 @@ def read(self): raise ValueError('native service clock moved backward') self.now=current self.observations.extend(response['events']) + self.native_observations.extend(response.get('native_command_events', [])) return response def command(self, command): @@ -104,6 +112,18 @@ def submit(self, request): raise ValueError('native service did not accept one request') self.requests[rid]=dict(request) + def try_submit(self, request): + """Use the existing optional gate; the caller owns deferred requests.""" + rid=request['request_id'] + if rid in self.requests: + raise ValueError('duplicate service request ID') + decision=self.command(dict(command='try_submit',request=request))['gate'] + if decision['submitted']: + accepted=dict(request) + accepted['issue_ns']=decision['backend_arrival_ns'] + self.requests[rid]=accepted + return decision + def until(self, horizon): response=self.command(dict(command='until',deadline_ns=horizon)) completion=response['completion'] diff --git a/scripts/eval/test_mqsim_service_client.py b/scripts/eval/test_mqsim_service_client.py index 28b200b..2606eea 100644 --- a/scripts/eval/test_mqsim_service_client.py +++ b/scripts/eval/test_mqsim_service_client.py @@ -9,6 +9,7 @@ ROOT=Path(__file__).resolve().parents[2] BINARY=Path(os.environ.get('HBFSIM_MQSIM_SERVICE_BINARY',ROOT/'build-eval-implementation/hbf_mqsim_service')) +ARTIFACT_ROOT=Path(os.environ.get('EQ3_TEST_ARTIFACT_ROOT',ROOT)).resolve() class ClientTests(unittest.TestCase): @@ -22,7 +23,8 @@ def setUp(self): self.profile.write_text(json.dumps(profile)) def test_live_native_clock_and_finish_receipt(self): - with MqsimService(BINARY, self.profile, self.directory/'attempt', timeout=20) as service: + with MqsimService(BINARY, self.profile, self.directory/'attempt', timeout=20, + artifact_root=ARTIFACT_ROOT) as service: service.submit(dict(request_id=1,issue_ns=0,logical_address=0,bytes=16384,operation='read')) self.assertIsNone(service.until(1000)) self.assertEqual(service.now,1000) @@ -35,15 +37,18 @@ def test_live_native_clock_and_finish_receipt(self): self.assertTrue((self.directory/'attempt/service-transcript.jsonl').is_file()) def test_invalid_request_retains_failure_transcript_and_owns_cleanup(self): + service=MqsimService(BINARY, self.profile, self.directory/'failure', timeout=20, + artifact_root=ARTIFACT_ROOT) with self.assertRaises(ValueError): - with MqsimService(BINARY, self.profile, self.directory/'failure', timeout=20) as service: + with service: service.submit(dict(request_id=1,issue_ns=0,logical_address=0,bytes=1,operation='read')) self.assertIsNotNone(service.process.returncode) self.assertNotEqual(service.process.returncode,0) self.assertTrue((self.directory/'failure/service-stderr.log').read_text()) def test_empty_fully_resident_service_can_finish(self): - with MqsimService(BINARY,self.profile,self.directory/'empty',timeout=20) as service: + with MqsimService(BINARY,self.profile,self.directory/'empty',timeout=20, + artifact_root=ARTIFACT_ROOT) as service: self.assertIsNone(service.until(2500)) self.assertEqual(service.finish()['issued'],0) diff --git a/src/eq3_thermal/cpu_service.cpp b/src/eq3_thermal/cpu_service.cpp index 055a045..af225cd 100644 --- a/src/eq3_thermal/cpu_service.cpp +++ b/src/eq3_thermal/cpu_service.cpp @@ -54,7 +54,8 @@ struct CpuService::Impl { observer(RuntimeMode::Shadow,thermal) { need(config.at("evidence")=="ENGINEERING_FIXTURE","physical service parameters not authorized"); const std::string layout=config.at("topology");need(layout=="mixed_direct"||layout=="relay"||layout=="dash"||layout=="all_hbf_direct","unknown topology"); - need(config.at("policy")=="none"||config.at("policy")=="hysteresis","unsupported policy"); + need(config.at("policy")=="none"||config.at("policy")=="hysteresis"|| + config.at("policy")=="hysteresis_escalation_priority_v2","unsupported policy"); need(tick(config.at("thermal_step_ns"))>0&&tick(config.at("sample_ns"))>0,"positive clocks required"); for(const auto& [key,value]:config.at("duration_ns").items())need(tick(value)>0,"zero service duration"); for(const auto& [key,value]:config.at("power_w").items())(void)number(value); @@ -108,8 +109,10 @@ struct CpuService::Impl { J route(const J& job)const { const std::string stack=job.at("stack"),op=job.at("op"),path=job.at("route"); const std::string physical=kind(stack);const auto die=tick(job.at("die")); - J result={{"resources",J::array()},{"power",J::object()},{"external_power_w",0.0},{"link_hops",1}}; + J result={{"resources",J::array()},{"control_endpoints",J::array()},{"power",J::object()},{"external_power_w",0.0},{"link_hops",1}}; auto add=[&](const std::string& r){result["resources"].push_back(r);}; + std::set controlled; + auto control=[&](const std::string& id){if(controlled.insert(id).second)result["control_endpoints"].push_back(id);}; const auto& power=config.at("power_w"); if(physical=="GDDR") { need(config.at("topology")=="all_hbf_direct"&&path=="direct"&&(op=="read"||op=="write"),"invalid external GDDR operation"); @@ -118,7 +121,7 @@ struct CpuService::Impl { need(dieapplied) { + if(c.at("pending").is_null()||c.at("pending").at("state")!=level(desired)) + c["pending"]={{"state",level(desired)},{"at_ns",now()+tick(p.at("action_delay_ns"))}}; + } else { + if(c.at("pending").is_null()||c.at("pending").at("state")!=level(desired)) + c["pending"]={{"state",level(desired)},{"at_ns",std::max(now()+tick(p.at("action_delay_ns")),tick(c.at("last_change"))+tick(p.at("min_dwell_ns")))}}; } } state["next_sample"]=now()+tick(config.at("sample_ns")); } void apply_controls() { for(auto& [id,c]:state["control"].items())if(!c.at("pending").is_null()&&tick(c.at("pending").at("at_ns"))<=now()) { - log("control",{{"stack",id},{"from",c.at("applied")},{"to",c.at("pending").at("state")},{"reason","SIMULATED_STACK_HOTSPOT_HYSTERESIS"}}); + const std::string reason=config.at("policy")=="hysteresis_escalation_priority_v2"? + "SIMULATED_STACK_HOTSPOT_HYSTERESIS_ESCALATION_PRIORITY_V2":"SIMULATED_STACK_HOTSPOT_HYSTERESIS"; + log("control",{{"stack",id},{"from",c.at("applied")},{"to",c.at("pending").at("state")},{"reason",reason}}); c["applied"]=c.at("pending").at("state");c["last_change"]=now();c["pending"]=nullptr; } } @@ -244,11 +257,11 @@ struct CpuService::Impl { J waiting=J::array(); for(auto job:state["queue"]) { bool blocked=tick(job.at("arrival_ns"))>now();const std::string stack=job.at("stack"); - if(stack!="gddr") { - const auto& c=state.at("control").at(stack);const int s=severity(c.at("applied")); + const auto mapping=route(job); + for(const auto& endpoint:mapping.at("control_endpoints")) { + const auto& c=state.at("control").at(endpoint.get());const int s=severity(c.at("applied")); blocked|=s==3||(!job.at("maintenance").get()&&(s>=2||(s==1&&now()()); if(blocked){waiting.push_back(job);continue;} const auto duration=tick(config.at("duration_ns").at(job.at("op").get())); @@ -260,7 +273,10 @@ struct CpuService::Impl { if(stack!="gddr") { const std::string op=job.at("op");auto& cohort=state["cohorts"][cohort_id(stack,tick(job.at("die")))]; if(op=="program"||op=="erase"){const auto key=op=="program"?"program_attempts":"erase_attempts";cohort[key]=tick(cohort.at(key))+1;} - auto& c=state["control"][stack];if(!job.at("maintenance").get()&&c.at("applied")=="Light")c["next_admit"]=now()+tick(config.at("control").at(stack).at("light_gap_ns")); + if(!job.at("maintenance").get())for(const auto& endpoint:mapping.at("control_endpoints")) { + const std::string id=endpoint;auto& c=state["control"][id]; + if(c.at("applied")=="Light")c["next_admit"]=now()+tick(config.at("control").at(id).at("light_gap_ns")); + } } observer.submit(activity(job,Phase::Issue));observer.submit(activity(job,Phase::Start));state["active"].push_back(job); log("start",{{"id",job.at("id")},{"resources",job.at("resources")},{"end_ns",job.at("end_ns")}}); @@ -295,6 +311,31 @@ struct CpuService::Impl { J report()const { J result=state;result["evidence"]="ENGINEERING_FIXTURE";result["topology"]=config.at("topology");result["policy"]=config.at("policy"); result["temperature_source"]="SIMULATED";result["token_throughput"]="UNAVAILABLE"; + result["admission_blocks"]=J::array(); + for(const auto& job:state.at("queue")) { + const auto mapping=route(job);J row={{"id",job.at("id")},{"arrival_ns",job.at("arrival_ns")}, + {"blocked_endpoints",J::array()},{"busy_resources",J::array()}}; + for(const auto& endpoint:mapping.at("control_endpoints")) { + const std::string id=endpoint;const auto& c=state.at("control").at(id);const int s=severity(c.at("applied")); + std::string reason;J retry=nullptr; + if(s==3)reason="SHUTDOWN"; + else if(!job.at("maintenance").get()&&s>=2)reason="SEVERE"; + else if(!job.at("maintenance").get()&&s==1&&now()())) + row["busy_resources"].push_back(resource); + if(tick(job.at("arrival_ns"))>now())row["not_before_arrival_ns"]=job.at("arrival_ns"); + if(!row.at("blocked_endpoints").empty()||!row.at("busy_resources").empty()||row.contains("not_before_arrival_ns")) + result["admission_blocks"].push_back(row); + } result["energy_j"]=observer.energy_j();result["external_energy_j"]=observer.external_energy_j(); result["external_gddr_temperature"]="UNAVAILABLE_OUTSIDE_PACKAGE"; result["temperature_k"]=J::object();for(const auto& [id,i]:nodes)result["temperature_k"][id]=observer.model()->temperatures_k()[i]; diff --git a/src/mqsim_adapter/mqsim_online.cpp b/src/mqsim_adapter/mqsim_online.cpp index e1561b0..780b0da 100644 --- a/src/mqsim_adapter/mqsim_online.cpp +++ b/src/mqsim_adapter/mqsim_online.cpp @@ -449,6 +449,9 @@ namespace hbfsim mqsim_request->Priority_class = IO_Flow_Priority_Class::URGENT; mqsim_request->IO_command_info = nullptr; mqsim_request->Data = nullptr; + // Patched MQSim carries this immutable identity into native transaction + // and command observations. It is not used by scheduling or completion. + mqsim_request->HBF_External_Request_ID = request.request_id; auto submission = new Impl::Submission{ .descriptor = request, diff --git a/tests/cpu/ptx_transform_test.cpp b/tests/cpu/ptx_transform_test.cpp index ae52dc3..b8eb41d 100644 --- a/tests/cpu/ptx_transform_test.cpp +++ b/tests/cpu/ptx_transform_test.cpp @@ -1,7 +1,8 @@ #include "ptx_memory_op.hpp" #include "transform.hpp" -#include +#include +#include #include #include #include @@ -17,7 +18,8 @@ #define CHECK(condition) \ do { \ if (!(condition)) { \ - return __LINE__; \ + std::fprintf(stderr, "CHECK failed at line %d: %s\n", __LINE__, #condition); \ + std::exit(1); \ } \ } while (false) @@ -39,7 +41,7 @@ std::string configured_ptx(std::string ptx) std::string read_fixture(const std::string& name) { std::ifstream input("tests/fixtures/ptx/" + name); - assert(input); + CHECK(input); return configured_ptx(std::string{ std::istreambuf_iterator{input}, std::istreambuf_iterator{}, @@ -52,40 +54,40 @@ int main() { const auto op = hbfsim::ptx::parse_memory_op( "@%p1 ld.global.v2.u32 {%r4,%r5}, [%rd8+16];"); - assert(op.has_value()); - assert(op->predicate == "@%p1"); - assert(op->kind == hbfsim::ptx::AccessKind::Read); - assert(op->bytes == 8); - assert(op->base_register == "%rd8"); - assert(op->offset == 16); + CHECK(op.has_value()); + CHECK(op->predicate == "@%p1"); + CHECK(op->kind == hbfsim::ptx::AccessKind::Read); + CHECK(op->bytes == 8); + CHECK(op->base_register == "%rd8"); + CHECK(op->offset == 16); const auto store = hbfsim::ptx::parse_memory_op( "st.global.release.gpu.u64 [%rd2-0x20], %rd3; // payload"); - assert(store.has_value()); - assert(store->kind == hbfsim::ptx::AccessKind::Write); - assert(store->bytes == 8); - assert(store->offset == -32); + CHECK(store.has_value()); + CHECK(store->kind == hbfsim::ptx::AccessKind::Write); + CHECK(store->bytes == 8); + CHECK(store->offset == -32); const auto nc = hbfsim::ptx::parse_memory_op( "ld.global.nc.L2::128B.v4.b32 {%r0,%r1,%r2,%r3}, [%rd4];"); - assert(nc.has_value()); - assert(nc->bytes == 16); + CHECK(nc.has_value()); + CHECK(nc->bytes == 16); const auto volatile_load = hbfsim::ptx::parse_memory_op( "ld.volatile.global.u8 %rd7, [%rd8+4096];"); - assert(volatile_load.has_value()); - assert(volatile_load->kind == hbfsim::ptx::AccessKind::Read); - assert(volatile_load->bytes == 1); - assert(volatile_load->base_register == "%rd8"); - assert(volatile_load->offset == 4096); + CHECK(volatile_load.has_value()); + CHECK(volatile_load->kind == hbfsim::ptx::AccessKind::Read); + CHECK(volatile_load->bytes == 1); + CHECK(volatile_load->base_register == "%rd8"); + CHECK(volatile_load->offset == 4096); const auto volatile_negative = hbfsim::ptx::parse_memory_op( "ld.volatile.global.u8 %rd7, [%rd8+-32768];"); - assert(volatile_negative.has_value()); - assert(volatile_negative->offset == -32768); + CHECK(volatile_negative.has_value()); + CHECK(volatile_negative->offset == -32768); - assert(!hbfsim::ptx::parse_memory_op("atom.global.add.u32 %r1, [%rd2], 1;")); - assert(!hbfsim::ptx::parse_memory_op("ld.u32 %r1, [%rd2];")); - assert(!hbfsim::ptx::parse_memory_op("ld.global.u32 %r1, [%r2+%r3];")); + CHECK(!hbfsim::ptx::parse_memory_op("atom.global.add.u32 %r1, [%rd2], 1;")); + CHECK(!hbfsim::ptx::parse_memory_op("ld.u32 %r1, [%rd2];")); + CHECK(!hbfsim::ptx::parse_memory_op("ld.global.u32 %r1, [%r2+%r3];")); const std::string spoofed_helper = configured_ptx(R"ptx(.version 8.7 .target sm_120 @@ -125,13 +127,13 @@ int main() .full_ptx = read_fixture("supported.ptx"), .to_patch_kernel = "kernel", }); - assert(result.modified); - assert(result.coverage.rewritten_instructions == 3); - assert(result.output_ptx.find("__hbfsim_resolve") != std::string::npos); - assert(result.output_ptx.find("__hbfsim_fault") != std::string::npos); - assert(result.output_ptx.find("@!%p1 bra $L__hbfsim_skip_") != + CHECK(result.modified); + CHECK(result.coverage.rewritten_instructions == 3); + CHECK(result.output_ptx.find("__hbfsim_resolve") != std::string::npos); + CHECK(result.output_ptx.find("__hbfsim_fault") != std::string::npos); + CHECK(result.output_ptx.find("@!%p1 bra $L__hbfsim_skip_") != std::string::npos); - assert(result.output_ptx.find("[%hbfsim_addr_") != std::string::npos); + CHECK(result.output_ptx.find("[%hbfsim_addr_") != std::string::npos); // The unsupported scan lists inline asm because the transform cannot see // through it. The pattern used to be `asm\s*\(`, which does not match @@ -178,7 +180,12 @@ int main() }); CHECK(mixed.modified); CHECK(mixed.coverage.rewritten_instructions == 1); - CHECK(mixed.coverage.unsupported_instructions == 1); + // transform_ptx reports all unsupported instructions, including ld.param. + // Only the plugin filters non-global opcodes when deciding admission. + CHECK(mixed.coverage.unsupported_instructions == 2); + CHECK(mixed.coverage.unsupported_opcodes.size() == 2); + CHECK(mixed.coverage.unsupported_opcodes[0] == "ld.param.u64"); + CHECK(mixed.coverage.unsupported_opcodes[1] == "ld.global.u32"); bool mixed_reports_global = false; for (const auto& opcode : mixed.coverage.unsupported_opcodes) { if (opcode.starts_with("ld.global")) { mixed_reports_global = true; } @@ -268,15 +275,15 @@ int main() .full_ptx = read_fixture("unsupported.ptx"), .to_patch_kernel = "unsupported_kernel", }); - assert(!rejected.modified); - assert(rejected.coverage.unsupported_instructions == 5); + CHECK(!rejected.modified); + CHECK(rejected.coverage.unsupported_instructions == 5); const auto excluded = hbfsim::ptx::transform_ptx({ .full_ptx = read_fixture("helper_exclusion.ptx"), .to_patch_kernel = "", }); - assert(!excluded.modified); - assert(excluded.coverage.excluded_functions == 2); + CHECK(!excluded.modified); + CHECK(excluded.coverage.excluded_functions == 2); // A global access the rewriter could not consume must be reported, never // skipped in silence. Silence lets a kernel be marked instrumented while diff --git a/tests/eq3_thermal/cpu_service_tests.cpp b/tests/eq3_thermal/cpu_service_tests.cpp index 1998f28..cd1b3b5 100644 --- a/tests/eq3_thermal/cpu_service_tests.cpp +++ b/tests/eq3_thermal/cpu_service_tests.cpp @@ -45,6 +45,192 @@ J request(std::string id,std::string stack="hbf0",std::string route="direct",std return {{"id",id},{"stack",stack},{"route",route},{"op",op},{"die",die},{"arrival_ns",0}, {"logical_bytes",64},{"physical_bytes",128},{"link_bytes",256},{"fail_fraction",0}}; } +void force_control(CpuService& service,const std::string& stack,const std::string& applied, + std::uint64_t next_sample=1000000000ULL) { + auto snapshot=J::parse(service.checkpoint()); + snapshot["state"]["control"][stack]["applied"]=applied; + snapshot["state"]["control"][stack]["suggested"]=applied; + snapshot["state"]["control"][stack]["pending"]=nullptr; + snapshot["state"]["next_sample"]=next_sample; + service.restore(snapshot.dump()); +} +void path_endpoint_admission() { + // A relay traverses the paired HBM base control domain. Shutdown there must + // prevent start without reserving any resource or creating forwarding heat. + CpuService shutdown(fixture("relay").dump());force_control(shutdown,"hbm0","Shutdown"); + shutdown.submit(request("blocked-relay","hbf0","relay").dump());shutdown.advance_to(10000000); + auto r=J::parse(shutdown.report()); + check(r["active"].empty()&&r["done"].empty()&&r["queue"].size()==1,"relay bypassed paired HBM Shutdown"); + check(r["resources"].empty(),"blocked relay leaked a partial reservation"); + check(r["admission_blocks"].size()==1&&r["admission_blocks"][0]["blocked_endpoints"].size()==1&& + r["admission_blocks"][0]["blocked_endpoints"][0]["stack"]=="hbm0"&& + r["admission_blocks"][0]["blocked_endpoints"][0]["reason"]=="SHUTDOWN"&& + r["admission_blocks"][0]["blocked_endpoints"][0]["retry_at_ns"].is_null(),"Shutdown blocker diagnostics missing"); + near(r["energy_j"]["hbf0_die0"],0);near(r["energy_j"]["hbf0_base"],0);near(r["energy_j"]["hbm0_base"],0); + + // DASH direct does not traverse the paired HBM endpoint and remains legal. + CpuService dash(fixture("dash").dump());force_control(dash,"hbm0","Shutdown"); + dash.submit(request("blocked-dash-relay","hbf0","relay").dump()); + dash.submit(request("legal-dash-direct","hbf0","direct", "read",1).dump());dash.advance_to(100000000); + r=J::parse(dash.report()); + check(r["done"].size()==1&&r["done"][0]["id"]=="legal-dash-direct"&&r["queue"].size()==1, + "unrelated paired endpoint blocked DASH direct or relay escaped"); + near(r["energy_j"]["hbm0_base"],0); + + // A successful foreground relay consumes the Light quota of each unique + // controlled endpoint, including its paired HBM base domain. + CpuService light(fixture("relay").dump());force_control(light,"hbf0","Light");force_control(light,"hbm0","Light"); + light.submit(request("light-relay","hbf0","relay").dump());light.advance_to(1); + r=J::parse(light.report()); + check(r["active"].size()==1,"two-endpoint Light relay was not admitted initially"); + check(r["control"]["hbf0"]["next_admit"]==250000000&&r["control"]["hbm0"]["next_admit"]==250000000, + "successful relay did not update every endpoint Light quota"); + auto hbm=request("light-hbm","hbm0");hbm["arrival_ns"]=1;light.submit(hbm.dump());light.advance_to(100000000); + r=J::parse(light.report());check(r["done"].size()==1&&r["queue"].size()==1,"paired Light quota was bypassed"); + light.advance_to(350000000);r=J::parse(light.report());check(r["done"].size()==2&&r["queue"].empty(),"Light retry did not complete uniquely"); + + // Joint admission honors the latest quota among traversed endpoints. + CpuService staggered(fixture("relay").dump()); + auto staggered_snapshot=J::parse(staggered.checkpoint()); + for(const auto* endpoint:{"hbf0","hbm0"}) { + staggered_snapshot["state"]["control"][endpoint]["applied"]="Light"; + staggered_snapshot["state"]["control"][endpoint]["suggested"]="Light"; + } + staggered_snapshot["state"]["control"]["hbf0"]["next_admit"]=50000000; + staggered_snapshot["state"]["control"]["hbm0"]["next_admit"]=150000000; + staggered_snapshot["state"]["next_sample"]=1000000000ULL;staggered.restore(staggered_snapshot.dump()); + staggered.submit(request("staggered-light","hbf0","relay").dump());staggered.advance_to(100000000); + r=J::parse(staggered.report());check(r["active"].empty()&&r["queue"].size()==1,"relay ignored the later endpoint Light quota"); + staggered.advance_to(250000000);r=J::parse(staggered.report()); + check(r["done"].size()==1&&r["done"][0]["start_ns"]==150000000,"relay did not retry at the joint Light boundary"); + + // The direct and relay DASH routes share the HBF upstream control endpoint. + CpuService shared(fixture("dash").dump());force_control(shared,"hbf0","Light"); + shared.submit(request("shared-direct","hbf0","direct").dump());shared.submit(request("shared-relay","hbf0","relay", "read",1).dump()); + shared.advance_to(200000000);r=J::parse(shared.report()); + check(r["done"].size()==1&&r["done"][0]["id"]=="shared-direct"&&r["queue"].size()==1, + "DASH routes bypassed their shared HBF Light quota"); + shared.advance_to(350000000);r=J::parse(shared.report()); + check(r["done"].size()==2&&r["done"][1]["start_ns"]==250000000,"shared endpoint retry was not unique/deterministic"); + + // Recovery becomes effective before admission at the same timestamp. A + // checkpoint taken while blocked must preserve the retry and completion. + auto recovering_config=fixture("relay");recovering_config["policy"]="hysteresis"; + CpuService recovering(recovering_config.dump());force_control(recovering,"hbm0","Shutdown",0); + recovering.submit(request("recovering-relay","hbf0","relay").dump());recovering.advance_to(20000000); + r=J::parse(recovering.report());check(r["queue"].size()==1&&r["active"].empty(),"relay started before endpoint recovery"); + CpuService resumed(recovering_config.dump());resumed.restore(recovering.checkpoint()); + recovering.advance_to(200000000);resumed.advance_to(200000000);check(recovering.report()==resumed.report(),"blocked endpoint checkpoint diverged"); + r=J::parse(recovering.report());check(r["done"].size()==1&&r["done"][0]["start_ns"]==30000000, + "same-timestamp recovery/admission ordering changed"); + + // Control changes never revoke in-flight work; they only gate later starts. + CpuService draining(fixture("relay").dump());draining.submit(request("inflight","hbf0","relay").dump());draining.advance_to(10000000); + force_control(draining,"hbm0","Shutdown");auto later=request("later","hbf0","relay");later["arrival_ns"]=10000000;draining.submit(later.dump()); + draining.advance_to(200000000);r=J::parse(draining.report()); + check(r["done"].size()==1&&r["done"][0]["id"]=="inflight"&&r["queue"].size()==1&&r["resources"].empty(), + "endpoint gate changed drain semantics or leaked resources"); + + // Existing maintenance policy permits maintenance through Severe (but not + // Shutdown); adding endpoint checks must preserve that explicit distinction. + CpuService maintenance(fixture("relay").dump()); + auto snapshot=J::parse(maintenance.checkpoint());snapshot["state"]["cohorts"]["hbf0:0"]["initial_age_ns"]=10000000000ULL; + snapshot["state"]["control"]["hbm0"]["applied"]="Severe";snapshot["state"]["control"]["hbm0"]["suggested"]="Severe"; + snapshot["state"]["next_sample"]=1000000000ULL;maintenance.restore(snapshot.dump());maintenance.advance_to(1); + r=J::parse(maintenance.report());bool found=false;for(const auto& job:r["active"]) + found|=job["maintenance"].get()&&job["stack"]=="hbf0"&&job["die"]==0; + check(found,"paired Severe endpoint changed maintenance exemption"); + CpuService light_maintenance(fixture("relay").dump());snapshot=J::parse(light_maintenance.checkpoint()); + snapshot["state"]["cohorts"]["hbf0:0"]["initial_age_ns"]=10000000000ULL; + snapshot["state"]["control"]["hbm0"]["applied"]="Light";snapshot["state"]["control"]["hbm0"]["suggested"]="Light"; + snapshot["state"]["control"]["hbm0"]["next_admit"]=700000000ULL;snapshot["state"]["next_sample"]=1000000000ULL; + light_maintenance.restore(snapshot.dump());light_maintenance.advance_to(1);r=J::parse(light_maintenance.report()); + found=false;for(const auto& job:r["active"])found|=job["maintenance"].get()&&job["stack"]=="hbf0"&&job["die"]==0; + check(found&&r["control"]["hbm0"]["next_admit"]==700000000ULL,"paired Light endpoint changed maintenance exemption/quota"); + CpuService stopped_maintenance(fixture("relay").dump());snapshot=J::parse(stopped_maintenance.checkpoint()); + snapshot["state"]["cohorts"]["hbf0:0"]["initial_age_ns"]=10000000000ULL; + snapshot["state"]["control"]["hbm0"]["applied"]="Shutdown";snapshot["state"]["control"]["hbm0"]["suggested"]="Shutdown"; + snapshot["state"]["next_sample"]=1000000000ULL;stopped_maintenance.restore(snapshot.dump());stopped_maintenance.advance_to(1); + r=J::parse(stopped_maintenance.report()); + found=false;for(const auto& job:r["queue"])found|=job["maintenance"].get()&&job["stack"]=="hbf0"&&job["die"]==0; + check(found&&r["active"].empty()&&r["resources"].empty(),"paired Shutdown endpoint admitted maintenance or leaked resources"); +} +void control_pending_contract() { + auto c=fixture();c["policy"]="hysteresis"; + CpuService cancel(c.dump());auto snapshot=J::parse(cancel.checkpoint()); + snapshot["state"]["control"]["hbf0"]["pending"]={{"state","Light"},{"at_ns",20000000}}; + cancel.restore(snapshot.dump());cancel.advance_to(1);auto r=J::parse(cancel.report()); + check(r["control"]["hbf0"]["applied"]=="Normal"&&r["control"]["hbf0"]["pending"].is_null(), + "desired==applied did not cancel a stale pending action"); + + auto hot=fixture();hot["policy"]="hysteresis"; + hot["control"]["hbf0"]["light_k"]=299.0;hot["control"]["hbf0"]["severe_k"]=299.1;hot["control"]["hbf0"]["shutdown_k"]=299.2; + CpuService escalate(hot.dump());snapshot=J::parse(escalate.checkpoint()); + snapshot["state"]["control"]["hbf0"]["pending"]={{"state","Light"},{"at_ns",20000000}}; + escalate.restore(snapshot.dump());escalate.advance_to(1);r=J::parse(escalate.report()); + check(r["control"]["hbf0"]["pending"]["state"]=="Shutdown"&&r["control"]["hbf0"]["pending"]["at_ns"]==30000000, + "more severe sample did not replace stale pending action"); + escalate.advance_to(30000001);r=J::parse(escalate.report());unsigned transitions=0; + for(const auto& row:r["log"])if(row["kind"]=="control"&&row["stack"]=="hbf0") { + ++transitions;check(row["from"]=="Normal"&&row["to"]=="Shutdown"&&row["time_ns"]==30000000, + "sample/action timestamp ordering changed"); + } + check(transitions==1,"stale pending action executed before escalation"); +} +void escalation_priority_v2_contract() { + auto hot=[](const std::string& policy) { + auto c=fixture();c["policy"]=policy;c["control"]["hbf0"]["light_k"]=299.0; + c["control"]["hbf0"]["severe_k"]=299.1;c["control"]["hbf0"]["shutdown_k"]=299.2; + c["control"]["hbf0"]["action_delay_ns"]=20000000;c["control"]["hbf0"]["min_dwell_ns"]=100000000; + return c; + }; + auto force_applied=[](CpuService& service,const std::string& applied,J pending=nullptr) { + auto snapshot=J::parse(service.checkpoint());auto& state=snapshot["state"]["control"]["hbf0"]; + state["applied"]=applied;state["suggested"]=applied;state["last_change"]=0;state["pending"]=pending; + snapshot["state"]["next_sample"]=0;service.restore(snapshot.dump()); + }; + + CpuService legacy(hot("hysteresis").dump());force_applied(legacy,"Light");legacy.advance_to(1); + auto r=J::parse(legacy.report()); + check(r["control"]["hbf0"]["pending"]["state"]=="Shutdown"&&r["control"]["hbf0"]["pending"]["at_ns"]==100000000, + "legacy hysteresis dwell contract changed"); + + const auto v2_config=hot("hysteresis_escalation_priority_v2");CpuService priority(v2_config.dump()); + force_applied(priority,"Light");priority.advance_to(10000001);r=J::parse(priority.report()); + check(r["control"]["hbf0"]["pending"]["state"]=="Shutdown"&&r["control"]["hbf0"]["pending"]["at_ns"]==20000000, + "v2 escalation was delayed by recovery dwell or same-target resampling"); + CpuService resumed(v2_config.dump());resumed.restore(priority.checkpoint()); + priority.advance_to(20000001);resumed.advance_to(20000001);check(priority.report()==resumed.report(),"v2 pending checkpoint diverged"); + r=J::parse(priority.report());check(r["control"]["hbf0"]["applied"]=="Shutdown","v2 escalation omitted action delay boundary"); + + auto cool=fixture();cool["policy"]="hysteresis_escalation_priority_v2"; + cool["control"]["hbf0"]["action_delay_ns"]=20000000;cool["control"]["hbf0"]["min_dwell_ns"]=100000000; + CpuService recovery(cool.dump());force_applied(recovery,"Shutdown");recovery.advance_to(1);r=J::parse(recovery.report()); + check(r["control"]["hbf0"]["pending"]["state"]=="Normal"&&r["control"]["hbf0"]["pending"]["at_ns"]==100000000, + "v2 recovery bypassed hysteresis dwell"); + + auto held=cool;held["control"]["hbf0"]["light_k"]=299.0;held["control"]["hbf0"]["severe_k"]=299.1; + held["control"]["hbf0"]["shutdown_k"]=300.02;held["control"]["hbf0"]["hysteresis_k"]=.05; + CpuService hysteresis(held.dump());force_applied(hysteresis,"Shutdown");hysteresis.advance_to(1);r=J::parse(hysteresis.report()); + check(r["control"]["hbf0"]["applied"]=="Shutdown"&&r["control"]["hbf0"]["pending"].is_null(),"v2 recovery ignored hysteresis"); + + CpuService replace(v2_config.dump()); + force_applied(replace,"Severe",J{{"state","Normal"},{"at_ns",100000000}});replace.advance_to(1);r=J::parse(replace.report()); + check(r["control"]["hbf0"]["pending"]["state"]=="Shutdown"&&r["control"]["hbf0"]["pending"]["at_ns"]==20000000, + "v2 severe recommendation did not replace stale recovery"); + + auto light=fixture();light["policy"]="hysteresis_escalation_priority_v2";light["control"]["hbf0"]["light_k"]=299.0; + light["control"]["hbf0"]["severe_k"]=310.0;light["control"]["hbf0"]["shutdown_k"]=320.0; + CpuService retarget(light.dump());force_applied(retarget,"Normal",J{{"state","Shutdown"},{"at_ns",20000000}}); + retarget.advance_to(1);r=J::parse(retarget.report()); + check(r["control"]["hbf0"]["pending"]["state"]=="Light","v2 retained an obsolete stronger pending target"); + + CpuService ordered(v2_config.dump());force_applied(ordered,"Light");auto q=request("v2-boundary");q["arrival_ns"]=20000000; + ordered.submit(q.dump());ordered.advance_to(20000001);r=J::parse(ordered.report()); + check(r["control"]["hbf0"]["applied"]=="Shutdown"&&r["queue"].size()==1&&r["active"].empty()&& + r["admission_blocks"][0]["blocked_endpoints"][0]["reason"]=="SHUTDOWN", + "v2 same-timestamp sample/apply/admission ordering changed"); +} void resource_energy_topologies() { for(const auto& topology:{"mixed_direct","relay","dash","all_hbf_direct"}) { auto config=fixture(topology,std::string(topology)=="all_hbf_direct"?0:4);CpuService s(config.dump()); @@ -142,4 +328,4 @@ void actual_dispatcher() { std::cout<<"PASS actual RequestDispatcher Engine and shared completion consumer (CPU fixture, not live GPU)\n"; #endif } -int main(){try{resource_energy_topologies();maintenance_failures_restart();control_cooling();actual_dispatcher();std::cout<<"PASS four topology resources/energy; maintenance partial failure/commit/wear; checkpoint; per-stack control/drain/cooling\n";return 0;}catch(const std::exception& e){std::cerr< completions;std::vector events;double wall{},energy{};unsigned advice_samples{};}; +HbfRequest request(std::uint64_t id,std::uint64_t arrival=0) { + return {.request_id=id,.sequence=id,.arrival_ns=arrival, + .logical_address=(id-1)*16384,.bytes=16384,.operation=0}; +} Result run(Profile profile,RuntimeMode mode){ auto wall=std::chrono::steady_clock::now(); ThermalModelConfig model; @@ -20,8 +24,7 @@ Result run(Profile profile,RuntimeMode mode){ metadata.evidence="ENGINEERING_FIXTURE_REQUEST_OCCUPANCY"; metadata.component_power_w={{"fixture_hbf",1}}; // declared 1W per admitted request, NOT measured adapter.bind(id,metadata); - engine.submit({.request_id=id,.sequence=id,.arrival_ns=id<5?0ULL:15000ULL, - .logical_address=(id-1)*16384,.bytes=16384,.operation=0}); + engine.submit(request(id,id<5?0ULL:15000ULL)); } Result result; for(std::uint64_t horizon=10000000;horizon<=1100000000;horizon+=10000000){ @@ -47,6 +50,175 @@ Result run(Profile profile,RuntimeMode mode){ require(observer.solver_constructed()==(mode==RuntimeMode::Shadow),"solver mode mismatch"); result.wall=std::chrono::duration(std::chrono::steady_clock::now()-wall).count();return result; } + +struct GateResult { + std::vector completions; + std::vector events; + std::uint64_t admitted_ns{},external_wait_ns{}; +}; + +GateResult gated(Profile profile,RuntimeMode mode) { + ThermalModelConfig model; + model.nodes={{"fixture_hbf",PhysicalType::Hbf,LogicalRole::CapacityMemory, + "hbf0",std::nullopt,2,300,0,1,300}}; + ActivityObserver observer(mode,model);MqsimOnlineEngine engine(profile); + MqsimObserverAdapter observation(engine,observer); + auto metadata=[] { + ObservedEvent value;value.operation="read";value.source="demand"; + value.physical_type="HBF";value.stack_id="UNKNOWN_REQUEST_LEVEL"; + value.evidence="ENGINEERING_FIXTURE_REQUEST_OCCUPANCY"; + value.component_power_w={{"fixture_hbf",1}};return value; + }; + observation.bind(101,metadata());observation.bind(102,metadata()); + constexpr std::uint64_t recovery_ns=200000000; + MqsimSubmissionGateAdapter gate(engine,MqsimGateMode::Enabled, + [=](const HbfRequest& value,std::uint64_t now) { + if(value.request_id==102&&now1&&attempt.backend_arrival_ns==engine.current_time_ns()&& + attempt.external_wait_ns==engine.current_time_ns()&& + observer.model()->temperatures_k().at(0)<300.1, + "thermal cooling/advice did not control real submission"); + std::optional completion; + while(engine.pending()) { + const auto horizon=engine.current_time_ns()+1000000000; + completion=engine.run_next_completion_until(horizon);observation.drain_to_current_time(); + if(!completion)require(engine.current_time_ns()==horizon, + "thermal-gated request lost before horizon"); + } + require(completion&&completion->request_id==401&&observer.terminals().size()==1, + "thermal-gated MQSim completion was not delivered once"); +} + +void gate_contract(Profile profile) { + { + HbfCompletion baseline{}; + { + MqsimOnlineEngine direct(profile);direct.submit(request(11)); + const auto completion=direct.run_next_completion(); + require(completion.has_value(),"direct MQSim request failed");baseline=*completion; + } + MqsimOnlineEngine bypassed(profile); + MqsimSubmissionGateAdapter off(bypassed); + const auto attempt=off.try_submit(request(11)); + const auto completion=bypassed.run_next_completion(); + require(attempt.submitted&&attempt.external_wait_ns==0&&completion&& + baseline.modeled_completion_ns==completion->modeled_completion_ns&& + baseline.modeled_ns==completion->modeled_ns,"disabled gate changed MQSim service"); + } + const auto read=gated(profile,RuntimeMode::ReadOnly); + const auto shadow=gated(profile,RuntimeMode::Shadow); + require(read.completions.size()==2&&shadow.completions.size()==2&& + read.events==shadow.events&&read.admitted_ns==200000000&& + read.external_wait_ns==200000000,"gate modes changed service/event contract"); + for(std::size_t i=0;i0&&!off.advice_samples&&!read.advice_samples,"shadow advice not consumed"); + gate_contract(p); std::cout<<"PASS CPU_PATH_VERIFIED actual MQSim6requests; off/read_only/shadow; GPU_LIVE_NOT_TESTED\n" + <<"PASS MQSIM_GATE default-off; nonblocking defer/recovery; maintenance=UNSUPPORTED_CAPABILITY\n" <<"wall_s off="< +#include + +namespace live { +namespace tf=hbfsim::timing_future; +constexpr unsigned instruction=17, ring=2; constexpr std::uint64_t generation=7; +enum class Case:unsigned {Ready,Timeout,Shutdown,Generation,Native,Mixed}; +struct Result {std::uint64_t value,elapsed,begin,target,reservation;unsigned issue,poll,status,state,mask,leader;}; +} + +#ifdef HBFSIM_LIVE_DEVICE_IMAGE +#include "../../src/cuda_runtime/device/hbf_device.cu" +extern "C" __device__ unsigned timing_future_live_wait_entered=0; +__device__ __forceinline__ std::uint64_t live_clock(){std::uint64_t n;asm volatile("mov.u64 %0, %%globaltimer;":"=l"(n)::"memory");return n;} +extern "C" __global__ void timing_future_wait_live(live::Case kind,const std::uint32_t* modeled, + const std::uint32_t* native_value,live::Result* out,std::uint64_t fault_delay) +{ + using namespace live;const unsigned lane=threadIdx.x; + if(blockIdx.x==1){if(lane||!(kind==Case::Shutdown||kind==Case::Generation))return; + while(atomicAdd(&timing_future_live_wait_entered,0U)==0U){} + const auto start=live_clock();while(live_clock()-start(__hbfsim_timing_future_config_v1.control_alias); + if(kind==Case::Shutdown)h->shutdown=1;else h->control_generation=generation+1;__threadfence_system();return;} + if(kind!=Case::Mixed&&lane)return; + const unsigned index=kind==Case::Mixed?(lane<16?0:32):0; + const auto* source=kind==Case::Native?native_value:modeled+index; + tf::TimingFutureLaneMetadataV1 meta{}; + auto f=__hbfsim_timing_future_issue_v1(reinterpret_cast(source),4,instruction,0,&meta); + const auto issue=f.status;const auto native=std::uint64_t(*source); + const auto poll=__hbfsim_timing_future_poll_v1(&f,&meta,instruction,4); + if(kind==Case::Shutdown||kind==Case::Generation)atomicExch(&timing_future_live_wait_entered,1U); + const auto begin=live_clock();const auto got=__hbfsim_timing_future_wait_v1(&f,&meta,native,instruction,4,0);const auto end=live_clock(); + out[lane]={native,end-begin,begin,kind==Case::Timeout?f.deadline_ns:f.ready_ns,f.reservation_id, + issue,poll,got.status,unsigned(got.state),meta.group_mask,meta.group_leader}; +} +#else +#include +#include +#include +#include +#include +#include +#include +#include +namespace { +using namespace live;using hbfsim::device::SharedControlHeader;using hbfsim::device::SharedRangeRecord; +using hbfsim::device::SharedRequestSlot;using hbfsim::device::SharedCompletionSlot;using hbfsim::device::PageEntry; +void ck(CUresult s,const char* w){if(s==CUDA_SUCCESS)return;const char* x="CUDA error";cuGetErrorString(s,&x);throw std::runtime_error(std::string(w)+": "+x);} +void req(bool v,const char* w){if(!v)throw std::runtime_error(w);} +std::size_t control_bytes(){return sizeof(SharedControlHeader)+sizeof(SharedRangeRecord)*hbfsim::device::kRangeCapacity+sizeof(SharedRequestSlot)*ring+sizeof(SharedCompletionSlot)*ring+sizeof(PageEntry)*ring;} +const char* cname(Case c){const char* n[]={"ready","timeout","shutdown_midwait","generation_midwait","native","mixed_groups"};return n[unsigned(c)];} +struct Fixture{ + CUmodule mod{};CUfunction fn{};CUdeviceptr control{},values{},native{},results{},traces{}; + Fixture(const char* path){std::ifstream f(path,std::ios::binary);req(bool(f),"open PTX");std::string p{std::istreambuf_iterator(f),{}};ck(cuModuleLoadData(&mod,p.c_str()),"load PTX");ck(cuModuleGetFunction(&fn,mod,"timing_future_wait_live"),"get kernel"); + ck(cuMemAlloc(&control,control_bytes()),"alloc control");ck(cuMemAlloc(&values,256),"alloc values");ck(cuMemAlloc(&native,4),"alloc native");ck(cuMemAlloc(&results,32*sizeof(Result)),"alloc results");ck(cuMemAlloc(&traces,128*sizeof(tf::Trace)),"alloc traces");} + ~Fixture(){if(traces)cuMemFree(traces);if(results)cuMemFree(results);if(native)cuMemFree(native);if(values)cuMemFree(values);if(control)cuMemFree(control);if(mod)cuModuleUnload(mod);} + std::pair sym(const char* n){CUdeviceptr p{};std::size_t z{};ck(cuModuleGetGlobal(&p,&z,mod,n),n);return {p,z};} + templatevoid put(const char* n,const T& v){auto[p,z]=sym(n);req(z==sizeof(v),"symbol size");ck(cuMemcpyHtoD(p,&v,sizeof(v)),"write symbol");} + tf::Counters counts(){tf::Counters c{};auto[p,z]=sym("__hbfsim_timing_future_counters_v1");req(z==sizeof(c),"counter size");ck(cuMemcpyDtoH(&c,p,sizeof(c)),"read counter");return c;} + void reset(std::uint64_t latency,std::uint64_t timeout){std::vectorb(control_bytes());auto*h=reinterpret_cast(b.data());h->magic=hbfsim::device::kControlMagic;h->abi_version=4;h->header_bytes=sizeof(*h);h->region_bytes=b.size();h->ring_capacity=ring;h->range_capacity=hbfsim::device::kRangeCapacity;h->page_capacity=ring;h->range_count=1;h->range_offset=sizeof(*h);h->request_offset=h->range_offset+sizeof(SharedRangeRecord)*hbfsim::device::kRangeCapacity;h->completion_offset=h->request_offset+sizeof(SharedRequestSlot)*ring;h->page_offset=h->completion_offset+sizeof(SharedCompletionSlot)*ring;h->heartbeat_ns=1;h->request_timeout_ns=timeout;h->heartbeat_timeout_ns=10000000000ULL;h->time_scale=1;h->control_generation=generation;h->read_latency_ns=latency;h->program_latency_ns=1;h->aggregate_bandwidth_bytes_per_s=1000000000ULL;h->timing_model=1; + auto*r=reinterpret_cast(b.data()+h->range_offset);*r={values,256,0,1,1,1,0,0,0,128,0};std::uint32_t v[64];for(unsigned i=0;i<64;i++)v[i]=0xabc00000U+i;const std::uint32_t seed=0x13579bdfU;ck(cuMemcpyHtoD(control,b.data(),b.size()),"write control");ck(cuMemcpyHtoD(values,v,sizeof(v)),"write values");ck(cuMemcpyHtoD(native,&seed,4),"write native");ck(cuMemsetD8(results,0,32*sizeof(Result)),"clear results");ck(cuMemsetD8(traces,0,128*sizeof(tf::Trace)),"clear trace"); + tf::ModuleConfig cfg{};cfg.enabled=1;cfg.control_alias=control;cfg.control_generation=generation;cfg.trace_address=traces;cfg.trace_capacity=128;tf::Counters c{};const unsigned long long a=control,g=generation;const unsigned wait_entered=0;put("__hbfsim_control",a);put("__hbfsim_control_generation",g);put("__hbfsim_timing_future_config_v1",cfg);put("__hbfsim_timing_future_counters_v1",c);put("timing_future_live_wait_entered",wait_entered);} + std::vector run(Case c,unsigned threads,std::uint64_t delay=0){unsigned blocks=c==Case::Shutdown||c==Case::Generation?2:1;void*args[]={&c,&values,&native,&results,&delay};ck(cuLaunchKernel(fn,blocks,1,1,threads,1,1,0,nullptr,args,nullptr),"launch");ck(cuCtxSynchronize(),"sync");std::vectoro(threads);ck(cuMemcpyDtoH(o.data(),results,o.size()*sizeof(Result)),"read results");return o;} +}; +void receipt(Case k,const Result&r,const tf::Counters&c){std::printf("{\"case\":\"%s\",\"issue\":%u,\"poll\":%u,\"status\":%u,\"state\":%u,\"value\":%llu,\"elapsed_ns\":%llu,\"begin_ns\":%llu,\"target_ns\":%llu,\"issued\":%llu,\"pending\":%llu,\"ready\":%llu,\"consumed\":%llu,\"errors\":%llu,\"groups_issued\":%llu,\"groups_completed\":%llu}\n",cname(k),r.issue,r.poll,r.status,r.state,(unsigned long long)r.value,(unsigned long long)r.elapsed,(unsigned long long)r.begin,(unsigned long long)r.target,(unsigned long long)c.issued,(unsigned long long)c.pending,(unsigned long long)c.model_ready,(unsigned long long)c.consumed,(unsigned long long)c.terminal_error,(unsigned long long)c.groups_issued,(unsigned long long)c.groups_completed);} +} +int main(int ac,char**av){CUcontext ctx{};try{req(ac==2,"usage: test DEVICE.ptx");ck(cuInit(0),"init");CUdevice d{};ck(cuDeviceGet(&d,0),"device");ck(cuCtxCreate(&ctx,nullptr,0,d),"context");Fixture f(av[1]); + auto one=[&](Case k,std::uint64_t l,std::uint64_t t,std::uint64_t fault=0){f.reset(l,t);auto o=f.run(k,1,fault);auto c=f.counts();receipt(k,o[0],c);return std::pair{o[0],c};}; + auto[ready,rc]=one(Case::Ready,2000000,20000000);req(ready.issue==tf::kPending&&ready.poll==tf::kPending,"ready not pending");req(ready.status==tf::kReady&&ready.state==unsigned(tf::State::Consumed)&&ready.value==0xabc00000U,"ready result");req(ready.begin>=ready.target||ready.elapsed>=ready.target-ready.begin,"early ready");req(rc.issued==1&&rc.model_ready==1&&rc.consumed==1&&rc.pending==0,"ready counters"); + auto[timed,tc]=one(Case::Timeout,20000000,2000000);req(timed.issue==tf::kPending&&timed.poll==tf::kPending,"timeout not pending");req(timed.status==tf::kTimeout&&timed.state==unsigned(tf::State::TerminalError),"timeout result");req(timed.begin>=timed.target||timed.elapsed>=timed.target-timed.begin,"early timeout"); + auto[shut,sc]=one(Case::Shutdown,20000000,50000000,1000000);req(shut.poll==tf::kPending&&shut.status==tf::kDaemonLost,"shutdown result");auto[gen,gc]=one(Case::Generation,20000000,50000000,1000000);req(gen.poll==tf::kPending&&gen.status==tf::kUnsupported,"generation result"); + auto[nat,nc]=one(Case::Native,2000000,20000000);req(nat.issue==tf::kReady&&nat.poll==tf::kReady&&nat.status==tf::kReady&&nat.value==0x13579bdfU&&nat.reservation==0&&nc.native_loads==1&&nc.issued==0,"native result"); + f.reset(2000000,20000000);auto m=f.run(Case::Mixed,32);auto mc=f.counts();receipt(Case::Mixed,m[0],mc);req(std::all_of(m.begin(),m.end(),[](const Result&r){return r.issue==tf::kPending&&r.poll==tf::kPending&&r.status==tf::kReady&&r.state==unsigned(tf::State::Consumed);}),"mixed states");for(unsigned i=0;i<32;i++)req(m[i].value==0xabc00000U+(i<16?0:32),"mixed data");req(m[0].mask==0xffff&&m[0].leader==0&&m[16].mask==0xffff0000U&&m[16].leader==16&&mc.issued==32&&mc.consumed==32&&mc.pending==0&&mc.groups_issued==2&&mc.groups_completed==2,"mixed accounting"); + std::puts("{\"status\":\"PASS\",\"device_storage_upper_mib\":7}");cuCtxDestroy(ctx);return 0;}catch(const std::exception&e){std::fprintf(stderr,"FAIL: %s\n",e.what());if(ctx)cuCtxDestroy(ctx);return 1;}} +#endif diff --git a/tests/integration/test_mqsim_service.py b/tests/integration/test_mqsim_service.py index 862a709..c05b1e7 100644 --- a/tests/integration/test_mqsim_service.py +++ b/tests/integration/test_mqsim_service.py @@ -24,8 +24,9 @@ def request(self, request_id, issue=0): return dict(request_id=request_id, issue_ns=issue, logical_address=(request_id-1)*16384, bytes=16384, operation='read') - def invoke(self, commands, *, success=True, extra=()): - result = subprocess.run([str(BINARY), '--profile', str(self.profile), *extra], + def invoke(self, commands, *, success=True, extra=(), profile=None): + selected_profile = self.profile if profile is None else Path(profile) + result = subprocess.run([str(BINARY), '--profile', str(selected_profile), *extra], input=''.join(json.dumps(c)+'\n' for c in commands), capture_output=True, text=True, timeout=20) self.assertEqual(result.returncode == 0, success, result.stderr) @@ -77,6 +78,174 @@ def test_bandwidth_bound_is_not_returned_early(self): self.assertEqual(replies[3]['now_ns'], 163840000) self.assertEqual(replies[3]['completion']['reported_complete'], 163840000) + def test_default_off_gate_service_parity(self): + direct, _ = self.invoke([dict(command='submit', requests=[self.request(1)]), + dict(command='until', deadline_ns=100000), dict(command='finish')]) + gated, _ = self.invoke([dict(command='try_submit', request=self.request(1)), + dict(command='until', deadline_ns=100000), dict(command='finish')]) + self.assertEqual(direct[0]['submission_gate'], 'OFF') + self.assertTrue(gated[1]['gate']['submitted']) + self.assertEqual(gated[1]['gate']['external_wait_ns'], 0) + self.assertEqual(direct[2]['completion'], gated[2]['completion']) + direct_events = [event for reply in direct for event in reply.get('events', [])] + gated_events = [event for reply in gated for event in reply.get('events', [])] + self.assertEqual(direct_events, gated_events) + + def test_enabled_gate_uses_existing_horizon_and_retry(self): + target = 200000 + commands = [dict(command='try_submit', request=self.request(1)), + dict(command='until', deadline_ns=target), + dict(command='try_submit', request=self.request(1)), + dict(command='until', deadline_ns=1000000), dict(command='finish')] + replies, _ = self.invoke(commands, extra=('--gate-not-before-ns', str(target))) + self.assertEqual(replies[0]['submission_gate'], 'ENGINEERING_FIXTURE_ENABLED') + self.assertFalse(replies[1]['gate']['submitted']) + self.assertEqual(replies[1]['gate']['target_ns'], target) + self.assertEqual(replies[2]['now_ns'], target) + self.assertTrue(replies[3]['gate']['submitted']) + self.assertEqual(replies[3]['gate']['backend_arrival_ns'], target) + self.assertEqual(replies[3]['gate']['external_wait_ns'], target) + self.assertEqual(replies[4]['completion']['request_id'], 1) + self.assertEqual(replies[-1]['issued'], 1) + self.assertEqual(replies[-1]['completed'], 1) + + def test_native_command_observation_off_on_parity_and_identity(self): + request = self.request(1) + request['bytes'] = 32768 # two LPAs: native MQSim maps them independently + commands = [dict(command='submit', requests=[request]), + dict(command='until', deadline_ns=1000000), dict(command='finish')] + off, _ = self.invoke(commands) + on, _ = self.invoke(commands, extra=('--native-command-observations', 'on')) + self.assertEqual(off[0]['native_command_observations'], 'OFF') + self.assertEqual(on[0]['native_command_observations'], 'ON') + self.assertEqual(off[2]['completion'], on[2]['completion']) + self.assertEqual([e for r in off for e in r['events']], + [e for r in on for e in r['events']]) + self.assertFalse([e for r in off for e in r['native_command_events']]) + native = [e for r in on for e in r['native_command_events']] + self.assertTrue(native) + commands_by_id = {} + saw_request = False + request_transaction_ids = set() + request_channels = set() + for event in native: + commands_by_id.setdefault(event['command_id'], []).append(event) + self.assertTrue(event['transactions']) + for transaction in event['transactions']: + self.assertIsNone(transaction['stack']) + self.assertTrue(transaction['logical_page'] is None or + isinstance(transaction['logical_page'], int)) + saw_request |= transaction['external_request_id'] == 1 + if transaction['external_request_id'] == 1: + request_transaction_ids.add(transaction['transaction_id']) + request_channels.add(transaction['channel']) + self.assertIsInstance(transaction['transaction_id'], int) + self.assertGreater(transaction['transaction_id'], 0) + self.assertTrue(saw_request) + self.assertGreaterEqual(len(request_transaction_ids), 2) + self.assertGreaterEqual(len(request_channels), 2) + for phases in commands_by_id.values(): + phase_ids = [event['phase'] for event in phases] + times = [event['time_ns'] for event in phases] + self.assertEqual(phase_ids[:3], [0, 1, 2]) + self.assertEqual(times, sorted(times)) + issued_transactions = {t['transaction_id'] for t in phases[0]['transactions']} + for event in phases[1:3]: + self.assertEqual({t['transaction_id'] for t in event['transactions']}, + issued_transactions) + + def make_small_eight_stack_fixture(self): + profile_path = Path(self.temp.name)/'eight-stack-engineering-profile.json' + profile = json.loads((ROOT/'configs/profiles/nominal.json').read_text()) + profile.update(name='ENGINEERING_FIXTURE_SMALL_8_STACK_1_DIE_PER_STACK', + capacity_bytes=128 << 20, hbm_cache_bytes=64 << 20, + channels=8, dies_per_channel=1, planes_per_die=1, + pages_per_block=256, queue_depth=16, time_scale=1) + profile_path.write_text(json.dumps(profile)) + # Deliberately permuted: success proves the consumer uses the explicit + # physical channel groups, rather than guessing channel == stack index. + channels = [4, 0, 5, 1, 6, 2, 7, 3] + mapping = dict(schema_version=1, physical_kind='HBF', route='direct', + address_layout='GLOBAL_PAGE_STRIPE_V1', + plane_allocation_scheme='CWDP', page_bytes=profile['page_bytes'], + channels=8, dies_per_channel=1, + evidence='ENGINEERING_FIXTURE_SMALL_8_STACK_1_DIE_PER_STACK', + stacks=[dict(id=f'hbf{i}', declared_dies=1, channels=[channel]) + for i, channel in enumerate(channels)]) + map_path = Path(self.temp.name)/'eight-stack-map.json' + map_path.write_text(json.dumps(mapping)) + return profile_path, map_path, channels + + def test_explicit_eight_hbf_stack_map_reaches_native_channels(self): + profile_path, map_path, channels = self.make_small_eight_stack_fixture() + requests = [] + for i in range(8): + row = dict(request_id=i+1, issue_ns=0, stack_local_page=0, + bytes=16384, operation='read', stack=f'hbf{i}', route='direct') + requests.append(row) + commands = [dict(command='submit', requests=requests)] + commands.extend(dict(command='until', deadline_ns=10_000_000) for _ in requests) + commands.append(dict(command='finish')) + replies, _ = self.invoke(commands, profile=profile_path, + extra=('--stack-map', str(map_path))) + self.assertEqual(replies[0]['stack_mapping'], + 'ACTUAL_MQSIM_CHANNEL_PARTITIONED_HBF_STACKS') + self.assertEqual(replies[0]['stack_mapping_evidence'], + 'ENGINEERING_FIXTURE_CONFIGURATION_NOT_RESEARCH_GEOMETRY') + self.assertEqual(replies[0]['native_command_observations'], 'ON') + self.assertEqual(replies[-1]['issued'], 8) + self.assertEqual(replies[-1]['completed'], 8) + completions = [r['completion']['request_id'] for r in replies + if r.get('completion') is not None] + self.assertEqual(set(completions), set(range(1, 9))) + self.assertEqual(len(completions), 8) + demand = {} + for reply in replies: + for event in reply.get('native_command_events', []): + for transaction in event['transactions']: + request_id = transaction['external_request_id'] + if request_id in range(1, 9): + demand.setdefault(request_id, transaction) + self.assertEqual(transaction['stack'], f'hbf{request_id-1}') + self.assertEqual(transaction['expected_stack'], f'hbf{request_id-1}') + self.assertEqual(transaction['channel'], channels[request_id-1]) + self.assertEqual(transaction['external_logical_page'], request_id-1) + self.assertEqual(transaction['backend_logical_page'], channels[request_id-1]) + self.assertEqual(set(demand), set(range(1, 9))) + + def test_enabled_stack_map_rejects_mismatch_and_multi_page(self): + profile_path, map_path, _ = self.make_small_eight_stack_fixture() + mismatch = dict(request_id=1, issue_ns=0, logical_address=0, stack_local_page=0, + bytes=16384, operation='read', stack='hbf1', route='direct') + multi_page = dict(request_id=1, issue_ns=0, stack_local_page=0, bytes=32768, + operation='read', stack='hbf0', route='direct') + for row in (mismatch, multi_page): + with self.subTest(row=row): + self.invoke([dict(command='submit', requests=[row])], success=False, + profile=profile_path, extra=('--stack-map', str(map_path))) + + def test_stack_map_configuration_rejects_profile_overlap_and_incomplete_groups(self): + profile_path, map_path, _ = self.make_small_eight_stack_fixture() + base = json.loads(map_path.read_text()) + variants = [] + profile_mismatch = json.loads(json.dumps(base)) + profile_mismatch['channels'] = 9 + variants.append((profile_mismatch, 'stack map does not match')) + overlap = json.loads(json.dumps(base)) + overlap['stacks'][1]['channels'] = overlap['stacks'][0]['channels'] + variants.append((overlap, 'overlapping or invalid mqsim channel group')) + incomplete = json.loads(json.dumps(base)) + incomplete['stacks'].pop() + variants.append((incomplete, 'invalid mqsim stack-map geometry')) + for index, (config, expected_error) in enumerate(variants): + with self.subTest(index=index): + invalid_path = Path(self.temp.name)/f'invalid-stack-map-{index}.json' + invalid_path.write_text(json.dumps(config)) + _, failure = self.invoke([dict(command='finish')], success=False, + profile=profile_path, + extra=('--stack-map', str(invalid_path))) + self.assertIn(expected_error, failure.stderr.lower()) + if __name__ == '__main__': unittest.main() diff --git a/tests/integration/test_timing_future_device_ptx.py b/tests/integration/test_timing_future_device_ptx.py index 907e5fe..0dd1495 100644 --- a/tests/integration/test_timing_future_device_ptx.py +++ b/tests/integration/test_timing_future_device_ptx.py @@ -29,9 +29,13 @@ def check_executable_logic(helper): assert 'isspacep.local' in body,'kernel-local token/metadata validation' assert '%globaltimer' in body and 'ld.acquire.sys' in body,'fresh clock and binding/liveness reads' labels={m[1]:m.start() for m in re.finditer(r'(?m)^([\w$]+):',wait)} - backedges=[(labels[m[1]],m.start()) for m in re.finditer(r'\bbra\s+([\w$]+)\s*;',wait) + backedges=[(labels[m[1]],m.start()) for m in re.finditer(r'\bbra(?:\.uni)?\s+([\w$]+)\s*;',wait) if m[1] in labels and labels[m[1]] sequence[-1][0]: + raise ValueError("window is outside available data") + result = [(start, interpolate(sequence, start))] + result.extend((time_s, value) for time_s, value in sequence if start < time_s < end) + result.append((end, interpolate(sequence, end))) + return result + + +def crossing_events(sequence, threshold): + events = [] + for left, right in zip(sequence, sequence[1:]): + t0, v0 = left + time_s, value = right + if (v0 < threshold <= value) or (value < threshold <= v0): + events.append({"direction": "up" if value > v0 else "down", + "time_s": t0 + (time_s - t0) * + (threshold - v0) / (value - v0)}) + return events + + +def crossing_score(reference, candidate, threshold, uncertainty_k, + minimum_time_s, fraction): + reference_events = crossing_events(reference, threshold) + candidate_events = crossing_events(candidate, threshold) + ambiguous = any(abs(left[1] - threshold) <= uncertainty_k and + abs(right[1] - threshold) <= uncertainty_k + for left, right in zip(reference, reference[1:])) + if not reference_events and not candidate_events: + legacy_status = "NOT_APPLICABLE" + elif len(reference_events) != len(candidate_events): + legacy_status = "FAIL" + else: + legacy_status = "PASS" + for expected, actual in zip(reference_events, candidate_events): + allowed = max(minimum_time_s, fraction * expected["time_s"]) + if (expected["direction"] != actual["direction"] or + abs(expected["time_s"] - actual["time_s"]) > allowed): + legacy_status = "FAIL" + break + status = "THRESHOLD_AMBIGUOUS" if ambiguous else legacy_status + return {"threshold_k": threshold, "uncertainty_band_k": uncertainty_k, + "reference": reference_events, "candidate": candidate_events, + "status": status, "legacy_v1_status": legacy_status, + "ambiguity_rule": "two consecutive reference observations within threshold +/- L", + "ambiguity_rule_basis": "ENGINEERING_INFERENCE", + "ambiguity_is_global_numeric_veto": False} + + +def sensor_window_score(reference, candidate, window, is_hotspot, is_control, + thresholds, crossing_min_time_s, crossing_fraction): + start, end = window["start_s"], window["end_s"] + reference = clip(reference, start, end) + candidate = clip(candidate, start, end) + if [item[0] for item in reference] != [item[0] for item in candidate]: + raise ValueError("reference and candidate window timestamps differ") + signed_errors = [right[1] - left[1] for left, right in zip(reference, candidate)] + errors = [abs(value) for value in signed_errors] + interval_areas = [] + for index in range(len(errors) - 1): + duration_s = reference[index + 1][0] - reference[index][0] + left, right = signed_errors[index], signed_errors[index + 1] + if left * right >= 0: + area = duration_s * (abs(left) + abs(right)) * .5 + else: + magnitudes = abs(left) + abs(right) + area = duration_s * (left * left + right * right) / (2 * magnitudes) + interval_areas.append(area) + integral = math.fsum(interval_areas) + duration = end - start + mae = integral / duration + arithmetic_mae = math.fsum(errors) / len(errors) + amplitude = max(value for _, value in reference) - min(value for _, value in reference) + limit = min(V2_ABSOLUTE_CAP_K, V2_BASE_K + V2_AMPLITUDE_FRACTION * amplitude) + maximum = max(errors) + crossings = [crossing_score(reference, candidate, threshold, limit, + crossing_min_time_s, crossing_fraction) + for threshold in thresholds] + crossing_pass = all(item["legacy_v1_status"] != "FAIL" for item in crossings) + maximum_pass = (not (is_hotspot or is_control) or + maximum <= MAX_HOTSPOT_CONTROL_ERROR_K) + return {"window_id": window["id"], "window_kind": window["kind"], + "start_s": start, "end_s": end, "duration_s": duration, + "reference_amplitude_k": amplitude, "time_weighted_mae_k": mae, + "v2_mae_limit_k": limit, "max_abs_error_k": maximum, + "is_grid_hotspot": is_hotspot, "is_control_sensor": is_control, + "hotspot_control_max_limit_k": MAX_HOTSPOT_CONTROL_ERROR_K, + "crossings": crossings, + "v2_pass": mae <= limit and maximum_pass and crossing_pass, + "sample_arithmetic_mae_k_for_comparison": arithmetic_mae} + + +def exact_legacy_v1(reference, candidate, initial_time_s, initial_by_sensor, + thresholds, crossing_min_time_s, crossing_fraction): + """Reproduce the historical full-trajectory arithmetic score exactly.""" + rows = [] + for sensor in sorted(reference): + ref = reference[sensor] + cand = candidate[sensor] + errors = [abs(left[1] - right[1]) for left, right in zip(ref, cand)] + arithmetic_mae = math.fsum(errors) / len(errors) + amplitude = max(value for _, value in ref) - min(value for _, value in ref) + maximum = max(errors) + initial = initial_by_sensor[sensor] + crossings = [crossing_score( + with_initial(ref, initial_time_s, initial), + with_initial(cand, initial_time_s, initial), threshold, + 0.0, crossing_min_time_s, crossing_fraction) + for threshold in thresholds] + crossing_pass = all(item["legacy_v1_status"] != "FAIL" + for item in crossings) + normalized = arithmetic_mae / max(amplitude, 1.0) + passed = (arithmetic_mae <= 1.0 and normalized <= LEGACY_NMAE_LIMIT and + ("hotspot" not in sensor or maximum <= MAX_HOTSPOT_CONTROL_ERROR_K) and + crossing_pass) + rows.append({"sensor_id": sensor, + "mean_semantics": "sample_arithmetic_mean_without_synthetic_initial", + "mae_k": arithmetic_mae, "mae_limit_k": 1.0, + "reference_amplitude_k": amplitude, + "normalized_mae": normalized, + "normalized_mae_limit": LEGACY_NMAE_LIMIT, + "max_abs_error_k": maximum, + "hotspot_max_limit_k": MAX_HOTSPOT_CONTROL_ERROR_K, + "crossings": crossings, "pass": passed}) + return {"status": "PASS" if all(row["pass"] for row in rows) + else "NUMERICAL_FAIL", + "exact_historical_formula": True, + "retained_not_v2_veto": True, + "sensors": rows} + + +def energy_score(receipt): + try: + input_j = _finite(receipt["total_input_energy_j"], "total_input_energy_j") + stored_j = _finite(receipt["stored_energy_change_j"], "stored_energy_change_j") + boundary_j = _finite(receipt["boundary_loss_j"], "boundary_loss_j") + except KeyError as error: + raise ValueError(f"energy receipt missing {error.args[0]}") from error + residual_j = input_j - boundary_j - stored_j + relative = abs(residual_j) / max(input_j, 1.0) + return {"total_input_energy_j": input_j, "boundary_loss_j": boundary_j, + "stored_energy_change_j": stored_j, "residual_j": residual_j, + "relative_residual": relative, "limit": ENERGY_RELATIVE_LIMIT, + "pass": relative <= ENERGY_RELATIVE_LIMIT} + + +def analyze(reference_path, candidate_path, method, energy_receipt): + reference = read_series(reference_path) + candidate = read_series(candidate_path) + if set(reference) != set(candidate): + raise ValueError("sensor coverage differs") + for sensor in reference: + if [item[0] for item in reference[sensor]] != [item[0] for item in candidate[sensor]]: + raise ValueError(f"timestamp coverage differs for {sensor}") + checked = validate_method(method, set(reference)) + scores = [] + for sensor in sorted(reference): + initial = checked["initial_by_sensor"][sensor] + ref = with_initial(reference[sensor], checked["initial_time_s"], initial) + cand = with_initial(candidate[sensor], checked["initial_time_s"], initial) + for window in checked["windows"]: + score = sensor_window_score( + ref, cand, window, sensor in checked["hotspots"], + sensor in checked["controls"], checked["thresholds"], + checked["crossing_min_time_s"], checked["crossing_fraction"]) + score["sensor_id"] = sensor + scores.append(score) + energy = energy_score(energy_receipt) + ambiguous = any(crossing["status"] == "THRESHOLD_AMBIGUOUS" + for score in scores for crossing in score["crossings"]) + v2_pass = energy["pass"] and all(score["v2_pass"] for score in scores) + legacy = exact_legacy_v1( + reference, candidate, checked["initial_time_s"], + checked["initial_by_sensor"], checked["thresholds"], + checked["crossing_min_time_s"], checked["crossing_fraction"]) + legacy["energy_reported_separately"] = energy + status = ("PASS_WITH_THRESHOLD_AMBIGUITY" if v2_pass and ambiguous else + "PASS" if v2_pass else "NUMERICAL_FAIL_WITH_THRESHOLD_AMBIGUITY" + if ambiguous else "NUMERICAL_FAIL") + return {"schema_version": "eq3-acceptance-v2-result-v1", + "analysis_kind": "fast_against_selected_reference", + "status": status, "physical_calibration": False, + "reference_qualification_inherited": False, + "method": method, "energy": energy, "scores": scores, + "legacy_v1": legacy} + + +def main(): + parser = argparse.ArgumentParser() + parser.add_argument("--reference", type=Path, required=True) + parser.add_argument("--candidate", type=Path, required=True) + parser.add_argument("--method", type=Path, required=True) + parser.add_argument("--energy-receipt", type=Path, required=True) + parser.add_argument("--output", type=Path, required=True) + args = parser.parse_args() + method = json.loads(args.method.read_text()) + energy = json.loads(args.energy_receipt.read_text()) + result = analyze(args.reference, args.candidate, method, energy) + with args.output.open("x") as output: + json.dump(result, output, indent=2) + print(json.dumps({"status": result["status"], + "legacy_v1": result["legacy_v1"]["status"], + "sensor_window_scores": len(result["scores"])})) + + +if __name__ == "__main__": + main() diff --git a/tools/eq3_acceptance_v3.py b/tools/eq3_acceptance_v3.py new file mode 100644 index 0000000..6be6212 --- /dev/null +++ b/tools/eq3_acceptance_v3.py @@ -0,0 +1,322 @@ +"""Retrospective EQ3 acceptance v3; consumes existing CSVs and never solves.""" +import argparse +import csv +import json +import math +from decimal import Decimal, ROUND_HALF_EVEN +from pathlib import Path + +from eq3_acceptance_v2 import (ENERGY_RELATIVE_LIMIT, MAX_HOTSPOT_CONTROL_ERROR_K, + V2_ABSOLUTE_CAP_K, V2_AMPLITUDE_FRACTION, + V2_BASE_K, energy_score) + +NS_PER_S = Decimal("1000000000") + + +def _finite(value, label): + value = float(value) + if not math.isfinite(value): + raise ValueError(f"{label} must be finite") + return value + + +def _ns(value, label): + raw = Decimal(str(value)) * NS_PER_S + tick = int(raw.to_integral_value(rounding=ROUND_HALF_EVEN)) + if abs(raw - tick) > Decimal("0.5"): + raise ValueError(f"{label} cannot be normalized to integer ns") + return tick + + +def read_series(path): + result = {} + with path.open(newline="") as source: + reader = csv.DictReader(source) + required = {"time_s", "sensor_id", "temperature_k"} + if not reader.fieldnames or not required.issubset(reader.fieldnames): + raise ValueError("sensor CSV must contain time_s,sensor_id,temperature_k in K") + for row in reader: + sensor = row["sensor_id"] + if not sensor: + raise ValueError("sensor_id must not be empty") + time_ns = _ns(row["time_s"], "time_s") + temperature = _finite(row["temperature_k"], "temperature_k") + sequence = result.setdefault(sensor, []) + if sequence and time_ns <= sequence[-1][0]: + raise ValueError(f"non-increasing or duplicate normalized time for {sensor}") + sequence.append((time_ns, temperature)) + if not result: + raise ValueError("sensor CSV is empty") + return result + + +def validate_method(method, sensors): + if method.get("schema_version") != "eq3-acceptance-v3-method-v1": + raise ValueError("unsupported acceptance v3 method schema") + if method.get("temperature_unit") != "K" or method.get("time_unit") != "ns": + raise ValueError("v3 requires temperature_unit K and time_unit ns") + ids = method.get("sensor_ids") + if not isinstance(ids, list) or not ids or len(ids) != len(set(ids)) or set(ids) != set(sensors): + raise ValueError("CSV sensor coverage differs from unique method sensor_ids") + initial_ns = _ns(method.get("initial_time_s"), "initial_time_s") + initial = method.get("initial_temperature_k") + if isinstance(initial, dict): + if set(initial) != set(sensors): + raise ValueError("per-sensor initial temperature coverage differs") + initials = {key: _finite(value, "initial_temperature_k") for key, value in initial.items()} + else: + value = _finite(initial, "initial_temperature_k") + initials = {sensor: value for sensor in sensors} + windows, seen, kinds = [], set(), set() + for window in method.get("windows", []): + wid, kind = window.get("id"), window.get("kind") + start, end = _ns(window.get("start_s"), "window start_s"), _ns(window.get("end_s"), "window end_s") + if not wid or wid in seen or kind not in {"full", "excitation", "cooling"} or not initial_ns <= start < end: + raise ValueError("invalid or duplicate v3 window") + windows.append({"id": wid, "kind": kind, "start_ns": start, "end_ns": end}) + seen.add(wid); kinds.add(kind) + if not {"full", "excitation", "cooling"}.issubset(kinds): + raise ValueError("full, excitation, and cooling windows must be preregistered") + hotspots, controls = set(method.get("hotspot_sensor_ids", [])), set(method.get("control_sensor_ids", [])) + if not hotspots or not hotspots.issubset(sensors) or not controls.issubset(sensors): + raise ValueError("invalid hotspot/control sensor coverage") + thresholds = [_finite(x, "crossing threshold") for x in method.get("crossing_thresholds_k", [])] + if not thresholds: + raise ValueError("crossing_thresholds_k must be nonempty") + q_ref = _finite(method.get("reference_quantization_k", method.get("quantization_k")), + "reference_quantization_k") + q_cand = _finite(method.get("candidate_quantization_k", method.get("quantization_k")), + "candidate_quantization_k") + if q_ref < 0 or q_cand < 0 or method.get("quantization_mode") != "nearest_rounding": + raise ValueError("v3 requires nonnegative nearest-rounding quantization") + minimum_ns = _ns(method.get("crossing_min_time_s"), "crossing_min_time_s") + fraction = _finite(method.get("crossing_fraction"), "crossing_fraction") + if minimum_ns < 0 or fraction < 0: + raise ValueError("crossing tolerances must be nonnegative") + return {"initial_ns": initial_ns, "initials": initials, "windows": windows, + "hotspots": hotspots, "controls": controls, "thresholds": thresholds, + "q_ref": q_ref, "q_cand": q_cand, + "minimum_ns": minimum_ns, "fraction": fraction} + + +def with_initial(sequence, time_ns, temperature): + if sequence[0][0] < time_ns: + raise ValueError("CSV begins before declared initial time") + if sequence[0][0] == time_ns: + if sequence[0][1] != temperature: + raise ValueError("CSV initial temperature differs from declared state") + return sequence + return [(time_ns, temperature), *sequence] + + +def _side(value, threshold, half_q): + if value + half_q < threshold: + return -1 + if value - half_q > threshold: + return 1 + return 0 + + +def crossing_events(sequence, threshold, quantization_k): + """Extract certain directed events once across the complete trajectory.""" + half = quantization_k / 2.0 + sides = [_side(value, threshold, half) for _, value in sequence] + events, indeterminate = [], [] + last_definite = None + ambiguous_start = None + event_id = 0 + for index, side in enumerate(sides): + if side == 0: + if ambiguous_start is None: + ambiguous_start = index + continue + if last_definite is not None and side != sides[last_definite]: + event_id += 1 + left_time, left_value = sequence[last_definite] + right_time, right_value = sequence[index] + possible = [] + for left in (left_value-half, left_value+half): + for right in (right_value-half, right_value+half): + if right == left: + continue + fraction = (threshold-left)/(right-left) + if 0 <= fraction <= 1: + possible.append(left_time+(right_time-left_time)*fraction) + if not possible: + raise AssertionError("opposite definite sides must bracket threshold") + events.append({"event_id": event_id, + "direction": "up" if side > 0 else "down", + "time_lo_ns": math.floor(min(possible)), + "time_hi_ns": math.ceil(max(possible)), + "quantization_bridged": ambiguous_start is not None}) + elif ambiguous_start is not None: + # The trace returned to the same certain side; a crossing pair may + # have occurred inside the quantization band and cannot be scored. + indeterminate.append({"time_lo_ns": sequence[ambiguous_start][0], + "time_hi_ns": sequence[index][0], + "reason": "threshold_overlap_returned_to_same_side"}) + last_definite = index + ambiguous_start = None + if ambiguous_start is not None: + indeterminate.append({"time_lo_ns": sequence[ambiguous_start][0], + "time_hi_ns": sequence[-1][0], + "reason": "threshold_overlap_at_trajectory_end"}) + if last_definite is None: + indeterminate = [{"time_lo_ns": sequence[0][0], "time_hi_ns": sequence[-1][0], + "reason": "entire_trajectory_overlaps_threshold"}] + return events, indeterminate + + +def _interval_gap(left, right): + return max(0, left["time_lo_ns"] - right["time_hi_ns"], + right["time_lo_ns"] - left["time_hi_ns"]) + + +def _window_assignment(lo, hi, windows): + terminal = max(window["end_ns"] for window in windows) + certain, possible = [], [] + for window in windows: + closed_end = window["end_ns"] == terminal + contains_hi = hi <= window["end_ns"] if closed_end else hi < window["end_ns"] + if window["start_ns"] <= lo and contains_hi: + certain.append(window["id"]) + elif hi >= window["start_ns"] and (lo < window["end_ns"] or (closed_end and lo == window["end_ns"])): + possible.append(window["id"]) + return {"certain_window_ids": certain, "possible_window_ids": possible, + "status": ("INDETERMINATE_WINDOW_ASSIGNMENT" if possible else + "ASSIGNED" if certain else "OUTSIDE_DECLARED_WINDOWS"), + "window_semantics": "[start,end), global terminal end included"} + + +def crossing_score(reference, candidate, threshold, q_ref, q_cand, + minimum_ns, fraction, windows): + ref, ref_ind = crossing_events(reference, threshold, q_ref) + cand, cand_ind = crossing_events(candidate, threshold, q_cand) + matched, failure = [], None + # A count difference is definite only when neither trace contains an + # unresolved threshold-band excursion. Otherwise quantization could hide + # the event, so the result must remain indeterminate. + if len(ref) != len(cand) and not (ref_ind or cand_ind): + failure = "DEFINITE_MISSING_OR_EXTRA_CROSSING" + elif len(ref) == len(cand): + for expected, actual in zip(ref, cand): + midpoint = (expected["time_lo_ns"] + expected["time_hi_ns"]) // 2 + allowed = max(minimum_ns, int(round(fraction * midpoint))) + gap = _interval_gap(expected, actual) + if expected["direction"] != actual["direction"]: + failure = "WRONG_DIRECTION" + elif gap > allowed: + failure = "CROSSING_TIMEOUT" + lo = min(expected["time_lo_ns"], actual["time_lo_ns"]) + hi = max(expected["time_hi_ns"], actual["time_hi_ns"]) + matched.append({"reference_event_id": expected["event_id"], + "candidate_event_id": actual["event_id"], + "direction": expected["direction"], "interval_gap_ns": gap, + "allowed_ns": allowed, "reference": expected, "candidate": actual, + "window_assignment": _window_assignment(lo, hi, windows)}) + if failure: + break + window_ambiguous = any(row["window_assignment"]["status"].startswith("INDETERMINATE") for row in matched) + if failure: + status = "FAIL" + elif ref_ind or cand_ind or window_ambiguous: + status = "INDETERMINATE_QUANTIZATION" + else: + status = "PASS" if ref else "NOT_APPLICABLE" + return {"threshold_k": threshold, "reference_quantization_k": q_ref, + "candidate_quantization_k": q_cand, "status": status, + "failure_reason": failure, "reference_events": ref, "candidate_events": cand, + "reference_indeterminate": ref_ind, "candidate_indeterminate": cand_ind, + "matches": matched, "extraction_scope": "complete_trajectory_once"} + + +def _interpolate(sequence, tick): + for t, value in sequence: + if t == tick: + return value + for left, right in zip(sequence, sequence[1:]): + if left[0] < tick < right[0]: + fraction = (tick - left[0]) / (right[0] - left[0]) + return left[1] + fraction * (right[1] - left[1]) + raise ValueError("window boundary outside available data") + + +def window_score(reference, candidate, window, is_hotspot, is_control): + start, end = window["start_ns"], window["end_ns"] + if start < reference[0][0] or start < candidate[0][0] or end > reference[-1][0] or end > candidate[-1][0]: + raise ValueError("window outside available data") + ticks = sorted({start, end, *(t for t, _ in reference if start < t < end), + *(t for t, _ in candidate if start < t < end)}) + refs, cands = [_interpolate(reference, t) for t in ticks], [_interpolate(candidate, t) for t in ticks] + signed = [b-a for a, b in zip(refs, cands)] + area = 0.0 + for index, (left, right) in enumerate(zip(signed, signed[1:])): + duration = (ticks[index+1]-ticks[index]) / 1e9 + if left * right >= 0: + area += duration * (abs(left)+abs(right)) / 2 + else: + area += duration * (left*left+right*right) / (2*(abs(left)+abs(right))) + duration = (end-start)/1e9 + mae = area/duration + amplitude = max(refs)-min(refs) + limit = min(V2_ABSOLUTE_CAP_K, V2_BASE_K+V2_AMPLITUDE_FRACTION*amplitude) + maximum = max(abs(x) for x in signed) + passed = mae <= limit and (not (is_hotspot or is_control) or maximum <= MAX_HOTSPOT_CONTROL_ERROR_K) + return {"window_id": window["id"], "window_kind": window["kind"], + "start_ns": start, "end_ns": end, "time_weighted_mae_k": mae, + "v2_mae_limit_k": limit, "reference_amplitude_k": amplitude, + "max_abs_error_k": maximum, "hotspot_control_max_limit_k": MAX_HOTSPOT_CONTROL_ERROR_K, + "temperature_pass": passed, "comparison_grid": "union_of_integer_ns_sample_times"} + + +def analyze(reference_path, candidate_path, method, energy_receipt): + reference, candidate = read_series(reference_path), read_series(candidate_path) + if set(reference) != set(candidate): + raise ValueError("sensor coverage differs") + checked = validate_method(method, reference) + scores, crossings = [], [] + for sensor in sorted(reference): + ref = with_initial(reference[sensor], checked["initial_ns"], checked["initials"][sensor]) + cand = with_initial(candidate[sensor], checked["initial_ns"], checked["initials"][sensor]) + for window in checked["windows"]: + row = window_score(ref, cand, window, sensor in checked["hotspots"], sensor in checked["controls"]) + row["sensor_id"] = sensor; scores.append(row) + for threshold in checked["thresholds"]: + row = crossing_score(ref, cand, threshold, checked["q_ref"], checked["q_cand"], + checked["minimum_ns"], checked["fraction"], checked["windows"]) + row["sensor_id"] = sensor; crossings.append(row) + energy = energy_score(energy_receipt) + numeric_fail = not energy["pass"] or any(not row["temperature_pass"] for row in scores) + crossing_fail = any(row["status"] == "FAIL" for row in crossings) + indeterminate = any(row["status"] == "INDETERMINATE_QUANTIZATION" for row in crossings) + status = ("NUMERICAL_FAIL" if numeric_fail else "CROSSING_FAIL" if crossing_fail else + "INDETERMINATE_QUANTIZATION" if indeterminate else "PASS") + return {"schema_version": "eq3-acceptance-v3-result-v1", "status": status, + "analysis_kind": "RETROSPECTIVE_EXISTING_RAW", "physical_calibration": False, + "reference_qualification_inherited": False, "method": method, "energy": energy, + "temperature_scores": scores, "crossing_scores": crossings, + "unchanged_limits": {"v2_absolute_cap_k": V2_ABSOLUTE_CAP_K, + "v2_base_k": V2_BASE_K, + "v2_amplitude_fraction": V2_AMPLITUDE_FRACTION, + "hotspot_control_max_k": MAX_HOTSPOT_CONTROL_ERROR_K, + "energy_relative_limit": ENERGY_RELATIVE_LIMIT}} + + +def main(): + parser = argparse.ArgumentParser() + parser.add_argument("--reference", type=Path, required=True) + parser.add_argument("--candidate", type=Path, required=True) + parser.add_argument("--method", type=Path, required=True) + parser.add_argument("--energy-receipt", type=Path, required=True) + parser.add_argument("--output", type=Path, required=True) + args = parser.parse_args() + result = analyze(args.reference, args.candidate, json.loads(args.method.read_text()), + json.loads(args.energy_receipt.read_text())) + with args.output.open("x") as output: + json.dump(result, output, indent=2) + print(json.dumps({"status": result["status"], "temperature_scores": len(result["temperature_scores"]), + "crossing_scores": len(result["crossing_scores"])})) + + +if __name__ == "__main__": + main() diff --git a/tools/eq3_all_source_cap.py b/tools/eq3_all_source_cap.py new file mode 100644 index 0000000..3feb835 --- /dev/null +++ b/tools/eq3_all_source_cap.py @@ -0,0 +1,132 @@ +"""Build an auditable all-source-cap event file without reading thermal output.""" +import argparse +import hashlib +import json +import math +from pathlib import Path + +from eq3_layered_ir import normalize + + +def sha256(path): + return hashlib.sha256(path.read_bytes()).hexdigest() + + +def number(value): + return format(float(value), ".17g") + + +def build(profile, power, grid): + order = power.get("group_order") + caps = power.get("caps_W") + if (not isinstance(order, list) or not isinstance(caps, list) or + len(order) != 17 or len(caps) != len(order) or len(set(order)) != len(order)): + raise ValueError("frozen all-source cap requires 17 unique ordered groups and caps") + if any(isinstance(value, bool) or not isinstance(value, (int, float)) or + not math.isfinite(value) or value < 0 for value in caps): + raise ValueError("caps_W must be finite and nonnegative") + slot_s = float(power["slot_s"]) + inline = {"id": "all_source_cap", "duration_s": slot_s, + "slots_W": [dict(zip(order, caps, strict=True))]} + ir = normalize(profile, power, inline) + interval = ir["power"]["intervals"] + if len(interval) != 1 or interval[0]["start_s"] != 0 or interval[0]["end_s"] != slot_s: + raise ValueError("all-source normalization must produce exactly one cap slot") + cells = grid.get("cells") + component_cells = grid.get("component_cells") + if not isinstance(cells, list) or not isinstance(component_cells, dict): + raise ValueError("RC grid lacks cells/component_cells") + components = {item["id"]: item for item in ir["components"]} + component_power = interval[0]["power_w"] + events = ["HBFSIM_EQ3_THERMAL_EVENTS 1"] + emitted_by_component = {} + node_cap_w = {} + used_indices = set() + for event_id, component_id in enumerate(sorted(component_power), 1): + watts = float(component_power[component_id]) + indices = component_cells.get(component_id) + if not isinstance(indices, list) or not indices: + raise ValueError(f"RC grid lacks powered component {component_id}") + if (any(isinstance(index, bool) or not isinstance(index, int) or + index < 0 or index >= len(cells) for index in indices) or + used_indices.intersection(indices)): + raise ValueError(f"RC grid has invalid or multiply owned cells for {component_id}") + used_indices.update(indices) + if any(cells[index].get("component") != component_id for index in indices): + raise ValueError(f"RC grid component ownership mismatch for {component_id}") + expected_volume = math.prod(components[component_id]["size_m"]) + actual_volume = math.fsum(float(cells[index]["volume_m3"]) for index in indices) + if not math.isclose(actual_volume, expected_volume, rel_tol=1e-10, abs_tol=1e-24): + raise ValueError(f"RC grid volume mismatch for {component_id}") + assignments = [] + emitted = 0.0 + for index in indices: + cell = cells[index] + energy = watts * slot_s * float(cell["volume_m3"]) / expected_volume + emitted += energy + node_cap_w[str(cell["id"])] = node_cap_w.get(str(cell["id"]), 0.0) + energy / slot_s + assignments.extend((str(cell["id"]), number(energy))) + emitted_by_component[component_id] = emitted / slot_s + events.append(" ".join([ + "activity", str(event_id), str(event_id), "external_heat", "external", + "0", "-1", "-1", "-1", "0", number(slot_s), number(slot_s), + *assignments])) + group_members = {} + for component in ir["components"]: + group = component.get("power_group") + if group in order: + group_members.setdefault(group, []).append(component["id"]) + for group, cap in zip(order, caps, strict=True): + actual = math.fsum(emitted_by_component.get(item, 0.0) + for item in group_members.get(group, [])) + if not math.isclose(actual, float(cap), rel_tol=1e-10, abs_tol=1e-10): + raise ValueError(f"emitted cap mismatch for group {group}") + total = math.fsum(emitted_by_component.values()) + expected_total = math.fsum(float(value) for value in caps) + if not math.isclose(total, expected_total, rel_tol=1e-10, abs_tol=1e-10): + raise ValueError("all-source cap total is not conserved") + receipt = { + "schema_version": "eq3-all-source-cap-v1", + "status": "CAP_INPUT_GENERATED_NOT_SOLVED", + "blind_trajectory_read": False, + "trace_id": "all_source_cap", + "duration_s": slot_s, + "group_order": order, + "caps_w": dict(zip(order, caps, strict=True)), + "group_members": {key: sorted(value) for key, value in sorted(group_members.items())}, + "component_cap_w": {key: emitted_by_component[key] + for key in sorted(emitted_by_component)}, + "node_cap_w": {key: node_cap_w[key] for key in sorted(node_cap_w)}, + "total_cap_w": total, + "event_count": len(events) - 1, + "powered_node_count": len(node_cap_w), + } + return "\n".join(events) + "\n", receipt + + +def main(): + parser = argparse.ArgumentParser() + parser.add_argument("--profile", type=Path, required=True) + parser.add_argument("--power", type=Path, required=True) + parser.add_argument("--rc-grid", type=Path, required=True) + parser.add_argument("--events-output", type=Path, required=True) + parser.add_argument("--receipt-output", type=Path, required=True) + args = parser.parse_args() + for output in (args.events_output, args.receipt_output): + if output.exists(): + raise FileExistsError(f"refusing to overwrite {output}") + events, receipt = build(json.loads(args.profile.read_text()), + json.loads(args.power.read_text()), + json.loads(args.rc_grid.read_text())) + receipt["source_sha256"] = { + "profile": sha256(args.profile), "power": sha256(args.power), + "rc_grid": sha256(args.rc_grid)} + args.events_output.write_text(events) + receipt["events_sha256"] = hashlib.sha256(events.encode()).hexdigest() + args.receipt_output.write_text(json.dumps(receipt, indent=2) + "\n") + print(json.dumps({"status": receipt["status"], "total_cap_w": receipt["total_cap_w"], + "event_count": receipt["event_count"]})) + + +if __name__ == "__main__": + main() diff --git a/tools/eq3_basic_fabric.py b/tools/eq3_basic_fabric.py new file mode 100644 index 0000000..ca39234 --- /dev/null +++ b/tools/eq3_basic_fabric.py @@ -0,0 +1,484 @@ +"""Default-off parameterized HBF/HBM transfer fabric for CPU composition. + +The caller supplies media-ready times. This module models only base SRAM +buffering and links; it does not issue media commands or call back into a +scheduler. All parameters are engineering SCENARIO_ASSUMPTION values. +""" +from __future__ import annotations + +import copy +import heapq + + +NANOSECONDS_PER_SECOND = 1_000_000_000 + + +def _integer(value, label, *, positive=False): + if isinstance(value, bool) or not isinstance(value, int): + raise ValueError(f"{label} must be an integer") + if value < (1 if positive else 0): + raise ValueError(f"{label} must be {'positive' if positive else 'non-negative'}") + return value + + +def _exact(value, keys, label): + if not isinstance(value, dict) or set(value) != set(keys): + raise ValueError(f"{label} must contain exactly {sorted(keys)}") + return value + + +def _stage(value, label): + value = _exact(value, ("latency_ns", "bandwidth_bytes_per_s"), label) + return { + "latency_ns": _integer(value["latency_ns"], f"{label}.latency_ns"), + "bandwidth_bytes_per_s": _integer( + value["bandwidth_bytes_per_s"], + f"{label}.bandwidth_bytes_per_s", + positive=True, + ), + } + + +def _duration(stage, byte_count): + transfer = (byte_count * NANOSECONDS_PER_SECOND + stage["bandwidth_bytes_per_s"] - 1) // stage[ + "bandwidth_bytes_per_s" + ] + return stage["latency_ns"] + transfer + + +def _normalize(config): + config = _exact(config, ("evidence", "hbf", "hbm"), "config") + if config["evidence"] != "SCENARIO_ASSUMPTION": + raise ValueError("basic fabric parameters must remain SCENARIO_ASSUMPTION") + if not isinstance(config["hbf"], dict) or not config["hbf"]: + raise ValueError("config.hbf must be a nonempty object") + if not isinstance(config["hbm"], dict): + raise ValueError("config.hbm must be an object") + hbm = {} + for stack, raw in sorted(config["hbm"].items()): + if not isinstance(stack, str) or not stack: + raise ValueError("HBM stack IDs must be nonempty strings") + raw = _exact(raw, ("bank_count", "bank_capacity_bytes", "gpu_link"), f"hbm.{stack}") + count = _integer(raw["bank_count"], f"hbm.{stack}.bank_count", positive=True) + if count != 2: + raise ValueError("basic HBM relay buffering requires exactly two banks") + hbm[stack] = { + "bank_count": count, + "bank_capacity_bytes": _integer( + raw["bank_capacity_bytes"], f"hbm.{stack}.bank_capacity_bytes", positive=True + ), + "gpu_link": _stage(raw["gpu_link"], f"hbm.{stack}.gpu_link"), + } + hbf = {} + used_pairs = set() + for stack, raw in sorted(config["hbf"].items()): + if not isinstance(stack, str) or not stack: + raise ValueError("HBF stack IDs must be nonempty strings") + raw = _exact( + raw, + ("pair", "bank_count", "bank_capacity_bytes", "fill", "direct_link", "relay_link"), + f"hbf.{stack}", + ) + count = _integer(raw["bank_count"], f"hbf.{stack}.bank_count", positive=True) + if count != 2: + raise ValueError("basic HBF buffering requires exactly two banks") + pair = raw["pair"] + if pair is not None: + if not isinstance(pair, str) or pair not in hbm or pair in used_pairs: + raise ValueError(f"hbf.{stack}.pair must name a unique configured HBM") + used_pairs.add(pair) + relay = _stage(raw["relay_link"], f"hbf.{stack}.relay_link") + else: + if raw["relay_link"] is not None: + raise ValueError(f"hbf.{stack}.relay_link requires an HBM pair") + relay = None + hbf[stack] = { + "pair": pair, + "bank_count": count, + "bank_capacity_bytes": _integer( + raw["bank_capacity_bytes"], f"hbf.{stack}.bank_capacity_bytes", positive=True + ), + "fill": _stage(raw["fill"], f"hbf.{stack}.fill"), + "direct_link": _stage(raw["direct_link"], f"hbf.{stack}.direct_link"), + "relay_link": relay, + } + if set(hbf) & set(hbm): + raise ValueError("stack IDs must be unique across HBF and HBM") + return {"evidence": "SCENARIO_ASSUMPTION", "hbf": hbf, "hbm": hbm} + + +class BasicFabric: + """Small deterministic event engine for media-ready byte transfers.""" + + def __init__(self, config): + self._config = _normalize(copy.deepcopy(config)) + self._now = 0 + self._sequence = 0 + self._jobs = {} + self._arrivals = [] + self._backend_ready = [] + self._active = [] + self._completions = [] + self._events = [] + self._hbf = { + stack: {"banks": [None] * row["bank_count"], "fill": None, "direct": None, "relay": None} + for stack, row in self._config["hbf"].items() + } + self._hbm = { + stack: {"banks": [None] * row["bank_count"], "gpu": None} + for stack, row in self._config["hbm"].items() + } + + @property + def time_ns(self): + return self._now + + def immutable_facts(self): + return copy.deepcopy( + { + "evidence": "SCENARIO_ASSUMPTION", + "enabled_by_default": False, + "media_model": "EXTERNAL_READY_TIMESTAMP", + "backend_delivered_mode": "RESERVE_SOURCE_BANK_THEN_MARK_READY_NO_DUPLICATE_FILL", + "duration_rule": "latency_ns + ceil(bytes*1e9/bandwidth_bytes_per_s)", + "pair_scope": "ONE_TO_ONE_PRIVATE_NO_PACKAGE_GLOBAL_LOCK", + "same_timestamp_order": "COMPLETE_RELEASE_ARRIVE_ADMIT", + "config": self._config, + } + ) + + def enqueue(self, request_id, source_stack, route, bytes, ready_ns): + if not isinstance(request_id, str) or not request_id or request_id in self._jobs: + raise ValueError("request_id must be a new nonempty string") + byte_count = _integer(bytes, "bytes", positive=True) + ready = _integer(ready_ns, "ready_ns") + if ready < self._now: + raise ValueError("ready_ns precedes current fabric time") + if source_stack in self._config["hbf"]: + if route not in ("direct", "relay"): + raise ValueError("HBF route must be direct or relay") + row = self._config["hbf"][source_stack] + if byte_count > row["bank_capacity_bytes"]: + raise ValueError("request exceeds HBF bank capacity; chunking is external") + if route == "relay": + pair = row["pair"] + if pair is None: + raise ValueError("relay requested for unpaired HBF") + if byte_count > self._config["hbm"][pair]["bank_capacity_bytes"]: + raise ValueError("request exceeds HBM bank capacity; chunking is external") + kind = "HBF" + elif source_stack in self._config["hbm"]: + if route != "direct": + raise ValueError("HBM source accepts direct route only") + kind = "HBM" + else: + raise ValueError("unknown source_stack") + sequence = self._sequence + self._sequence += 1 + job = { + "request_id": request_id, + "source_stack": source_stack, + "source_kind": kind, + "route": route, + "bytes": byte_count, + "ready_ns": ready, + "sequence": sequence, + "state": "WAIT_READY", + "hbf_bank": None, + "hbm_bank": None, + } + self._jobs[request_id] = job + heapq.heappush(self._arrivals, (ready, sequence, request_id)) + + def reserve_source(self, request_id, source_stack, route, bytes, arrival_ns): + """Reserve bounded controller storage before submitting a backend command. + + False means the caller must retain the request externally. A successful + reservation owns one source-base bank until GPU drain completion (or, + for HBF relay, until the relay has safely occupied an HBM bank). + """ + if not isinstance(request_id, str) or not request_id or request_id in self._jobs: + raise ValueError("request_id must be a new nonempty string") + byte_count = _integer(bytes, "bytes", positive=True) + arrival = _integer(arrival_ns, "arrival_ns") + if arrival > self._now: + raise ValueError("future request cannot reserve a bank before arrival") + if source_stack in self._config["hbf"]: + if route not in ("direct", "relay"): + raise ValueError("HBF route must be direct or relay") + row = self._config["hbf"][source_stack] + if byte_count > row["bank_capacity_bytes"]: + raise ValueError("request exceeds HBF bank capacity; chunking is external") + if route == "relay": + pair = row["pair"] + if pair is None: + raise ValueError("relay requested for unpaired HBF") + if byte_count > self._config["hbm"][pair]["bank_capacity_bytes"]: + raise ValueError("request exceeds HBM bank capacity; chunking is external") + kind = "HBF" + state = self._hbf[source_stack] + elif source_stack in self._config["hbm"]: + if route != "direct": + raise ValueError("HBM source accepts direct route only") + row = self._config["hbm"][source_stack] + if byte_count > row["bank_capacity_bytes"]: + raise ValueError("request exceeds HBM bank capacity; chunking is external") + kind = "HBM" + state = self._hbm[source_stack] + else: + raise ValueError("unknown source_stack") + bank = self._free_bank(state["banks"]) + if bank is None: + return False + sequence = self._sequence + self._sequence += 1 + job = { + "request_id": request_id, + "source_stack": source_stack, + "source_kind": kind, + "route": route, + "bytes": byte_count, + "ready_ns": None, + "arrival_ns": arrival, + "sequence": sequence, + "state": "BACKEND_RESERVED", + "hbf_bank": bank if kind == "HBF" else None, + "hbm_bank": bank if kind == "HBM" else None, + } + self._jobs[request_id] = job + state["banks"][bank] = request_id + self._events.append( + { + "kind": "reserve", + "request_id": request_id, + "resource": f"{source_stack}:bank:{bank}", + ("hbf_bank" if kind == "HBF" else "hbm_bank"): bank, + "bytes": byte_count, + "arrival_ns": arrival, + "time_ns": self._now, + } + ) + return True + + def reserve_hbf(self, request_id, source_stack, route, bytes, arrival_ns): + if source_stack not in self._config["hbf"]: + raise ValueError("reserve_hbf requires a configured HBF source") + return self.reserve_source(request_id, source_stack, route, bytes, arrival_ns) + + def mark_source_ready(self, request_id, backend_completion_ns): + if request_id not in self._jobs or self._jobs[request_id]["state"] != "BACKEND_RESERVED": + raise ValueError("request does not own a pending backend source reservation") + completion = _integer(backend_completion_ns, "backend_completion_ns") + if completion < self._now: + raise ValueError("backend completion precedes current fabric time") + job = self._jobs[request_id] + job["state"] = "BACKEND_READY_PENDING" + job["ready_ns"] = completion + heapq.heappush(self._backend_ready, (completion, job["sequence"], request_id)) + + def mark_hbf_ready(self, request_id, backend_completion_ns): + if request_id not in self._jobs or self._jobs[request_id]["source_kind"] != "HBF": + raise ValueError("mark_hbf_ready requires an HBF reservation") + self.mark_source_ready(request_id, backend_completion_ns) + + def next_event_ns(self): + times = [] + if self._arrivals: + times.append(self._arrivals[0][0]) + if self._backend_ready: + times.append(self._backend_ready[0][0]) + if self._active: + times.append(self._active[0][0]) + return min(times) if times else None + + def completions(self): + return tuple(copy.deepcopy(self._completions)) + + def events(self): + return tuple(copy.deepcopy(self._events)) + + def resource_state(self): + """Read-only ownership snapshot for coordinator conservation checks.""" + return copy.deepcopy({ + "time_ns": self._now, + "hbf": { + stack: {"banks": row["banks"], "fill": row["fill"], + "direct": row["direct"], "relay": row["relay"]} + for stack, row in self._hbf.items() + }, + "hbm": { + stack: {"banks": row["banks"], "gpu": row["gpu"]} + for stack, row in self._hbm.items() + }, + "unfinished": sorted( + request_id for request_id, job in self._jobs.items() + if job["state"] != "DONE" + ), + }) + + def _free_bank(self, banks): + return next((index for index, owner in enumerate(banks) if owner is None), None) + + def _start(self, job, stage_name, stage, resource, *, hbf_bank=None, hbm_bank=None): + start = self._now + end = start + _duration(stage, job["bytes"]) + job["state"] = stage_name + heapq.heappush(self._active, (end, job["sequence"], stage_name, job["request_id"])) + event = { + "kind": "start", + "stage": stage_name, + "request_id": job["request_id"], + "resource": resource, + "bytes": job["bytes"], + "start_ns": start, + "end_ns": end, + } + if hbf_bank is not None: + event["hbf_bank"] = hbf_bank + if hbm_bank is not None: + event["hbm_bank"] = hbm_bank + self._events.append(event) + + def _complete(self, stage_name, request_id): + job = self._jobs[request_id] + source = job["source_stack"] + if stage_name == "HBF_FILL": + self._hbf[source]["fill"] = None + job["state"] = "READY_HBF" + elif stage_name == "HBF_DIRECT": + self._hbf[source]["direct"] = None + self._hbf[source]["banks"][job["hbf_bank"]] = None + job["hbf_bank"] = None + self._finish(job) + elif stage_name == "HBF_RELAY": + self._hbf[source]["relay"] = None + self._hbf[source]["banks"][job["hbf_bank"]] = None + job["hbf_bank"] = None + job["state"] = "READY_HBM" + elif stage_name == "HBM_GPU": + hbm_stack = source if job["source_kind"] == "HBM" else self._config["hbf"][source]["pair"] + self._hbm[hbm_stack]["gpu"] = None + if job["hbm_bank"] is not None: + self._hbm[hbm_stack]["banks"][job["hbm_bank"]] = None + job["hbm_bank"] = None + self._finish(job) + else: + raise AssertionError("unknown active stage") + self._events.append( + {"kind": "complete", "stage": stage_name, "request_id": request_id, "bytes": job["bytes"], "time_ns": self._now} + ) + + def _finish(self, job): + job["state"] = "DONE" + self._completions.append( + { + "request_id": job["request_id"], + "source_stack": job["source_stack"], + "route": job["route"], + "bytes": job["bytes"], + "ready_ns": job["ready_ns"], + "completion_ns": self._now, + } + ) + + def _admit(self): + waiting = sorted(self._jobs.values(), key=lambda row: row["sequence"]) + for stack, row in self._config["hbf"].items(): + state = self._hbf[stack] + if state["fill"] is not None: + continue + bank = self._free_bank(state["banks"]) + job = next((j for j in waiting if j["source_stack"] == stack and j["state"] == "MEDIA_READY"), None) + if bank is not None and job is not None: + state["fill"] = job["request_id"] + state["banks"][bank] = job["request_id"] + job["hbf_bank"] = bank + self._start(job, "HBF_FILL", row["fill"], f"{stack}:fill", hbf_bank=bank) + + for job in waiting: + if job["state"] != "READY_HBF": + continue + stack = job["source_stack"] + row = self._config["hbf"][stack] + state = self._hbf[stack] + if job["route"] == "direct" and state["direct"] is None: + state["direct"] = job["request_id"] + self._start( + job, "HBF_DIRECT", row["direct_link"], f"{stack}:gpu-link", hbf_bank=job["hbf_bank"] + ) + elif job["route"] == "relay" and state["relay"] is None: + pair = row["pair"] + hbm_state = self._hbm[pair] + bank = self._free_bank(hbm_state["banks"]) + if bank is not None: + state["relay"] = job["request_id"] + hbm_state["banks"][bank] = job["request_id"] + job["hbm_bank"] = bank + self._start( + job, + "HBF_RELAY", + row["relay_link"], + f"{stack}->{pair}:relay-link", + hbf_bank=job["hbf_bank"], + hbm_bank=bank, + ) + + for stack, row in self._config["hbm"].items(): + state = self._hbm[stack] + if state["gpu"] is not None: + continue + candidates = [ + job + for job in waiting + if (job["source_kind"] == "HBM" and job["source_stack"] == stack and job["state"] == "HBM_READY") + or ( + job["source_kind"] == "HBF" + and self._config["hbf"][job["source_stack"]]["pair"] == stack + and job["state"] == "READY_HBM" + ) + ] + if candidates: + job = candidates[0] + state["gpu"] = job["request_id"] + self._start(job, "HBM_GPU", row["gpu_link"], f"{stack}:gpu-link", hbm_bank=job["hbm_bank"]) + + def advance(self, horizon_ns): + horizon = _integer(horizon_ns, "horizon_ns") + if horizon < self._now: + raise ValueError("fabric time cannot move backwards") + while True: + next_time = self.next_event_ns() + if next_time is None or next_time > horizon: + break + self._now = next_time + ending = [] + while self._active and self._active[0][0] == self._now: + ending.append(heapq.heappop(self._active)) + for _, _, stage_name, request_id in ending: + self._complete(stage_name, request_id) + while self._backend_ready and self._backend_ready[0][0] == self._now: + _, _, request_id = heapq.heappop(self._backend_ready) + job = self._jobs[request_id] + job["state"] = "READY_HBF" if job["source_kind"] == "HBF" else "HBM_READY" + self._events.append( + { + "kind": "backend_ready", + "request_id": request_id, + "bytes": job["bytes"], + "time_ns": self._now, + } + ) + while self._arrivals and self._arrivals[0][0] == self._now: + _, _, request_id = heapq.heappop(self._arrivals) + job = self._jobs[request_id] + job["state"] = "MEDIA_READY" if job["source_kind"] == "HBF" else "HBM_READY" + self._events.append( + { + "kind": "media_ready", + "request_id": request_id, + "bytes": job["bytes"], + "time_ns": self._now, + } + ) + self._admit() + self._now = horizon diff --git a/tools/eq3_basic_hbm.py b/tools/eq3_basic_hbm.py new file mode 100644 index 0000000..d168534 --- /dev/null +++ b/tools/eq3_basic_hbm.py @@ -0,0 +1,248 @@ +"""Minimal event-driven HBM media fixture for PARAMETRIC_HBM_SCENARIO use.""" +import copy +import math + + +SCHEMA = "eq3-basic-hbm-v1" +EVIDENCE = "PARAMETRIC_HBM_SCENARIO" +OPS = ("read", "write") + + +def _integer(value, label, *, positive=False): + if isinstance(value, bool) or not isinstance(value, int): + raise ValueError(f"{label} must be an integer") + if value < (1 if positive else 0): + raise ValueError(f"{label} is outside its allowed range") + return value + + +def _number(value, label, *, positive=False): + if isinstance(value, bool) or not isinstance(value, (int, float)): + raise ValueError(f"{label} must be numeric") + value = float(value) + if not math.isfinite(value) or value < (0 if not positive else math.nextafter(0.0, 1.0)): + raise ValueError(f"{label} is outside its allowed range") + return value + + +def scenario_example(stack_ids=("hbm0",)): + """Explicit example values; callers must opt in by using the returned config.""" + return { + "schema_version": SCHEMA, + "evidence_class": EVIDENCE, + "stacks": [{ + "stack_id": stack_id, + "media_latency_ns": {"read": 100, "write": 100}, + "media_latency_evidence": "SCENARIO_ASSUMPTION", + "media_bandwidth_Bps": {"read": 2_048_000_000_000, + "write": 2_048_000_000_000}, + "media_bandwidth_evidence": + "DERIVED_WITH_TRANSFER_ASSUMPTION: 2048-bit times 8 Gbit/s raw interface treated as media service bandwidth; not calibrated", + "energy_j_per_byte": {"read": None, "write": None}, + "energy_evidence": "UNKNOWN_UNPARAMETERIZED", + } for stack_id in stack_ids], + "claim_limit": "basic CPU timing scenario; not a real DRAM backend", + } + + +class BasicHbm: + """One FIFO media server per HBM stack; fabric service is intentionally external.""" + + def __init__(self, config): + if config.get("schema_version") != SCHEMA or config.get("evidence_class") != EVIDENCE: + raise ValueError("HBM config must explicitly select PARAMETRIC_HBM_SCENARIO") + rows = config.get("stacks") + if not isinstance(rows, list) or not rows: + raise ValueError("stacks must be a nonempty list") + self._stacks = {} + for row in rows: + stack_id = row.get("stack_id") + if not isinstance(stack_id, str) or not stack_id or stack_id in self._stacks: + raise ValueError("stack_id must be a unique nonempty string") + latency = row.get("media_latency_ns") + bandwidth = row.get("media_bandwidth_Bps") + if not isinstance(latency, dict) or set(latency) != set(OPS): + raise ValueError(f"{stack_id} requires read/write media_latency_ns") + if not isinstance(bandwidth, dict) or set(bandwidth) != set(OPS): + raise ValueError(f"{stack_id} requires read/write media_bandwidth_Bps") + latency = {op: _integer(latency[op], f"{stack_id}.{op}.latency") for op in OPS} + bandwidth = {op: _number(bandwidth[op], f"{stack_id}.{op}.bandwidth", + positive=True) for op in OPS} + energy = row.get("energy_j_per_byte", {op: None for op in OPS}) + if not isinstance(energy, dict) or set(energy) != set(OPS): + raise ValueError(f"{stack_id} energy_j_per_byte must cover read/write") + checked_energy = {} + for op in OPS: + checked_energy[op] = (None if energy[op] is None else + _number(energy[op], f"{stack_id}.{op}.energy")) + latency_evidence = row.get("media_latency_evidence") + bandwidth_evidence = row.get("media_bandwidth_evidence") + energy_evidence = row.get("energy_evidence", "UNKNOWN_UNPARAMETERIZED") + if not isinstance(latency_evidence, str) or not latency_evidence: + raise ValueError(f"{stack_id} requires media_latency_evidence") + if not isinstance(bandwidth_evidence, str) or not bandwidth_evidence: + raise ValueError(f"{stack_id} requires media_bandwidth_evidence") + if (any(value is not None for value in checked_energy.values()) and + (not isinstance(energy_evidence, str) or + energy_evidence.startswith("UNKNOWN"))): + raise ValueError(f"{stack_id} parameterized energy requires non-UNKNOWN evidence") + self._stacks[stack_id] = { + "latency": latency, "bandwidth": bandwidth, "energy": checked_energy, + "latency_evidence": latency_evidence, + "bandwidth_evidence": bandwidth_evidence, + "energy_evidence": energy_evidence, + "queue": [], "active": None, + } + self._now_ns = 0 + self._requests = {} + self._facts = [] + self._completions = [] + self._event_id = 0 + + @property + def now_ns(self): + return self._now_ns + + def capability(self): + return { + "schema_version": SCHEMA, + "evidence_class": EVIDENCE, + "operations": list(OPS), + "queue_model": "one FIFO single media server per stack", + "fabric_arbitration": "EXTERNAL_REQUIRED", + "refresh": "UNSUPPORTED_CAPABILITY", + "die": "UNKNOWN_NOT_MODELED", + "plane": "UNKNOWN_NOT_MODELED", + "real_dram_backend": False, + } + + def _emit(self, phase, request, time_ns, **extra): + self._event_id += 1 + fact = { + "event_id": self._event_id, "phase": phase, "time_ns": time_ns, + "request_id": request["request_id"], "stack_id": request["stack_id"], + "op": request["op"], "bytes": request["bytes"], + "resource": f"{request['stack_id']}:media", "die": "UNKNOWN", + "plane": "UNKNOWN", "evidence_class": EVIDENCE, + "reported_completion": False, + } + fact.update(extra) + self._facts.append(fact) + + def arrival(self, request): + required = {"request_id", "stack_id", "op", "bytes", "arrival_ns"} + if not isinstance(request, dict) or not required.issubset(request): + raise ValueError("arrival requires request_id, stack_id, op, bytes, arrival_ns") + request_id = request["request_id"] + if not isinstance(request_id, str) or not request_id or request_id in self._requests: + raise ValueError("request_id must be unique and nonempty") + stack_id = request["stack_id"] + if stack_id not in self._stacks: + raise ValueError("unknown HBM stack") + if request["op"] not in OPS: + raise ValueError("HBM media supports only read/write") + byte_count = _integer(request["bytes"], "bytes", positive=True) + arrival_ns = _integer(request["arrival_ns"], "arrival_ns") + if any(key in request for key in ("die", "plane")): + raise ValueError("die/plane must remain UNKNOWN; address-derived placement is unsupported") + self.advance(arrival_ns) + value = {"request_id": request_id, "stack_id": stack_id, "op": request["op"], + "bytes": byte_count, "arrival_ns": arrival_ns, "state": "ARRIVED"} + self._requests[request_id] = value + self._emit("arrival", value, arrival_ns) + return copy.deepcopy(value) + + def submit(self, request_id, time_ns=None): + if request_id not in self._requests: + raise ValueError("request must arrive before submit") + request = self._requests[request_id] + if request["state"] != "ARRIVED": + raise ValueError("request may be submitted exactly once") + selected_time = self._now_ns if time_ns is None else _integer(time_ns, "submit time") + self.advance(selected_time) + if selected_time < request["arrival_ns"]: + raise ValueError("submit precedes arrival") + request["submit_ns"] = selected_time + request["state"] = "QUEUED" + self._stacks[request["stack_id"]]["queue"].append(request_id) + self._emit("submit", request, selected_time) + self._start(request["stack_id"]) + return {"disposition": "ACCEPTED", "request_id": request_id, + "media_start_ns": request.get("media_start_ns")} + + def _duration_ns(self, request): + stack = self._stacks[request["stack_id"]] + transfer = math.ceil(request["bytes"] * 1_000_000_000 / + stack["bandwidth"][request["op"]]) + return stack["latency"][request["op"]] + transfer, transfer + + def _start(self, stack_id): + stack = self._stacks[stack_id] + if stack["active"] is not None or not stack["queue"]: + return + request = self._requests[stack["queue"].pop(0)] + duration, transfer = self._duration_ns(request) + request.update({"state": "MEDIA_ACTIVE", "media_start_ns": self._now_ns, + "media_end_ns": self._now_ns + duration}) + stack["active"] = request["request_id"] + self._emit( + "media_start", request, self._now_ns, media_end_ns=request["media_end_ns"], + media_duration_ns=duration, fixed_media_latency_ns=duration - transfer, + bandwidth_transfer_ns=transfer, + media_latency_evidence=stack["latency_evidence"], + media_bandwidth_evidence=stack["bandwidth_evidence"], + duration_formula="media_latency_ns + ceil(bytes*1e9/media_bandwidth_Bps)") + + def next_event_ns(self): + values = [self._requests[stack["active"]]["media_end_ns"] + for stack in self._stacks.values() if stack["active"] is not None] + return min(values) if values else None + + def advance(self, target_ns): + target_ns = _integer(target_ns, "target_ns") + if target_ns < self._now_ns: + raise ValueError("HBM clock cannot move backward") + while self.next_event_ns() is not None and self.next_event_ns() <= target_ns: + self._now_ns = self.next_event_ns() + completed_stacks = sorted( + stack_id for stack_id, stack in self._stacks.items() + if stack["active"] is not None and + self._requests[stack["active"]]["media_end_ns"] == self._now_ns) + for stack_id in completed_stacks: + stack = self._stacks[stack_id] + request = self._requests[stack["active"]] + request["state"] = "MEDIA_COMPLETE" + stack["active"] = None + energy_parameter = stack["energy"][request["op"]] + energy = None if energy_parameter is None else request["bytes"] * energy_parameter + energy_status = ("UNKNOWN_UNPARAMETERIZED" if energy is None else + stack["energy_evidence"]) + self._emit("media_complete", request, self._now_ns, + media_energy_j=energy, media_energy_status=energy_status, + requires_fabric=True) + self._completions.append({ + "phase": "MEDIA_DONE", "request_id": request["request_id"], + "time_ns": self._now_ns, "stack": stack_id, + "op": request["op"], "bytes": request["bytes"], + "requires_fabric": True, "reported_completion": False, + "die": "UNKNOWN", "plane": "UNKNOWN", + "media_energy_j": energy, "media_energy_status": energy_status, + "evidence_class": EVIDENCE, + }) + for stack_id in completed_stacks: + self._start(stack_id) + self._now_ns = target_ns + + def take_facts(self): + result, self._facts = self._facts, [] + return copy.deepcopy(result) + + def take_media_completions(self): + result, self._completions = self._completions, [] + return copy.deepcopy(result) + + def refresh(self, stack_id): + if stack_id not in self._stacks: + raise ValueError("unknown HBM stack") + return {"status": "UNSUPPORTED_CAPABILITY", "operation": "refresh", + "stack_id": stack_id, "reason": "basic HBM media fixture has no refresh backend"} diff --git a/tools/eq3_basic_system.py b/tools/eq3_basic_system.py new file mode 100644 index 0000000..3627907 --- /dev/null +++ b/tools/eq3_basic_system.py @@ -0,0 +1,348 @@ +"""Minimal CPU coordinator for the authorized EQ3 basic system fixtures. + +This module composes existing backends; it does not add a scheduler thread, +media timing, retry loop inside MQSim, or calibrated topology claim. +""" +from __future__ import annotations + +import copy + + +MODES = {"all_hbf_direct", "mixed_direct", "relay", "dash"} +MAX_HORIZON_NS = (1 << 64) - 1 + + +def _integer(value, label, *, positive=False): + if isinstance(value, bool) or not isinstance(value, int): + raise ValueError(f"{label} must be an integer") + if value < (1 if positive else 0): + raise ValueError(f"{label} is outside its allowed range") + return value + + +class UnsupportedComposition(RuntimeError): + pass + + +def engineering_fixture(mode, page_bytes): + """Return one explicit small four-topology software fixture declaration.""" + if mode not in MODES: + raise ValueError("unknown basic-system topology mode") + page_bytes = _integer(page_bytes, "page_bytes", positive=True) + hbf_count = 8 if mode == "all_hbf_direct" else 4 + hbm_count = 0 if mode == "all_hbf_direct" else 4 + + def stage(latency): + return {"latency_ns": latency, "bandwidth_bytes_per_s": 1_000_000_000_000} + + fabric = { + "evidence": "SCENARIO_ASSUMPTION", + "hbf": { + f"hbf{i}": { + "pair": f"hbm{i}" if mode in {"relay", "dash"} else None, + "bank_count": 2, "bank_capacity_bytes": page_bytes, + "fill": stage(1), "direct_link": stage(2), + "relay_link": stage(3) if mode in {"relay", "dash"} else None, + } for i in range(hbf_count) + }, + "hbm": { + f"hbm{i}": {"bank_count": 2, "bank_capacity_bytes": page_bytes, + "gpu_link": stage(4)} + for i in range(hbm_count) + }, + } + hbm = None if not hbm_count else { + "schema_version": "eq3-basic-hbm-v1", + "evidence_class": "PARAMETRIC_HBM_SCENARIO", + "stacks": [{ + "stack_id": f"hbm{i}", + "media_latency_ns": {"read": 10, "write": 10}, + "media_latency_evidence": "SCENARIO_ASSUMPTION", + "media_bandwidth_Bps": {"read": 1_000_000_000_000, + "write": 1_000_000_000_000}, + "media_bandwidth_evidence": "SCENARIO_ASSUMPTION", + "energy_j_per_byte": {"read": None, "write": None}, + "energy_evidence": "UNKNOWN_UNPARAMETERIZED", + } for i in range(hbm_count)], + "claim_limit": "small basic-system CPU fixture; not a real DRAM backend", + } + requests = [] + for i in range(hbf_count): + routes = (("direct", "relay", "direct", "relay") if mode == "dash" else + ("relay", "relay", "relay") if mode == "relay" else + ("direct", "direct", "direct")) + for local_page, route in enumerate(routes): + requests.append({ + "request_id": f"hbf{i}-{route}-{local_page}", "stack": f"hbf{i}", + "stack_local_page": local_page, "route": route, + "bytes": page_bytes, "arrival_ns": 0, "operation": "read", + }) + for i in range(hbm_count): + for local_index in range(3): + requests.append({ + "request_id": f"hbm{i}-direct-{local_index}", "stack": f"hbm{i}", + "route": "direct", "bytes": page_bytes, "arrival_ns": 0, + "operation": "read", + }) + return {"evidence": "ENGINEERING_FIXTURE_BASIC_SYSTEM", + "research_geometry": False, "mode": mode, + "fabric": fabric, "hbm": hbm, "requests": requests} + + +class BasicSystem: + """Single-threaded horizon consumer for MQSim, BasicHbm and BasicFabric.""" + + def __init__(self, mode, mqsim, hbm, fabric): + if mode not in MODES: + raise ValueError("unknown basic-system topology mode") + self.mode = mode + self.mqsim = mqsim + self.hbm = hbm + self.fabric = fabric + facts = fabric.immutable_facts() + self._hbf = set(facts["config"]["hbf"]) + self._hbm = set(facts["config"]["hbm"]) + if mode == "all_hbf_direct" and self._hbm: + raise ValueError("all_hbf_direct cannot configure in-package HBM") + if mode in {"relay", "dash"} and (not self._hbf or not self._hbm): + raise ValueError("relay and DASH require explicit HBF/HBM pairs") + self.now_ns = 0 + self._records = {} + self._waiting = [] + self._mq_pending = {} + self._next_backend_id = 1 + self._fabric_completion_count = 0 + self._hbm_facts = [] + + def _validate_request(self, raw, sequence): + required = {"request_id", "stack", "route", "bytes", "arrival_ns", "operation"} + if not isinstance(raw, dict) or not required.issubset(raw): + raise ValueError("request lacks required identity, path, extent, or arrival") + request_id = raw["request_id"] + if not isinstance(request_id, str) or not request_id or request_id in self._records: + raise ValueError("request_id must be unique and nonempty") + stack, route = raw["stack"], raw["route"] + if stack in self._hbf: + kind = "HBF" + if raw["operation"] != "read": + raise ValueError("basic MQSim HBF path is read-only") + if "stack_local_page" not in raw: + raise ValueError("HBF request requires persistent stack_local_page") + local_page = _integer(raw["stack_local_page"], "stack_local_page") + if self.mode in {"all_hbf_direct", "mixed_direct"} and route != "direct": + raise ValueError("this topology exposes only direct HBF package paths") + if self.mode == "relay" and route != "relay": + raise ValueError("cascaded topology exposes no HBF direct GPU path") + if self.mode == "dash" and route not in {"direct", "relay"}: + raise ValueError("DASH HBF route must be direct or relay") + elif stack in self._hbm: + kind, local_page = "HBM", None + if self.mode == "all_hbf_direct" or route != "direct": + raise ValueError("HBM is local/direct only where configured") + if raw["operation"] not in {"read", "write"}: + raise ValueError("basic HBM path supports read/write only") + else: + raise ValueError("request names an unconfigured stack") + value = { + "request_id": request_id, "sequence": sequence, "kind": kind, + "stack": stack, "route": route, + "bytes": _integer(raw["bytes"], "bytes", positive=True), + "arrival_ns": _integer(raw["arrival_ns"], "arrival_ns"), + "operation": raw["operation"], "stack_local_page": local_page, + "state": "EXTERNAL_WAIT", "backend_submit_ns": None, + "backend_completion_ns": None, "backend_media_ns": None, + "fabric_completion_ns": None, + } + page_bytes = getattr(self.mqsim, "header", {}).get("page_bytes") + if kind == "HBF" and page_bytes is not None and value["bytes"] != page_bytes: + raise ValueError("HBF request must equal the mapped MQSim profile page size") + self._records[request_id] = value + return value + + def _admit(self): + progress = False + for request in sorted(self._waiting, key=lambda row: row["sequence"]): + if request["state"] == "GATE_WAIT": + if request["gate_target_ns"] <= self.now_ns: + self._try_hbf_submit(request) + progress = True + continue + if request["state"] != "EXTERNAL_WAIT" or request["arrival_ns"] > self.now_ns: + continue + if not self.fabric.reserve_source( + request["request_id"], request["stack"], request["route"], + request["bytes"], request["arrival_ns"]): + continue + if request["kind"] == "HBF": + backend_id = self._next_backend_id + self._next_backend_id += 1 + request["backend_request_id"] = backend_id + request["state"] = "SOURCE_RESERVED" + self._try_hbf_submit(request) + else: + request["backend_submit_ns"] = self.now_ns + request["external_wait_ns"] = self.now_ns - request["arrival_ns"] + request["state"] = "BACKEND_PENDING" + self.hbm.arrival({ + "request_id": request["request_id"], "stack_id": request["stack"], + "op": request["operation"], "bytes": request["bytes"], + "arrival_ns": self.now_ns, + }) + self.hbm.submit(request["request_id"], self.now_ns) + progress = True + return progress + + def _try_hbf_submit(self, request): + backend_id = request["backend_request_id"] + payload = { + "request_id": backend_id, "issue_ns": self.now_ns, + "stack": request["stack"], + "stack_local_page": request["stack_local_page"], + "bytes": request["bytes"], "operation": "read", + # MQSim placement is always native direct; package relay is owned + # solely by BasicFabric and remains in the system record. + "route": "direct", + } + decision = self.mqsim.try_submit(payload) + request["gate_decision"] = copy.deepcopy(decision) + if decision["submitted"]: + request["backend_submit_ns"] = decision["backend_arrival_ns"] + request["external_wait_ns"] = request["backend_submit_ns"] - request["arrival_ns"] + request["state"] = "BACKEND_PENDING" + request.pop("gate_target_ns", None) + self._mq_pending[backend_id] = request["request_id"] + return + if decision["disposition"] == 1 and isinstance(decision.get("target_ns"), int): + if decision["target_ns"] <= self.now_ns: + raise RuntimeError("MQSim gate returned a nonfuture defer target") + request["gate_target_ns"] = decision["target_ns"] + request["state"] = "GATE_WAIT" + return + raise UnsupportedComposition( + f"UNSUPPORTED_COMPOSITION: MQSim gate rejected request: {decision.get('reason', '')}") + + def _raw_mqsim_completion(self, backend_id, reported_ns): + matches = [event for event in self.mqsim.observations + if event.get("kind") == 2 and event.get("request_id") == backend_id] + if len(matches) != 1: + raise UnsupportedComposition("UNSUPPORTED_COMPOSITION: missing unique MQSim media callback") + raw_ns = matches[0]["time_ns"] + if raw_ns != reported_ns: + raise UnsupportedComposition( + "UNSUPPORTED_COMPOSITION: MQSim aggregate bound differs from media callback") + return raw_ns + + def _accept_mqsim_completion(self, completion): + if completion is None: + return + backend_id = completion["request_id"] + if backend_id not in self._mq_pending: + raise RuntimeError("duplicate or unknown MQSim completion") + request = self._records[self._mq_pending.pop(backend_id)] + reported = completion["reported_complete"] + raw = self._raw_mqsim_completion(backend_id, reported) + request["backend_media_ns"] = raw + request["backend_completion_ns"] = reported + request["state"] = "FABRIC_PENDING" + self.fabric.mark_source_ready(request["request_id"], reported) + + def _accept_hbm_completions(self): + for completion in self.hbm.take_media_completions() if self.hbm is not None else (): + request = self._records[completion["request_id"]] + if request["state"] != "BACKEND_PENDING": + raise RuntimeError("duplicate or unknown HBM completion") + request["backend_media_ns"] = completion["time_ns"] + request["backend_completion_ns"] = completion["time_ns"] + request["state"] = "FABRIC_PENDING" + self.fabric.mark_source_ready(request["request_id"], completion["time_ns"]) + if self.hbm is not None: + self._hbm_facts.extend(self.hbm.take_facts()) + + def _accept_fabric_completions(self): + completions = self.fabric.completions() + for completion in completions[self._fabric_completion_count:]: + request = self._records[completion["request_id"]] + if request["state"] != "FABRIC_PENDING": + raise RuntimeError("duplicate or premature fabric completion") + request["fabric_completion_ns"] = completion["completion_ns"] + request["final_completion_ns"] = max( + request["backend_completion_ns"], completion["completion_ns"]) + request["state"] = "COMPLETE" + self._fabric_completion_count = len(completions) + + def _next_horizon(self, not_arrived): + candidates = [row["arrival_ns"] for row in not_arrived] + candidates.extend(row["gate_target_ns"] for row in self._records.values() + if row["state"] == "GATE_WAIT") + if self.hbm is not None and self.hbm.next_event_ns() is not None: + candidates.append(self.hbm.next_event_ns()) + if self.fabric.next_event_ns() is not None: + candidates.append(self.fabric.next_event_ns()) + if candidates: + return min(candidates) + if self._mq_pending: + return MAX_HORIZON_NS + return None + + def run(self, requests): + rows = list(requests) + for sequence, raw in enumerate(rows): + self._validate_request(raw, sequence) + not_arrived = sorted(self._records.values(), key=lambda row: (row["arrival_ns"], row["sequence"])) + self._waiting = list(not_arrived) + while any(row["state"] != "COMPLETE" for row in self._records.values()): + not_arrived = [row for row in not_arrived if row["arrival_ns"] > self.now_ns] + self._admit() + horizon = self._next_horizon(not_arrived) + if horizon is None or horizon < self.now_ns: + raise RuntimeError("basic system deadlocked with no legal event") + mq_completion = self.mqsim.until(horizon) + self.now_ns = self.mqsim.now + if self.hbm is not None: + self.hbm.advance(self.now_ns) + self._accept_mqsim_completion(mq_completion) + self._accept_hbm_completions() + self.fabric.advance(self.now_ns) + self._accept_fabric_completions() + receipt = self.mqsim.finish() + state = self.fabric.resource_state() + leaked = state["unfinished"] or any( + any(owner is not None for owner in row["banks"]) or + any(row[name] is not None for name in ("fill", "direct", "relay")) + for row in state["hbf"].values()) or any( + any(owner is not None for owner in row["banks"]) or row["gpu"] is not None + for row in state["hbm"].values()) + if leaked: + raise RuntimeError("fabric resource ownership leaked after completion") + completions = [copy.deepcopy(row) for row in sorted( + self._records.values(), key=lambda row: row["sequence"])] + return { + "schema_version": "eq3-basic-system-v1", + "evidence": "ENGINEERING_FIXTURE_BASIC_SYSTEM", + "topology_mode": self.mode, + "time_ns": self.now_ns, + "completions": completions, + "stack_request_counts": { + stack: sum(row["stack"] == stack for row in completions) + for stack in sorted(self._hbf | self._hbm) + }, + "mqsim_receipt": receipt, + "mqsim_native_events": copy.deepcopy(self.mqsim.native_observations), + "hbm_facts": copy.deepcopy(self._hbm_facts), + "fabric_events": list(self.fabric.events()), + "fabric_resource_state": state, + "limits": { + "hbf_backend": "ACTUAL_MQSIM_CHANNEL_PARTITIONED_HBF_STACKS", + "hbm_backend": "PARAMETRIC_HBM_SCENARIO", + "fabric": "SCENARIO_ASSUMPTION", + "capacity_validation": "UNAVAILABLE_NOT_RESEARCH_GEOMETRY", + "target_throughput_12_8_to_24_5_TBps": "NOT_VALIDATED", + "energy": "UNKNOWN_WHERE_UNPARAMETERIZED", + "thermal_coupling": "NOT_CONNECTED_REUSES_EXISTING_SEPARATE_GATE_FIXTURE", + "research_geometry": False, + }, + "external_devices": ([{ + "kind": "GDDR", "identity": "PHYSICAL_EXTERNAL_GDDR", + "service": "UNAVAILABLE", "temperature": "UNAVAILABLE", + }] if self.mode == "all_hbf_direct" else []), + } diff --git a/tools/eq3_basic_system_actual.py b/tools/eq3_basic_system_actual.py new file mode 100644 index 0000000..5686e13 --- /dev/null +++ b/tools/eq3_basic_system_actual.py @@ -0,0 +1,150 @@ +#!/usr/bin/env python3 +"""Run the four fixed basic-system cases through the actual MQSim CPU service.""" +from __future__ import annotations + +import argparse +import copy +import hashlib +import json +from pathlib import Path +import shutil +import sys + +ROOT = Path(__file__).resolve().parents[1] +sys.path.insert(0, str(ROOT / "scripts" / "eval")) + +from mqsim_service import MqsimService # noqa: E402 +from eq3_basic_fabric import BasicFabric # noqa: E402 +from eq3_basic_hbm import BasicHbm # noqa: E402 +from eq3_basic_system import BasicSystem, engineering_fixture # noqa: E402 + + +def _sha256(path): + digest = hashlib.sha256() + with path.open("rb") as source: + for chunk in iter(lambda: source.read(1024 * 1024), b""): + digest.update(chunk) + return digest.hexdigest() + + +def derive_mqsim_case(base_profile, template_map, mode): + hbf_count = 8 if mode == "all_hbf_direct" else 4 + required_profile = ("capacity_bytes", "page_bytes", "planes_per_die", "pages_per_block") + if any(type(base_profile.get(key)) is not int or base_profile[key] <= 0 + for key in required_profile): + raise ValueError("base MQSim fixture profile lacks positive integer geometry") + if base_profile.get("channels") != 8 or base_profile.get("dies_per_channel") != 1: + raise ValueError("base MQSim fixture must declare 8 channels and one die per channel") + if base_profile.get("plane_allocation_scheme", "CWDP") != "CWDP": + raise ValueError("base MQSim fixture must use CWDP placement") + if (template_map.get("schema_version") != 1 or + template_map.get("physical_kind") != "HBF" or + template_map.get("route") != "direct" or + template_map.get("address_layout") != "GLOBAL_PAGE_STRIPE_V1" or + template_map.get("plane_allocation_scheme") != "CWDP" or + template_map.get("page_bytes") != base_profile["page_bytes"] or + template_map.get("channels") != 8 or + template_map.get("dies_per_channel") != 1): + raise ValueError("template stack map does not match the base 8-HBF fixture") + rows = template_map.get("stacks") + if (not isinstance(rows, list) or len(rows) != 8 or + [row.get("id") for row in rows] != [f"hbf{i}" for i in range(8)] or + any(row.get("declared_dies") != 1 or row.get("channels") != [i] + for i, row in enumerate(rows))): + raise ValueError("template stack map must explicitly name hbf0 through hbf7") + profile = copy.deepcopy(base_profile) + profile["name"] = f"ENGINEERING_FIXTURE_{mode}_{hbf_count}HBF_1DIE" + profile["channels"] = hbf_count + profile["dies_per_channel"] = 1 + profile["plane_allocation_scheme"] = "CWDP" + denominator = (profile["page_bytes"] * profile["channels"] * + profile["dies_per_channel"] * profile["planes_per_die"] * + profile["pages_per_block"]) + if profile["capacity_bytes"] % denominator or profile["capacity_bytes"] // denominator < 4: + raise ValueError("derived MQSim fixture geometry has invalid blocks per plane") + stack_map = copy.deepcopy(template_map) + stack_map.update({"page_bytes": profile["page_bytes"], "channels": hbf_count, + "dies_per_channel": 1, + "evidence": f"DERIVED_ENGINEERING_FIXTURE_{mode}"}) + stack_map["stacks"] = [ + {"id": f"hbf{i}", "declared_dies": 1, "channels": [i]} + for i in range(hbf_count) + ] + return profile, stack_map + + +def main(): + parser = argparse.ArgumentParser() + parser.add_argument("--binary", type=Path, required=True) + parser.add_argument("--profile", type=Path, required=True) + parser.add_argument("--stack-map", type=Path, required=True) + parser.add_argument("--output", type=Path, required=True) + parser.add_argument("--artifact-root", type=Path, default=ROOT) + parser.add_argument("--timeout", type=float, default=30) + args = parser.parse_args() + output = args.output.resolve() + output.mkdir(parents=True, exist_ok=False) + base_profile_path = args.profile.resolve(strict=True) + template_map_path = args.stack_map.resolve(strict=True) + base_profile = json.loads(base_profile_path.read_text()) + template_map = json.loads(template_map_path.read_text()) + page_bytes = base_profile["page_bytes"] + summaries = [] + for mode in ("all_hbf_direct", "mixed_direct", "relay", "dash"): + fixture = engineering_fixture(mode, page_bytes) + directory = output / mode + directory.mkdir() + derived_profile, derived_map = derive_mqsim_case(base_profile, template_map, mode) + source_profile_copy = directory / "mqsim-profile.source.json" + source_map_copy = directory / "mqsim-stack-map.source.json" + shutil.copyfile(base_profile_path, source_profile_copy) + shutil.copyfile(template_map_path, source_map_copy) + profile_path = directory / "mqsim-profile.derived.json" + map_path = directory / "mqsim-stack-map.derived.json" + profile_path.write_text(json.dumps(derived_profile, indent=2, sort_keys=True) + "\n") + map_path.write_text(json.dumps(derived_map, indent=2, sort_keys=True) + "\n") + (directory / "mqsim-input-provenance.json").write_text(json.dumps({ + "base_profile": str(base_profile_path), + "base_profile_sha256": _sha256(base_profile_path), + "saved_base_profile": source_profile_copy.name, + "template_stack_map": str(template_map_path), + "template_stack_map_sha256": _sha256(template_map_path), + "saved_template_stack_map": source_map_copy.name, + "derivation": "channels=topology HBF count; one channel and one die per HBF; CWDP explicit; capacity unchanged", + "mode": mode, + }, indent=2, sort_keys=True) + "\n") + try: + with MqsimService( + args.binary, profile_path, directory / "service", timeout=args.timeout, + artifact_root=args.artifact_root, stack_map=map_path, + native_observations=True) as mqsim: + hbm = None if fixture["hbm"] is None else BasicHbm(fixture["hbm"]) + result = BasicSystem( + mode, mqsim, hbm, BasicFabric(fixture["fabric"])).run(fixture["requests"]) + result["fixture"] = fixture + (directory / "result.json").write_text( + json.dumps(result, indent=2, sort_keys=True, allow_nan=False) + "\n") + summaries.append({"mode": mode, "status": "PASS", + "requests": len(result["completions"]), + "time_ns": result["time_ns"]}) + except BaseException as error: + failure = {"mode": mode, "status": "FAIL", + "error_type": type(error).__name__, "error": str(error), + "fixture": fixture} + (directory / "result.json").write_text( + json.dumps(failure, indent=2, sort_keys=True, allow_nan=False) + "\n") + summaries.append(failure) + (output / "summary.json").write_text( + json.dumps({"status": "FAIL", "cases": summaries}, indent=2, + sort_keys=True, allow_nan=False) + "\n") + raise + summary = {"schema_version": "eq3-basic-system-actual-v1", + "status": "PASS", "cases": summaries, + "claim_limit": "small CPU engineering fixtures; not research geometry"} + (output / "summary.json").write_text( + json.dumps(summary, indent=2, sort_keys=True, allow_nan=False) + "\n") + print(json.dumps(summary, sort_keys=True)) + + +if __name__ == "__main__": + main() diff --git a/tools/eq3_campaign_compare.py b/tools/eq3_campaign_compare.py index 29cdcc8..80903ed 100644 --- a/tools/eq3_campaign_compare.py +++ b/tools/eq3_campaign_compare.py @@ -13,17 +13,22 @@ def read(path): seq.append((t,v)) return result -def crossings(seq,threshold): +def crossings(seq,threshold,initial_k=300.,initial_time_s=0.): # All transitions, both heating and cooling, linear interpolation of the - # registered0.1s observations; t0 explicitly from declared300K initial state. - out=[];previous=(0.,300.) + # registered observations; t0 comes from the selected analysis contract. + if not math.isfinite(initial_k) or not math.isfinite(initial_time_s): + raise ValueError('initial state must be finite') + out=[];previous=(initial_time_s,initial_k) for current in seq: t0,v0=previous;t,v=current if (v0v0 else 'down',t0+(t-t0)*(threshold-v0)/(v-v0))) previous=current return out -def compare(reference,candidate,mode): +def compare(reference,candidate,mode,initial_k=300.,probes_k=(301.,330.)): + probes_k=tuple(float(value) for value in probes_k) + if not math.isfinite(initial_k) or not probes_k or any(not math.isfinite(v) for v in probes_k): + raise ValueError('initial temperature and probes must be finite and probes nonempty') ref=read(reference);cand=read(candidate) if set(ref)!=set(cand):raise ValueError('sensor coverage differs') rows=[] @@ -33,15 +38,15 @@ def compare(reference,candidate,mode): errors=[abs(a[1]-b[1]) for a,b in zip(rv,cv)];mae=math.fsum(errors)/len(errors);maximum=max(errors) scale=max(max(v for t,v in rv)-min(v for t,v in rv),1.) crossing=[] - for threshold in [301,330]: - a=crossings(rv,threshold);b=crossings(cv,threshold) + for threshold in probes_k: + a=crossings(rv,threshold,initial_k);b=crossings(cv,threshold,initial_k) ok=len(a)==len(b) and all(x[0]==y[0] and abs(x[1]-y[1])<=max(.2,.05*x[1]) for x,y in zip(a,b)) crossing.append({'probe_k':threshold,'reference':a,'candidate':b,'status':'NOT_APPLICABLE' if not a and not b else 'PASS' if ok else 'FAIL'}) ok=maximum<=.25 if mode=='reference' else (mae<=1 and mae/scale<=.05 and ('hotspot' not in sid or maximum<=2) and all(x['status']!='FAIL' for x in crossing)) rows.append({'sensor_id':sid,'mae_k':mae,'max_abs_k':maximum,'normalized_mae':mae/scale,'reference_range_k':scale,'crossings':crossing,'passed':ok}) - return {'status':'PASS' if all(r['passed'] for r in rows) else 'NUMERICAL_FAIL','mode':mode,'max_abs_k':max(r['max_abs_k'] for r in rows),'worst_mae_k':max(r['mae_k'] for r in rows),'worst_normalized_mae':max(r['normalized_mae'] for r in rows),'sensors':rows,'crossing_semantics':'all ascending and descending transitions interpolated on0.1s observations; declared initial300K','physical_calibration':False} + return {'status':'PASS' if all(r['passed'] for r in rows) else 'NUMERICAL_FAIL','mode':mode,'max_abs_k':max(r['max_abs_k'] for r in rows),'worst_mae_k':max(r['mae_k'] for r in rows),'worst_normalized_mae':max(r['normalized_mae'] for r in rows),'sensors':rows,'analysis_initial_k':initial_k,'analysis_probes_k':list(probes_k),'crossing_semantics':f'all ascending and descending transitions interpolated on registered observations; selected initial {initial_k:g}K','physical_calibration':False} if __name__=='__main__': - p=argparse.ArgumentParser();p.add_argument('--reference',type=Path,required=True);p.add_argument('--candidate',type=Path,required=True);p.add_argument('--mode',choices=['reference','rc'],required=True);p.add_argument('--output',type=Path,required=True);a=p.parse_args();r=compare(a.reference,a.candidate,a.mode) + p=argparse.ArgumentParser();p.add_argument('--reference',type=Path,required=True);p.add_argument('--candidate',type=Path,required=True);p.add_argument('--mode',choices=['reference','rc'],required=True);p.add_argument('--output',type=Path,required=True);p.add_argument('--initial-k',type=float,default=300.);p.add_argument('--probe-k',type=float,action='append',dest='probes_k');a=p.parse_args();r=compare(a.reference,a.candidate,a.mode,a.initial_k,a.probes_k or (301.,330.)) with a.output.open('x') as f:json.dump(r,f,indent=2) print(json.dumps({k:v for k,v in r.items() if k!='sensors'})) diff --git a/tools/eq3_campaign_gate.py b/tools/eq3_campaign_gate.py index 0ae8c1d..7246d12 100644 --- a/tools/eq3_campaign_gate.py +++ b/tools/eq3_campaign_gate.py @@ -1,5 +1,5 @@ """Small stage-scope gate, not a user-signature synthesizer or scheduler.""" -import hashlib,json,re,copy +import hashlib,json,re,copy,math from pathlib import Path from eq3_experiment_gate import _fail,canonical_manifest_hash @@ -15,6 +15,25 @@ def resolve(root,path): if root not in p.parents or not p.is_file():_fail('STAGE_PATH_INVALID',path) return p +def resource_limits(scope): + """Return a validated per-experiment envelope, or the legacy fixed one.""" + limits=scope['limits'] + if scope.get('resource_policy')!='PER_EXPERIMENT_USER_CONFIRMED': + return {'task_ram_gib':16,'process_ram_gib':12,'threads':1,'watchdog_s':600, + 'point_disk_gib':4,'task_disk_gib':20,'min_free_disk_gib':0,'gpu':0,'cloud':0} + required={'task_ram_gib','process_ram_gib','threads','watchdog_s','point_disk_gib', + 'task_disk_gib','min_free_disk_gib','gpu','cloud'} + if set(limits)!=required:_fail('STAGE_RESOURCE_POLICY_INVALID','per-experiment limits incomplete') + if any(isinstance(limits[x],bool) or not isinstance(limits[x],(int,float)) or not math.isfinite(limits[x]) for x in required): + _fail('STAGE_RESOURCE_POLICY_INVALID','per-experiment limits must be numeric') + if not all(limits[x]>0 for x in ('task_ram_gib','process_ram_gib','threads','watchdog_s','point_disk_gib','task_disk_gib')): + _fail('STAGE_RESOURCE_POLICY_INVALID','per-experiment positive limits required') + if limits['process_ram_gib']>limits['task_ram_gib'] or limits['point_disk_gib']>limits['task_disk_gib'] or limits['min_free_disk_gib']<0: + _fail('STAGE_RESOURCE_POLICY_INVALID','per-experiment resource hierarchy invalid') + if int(limits['threads'])!=limits['threads'] or limits['gpu']!=0 or limits['cloud']!=0: + _fail('STAGE_RESOURCE_POLICY_INVALID','integer threads and zero GPU/cloud required') + return limits + def validate_stage(authorization,manifest,root): root=Path(root).resolve() if authorization.get('schema_version')!='eq3-stage-authorization-v1' or authorization.get('record_kind')!='USER_STAGE_AUTHORIZATION' or authorization.get('is_test_fixture') is not False: @@ -63,7 +82,11 @@ def validate_stage(authorization,manifest,root): expected_hash=canonical(prefix_ir(full,end)) if canonical(ir)!=expected_hash:_fail('STAGE_PHYSICS_CHANGED','IR differs from frozen physical/power/sensor input or exact authorized prefix') r=manifest['resource_budget']['requested'] - if r['ram_gib']>12 or r['disk_gib']>4 or r['build_threads']!=1 or r['gpu_compute_minutes']!=0 or r['cpu_configurations']!=1 or r['executions_per_configuration']!=1:_fail('STAGE_RESOURCE_EXCEEDED','stage envelope') + limits=resource_limits(scope) + if (r['ram_gib']>limits['process_ram_gib'] or r['disk_gib']>limits['point_disk_gib'] or + r['build_threads']!=limits['threads'] or r['gpu_compute_minutes']!=0 or + r['cpu_configurations']!=1 or r['executions_per_configuration']!=1): + _fail('STAGE_RESOURCE_EXCEEDED','stage envelope') if contract['limits']!=scope['limits']:_fail('STAGE_RESOURCE_EXCEEDED','child must retain stage safety limits') if contract['fit_parameters']!=0 or contract['rom_count']!=0:_fail('STAGE_METHOD_OUT_OF_SCOPE','no fit/ROM') if not 01e-8 or round(ratio)&(round(ratio)-1):_fail('STAGE_NUMERICS_INVALID','non-dyadic reference grid') - return {'status':'AUTHORIZED_BY_USER_STAGE_SCOPE','stage_id':scope['stage_id'],'point_id':contract['point_id']} + return {'status':'AUTHORIZED_BY_USER_STAGE_SCOPE','stage_id':scope['stage_id'],'point_id':contract['point_id'], + 'resource_policy':scope.get('resource_policy','LEGACY_FIXED'),'resource_limits':limits} diff --git a/tools/eq3_campaign_prepare.py b/tools/eq3_campaign_prepare.py index e3fdf48..597604b 100644 --- a/tools/eq3_campaign_prepare.py +++ b/tools/eq3_campaign_prepare.py @@ -5,18 +5,27 @@ def sha(p): with p.open('rb') as f:return hashlib.file_digest(f,'sha256').hexdigest() -def prepare(root,point,generated,trace,step,mesh,policy,family,deps,reason='',binary=None,model_lock=None,reference_environment=None): - code=root/'eq3_thermal/worktree';campaign=root/'eq3_thermal/plans/campaign-v1';plan=campaign/'points'/point - auth_path=campaign/'authorization-v3.json' - auth=json.loads(auth_path.read_text());scope=json.loads((root/auth['scope_path']).read_text());g=generated.resolve();r=json.loads((g/'generation_receipt.json').read_text()) - base=root/'eq3_thermal/plans/layered-R02-v1';m=json.loads((base/'experiment_manifest.json').read_text());l=json.loads((base/'launch.json').read_text()) +def prepare(root,point,generated,trace,step,mesh,policy,family,deps,reason='',binary=None,model_lock=None,reference_environment=None, + code_root=None,campaign_path=None,authorization_path=None,base_plan=None,preflight=None,regression_evidence=None,diagnostic_domain_version=None): + root=root.resolve() + def local(value,default): + p=Path(value) if value is not None else Path(default);p=(p if p.is_absolute() else root/p).resolve() + if p!=root and root not in p.parents:raise ValueError('path escapes task root') + return p + code=local(code_root,'eq3_thermal/worktree');campaign=local(campaign_path,'eq3_thermal/plans/campaign-v1');plan=campaign/'points'/point + auth_path=local(authorization_path,campaign/'authorization-v3.json') + if not code.is_dir() or not campaign.is_dir() or not auth_path.is_file():raise ValueError('code, campaign, or authorization path unavailable') + auth=json.loads(auth_path.read_text());scope=json.loads(local(auth['scope_path'],auth['scope_path']).read_text());g=generated.resolve() + if root not in g.parents or not g.is_dir():raise ValueError('generated input path outside task root or unavailable') + r=json.loads((g/'generation_receipt.json').read_text()) + base=local(base_plan,'eq3_thermal/plans/layered-R02-v1');m=json.loads((base/'experiment_manifest.json').read_text());l=json.loads((base/'launch.json').read_text()) def art(p,role,identity=None): rel=p.resolve().relative_to(root).as_posix();return {'path':rel,'logical_id':identity or rel,'semantic_role':role,'sha256':sha(p)} def write(name,obj): with (plan/name).open('x') as f:json.dump(obj,f,indent=2) rev,diff=observed_git_state(code) if diff!=hashlib.sha256(b'').hexdigest():raise ValueError('commit changes before freezing child') - contract={'stage_id':scope['stage_id'],'authorization_class':'AUTHORIZED_BY_USER_STAGE_SCOPE','point_id':point,'family':family,'trace':trace,'step_s':step,'mesh_um':mesh,'limits':scope['limits'],'fit_parameters':0,'rom_count':0,'output_policy':policy,'derivation_reason':reason,'dependencies':deps,'model_lock':model_lock,'purpose':'same-physics numerical validation; see registered family/dependencies','changes_from_parent':reason or 'registered numerical point','expected_runtime':'UNKNOWN; watchdog600s, not an estimate','expected_information':'reference discretization or same-equation RC validation, not physical calibration'} + contract={'stage_id':scope['stage_id'],'authorization_class':'AUTHORIZED_BY_USER_STAGE_SCOPE','point_id':point,'family':family,'trace':trace,'step_s':step,'mesh_um':mesh,'limits':scope['limits'],'fit_parameters':0,'rom_count':0,'output_policy':policy,'derivation_reason':reason,'dependencies':deps,'model_lock':model_lock,'purpose':'same-physics numerical validation; see registered family/dependencies','changes_from_parent':reason or 'registered numerical point','expected_runtime':f'UNKNOWN; watchdog{scope["limits"].get("watchdog_s",600)}s, not an estimate','expected_information':'reference discretization or same-equation RC validation, not physical calibration'} plan.mkdir(parents=True,exist_ok=False);(plan/'runs').mkdir();write('child.json',contract) files=[art(p,'generated physical/numerical input') for p in sorted(g.iterdir()) if p.is_file()] original_engine=m['dependencies'][0] @@ -41,8 +50,10 @@ def write(name,obj): l['command']['cwd']=g.relative_to(root).as_posix();l['output']['path']=(plan/'runs'/point).relative_to(root).as_posix() if family in ('rc','rc_pilot'): l['backend']=art(binary,'sparse same-equation RC backend','rc-backend');l['command']['argv']=[l['backend']['path'],'--run','--model','model.txt','--events','events.txt','--step-s',str(step),'--slot-s','.5','--sample-s','.1','--end-s',str(r['duration_s']),'--min-k','300','--max-k','400'] + if diagnostic_domain_version is not None: + l['command']['argv'] += ['--model-sha256',sha(g/'model.txt'),'--events-sha256',sha(g/'events.txt'),'--runner-source-sha256',sha(code/'tools/eq3_campaign_rc_runner.cpp'),'--domain-version',diagnostic_domain_version] elif policy in ('lossless_gzip','lossless_EQ3TMK1'): - l['backend']=art(code/'tools/eq3_campaign_stream.py','byte-lossless output adapter','stream-backend');nx,ny,nz=r['reference_shape'];l['command']['argv']=[l['backend']['path'],'--backend',reference_engine['path'],'--stack','package.stk','--layers',str(nz),'--frames',str(round(r['duration_s']/step)),'--nx',str(nx),'--ny',str(ny),'--watchdog','600'] + l['backend']=art(code/'tools/eq3_campaign_stream.py','byte-lossless output adapter','stream-backend');nx,ny,nz=r['reference_shape'];l['command']['argv']=[l['backend']['path'],'--backend',reference_engine['path'],'--stack','package.stk','--layers',str(nz),'--frames',str(round(r['duration_s']/step)),'--nx',str(nx),'--ny',str(ny),'--watchdog',str(scope['limits'].get('watchdog_s',600)),'--process-ram-gib',str(scope['limits'].get('process_ram_gib',12)),'--max-output-gib',str(scope['limits'].get('point_disk_gib',4))] if policy=='lossless_EQ3TMK1': codec=art(root/'eq3_thermal/build/campaign-codec-v1/libeq3_campaign_nativecodec.so','byte-lossless native field codec','native-field-codec') m['dependencies'].append(codec) @@ -50,22 +61,36 @@ def write(name,obj): else: l['backend']=reference_engine l['command']['argv']=[l['backend']['path'],'package.stk'] + if scope.get('resource_policy')=='PER_EXPERIMENT_USER_CONFIRMED': + limits=scope['limits'];l['limits'].update(threads=limits['threads'],ram_gib=limits['process_ram_gib'],watchdog_seconds=limits['watchdog_s'],gpu_compute_minutes=0) + l['output']['max_new_gib']=limits['point_disk_gib'] + requested=m['resource_budget']['requested'];requested.update(build_threads=limits['threads'],ram_gib=limits['process_ram_gib'],disk_gib=limits['point_disk_gib'],gpu_compute_minutes=0,cpu_configurations=1,executions_per_configuration=1) + declared=m['resource_budget']['limits'];declared.update(build_threads=limits['threads'],ram_gib=limits['process_ram_gib'],disk_gib=limits['point_disk_gib'],gpu_compute_minutes=0,cpu_configurations=1,executions_per_configuration=1) write('launch.json',l) m.update(experiment_id=l['experiment_id'],version=l['version'],approval_status='USER_APPROVED') # USER_APPROVED describes stage eligibility; not a fabricated per-point signature. - m['code']={'repository_revision':rev,'dirty_diff_sha256':diff,'artifacts':[art(code/p['path'].split('eq3_thermal/worktree/')[1],'source') for p in m['code']['artifacts']]} + def source_relative(item): + parts=Path(item['path']).parts + if 'worktree' not in parts:raise ValueError('base source artifact has no portable checkout-relative path') + return Path(*parts[parts.index('worktree')+1:]) + m['code']={'repository_revision':rev,'dirty_diff_sha256':diff,'artifacts':[art(code/source_relative(p),'source') for p in m['code']['artifacts']]} seen={x['path'] for x in m['code']['artifacts']} tracked=subprocess.check_output(['git','-C',str(code),'ls-files','tools/eq3_campaign*'],text=True).splitlines() for p in [code/name for name in tracked]: if p.is_file() and p.relative_to(root).as_posix() not in seen:m['code']['artifacts'].append(art(p,'stage execution source')) if l['backend']['logical_id']!=original_engine['logical_id']:m['dependencies'].append(l['backend']) m['dependencies'].append(art(root/'tools/collect_experiment_metadata.py','prelaunch environment inventory and metadata validator','metadata-collector')) - m['dependencies'].append(art(campaign/'FOLLOWUP_DIAGNOSTIC_PREFLIGHT.md','stage-derived diagnostic preflight')) + preflight_path=local(preflight,campaign/'FOLLOWUP_DIAGNOSTIC_PREFLIGHT.md') + if preflight is not None and not preflight_path.is_file():raise ValueError('explicit preflight evidence unavailable') + if preflight_path.is_file():m['dependencies'].append(art(preflight_path,'stage-derived diagnostic preflight')) if family in ('rc','rc_pilot'): m['dependencies'] += [art(root/'environments/eq3-thermal-campaign-rc-v2/manifest.json','frozen sparse CPU environment'),art(root/'eq3_thermal/build/campaign-rc-v2/validation_receipt.json','sparse build and fixed equivalence evidence')] m['inputs']=files+[p for p in m['inputs'] if p['semantic_role']=='source scientific input']+[art(plan/'child.json','stage derived child scope','stage-child-contract'),art(plan/'launch.json','single run launch binding','layered-launch-manifest')] # Preserve original fixed-test evidence; current campaign regression is also bound. - m['prerequisites']=[m['prerequisites'][0],{'id':'campaign-regression','status':'PASSED','evidence':art(campaign/'software-tests-v6.log','fixed regression raw evidence')}] + regression_path=local(regression_evidence,campaign/'software-tests-v6.log') + if regression_evidence is not None and not regression_path.is_file():raise ValueError('explicit regression evidence unavailable') + if regression_path.is_file(): + m['prerequisites']=[m['prerequisites'][0],{'id':'campaign-regression','status':'PASSED','evidence':art(regression_path,'fixed regression raw evidence')}] s=m['scientific_config'];s['workload_and_initial_state'].update(trace=trace,duration_s=r['duration_s'],input_energy_j=r['input_energy_j']) s['research_question_and_hypothesis']['question']=f'{point}: same-physics {family} numerical/resource validation at step{step}s and reference mesh{mesh}um; no physical calibration' s['geometry_materials_boundaries'].update(shape=r['reference_shape'],cells=r['reference_cells'],rc_nodes=r['rc_nodes']) @@ -82,4 +107,4 @@ def write(name,obj): write('experiment_manifest.json',m);print(json.dumps({'point':point,'manifest_hash':m['canonical_manifest_hash'],'status':'AUTHORIZED_BY_USER_STAGE_SCOPE','launch_performed':False}));return plan if __name__=='__main__': - p=argparse.ArgumentParser();p.add_argument('--root',type=Path,required=True);p.add_argument('--point',required=True);p.add_argument('--generated',type=Path,required=True);p.add_argument('--trace',required=True);p.add_argument('--step',type=float,required=True);p.add_argument('--mesh',type=float,default=0);p.add_argument('--policy',choices=['full_text','lossless_gzip','lossless_EQ3TMK1'],required=True);p.add_argument('--family',choices=['reference','reference_pilot','rc','rc_pilot','numerical_refinement'],required=True);p.add_argument('--dependencies',type=Path,required=True);p.add_argument('--reason',default='');p.add_argument('--binary',type=Path);p.add_argument('--reference-environment',type=Path);a=p.parse_args();prepare(a.root.resolve(),a.point,a.generated,a.trace,a.step,a.mesh,a.policy,a.family,json.loads(a.dependencies.read_text()),a.reason,a.binary,reference_environment=a.reference_environment) + p=argparse.ArgumentParser();p.add_argument('--root',type=Path,required=True);p.add_argument('--point',required=True);p.add_argument('--generated',type=Path,required=True);p.add_argument('--trace',required=True);p.add_argument('--step',type=float,required=True);p.add_argument('--mesh',type=float,default=0);p.add_argument('--policy',choices=['full_text','lossless_gzip','lossless_EQ3TMK1'],required=True);p.add_argument('--family',choices=['reference','reference_pilot','rc','rc_pilot','numerical_refinement'],required=True);p.add_argument('--dependencies',type=Path,required=True);p.add_argument('--reason',default='');p.add_argument('--binary',type=Path);p.add_argument('--reference-environment',type=Path);p.add_argument('--code-root',type=Path);p.add_argument('--campaign',type=Path);p.add_argument('--authorization-path',type=Path);p.add_argument('--base-plan',type=Path);p.add_argument('--preflight',type=Path);p.add_argument('--regression-evidence',type=Path);p.add_argument('--diagnostic-domain-version');a=p.parse_args();prepare(a.root.resolve(),a.point,a.generated,a.trace,a.step,a.mesh,a.policy,a.family,json.loads(a.dependencies.read_text()),a.reason,a.binary,reference_environment=a.reference_environment,code_root=a.code_root,campaign_path=a.campaign,authorization_path=a.authorization_path,base_plan=a.base_plan,preflight=a.preflight,regression_evidence=a.regression_evidence,diagnostic_domain_version=a.diagnostic_domain_version) diff --git a/tools/eq3_campaign_rc_runner.cpp b/tools/eq3_campaign_rc_runner.cpp index cbcfd65..6032c1f 100644 --- a/tools/eq3_campaign_rc_runner.cpp +++ b/tools/eq3_campaign_rc_runner.cpp @@ -29,10 +29,17 @@ using SparseMatrix = Eigen::SparseMatrix; using SparseSolver = Eigen::SimplicialLDLT>; +std::string json_string(const std::string& value); + struct Options { std::string model_path, events_path; + std::string model_sha256{"UNKNOWN_NOT_SUPPLIED"}; + std::string events_sha256{"UNKNOWN_NOT_SUPPLIED"}; + std::string runner_source_sha256{"UNKNOWN_NOT_SUPPLIED"}; + std::string domain_version{"eq3-runner-cli-temperature-domain-v1"}; double step_s{}, slot_s{}, end_s{}, sample_s{}, min_k{}, max_k{}; - bool inspect_only{}, run{}, equilibrium_diagnostic{}; + double envelope_limit_k{}; + bool inspect_only{}, run{}, equilibrium_diagnostic{}, steady_envelope{}; }; struct Schedule { @@ -78,6 +85,11 @@ Options options_from(int argc, char** argv) { options.equilibrium_diagnostic = true; continue; } + if (argument == "--steady-envelope") { + require(!options.steady_envelope, "duplicate --steady-envelope"); + options.steady_envelope = true; + continue; + } if (argument == "--inspect-only") { require(!options.inspect_only, "duplicate --inspect-only"); options.inspect_only = true; @@ -91,21 +103,31 @@ Options options_from(int argc, char** argv) { const bool takes_value = argument == "--model" || argument == "--events" || argument == "--step-s" || argument == "--slot-s" || argument == "--end-s" || argument == "--sample-s" || - argument == "--min-k" || argument == "--max-k"; + argument == "--min-k" || argument == "--max-k" || + argument == "--envelope-limit-k" || + argument == "--model-sha256" || argument == "--events-sha256" || + argument == "--runner-source-sha256" || argument == "--domain-version"; require(takes_value, "unknown argument " + argument); require(index + 1 < argc, "missing value after " + argument); const std::string value = argv[++index]; if (argument == "--model") options.model_path = value; else if (argument == "--events") options.events_path = value; + else if (argument == "--model-sha256") options.model_sha256 = value; + else if (argument == "--events-sha256") options.events_sha256 = value; + else if (argument == "--runner-source-sha256") options.runner_source_sha256 = value; + else if (argument == "--domain-version") options.domain_version = value; else if (argument == "--step-s") options.step_s = number(value, argument); else if (argument == "--slot-s") options.slot_s = number(value, argument); else if (argument == "--end-s") options.end_s = number(value, argument); else if (argument == "--sample-s") options.sample_s = number(value, argument); else if (argument == "--min-k") options.min_k = number(value, argument); else if (argument == "--max-k") options.max_k = number(value, argument); + else if (argument == "--envelope-limit-k") + options.envelope_limit_k = number(value, argument); } - require(int(options.inspect_only) + int(options.run) + int(options.equilibrium_diagnostic) == 1, - "exactly one of --inspect-only, --equilibrium-diagnostic or --run is required"); + require(int(options.inspect_only) + int(options.run) + + int(options.equilibrium_diagnostic) + int(options.steady_envelope) == 1, + "exactly one of --inspect-only, --equilibrium-diagnostic, --steady-envelope or --run is required"); require(!options.model_path.empty() && !options.events_path.empty(), "--model and --events are required"); require(options.step_s > 0 && options.slot_s > 0 && options.end_s > 0 && @@ -113,6 +135,11 @@ Options options_from(int argc, char** argv) { "step, slot, end, and sample seconds must be positive"); require(options.min_k > 0 && options.max_k > options.min_k, "--min-k and --max-k must satisfy 0 < min < max"); + require(!options.steady_envelope || options.envelope_limit_k > 0, + "--steady-envelope requires positive --envelope-limit-k"); + require(!options.model_sha256.empty() && !options.events_sha256.empty() && + !options.runner_source_sha256.empty() && !options.domain_version.empty(), + "diagnostic identity values must not be empty"); return options; } @@ -284,6 +311,128 @@ SparseMatrix system_matrix(const ThermalModelConfig& config, double step_s) { return matrix; } +SparseMatrix steady_matrix(const ThermalModelConfig& config) { + require(config.nodes.size() <= static_cast(std::numeric_limits::max()), + "node count exceeds Eigen int index range"); + const int n = static_cast(config.nodes.size()); + std::vector> entries; + entries.reserve(config.nodes.size() + 4 * config.edges.size()); + double boundary_total = 0; + for (int index = 0; index < n; ++index) { + const double boundary = + config.nodes[static_cast(index)].boundary_conductance_w_per_k; + entries.emplace_back(index, index, boundary); + boundary_total += boundary; + } + require(boundary_total > 0, "steady envelope requires at least one heat-rejection boundary"); + for (const auto& edge : config.edges) { + if (!config.direct_intercomponent_edges_enabled && + edge.kind == ConductanceKind::InterComponent) + continue; + const int a = static_cast(edge.node_a); + const int b = static_cast(edge.node_b); + const double conductance = edge.conductance_w_per_k; + entries.emplace_back(a, a, conductance); + entries.emplace_back(b, b, conductance); + entries.emplace_back(a, b, -conductance); + entries.emplace_back(b, a, -conductance); + } + SparseMatrix matrix(n, n); + matrix.setFromTriplets(entries.begin(), entries.end()); + matrix.makeCompressed(); + return matrix; +} + +void steady_envelope(const ThermalModelConfig& config, + const std::vector& activities, + const Options& options) { + const Eigen::Index n = static_cast(config.nodes.size()); + Eigen::VectorXd fixed_rhs(n), cap_power = Eigen::VectorXd::Zero(n); + for (Eigen::Index index = 0; index < n; ++index) { + const auto& node = config.nodes[static_cast(index)]; + fixed_rhs[index] = node.static_power_w + + node.boundary_conductance_w_per_k * node.boundary_temperature_k; + } + double cap_total_w = 0; + for (const auto& activity : activities) { + const double duration = activity.end_time_s - activity.start_time_s; + require(duration > 0, "steady cap activity must have positive duration"); + for (const auto& assignment : activity.node_energy) { + const double power = assignment.energy_j / duration; + require(std::isfinite(power) && power >= 0, + "steady cap activity power must be finite and nonnegative"); + cap_power[static_cast(assignment.node_index)] += power; + cap_total_w += power; + } + } + require(cap_total_w > 0, "steady envelope requires positive all-source cap power"); + const SparseMatrix matrix = steady_matrix(config); + SparseSolver solver; + const auto factor_start = std::chrono::steady_clock::now(); + solver.compute(matrix); + const double factor_seconds = std::chrono::duration( + std::chrono::steady_clock::now() - factor_start).count(); + require(solver.info() == Eigen::Success, "steady L factorization failed"); + const Eigen::VectorXd fixed = solver.solve(fixed_rhs); + require(solver.info() == Eigen::Success && fixed.allFinite(), + "steady fixed-source solve failed"); + const Eigen::VectorXd cap_rise = solver.solve(cap_power); + require(solver.info() == Eigen::Success && cap_rise.allFinite(), + "steady cap solve failed"); + require(cap_rise.minCoeff() >= -1e-10, + "steady cap response violates positive-network monotonicity"); + const double fixed_residual = + (matrix * fixed - fixed_rhs).lpNorm(); + const double cap_residual = + (matrix * cap_rise - cap_power).lpNorm(); + const std::vector alphas{1.0, 0.75, 0.5, 0.25}; + double selected = -1; + std::ostringstream candidates; + candidates << std::setprecision(17) << '['; + for (std::size_t number = 0; number < alphas.size(); ++number) { + const double alpha = alphas[number]; + const Eigen::VectorXd envelope = fixed + alpha * cap_rise; + bool initial_covered = true; + for (Eigen::Index index = 0; index < n; ++index) + initial_covered &= config.nodes[static_cast(index)].initial_temperature_k <= + envelope[index] + 1e-10; + const bool within = envelope.maxCoeff() <= options.envelope_limit_k; + if (selected < 0 && within && initial_covered) selected = alpha; + Eigen::Index maximum_index{}; + const double maximum = envelope.maxCoeff(&maximum_index); + if (number) candidates << ','; + candidates << "{\"alpha\":" << alpha << ",\"max_k\":" << maximum + << ",\"max_node\":" + << json_string(config.nodes[static_cast(maximum_index)].id) + << ",\"within_limit\":" << (within ? "true" : "false") + << ",\"initial_covered\":" << (initial_covered ? "true" : "false") + << '}'; + } + candidates << ']'; + std::cout << std::setprecision(17) + << "{\"schema_version\":\"eq3-steady-envelope-v1\"," + << "\"status\":" << json_string(selected > 0 ? "PREDICTED_ENVELOPE" + : "DOMAIN_REDESIGN_REQUIRED") + << ",\"workload_executed\":false,\"reference_qualified\":false," + << "\"model_sha256\":" << json_string(options.model_sha256) + << ",\"events_sha256\":" << json_string(options.events_sha256) + << ",\"runner_source_sha256\":" << json_string(options.runner_source_sha256) + << ",\"domain_version\":" << json_string(options.domain_version) + << ",\"declared_temperature_domain_k\":[" << options.min_k << ',' + << options.max_k << "]," + << "\"cap_total_w\":" << cap_total_w + << ",\"envelope_limit_k\":" << options.envelope_limit_k + << ",\"selected_alpha\":"; + if (selected > 0) std::cout << selected; + else std::cout << "null"; + std::cout << ",\"matrix_nnz\":" << matrix.nonZeros() + << ",\"factor_L_nnz\":" << solver.matrixL().nestedExpression().nonZeros() + << ",\"factor_seconds\":" << factor_seconds + << ",\"fixed_residual_inf\":" << fixed_residual + << ",\"cap_residual_inf\":" << cap_residual + << ",\"candidates\":" << candidates.str() << "}\n"; +} + // Fixed zero-source equilibrium diagnostic, never the supplied workload. void equilibrium_diagnostic(const ThermalModelConfig& config, double step_s) { const double reference = config.nodes.front().initial_temperature_k; @@ -350,6 +499,247 @@ void emit(double time_s, const ThermalModelConfig& config, << temperature[static_cast(index)] << '\n'; } +std::string json_string(const std::string& value) { + std::ostringstream out; + out << '"'; + for (const unsigned char character : value) { + switch (character) { + case '"': out << "\\\""; break; + case '\\': out << "\\\\"; break; + case '\b': out << "\\b"; break; + case '\f': out << "\\f"; break; + case '\n': out << "\\n"; break; + case '\r': out << "\\r"; break; + case '\t': out << "\\t"; break; + default: + if (character < 0x20) + out << "\\u" << std::hex << std::setfill('0') << std::setw(4) + << static_cast(character) << std::dec << std::setfill(' '); + else + out << character; + } + } + return out.str() + '"'; +} + +std::string csv_string(const std::string& value) { + std::string result{"\""}; + for (const char character : value) { + if (character == '"') result += '"'; + result += character; + } + return result + '"'; +} + +std::string json_number_or_null(double value) { + if (!std::isfinite(value)) return "null"; + std::ostringstream output; + output << std::setprecision(17) << value; + return output.str(); +} + +void write_state(const std::filesystem::path& final, + const ThermalModelConfig& config, + const Eigen::VectorXd& temperature) { + const std::filesystem::path temporary{final.string() + ".tmp"}; + require(!std::filesystem::exists(final) && !std::filesystem::exists(temporary), + "refusing to overwrite failure state " + final.string()); + std::ofstream output(temporary, std::ios::out | std::ios::trunc); + if (!output) throw std::runtime_error("cannot create failure state " + final.string()); + output << "node_index,node_id,group_id,die_index,temperature_k\n" + << std::setprecision(17); + for (std::size_t index = 0; index < config.nodes.size(); ++index) { + const auto& node = config.nodes[index]; + output << index << ',' << csv_string(node.id) << ',' << csv_string(node.group_id) << ','; + if (node.die_index) output << *node.die_index; + else output << "UNKNOWN"; + output << ',' << temperature[static_cast(index)] << '\n'; + } + output.close(); + if (!output) throw std::runtime_error("failed writing failure state " + final.string()); + std::filesystem::rename(temporary, final); +} + +long double stored_energy(const ThermalModelConfig& config, + const Eigen::VectorXd& theta, double origin) { + long double result = 0; + for (Eigen::Index index = 0; index < theta.size(); ++index) { + const auto& node = config.nodes[static_cast(index)]; + result += node.heat_capacity_j_per_k * + (static_cast(theta[index]) - + (node.initial_temperature_k - origin)); + } + return result; +} + +void unknown_trial_diagnostic(const Options& options, + const ThermalModelConfig& config, + const Eigen::VectorXd& last_valid_temperature, + const Eigen::VectorXd& last_valid_theta, + const Eigen::VectorXd& applied, + const Eigen::VectorXd& power, + std::uint64_t completed_steps, + std::uint64_t trial_step, std::uint64_t slot, + double last_valid_time_s, double trial_time_s, + long double boundary_loss_j, double origin_k, + const std::string& reason) { + const std::filesystem::path final{"rc_failure_diagnostic.json"}; + const std::filesystem::path temporary{"rc_failure_diagnostic.json.tmp"}; + require(!std::filesystem::exists(final) && !std::filesystem::exists(temporary), + "refusing to overwrite RC failure diagnostic"); + write_state("rc_failure_last_valid.csv", config, last_valid_temperature); + long double activity_completed = 0; + for (Eigen::Index index = 0; index < applied.size(); ++index) + activity_completed += applied[index]; + const long double static_completed = static_energy(config, last_valid_time_s); + const long double stored_completed = stored_energy(config, last_valid_theta, origin_k); + const long double residual_completed = activity_completed + static_completed - + stored_completed - boundary_loss_j; + std::ofstream output(temporary, std::ios::out | std::ios::trunc); + if (!output) throw std::runtime_error("cannot create RC failure diagnostic"); + output << std::setprecision(17) + << "{\n" + << " \"schema_version\": \"eq3-campaign-sparse-rc-failure-v1\",\n" + << " \"status\": \"NUMERICAL_FAILURE\",\n" + << " \"exit_reason\": " << json_string(reason) << ",\n" + << " \"failure_returned_to_caller\": true,\n" + << " \"last_valid_state_file\": \"rc_failure_last_valid.csv\",\n" + << " \"trial_state_file\": null,\n" + << " \"trial_state_status\": \"UNKNOWN_SOLVER_FAILURE\",\n" + << " \"last_valid_time_s\": " << last_valid_time_s << ",\n" + << " \"trial_target_time_s\": " << trial_time_s << ",\n" + << " \"completed_steps\": " << completed_steps << ",\n" + << " \"trial_step_1_based\": " << trial_step << ",\n" + << " \"step_s\": " << options.step_s << ",\n" + << " \"input_slot_index_0_based\": " << slot << ",\n" + << " \"input_interval_start_s\": " << static_cast(slot) * options.slot_s << ",\n" + << " \"input_interval_end_s\": " << static_cast(slot + 1) * options.slot_s << ",\n" + << " \"trial_activity_power_w\": " << json_number_or_null(power.sum()) << ",\n" + << " \"domain_version\": " << json_string(options.domain_version) << ",\n" + << " \"domain_min_k\": " << options.min_k << ",\n" + << " \"domain_max_k\": " << options.max_k << ",\n" + << " \"model_sha256\": " << json_string(options.model_sha256) << ",\n" + << " \"events_sha256\": " << json_string(options.events_sha256) << ",\n" + << " \"runner_source_sha256\": " << json_string(options.runner_source_sha256) << ",\n" + << " \"completed_interval_energy\": {\"activity_input_j\":" << activity_completed + << ",\"static_input_j\":" << static_completed + << ",\"stored_energy_change_j\":" << stored_completed + << ",\"boundary_loss_j\":" << boundary_loss_j + << ",\"energy_residual_j\":" << residual_completed << "},\n" + << " \"failed_trial_step_energy\": {\"status\":\"UNKNOWN_NOT_INTEGRATED\"," + "\"activity_input_j\":null,\"static_input_j\":null," + "\"stored_energy_change_j\":null,\"boundary_loss_j\":null," + "\"energy_residual_j\":null}\n" + << "}\n"; + output.close(); + if (!output) throw std::runtime_error("failed writing RC failure diagnostic"); + std::filesystem::rename(temporary, final); +} + +void failure_diagnostic(const Options& options, const ThermalModelConfig& config, + const Eigen::VectorXd& last_valid_temperature, + const Eigen::VectorXd& trial_temperature, + const Eigen::VectorXd& last_valid_theta, + const Eigen::VectorXd& applied, + const Eigen::VectorXd& power, + std::uint64_t completed_steps, std::uint64_t trial_step, + std::uint64_t slot, double last_valid_time_s, + double trial_time_s, long double boundary_loss_j, + double origin_k, const std::string& reason) { + const std::filesystem::path final{"rc_failure_diagnostic.json"}; + const std::filesystem::path temporary{"rc_failure_diagnostic.json.tmp"}; + require(!std::filesystem::exists(final) && !std::filesystem::exists(temporary), + "refusing to overwrite RC failure diagnostic"); + write_state("rc_failure_last_valid.csv", config, last_valid_temperature); + write_state("rc_failure_trial.csv", config, trial_temperature); + + std::vector offending; + std::size_t nonfinite_count = 0; + for (Eigen::Index index = 0; index < trial_temperature.size(); ++index) { + const double value = trial_temperature[index]; + if (!std::isfinite(value)) ++nonfinite_count; + if (!std::isfinite(value) || value < options.min_k || value > options.max_k) + offending.push_back(static_cast(index)); + } + long double activity_completed = 0; + for (Eigen::Index index = 0; index < applied.size(); ++index) + activity_completed += applied[index]; + const long double static_completed = static_energy(config, last_valid_time_s); + const long double stored_completed = stored_energy(config, last_valid_theta, origin_k); + const long double residual_completed = activity_completed + static_completed - + stored_completed - boundary_loss_j; + const bool finite_trial = trial_temperature.allFinite(); + + std::ofstream output(temporary, std::ios::out | std::ios::trunc); + if (!output) throw std::runtime_error("cannot create RC failure diagnostic"); + output << std::setprecision(17) + << "{\n" + << " \"schema_version\": \"eq3-campaign-sparse-rc-failure-v1\",\n" + << " \"status\": " + << json_string(nonfinite_count ? "NUMERICAL_FAILURE" : "DOMAIN_FAILURE") << ",\n" + << " \"exit_reason\": " << json_string(reason) << ",\n" + << " \"failure_returned_to_caller\": true,\n" + << " \"temperature_clamping\": false,\n" + << " \"last_valid_state_file\": \"rc_failure_last_valid.csv\",\n" + << " \"trial_state_file\": \"rc_failure_trial.csv\",\n" + << " \"last_valid_time_s\": " << last_valid_time_s << ",\n" + << " \"trial_target_time_s\": " << trial_time_s << ",\n" + << " \"completed_steps\": " << completed_steps << ",\n" + << " \"trial_step_1_based\": " << trial_step << ",\n" + << " \"step_s\": " << options.step_s << ",\n" + << " \"input_slot_index_0_based\": " << slot << ",\n" + << " \"input_interval_start_s\": " << static_cast(slot) * options.slot_s << ",\n" + << " \"input_interval_end_s\": " << static_cast(slot + 1) * options.slot_s << ",\n" + << " \"trial_activity_power_w\": " << json_number_or_null(power.sum()) << ",\n" + << " \"domain_version\": " << json_string(options.domain_version) << ",\n" + << " \"domain_min_k\": " << options.min_k << ",\n" + << " \"domain_max_k\": " << options.max_k << ",\n" + << " \"model_path\": " << json_string(options.model_path) << ",\n" + << " \"events_path\": " << json_string(options.events_path) << ",\n" + << " \"model_sha256\": " << json_string(options.model_sha256) << ",\n" + << " \"events_sha256\": " << json_string(options.events_sha256) << ",\n" + << " \"runner_source_sha256\": " << json_string(options.runner_source_sha256) << ",\n" + << " \"coordinate_status\": \"UNKNOWN_NOT_EXPOSED_BY_MODEL_TEXT_API\",\n" + << " \"nonfinite_trial_nodes\": " << nonfinite_count << ",\n" + << " \"trial_extrema_available\": " << (finite_trial ? "true" : "false") << ",\n"; + if (finite_trial) + output << " \"trial_min_k\": " << trial_temperature.minCoeff() << ",\n" + << " \"trial_max_k\": " << trial_temperature.maxCoeff() << ",\n"; + else + output << " \"trial_min_k\": null,\n \"trial_max_k\": null,\n"; + output << " \"offending_nodes\": ["; + for (std::size_t position = 0; position < offending.size(); ++position) { + if (position) output << ','; + const std::size_t index = offending[position]; + const auto& node = config.nodes[index]; + output << "{\"index\":" << index << ",\"id\":" << json_string(node.id) + << ",\"group_id\":" << json_string(node.group_id) << ",\"die_index\":"; + if (node.die_index) output << *node.die_index; + else output << "null"; + output << ",\"temperature_k\":"; + const double value = trial_temperature[static_cast(index)]; + if (std::isfinite(value)) output << value; + else output << json_string(std::isnan(value) ? "NaN" : value > 0 ? "+Infinity" : "-Infinity"); + output << '}'; + } + output << "],\n" + << " \"completed_interval_energy\": {\n" + << " \"activity_input_j\": " << activity_completed << ",\n" + << " \"static_input_j\": " << static_completed << ",\n" + << " \"stored_energy_change_j\": " << stored_completed << ",\n" + << " \"boundary_loss_j\": " << boundary_loss_j << ",\n" + << " \"energy_residual_j\": " << residual_completed << "\n" + << " },\n" + << " \"failed_trial_step_energy\": {\"status\":\"UNKNOWN_NOT_INTEGRATED\"," + "\"activity_input_j\":null,\"static_input_j\":null," + "\"stored_energy_change_j\":null,\"boundary_loss_j\":null," + "\"energy_residual_j\":null}\n" + << "}\n"; + output.close(); + if (!output) throw std::runtime_error("failed writing RC failure diagnostic"); + std::filesystem::rename(temporary, final); +} + void receipt(const Options& options, const Schedule& schedule, const ThermalModelConfig& config, std::size_t matrix_nnz, double alignment_error_s, double min_observed_k, @@ -412,9 +802,12 @@ void run(const ThermalModelConfig& config, const std::vector& activities, const Options& options, const Schedule& schedule, double alignment_error_s) { - require(!std::filesystem::exists("rc_energy_receipt.json") && - !std::filesystem::exists("rc_energy_receipt.json.tmp"), - "refusing to overwrite RC energy receipt"); + for (const char* path : {"rc_energy_receipt.json", "rc_energy_receipt.json.tmp", + "rc_failure_diagnostic.json", "rc_failure_diagnostic.json.tmp", + "rc_failure_last_valid.csv", "rc_failure_last_valid.csv.tmp", + "rc_failure_trial.csv", "rc_failure_trial.csv.tmp"}) + require(!std::filesystem::exists(path), + std::string("refusing to overwrite RC output ") + path); const SparseMatrix matrix = system_matrix(config, options.step_s); SparseSolver solver; const auto factor_start = std::chrono::steady_clock::now(); @@ -437,14 +830,6 @@ void run(const ThermalModelConfig& config, (node.boundary_temperature_k - origin); } double minimum = temperature.minCoeff(), maximum = temperature.maxCoeff(); - const auto check_domain = [&] { - require(temperature.allFinite(), "sparse solve produced nonfinite temperature"); - minimum = std::min(minimum, temperature.minCoeff()); - maximum = std::max(maximum, temperature.maxCoeff()); - require(temperature.minCoeff() >= options.min_k && - temperature.maxCoeff() <= options.max_k, - "temperature left required domain"); - }; std::cout << "time_s,node_id,temperature_k\n" << std::setprecision(17); emit(0.0, config, temperature); long double boundary_loss = 0; @@ -455,10 +840,39 @@ void run(const ThermalModelConfig& config, ++global; const Eigen::VectorXd rhs = capacity_over_dt.cwiseProduct(theta) + static_and_boundary + power; - theta = solver.solve(rhs); - temperature = theta.array() + origin; - require(solver.info() == Eigen::Success, "sparse LDLT solve failed"); - check_domain(); + const Eigen::VectorXd trial_theta = solver.solve(rhs); + const Eigen::VectorXd trial_temperature = trial_theta.array() + origin; + const double trial_time = local == schedule.steps_per_slot + ? static_cast(slot + 1) * options.slot_s + : static_cast(slot) * options.slot_s + + static_cast(local) * options.step_s; + const double last_valid_time = local == 1 + ? static_cast(slot) * options.slot_s + : static_cast(slot) * options.slot_s + + static_cast(local - 1) * options.step_s; + if (solver.info() != Eigen::Success) { + unknown_trial_diagnostic(options, config, temperature, theta, applied, power, + global - 1, global, slot, last_valid_time, + trial_time, boundary_loss, origin, + "sparse LDLT solve failed"); + fail("sparse LDLT solve failed"); + } + const bool finite = trial_temperature.allFinite(); + const bool in_domain = finite && trial_temperature.minCoeff() >= options.min_k && + trial_temperature.maxCoeff() <= options.max_k; + if (!in_domain) { + failure_diagnostic(options, config, temperature, trial_temperature, theta, + applied, power, global - 1, global, slot, + last_valid_time, trial_time, boundary_loss, + origin, finite ? "temperature left required domain" + : "sparse solve produced nonfinite temperature"); + fail(finite ? "temperature left required domain" + : "sparse solve produced nonfinite temperature"); + } + theta = trial_theta; + temperature = trial_temperature; + minimum = std::min(minimum, temperature.minCoeff()); + maximum = std::max(maximum, temperature.maxCoeff()); for (Eigen::Index index = 0; index < n; ++index) { const auto& node = config.nodes[static_cast(index)]; boundary_loss += static_cast(options.step_s) * @@ -476,14 +890,11 @@ void run(const ThermalModelConfig& config, } } } - long double applied_energy = 0, stored = 0; + long double applied_energy = 0; for (Eigen::Index index = 0; index < n; ++index) { applied_energy += applied[index]; - const auto& node = config.nodes[static_cast(index)]; - stored += node.heat_capacity_j_per_k * - (static_cast(theta[index]) - - (node.initial_temperature_k - origin)); } + const long double stored = stored_energy(config, theta, origin); receipt(options, schedule, config, static_cast(matrix.nonZeros()), alignment_error_s, minimum, maximum, declared_activity_energy(activities), applied_energy, stored, @@ -509,6 +920,10 @@ int main(int argc, char** argv) try { equilibrium_diagnostic(config, options.step_s); return 0; } + if (options.steady_envelope) { + steady_envelope(config, activities, options); + return 0; + } run(config, activities, options, schedule, alignment_error); return 0; } catch (const std::exception& error) { diff --git a/tools/eq3_campaign_run.py b/tools/eq3_campaign_run.py index 46a11a3..c6d70f1 100644 --- a/tools/eq3_campaign_run.py +++ b/tools/eq3_campaign_run.py @@ -1,27 +1,43 @@ """Operational limits/measurements only; never modify bound science or retry.""" -import argparse, hashlib, json, os, resource, signal, subprocess, sys, time +import argparse, hashlib, json, os, resource, shutil, signal, subprocess, sys, time from pathlib import Path p=argparse.ArgumentParser();p.add_argument('mode',choices=['solve','observe','observe-audit']);p.add_argument('--root',type=Path,required=True);p.add_argument('--point',type=Path,required=True) a=p.parse_args(); root=a.root.resolve(); plan=a.point.resolve() -code=root/'eq3_thermal/worktree'; launch=json.loads((plan/'launch.json').read_text());run=root/launch['output']['path']; derived=plan/'derived' +if plan!=root and root not in plan.parents:raise SystemExit('Point path escapes task root') +launch=json.loads((plan/'launch.json').read_text());contract=json.loads((plan/'child.json').read_text());manifest=json.loads((plan/'experiment_manifest.json').read_text()) +def local(value,must_file=False): + path=(root/value).resolve() + if path!=root and root not in path.parents:raise SystemExit('Bound path escapes task root') + if must_file and not path.is_file():raise SystemExit('Bound file unavailable') + return path +code=local(manifest['execution_context']['code_root']);run=local(launch['output']['path']); derived=plan/'derived' +if not code.is_dir():raise SystemExit('Bound code root unavailable') if a.mode=='observe-audit':derived=plan/'derived-invalid-domain-audit' receipt=plan/(a.mode+'-operational.json') if receipt.exists(): raise SystemExit('No operational retry/overwrite allowed') gib=1024**3 -resource.setrlimit(resource.RLIMIT_AS,(512*1024**2,12*gib)) +authorization_path=local(manifest.get('execution_context',{}).get('authorization_path','eq3_thermal/plans/campaign-v1/authorization.json'),True) +authorization=json.loads(authorization_path.read_text());scope_path=local(authorization['scope_path'],True) +if hashlib.sha256(scope_path.read_bytes()).hexdigest()!=authorization['scope_sha256']:raise SystemExit('Bound stage scope changed') +scope=json.loads(scope_path.read_text()) +if contract['stage_id']!=scope['stage_id']:raise SystemExit('Child stage differs from bound scope') +if contract['limits']!=scope['limits']:raise SystemExit('Child resource limits differ from bound stage scope') +if scope.get('resource_policy')=='PER_EXPERIMENT_USER_CONFIRMED': + limits=scope['limits'];process_ram=limits['process_ram_gib'];task_ram=limits['task_ram_gib'];watchdog=limits['watchdog_s'] + point_disk=limits['point_disk_gib'];task_disk=limits['task_disk_gib'];min_free=limits['min_free_disk_gib'] +else: + process_ram,task_ram,watchdog,point_disk,task_disk,min_free=12,16,600,4,20,0 +resource.setrlimit(resource.RLIMIT_AS,(512*1024**2,int(process_ram*gib))) cpu=min(os.sched_getaffinity(0));os.sched_setaffinity(0,{cpu}) for key in ('OMP_NUM_THREADS','OPENBLAS_NUM_THREADS','MKL_NUM_THREADS','NUMEXPR_NUM_THREADS','BLIS_NUM_THREADS','VECLIB_MAXIMUM_THREADS'):os.environ[key]='1' os.environ['CUDA_VISIBLE_DEVICES']='' -contract=json.loads((plan/'child.json').read_text()) -manifest=json.loads((plan/'experiment_manifest.json').read_text()) for dependency in manifest['dependencies']: if dependency['logical_id']=='native-field-codec': library=root/dependency['path'] if hashlib.sha256(library.read_bytes()).hexdigest()!=dependency['sha256']: raise SystemExit('Bound codec library changed') os.environ['EQ3_CAMPAIGN_NATIVECODEC']=str(library) -authorization_path=root/manifest.get('execution_context',{}).get('authorization_path','eq3_thermal/plans/campaign-v1/authorization.json') if a.mode=='solve': argv=[sys.executable,'-B',str(code/'tools/eq3_layered_launch.py'),'run','--root',str(root),'--code-root',str(code),'--manifest',str(plan/'experiment_manifest.json'),'--launch',str(plan/'launch.json'),'--approval',str(authorization_path)] else: @@ -34,7 +50,7 @@ def bytes_in(path):return sum(x.stat().st_size for x in path.rglob('*') if x.is_file()) initial_task_bytes=bytes_in(root/'eq3_thermal') -if initial_task_bytes>=20*gib:raise SystemExit('Insufficient conservative task disk allowance') +if initial_task_bytes>=task_disk*gib or shutil.disk_usage(root).free16*gib/1024:reason='TASK_RSS_LIMIT' - if size>4*gib:reason='COMBINED_OUTPUT_LIMIT' - if bytes_in(root/'eq3_thermal')>20*gib:reason='TASK_DISK_LIMIT' - if a.mode!='solve' and time.monotonic()-started>600:reason='POSTPROCESS_WATCHDOG' + if rss>task_ram*gib/1024:reason='TASK_RSS_LIMIT' + if size>point_disk*gib:reason='COMBINED_OUTPUT_LIMIT' + if bytes_in(root/'eq3_thermal')>task_disk*gib or shutil.disk_usage(root).freewatchdog:reason='WATCHDOG' if reason: for pid in ids-{os.getpid()}: try:os.kill(pid,signal.SIGKILL) @@ -96,8 +112,9 @@ def limits(): 'affinity_cpu':cpu,'affinity_is_operational_not_scientific':True, 'initial_task_disk_bytes':initial_task_bytes,'plan_output_bytes':bytes_in(plan), 'task_disk_bytes_after':bytes_in(root/'eq3_thermal'),'task_disk_accounting':'entire eq3_thermal tree conservative, includes historical files', - 'task_memory_mechanism':'serial phases, <=512MiB launcher + <=12GiB single solver + <=512MiB supervisor; descendant RSS monitor16GiB; no cgroup changes', - 'disk_mechanism':'bound launcher4GiB raw monitor plus outer4GiB combined plan/raw/derived monitor; sampled not filesystem quota', + 'resource_policy':scope.get('resource_policy','LEGACY_FIXED'),'bound_limits':scope['limits'], + 'task_memory_mechanism':f'serial phases, <=512MiB launcher + <={process_ram}GiB single process; descendant RSS monitor{task_ram}GiB; no cgroup changes', + 'disk_mechanism':f'bound launcher{point_disk}GiB raw monitor plus outer{point_disk}GiB combined monitor; task{task_disk}GiB and free reserve{min_free}GiB; sampled not filesystem quota', 'supervisor_sha256':hashlib.sha256(Path(__file__).read_bytes()).hexdigest()} with receipt.open('x') as f:json.dump(result,f,indent=2) if a.mode=='solve': diff --git a/tools/eq3_campaign_stream.py b/tools/eq3_campaign_stream.py index 0cbac0a..1dd78ed 100755 --- a/tools/eq3_campaign_stream.py +++ b/tools/eq3_campaign_stream.py @@ -141,8 +141,9 @@ def replay(source, destination, nx, ny, frames): def run_stream(argv, directory, layers, frames, nx, ny, watchdog=600, - max_bytes=4 * 1024**3, policy='gzip', codec_library=None): - if min(layers, frames, nx, ny) < 1 or not 0 < watchdog <= 600: + max_bytes=4 * 1024**3, policy='gzip', codec_library=None, + process_ram_gib=12): + if min(layers, frames, nx, ny) < 1 or watchdog <= 0 or process_ram_gib <= 0: raise ValueError('invalid dimensions or watchdog') if policy not in ('gzip', 'lossless_EQ3TMK1'): raise ValueError('unknown output policy') @@ -161,7 +162,7 @@ def run_stream(argv, directory, layers, frames, nx, ny, watchdog=600, process = None status, error, records = 'FAILED', None, [] def child_limits(): - resource.setrlimit(resource.RLIMIT_AS, (12 * 1024**3, 12 * 1024**3)) + limit=int(process_ram_gib * 1024**3);resource.setrlimit(resource.RLIMIT_AS, (limit, limit)) try: for z in range(layers): pipe = directory / f'field_{z}.txt' @@ -243,14 +244,17 @@ def main(): for flag in ('layers', 'frames', 'nx', 'ny'): p.add_argument('--' + flag, type=int, required=True) p.add_argument('--watchdog', type=float, default=600) + p.add_argument('--process-ram-gib', type=float, default=12) + p.add_argument('--max-output-gib', type=float, default=4) p.add_argument('--policy', choices=('gzip','lossless_EQ3TMK1'), default='gzip') p.add_argument('--codec-library') a = p.parse_args() # A low soft ceiling for this bridge must not propagate to the solver. hard = resource.getrlimit(resource.RLIMIT_AS)[1] - if hard != resource.RLIM_INFINITY and hard < 12 * 1024**3: + approved_bytes=int(a.process_ram_gib * 1024**3) + if a.process_ram_gib <= 0 or a.max_output_gib <= 0 or hard != resource.RLIM_INFINITY and hard < approved_bytes: raise ValueError('outer hard AS limit cannot accommodate approved solver ceiling') - resource.setrlimit(resource.RLIMIT_AS, (512 * 1024**2, 12 * 1024**3)) + resource.setrlimit(resource.RLIMIT_AS, (512 * 1024**2, approved_bytes)) os.sched_setaffinity(0, {min(os.sched_getaffinity(0))}) backend = Path(a.backend) if not backend.is_absolute(): @@ -260,7 +264,7 @@ def main(): codec_library = Path(os.environ.get('EQ3_ARTIFACT_ROOT', str(Path.cwd()))) / codec_library run_stream([str(backend.resolve()), a.stack], Path.cwd(), a.layers, a.frames, a.nx, a.ny, a.watchdog, - policy=a.policy, codec_library=codec_library) + max_bytes=int(a.max_output_gib*1024**3),policy=a.policy, codec_library=codec_library,process_ram_gib=a.process_ram_gib) if __name__ == '__main__': diff --git a/tools/eq3_cpu_load_compare.py b/tools/eq3_cpu_load_compare.py new file mode 100644 index 0000000..5251508 --- /dev/null +++ b/tools/eq3_cpu_load_compare.py @@ -0,0 +1,35 @@ +"""Aggregate exactly nine reviewed arms without manufacturing a winner.""" +import argparse,json +from pathlib import Path +from eq3_cpu_load_matrix import POLICIES,RATES + +def compare(rows): + keyed={(x['rate_per_stack_rps'],x['policy']):x for x in rows} + if set(keyed)!={(r,p) for r in RATES for p in POLICIES}:raise ValueError('exact approved nine-point matrix required') + output={'matrix_id':'EQ3-D4-SUSTAINED-LOAD-v2','evidence':'ENGINEERING_FIXTURE','rates':{}, + 'interpretation':'descriptive deterministic comparison; negative results retained; no winner or product claim'} + for rate in RATES: + arms={p:keyed[(rate,p)] for p in POLICIES};assert len({x['workload_identity_sha256'] for x in arms.values()})==1 + baseline=arms['none'];table={} + for policy,x in arms.items(): + p95=x['latency_completed_s']['p95'];base_p95=baseline['latency_completed_s']['p95'] + table[policy]={'foreground_complete':x['foreground_complete'],'foreground_unfinished':x['foreground_unfinished'], + 'completed_logical_bytes':x['completed_logical_bytes'],'max_temperature_k':x['max_temperature_k'], + 'latency_completed_p95_s':p95,'maintenance_queued':x['maintenance']['queued'], + 'maintenance_overdue_cohorts':x['maintenance']['overdue_cohorts'],'max_data_age_s':x['max_data_age_s'], + 'package_dynamic_energy_j':x['package_dynamic_energy_j'],'external_energy_j':x['external_energy_j'], + 'delta_vs_none':{'foreground_complete':x['foreground_complete']-baseline['foreground_complete'], + 'foreground_unfinished':x['foreground_unfinished']-baseline['foreground_unfinished'], + 'max_temperature_k':x['max_temperature_k']-baseline['max_temperature_k'], + 'latency_completed_p95_s':None if p95 is None or base_p95 is None else p95-base_p95, + 'maintenance_queued':x['maintenance']['queued']-baseline['maintenance']['queued']}, + 'negative_flags':{'lower_temperature_with_more_backlog':x['max_temperature_k']baseline['foreground_unfinished'], + 'no_temperature_improvement':policy!='none' and x['max_temperature_k']>=baseline['max_temperature_k'], + 'maintenance_backlog_increase':x['maintenance']['queued']>baseline['maintenance']['queued']}} + output['rates'][str(rate)]={'workload_identity_sha256':baseline['workload_identity_sha256'],'arms':table} + return output + +def main(): + p=argparse.ArgumentParser();p.add_argument('--reviews',type=Path,nargs='+',required=True);a=p.parse_args() + print(json.dumps(compare([json.loads(x.read_text()) for x in a.reviews]),indent=2)) +if __name__=='__main__':main() diff --git a/tools/eq3_cpu_load_matrix.py b/tools/eq3_cpu_load_matrix.py new file mode 100644 index 0000000..614d49f --- /dev/null +++ b/tools/eq3_cpu_load_matrix.py @@ -0,0 +1,60 @@ +"""Generate the approved nine-point sustained-load CPU engineering matrix. + +This reuses the frozen Near engineering fixture. It changes only request arrival +rate, observation horizon, and the explicitly selected controller policy. +Generation performs no simulation. +""" +import argparse,hashlib,json +from pathlib import Path +from eq3_cpu_fixture import make_case + +RATES=(5,10,25) +POLICIES=('none','hysteresis','hysteresis_escalation_priority_v2') +ARRIVAL_S=20 +END_S=30 + +def canonical_hash(value): + return hashlib.sha256(json.dumps(value,sort_keys=True,separators=(',',':')).encode()).hexdigest() + +def make_load_case(code,rate,policy): + if rate not in RATES or policy not in POLICIES:raise ValueError('unapproved matrix coordinate') + case=make_case(code,'Near',policy);stacks=case['config']['stacks'];interval=1_000_000_000//rate + requests=[] + for i in range(rate*ARRIVAL_S): + for stack in stacks: + requests.append({'id':f'{stack["id"]}:{i}','stack':stack['id'],'die':i%stack['die_count'], + 'route':'direct','op':'read','arrival_ns':i*interval,'logical_bytes':4096, + 'physical_bytes':4096,'link_bytes':4096,'fail_fraction':0}) + case.update(matrix_id='EQ3-D4-SUSTAINED-LOAD-v2',scenario_label=f'rate{rate:02d}',rate_per_stack_rps=rate, + arrival_window_ns=ARRIVAL_S*1_000_000_000,recovery_window_ns=(END_S-ARRIVAL_S)*1_000_000_000, + end_ns=END_S*1_000_000_000,requests=requests) + identity_config=dict(case['config']);identity_config.pop('policy') + case['workload_identity_sha256']=canonical_hash({'config_without_policy':identity_config,'requests':requests, + 'arrival_window_ns':case['arrival_window_ns'],'end_ns':case['end_ns']}) + rho=rate*case['config']['duration_ns']['read']*1e-9 + case['predictions']['foreground_stack_utilization_before_maintenance']=rho + case['predictions']['interpretation']='offered load varies only by the approved per-stack rate; backlog and maintenance interference are valid outcomes' + case['screening']={'foreground_base_utilization_per_stack':rho,'light_capacity_rps':1e9/case['config']['control'][stacks[0]['id']]['light_gap_ns'], + 'hbf_maintenance_base_utilization':16*(case['config']['duration_ns']['read']+case['config']['duration_ns']['program'])*1e-9/2, + 'hbm_maintenance_base_utilization':12*case['config']['duration_ns']['dram_refresh']*1e-9/1, + 'prediction':'25rps exceeds the 12.5rps Light admission ceiling before maintenance; backlog is an allowed negative result'} + case['claim_limits'].extend(['fixed 20s arrival plus 10s recovery window; not steady state', + 'rate labels are offered load, not promised thermal states','one deterministic run per arm; no confidence interval']) + return case + +def main(): + p=argparse.ArgumentParser();p.add_argument('--output',type=Path,required=True);a=p.parse_args() + a.output.mkdir(parents=True,exist_ok=False);code=Path(__file__).resolve().parents[1];receipts=[] + for rate in RATES: + identities=set() + for policy in POLICIES: + case=make_load_case(code,rate,policy);identities.add(case['workload_identity_sha256']) + path=a.output/f'rate{rate:02d}-{policy}.json';path.write_text(json.dumps(case,indent=2)+'\n') + receipts.append({'file':path.name,'sha256':hashlib.sha256(path.read_bytes()).hexdigest(), + 'rate_per_stack_rps':rate,'policy':policy,'requests':len(case['requests']), + 'workload_identity_sha256':case['workload_identity_sha256']}) + assert len(identities)==1 + (a.output/'generation_receipt.json').write_text(json.dumps({'kind':'STATIC_INPUT_ONLY','matrix_id':'EQ3-D4-SUSTAINED-LOAD-v2', + 'cases':receipts,'rates_per_stack_rps':RATES,'policies':POLICIES,'arrival_s':ARRIVAL_S,'recovery_s':END_S-ARRIVAL_S, + 'interpretation':'ENGINEERING_FIXTURE; no physical/product/P5 claim'},indent=2)+'\n') +if __name__=='__main__':main() diff --git a/tools/eq3_cpu_load_review.py b/tools/eq3_cpu_load_review.py new file mode 100644 index 0000000..c39219b --- /dev/null +++ b/tools/eq3_cpu_load_review.py @@ -0,0 +1,77 @@ +"""Independent detailed review for one sustained-load CpuService result.""" +import argparse,collections,hashlib,json,math +from pathlib import Path +from eq3_cpu_review import review as base_review + +def percentile(values,q): + if not values:return None + x=sorted(values);position=(len(x)-1)*q;lo=math.floor(position);hi=math.ceil(position) + return x[lo] if lo==hi else x[lo]*(hi-position)+x[hi]*(position-lo) + +def queue_trace(jobs,end_ns,step_ns=1_000_000_000): + rows=[] + for t in range(0,end_ns+1,step_ns): + row={'time_ns':t} + for maintenance,label in ((False,'foreground'),(True,'maintenance')): + selected=[j for j in jobs if j['maintenance']==maintenance and j['arrival_ns']<=t] + row[label+'_queued']=sum('start_ns' not in j or j['start_ns']>t for j in selected) + row[label+'_inflight']=sum(j.get('start_ns',t+1)<=t=limits[key]: + count+=1;duration+=max(0,(rows[i+1]['time_ns'] if i+10 for x in overdue),'max_overdue_s':max(overdue)*1e-9}, + component_energy_j=result['energy_j'],package_dynamic_energy_j=sum(result['energy_j'].values()), + external_energy_j=result['external_energy_j'],negative_results_are_valid=True) + assert summary['offered_foreground']==len(case['requests']) + assert summary['foreground_complete']+summary['foreground_failed']+summary['foreground_unfinished']==summary['offered_foreground'] + return summary + +def main(): + p=argparse.ArgumentParser();p.add_argument('--input',type=Path,required=True);p.add_argument('--raw',type=Path,required=True);a=p.parse_args() + case=json.loads(a.input.read_text());raw=json.loads(a.raw.read_text());result=detailed_review(case,raw) + result['input_sha256']=hashlib.sha256(a.input.read_bytes()).hexdigest();result['raw_sha256']=hashlib.sha256(a.raw.read_bytes()).hexdigest() + print(json.dumps(result,indent=2)) +if __name__=='__main__':main() diff --git a/tools/eq3_domain_v2_input.py b/tools/eq3_domain_v2_input.py new file mode 100644 index 0000000..4de88e9 --- /dev/null +++ b/tools/eq3_domain_v2_input.py @@ -0,0 +1,144 @@ +"""Derive train/development DOMAIN_V2 inputs from one frozen steady alpha.""" +import argparse +import copy +import hashlib +import json +import math +from pathlib import Path + + +ALPHAS = (1.0, 0.75, 0.5, 0.25) +ALLOWED_TRACES = ("train", "development") + + +def sha256(path): + return hashlib.sha256(path.read_bytes()).hexdigest() + + +def finite_nonnegative(value, label): + if isinstance(value, bool) or not isinstance(value, (int, float)): + raise ValueError(f"{label} must be numeric") + value = float(value) + if not math.isfinite(value) or value < 0: + raise ValueError(f"{label} must be finite and nonnegative") + return value + + +def selected_alpha(receipt): + if (receipt.get("schema_version") != "eq3-steady-envelope-v1" or + receipt.get("status") != "PREDICTED_ENVELOPE" or + receipt.get("reference_qualified") is not False): + raise ValueError("steady receipt is not a conditional predicted envelope") + alpha = finite_nonnegative(receipt.get("selected_alpha"), "selected_alpha") + if alpha not in ALPHAS: + raise ValueError("selected_alpha is outside the frozen candidate set") + candidates = receipt.get("candidates") + if not isinstance(candidates, list) or [item.get("alpha") for item in candidates] != list(ALPHAS): + raise ValueError("steady receipt does not retain the frozen candidate order") + eligible = [float(item["alpha"]) for item in candidates + if item.get("within_limit") is True and item.get("initial_covered") is True] + if not eligible or alpha != eligible[0]: + raise ValueError("selected_alpha is not the largest eligible frozen candidate") + return alpha + + +def derive(power, receipt, traces): + traces = tuple(traces) + if not traces or len(traces) != len(set(traces)) or any(item not in ALLOWED_TRACES for item in traces): + raise ValueError("only unique train/development traces may be derived") + alpha = selected_alpha(receipt) + order = power.get("group_order") + caps = power.get("caps_W") + if (not isinstance(order, list) or len(order) != 17 or len(set(order)) != 17 or + not isinstance(caps, list) or len(caps) != len(order)): + raise ValueError("power source ledger must retain 17 ordered groups and caps") + original_caps = [finite_nonnegative(value, f"caps_W[{index}]") + for index, value in enumerate(caps)] + source_traces = power.get("traces") + if not isinstance(source_traces, dict): + raise ValueError("power source ledger lacks traces") + derived_traces = {} + energy = {} + slot_s = finite_nonnegative(power.get("slot_s"), "slot_s") + if slot_s <= 0: + raise ValueError("slot_s must be positive") + for trace_id in traces: + source = source_traces.get(trace_id) + if not isinstance(source, dict): + raise ValueError(f"missing requested trace {trace_id}") + slots = source.get("slots_W") + if not isinstance(slots, list) or not slots: + raise ValueError(f"trace {trace_id} lacks slots_W") + scaled_slots = [] + original_j = 0.0 + scaled_j = 0.0 + for slot_index, raw in enumerate(slots): + if not isinstance(raw, list) or len(raw) != len(order): + raise ValueError(f"trace {trace_id} slot {slot_index} violates frozen vector order") + values = [finite_nonnegative(value, f"{trace_id}.slots_W[{slot_index}]") + for value in raw] + if any(value > cap + 1e-12 for value, cap in zip(values, original_caps, strict=True)): + raise ValueError(f"trace {trace_id} exceeds the public group cap") + scaled = [value * alpha for value in values] + scaled_slots.append(scaled) + original_j += slot_s * math.fsum(values) + scaled_j += slot_s * math.fsum(scaled) + trace = copy.deepcopy(source) + trace["slots_W"] = scaled_slots + derived_traces[trace_id] = trace + energy[trace_id] = {"original_energy_j": original_j, + "scaled_energy_j": scaled_j} + derived = {key: copy.deepcopy(value) for key, value in power.items() if key != "traces"} + derived["schema_version"] = "eq3-domain-v2-power-v1" + derived["status"] = "DOMAIN_V2_IN_RANGE_DERIVED_INPUT" + derived["caps_W"] = [value * alpha for value in original_caps] + derived["traces"] = derived_traces + derived["domain_v2"] = { + "alpha": alpha, + "alpha_candidates": list(ALPHAS), + "source_semantics": "uniform scale of every variable source group; static model sources unchanged", + "blind_trace_included": False, + "future_blind_rule": "after explicit unseal, apply this same frozen alpha without reselection", + "reference_qualification_inherited": False, + } + result = { + "schema_version": "eq3-domain-v2-input-receipt-v1", + "status": "DERIVED_INPUT_NOT_EXECUTED", + "alpha": alpha, + "traces_read": list(traces), + "blind_trajectory_read": False, + "future_blind_rule": "same frozen alpha after dependency-controlled unseal", + "group_count": len(order), + "original_caps_w": dict(zip(order, original_caps, strict=True)), + "scaled_caps_w": dict(zip(order, derived["caps_W"], strict=True)), + "energy": energy, + "reference_qualification_inherited": False, + } + return derived, result + + +def main(): + parser = argparse.ArgumentParser() + parser.add_argument("--power", type=Path, required=True) + parser.add_argument("--steady-receipt", type=Path, required=True) + parser.add_argument("--trace", action="append", required=True) + parser.add_argument("--output", type=Path, required=True) + parser.add_argument("--receipt-output", type=Path, required=True) + args = parser.parse_args() + for output in (args.output, args.receipt_output): + if output.exists(): + raise FileExistsError(f"refusing to overwrite {output}") + derived, receipt = derive(json.loads(args.power.read_text()), + json.loads(args.steady_receipt.read_text()), args.trace) + receipt["source_sha256"] = { + "power": sha256(args.power), "steady_receipt": sha256(args.steady_receipt)} + text = json.dumps(derived, indent=2) + "\n" + receipt["derived_power_sha256"] = hashlib.sha256(text.encode()).hexdigest() + args.output.write_text(text) + args.receipt_output.write_text(json.dumps(receipt, indent=2) + "\n") + print(json.dumps({"status": receipt["status"], "alpha": receipt["alpha"], + "traces": receipt["traces_read"]})) + + +if __name__ == "__main__": + main() diff --git a/tools/eq3_experiment_gate.py b/tools/eq3_experiment_gate.py index 6412e84..61a5f9a 100644 --- a/tools/eq3_experiment_gate.py +++ b/tools/eq3_experiment_gate.py @@ -314,14 +314,18 @@ def validate_gate(manifest, approval, root, approval_path=None, _fail("CODE_DIFF_MISMATCH", "current tracked Git diff differs from the bound diff") if approval is None: _fail("PENDING_USER_APPROVAL", "no independently supplied user confirmation record was provided") + stage = None if approval.get('record_kind') == 'USER_STAGE_AUTHORIZATION': from eq3_campaign_gate import validate_stage - validate_stage(approval, manifest, root) + stage = validate_stage(approval, manifest, root) else: validate_approval(approval, manifest, expected_hash, approval_path) - return {"status": "READY_FOR_SUBMISSION", "experiment_id": manifest["experiment_id"], + result={"status": "READY_FOR_SUBMISSION", "experiment_id": manifest["experiment_id"], "version": manifest["version"], "canonical_manifest_hash": expected_hash, "launch_performed": False} + if stage: + result.update(resource_policy=stage['resource_policy'],resource_limits=stage['resource_limits']) + return result def main(argv=None): diff --git a/tools/eq3_layered_launch.py b/tools/eq3_layered_launch.py index ac09553..a2cd08f 100644 --- a/tools/eq3_layered_launch.py +++ b/tools/eq3_layered_launch.py @@ -248,7 +248,13 @@ def validate_launch(manifest, approval, launch, launch_path, root, approval_path ram_gib = _number(launch["limits"]["ram_gib"], "limits.ram_gib") watchdog = _number(launch["limits"]["watchdog_seconds"], "limits.watchdog_seconds") gpu_minutes = _number(launch["limits"]["gpu_compute_minutes"], "limits.gpu_compute_minutes") - if (threads != MAX_THREADS or not 0 < ram_gib <= MAX_RAM_GIB or + if gate.get("resource_policy") == "PER_EXPERIMENT_USER_CONFIRMED": + limits=gate["resource_limits"] + if (threads != limits["threads"] or ram_gib != limits["process_ram_gib"] or + watchdog != limits["watchdog_s"] or output_gib != limits["point_disk_gib"] or + gpu_minutes != 0): + _fail("LAUNCH_LIMIT_EXCEEDED", "launch limits differ from the bound per-experiment stage scope") + elif (threads != MAX_THREADS or not 0 < ram_gib <= MAX_RAM_GIB or not 0 < watchdog <= MAX_WATCHDOG_SECONDS or gpu_minutes != 0 or not 0 < output_gib <= MAX_OUTPUT_GIB): _fail("LAUNCH_LIMIT_EXCEEDED", "launch exceeds 1 thread, 12 GiB RAM, 600 s, 4 GiB, or zero-GPU limits") @@ -258,8 +264,11 @@ def validate_launch(manifest, approval, launch, launch_path, root, approval_path requested["cpu_configurations"] != 1 or requested["executions_per_configuration"] != 1): _fail("LAUNCH_LIMIT_EXCEEDED", "scientific resource request is not covered by this single-run envelope") - return {"status": "READY_TO_LAUNCH", "experiment_id": gate["experiment_id"], + result={"status": "READY_TO_LAUNCH", "experiment_id": gate["experiment_id"], "version": gate["version"], "run_id": run_id, "launch_performed": False} + if gate.get("resource_policy") == "PER_EXPERIMENT_USER_CONFIRMED": + result.update(resource_policy=gate["resource_policy"],resource_limits=gate["resource_limits"]) + return result def _output_bytes(path): @@ -420,6 +429,12 @@ def execute_launch(manifest, approval, launch, launch_path, root, approval_path= backend = _relative(root, launch["backend"]["path"], "backend.path", must_be_file=True) input_directory = _relative(root, launch["command"]["cwd"], "command.cwd", must_be_dir=True) output = _relative(root, launch["output"]["path"], "output.path") + if result.get("resource_policy") == "PER_EXPERIMENT_USER_CONFIRMED": + limits=result["resource_limits"];task_root=root/"eq3_thermal" if (root/"eq3_thermal").is_dir() else root + point_bytes=int(limits["point_disk_gib"]*1024**3) + if (_output_bytes(task_root)+point_bytes>limits["task_disk_gib"]*1024**3 or + shutil.disk_usage(root).free end_s + 1e-9: + raise ValueError("prefix retained an out-of-window timestamp") + return {"rows": rows, "frames": len(times), "first_time_s": min(times), + "last_time_s": max(times)} + + +def common_window(coarse, middle, fine, end_s=4.0, initial_k=300.0, + probes_k=(301.0, 330.0)): + sources = (coarse, middle, fine) + with tempfile.TemporaryDirectory() as root: + reduced = [] + coverage = [] + for index, source in enumerate(sources): + target = Path(root) / f"grid-{index}.csv" + coverage.append(prefix(source, target, end_s)) + reduced.append(target) + coarse_middle = compare(reduced[0], reduced[1], "reference", + initial_k, probes_k) + middle_fine = compare(reduced[1], reduced[2], "reference", + initial_k, probes_k) + return { + "status": "DIAGNOSTIC_ONLY", + "scope": (f"common [first observation,{end_s:g}s] window; does not replace " + "full memory/base excitation or establish convergence order"), + "initial_k": initial_k, + "probes_k": list(probes_k), + "coverage": [dict(source=str(path), **item) + for path, item in zip(sources, coverage)], + "coarse_to_middle": coarse_middle, + "middle_to_fine": middle_fine, + } + + +def main(): + parser = argparse.ArgumentParser() + parser.add_argument("--coarse", type=Path, required=True) + parser.add_argument("--middle", type=Path, required=True) + parser.add_argument("--fine", type=Path, required=True) + parser.add_argument("--end-s", type=float, default=4.0) + parser.add_argument("--initial-k", type=float, default=300.0) + parser.add_argument("--probe-k", type=float, action="append", dest="probes_k") + parser.add_argument("--output", type=Path, required=True) + args = parser.parse_args() + if args.end_s <= 0: + parser.error("--end-s must be positive") + result = common_window(args.coarse, args.middle, args.fine, args.end_s, + args.initial_k, args.probes_k or (301.0, 330.0)) + with args.output.open("x") as output: + json.dump(result, output, indent=2) + print(json.dumps({ + "status": result["status"], + "coarse_middle_max_abs_k": result["coarse_to_middle"]["max_abs_k"], + "middle_fine_max_abs_k": result["middle_to_fine"]["max_abs_k"], + })) + + +if __name__ == "__main__": + main() diff --git a/tools/eq3_plot_local_field.py b/tools/eq3_plot_local_field.py new file mode 100644 index 0000000..b08bb08 --- /dev/null +++ b/tools/eq3_plot_local_field.py @@ -0,0 +1,76 @@ +"""Plot an already-derived EQ3 local-field diagnostic without reading raw fields.""" +import argparse +import json +from pathlib import Path + +import matplotlib +matplotlib.use("Agg") +import matplotlib.pyplot as plt +import numpy as np + + +def matrix(rows, value, x_key, y_key): + xs = sorted({row[x_key][0] for row in rows}) + ys = sorted({row[y_key][1] for row in rows}) + xmap = {x: index for index, x in enumerate(xs)} + ymap = {y: index for index, y in enumerate(ys)} + array = np.full((len(ys), len(xs)), np.nan) + for row in rows: + array[ymap[row[y_key][1]], xmap[row[x_key][0]]] = row[value] + return np.asarray(xs) * 1e3, np.asarray(ys) * 1e3, array + + +def main(): + parser = argparse.ArgumentParser() + parser.add_argument("diagnostic", type=Path) + parser.add_argument("output", type=Path) + args = parser.parse_args() + data = json.loads(args.diagnostic.read_text()) + coarse_rows = data["coarse_distribution"] + fine_rows = data["fine_distribution"] + coarse_x, coarse_y, coarse = matrix( + coarse_rows, "coarse_temperature_k", "coarse_center_m", "coarse_center_m") + fine_x, fine_y, fine = matrix( + fine_rows, "temperature_k", "center_m", "center_m") + _, _, difference = matrix( + coarse_rows, "coarse_minus_coarsened_fine_k", + "coarse_center_m", "coarse_center_m") + minimum = min(float(np.nanmin(coarse)), float(np.nanmin(fine))) + maximum = max(float(np.nanmax(coarse)), float(np.nanmax(fine))) + extent = [16, 28, 0, 16] + figure, axes = plt.subplots(1, 3, figsize=(11.8, 3.65), constrained_layout=True) + images = [ + axes[0].imshow(coarse, origin="lower", extent=extent, aspect="equal", + interpolation="nearest", vmin=minimum, vmax=maximum, + cmap="inferno"), + axes[1].imshow(fine, origin="lower", extent=extent, aspect="equal", + interpolation="nearest", vmin=minimum, vmax=maximum, + cmap="inferno"), + ] + delta = max(abs(float(np.nanmin(difference))), abs(float(np.nanmax(difference)))) + images.append(axes[2].imshow(difference, origin="lower", extent=extent, + aspect="equal", interpolation="nearest", + vmin=-delta, vmax=delta, cmap="coolwarm")) + axes[0].set_title("2 mm field") + axes[1].set_title("1 mm field") + axes[2].set_title("2 mm − averaged 1 mm") + for axis in axes: + axis.set_xlabel("x (mm)") + axis.set_ylabel("y (mm)") + axis.plot([16, 28, 28, 16, 16], [0, 0, 16, 16, 0], color="cyan", lw=1) + axes[0].plot(data["coarse"]["hotspot_center_m"][0] * 1e3, + data["coarse"]["hotspot_center_m"][1] * 1e3, "c+", ms=10, mew=1.5) + axes[1].plot(data["fine"]["hotspot_center_m"][0] * 1e3, + data["fine"]["hotspot_center_m"][1] * 1e3, "c+", ms=10, mew=1.5) + axes[1].annotate("active HBM2\ncorner", xy=(16, 16), xytext=(21, 14), + color="white", fontsize=8, + arrowprops={"arrowstyle": "->", "color": "white", "lw": 0.8}) + figure.colorbar(images[1], ax=axes[:2], label="temperature (K)", shrink=.82) + figure.colorbar(images[2], ax=axes[2], label="difference (K)", shrink=.82) + figure.suptitle("hbm3.base, z=1.205–1.255 mm, t=15 s") + figure.savefig(args.output, dpi=180) + plt.close(figure) + + +if __name__ == "__main__": + main() diff --git a/tools/eq3_q4_all_hbf_profile.py b/tools/eq3_q4_all_hbf_profile.py new file mode 100644 index 0000000..fed9536 --- /dev/null +++ b/tools/eq3_q4_all_hbf_profile.py @@ -0,0 +1,209 @@ +#!/usr/bin/env python3 +"""Build the Q4 eight-HBF package profile from the frozen mixed profile. + +This is a geometry/template conversion only. It does not calibrate power, +start a solver, or give the resulting grid the P2 spatial qualification. +""" + +from __future__ import annotations + +import argparse +import copy +import json +from pathlib import Path +from typing import Any, Mapping + + +PAIRING = ( + ("hbm0", "hbf0", "hbf4"), + ("hbm1", "hbf1", "hbf5"), + ("hbm2", "hbf2", "hbf6"), + ("hbm3", "hbf3", "hbf7"), +) + + +def _by_id(rows: list[dict[str, Any]], what: str) -> dict[str, dict[str, Any]]: + result = {str(row.get("id")): row for row in rows} + if len(result) != len(rows) or "None" in result: + raise ValueError(f"{what} identities must be present and unique") + return result + + +def _replace_prefix(value: Any, source: str, target: str) -> Any: + if isinstance(value, str) and (value == source or value.startswith(source + ".")): + return target + value[len(source):] + return value + + +def convert_profile(source: Mapping[str, Any]) -> dict[str, Any]: + """Return an all-HBF deep copy, rejecting anything outside the frozen shape.""" + result = copy.deepcopy(dict(source)) + placements = list(result.get("placements", [])) + blocks = list(result.get("blocks", [])) + placement_by_id = _by_id(placements, "placement") + + expected = {"gpu", *(x for pair in PAIRING for x in pair[:2])} + if set(placement_by_id) != expected: + raise ValueError( + "source must be the one-GPU, hbm0..3, hbf0..3 mixed candidate; " + f"found {sorted(placement_by_id)}" + ) + + block_ids = {str(block.get("id")) for block in blocks} + if len(block_ids) != len(blocks): + raise ValueError("block identities must be present and unique") + + new_placements: list[dict[str, Any]] = [copy.deepcopy(placement_by_id["gpu"])] + new_placements.extend(copy.deepcopy(placement_by_id[f"hbf{i}"]) for i in range(4)) + new_blocks = [ + copy.deepcopy(block) + for block in blocks + if not any(str(block["id"]).startswith(f"hbm{i}.") for i in range(4)) + ] + + for old_hbm, template_hbf, new_hbf in PAIRING: + target = placement_by_id[old_hbm] + template = placement_by_id[template_hbf] + if target.get("footprint_um") != template.get("footprint_um"): + raise ValueError( + f"{old_hbm} and {template_hbf} footprint/orientation differ; rotation is unsupported" + ) + if target.get("array_die_count") != 12 or template.get("array_die_count") != 16: + raise ValueError(f"unexpected die counts for {old_hbm}/{template_hbf}") + translated = copy.deepcopy(template) + translated.update({ + "id": new_hbf, + "xy_um": copy.deepcopy(target["xy_um"]), + "array_die_count": 16, + }) + new_placements.append(translated) + + template_blocks = [ + block for block in blocks if str(block["id"]).startswith(template_hbf + ".") + ] + if len(template_blocks) != 35: + raise ValueError(f"{template_hbf} must contain the complete 35-block 16-die template") + die_indices = sorted( + int(block["die_index"]) + for block in template_blocks + if str(block["id"]).startswith(template_hbf + ".die") + ) + if die_indices != list(range(16)): + raise ValueError(f"{template_hbf} does not contain exactly die0..die15") + + delta = [target["xy_um"][i] - template["xy_um"][i] for i in range(2)] + for original in template_blocks: + cloned = copy.deepcopy(original) + cloned["id"] = _replace_prefix(cloned["id"], template_hbf, new_hbf) + cloned["device"] = "HBF" + if "parent_device_id" in cloned: + cloned["parent_device_id"] = _replace_prefix( + cloned["parent_device_id"], template_hbf, new_hbf + ) + if cloned.get("power_group") is not None: + cloned["power_group"] = _replace_prefix( + cloned["power_group"], template_hbf, new_hbf + ) + cloned["xyz_um"][0] += delta[0] + cloned["xyz_um"][1] += delta[1] + new_blocks.append(cloned) + + result["profile_id"] = "eq3-all-hbf-direct-8-external-gddr-conditional-v1" + result["status"] = "ENGINEERING_MODEL; STATIC_CONVERSION_VALIDATION_REQUIRED" + result["purpose"] = ( + "Q4 complete-package thermal geometry: eight HBF stacks and one external " + "physical GDDR identity outside the package thermal domain" + ) + result["approval_scope"] = ( + "USER_CONFIRMED_EQ3_ISOLATED_MAINTENANCE_CAMPAIGN_V1; geometry/template " + "conversion only; no P2 spatial qualification inherited" + ) + result["placements"] = new_placements + result["blocks"] = new_blocks + + selection = result.setdefault("device_selection", {}) + selection.setdefault("hbm", {})["count"] = 0 + selection.setdefault("hbf", {})["count"] = 8 + selection["gddr"] = { + "profile_id": "external_physical_gddr_identity_v1", + "physical_type": "GDDR", + "package_geometry_modeled": False, + "package_temperature": "UNAVAILABLE", + "service_energy_scope": "SYSTEM_ONLY_NOT_PACKAGE_HEAT", + } + + hbf_ids = [f"hbf{i}" for i in range(8)] + result["data_topology"] = { + "kind": "all_hbf_direct", + "stack_count": 8, + "hbm_count": 0, + "hbf_count": 8, + "coordinates_do_not_change_on_graph_rewiring": True, + "gpu_links": [["gpu", identity] for identity in hbf_ids] + + [["gpu", "external_fast_memory"]], + "external_fast_memory_profile": copy.deepcopy(selection["gddr"]), + "hbf_gpu_facing_phy": result.get("data_topology", {}).get( + "hbf_gpu_facing_phy", "UNKNOWN_BLOCKING for service/area claims" + ), + "link_and_nand_arbitration": "PROVIDED_BY_ISOLATED_EXPERIMENT_FABRIC_NOT_THERMAL_MODEL", + } + result.setdefault("non_claims", []).append( + "Q4 geometry does not inherit P2 grid convergence or external GDDR package temperature" + ) + result["conversion_provenance"] = { + "kind": "USER_CONFIRMED_TEMPLATE_TRANSLATION", + "pairs": [ + {"removed_slot": old, "template": template, "new_stack": new} + for old, template, new in PAIRING + ], + "preserved": ["package", "materials", "boundaries", "background", "gpu"], + } + return result + + +def fixture_power(profile: Mapping[str, Any]) -> dict[str, Any]: + """One 20 ms all-source mapping fixture; values are not research workload data.""" + powered = sorted( + str(block["id"]) + for block in profile["blocks"] + if block.get("powered", block.get("power_group") is not None) + ) + values = {identity: 1.0 for identity in powered} + return { + "schema_version": "eq3-q4-engineering-power-fixture-v1", + "status": "ENGINEERING_FIXTURE", + "evidence_kind": "SCENARIO_ASSUMPTION", + "claim_scope": "component mapping and energy conservation only", + "mode": "per_component", + "power_unit": "W", + "initial_k": 300.0, + "traces": { + "mapping_20ms": { + "duration_s": 0.02, + "intervals": [{"start_s": 0.0, "end_s": 0.02, "power_w": values}], + } + }, + } + + +def _write_new(path: Path, value: Mapping[str, Any]) -> None: + path.parent.mkdir(parents=True, exist_ok=True) + with path.open("x", encoding="utf-8") as stream: + json.dump(value, stream, indent=2, sort_keys=True, allow_nan=False) + stream.write("\n") + + +def main() -> None: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--input-profile", type=Path, required=True) + parser.add_argument("--output-profile", type=Path, required=True) + parser.add_argument("--output-fixture-power", type=Path, required=True) + args = parser.parse_args() + source = json.loads(args.input_profile.read_text(encoding="utf-8")) + converted = convert_profile(source) + _write_new(args.output_profile, converted) + _write_new(args.output_fixture_power, fixture_power(converted)) + + +if __name__ == "__main__": + main() diff --git a/tools/eq3_sensor_time_canonicalize.py b/tools/eq3_sensor_time_canonicalize.py new file mode 100644 index 0000000..1267129 --- /dev/null +++ b/tools/eq3_sensor_time_canonicalize.py @@ -0,0 +1,85 @@ +"""Canonicalize only a sensor CSV's floating timestamp spelling on a fixed grid.""" +import argparse +import csv +import hashlib +import json +import math +from decimal import Decimal +from pathlib import Path + + +def digest(path): + return hashlib.sha256(Path(path).read_bytes()).hexdigest() + + +def row_digest(rows): + value = hashlib.sha256() + for row in rows: + value.update(json.dumps(row, ensure_ascii=False, separators=(",", ":")).encode()) + value.update(b"\n") + return value.hexdigest() + + +def canonicalize(source_path, output_path, quantum_s, duration_s, tolerance_s): + expected_count = round(duration_s / quantum_s) + if not math.isclose(expected_count * quantum_s, duration_s, abs_tol=tolerance_s): + raise ValueError("duration is not an integer number of timestamp quanta") + counts = {} + max_adjustment = 0.0 + non_time = [] + decimal_quantum = Decimal(str(quantum_s)) + with source_path.open(newline="") as source, output_path.open("x", newline="") as output: + reader = csv.DictReader(source) + if not reader.fieldnames or "time_s" not in reader.fieldnames or "sensor_id" not in reader.fieldnames: + raise ValueError("CSV must contain time_s and sensor_id") + writer = csv.DictWriter(output, fieldnames=reader.fieldnames) + writer.writeheader() + for row in reader: + sensor = row["sensor_id"] + index = counts.get(sensor, 0) + 1 + if index > expected_count: + raise ValueError(f"too many timestamps for {sensor}") + actual = float(row["time_s"]) + expected_decimal = index * decimal_quantum + expected = float(expected_decimal) + adjustment = abs(actual - expected) + if not math.isfinite(actual) or adjustment > tolerance_s: + raise ValueError(f"timestamp is off the declared grid for {sensor} at row {index}") + max_adjustment = max(max_adjustment, adjustment) + counts[sensor] = index + non_time.append([row[name] for name in reader.fieldnames if name != "time_s"]) + row["time_s"] = format(expected_decimal, "f") + writer.writerow(row) + if not counts or set(counts.values()) != {expected_count}: + raise ValueError("sensor timestamp coverage is incomplete") + return {"sensor_count": len(counts), "timestamps_per_sensor": expected_count, + "max_abs_time_adjustment_s": max_adjustment, + "non_time_rows_sha256": row_digest(non_time)} + + +def main(): + parser = argparse.ArgumentParser() + parser.add_argument("--input", type=Path, required=True) + parser.add_argument("--output", type=Path, required=True) + parser.add_argument("--receipt", type=Path, required=True) + parser.add_argument("--quantum-s", type=float, required=True) + parser.add_argument("--duration-s", type=float, required=True) + parser.add_argument("--tolerance-s", type=float, default=1e-12) + args = parser.parse_args() + if args.quantum_s <= 0 or args.duration_s <= 0 or args.tolerance_s < 0: + raise ValueError("quantum/duration must be positive and tolerance nonnegative") + result = canonicalize(args.input, args.output, args.quantum_s, + args.duration_s, args.tolerance_s) + result.update(schema_version="eq3-sensor-time-canonicalization-v1", + status="LOSSLESS_NON_TIME_FIELDS_TIMESTAMP_SPELLING_ONLY", + input_sha256=digest(args.input), output_sha256=digest(args.output), + quantum_s=args.quantum_s, duration_s=args.duration_s, + tolerance_s=args.tolerance_s) + with args.receipt.open("x") as output: + json.dump(result, output, indent=2) + output.write("\n") + print(json.dumps(result, separators=(",", ":"))) + + +if __name__ == "__main__": + main() diff --git a/tools/eq3_spatial_diagnostic.py b/tools/eq3_spatial_diagnostic.py new file mode 100644 index 0000000..8169ee8 --- /dev/null +++ b/tools/eq3_spatial_diagnostic.py @@ -0,0 +1,212 @@ +"""Windowed adjacent-grid diagnostics for completed EQ3 sensor CSVs. + +This is a read-only postprocessor. It deliberately does not estimate a +Richardson order: one global maximum, potentially from different sensors and +times on each grid, is not a valid convergence observable. +""" +import argparse +import csv +import json +import math +from pathlib import Path + +from eq3_acceptance_v2 import clip, read_series, validate_method, with_initial + + +GRID_INPUTS = (("R01", "4mm"), ("R02", "2mm"), ("R03", "1mm")) + + +def _cell_rows(path, series): + """Read only provenance fields omitted by acceptance-v2's shared reader.""" + cells = {} + with path.open(newline="") as source: + reader = csv.DictReader(source) + required = {"time_s", "sensor_id", "temperature_k"} + if not reader.fieldnames or not required.issubset(reader.fieldnames): + raise ValueError("sensor CSV must contain time_s,sensor_id,temperature_k in K") + for row in reader: + key = (row["sensor_id"], float(row["time_s"])) + if key in cells: + raise ValueError(f"duplicate raw observation for {key[0]} at {key[1]}") + cell = row.get("hotspot_cell_id") or None + cells[key] = {"temperature_k": float(row["temperature_k"]), + "hotspot_cell_id": cell} + expected = {(sensor, time_s): temperature + for sensor, values in series.items() + for time_s, temperature in values} + if set(cells) != set(expected): + raise ValueError("raw provenance rows differ from parsed sensor coverage") + for key, temperature in expected.items(): + if cells[key]["temperature_k"] != temperature: + raise ValueError("raw provenance temperature differs from parsed series") + return cells + + +def _sensor_group(sensor): + leaf = sensor.rsplit(":", 1)[-1] + if leaf == "mean": + return "mean" + if leaf == "hotspot" or "hotspot" in leaf: + return "hotspot" + return "other" + + +def _time_weighted_mae(reference, candidate): + signed = [right[1] - left[1] for left, right in zip(reference, candidate)] + areas = [] + for index in range(len(signed) - 1): + duration = reference[index + 1][0] - reference[index][0] + left, right = signed[index], signed[index + 1] + if left * right >= 0: + areas.append(duration * (abs(left) + abs(right)) * 0.5) + else: + magnitude = abs(left) + abs(right) + areas.append(duration * (left * left + right * right) / + (2 * magnitude)) + return math.fsum(areas) / (reference[-1][0] - reference[0][0]) + + +def _sensor_score(sensor, window, reference, candidate, ref_cells, cand_cells): + start, end = window["start_s"], window["end_s"] + ref_window = clip(reference, start, end) + cand_window = clip(candidate, start, end) + if [item[0] for item in ref_window] != [item[0] for item in cand_window]: + raise ValueError(f"window timestamp coverage differs for {sensor}") + + # Keep the maximum tied to an actual CSV observation so that cell identity + # and both temperatures remain exact rather than inferred at a boundary. + observed = [] + for time_s, reference_k in reference: + if start <= time_s <= end and (sensor, time_s) in ref_cells: + candidate_k = cand_cells[(sensor, time_s)]["temperature_k"] + observed.append((abs(candidate_k - reference_k), time_s, + reference_k, candidate_k)) + if not observed: + raise ValueError(f"window {window['id']} contains no raw observation for {sensor}") + error, time_s, reference_k, candidate_k = max(observed, key=lambda row: row[0]) + point = { + "sensor_id": sensor, + "time_s": time_s, + "reference_temperature_k": reference_k, + "candidate_temperature_k": candidate_k, + "abs_error_k": error, + "reference_hotspot_cell_id": ref_cells[(sensor, time_s)]["hotspot_cell_id"], + "candidate_hotspot_cell_id": cand_cells[(sensor, time_s)]["hotspot_cell_id"], + "observation_semantics": "exact_registered_csv_row", + } + return {"sensor_id": sensor, "group": _sensor_group(sensor), + "time_weighted_mae_k": _time_weighted_mae(ref_window, cand_window), + "max_registered_abs_error_k": error, + "worst_registered_point": point} + + +def _group_scores(sensor_scores): + output = {} + for group in ("mean", "hotspot", "other"): + members = [score for score in sensor_scores if score["group"] == group] + if not members: + output[group] = {"sensor_count": 0, "status": "NOT_APPLICABLE"} + continue + worst = max(members, key=lambda item: item["max_registered_abs_error_k"]) + output[group] = { + "sensor_count": len(members), + "mean_sensor_time_weighted_mae_k": ( + math.fsum(item["time_weighted_mae_k"] for item in members) / + len(members)), + "max_registered_abs_error_k": worst["max_registered_abs_error_k"], + "worst_registered_point": worst["worst_registered_point"], + } + return output + + +def analyze(r01_path, r02_path, r03_path, method): + paths = (Path(r01_path), Path(r02_path), Path(r03_path)) + + # Validate every input before constructing either adjacent-grid result. + # This is the key guard against reporting a partial, still-running R03. + series = [read_series(path) for path in paths] + sensors = set(series[0]) + checked = validate_method(method, sensors) + for run_id, run_series in zip(("R01", "R02", "R03"), series): + if set(run_series) != sensors: + raise ValueError(f"{run_id} sensor coverage differs") + for sensor in sorted(sensors): + expected_times = [item[0] for item in series[0][sensor]] + for run_id, run_series in zip(("R02", "R03"), series[1:]): + if [item[0] for item in run_series[sensor]] != expected_times: + raise ValueError(f"{run_id} timestamp coverage differs for {sensor}") + initial = checked["initial_by_sensor"][sensor] + for run_series in series: + full = with_initial(run_series[sensor], checked["initial_time_s"], initial) + for window in checked["windows"]: + clip(full, window["start_s"], window["end_s"]) + + cells = [_cell_rows(path, run_series) + for path, run_series in zip(paths, series)] + pairs = [] + for left_index in range(2): + right_index = left_index + 1 + left_id, left_grid = GRID_INPUTS[left_index] + right_id, right_grid = GRID_INPUTS[right_index] + windows = [] + for window in checked["windows"]: + scores = [] + for sensor in sorted(sensors): + initial = checked["initial_by_sensor"][sensor] + reference = with_initial(series[left_index][sensor], + checked["initial_time_s"], initial) + candidate = with_initial(series[right_index][sensor], + checked["initial_time_s"], initial) + scores.append(_sensor_score(sensor, window, reference, candidate, + cells[left_index], cells[right_index])) + windows.append({"window_id": window["id"], "window_kind": window["kind"], + "start_s": window["start_s"], "end_s": window["end_s"], + "groups": _group_scores(scores), "sensors": scores}) + pairs.append({"reference_run_id": left_id, "reference_grid": left_grid, + "candidate_run_id": right_id, "candidate_grid": right_grid, + "windows": windows}) + + return { + "schema_version": "eq3-spatial-diagnostic-v1", + "analysis_kind": "completed_adjacent_grid_same_window_postprocess", + "input_status": "ALL_THREE_COMPLETE_AND_COVERAGE_MATCHED", + "temperature_unit": "K", + "group_rule": "sensor leaf mean -> mean; leaf containing hotspot -> hotspot; else other", + "maximum_semantics": "maximum over exact registered CSV rows inside each window", + "time_weighted_mae_semantics": "linear interpolation at v2 window boundaries", + "inputs": [{"run_id": run_id, "grid": grid, "sensor_csv": str(path)} + for (run_id, grid), path in zip(GRID_INPUTS, paths)], + "richardson_order": "NOT_COMPUTED", + "richardson_reason": ( + "No global-maximum Richardson estimate: sensor and time identity must remain fixed"), + "pairs": pairs, + } + + +def write_result(path, result): + with Path(path).open("x") as output: + json.dump(result, output, indent=2) + output.write("\n") + + +def main(): + parser = argparse.ArgumentParser() + parser.add_argument("--r01", type=Path, required=True, + help="completed R01/4mm sensors.csv") + parser.add_argument("--r02", type=Path, required=True, + help="completed R02/2mm sensors.csv") + parser.add_argument("--r03", type=Path, required=True, + help="completed R03/1mm sensors.csv") + parser.add_argument("--method", type=Path, required=True, + help="EQ3 acceptance-v2 method JSON") + parser.add_argument("--output", type=Path, required=True) + args = parser.parse_args() + method = json.loads(args.method.read_text()) + result = analyze(args.r01, args.r02, args.r03, method) + write_result(args.output, result) + print(json.dumps({"status": result["input_status"], + "adjacent_pairs": len(result["pairs"])})) + + +if __name__ == "__main__": + main() diff --git a/tools/test_eq3_acceptance_v2.py b/tools/test_eq3_acceptance_v2.py new file mode 100644 index 0000000..d5a48da --- /dev/null +++ b/tools/test_eq3_acceptance_v2.py @@ -0,0 +1,177 @@ +import csv +import json +import tempfile +import unittest +from pathlib import Path + +from eq3_acceptance_v2 import analyze + + +SENSORS = ("mean", "grid_hotspot", "control") + + +def method(**updates): + value = { + "schema_version": "eq3-acceptance-v2-method-v1", + "temperature_unit": "K", + "sensor_ids": list(SENSORS), + "initial_time_s": 0.0, + "initial_temperature_k": 300.0, + "windows": [ + {"id": "full", "kind": "full", "start_s": 0.0, "end_s": 4.0}, + {"id": "excited", "kind": "excitation", "start_s": 0.0, "end_s": 2.0}, + {"id": "cooling", "kind": "cooling", "start_s": 2.0, "end_s": 4.0}, + ], + "hotspot_sensor_ids": ["grid_hotspot"], + "control_sensor_ids": ["control"], + "crossing_thresholds_k": [330.0], + "crossing_min_time_s": 0.2, + "crossing_fraction": 0.05, + } + value.update(updates) + return value + + +def energy(residual=0.0): + return {"total_input_energy_j": 10.0, + "stored_energy_change_j": 4.0, + "boundary_loss_j": 6.0 - residual} + + +def write_csv(path, values, times=(1.0, 2.0, 4.0), sensors=SENSORS): + with path.open("w", newline="") as output: + writer = csv.writer(output) + writer.writerow(["time_s", "sensor_id", "temperature_k"]) + for sensor in sensors: + sequence = values[sensor] + for time_s, temperature in zip(times, sequence): + writer.writerow([time_s, sensor, temperature]) + + +class AcceptanceV2Tests(unittest.TestCase): + def run_analysis(self, reference_values, candidate_values, selected_method=None, + selected_energy=None, reference_times=(1.0, 2.0, 4.0), + candidate_times=None, candidate_sensors=SENSORS): + with tempfile.TemporaryDirectory() as root: + root = Path(root) + reference, candidate = root / "reference.csv", root / "candidate.csv" + write_csv(reference, reference_values, reference_times) + write_csv(candidate, candidate_values, + candidate_times or reference_times, candidate_sensors) + return analyze(reference, candidate, selected_method or method(), + selected_energy or energy()) + + @staticmethod + def constant(value): + return {sensor: [value, value, value] for sensor in SENSORS} + + def score(self, result, sensor="mean", window="excited"): + return next(item for item in result["scores"] + if item["sensor_id"] == sensor and item["window_id"] == window) + + def test_constant_and_low_rise_v2_boundaries(self): + reference = self.constant(300.0) + candidate = self.constant(300.25) + result = self.run_analysis(reference, candidate) + score = self.score(result) + self.assertEqual(score["reference_amplitude_k"], 0) + self.assertAlmostEqual(score["v2_mae_limit_k"], .25) + self.assertAlmostEqual(score["time_weighted_mae_k"], .1875) + self.assertTrue(score["v2_pass"]) + + reference = {sensor: [300.5, 301.0, 301.0] for sensor in SENSORS} + candidate = {sensor: [300.79, 301.29, 301.29] for sensor in SENSORS} + result = self.run_analysis(reference, candidate) + score = self.score(result) + self.assertEqual(score["reference_amplitude_k"], 1.0) + self.assertEqual(score["v2_mae_limit_k"], .3) + self.assertTrue(score["v2_pass"]) + + def test_just_over_constant_limit_fails(self): + reference = self.constant(300.0) + candidate = self.constant(300.34) + result = self.run_analysis(reference, candidate) + self.assertFalse(self.score(result)["v2_pass"]) + self.assertEqual(result["status"], "NUMERICAL_FAIL") + + def test_nonuniform_time_weighting_and_legacy_arithmetic_are_distinct(self): + reference = self.constant(300.0) + candidate = {sensor: [300.0, 302.0, 302.0] for sensor in SENSORS} + result = self.run_analysis(reference, candidate) + score = self.score(result, window="full") + self.assertAlmostEqual(score["time_weighted_mae_k"], 1.25) + legacy = next(item for item in result["legacy_v1"]["sensors"] + if item["sensor_id"] == "mean") + self.assertAlmostEqual(legacy["mae_k"], 4.0 / 3.0) + self.assertEqual(legacy["mean_semantics"], + "sample_arithmetic_mean_without_synthetic_initial") + + def test_signed_error_zero_crossing_is_integrated_exactly(self): + reference = self.constant(300.0) + candidate = {sensor: [301.0, 299.0, 300.0] for sensor in SENSORS} + result = self.run_analysis(reference, candidate) + score = self.score(result, window="full") + self.assertAlmostEqual(score["time_weighted_mae_k"], .5) + + def test_excitation_window_cannot_be_diluted_by_cooling(self): + times = (.5, 1.0, 2.0, 4.0) + reference = {sensor: [300.0] * 4 for sensor in SENSORS} + candidate = {sensor: [301.2, 300.0, 300.0, 300.0] for sensor in SENSORS} + result = self.run_analysis(reference, candidate, reference_times=times) + full = self.score(result, window="full") + excited = self.score(result, window="excited") + self.assertLessEqual(full["time_weighted_mae_k"], full["v2_mae_limit_k"]) + self.assertGreater(excited["time_weighted_mae_k"], excited["v2_mae_limit_k"]) + + def test_hotspot_and_control_keep_two_kelvin_maximum(self): + times = (.01, .02, 2.0, 4.0) + reference = {sensor: [300.0] * 4 for sensor in SENSORS} + candidate = {sensor: [300.0] * 4 for sensor in SENSORS} + candidate["grid_hotspot"] = [302.1, 300.0, 300.0, 300.0] + candidate["control"] = [302.1, 300.0, 300.0, 300.0] + result = self.run_analysis(reference, candidate, reference_times=times) + self.assertLess(self.score(result, "grid_hotspot")["time_weighted_mae_k"], .25) + self.assertFalse(self.score(result, "grid_hotspot")["v2_pass"]) + self.assertFalse(self.score(result, "control")["v2_pass"]) + + def test_energy_limit_is_independent(self): + result = self.run_analysis(self.constant(300), self.constant(300), + selected_energy=energy(.02)) + self.assertFalse(result["energy"]["pass"]) + self.assertEqual(result["status"], "NUMERICAL_FAIL") + + def test_threshold_ambiguity_is_reported_without_forcing_numeric_fail(self): + values = {sensor: [300.9, 301.1, 302.0] for sensor in SENSORS} + selected = method(crossing_thresholds_k=[301.0]) + result = self.run_analysis(values, values, selected_method=selected) + self.assertEqual(result["status"], "PASS_WITH_THRESHOLD_AMBIGUITY") + crossing = self.score(result)["crossings"][0] + self.assertEqual(crossing["status"], "THRESHOLD_AMBIGUOUS") + self.assertEqual(crossing["legacy_v1_status"], "PASS") + + def test_missing_frame_sensor_and_unit_mutation_are_rejected(self): + values = self.constant(300) + with self.assertRaisesRegex(ValueError, "timestamp coverage differs"): + self.run_analysis(values, values, candidate_times=(1.0, 2.1, 4.0)) + with self.assertRaisesRegex(ValueError, "sensor coverage differs"): + self.run_analysis(values, values, + candidate_sensors=("mean", "grid_hotspot")) + with self.assertRaisesRegex(ValueError, "preregistered sensor_ids"): + self.run_analysis(values, values, + selected_method=method(sensor_ids=["mean", "grid_hotspot"])) + with self.assertRaisesRegex(ValueError, "temperature_unit must be K"): + self.run_analysis(values, values, + selected_method=method(temperature_unit="degC")) + + def test_all_three_window_kinds_are_required(self): + selected = method(windows=[ + {"id": "full", "kind": "full", "start_s": 0, "end_s": 4}, + {"id": "excited", "kind": "excitation", "start_s": 0, "end_s": 2}, + ]) + with self.assertRaisesRegex(ValueError, "must be preregistered"): + self.run_analysis(self.constant(300), self.constant(300), + selected_method=selected) + + +if __name__ == "__main__": + unittest.main() diff --git a/tools/test_eq3_acceptance_v3.py b/tools/test_eq3_acceptance_v3.py new file mode 100644 index 0000000..e4ca99c --- /dev/null +++ b/tools/test_eq3_acceptance_v3.py @@ -0,0 +1,124 @@ +import csv +import tempfile +import unittest +from pathlib import Path + +from eq3_acceptance_v3 import analyze + + +def method(end=4.0, windows=None, threshold=301.0, minimum=.2): + return {"schema_version":"eq3-acceptance-v3-method-v1", "temperature_unit":"K", + "time_unit":"ns", "sensor_ids":["s"], "initial_time_s":0, + "initial_temperature_k":300.0, + "windows":windows or [ + {"id":"full","kind":"full","start_s":0,"end_s":end}, + {"id":"heat","kind":"excitation","start_s":0,"end_s":end/2}, + {"id":"cool","kind":"cooling","start_s":end/2,"end_s":end}], + "hotspot_sensor_ids":["s"], "control_sensor_ids":[], + "crossing_thresholds_k":[threshold], "crossing_min_time_s":minimum, + "crossing_fraction":.05, "quantization_k":.001, + "quantization_mode":"nearest_rounding"} + + +ENERGY={"total_input_energy_j":10,"stored_energy_change_j":4,"boundary_loss_j":6} + + +class AcceptanceV3Tests(unittest.TestCase): + def run_case(self, ref, cand, selected=None): + with tempfile.TemporaryDirectory() as root: + root=Path(root); rp=root/'r.csv'; cp=root/'c.csv' + for path,rows in ((rp,ref),(cp,cand)): + with path.open('w',newline='') as out: + writer=csv.writer(out);writer.writerow(['time_s','sensor_id','temperature_k']) + for t,v in rows:writer.writerow([t,'s',v]) + return analyze(rp,cp,selected or method(end=max(ref[-1][0],cand[-1][0])),ENERGY) + + def crossing(self,result): + return result['crossing_scores'][0] + + def test_same_window_up_crossing_passes(self): + rows=[(1,300.5),(2,301.5),(4,302)] + result=self.run_case(rows,rows) + self.assertEqual(self.crossing(result)['status'],'PASS') + self.assertEqual(self.crossing(result)['matches'][0]['direction'],'up') + + def test_boundary_event_matches_globally_then_is_indeterminate_once(self): + windows=[{"id":"full","kind":"full","start_s":0,"end_s":40}, + {"id":"heat","kind":"excitation","start_s":0,"end_s":39}, + {"id":"cool","kind":"cooling","start_s":39,"end_s":40}] + ref=[(38.9,300.9),(39.0,301.0),(39.1,301.1),(40,301.2)] + cand=[(38.9,300.9),(39.0,300.99),(39.1,301.1),(40,301.2)] + result=self.run_case(ref,cand,method(40,windows)) + crossing=self.crossing(result) + self.assertEqual(crossing['status'],'INDETERMINATE_QUANTIZATION') + self.assertEqual(len(crossing['matches']),1) + assignment=crossing['matches'][0]['window_assignment'] + self.assertEqual(assignment['status'],'INDETERMINATE_WINDOW_ASSIGNMENT') + self.assertIn('full',assignment['certain_window_ids']) + + def test_threshold_plateau_return_is_not_pass(self): + ref=[(1,300.9),(2,301.0),(3,300.9),(4,300.8)] + result=self.run_case(ref,ref) + self.assertEqual(self.crossing(result)['status'],'INDETERMINATE_QUANTIZATION') + self.assertTrue(self.crossing(result)['reference_indeterminate']) + + def test_exact_equal_between_opposite_states_is_interval_event(self): + rows=[(1,300.9),(2,301.0),(3,301.1),(5,301.2)] + windows=[{"id":"full","kind":"full","start_s":0,"end_s":5}, + {"id":"heat","kind":"excitation","start_s":0,"end_s":4}, + {"id":"cool","kind":"cooling","start_s":4,"end_s":5}] + crossing=self.crossing(self.run_case(rows,rows,method(5,windows))) + self.assertEqual(crossing['status'],'PASS') + self.assertTrue(crossing['reference_events'][0]['quantization_bridged']) + self.assertLess(crossing['reference_events'][0]['time_lo_ns'],2_000_000_000) + self.assertGreater(crossing['reference_events'][0]['time_hi_ns'],2_000_000_000) + + def test_ambiguous_excursion_cannot_be_called_definite_missing(self): + ref=[(1,300.0),(2,302.0),(4,302.0)] + cand=[(1,300.0),(2,301.0),(4,300.0)] + crossing=self.crossing(self.run_case(ref,cand)) + self.assertEqual(crossing['status'],'INDETERMINATE_QUANTIZATION') + self.assertIsNone(crossing['failure_reason']) + + def test_no_crossing_is_not_applicable(self): + rows=[(1,300),(2,300.2),(4,300.3)] + self.assertEqual(self.crossing(self.run_case(rows,rows))['status'],'NOT_APPLICABLE') + + def test_one_sided_definite_missing_crossing_fails(self): + ref=[(1,300),(2,302),(4,302)] + cand=[(1,300),(2,300.2),(4,300.3)] + crossing=self.crossing(self.run_case(ref,cand)) + self.assertEqual(crossing['status'],'FAIL') + self.assertEqual(crossing['failure_reason'],'DEFINITE_MISSING_OR_EXTRA_CROSSING') + + def test_multiple_oscillations_keep_direction_and_order(self): + rows=[(.5,300),(1,302),(1.5,300),(2,302),(4,302)] + crossing=self.crossing(self.run_case(rows,rows)) + self.assertEqual([x['direction'] for x in crossing['reference_events']],['up','down','up']) + self.assertEqual(crossing['status'],'PASS') + + def test_different_sampling_and_integer_time_normalization(self): + ref=[(.3,300.0),(1.0,302.0),(2.0,302.0),(4.0,302.0)] + cand=[(.30000000000000004,300.0),(.5,300.5),(1.0,302.0),(3.0,302.0),(4.0,302.0)] + result=self.run_case(ref,cand) + self.assertEqual(self.crossing(result)['status'],'PASS') + self.assertAlmostEqual(result['temperature_scores'][0]['time_weighted_mae_k'],.00625) + + def test_cooling_direction_and_timeout_failure(self): + selected=method(threshold=301,minimum=.1) + ref=[(1,302),(2,300),(4,300)] + cand=[(1,302),(3,302),(4,300)] + crossing=self.crossing(self.run_case(ref,cand,selected)) + self.assertEqual(crossing['reference_events'][-1]['direction'],'down') + self.assertEqual(crossing['status'],'FAIL') + self.assertEqual(crossing['failure_reason'],'CROSSING_TIMEOUT') + + def test_temperature_and_energy_limits_are_unchanged(self): + rows=[(1,300),(2,300),(4,300)] + result=self.run_case(rows,[(1,303),(2,303),(4,303)]) + self.assertEqual(result['status'],'NUMERICAL_FAIL') + self.assertEqual(result['unchanged_limits']['energy_relative_limit'],.001) + self.assertEqual(result['unchanged_limits']['hotspot_control_max_k'],2.0) + + +if __name__=='__main__':unittest.main() diff --git a/tools/test_eq3_all_source_cap.py b/tools/test_eq3_all_source_cap.py new file mode 100644 index 0000000..c5e33bc --- /dev/null +++ b/tools/test_eq3_all_source_cap.py @@ -0,0 +1,57 @@ +import json +import unittest +from pathlib import Path + +from eq3_all_source_cap import build +from eq3_layered_ir import normalize + + +ROOT = Path(__file__).resolve().parents[1] + + +class AllSourceCapTests(unittest.TestCase): + @classmethod + def setUpClass(cls): + cls.profile = json.loads( + (ROOT / "configs/eq3_thermal/research/candidate_profile.json").read_text()) + cls.power = json.loads( + (ROOT / "configs/eq3_thermal/research/calibration_power.json").read_text()) + + def synthetic_grid(self): + inline = {"id": "cap", "duration_s": self.power["slot_s"], + "slots_W": [dict(zip(self.power["group_order"], + self.power["caps_W"], strict=True))]} + ir = normalize(self.profile, self.power, inline) + cells = [] + component_cells = {} + for component in ir["components"]: + if not component["powered"]: + continue + component_cells[component["id"]] = [len(cells)] + cells.append({"id": f"n{len(cells)}", "component": component["id"], + "volume_m3": component["volume_m3"]}) + return {"cells": cells, "component_cells": component_cells} + + def test_all_seventeen_public_caps_are_mapped_and_conserved(self): + events, receipt = build(self.profile, self.power, self.synthetic_grid()) + self.assertEqual(receipt["status"], "CAP_INPUT_GENERATED_NOT_SOLVED") + self.assertFalse(receipt["blind_trajectory_read"]) + self.assertEqual(len(receipt["caps_w"]), 17) + self.assertAlmostEqual(receipt["total_cap_w"], 840.0) + self.assertAlmostEqual(sum(receipt["node_cap_w"].values()), 840.0) + self.assertEqual(len(events.splitlines()), receipt["event_count"] + 1) + self.assertEqual(set(receipt["group_members"]), set(self.power["group_order"])) + + def test_cap_identity_or_grid_coverage_cannot_silently_fallback(self): + bad = dict(self.power) + bad["group_order"] = self.power["group_order"][:-1] + with self.assertRaisesRegex(ValueError, "17 unique"): + build(self.profile, bad, self.synthetic_grid()) + grid = self.synthetic_grid() + grid["component_cells"].pop(next(iter(grid["component_cells"]))) + with self.assertRaisesRegex(ValueError, "lacks powered component"): + build(self.profile, self.power, grid) + + +if __name__ == "__main__": + unittest.main() diff --git a/tools/test_eq3_basic_fabric.py b/tools/test_eq3_basic_fabric.py new file mode 100644 index 0000000..e9f8f4d --- /dev/null +++ b/tools/test_eq3_basic_fabric.py @@ -0,0 +1,183 @@ +import unittest + +from eq3_basic_fabric import BasicFabric + + +def stage(latency=10, bandwidth=1_000_000_000): + return {"latency_ns": latency, "bandwidth_bytes_per_s": bandwidth} + + +def fixture(): + return { + "evidence": "SCENARIO_ASSUMPTION", + "hbf": { + f"hbf{i}": { + "pair": f"hbm{i}", + "bank_count": 2, + "bank_capacity_bytes": 1024, + "fill": stage(), + "direct_link": stage(20), + "relay_link": stage(30), + } + for i in range(2) + }, + "hbm": { + f"hbm{i}": { + "bank_count": 2, + "bank_capacity_bytes": 1024, + "gpu_link": stage(40), + } + for i in range(2) + }, + } + + +class BasicFabricTests(unittest.TestCase): + def test_direct_and_relay_exact_timing_and_bytes(self): + direct = BasicFabric(fixture()) + direct.enqueue("d", "hbf0", "direct", 100, 0) + self.assertEqual(direct.next_event_ns(), 0) + direct.advance(230) + self.assertEqual(direct.completions()[0]["completion_ns"], 230) + self.assertEqual(direct.completions()[0]["bytes"], 100) + + relay = BasicFabric(fixture()) + relay.enqueue("r", "hbf0", "relay", 100, 0) + relay.advance(380) + self.assertEqual(relay.completions()[0]["completion_ns"], 380) + starts = [e for e in relay.events() if e["kind"] == "start"] + self.assertEqual([e["stage"] for e in starts], ["HBF_FILL", "HBF_RELAY", "HBM_GPU"]) + self.assertTrue(all(e["bytes"] == 100 for e in starts)) + + def test_two_hbf_banks_allow_direct_relay_overlap(self): + fabric = BasicFabric(fixture()) + fabric.enqueue("direct", "hbf0", "direct", 100, 0) + fabric.enqueue("relay", "hbf0", "relay", 100, 0) + fabric.advance(400) + starts = {(e["request_id"], e["stage"]): e for e in fabric.events() if e["kind"] == "start"} + direct = starts[("direct", "HBF_DIRECT")] + relay = starts[("relay", "HBF_RELAY")] + self.assertLess(relay["start_ns"], direct["end_ns"]) + self.assertNotEqual(direct["hbf_bank"], relay["hbf_bank"]) + + def test_hbm_local_and_relay_share_gpu_link(self): + fabric = BasicFabric(fixture()) + fabric.enqueue("local", "hbm0", "direct", 100, 240) + fabric.enqueue("relay", "hbf0", "relay", 100, 0) + fabric.advance(520) + starts = [e for e in fabric.events() if e["kind"] == "start" and e["stage"] == "HBM_GPU"] + self.assertEqual([(e["request_id"], e["start_ns"], e["end_ns"]) for e in starts], + [("local", 240, 380), ("relay", 380, 520)]) + + def test_pairs_have_no_package_global_lock(self): + fabric = BasicFabric(fixture()) + fabric.enqueue("r0", "hbf0", "relay", 100, 0) + fabric.enqueue("r1", "hbf1", "relay", 100, 0) + fabric.advance(380) + self.assertEqual({row["request_id"]: row["completion_ns"] for row in fabric.completions()}, + {"r0": 380, "r1": 380}) + + def test_bank_backpressure_and_complete_release_admit_order(self): + fabric = BasicFabric(fixture()) + for request_id in ("a", "b", "c"): + fabric.enqueue(request_id, "hbf0", "direct", 100, 0) + fabric.advance(231) + starts = {(e["request_id"], e["stage"]): e for e in fabric.events() if e["kind"] == "start"} + self.assertEqual(starts[("c", "HBF_FILL")]["start_ns"], 230) + self.assertEqual(starts[("b", "HBF_DIRECT")]["start_ns"], 230) + at_230 = [e for e in fabric.events() if (e.get("time_ns") == 230 or e.get("start_ns") == 230)] + self.assertEqual(at_230[0]["kind"], "complete") + + def test_unique_ids_validation_and_snapshot_isolation(self): + fabric = BasicFabric(fixture()) + fabric.enqueue("a", "hbf0", "direct", 100, 0) + with self.assertRaises(ValueError): + fabric.enqueue("a", "hbf0", "direct", 100, 0) + with self.assertRaises(ValueError): + fabric.enqueue("large", "hbf0", "direct", 2048, 0) + with self.assertRaises(ValueError): + fabric.enqueue("bad-route", "hbm0", "relay", 100, 0) + fabric.advance(1) + with self.assertRaises(ValueError): + fabric.enqueue("past", "hbf0", "direct", 100, 0) + facts = fabric.immutable_facts() + facts["config"]["hbf"]["hbf0"]["bank_count"] = 99 + self.assertEqual(fabric.immutable_facts()["config"]["hbf"]["hbf0"]["bank_count"], 2) + + def test_unpaired_hbf_supports_direct_only(self): + config = fixture() + config["hbf"]["hbf0"]["pair"] = None + config["hbf"]["hbf0"]["relay_link"] = None + fabric = BasicFabric(config) + fabric.enqueue("d", "hbf0", "direct", 100, 0) + with self.assertRaises(ValueError): + fabric.enqueue("r", "hbf0", "relay", 100, 0) + + def test_backend_delivered_reservation_is_bounded_and_skips_fill(self): + fabric = BasicFabric(fixture()) + self.assertTrue(fabric.reserve_hbf("direct", "hbf0", "direct", 100, 0)) + self.assertTrue(fabric.reserve_hbf("relay", "hbf0", "relay", 100, 0)) + self.assertFalse(fabric.reserve_hbf("outside", "hbf0", "direct", 100, 0)) + self.assertNotIn("outside", [e.get("request_id") for e in fabric.events()]) + fabric.mark_hbf_ready("direct", 100) + fabric.mark_hbf_ready("relay", 100) + fabric.advance(221) + starts = [e for e in fabric.events() if e["kind"] == "start"] + self.assertFalse(any(e["stage"] == "HBF_FILL" for e in starts)) + self.assertEqual({e["stage"] for e in starts if e["start_ns"] == 100}, {"HBF_DIRECT", "HBF_RELAY"}) + self.assertTrue(fabric.reserve_hbf("outside", "hbf0", "direct", 100, 0)) + + def test_backend_ready_validation_and_same_timestamp_release(self): + fabric = BasicFabric(fixture()) + with self.assertRaises(ValueError): + fabric.reserve_hbf("future", "hbf0", "direct", 100, 1) + self.assertTrue(fabric.reserve_hbf("a", "hbf0", "direct", 100, 0)) + with self.assertRaises(ValueError): + fabric.mark_hbf_ready("missing", 0) + fabric.mark_hbf_ready("a", 100) + with self.assertRaises(ValueError): + fabric.mark_hbf_ready("a", 100) + fabric.advance(220) + events = fabric.events() + ready = next(i for i, e in enumerate(events) if e["kind"] == "backend_ready") + start = next(i for i, e in enumerate(events) if e["kind"] == "start" and e["stage"] == "HBF_DIRECT") + self.assertLess(ready, start) + + def test_hbm_backend_uses_same_bounded_banks_as_relay(self): + fabric = BasicFabric(fixture()) + self.assertTrue(fabric.reserve_source("local0", "hbm0", "direct", 100, 0)) + self.assertTrue(fabric.reserve_source("local1", "hbm0", "direct", 100, 0)) + self.assertFalse(fabric.reserve_source("outside", "hbm0", "direct", 100, 0)) + fabric.mark_source_ready("local0", 100) + fabric.mark_source_ready("local1", 100) + fabric.advance(240) + self.assertEqual([row["request_id"] for row in fabric.completions()], ["local0"]) + self.assertTrue(fabric.reserve_source("outside", "hbm0", "direct", 100, 0)) + + relay = BasicFabric(fixture()) + self.assertTrue(relay.reserve_source("local0", "hbm0", "direct", 100, 0)) + self.assertTrue(relay.reserve_source("local1", "hbm0", "direct", 100, 0)) + relay.mark_source_ready("local0", 0) + relay.mark_source_ready("local1", 0) + relay.enqueue("relay", "hbf0", "relay", 100, 0) + relay.advance(1_000) + relay_start = next(e for e in relay.events() if e["kind"] == "start" and e["stage"] == "HBF_RELAY") + self.assertEqual(relay_start["start_ns"], 140) + + def test_invalid_config_rejected(self): + config = fixture() + config["hbf"]["hbf0"]["bank_count"] = 1 + with self.assertRaises(ValueError): + BasicFabric(config) + config = fixture() + config["hbf"]["hbf1"]["pair"] = "hbm0" + with self.assertRaises(ValueError): + BasicFabric(config) + config = fixture() + config["evidence"] = "CALIBRATED" + with self.assertRaises(ValueError): + BasicFabric(config) + + +if __name__ == "__main__": + unittest.main() diff --git a/tools/test_eq3_basic_hbm.py b/tools/test_eq3_basic_hbm.py new file mode 100644 index 0000000..07a1023 --- /dev/null +++ b/tools/test_eq3_basic_hbm.py @@ -0,0 +1,118 @@ +import unittest + +from eq3_basic_hbm import BasicHbm, scenario_example + + +def config(): + value = scenario_example(("hbm0", "hbm1")) + value["stacks"][0]["media_latency_ns"] = {"read": 10, "write": 20} + value["stacks"][0]["media_bandwidth_Bps"] = {"read": 1_000_000_000, + "write": 2_000_000_000} + value["stacks"][1]["media_latency_ns"] = {"read": 5, "write": 7} + value["stacks"][1]["media_bandwidth_Bps"] = {"read": 1_000_000_000, + "write": 1_000_000_000} + return value + + +class BasicHbmTests(unittest.TestCase): + def test_explicit_latency_plus_ceiling_bandwidth_formula(self): + hbm = BasicHbm(config()) + hbm.arrival({"request_id": "r", "stack_id": "hbm0", "op": "read", + "bytes": 5, "arrival_ns": 0}) + hbm.submit("r") + self.assertEqual(hbm.next_event_ns(), 15) + hbm.advance(15) + facts = hbm.take_facts() + start = next(item for item in facts if item["phase"] == "media_start") + self.assertEqual(start["fixed_media_latency_ns"], 10) + self.assertEqual(start["bandwidth_transfer_ns"], 5) + completion = hbm.take_media_completions()[0] + self.assertEqual(completion["time_ns"], 15) + self.assertEqual(completion["phase"], "MEDIA_DONE") + self.assertTrue(completion["requires_fabric"]) + self.assertFalse(completion["reported_completion"]) + + def test_same_stack_fifo_and_different_stacks_progress_in_parallel(self): + hbm = BasicHbm(config()) + for request in ( + {"request_id": "a", "stack_id": "hbm0", "op": "read", "bytes": 10, + "arrival_ns": 0}, + {"request_id": "b", "stack_id": "hbm0", "op": "write", "bytes": 10, + "arrival_ns": 0}, + {"request_id": "c", "stack_id": "hbm1", "op": "read", "bytes": 10, + "arrival_ns": 0}): + hbm.arrival(request) + hbm.submit(request["request_id"]) + self.assertEqual(hbm.next_event_ns(), 15) + hbm.advance(20) + ready = {item["request_id"]: item["time_ns"] for item in hbm.take_media_completions()} + self.assertEqual(ready, {"c": 15, "a": 20}) + self.assertEqual(hbm.next_event_ns(), 45) + hbm.advance(45) + self.assertEqual(hbm.take_media_completions()[0]["request_id"], "b") + + def test_arrival_submit_media_and_fabric_boundaries_are_distinct(self): + hbm = BasicHbm(config()) + hbm.arrival({"request_id": "r", "stack_id": "hbm0", "op": "write", + "bytes": 1, "arrival_ns": 7}) + self.assertIsNone(hbm.next_event_ns()) + self.assertEqual([item["phase"] for item in hbm.take_facts()], ["arrival"]) + hbm.submit("r", 9) + facts = hbm.take_facts() + self.assertEqual([item["phase"] for item in facts], ["submit", "media_start"]) + self.assertTrue(all(not item["reported_completion"] for item in facts)) + hbm.advance(30) + completion = hbm.take_media_completions()[0] + self.assertEqual(completion["time_ns"], 30) + self.assertTrue(completion["requires_fabric"]) + + def test_unknown_energy_die_plane_are_not_invented(self): + hbm = BasicHbm(config()) + hbm.arrival({"request_id": "r", "stack_id": "hbm0", "op": "read", + "bytes": 1, "arrival_ns": 0}) + hbm.submit("r") + hbm.advance(hbm.next_event_ns()) + completion = hbm.take_media_completions()[0] + self.assertIsNone(completion["media_energy_j"]) + self.assertEqual(completion["media_energy_status"], "UNKNOWN_UNPARAMETERIZED") + self.assertEqual(completion["die"], "UNKNOWN") + self.assertEqual(completion["plane"], "UNKNOWN") + + def test_refresh_is_explicitly_unsupported_and_capabilities_are_limited(self): + hbm = BasicHbm(config()) + self.assertEqual(hbm.refresh("hbm0")["status"], "UNSUPPORTED_CAPABILITY") + capability = hbm.capability() + self.assertFalse(capability["real_dram_backend"]) + self.assertEqual(capability["fabric_arbitration"], "EXTERNAL_REQUIRED") + + def test_invalid_lifecycle_and_address_placement_are_rejected(self): + hbm = BasicHbm(config()) + with self.assertRaisesRegex(ValueError, "arrive before"): + hbm.submit("missing") + with self.assertRaisesRegex(ValueError, "die/plane"): + hbm.arrival({"request_id": "bad", "stack_id": "hbm0", "op": "read", + "bytes": 1, "arrival_ns": 0, "die": 0}) + hbm.arrival({"request_id": "r", "stack_id": "hbm0", "op": "read", + "bytes": 1, "arrival_ns": 0}) + hbm.submit("r") + with self.assertRaisesRegex(ValueError, "exactly once"): + hbm.submit("r") + with self.assertRaisesRegex(ValueError, "allowed range|backward"): + hbm.advance(-1) + + def test_parameter_values_require_evidence_and_unknown_energy_stays_null(self): + accepted = BasicHbm(config()) + self.assertEqual(accepted.capability()["evidence_class"], + "PARAMETRIC_HBM_SCENARIO") + value = config() + del value["stacks"][0]["media_latency_evidence"] + with self.assertRaisesRegex(ValueError, "media_latency_evidence"): + BasicHbm(value) + value = config() + value["stacks"][0]["energy_j_per_byte"]["read"] = 1e-12 + with self.assertRaisesRegex(ValueError, "non-UNKNOWN evidence"): + BasicHbm(value) + + +if __name__ == "__main__": + unittest.main() diff --git a/tools/test_eq3_basic_system.py b/tools/test_eq3_basic_system.py new file mode 100644 index 0000000..6da3f04 --- /dev/null +++ b/tools/test_eq3_basic_system.py @@ -0,0 +1,192 @@ +import heapq +import unittest + +from eq3_basic_fabric import BasicFabric +from eq3_basic_hbm import BasicHbm +from eq3_basic_system import BasicSystem, UnsupportedComposition, engineering_fixture +from eq3_basic_system_actual import derive_mqsim_case + + +def fabric_config(mode): + return engineering_fixture(mode, 64)["fabric"] + + +def hbm_fixture(mode): + value = engineering_fixture(mode, 64)["hbm"] + return None if value is None else BasicHbm(value) + + +class FakeMqsim: + """Existing-client-shaped deterministic backend used only by fixed tests.""" + + def __init__(self, *, aggregate_delay=0, defer_until=None): + self.now = 0 + self.aggregate_delay = aggregate_delay + self.defer_until = defer_until + self.requests = {} + self.completions = {} + self.observations = [] + self.native_observations = [] + self._pending = [] + self.submitted = [] + + def submit(self, request): + request = dict(request) + request_id = request["request_id"] + if request_id in self.requests: + raise ValueError("duplicate fake request") + self.requests[request_id] = request + self.submitted.append(request) + raw = request["issue_ns"] + 50 + heapq.heappush(self._pending, (raw + self.aggregate_delay, request_id, raw)) + for kind in (0, 1): + self.observations.append({ + "kind": kind, "request_id": request_id, + "arrival_ns": request["issue_ns"], "time_ns": request["issue_ns"], + "reported_complete": 0, "bytes": request["bytes"], + "device_outstanding": 1, + }) + + def try_submit(self, request): + if self.defer_until is not None and self.now < self.defer_until: + return {"submitted": False, "disposition": 1, "reason": "fixed gate", + "original_arrival_ns": request["issue_ns"], + "evaluated_ns": self.now, "target_ns": self.defer_until, + "backend_arrival_ns": None, "external_wait_ns": None} + self.submit(request) + return {"submitted": True, "disposition": 0, "reason": "gate disabled", + "original_arrival_ns": request["issue_ns"], + "evaluated_ns": self.now, "target_ns": None, + "backend_arrival_ns": request["issue_ns"], "external_wait_ns": 0} + + def until(self, horizon): + if self._pending and self._pending[0][0] <= horizon: + reported, request_id, raw = heapq.heappop(self._pending) + self.now = reported + request = self.requests[request_id] + self.observations.append({ + "kind": 2, "request_id": request_id, + "arrival_ns": request["issue_ns"], "time_ns": raw, + "reported_complete": reported, "bytes": request["bytes"], + "device_outstanding": 0, + }) + completion = {"request_id": request_id, "reported_complete": reported, "status": 0} + self.completions[request_id] = completion + return completion + self.now = horizon + return None + + def finish(self): + if self._pending or set(self.requests) != set(self.completions): + raise ValueError("unfinished fake MQSim") + total = sum(row["bytes"] for row in self.requests.values()) + return {"status": "FINISHED", "issued": len(self.requests), + "completed": len(self.completions), "issued_bytes": total, + "completed_bytes": total, "pending": 0} + + +def requests_for(mode): + return engineering_fixture(mode, 64)["requests"] + + +class BasicSystemTests(unittest.TestCase): + def build(self, mode, *, mqsim=None): + mqsim = FakeMqsim() if mqsim is None else mqsim + return BasicSystem(mode, mqsim, hbm_fixture(mode), BasicFabric(fabric_config(mode))), mqsim + + def test_fixed_four_topology_chains_conserve_identity_and_resources(self): + for mode in ("all_hbf_direct", "mixed_direct", "relay", "dash"): + with self.subTest(mode=mode): + system, mqsim = self.build(mode) + requests = requests_for(mode) + result = system.run(requests) + self.assertEqual(result["topology_mode"], mode) + self.assertEqual(len(result["completions"]), len(requests)) + self.assertEqual({row["request_id"] for row in result["completions"]}, + {row["request_id"] for row in requests}) + self.assertTrue(all(row["state"] == "COMPLETE" for row in result["completions"])) + self.assertTrue(all(row["external_wait_ns"] >= 0 for row in result["completions"])) + self.assertTrue(all(row["final_completion_ns"] >= row["backend_completion_ns"] + for row in result["completions"])) + state = result["fabric_resource_state"] + self.assertFalse(state["unfinished"]) + self.assertTrue(all(not any(row["banks"]) for row in state["hbf"].values())) + self.assertTrue(all(not any(row["banks"]) for row in state["hbm"].values())) + self.assertTrue(all(row["route"] == "direct" for row in mqsim.submitted)) + expected = {stack: sum(row["stack"] == stack for row in requests) + for stack in result["stack_request_counts"]} + self.assertEqual(result["stack_request_counts"], expected) + self.assertEqual(bool(result["external_devices"]), mode == "all_hbf_direct") + if mode == "all_hbf_direct": + self.assertEqual(result["external_devices"][0]["service"], "UNAVAILABLE") + self.assertEqual(result["external_devices"][0]["temperature"], "UNAVAILABLE") + self.assertEqual(result["limits"]["energy"], + "UNKNOWN_WHERE_UNPARAMETERIZED") + + def test_actual_case_derivation_matches_each_hbf_axis(self): + base = {"name": "base", "capacity_bytes": 128 << 20, "page_bytes": 16384, + "channels": 8, "dies_per_channel": 1, "planes_per_die": 1, + "pages_per_block": 256} + template = {"schema_version": 1, "physical_kind": "HBF", "route": "direct", + "address_layout": "GLOBAL_PAGE_STRIPE_V1", + "plane_allocation_scheme": "CWDP", "page_bytes": 16384, + "channels": 8, "dies_per_channel": 1, + "stacks": [{"id": f"hbf{i}", "declared_dies": 1, "channels": [i]} + for i in range(8)]} + for mode in ("all_hbf_direct", "mixed_direct", "relay", "dash"): + profile, mapping = derive_mqsim_case(base, template, mode) + expected = 8 if mode == "all_hbf_direct" else 4 + self.assertEqual(profile["channels"], expected) + self.assertEqual(len(mapping["stacks"]), expected) + self.assertEqual([row["channels"] for row in mapping["stacks"]], + [[i] for i in range(expected)]) + fixture = engineering_fixture(mode, 64) + self.assertTrue(all(sum(row["stack"] == stack for row in fixture["requests"]) >= 3 + for stack in fixture["fabric"]["hbf"] | fixture["fabric"]["hbm"])) + if mode == "mixed_direct": + self.assertTrue(all(row["pair"] is None and row["relay_link"] is None + for row in fixture["fabric"]["hbf"].values())) + + def test_cascaded_direct_and_non_dash_route_mix_are_rejected(self): + system, _ = self.build("relay") + bad = requests_for("relay")[0] + bad["route"] = "direct" + with self.assertRaisesRegex(ValueError, "no HBF direct"): + system.run([bad]) + system, _ = self.build("mixed_direct") + bad = requests_for("mixed_direct")[0] + bad["route"] = "relay" + with self.assertRaisesRegex(ValueError, "only direct"): + system.run([bad]) + + def test_mqsim_aggregate_bound_is_not_composed_twice(self): + system, _ = self.build("all_hbf_direct", mqsim=FakeMqsim(aggregate_delay=1)) + with self.assertRaisesRegex(UnsupportedComposition, "UNSUPPORTED_COMPOSITION"): + system.run([requests_for("all_hbf_direct")[0]]) + + def test_full_source_banks_leave_third_request_in_external_wait(self): + system, _ = self.build("all_hbf_direct") + rows = [] + for index in range(3): + rows.append({"request_id": f"queued-{index}", "stack": "hbf0", + "stack_local_page": index, "route": "direct", "bytes": 64, + "arrival_ns": 0, "operation": "read"}) + result = system.run(rows) + by_id = {row["request_id"]: row for row in result["completions"]} + self.assertEqual(by_id["queued-0"]["external_wait_ns"], 0) + self.assertEqual(by_id["queued-1"]["external_wait_ns"], 0) + self.assertGreater(by_id["queued-2"]["external_wait_ns"], 0) + self.assertFalse(result["fabric_resource_state"]["unfinished"]) + + def test_gate_defer_uses_horizon_and_submits_reserved_request_once(self): + mqsim = FakeMqsim(defer_until=25) + system, _ = self.build("all_hbf_direct", mqsim=mqsim) + result = system.run([requests_for("all_hbf_direct")[0]]) + completion = result["completions"][0] + self.assertEqual(completion["backend_submit_ns"], 25) + self.assertEqual(completion["external_wait_ns"], 25) + self.assertEqual(len(mqsim.submitted), 1) + + +if __name__ == "__main__": + unittest.main() diff --git a/tools/test_eq3_campaign_compare.py b/tools/test_eq3_campaign_compare.py index 6cbd70a..8f7c878 100644 --- a/tools/test_eq3_campaign_compare.py +++ b/tools/test_eq3_campaign_compare.py @@ -5,6 +5,8 @@ class CompareTests(unittest.TestCase): def test_crossings_include_cooling_and_declared_initial(self): self.assertEqual(crossings([(1,302),(2,300)],301),[('up',.5),('down',1.5)]) + self.assertEqual(crossings([(1,312),(2,310)],311,310), + [('up',.5),('down',1.5)]) def test_normalization_not_absolute_temperature(self): with tempfile.TemporaryDirectory() as tmp: a=Path(tmp)/'a';b=Path(tmp)/'b' @@ -12,6 +14,21 @@ def test_normalization_not_absolute_temperature(self): with path.open('w',newline='') as f: w=csv.writer(f);w.writerow(['time_s','sensor_id','temperature_k']);w.writerows([[.1,'component:a:mean',300+offset],[.2,'component:a:mean',300.1+offset]]) r=compare(a,b,'rc');self.assertEqual(r['status'],'NUMERICAL_FAIL');self.assertAlmostEqual(r['sensors'][0]['normalized_mae'],.2) + self.assertEqual(r['analysis_initial_k'],300) + self.assertEqual(r['analysis_probes_k'],[301,330]) self.assertEqual(compare(a,b,'reference')['status'],'PASS') + def test_explicit_initial_and_probes_are_reported_and_consumed(self): + with tempfile.TemporaryDirectory() as tmp: + a=Path(tmp)/'a';b=Path(tmp)/'b' + for path in (a,b): + with path.open('w',newline='') as f: + w=csv.writer(f);w.writerow(['time_s','sensor_id','temperature_k']) + w.writerows([[.1,'component:a:mean',311],[.2,'component:a:mean',312]]) + r=compare(a,b,'rc',initial_k=310,probes_k=(310.5,311.5)) + self.assertEqual(r['analysis_initial_k'],310) + self.assertEqual(r['analysis_probes_k'],[310.5,311.5]) + self.assertEqual([x['status'] for x in r['sensors'][0]['crossings']], + ['PASS','PASS']) + if __name__=='__main__':unittest.main() diff --git a/tools/test_eq3_campaign_gate.py b/tools/test_eq3_campaign_gate.py index d4b5f22..3685f68 100644 --- a/tools/test_eq3_campaign_gate.py +++ b/tools/test_eq3_campaign_gate.py @@ -30,6 +30,29 @@ def test_reject_unlocked_blind(self): def test_reject_ram16_per_process(self): self.m['resource_budget']['requested']['ram_gib']=16 with self.assertRaises(GateError):self.check() + def enable_per_experiment_resources(self): + self.scope['resource_policy']='PER_EXPERIMENT_USER_CONFIRMED' + self.scope['limits']={'task_ram_gib':24,'process_ram_gib':16,'threads':1,'watchdog_s':3600, + 'point_disk_gib':6,'task_disk_gib':40,'min_free_disk_gib':10,'gpu':0,'cloud':0} + self.c['limits']=self.scope['limits'];self.m['resource_budget']['requested'].update(ram_gib=16,disk_gib=6) + self.write('scope.json',json.dumps(self.scope));self.a['scope_sha256']=sha(self.root/'scope.json');self.write('child.json',json.dumps(self.c)) + def test_explicit_per_experiment_resources_replace_legacy_caps(self): + self.enable_per_experiment_resources();result=self.check() + self.assertEqual(result['resource_policy'],'PER_EXPERIMENT_USER_CONFIRMED') + self.assertEqual(result['resource_limits']['watchdog_s'],3600) + def test_per_experiment_child_must_bind_exact_scope_limits(self): + self.enable_per_experiment_resources();self.c['limits']['watchdog_s']=3599;self.write('child.json',json.dumps(self.c)) + with self.assertRaises(GateError):self.check() + def test_per_experiment_resources_reject_nonfinite(self): + self.enable_per_experiment_resources();self.scope['limits']['watchdog_s']=float('inf');self.c['limits']=self.scope['limits'] + self.write('scope.json',json.dumps(self.scope));self.a['scope_sha256']=sha(self.root/'scope.json');self.write('child.json',json.dumps(self.c)) + with self.assertRaises(GateError):self.check() + def test_resource_values_without_explicit_policy_keep_legacy_caps(self): + self.scope['limits']={'task_ram_gib':24,'process_ram_gib':16,'threads':1,'watchdog_s':3600, + 'point_disk_gib':6,'task_disk_gib':40,'min_free_disk_gib':10,'gpu':0,'cloud':0} + self.c['limits']=self.scope['limits'];self.m['resource_budget']['requested']['ram_gib']=16 + self.write('scope.json',json.dumps(self.scope));self.a['scope_sha256']=sha(self.root/'scope.json');self.write('child.json',json.dumps(self.c)) + with self.assertRaises(GateError):self.check() def test_reject_outside_family(self): self.c['family']='GPU';self.write('child.json',json.dumps(self.c)) with self.assertRaises(GateError):self.check() diff --git a/tools/test_eq3_campaign_rc_runner.py b/tools/test_eq3_campaign_rc_runner.py index 1e7bc5d..fb989c4 100644 --- a/tools/test_eq3_campaign_rc_runner.py +++ b/tools/test_eq3_campaign_rc_runner.py @@ -5,6 +5,7 @@ """ import csv +import hashlib import io import json import os @@ -15,7 +16,6 @@ ROOT = Path(__file__).resolve().parents[1] -WORKSPACE = ROOT.parent MODEL = """HBFSIM_EQ3_THERMAL_MODEL 1 coupling on @@ -41,6 +41,8 @@ def setUpClass(cls): ) cls.sparse = Path(sparse).resolve() cls.dense = Path(dense).resolve() + generated = os.environ.get("EQ3_GENERATED_ROOT") + cls.generated_root = Path(generated).resolve() if generated else None for binary in (cls.sparse, cls.dense): if not binary.is_file() or not os.access(binary, os.X_OK): raise unittest.SkipTest(f"runner is not executable: {binary}") @@ -102,6 +104,26 @@ def test_sparse_matches_existing_dense_equation_on_fixed_fixture(self): ): self.assertAlmostEqual(dense_energy[field], sparse_energy[field], places=9) + def test_two_node_one_step_matches_independent_hand_solution(self): + with tempfile.TemporaryDirectory() as root: + root=Path(root);model,events=root/'model.txt',root/'events.txt' + model.write_text(MODEL) + events.write_text('HBFSIM_EQ3_THERMAL_EVENTS 1\n') + command=self.command(self.sparse,model,events,'--run',end='.01') + command[command.index('--slot-s')+1]='.01' + command[command.index('--sample-s')+1]='.01' + result=self.execute(command,root) + self.assertEqual(result.returncode,0,result.stderr) + values=self.csv_values(result.stdout) + a=2/.01+.5+1;d=3/.01+.2+1 + rhs_a=2/.01*300+1+.5*295 + rhs_b=3/.01*301+.5+.2*296 + determinant=a*d-1 + expected_a=(d*rhs_a+rhs_b)/determinant + expected_b=(rhs_a+a*rhs_b)/determinant + self.assertAlmostEqual(values[(.01,'a')],expected_a,places=11) + self.assertAlmostEqual(values[(.01,'b')],expected_b,places=11) + def test_inspect_does_not_create_solver_or_receipt(self): with tempfile.TemporaryDirectory() as root: root = Path(root) @@ -140,21 +162,198 @@ def test_zero_source_equilibrium_exact_with_nonhardcoded_origin(self): for field in ('stored_energy_change_j','boundary_loss_j','energy_residual_j'): self.assertEqual(receipt[field],0) + def test_steady_envelope_solves_fixed_and_all_source_cap_rhs(self): + with tempfile.TemporaryDirectory() as root: + root = Path(root) + model, events = root / 'model.txt', root / 'events.txt' + model.write_text('HBFSIM_EQ3_THERMAL_MODEL 1\ncoupling on\n' + 'node a hbf capacity_memory s0 0 1 300 1 2 300\n') + events.write_text('HBFSIM_EQ3_THERMAL_EVENTS 1\n' + 'activity 1 cap external_heat external 0 -1 -1 -1 0 1 1 a 4\n') + command = self.command(self.sparse, model, events, + '--steady-envelope', end='1') + command += ['--envelope-limit-k', '302'] + result = self.execute(command, root) + self.assertEqual(result.returncode, 0, result.stderr) + receipt = json.loads(result.stdout) + self.assertEqual(receipt['status'], 'PREDICTED_ENVELOPE') + self.assertEqual(receipt['selected_alpha'], .75) + self.assertEqual(receipt['cap_total_w'], 4) + self.assertAlmostEqual(receipt['candidates'][0]['max_k'], 302.5) + self.assertLess(receipt['fixed_residual_inf'], 1e-12) + self.assertLess(receipt['cap_residual_inf'], 1e-12) + + def test_steady_envelope_rejects_network_without_heat_outlet(self): + with tempfile.TemporaryDirectory() as root: + root = Path(root) + model, events = root / 'model.txt', root / 'events.txt' + model.write_text('HBFSIM_EQ3_THERMAL_MODEL 1\ncoupling on\n' + 'node a hbf capacity_memory s0 0 1 300 0 0 300\n') + events.write_text('HBFSIM_EQ3_THERMAL_EVENTS 1\n' + 'activity 1 cap external_heat external 0 -1 -1 -1 0 1 1 a 4\n') + command = self.command(self.sparse, model, events, + '--steady-envelope', end='1') + command += ['--envelope-limit-k', '380'] + result = self.execute(command, root) + self.assertNotEqual(result.returncode, 0) + self.assertIn('at least one heat-rejection boundary', result.stderr) + + def test_steady_envelope_does_not_search_below_frozen_alpha_set(self): + with tempfile.TemporaryDirectory() as root: + root = Path(root) + model, events = root / 'model.txt', root / 'events.txt' + model.write_text('HBFSIM_EQ3_THERMAL_MODEL 1\ncoupling on\n' + 'node a hbf capacity_memory s0 0 1 300 0 2 300\n') + events.write_text('HBFSIM_EQ3_THERMAL_EVENTS 1\n' + 'activity 1 cap external_heat external 0 -1 -1 -1 0 1 1 a 400\n') + command = self.command(self.sparse, model, events, + '--steady-envelope', end='1') + command += ['--envelope-limit-k', '320'] + result = self.execute(command, root) + self.assertEqual(result.returncode, 0, result.stderr) + receipt = json.loads(result.stdout) + self.assertEqual(receipt['status'], 'DOMAIN_REDESIGN_REQUIRED') + self.assertIsNone(receipt['selected_alpha']) + self.assertEqual([item['alpha'] for item in receipt['candidates']], + [1, .75, .5, .25]) + + def test_node_reordering_preserves_small_model_solution(self): + with tempfile.TemporaryDirectory() as root: + root=Path(root);events=root/'events.txt' + events.write_text('HBFSIM_EQ3_THERMAL_EVENTS 1\n' + 'activity 1 q external_heat external 0 -1 -1 -1 0 0.2 0.2 a 1\n') + models=[ + 'node a hbf capacity_memory s0 0 2 300 0.1 0.5 295\n' + 'node b hbm fast_memory s1 0 3 301 0.2 0.2 296\n' + 'node c gpu compute gpu 0 4 302 0.3 0.4 297\n' + 'edge a b 1 component\nedge b c 2 component\n', + 'node c gpu compute gpu 0 4 302 0.3 0.4 297\n' + 'node a hbf capacity_memory s0 0 2 300 0.1 0.5 295\n' + 'node b hbm fast_memory s1 0 3 301 0.2 0.2 296\n' + 'edge a b 1 component\nedge b c 2 component\n'] + values=[];receipts=[] + for number,body in enumerate(models): + directory=root/str(number);directory.mkdir();model=directory/'model.txt' + model.write_text('HBFSIM_EQ3_THERMAL_MODEL 1\ncoupling on\n'+body) + result=self.execute(self.command(self.sparse,model,events,'--run'),directory) + self.assertEqual(result.returncode,0,result.stderr) + values.append(self.csv_values(result.stdout)) + receipts.append(json.loads((directory/'rc_energy_receipt.json').read_text())) + self.assertEqual(values[0].keys(),values[1].keys()) + for key in values[0]: + self.assertAlmostEqual(values[0][key],values[1][key],places=11) + for field in ('total_input_energy_j','stored_energy_change_j', + 'boundary_loss_j','energy_residual_j'): + self.assertAlmostEqual(receipts[0][field],receipts[1][field],places=11) + def test_real_domain_violation_not_clamped(self): with tempfile.TemporaryDirectory() as root: root=Path(root) model,events=root/'model.txt',root/'events.txt' + model.write_text('HBFSIM_EQ3_THERMAL_MODEL 1\ncoupling on\n' + 'node a hbf capacity_memory s0 0 1 300 0 0 300\n') + events.write_text('HBFSIM_EQ3_THERMAL_EVENTS 1\n' + 'activity 1 q external_heat external 0 -1 -1 -1 0 0.2 0.2 a 30\n') + command=self.command(self.sparse,model,events,'--run') + command[command.index('--max-k')+1]='302' + model_sha=hashlib.sha256(model.read_bytes()).hexdigest() + events_sha=hashlib.sha256(events.read_bytes()).hexdigest() + source_sha=hashlib.sha256((ROOT/'tools/eq3_campaign_rc_runner.cpp').read_bytes()).hexdigest() + command += ['--model-sha256',model_sha, + '--events-sha256',events_sha, + '--runner-source-sha256',source_sha, + '--domain-version','fixture-domain-v1'] + result=self.execute(command,root) + print('DOMAIN_REPRO_STDERR:',result.stderr.strip()) + self.assertNotEqual(result.returncode,0) + self.assertIn('temperature left required domain',result.stderr) + diagnostic_path=root/'rc_failure_diagnostic.json' + if diagnostic_path.exists(): + print('DOMAIN_FAILURE_DIAGNOSTIC:',diagnostic_path.read_text()) + print('DOMAIN_LAST_VALID_STATE:',(root/'rc_failure_last_valid.csv').read_text()) + print('DOMAIN_TRIAL_STATE:',(root/'rc_failure_trial.csv').read_text()) + diagnostic=json.loads(diagnostic_path.read_text()) + self.assertEqual(diagnostic['status'],'DOMAIN_FAILURE') + self.assertTrue(diagnostic['failure_returned_to_caller']) + self.assertFalse(diagnostic['temperature_clamping']) + self.assertEqual(diagnostic['last_valid_time_s'],.01) + self.assertEqual(diagnostic['trial_target_time_s'],.02) + self.assertEqual(diagnostic['completed_steps'],1) + self.assertEqual(diagnostic['trial_step_1_based'],2) + self.assertEqual(diagnostic['input_interval_start_s'],0) + self.assertEqual(diagnostic['input_interval_end_s'],.1) + self.assertEqual(diagnostic['model_sha256'],model_sha) + self.assertEqual(diagnostic['events_sha256'],events_sha) + self.assertEqual(diagnostic['runner_source_sha256'],source_sha) + self.assertEqual(diagnostic['domain_version'],'fixture-domain-v1') + self.assertTrue(diagnostic['offending_nodes']) + self.assertEqual(diagnostic['offending_nodes'][0]['id'],'a') + self.assertEqual(diagnostic['failed_trial_step_energy']['status'], + 'UNKNOWN_NOT_INTEGRATED') + self.assertIsNone(diagnostic['failed_trial_step_energy']['energy_residual_j']) + completed=diagnostic['completed_interval_energy'] + self.assertAlmostEqual(completed['activity_input_j'],1.5) + self.assertAlmostEqual(completed['stored_energy_change_j'],1.5) + self.assertAlmostEqual(completed['energy_residual_j'],0) + self.assertTrue((root/'rc_failure_last_valid.csv').is_file()) + self.assertTrue((root/'rc_failure_trial.csv').is_file()) + + def test_nonfinite_input_is_rejected_before_trial(self): + with tempfile.TemporaryDirectory() as root: + root=Path(root);model,events=root/'model.txt',root/'events.txt' + model.write_text('HBFSIM_EQ3_THERMAL_MODEL 1\ncoupling on\n' + 'node a hbf capacity_memory s0 0 nan 300 0 1 290\n') + events.write_text('HBFSIM_EQ3_THERMAL_EVENTS 1\n') + result=self.execute(self.command(self.sparse,model,events,'--run'),root) + self.assertNotEqual(result.returncode,0) + self.assertIn('malformed node record',result.stderr) + self.assertFalse((root/'rc_failure_diagnostic.json').exists()) + + def test_nonfinite_trial_is_recorded_as_numerical_failure(self): + with tempfile.TemporaryDirectory() as root: + root=Path(root);model,events=root/'model.txt',root/'events.txt' + model.write_text('HBFSIM_EQ3_THERMAL_MODEL 1\ncoupling on\n' + 'node a hbf capacity_memory s0 0 1 300 0 0 300\n') + events.write_text('HBFSIM_EQ3_THERMAL_EVENTS 1\n' + 'activity 1 q external_heat external 0 -1 -1 -1 0 0.2 0.2 a 1e308\n') + result=self.execute(self.command(self.sparse,model,events,'--run'),root) + self.assertNotEqual(result.returncode,0) + diagnostic=json.loads((root/'rc_failure_diagnostic.json').read_text()) + self.assertEqual(diagnostic['status'],'NUMERICAL_FAILURE') + self.assertGreater(diagnostic['nonfinite_trial_nodes'],0) + self.assertFalse(diagnostic['trial_extrema_available']) + self.assertIsNone(diagnostic['trial_max_k']) + + def test_failure_evidence_is_never_overwritten(self): + with tempfile.TemporaryDirectory() as root: + root=Path(root);model,events=root/'model.txt',root/'events.txt' + model.write_text('HBFSIM_EQ3_THERMAL_MODEL 1\ncoupling on\n' + 'node a hbf capacity_memory s0 0 1 300 0 1 290\n') + events.write_text('HBFSIM_EQ3_THERMAL_EVENTS 1\n') + marker=root/'rc_failure_diagnostic.json';marker.mkdir() + result=self.execute(self.command(self.sparse,model,events,'--run'),root) + self.assertNotEqual(result.returncode,0) + self.assertIn('refusing to overwrite RC output',result.stderr) + self.assertTrue(marker.is_dir()) + + def test_failure_evidence_write_error_still_returns_failure(self): + if not Path('/proc').is_dir(): + self.skipTest('/proc read-only pseudo-filesystem is unavailable') + with tempfile.TemporaryDirectory() as root: + root=Path(root);model,events=root/'model.txt',root/'events.txt' model.write_text('HBFSIM_EQ3_THERMAL_MODEL 1\ncoupling on\n' 'node a hbf capacity_memory s0 0 1 300 0 1 290\n') events.write_text('HBFSIM_EQ3_THERMAL_EVENTS 1\n') command=self.command(self.sparse,model,events,'--run') command[command.index('--min-k')+1]='300' - result=self.execute(command,root) + result=self.execute(command,Path('/proc')) self.assertNotEqual(result.returncode,0) - self.assertIn('temperature left required domain',result.stderr) + self.assertIn('cannot create failure state',result.stderr) def test_candidate_equilibrium_diagnostic_when_available(self): - generated=WORKSPACE/'generated/layered-v3/train-4mm-20ms' + if self.generated_root is None: + self.skipTest('EQ3_GENERATED_ROOT was not explicitly supplied') + generated=self.generated_root/'layered-v3/train-4mm-20ms' if not generated.is_dir(): self.skipTest('generated candidate unavailable') with tempfile.TemporaryDirectory() as root: @@ -170,7 +369,9 @@ def test_candidate_equilibrium_diagnostic_when_available(self): self.assertLess(receipt['absolute_max_equilibrium_error_k'],1e-8) def test_generated_candidate_inspect_when_available(self): - generated = WORKSPACE / "generated/layered-v1/train-4mm-20ms" + if self.generated_root is None: + self.skipTest('EQ3_GENERATED_ROOT was not explicitly supplied') + generated = self.generated_root / "layered-v1/train-4mm-20ms" if not generated.is_dir(): self.skipTest("generated candidate is unavailable") with tempfile.TemporaryDirectory() as root: diff --git a/tools/test_eq3_campaign_stream.py b/tools/test_eq3_campaign_stream.py index 5bf3dd4..0981f6b 100644 --- a/tools/test_eq3_campaign_stream.py +++ b/tools/test_eq3_campaign_stream.py @@ -66,6 +66,18 @@ def test_abnormal_producer_exit(self): with self.assertRaisesRegex(ValueError, 'producer failed'): self.producer(tmp, mode='fail') + def test_bound_outer_policy_can_supply_longer_watchdog_and_process_ram(self): + with tempfile.TemporaryDirectory() as tmp: + result=run_stream([sys.executable,'-c',PRODUCER,'1','1','1','1','ok'],tmp,1,1,1,1, + watchdog=3600,process_ram_gib=.25) + self.assertEqual(result['status'],'PASS') + with tempfile.TemporaryDirectory() as tmp: + with self.assertRaisesRegex(ValueError,'invalid dimensions or watchdog'): + run_stream([sys.executable,'-c','pass'],tmp,1,1,1,1,watchdog=0) + with tempfile.TemporaryDirectory() as tmp: + with self.assertRaisesRegex(ValueError,'point output budget exceeded'): + run_stream([sys.executable,'-c',PRODUCER,'1','1','1','1','ok'],tmp,1,1,1,1,watchdog=10,max_bytes=1) + def test_replay_and_corrupt_footer(self): with tempfile.TemporaryDirectory() as tmp: source, compressed = Path(tmp) / 'field', Path(tmp) / 'field.gz' diff --git a/tools/test_eq3_cpu_load_matrix.py b/tools/test_eq3_cpu_load_matrix.py new file mode 100644 index 0000000..52559dd --- /dev/null +++ b/tools/test_eq3_cpu_load_matrix.py @@ -0,0 +1,34 @@ +import json,tempfile,unittest +from pathlib import Path +from eq3_cpu_load_compare import compare +from eq3_cpu_load_matrix import POLICIES,RATES,make_load_case +from eq3_cpu_load_review import percentile,queue_trace + +class SustainedLoadMatrixTests(unittest.TestCase): + @classmethod + def setUpClass(cls):cls.code=Path(__file__).resolve().parents[1] + def test_exact_matrix_and_policy_fairness(self): + for rate in RATES: + cases=[make_load_case(self.code,rate,p) for p in POLICIES] + self.assertEqual({len(x['requests']) for x in cases},{rate*20*8}) + self.assertEqual(len({x['workload_identity_sha256'] for x in cases}),1) + self.assertTrue(all(x['end_ns']==30_000_000_000 and max(r['arrival_ns'] for r in x['requests'])<20_000_000_000 for x in cases)) + self.assertAlmostEqual(make_load_case(self.code,25,'none')['screening']['foreground_base_utilization_per_stack'],.5) + self.assertEqual(make_load_case(self.code,25,'none')['screening']['light_capacity_rps'],12.5) + def test_review_helpers_include_censoring(self): + jobs=[{'maintenance':False,'arrival_ns':0,'start_ns':5,'end_ns':10}, + {'maintenance':False,'arrival_ns':0},{'maintenance':True,'arrival_ns':0,'start_ns':0,'end_ns':20}] + trace=queue_trace(jobs,20,10);self.assertEqual(trace[0]['foreground_queued'],2);self.assertEqual(trace[1]['foreground_queued'],1) + self.assertEqual(percentile([1,2,3],.5),2);self.assertAlmostEqual(percentile([1,3],.95),2.9) + def test_aggregate_retains_negative_result(self): + rows=[] + for rate in RATES: + for i,policy in enumerate(POLICIES): + rows.append({'rate_per_stack_rps':rate,'policy':policy,'workload_identity_sha256':str(rate), + 'foreground_complete':100-i,'foreground_unfinished':i,'completed_logical_bytes':4096*(100-i), + 'max_temperature_k':300-i,'latency_completed_s':{'p95':.1+i},'maintenance':{'queued':i,'overdue_cohorts':i}, + 'max_data_age_s':i,'package_dynamic_energy_j':10-i,'external_energy_j':1}) + result=compare(rows);self.assertTrue(result['rates']['5']['arms']['hysteresis']['negative_flags']['lower_temperature_with_more_backlog']) + self.assertNotIn('winner',result) + self.assertTrue(all('winner' not in group for group in result['rates'].values())) +if __name__=='__main__':unittest.main() diff --git a/tools/test_eq3_domain_v2_input.py b/tools/test_eq3_domain_v2_input.py new file mode 100644 index 0000000..fb0dd75 --- /dev/null +++ b/tools/test_eq3_domain_v2_input.py @@ -0,0 +1,73 @@ +import unittest + +from eq3_domain_v2_input import derive + + +def power_fixture(): + order = [f"g{index}" for index in range(17)] + return { + "schema_version": "source-v1", "slot_s": .5, + "group_order": order, "caps_W": [100.0] * 17, + "traces": { + "train": {"duration_s": 1.0, + "slots_W": [[0.0] * 17, [float(index) for index in range(17)]]}, + "development": {"duration_s": .5, "slots_W": [[100.0] * 17]}, + "new_blind": {"slots_W": "POISON_MUST_NOT_BE_TRAVERSED"}, + }, + } + + +def receipt_fixture(alpha=.5): + return { + "schema_version": "eq3-steady-envelope-v1", + "status": "PREDICTED_ENVELOPE", "reference_qualified": False, + "selected_alpha": alpha, + "candidates": [ + {"alpha": 1.0, "within_limit": False, "initial_covered": True}, + {"alpha": .75, "within_limit": False, "initial_covered": True}, + {"alpha": .5, "within_limit": True, "initial_covered": True}, + {"alpha": .25, "within_limit": True, "initial_covered": True}, + ], + } + + +class DomainV2InputTests(unittest.TestCase): + def test_one_alpha_scales_all_groups_and_does_not_read_blind(self): + derived, receipt = derive(power_fixture(), receipt_fixture(), + ["train", "development"]) + self.assertEqual(set(derived["traces"]), {"train", "development"}) + self.assertFalse(derived["domain_v2"]["blind_trace_included"]) + self.assertFalse(receipt["blind_trajectory_read"]) + self.assertEqual(derived["caps_W"], [50.0] * 17) + self.assertEqual(derived["traces"]["train"]["slots_W"][1], + [index * .5 for index in range(17)]) + self.assertEqual(derived["traces"]["development"]["slots_W"][0], + [50.0] * 17) + for value in receipt["energy"].values(): + self.assertAlmostEqual(value["scaled_energy_j"], + .5 * value["original_energy_j"]) + + def test_blind_request_and_nonlargest_or_unknown_alpha_are_rejected(self): + with self.assertRaisesRegex(ValueError, "train/development"): + derive(power_fixture(), receipt_fixture(), ["new_blind"]) + wrong = receipt_fixture(.25) + with self.assertRaisesRegex(ValueError, "largest eligible"): + derive(power_fixture(), wrong, ["train"]) + outside = receipt_fixture() + outside["selected_alpha"] = .1 + with self.assertRaisesRegex(ValueError, "candidate set"): + derive(power_fixture(), outside, ["train"]) + + def test_missing_group_or_cap_violation_is_rejected(self): + source = power_fixture() + source["group_order"] = source["group_order"][:-1] + with self.assertRaisesRegex(ValueError, "17 ordered"): + derive(source, receipt_fixture(), ["train"]) + source = power_fixture() + source["traces"]["train"]["slots_W"][0][0] = 101 + with self.assertRaisesRegex(ValueError, "exceeds"): + derive(source, receipt_fixture(), ["train"]) + + +if __name__ == "__main__": + unittest.main() diff --git a/tools/test_eq3_layered_launch.py b/tools/test_eq3_layered_launch.py index c216c6a..38cd023 100644 --- a/tools/test_eq3_layered_launch.py +++ b/tools/test_eq3_layered_launch.py @@ -277,6 +277,18 @@ def test_twelve_gib_process_ceiling_is_allowed(self): result = self.validate(manifest=manifest, approval=approval, launch=launch) self.assertEqual(result["status"], "READY_TO_LAUNCH") + def test_explicit_stage_resource_policy_uses_bound_scope_limits(self): + launch=copy.deepcopy(self.launch);launch['limits'].update(ram_gib=16,watchdog_seconds=3600) + launch['output']['max_new_gib']=8;manifest,approval=self.rebuild_binding(launch) + manifest['resource_budget']['requested'].update(ram_gib=16,disk_gib=8) + stage={'status':'READY_FOR_SUBMISSION','experiment_id':manifest['experiment_id'],'version':manifest['version'], + 'resource_policy':'PER_EXPERIMENT_USER_CONFIRMED','resource_limits':{'task_ram_gib':20,'process_ram_gib':16, + 'threads':1,'watchdog_s':3600,'point_disk_gib':8,'task_disk_gib':40,'min_free_disk_gib':10,'gpu':0,'cloud':0}} + with mock.patch('eq3_layered_launch.validate_gate',return_value=stage): + self.assertEqual(self.validate(manifest=manifest,approval=approval,launch=launch)['status'],'READY_TO_LAUNCH') + launch['limits']['watchdog_seconds']=3599 + self.assert_refused('LAUNCH_LIMIT_EXCEEDED',manifest=manifest,approval=approval,launch=launch) + def test_each_hard_safety_ceiling_is_refused(self): changes = ( ("limits", "threads", 2), diff --git a/tools/test_eq3_q4_all_hbf_profile.py b/tools/test_eq3_q4_all_hbf_profile.py new file mode 100644 index 0000000..2ffb6ea --- /dev/null +++ b/tools/test_eq3_q4_all_hbf_profile.py @@ -0,0 +1,94 @@ +import copy +import json +from pathlib import Path +import tempfile +import unittest + +from eq3_layered_ir import normalize +from eq3_q4_all_hbf_profile import PAIRING, convert_profile, fixture_power, _write_new + + +ROOT = Path(__file__).resolve().parents[1] + + +class Q4AllHbfProfileTests(unittest.TestCase): + @classmethod + def setUpClass(cls): + cls.source = json.loads( + (ROOT / "configs/eq3_thermal/research/candidate_profile.json").read_text() + ) + + def test_complete_templates_translate_without_relabeling_hbm(self): + q4 = convert_profile(self.source) + placements = {row["id"]: row for row in q4["placements"]} + self.assertEqual({"gpu", *(f"hbf{i}" for i in range(8))}, set(placements)) + self.assertFalse(any(block["id"].startswith("hbm") for block in q4["blocks"])) + for old, template, new in PAIRING: + self.assertEqual(16, placements[new]["array_die_count"]) + self.assertEqual( + next(x for x in self.source["placements"] if x["id"] == old)["xy_um"], + placements[new]["xy_um"], + ) + source_blocks = { + block["id"].replace(template, new, 1): block + for block in self.source["blocks"] + if block["id"].startswith(template + ".") + } + new_blocks = { + block["id"]: block for block in q4["blocks"] if block["id"].startswith(new + ".") + } + self.assertEqual(set(source_blocks), set(new_blocks)) + self.assertEqual(list(range(16)), sorted( + block["die_index"] for block in new_blocks.values() if ".die" in block["id"] + )) + for identity, cloned in new_blocks.items(): + original = source_blocks[identity] + self.assertEqual(original["size_um"], cloned["size_um"]) + self.assertEqual(original["xyz_um"][2], cloned["xyz_um"][2]) + self.assertEqual(original["material"], cloned["material"]) + + def test_package_physics_preserved_and_external_gddr_outside_domain(self): + q4 = convert_profile(self.source) + for key in ("package_size_um", "materials", "background", "boundaries", "model_domain"): + self.assertEqual(self.source[key], q4[key]) + topology = q4["data_topology"] + self.assertEqual("all_hbf_direct", topology["kind"]) + self.assertEqual(8, topology["hbf_count"]) + self.assertEqual(0, topology["hbm_count"]) + self.assertFalse(topology["external_fast_memory_profile"]["package_geometry_modeled"]) + self.assertEqual("UNAVAILABLE", q4["device_selection"]["gddr"]["package_temperature"]) + + def test_normalizer_energy_and_sensor_contract(self): + q4 = convert_profile(self.source) + power = fixture_power(q4) + ir = normalize(q4, power, "mapping_20ms") + self.assertEqual("all_hbf_direct", ir["topology"]["mode"]) + self.assertEqual(8, ir["topology"]["hbf_count"]) + self.assertEqual(0, ir["topology"]["hbm_count"]) + self.assertEqual(287, len(ir["components"])) + self.assertEqual(307, len(ir["sensors"])) + self.assertEqual(137, sum(component["powered"] for component in ir["components"])) + self.assertAlmostEqual(2.74, ir["power"]["total_energy_j"], places=12) + external = next(x for x in ir["devices"] if x["id"] == "external_fast_memory") + self.assertFalse(external["package_geometry_modeled"]) + + def test_rejects_rotation_or_incomplete_template(self): + rotated = copy.deepcopy(self.source) + next(x for x in rotated["placements"] if x["id"] == "hbm0")["footprint_um"] = [16000, 12000] + with self.assertRaisesRegex(ValueError, "orientation"): + convert_profile(rotated) + incomplete = copy.deepcopy(self.source) + incomplete["blocks"] = [x for x in incomplete["blocks"] if x["id"] != "hbf0.die15"] + with self.assertRaisesRegex(ValueError, "complete 35-block"): + convert_profile(incomplete) + + def test_outputs_are_create_only(self): + with tempfile.TemporaryDirectory() as directory: + path = Path(directory) / "out.json" + _write_new(path, {"a": 1}) + with self.assertRaises(FileExistsError): + _write_new(path, {"a": 2}) + + +if __name__ == "__main__": + unittest.main() diff --git a/tools/test_eq3_sensor_time_canonicalize.py b/tools/test_eq3_sensor_time_canonicalize.py new file mode 100644 index 0000000..a26f152 --- /dev/null +++ b/tools/test_eq3_sensor_time_canonicalize.py @@ -0,0 +1,75 @@ +import csv +import hashlib +import tempfile +import unittest +from pathlib import Path + +from eq3_sensor_time_canonicalize import canonicalize + + +FIELDS = ["time_s", "sensor_id", "temperature_k", "hotspot_cell_id"] + + +def write_csv(path, rows): + with path.open("w", newline="") as output: + writer = csv.DictWriter(output, fieldnames=FIELDS) + writer.writeheader() + writer.writerows(rows) + + +def read_csv(path): + with path.open(newline="") as source: + return list(csv.DictReader(source)) + + +class SensorTimeCanonicalizeTest(unittest.TestCase): + def test_binary_spelling_only_and_non_time_lossless(self): + rows = [ + {"time_s": time_s, "sensor_id": sensor, + "temperature_k": f"{300 + index / 10:.17g}", + "hotspot_cell_id": f"{sensor}-cell-{index}"} + for sensor, times in ( + ("a", ["0.1", "0.20000000000000001", "0.30000000000000004"]), + ("b", ["0.10000000000000001", "0.2", "0.29999999999999999"]), + ) + for index, time_s in enumerate(times, 1) + ] + with tempfile.TemporaryDirectory() as directory: + source = Path(directory) / "source.csv" + output = Path(directory) / "canonical.csv" + write_csv(source, rows) + before = hashlib.sha256(source.read_bytes()).hexdigest() + receipt = canonicalize(source, output, 0.1, 0.3, 1e-12) + self.assertEqual(hashlib.sha256(source.read_bytes()).hexdigest(), before) + self.assertLessEqual(receipt["max_abs_time_adjustment_s"], 1e-14) + actual = read_csv(output) + self.assertEqual([row["time_s"] for row in actual], + ["0.1", "0.2", "0.3", "0.1", "0.2", "0.3"]) + names = [name for name in FIELDS if name != "time_s"] + self.assertEqual([[row[name] for name in names] for row in actual], + [[row[name] for name in names] for row in rows]) + + def test_rejects_real_shift_duplicate_missing_and_coverage_change(self): + valid = [ + {"time_s": str(index / 10), "sensor_id": sensor, + "temperature_k": "300", "hotspot_cell_id": sensor} + for sensor in ("a", "b") for index in (1, 2, 3) + ] + cases = { + "real shift": [dict(row, time_s="0.1001") if row["sensor_id"] == "a" and row["time_s"] == "0.1" else row for row in valid], + "duplicate": [dict(row, time_s="0.1") if row["sensor_id"] == "a" and row["time_s"] == "0.2" else row for row in valid], + "missing": [row for row in valid if not (row["sensor_id"] == "a" and row["time_s"] == "0.2")], + "coverage change": [row for row in valid if not (row["sensor_id"] == "b" and row["time_s"] == "0.3")], + } + with tempfile.TemporaryDirectory() as directory: + for label, rows in cases.items(): + with self.subTest(label=label): + source = Path(directory) / f"{label.replace(' ', '-')}.csv" + output = Path(directory) / f"{label.replace(' ', '-')}-out.csv" + write_csv(source, rows) + with self.assertRaises(ValueError): + canonicalize(source, output, 0.1, 0.3, 1e-12) + + +if __name__ == "__main__": + unittest.main() diff --git a/tools/test_eq3_spatial_diagnostic.py b/tools/test_eq3_spatial_diagnostic.py new file mode 100644 index 0000000..7566164 --- /dev/null +++ b/tools/test_eq3_spatial_diagnostic.py @@ -0,0 +1,111 @@ +import csv +import tempfile +import unittest +from pathlib import Path + +from eq3_spatial_diagnostic import analyze, write_result + + +SENSORS = ("component:hbf0.die0:mean", "component:hbf0.die0:hotspot") + + +def method(): + return { + "schema_version": "eq3-acceptance-v2-method-v1", + "temperature_unit": "K", + "sensor_ids": list(SENSORS), + "initial_time_s": 0.0, + "initial_temperature_k": 300.0, + "windows": [ + {"id": "full", "kind": "full", "start_s": 0.0, "end_s": 4.0}, + {"id": "heat", "kind": "excitation", "start_s": 0.0, "end_s": 2.0}, + {"id": "cool", "kind": "cooling", "start_s": 2.0, "end_s": 4.0}, + ], + "hotspot_sensor_ids": [SENSORS[1]], + "control_sensor_ids": [], + "crossing_thresholds_k": [301.0], + "crossing_min_time_s": 0.2, + "crossing_fraction": 0.05, + } + + +def write_csv(path, offset, times=(1.0, 2.0, 4.0), sensors=SENSORS): + with path.open("w", newline="") as output: + writer = csv.writer(output) + writer.writerow(["time_s", "sensor_id", "temperature_k", "hotspot_cell_id"]) + for sensor in sensors: + is_hotspot = sensor.endswith(":hotspot") + for index, time_s in enumerate(times): + base = (300.0, 302.0, 301.0)[index] + temperature = base + offset * (index + 1) * (2 if is_hotspot else 1) + cell = f"n{int(offset * 10)}_{index}" if is_hotspot else "" + writer.writerow([time_s, sensor, temperature, cell]) + + +class SpatialDiagnosticTests(unittest.TestCase): + def setUp(self): + self.temporary = tempfile.TemporaryDirectory() + self.root = Path(self.temporary.name) + self.paths = [self.root / f"R0{index}.csv" for index in range(1, 4)] + for path, offset in zip(self.paths, (0.0, 0.1, 0.3)): + write_csv(path, offset) + + def tearDown(self): + self.temporary.cleanup() + + def run_analysis(self): + return analyze(*self.paths, method()) + + def test_adjacent_pairs_windows_groups_and_exact_hotspot_cell(self): + result = self.run_analysis() + self.assertEqual(result["richardson_order"], "NOT_COMPUTED") + self.assertEqual(len(result["pairs"]), 2) + first = result["pairs"][0] + self.assertEqual((first["reference_grid"], first["candidate_grid"]), + ("4mm", "2mm")) + self.assertEqual({window["window_kind"] for window in first["windows"]}, + {"full", "excitation", "cooling"}) + full = next(window for window in first["windows"] + if window["window_kind"] == "full") + self.assertEqual(full["groups"]["mean"]["sensor_count"], 1) + self.assertEqual(full["groups"]["hotspot"]["sensor_count"], 1) + point = full["groups"]["hotspot"]["worst_registered_point"] + self.assertEqual(point["sensor_id"], SENSORS[1]) + self.assertEqual(point["time_s"], 4.0) + self.assertEqual(point["reference_hotspot_cell_id"], "n0_2") + self.assertEqual(point["candidate_hotspot_cell_id"], "n1_2") + self.assertAlmostEqual(point["abs_error_k"], 0.6) + + def test_excitation_and_cooling_are_scored_separately(self): + windows = self.run_analysis()["pairs"][0]["windows"] + heat = next(window for window in windows if window["window_kind"] == "excitation") + cool = next(window for window in windows if window["window_kind"] == "cooling") + heat_mae = heat["groups"]["mean"]["mean_sensor_time_weighted_mae_k"] + cool_mae = cool["groups"]["mean"]["mean_sensor_time_weighted_mae_k"] + self.assertNotEqual(heat_mae, cool_mae) + + def test_incomplete_r03_rejects_whole_analysis(self): + write_csv(self.paths[2], 0.3, times=(1.0, 2.0)) + with self.assertRaisesRegex(ValueError, "R03 timestamp coverage differs"): + self.run_analysis() + + def test_r03_sensor_coverage_mismatch_is_rejected(self): + write_csv(self.paths[2], 0.3, sensors=(SENSORS[0],)) + with self.assertRaisesRegex(ValueError, "R03 sensor coverage differs"): + self.run_analysis() + + def test_postprocessor_does_not_modify_raw_inputs(self): + before = [path.read_bytes() for path in self.paths] + self.run_analysis() + self.assertEqual([path.read_bytes() for path in self.paths], before) + + def test_output_is_create_only(self): + output = self.root / "result.json" + output.write_text("preserve") + with self.assertRaises(FileExistsError): + write_result(output, self.run_analysis()) + self.assertEqual(output.read_text(), "preserve") + + +if __name__ == "__main__": + unittest.main()