diff --git a/.github/workflows/macos.yml b/.github/workflows/macos.yml index bb1d683e..01adf17e 100644 --- a/.github/workflows/macos.yml +++ b/.github/workflows/macos.yml @@ -38,12 +38,10 @@ jobs: - name: Try building extensions run: | pdm run build-ext - pdm run build-ext-test - - - run: pdm run build-ext-ref - run: cargo install mdbook-toc - run: cargo install mdbook-katex --version 0.10.0-alpha - uses: taiki-e/install-action@mdbook - - run: pdm run test-refsol + - name: Run printed shipped-day reference CI manifest + run: pdm run python scripts/week2_shipped_day_ci.py diff --git a/README.md b/README.md index 04f51a3f..6235540e 100644 --- a/README.md +++ b/README.md @@ -27,12 +27,10 @@ The course follows a four-week learning path: - **Week 1: From Matmul to Text.** Build a Qwen3 model directly from `mlx.core` array operations: attention, RoPE, GQA, RMSNorm, the MLP, sampling, and the autoregressive loop. -- **Week 2: A Step Closer to vLLM.** Add a KV cache, establish a - synchronized MLX baseline, and let matched benchmarks choose each - optimization. The causal path moves from quantized decode matvec to fused - model kernels and SIMD-matrix prefill; decode attention is an optional - workload-conditioned lab, and split-K stays only where a measured short - shape supports it. +- **Week 2: A Faster Single Request.** The current Day 1 route adds + `kv-cache` and request-bounded `capacity-cache`, then measures both against + the Week 1 full-prefix control. Later packed-W4, SIMD, fused-primitive, and + tiled-attention lessons are planned; their checkpoints are not Day 1 gates. - **Week 3: Build a Mini vLLM.** Introduce continuous batching and chunked admission, then make paged KV the canonical serving layout. Decode attention and FlashAttention learn to read pages directly so @@ -109,13 +107,7 @@ one explicit byte range through the existing loop. | 1.5 | Load the Model | ✅ | ✅ | ✅ | ✅ | | 1.6 | Generate Responses (aka Decoding) | ✅ | ✅ | ✅ | ✅ | | 1.7 | Sampling | ✅ | ✅ | ✅ | ✅ | -| 2.1 | KV Cache | ✅ | ✅ | ✅ | 🚧 | -| 2.2 | Benchmarking and Profiling | ✅ | ✅ | ✅ | 🚧 | -| 2.3 | Quantize the Model | ✅ | ✅ | ✅ | 🚧 | -| 2.4 | Fused Model Kernels | ✅ | ✅ | ✅ | 🚧 | -| 2.5 | SIMD-Matrix Prefill | ✅ | ✅ | ✅ | 🚧 | -| 2.6 (optional) | Workload-Conditioned Operator Lab | ✅ | ✅ | ✅ | 🚧 | -| 2.7 | Conditional Split-K and Final Decision | ✅ | ✅ | ✅ | 🚧 | +| 2.1 | Cache and Measure (`kv-cache`, `capacity-cache`) | 🚧 | 🚧 | ✅ | 🚧 | | 3.1 | Continuous Batching | ✅ | ✅ | ✅ | 🚧 | | 3.2 | Chunked Prefill | ✅ | ✅ | ✅ | 🚧 | | 3.3 | Paged KV Cache | ✅ | ✅ | ✅ | 🚧 | @@ -133,6 +125,8 @@ one explicit byte range through the existing loop. | 4.8 | Fork, Steer, and Select | ✅ | ✅ | ✅ | 🚧 | | 4.9 | Bound Tool Evidence | ✅ | ✅ | ✅ | 🚧 | +The older Week 2 chapter URLs remain available as [historical material](book/src/week2-02-benchmark-profile.md). Their former Day 2–7 tests and checkpoint commands are not part of the current Day 1 learner route. + Other topics not covered include quantized or compressed KV caches, cross-request prefix caching, fine-tuning, and long-context techniques. diff --git a/benches/bench.py b/benches/bench.py index d3f4398f..36f66560 100644 --- a/benches/bench.py +++ b/benches/bench.py @@ -86,16 +86,7 @@ def parse_args() -> argparse.Namespace: ) parser.add_argument( "--week2-checkpoint", - choices=( - "kv-cache", - "quantized-matvec", - "rmsnorm", - "rope", - "swiglu", - "simd-matmul", - "decode-attention", - "split-k", - ), + choices=("kv-cache", "capacity-cache"), help="run one cumulative Week 2 end-to-end checkpoint", ) parser.add_argument("--device", type=str, default="gpu", choices=["cpu", "gpu"]) @@ -165,12 +156,15 @@ def validate_args(args: argparse.Namespace) -> None: and args.device != "gpu" and ( args.loader == "week3" - or (args.loader == "week2" and args.week2_checkpoint != "kv-cache") + or ( + args.loader == "week2" + and args.week2_checkpoint not in (None, "kv-cache", "capacity-cache") + ) ) ): raise ValueError( - "The completed Week 2 and Week 3 custom-kernel models are GPU-only; " - "use the Week 2 kv-cache checkpoint for the readable pre-kernel path" + "Week 3 custom-kernel models are GPU-only; " + "Day 1 Week 2 checkpoints use the readable path" ) if args.disable_paged_attention and args.loader != "week3": raise ValueError("--disable-paged-attention requires --loader week3") diff --git a/benches/bench_course_progression.py b/benches/bench_course_progression.py index c1ab5212..6b0d21d7 100644 --- a/benches/bench_course_progression.py +++ b/benches/bench_course_progression.py @@ -42,59 +42,17 @@ class Throughput: WEEK1_VARIANT, Variant( "week2-kv-cache", - "2.1 KV cache", + "2.1 Reuse the prefix", "ref", "week2", ("--week2-checkpoint", "kv-cache"), ), Variant( - "week2-quantized-matvec", - "2.3 Quantized matvec", + "week2-capacity-cache", + "2.1 + Bound KV-cache movement", "ref", "week2", - ("--week2-checkpoint", "quantized-matvec"), - ), - Variant( - "week2-rmsnorm", - "2.4 Fast RMSNorm", - "ref", - "week2", - ("--week2-checkpoint", "rmsnorm"), - ), - Variant( - "week2-rope", - "2.4 + Fast RoPE", - "ref", - "week2", - ("--week2-checkpoint", "rope"), - ), - Variant( - "week2-swiglu", - "2.4 + Fused SwiGLU", - "ref", - "week2", - ("--week2-checkpoint", "swiglu"), - ), - Variant( - "week2-simd-matmul", - "2.5 SIMD matrix prefill", - "ref", - "week2", - ("--week2-checkpoint", "simd-matmul"), - ), - Variant( - "week2-decode-attention", - "2.6 Optional decode attention", - "ref", - "week2", - ("--week2-checkpoint", "decode-attention"), - ), - Variant( - "week2-split-k", - "2.7 Split-K prefill", - "ref", - "week2", - ("--week2-checkpoint", "split-k"), + ("--week2-checkpoint", "capacity-cache"), ), MLX_VARIANT, ) diff --git a/benches/profile_week2_kernels.py b/benches/profile_week2_kernels.py index a399089e..6c9fd0ec 100644 --- a/benches/profile_week2_kernels.py +++ b/benches/profile_week2_kernels.py @@ -25,13 +25,7 @@ DEFAULT_CASES = ( "kv-cache:decode:128", - "quantized-matvec:decode:128", - "swiglu:decode:128", - "simd-matmul:prefill:128", - "simd-matmul:prefill:32", - "decode-attention:decode:128", - "decode-attention:prefill:128", - "split-k:prefill:32", + "capacity-cache:decode:128", ) PROMPT_RULE = "synthetic-token-ids" PREFILL_LOGITS = "all" diff --git a/benches/week2_gpudebug.py b/benches/week2_gpudebug.py index fc22e093..99af6d50 100644 --- a/benches/week2_gpudebug.py +++ b/benches/week2_gpudebug.py @@ -19,13 +19,7 @@ PREFILL_LOGITS = "all" KNOWN_CHECKPOINTS = ( "kv-cache", - "quantized-matvec", - "rmsnorm", - "rope", - "swiglu", - "simd-matmul", - "decode-attention", - "split-k", + "capacity-cache", ) diff --git a/book/src/SUMMARY.md b/book/src/SUMMARY.md index 68705cde..be1578ed 100644 --- a/book/src/SUMMARY.md +++ b/book/src/SUMMARY.md @@ -13,15 +13,15 @@ - [The Qwen3 Model](./week1-05-qwen3-model.md) - [Generating the Response](./week1-06-generate-response.md) - [Sampling and Preparing for Week 2](./week1-07-sampling-prepare.md) -- [🚧 Week 2: A Step Closer to vLLM](./week2-overview.md) - - [🚧 KV Cache](./week2-01-kv-cache.md) - - [🚧 Benchmarking and Profiling](./week2-02-benchmark-profile.md) - - [🚧 Optional: Metal Profiling](./week2-advanced-profiling.md) - - [🚧 Quantize the Model](./week2-03-quantize-model.md) - - [🚧 Fused Model Kernels](./week2-04-fused-model-kernels.md) - - [🚧 SIMD-Matrix Prefill](./week2-05-simd-matrix-prefill.md) - - [🚧 Day 6 (Optional): Workload-Conditioned Operator Lab](./week2-06-operator-lab.md) - - [🚧 Conditional Split-K and Final Decision](./week2-07-split-k-prefill.md) +- [🚧 Week 2: A Faster Single Request](./week2-overview.md) + - [🚧 Day 1: Cache and Measure](./week2-01-kv-cache.md) + - [Historical Week 2 lesson addresses](./week2-02-benchmark-profile.md) + - [Earlier quantization lesson](./week2-03-quantize-model.md) + - [Earlier fused-kernel lesson](./week2-04-fused-model-kernels.md) + - [Earlier SIMD-prefill lesson](./week2-05-simd-matrix-prefill.md) + - [Earlier bounded-decode lab](./week2-06-operator-lab.md) + - [Earlier Split-K lab](./week2-07-split-k-prefill.md) + - [Earlier optional macOS capture lab](./week2-advanced-profiling.md) - [🚧 Week 3: Build a Mini vLLM](./week3-overview.md) - [🚧 Continuous Batching](./week3-01-continuous-batching.md) - [🚧 Chunked Prefill](./week3-02-chunked-prefill.md) diff --git a/book/src/appendix-performance.md b/book/src/appendix-performance.md index 827b7770..2040bf14 100644 --- a/book/src/appendix-performance.md +++ b/book/src/appendix-performance.md @@ -1,8 +1,10 @@ # 🚧 Appendix: Performance Evidence Ledger -> **Status: Experimental, single-machine evidence.** See the -> [Week 2 verification matrix](./week2-overview.md#verification-status) before -> treating a correctness, integration, or performance result as broader proof. +> **Historical evidence from an earlier full Week 2 course state.** The +> [current Week 2 route](./week2-overview.md) ships Day 1 only. Later +> checkpoint labels, commands, and measured results below describe the older +> source tree; they are not runnable gates or performance results for this +> checkout. This appendix records the measurements that determined the course order. The numbers are not additive promises: after one bottleneck shrinks, every other diff --git a/book/src/course-roadmap.svg b/book/src/course-roadmap.svg index 01034c28..ee2f2329 100644 --- a/book/src/course-roadmap.svg +++ b/book/src/course-roadmap.svg @@ -1,6 +1,6 @@ - Tiny-LLM course roadmap, Week 2 operator off-ramps, and entry paths - The cumulative course implementation proceeds from setup and Week 1 through seven Week 2 days and Week 3 into Week 4. Week 2 Days 1 and 2 establish required state and measurement. Days 3 through 7 preserve course interfaces but allow manual substitution of corresponding MLX operators instead of custom kernels. This per-operator path is different from the full-MLX solution, which bypasses the course stack. Week 4 keeps setup plus Weeks 1 through 3 as course prerequisites, while the deterministic scripted-model tests for Days 1 through 7 can run after setup without a working serving implementation. Day 8 reconnects the scripted sequence to the Week 3 tokenizer and KV cache. + Historical Tiny-LLM roadmap: earlier seven-day Week 2 operator routes + This historical roadmap showed a former seven-day Week 2 course route from setup and Week 1 through Week 3 into Week 4. In that route, Week 2 Days 1 and 2 established state and measurement. Days 3 through 7 preserved course interfaces but allowed manual substitution of corresponding MLX operators instead of custom kernels. That per-operator path differed from the full-MLX solution, which bypassed the course stack. Week 4 kept setup plus Weeks 1 through 3 as course prerequisites, while deterministic scripted-model tests for Days 1 through 7 could run after setup without a working serving implementation. Day 8 reconnected the scripted sequence to the Week 3 tokenizer and KV cache. Consult the current book navigation for today's Week 2 sequence.
- Tiny-LLM roadmap. The cumulative interface and state path runs from Week 1 through the seven Week 2 days and Week 3 into Week 4. Week 2 Days 3 through 7 show optional MLX operator off-ramps that preserve the course interfaces; they are different from the full-MLX model baseline. Week 4 keeps the course prerequisite of setup plus Weeks 1 through 3, while its deterministic scripted-model tests for Days 1 through 7 can run after setup. Day 8 joins the scripted sequence to the real-model path, and Day 9 continues the Week 4 sequence. -
- -

On a narrow screen, scroll the roadmap -horizontally; when the roadmap is focused, the left and right arrow keys move -through it without changing chapters. Its labels stay at their readable desktop -size.

- - - -Solid arrows in the diagram are **interface and state prerequisites**. They do -not mean that you must hand-write every earlier optimization. A dashed border -marks a custom operator that you may replace locally with its MLX equivalent -while keeping the surrounding course interface. The reference and full-MLX -lanes let you observe a completed system, but they do not fill in unfinished -functions in `src/tiny_llm`. +The current route runs from Week 1 to [Week 2 Day 1](./week2-01-kv-cache.md): +first `kv-cache`, then `capacity-cache`. The later Week 2 checkpoints are +planned, so a Day 1 checkout does not offer a completed Week 2 → Week 3 learner +path. The [earlier full-course roadmap diagram](./course-roadmap.svg) is +retained as historical context; its seven-day Week 2 order is not this +checkout's navigation. + +The reference and full-MLX models are useful controls, but they do not fill +unfinished functions in `src/tiny_llm`. A later custom kernel may have an MLX +operator substitution at the same interface; Day 1's cache state has no such +operator shortcut. | Your goal | Start here | What earlier implementation is required? | | --- | --- | --- | -| Build the whole serving system | Week 1, then follow the solid arrows | Each week uses interfaces and mechanisms established by the previous week. | -| Skip a Week 2 kernel optimization | Keep that day's course interface and wire the corresponding MLX operator at the seam | The earlier model, state, and interface work still needs to exist. This is a local code choice, not a CLI flag. | +| Build the currently shipped cache path | Week 1, then Week 2 Day 1 | Implement both cache checkpoints and their matched measurement. | +| Study a later Week 2 kernel | Read its historical page while waiting for the staged lesson | Its commands and checkpoint are not a Day 1 gate. | | Read or experiment with a later week | Open that chapter and use `tiny_llm_ref` | None in your learner tree. Run the supplied reference tests or reference loader. | | Compare with the production-library baseline | Use `--solution mlx` | None, but this runs the full MLX model and bypasses the course implementation. | | Run the Week 4 Days 1–7 deterministic tests before finishing the serving stack | After setup, run the supplied scripted-model tests | The tests do not need a working serving implementation. The course still assumes setup plus Weeks 1–3 before Week 4; follow the Week 4 days in order, and Day 8's real-model bridge needs the Week 3 model/tokenizer/KV-cache boundary. | The cumulative dependencies are deliberate: -- **Week 1 → Week 2:** Week 2 starts from the readable Qwen3 model and replaces - costs one measured mechanism at a time: first the generation algorithm and - KV cache, then quantized and fused kernels. Days 1–2 establish state and a - repeatable measurement; Days 3–5 follow the dominant cost, Day 6 is an - optional operator lab, and Day 7 closes with a conditional schedule decision. +- **Week 1 → Week 2:** Day 1 keeps the readable model, adds request-owned + dense K/V reuse and bounded storage, and compares the same request at both + cache checkpoints. Later packed-weight and custom-kernel work is planned. - **Week 2 → Week 3:** Week 3 selects MLX quantized projections, but it keeps course-owned normalization, activation, cache, attention, paging, batching, and scheduling. This is an explicit operator seam, not “use the MLX model for @@ -116,7 +77,7 @@ The cumulative dependencies are deliberate: harness to the real tokenizer and KV cache, so that checkpoint needs a working Week 3 path. -> **Is Week 2 required for Week 3? The Week 2 interfaces are; every Week 2 +> **Is Week 2 required for Week 3? Its interfaces are; every future > optimization is not.** The current Week 3 starter reuses the Week 2 model > shell, dense-cache contract, packed-weight plumbing, normalization, > activation, attention, and matrix-fragment interfaces. You may preserve @@ -127,45 +88,34 @@ The cumulative dependencies are deliberate: > the entire Week 2 implementation would require a supplied hybrid starting > checkpoint; that checkpoint does not exist today. -### Week 2 operator off-ramps - -Week 2 separates the mechanism you need later from the kernel you are invited -to optimize. If your goal is to continue into Week 3 rather than implement -every Metal kernel, you can make these explicit local substitutions: +### Later Week 2 operator off-ramps -| Week 2 day | Keep in the course stack | Optional MLX substitution | -| --- | --- | --- | -| Days 1–2 | Dense KV-cache state, the Week 2 model boundary, and the matched measurement method | None; these are state and methodology rather than replaceable operators. | -| Day 3 | Packed-weight containers, quantized embedding/model wiring, and the `quantized_linear` interface | Route projections through `mx.quantized_matmul` instead of the custom matrix-vector kernel. | -| Day 4 | The Week 2 norm, position, and activation call sites | Use the corresponding MLX RMSNorm/RoPE operators and an MLX SiLU-based SwiGLU composition instead of the custom fused kernels. | -| Day 5 | The quantized-projection interface and matrix-shaped dispatch boundary | Keep using the Day 3 MLX projection seam instead of implementing the SIMD-matrix schedule. | -| Day 6 (optional) | The dense-cache attention interface and its shape/mask adapter | Use `mx.fast.scaled_dot_product_attention` instead of the supplied bounded decode-attention branch. | -| Day 7 | The Day 5 unsplit projection fallback and measured dispatch boundary | Keep the unsplit path rather than implementing Split-K where your measurement does not support it. | - -Only the quantized-projection seam is already selected by canonical Week 3. -The Day 4 and optional Day 6 alternatives require you to wire the MLX call at the -existing course interface; there is no `--use-mlx-for-day` command. These -off-ramps let you study later mechanisms, but they do not complete the skipped -day's custom-kernel exercises, implementation-specific tests, or performance -claims. +Day 1 requires cache state and matched measurement; it has no replaceable +custom kernel. The earlier full-course book describes optional MLX substitutions +for later operators, but those are not shipped Day 1 checkpoints. Their +mechanisms and old addresses remain in the +[historical Week 2 pages](./week2-02-benchmark-profile.md). Selecting +`--solution mlx` runs a separate complete model, not a hybrid that completes +learner cache TODOs. To run a completed checkpoint without solving it first: ```bash -# Build the supplied reference extension once after setup or a clean checkout. -pdm run build-ext-ref - -# Run one supplied reference test group. -pdm run test-refsol --week 3 --day 1 +# Run the Day 1 reference tests after setup. +pdm run test-refsol --week 2 --day 1 # Run a completed course model. -pdm run main --solution ref --loader week3 +pdm run main --solution tiny_llm_ref --loader week2 --week2-checkpoint kv-cache # Run the separate full-MLX baseline. pdm run main --solution mlx ``` -`--solution ref` runs the supplied implementation end to end. `--solution mlx` +The Day 1 reference tests do not require `pdm run build-ext-ref`. That +reference-native build is deferred to Day 2; its current W4 Metal compilation +issue should not block the Python cache checkpoints here. + +`--solution tiny_llm_ref` runs the supplied implementation end to end. `--solution mlx` runs MLX end to end. Neither command composes “earlier weeks from the reference or MLX, this week's TODOs from my learner tree.” Per-operator substitution is a manual code edit that preserves the course interface; it is not a third @@ -183,10 +133,10 @@ default batch settings. | Unified memory | Week 1 | Week 2 | Week 3 | Week 4 | | --- | --- | --- | --- | --- | -| 8 GB | 0.6B / 0.6B | 0.6B / 1.7B[^week2-dense] | 0.6B / 1.7B | 0.6B / 1.7B | -| 16 GB | 0.6B / 1.7B | 4B / 8B[^week2-dense] | 4B / 8B | 4B / 8B | -| 18 GB | 0.6B / 1.7B | 4B / 8B[^week2-dense] | 4B / 8B | 4B / 8B | -| 24 GB | 0.6B / 1.7B | 4B / 8B[^week2-dense] | 4B / 8B | 4B / 8B | +| 8 GB | 0.6B / 0.6B | 0.6B / 0.6B[^week2-dense] | 0.6B / 1.7B | 0.6B / 1.7B | +| 16 GB | 0.6B / 1.7B | 0.6B / 1.7B[^week2-dense] | 4B / 8B | 4B / 8B | +| 18 GB | 0.6B / 1.7B | 0.6B / 1.7B[^week2-dense] | 4B / 8B | 4B / 8B | +| 24 GB | 0.6B / 1.7B | 0.6B / 1.7B[^week2-dense] | 4B / 8B | 4B / 8B | | 32 GB | 4B / 8B | 4B / 8B | 4B / 30B-A3B[^moe] | 4B / 30B-A3B[^moe] | | 36 GB | 4B / 8B | 4B / 8B | 4B / 30B-A3B[^moe] | 4B / 30B-A3B[^moe] | | 48 GB | 4B / 8B | 4B / 8B | 4B / 30B-A3B[^moe] | 4B / 30B-A3B[^moe] | @@ -194,9 +144,9 @@ default batch settings. Week 1 reads an official 4-bit checkpoint but materializes its linear and embedding weights in BF16. On an 8 GB Mac, keep the required path at 0.6B. On a 16–24 GB Mac, use 0.6B for the required work and treat 1.7B as an upper-end experiment. -Week 2 Days 1–2 -retain that dense BF16 model; Day 3 keeps weights packed for the quantized-matvec checkpoint. Weeks 3 -and 4 inherit that packed path. More memory still helps after reaching the largest +Week 2 Day 1 retains that dense BF16 model. A later lesson will keep +weights packed for the quantized-matvec checkpoint; the Week 3 and 4 paths +expect that later interface. More memory still helps after reaching the largest supported model because prompt length, batch size, KV caches, compilation, macOS, and other applications all share the same pool. These ceilings are therefore planning guidance, not a guarantee that every workload will avoid memory pressure. @@ -206,8 +156,8 @@ pressure. [M4 Mac mini](https://support.apple.com/en-us/121555), and [M5 MacBook Air](https://support.apple.com/en-us/126320) specifications. Higher-memory configurations are outside this table. -[^week2-dense]: Week 2 Days 1–2 use the dense Week 1 loader, so keep using the Week 1 recommendation - until the packed quantized-matvec path is complete on Day 3. The larger Week 2 entries apply after that checkpoint. +[^week2-dense]: Week 2 Day 1 uses dense BF16 weights and the same model-size + guidance as Week 1. Larger packed-model options belong to a later lesson. [^moe]: 30B-A3B requires the optional Week 3 MoE implementation. In Week 4, select the Week 3 loader. Use batch size one and a short context when approaching this ceiling; 4B remains the required-course target. diff --git a/book/src/week1-07-sampling-prepare.md b/book/src/week1-07-sampling-prepare.md index c8c19aeb..c84a0d66 100644 --- a/book/src/week1-07-sampling-prepare.md +++ b/book/src/week1-07-sampling-prepare.md @@ -95,10 +95,12 @@ Larger models remain optional and are not a Day 7 completion gate. ## Task 2: Prepare for Week 2 -Week 2 Days 1 and 2 introduce KV caching in Python, so you can begin them before -the custom-extension toolchain is ready. Starting on Day 3, the C++ and Metal -work requires full Xcode, its command-line tools, the Metal compiler, and CMake -3.27 or newer. +Day 1's KV-cache implementation is in Python, but its supplied tests import +the Week 2 native extension when they collect. Prepare and build that extension +before the first Day 1 test. This toolchain requires full Xcode, its +command-line tools, the Metal compiler, and CMake 3.27 or newer. A later +packed-W4 lesson will use Metal for the work itself; that lesson is not part +of the current Day 1 route. 1. **Install Xcode:** @@ -177,13 +179,14 @@ The other exported extension names are fail-closed starter stubs labeled with the Week 2 or Week 3 checkpoint that implements them; this setup check calls only `axpby`. -If you are new to C++ or Metal, try a few small exercises before the custom -kernel work on Day 3. For example, implement element-wise operations such as +If you are new to C++ or Metal, try a few small exercises before the later +custom-kernel lessons. For example, implement element-wise operations such as `exp`, `sin`, and `cos`, then use them in place of the corresponding MLX operations in your model implementation. That completes Week 1: you now have a single-request Python inference loop that loads Qwen3, computes logits, samples tokens, and streams a response. Week 2 -first adds KV caching in Python, then begins the custom Metal kernel path. +first adds KV caching in Python with the extension built for test collection. +The custom Metal kernel path follows in a later release. {{#include copyright.md}} diff --git a/book/src/week2-01-kv-cache.md b/book/src/week2-01-kv-cache.md index d17dbb4d..3dccdd1e 100644 --- a/book/src/week2-01-kv-cache.md +++ b/book/src/week2-01-kv-cache.md @@ -1,7 +1,8 @@ -# 🚧 Week 2 Day 1: KV Cache +# 🚧 Week 2 Day 1: Reuse the Prefix, Then Bound the Cache Your Week 1 Qwen model already generates by rerunning the full prefix. Day 1 -keeps that path intact while you complete four separate Week 2 shells: +keeps that path intact while you first complete the `kv-cache` checkpoint, then +bound the request-owned storage for `capacity-cache`. The four initial shells are: - `src/tiny_llm/kv_cache.py::TinyKvFullCache` stores one layer's dense K/V; - `src/tiny_llm/qwen3_week2.py::Qwen3ModelWeek2` threads cache state and @@ -11,15 +12,20 @@ keeps that path intact while you complete four separate Week 2 shells: Together, these pieces make prefill populate the cache and make decode send only the new token. The starter already supplies the Week 1 operators and the -model-loading boundary. Start with the focused learner gate: +model-loading boundary. Complete the Week 1 Day 7 toolchain setup first. The +Day 1 tests import the native Week 2 extension even though this first cache +change is in Python, so build it before running the focused learner gate: ```bash -pdm run test --week 2 --day 1 +pdm run build-ext +pdm run test --week 2 --day 1 -- -k 'not capacity' ``` -When it passes, run the `kv-cache` checkpoint shown in Task 4. That live call -puts the cache into the generation loop instead of exercising it only as an -isolated data structure. +The three selected tests cover the initial `kv-cache` work, not the later +capacity checkpoint. They may fail at the learner-owned seams until you finish +Tasks 1–3. Rerun this focused gate after connecting the serving loop in Task +4, then run the `kv-cache` product command there. The whole-day gate follows +the capacity checkpoint below. Each attention layer can then reuse the keys and values from previous tokens instead of recomputing the entire prefix at every step. @@ -72,7 +78,7 @@ Assume that each attention head has dimension `D = 4`: ``` L = 3 -Q x K^T = +Q x K^T = 1 1 1 1 1 2 3 1x1 -inf -inf 2 2 2 2 1 2 3 2x1 2x2 -inf 3 3 3 3 1 2 3 3x1 3x2 3x3 @@ -102,17 +108,17 @@ K in cache: [a b c d] represent cached values L = 1, S = 3 -Q x K^T = +Q x K^T = (⬇️ is K not transposed) - [1 1 1 1] - [2 2 2 2] + [1 1 1 1] + [2 2 2 2] 3 3 3 3 3 3 3 3 3x1 3x2 3x3 L = 1, S = 4 -Q x K^T = +Q x K^T = (⬇️ is K not transposed) - [1 1 1 1] - [2 2 2 2] + [1 1 1 1] + [2 2 2 2] [3 3 3 3] 4 4 4 4 4 4 4 4 4x1 4x2 4x3 4x4 ``` @@ -161,8 +167,9 @@ Keep this first cache deliberately simple and dense. Each `mx.concat` allocates a larger buffer and copies the previous K/V contents. Across a token-by-token decode of length `S`, those copies add up to `O(S²)` bytes even though the cache avoids `O(S²)` prefix recomputation. The reference cache records that traffic -as `growth_copy_bytes` so the profiler can separate it from attention. Week 3 -replaces repeated concatenation with preallocated pages for serving. +as `growth_copy_bytes` so the profiler can separate it from attention. The +second checkpoint in this lesson replaces repeated concatenation with a +request-bounded allocation. Week 3 later introduces pages for serving. ## Task 2: Build the Cached Week 2 Model @@ -224,8 +231,8 @@ Implement `create_kv_cache` so each request receives one cache handle per Transformer layer. Pass the matching cache through each block, and keep the caller's offset equal to the cache's logical length. -The Day 1 test checks this request-scoped lifecycle together with the cache and -model work from the earlier tasks. +The first Day 1 gate checks this request-scoped lifecycle together with the +cache and model work from the earlier tasks. ## Task 4: Connect the Serving Loop @@ -252,37 +259,349 @@ You can test your solution with: ```bash pdm run main --solution tiny_llm --loader week2 \ - --week2-checkpoint kv-cache --model qwen3-4b + --week2-checkpoint kv-cache --model qwen3-0.6b --max-tokens 16 ``` You can also run the same loop with the reference solution: ```bash pdm run main --solution tiny_llm_ref --loader week2 \ - --week2-checkpoint kv-cache --model qwen3-4b + --week2-checkpoint kv-cache --model qwen3-0.6b --max-tokens 16 +``` + +## Measure the First Cache Change + +Before adding capacity, compare the Week 1 full-prefix loop with this +`kv-cache` checkpoint. The progression runner starts each variant in a fresh +process, uses the cached Qwen3-0.6B model with the same 128-token prompt and +129-token output for both variants, and records the workload and results in +JSON: + +```bash +pdm run bench-week2-progression --offline --solution tiny_llm --suite week2 \ + --repeats 2 --variant week1 --variant week2-kv-cache \ + --model qwen3-0.6b --input-len 128 --output-len 129 --warmup 2 \ + --prefill-logits all --json-output week2-day1-cache.json +``` + +Keep `week2-day1-cache.json` as the baseline for the capacity checkpoint +below. Week 1 requires `--prefill-logits all`; this is a matched algorithm +comparison, not a serving-only prefill measurement. Record the observation +and workload identity without assuming its speedup holds on another machine. + + +## Second checkpoint: bound the request cache + +The dense cache has removed repeated model computation, but its concatenation +still copies the old K/V prefix on each append. Physical capacity and logical +length are different: only the logical prefix is visible to attention. The +`capacity-cache` checkpoint keeps the same serving loop and readable operators. +It allocates storage from the request's prompt-plus-output bound, writes new +K/V into a slice, and returns a view of the active prefix. Capacity can raise +peak memory for some shapes while removing repeated prefix copies. + +## First Diagnostic: Hide Unused Capacity + +```bash +pdm run test --week 2 --day 1 -- -k logical_prefix +``` + +The expected first failure points at `_logical_key_values` or the capacity +branch of `update_and_fetch`. For physical arrays shaped `B, H, capacity, D`, +attention must see only `:offset`: + +```text +physical storage: [token 0][token 1][unused][unused] +logical prefix: [token 0][token 1] +offset = 2, capacity = 4 +``` + +Do not infer logical length from the backing array's shape. + +## Allocate from the Request Bound + +The generation loop knows the prompt length and maximum number of new tokens. +Use that request-local bound when it creates each layer cache. Capacity is not a +global maximum and must not grow beyond the request's declared budget. + +On the first append, allocate K/V storage for the full capacity. On every +append: + +1. compute `end = offset + L_new`; +2. reject `end > capacity` before changing storage, offset, or counters; +3. use `mx.slice_update` on sequence axis 2; +4. advance `offset` only after the write is valid; +5. return `storage[:, :, :offset, :]` for both K and V. + +That ordering makes overflow transactional. A rejected append leaves the old +logical cache usable. + +## Make Movement Observable + +The cache exposes three counter categories: + +| Counter | Meaning | Expected capacity behavior | +|---|---|---| +| `logical_copy_bytes` | old logical K/V copied by concatenation | zero | +| `physical_growth_copy_bytes` | old K/V copied while growing storage | zero | +| `slice_write_bytes` | newly written K/V bytes | increases by each append's K/V size | +| `growth_copy_bytes` | legacy total for growth copies | zero | + +Counters are mechanism evidence. They explain which bytes moved; they do not by +themselves establish lower complete-request latency or peak memory. + +## Reset and Rewind Without Leaking a Suffix + +`rewind(n)` shortens the logical length and rejects negative or oversized +rewinds. The next append may reuse the abandoned physical slots, but attention +must not see values beyond the new offset. `reset()` returns the logical cache +to length zero. It clears storage for the unbounded fallback but retains the +request-bounded allocation. All four movement counters remain +lifetime-cumulative across rewind and reset, so measure their deltas when you +need per-request evidence. + +Run the state-transition witness before the product: + +```bash +pdm run test --week 2 --day 1 -- -k 'rewind or overflow' +``` + +Test the sequence append → rewind → append as well as an overflow after valid +data. Those cases catch implementations that expose physical capacity as +logical state or mutate before validation. + +## Complete the `capacity-cache` Checkpoint + +```bash +pdm run test --week 2 --day 1 +pdm run main --solution tiny_llm --loader week2 \ + --week2-checkpoint capacity-cache --model qwen3-0.6b --max-tokens 16 +``` + +The predecessor fallback is `kv-cache`: it keeps the same generation algorithm +and readable model but uses concatenation. If bounded allocation cannot be +established, return to that checkpoint rather than exposing unused storage. + + +## Compare the two cache checkpoints + +After the focused gates, use identical prompts and output bounds for the two +public checkpoints. These coarse commands include process startup; use the +supplied progression runner below when you need separated prefill/decode +measurements and fresh-process repeats. + +```bash +/usr/bin/time -p pdm run main --solution tiny_llm --loader week2 \ + --week2-checkpoint kv-cache --model qwen3-0.6b --max-tokens 16 +/usr/bin/time -p pdm run main --solution tiny_llm --loader week2 \ + --week2-checkpoint capacity-cache --model qwen3-0.6b --max-tokens 16 +``` + +The cache counters distinguish old logical-prefix copies, physical growth +copies, and new slice writes. A counter difference establishes the mechanism; +it does not alone prove a complete-request speedup. A historical exact-mechanism +run observed a **+88.0 MiB / +2.276%** temporal 2K/512 peak-memory tradeoff. +That is historical evidence, not a result from this checkout. + +Extend the first JSON baseline with the new capacity checkpoint using the +same model, lengths, warmups, and `all`-logit workload. Save a second JSON +file so the original Week 1 versus `kv-cache` observation remains available: + +```bash +pdm run bench-week2-progression --offline --solution tiny_llm --suite week2 \ + --repeats 2 --variant week1 --variant week2-kv-cache \ + --variant week2-capacity-cache --model qwen3-0.6b \ + --input-len 128 --output-len 129 --warmup 2 \ + --prefill-logits all --json-output week2-day1-cache-ladder.json +``` + +Compare these rows with `week2-day1-cache.json` only when the recorded +workload and device match. The serving comparisons below use +`--prefill-logits last`, so keep their results separate from this ladder. + +## Benchmark the Cached Model + +Before changing the model, make the comparison trustworthy. Prefill processes +many prompt tokens at once, while decode usually processes one token per +request. At this checkpoint, decode repeatedly reads dense BF16 projection +weights. Because a change can help one phase while hurting the other, +`benches/bench.py` reports them separately: + +- prefill tokens per second: prompt tokens divided by prefill time; +- decode tokens per second: generated tokens after the first token divided by + decode time. + +The first generated token is part of prefill. Leaving it out of decode keeps +prompt length from distorting the decode number. + +Decide what prefill should return before comparing implementations. Prompt +scoring needs logits for every position; serving needs only the final prompt +logit. Use `--prefill-logits all` for the former and +`--prefill-logits last` for the latter. The runner applies one choice to your +solution and MLX alike, so the two rows do the same work. + +Keep the Week 2 generation algorithm matched too. Both sides use a KV cache: +prefill the prompt once, then pass only the newly generated token on each +decode step. A cached MLX baseline against a full-prefix solution would compare +two different algorithms instead of locating the next optimization target. + +### Record a Matched Baseline + +Use the same model, prompt length, output length, device, and warmup count for +your solution and MLX. The required baseline uses the 0.6B model downloaded +in setup: + +```bash +pdm run bench --solution tiny_llm --loader week2 \ + --week2-checkpoint capacity-cache --model qwen3-0.6b \ + --num-seqs 1 --min-input-len 128 --max-input-len 128 \ + --min-output-len 65 --max-output-len 65 --warmup 2 \ + --prefill-logits last + +pdm run bench --solution mlx --loader week2 --model qwen3-0.6b \ + --num-seqs 1 --min-input-len 128 --max-input-len 128 \ + --min-output-len 65 --max-output-len 65 --warmup 2 \ + --prefill-logits last ``` -## Integrate and Measure +Use `--solution tiny_llm_ref` with the same arguments when you want to compare +your solution with the reference solution instead of MLX. -Finish Day 1 with a matched Week 1 versus cached Week 2 observation. The runner -uses fresh processes, applies the same Qwen3-4B 128×129 workload to both rows, -and writes the configuration beside the result: +Or run the cumulative ladder in fresh processes: ```bash -pdm run bench-week2-progression --offline --solution tiny_llm --repeats 2 \ - --variant week1 --variant week2-kv-cache \ - --model qwen3-4b --input-len 128 --output-len 129 --warmup 2 \ - --json-output week2-day1-cache.json +pdm run bench-week2-progression --offline --repeats 2 \ + --solution tiny_llm --suite week2 \ + --variant week2-capacity-cache --variant mlx \ + --model qwen3-0.6b --input-len 128 --output-len 129 --warmup 2 \ + --prefill-logits last --json-output week2-day1-baseline.json +``` + +Benchmark on an otherwise idle machine. Stop other CPU- and GPU-intensive +workloads, keep power mode and ambient conditions fixed, and wait for a stable +temperature before comparing runs. Repeat each command, report the median, and +record the hardware, MLX and mlx-lm versions, prefill-logit mode, and exact +model. After a dependency upgrade, remeasure MLX instead of carrying the old +baseline forward. + +### Synchronize Lazy Work + +MLX builds computation graphs lazily. Timing only the Python call measures +graph construction instead of GPU execution, so every timed iteration must +evaluate its output: + +```python +start = perf_counter() +output = function() +mx.eval(output) +elapsed = perf_counter() - start ``` -Keep this JSON as Day 2's baseline. Its useful result is the matched observation -and recorded workload identity, not a speedup claim for another model, prompt -length, output length, or device. +The benchmark must also call the cache release hook after warmups and timed +runs. That lets caches return owned or shared resources even when a run fails; +the supplied benchmark-lifecycle test covers both paths. + +## Attribute the Cached Model + +Next, attribute the same cached-decode workload. Keep the learner solution, +model, decode phase, and 128-token context fixed: + +```bash +pdm run profile-week2-kernels --solution tiny_llm --model qwen3-0.6b \ + --case capacity-cache:decode:128 --warmup 4 --iterations 12 \ + --json-output week2-day1-attribution.json +``` + +The result identifies its source, checkpoint, phase, token count, prompt rule, +software, host, category medians, and category shares without depending on a +private function name or Metal symbol. An earlier checked M4 Pro run attributed +81.5% of cached-decode time to dense projections at the then-current `kv-cache` +checkpoint. This is a +historical example, not a measurement of your bounded-cache checkout; another +device or shape may point elsewhere. + +Turn the observation into a decision with three sentences: + +1. “Dense projections dominate this exact cached-decode workload.” +2. “Packing W4 weights and changing only the selected projection path should + reduce that category and improve matched decode.” +3. “I will reject or revise the hypothesis if projection time does not fall or + complete-model decode regresses under the same workload.” + +Substitute the category you observed for the checked example. Your required +work ends with the benchmark, attribution, and decision record. The +[earlier macOS 27 capture lab](./week2-advanced-profiling.md) is historical; no trace, +`gpudebug` output, screenshot, or device-specific counter gates the W4 lesson. + +For an optional deeper run with Qwen3-4B, use a device with enough unified +memory for the dense BF16 Week 2 model (the [model table](./preface.md#choose-a-model-for-your-mac) +recommends at least 32 GB), and download/cache it before using `--offline`: + +```bash +hf download Qwen/Qwen3-4B-MLX-4bit +``` + +Then rerun the matched progression, baseline, and attribution commands with +`--model qwen3-4b` on every compared row. Keep their JSON and conclusions +separate from the required 0.6B observations; model size changes the workload. + +## Why Quantize: The Decode Roofline + +The measurement now has a hardware reason to test. LLM decode is typically +**memory-bandwidth bound**: each token reads the model's weights while doing +relatively little work with them. Use the dimensions in the official +[Qwen3-4B configuration](https://huggingface.co/Qwen/Qwen3-4B/blob/main/config.json) +to calculate an illustrative 4B ideal bound. This is a model-size calculation, +not the measured roofline of your required 0.6B run: + +```plain +Qwen3-4B dimensions: + hidden size h = 2,560 + MLP size i = 9,728 + query width q = 4,096 + key/value width kv = 1,024 + layers L = 36 + vocabulary V = 151,936 + +Projection weights per layer: + Q and O: 2 × h × q = 20,971,520 + K and V: 2 × h × kv = 5,242,880 + MLP: 3 × h × i = 74,711,040 + total per layer = 100,925,440 + +All transformer layers: L × 100,925,440 = 3,633,315,840 +Tied vocabulary head: V × h = 388,956,160 +Total streamed weights: 4,022,272,000 + +FLOPs per token: 2 × 4,022,272,000 = 8.045 GFLOPs +``` + +Count the tied embedding matrix once as the vocabulary projection. The +single-row embedding lookup, normalization weights, activations, KV reads, and +attention work are omitted, so the result is an upper bound for linear layers +rather than a prediction of complete-model throughput. A dense FP16 or BF16 +weight occupies two bytes: + +```plain +4,022,272,000 weights × 2 bytes = 8.045 GB per token +arithmetic intensity = 8.045 GFLOPs / 8.045 GB = 1.0 FLOP/byte +``` -Day 1 changes the generation algorithm by removing full-prefix recomputation, -so measure it with the end-to-end benchmark rather than inventing a -shader-level limiter from a GPU trace. On Day 2, attribute this exact cached -workload and turn the observation into a falsifiable next change. Begin Day 3 -only after that evidence names dense projections. +FP16 and BF16 divide their 16 bits differently: FP16 gives more bits to the +significand, while BF16 gives more bits to the exponent. That affects numerical +range and precision, but not this bandwidth calculation. The course uses BF16 +for activations and outputs. + +| Dense weight format | Bits per weight | Bytes per weight | Streamed weight bytes per token | Weight arithmetic intensity | +|---|---:|---:|---:|---:| +| FP16 | 16 | 2 | 8.045 GB | 1.0 FLOP/byte | +| BF16 | 16 | 2 | 8.045 GB | 1.0 FLOP/byte | + +This is the baseline to improve: both dense formats must stream roughly 8 GB +of projection weights to generate one token. Save the matched benchmark result, +and keep it for comparison when the packed-W4 Day 2 lesson ships. The +[earlier quantization lesson](./week2-03-quantize-model.md) preserves that +mechanism, but its old checkpoint commands are historical in this Day 1 +checkout. {{#include copyright.md}} diff --git a/book/src/week2-02-benchmark-profile.md b/book/src/week2-02-benchmark-profile.md index bd4309e3..9d6daaf3 100644 --- a/book/src/week2-02-benchmark-profile.md +++ b/book/src/week2-02-benchmark-profile.md @@ -1,4 +1,11 @@ -# 🚧 Week 2 Day 2: Benchmarking and Profiling +# Historical Week 2: Benchmarking And Profiling + +> **Earlier lesson address.** This page preserves the benchmarking and profiling +> explanation and its original links. The current Week 2 learner route +> ships [Day 1: Cache and Measure](./week2-01-kv-cache.md) only. +> Later checkpoints, day numbers, tests, and commands below belong to an +> earlier all-days course state; do not use them as gates for this checkout. + Day 1 leaves you with a cached BF16 model and a working `kv-cache` checkpoint. Day 2 does not add another model operator. The supplied benchmark and portable diff --git a/book/src/week2-03-quantize-model.md b/book/src/week2-03-quantize-model.md index cf88067e..81dec9e3 100644 --- a/book/src/week2-03-quantize-model.md +++ b/book/src/week2-03-quantize-model.md @@ -1,4 +1,11 @@ -# 🚧 Week 2 Day 3: Quantize the Model +# Historical Week 2: Packed W4 Quantization + +> **Earlier lesson address.** This page preserves the packed W4 quantization +> explanation and its original links. The current Week 2 learner route +> ships [Day 1: Cache and Measure](./week2-01-kv-cache.md) only. +> Later checkpoints, day numbers, tests, and commands below belong to an +> earlier all-days course state; do not use them as gates for this checkout. + Day 2 leaves you with a synchronized dense BF16 baseline. The Day 3 starter already supplies the packed-weight container and its @@ -240,9 +247,10 @@ The advertised bandwidths come from Apple's specifications for [M2 Pro and Max](https://www.apple.com/newsroom/2023/01/apple-unveils-m2-pro-and-m2-max-next-generation-chips-for-next-level-workflows/), [M2 Ultra](https://www.apple.com/newsroom/2023/06/apple-introduces-m2-ultra/), [M3 Pro and Max](https://support.apple.com/en-us/117736), -[M3 Ultra](https://www.apple.com/mac-studio/), and -[M4 Pro and Max](https://support.apple.com/en-us/121553). Apple's current Mac -Studio pairs M4 Max with M3 Ultra, so there is no M4 Ultra row. +[M3 Ultra in the 2025 Mac Studio](https://support.apple.com/en-us/122211), and +[M4 Pro and Max](https://support.apple.com/en-us/121553). Apple's 2025 Mac +Studio specifications listed M4 Max and M3 Ultra configurations, but no M4 +Ultra configuration; this historical table therefore has no M4 Ultra row. These values assume peak advertised bandwidth, one read of every projection weight, and no other traffic or work. A complete model also reads activations diff --git a/book/src/week2-04-fused-model-kernels.md b/book/src/week2-04-fused-model-kernels.md index bf6f7d9b..6fd13b68 100644 --- a/book/src/week2-04-fused-model-kernels.md +++ b/book/src/week2-04-fused-model-kernels.md @@ -1,4 +1,11 @@ -# 🚧 Week 2 Day 4: Fused Model Kernels +# Historical Week 2: Fused Model Kernels + +> **Earlier lesson address.** This page preserves the fused model kernels +> explanation and its original links. The current Week 2 learner route +> ships [Day 1: Cache and Measure](./week2-01-kv-cache.md) only. +> Later checkpoints, day numbers, tests, and commands below belong to an +> earlier all-days course state; do not use them as gates for this checkout. + Day 3 leaves the cached model using packed projections. Day 4 keeps the Week 1 Python equations as readable oracles and completes three separate extension diff --git a/book/src/week2-05-simd-matrix-prefill.md b/book/src/week2-05-simd-matrix-prefill.md index 6ad1b78d..96e6a1c8 100644 --- a/book/src/week2-05-simd-matrix-prefill.md +++ b/book/src/week2-05-simd-matrix-prefill.md @@ -1,4 +1,11 @@ -# 🚧 Week 2 Day 5: SIMD-Matrix Prefill +# Historical Week 2: Simd Matrix Prefill + +> **Earlier lesson address.** This page preserves the SIMD matrix prefill +> explanation and its original links. The current Week 2 learner route +> ships [Day 1: Cache and Measure](./week2-01-kv-cache.md) only. +> Later checkpoints, day numbers, tests, and commands below belong to an +> earlier all-days course state; do not use them as gates for this checkout. + Day 4 ends with a decision, not a predetermined kernel. Re-profile the fixed 128-token prefill and name the dominant category before changing code. On the diff --git a/book/src/week2-06-operator-lab.md b/book/src/week2-06-operator-lab.md index 1f4724a7..e4b86b66 100644 --- a/book/src/week2-06-operator-lab.md +++ b/book/src/week2-06-operator-lab.md @@ -1,4 +1,11 @@ -# 🚧 Week 2 Day 6 (Optional): Workload-Conditioned Operator Lab +# Historical Week 2: Bounded Decode-Attention Operator Lab + +> **Earlier lesson address.** This page preserves the bounded decode-attention operator lab +> explanation and its original links. The current Week 2 learner route +> ships [Day 1: Cache and Measure](./week2-01-kv-cache.md) only. +> Later checkpoints, day numbers, tests, and commands below belong to an +> earlier all-days course state; do not use them as gates for this checkout. + Day 5 restores the matrix-shaped projection path selected by the fixed 128-token prefill profile. Day 6 asks a different question: can a secondary diff --git a/book/src/week2-07-split-k-prefill.md b/book/src/week2-07-split-k-prefill.md index 50bffd44..7d394db2 100644 --- a/book/src/week2-07-split-k-prefill.md +++ b/book/src/week2-07-split-k-prefill.md @@ -1,4 +1,11 @@ -# 🚧 Week 2 Day 7: Conditional Split-K and Final Decision +# Historical Week 2: Conditional Split-K + +> **Earlier lesson address.** This page preserves the conditional Split-K +> explanation and its original links. The current Week 2 learner route +> ships [Day 1: Cache and Measure](./week2-01-kv-cache.md) only. +> Later checkpoints, day numbers, tests, and commands below belong to an +> earlier all-days course state; do not use them as gates for this checkout. + Day 5 leaves a reusable 32×32×32 SIMD-matrix projection and an exact unsplit fallback. Day 6 is an optional branch and is not inherited here: the `split-k` diff --git a/book/src/week2-advanced-profiling.md b/book/src/week2-advanced-profiling.md index 9bc4f437..eef8df15 100644 --- a/book/src/week2-advanced-profiling.md +++ b/book/src/week2-advanced-profiling.md @@ -1,4 +1,11 @@ -# Optional: Inspect a Week 2 Capture on macOS 27 +# Historical Week 2: Optional Macos Capture + +> **Earlier lesson address.** This page preserves the optional macOS capture +> explanation and its original links. The current Week 2 learner route +> ships [Day 1: Cache and Measure](./week2-01-kv-cache.md) only. +> Later checkpoints, day numbers, tests, and commands below belong to an +> earlier all-days course state; do not use them as gates for this checkout. + The synchronized product benchmark and portable operator-attribution runner are sufficient for every required Week 2 checkpoint. This page is an optional diff --git a/book/src/week2-kernel-profile.svg b/book/src/week2-kernel-profile.svg index c4567089..5b084752 100644 --- a/book/src/week2-kernel-profile.svg +++ b/book/src/week2-kernel-profile.svg @@ -1,6 +1,6 @@ - Week 2 Qwen3-4B kernel-time attribution by checkpoint - Eight stacked bars follow the chronological Week 2 order: cached decode on Day 2, packed W4 on Day 3, fused model kernels on Day 4, the pre-SIMD control immediately before the SIMD prefill rows on Day 5, the optional decode-attention lab on Day 6, and Split-K on Day 7. Each bar normalizes synchronized kernel-group time across projections, attention, normalization position and activation kernels, and KV growth. + Historical seven-day Week 2 Qwen3-4B kernel-time attribution + These eight historical stacked bars followed the former seven-day Week 2 order: cached decode on Day 2, packed W4 on Day 3, fused model kernels on Day 4, the pre-SIMD control immediately before the SIMD prefill rows on Day 5, the optional decode-attention lab on Day 6, and Split-K on Day 7. Each bar normalized synchronized kernel-group time across projections, attention, normalization position and activation kernels, and KV growth. The day labels do not describe the current book sequence.