diff --git a/.obsidian/workspace.json b/.obsidian/workspace.json index 58a8978..23e775d 100644 --- a/.obsidian/workspace.json +++ b/.obsidian/workspace.json @@ -13,12 +13,12 @@ "state": { "type": "markdown", "state": { - "file": "design-docs/DD-002-buffer-pool-manager.md", + "file": "design-docs/DD-004-heap-pages-and-table-heap.md", "mode": "preview", "source": false }, "icon": "lucide-file", - "title": "DD-002-buffer-pool-manager" + "title": "DD-004-heap-pages-and-table-heap" } } ] @@ -188,19 +188,19 @@ }, "active": "213624f19b44c94b", "lastOpenFiles": [ - "build/debug/CMakeFiles/kernsql.dir/link.d", - "build/debug/stt8jRT2", - "build/debug/stTk6Uc9", - "build/debug/CMakeFiles/kernsql_lib.dir/src/buffer/buffer_pool_manager.cpp.o.d", - "src/buffer/buffer_pool_manager.cpp.tmp.16764.7acecd0a7cef", - "src/buffer/buffer_pool_manager.cpp.tmp.16764.6d12f137211c", - "build/debug/stB1ka1L", - "build/debug/stMIkJFI", - "build/debug/stCuaJp3", - "build/debug/stI9cO3C", - "build/debug/CMakeFiles/kernsql_lib.dir/src/buffer/page_guard.cpp.o.d", - "design-docs/DD-003-threading-model.md", + "test/heap/table_heap_test.cpp", + "test/heap/table_heap_test.cpp.tmp.18698.cd124d9f8f82", + "build/debug/lib/stSWVPH6", + "build/debug/lib/stX2r8JG", + "build/debug/lib/st86nMd7", + "build/debug/lib/stOHmJLK", + "build/debug/lib/stl0srJW", + "build/debug/lib/stiFWUPf", + "build/debug/lib/staf5uB5", + "build/debug/lib/stwxl6qH", + "build/debug/_deps/googletest-build/googletest/CMakeFiles/gtest.dir/src/gtest-all.cc.o.d", "design-docs/DD-002-buffer-pool-manager.md", + "design-docs/DD-003-threading-model.md", "design-docs/DD-002-buffer-manager.md", "design-docs/notes/Buffer-Pool-Manager.md", "design-docs.md", diff --git a/CMakeLists.txt b/CMakeLists.txt index 98769b0..9d4b497 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -10,7 +10,9 @@ add_library(kernsql_lib STATIC src/common/logger.cpp src/storage/disk_manager.cpp src/buffer/buffer_pool_manager.cpp - src/buffer/page_guard.cpp) + src/buffer/page_guard.cpp + src/heap/heap_page.cpp + src/heap/table_heap.cpp) target_include_directories(kernsql_lib PUBLIC src) target_compile_options(kernsql_lib PUBLIC -Wall -Wextra -Wpedantic -Wshadow -Wconversion -Werror=format) diff --git a/design-docs/DD-001-storage-file-layout.md b/design-docs/DD-001-storage-file-layout.md index 719f28b..0d3cd7a 100644 --- a/design-docs/DD-001-storage-file-layout.md +++ b/design-docs/DD-001-storage-file-layout.md @@ -51,6 +51,15 @@ single page (the catalog first, but every heap table and index eventually) grows current last page. Page 1 is only the catalog's *first* page, not its full extent — scanning the catalog means following `next_page_id` from page 1 until `INVALID_PAGE`. +> **A page is in at most one chain at a time, and `page_type` decides which chain that is.** +> `FREE` → the next free page in the freelist. `HEAP` or `CATALOG` → the next page of that +> object. `INDEX_LEAF` → the right sibling. There is no page state in which two of those +> readings are simultaneously valid, which is what makes one field safe to share across all of +> them: a page leaves its old chain in the same operation that changes its `page_type`. +> `DeallocatePage` is the concrete case — it stamps `FREE` and threads the page onto the +> freelist together, so the heap link it used to hold is gone by the time anyone can read the +> field as a freelist link. + Index root pages are the one exception to "storage doesn't know about layout": since a B+tree's root page can change across root splits, it cannot be a fixed well-known page like page 1. Each index's current root page id is stored as mutable data in the catalog, not in @@ -64,12 +73,19 @@ offset 0, so the format is uniform regardless of what currently owns the page. | Field | Type | Size | Purpose | |---|---|---|---| | `page_type` | `uint8_t` (enum) | 1 | Distinguishes meta / heap / index-internal / index-leaf / free. Lets any page be sanity-checked against how it's being interpreted. | -| `next_page_id` | `page_id_t` (int32) | 4 | Forward chain link. Meaning depends on `page_type`: next page in a heap/catalog chain, next free page in the freelist, or right-sibling for a B+tree node (leaf range scans and B-link–style crabbing reuse this same field). | +| `flags` | `uint8_t` | 1 | Reserved; always zero. | +| `format_version` | `uint16_t` | 2 | `PAGE_FORMAT_VERSION`. Stamped on every header written; nothing reads it back yet, which is the point — a future format change has a discriminator to branch on. | +| `page_id` | `page_id_t` (int32) | 4 | The page's own id, redundant with the offset it was read from and redundant on purpose: derive identity from the offset alone and a disagreement is undetectable by construction. Catches misdirected writes, off-by-one page arithmetic, torn extends. InnoDB's `FIL_PAGE_OFFSET`. | +| `next_page_id` | `page_id_t` (int32) | 4 | Forward chain link. Which chain is decided by `page_type` — see the invariant above. | | `prev_page_id` | `page_id_t` (int32) | 4 | Reserved for future backward traversal. Unused today. | -| `page_lsn` | `lsn_t` (uint64) | 8 | LSN of the last WAL record applied to this page. Enables idempotent redo during recovery (skip reapplying a log record if `page_lsn >= record_lsn`) and is the field the buffer pool's write-ahead rule will check before flushing a dirty page. Populated in this branch even though WAL itself lands later, to avoid a page-format break. | - -Total: ~17 bytes, aligned to 24 bytes. Leaves ~4072 bytes of `PAGE_SIZE` for page-type-specific -content (slot directory + tuple data for heap pages, key/pointer arrays for index pages). +| `page_lsn` | `lsn_t` (uint64) | 8 | LSN of the last WAL record applied to this page. Enables idempotent redo during recovery (skip reapplying a log record if `page_lsn >= record_lsn`) and is the field the buffer pool's write-ahead rule will check before flushing a dirty page. Reserved even though no WAL is planned, to avoid a page-format break. | +| `checksum` | `uint32_t` | 4 | Reserved. See Non-goals. | +| `reserved` | `uint32_t` | 4 | Padding, explicit rather than implicit so `ReadFrom`/`WriteTo` move a fully-defined 32 bytes. | + +Total: exactly **32 bytes** (`PAGE_HEADER_SIZE`), asserted in `page_header.hpp`. Field order is +chosen so every member lands on its natural alignment with no implicit padding. That leaves +**4064 bytes** (`PAGE_BODY_SIZE`) for page-type-specific content — and 4064, not 4096, is what a +page guard hands out, so no caller above the buffer pool can name the header bytes as raw memory. For heap pages specifically, that remaining space is a slotted page: a slot directory grows forward from right after the header, tuple data grows backward from the end of the page, and diff --git a/design-docs/DD-004-heap-pages-and-table-heap.md b/design-docs/DD-004-heap-pages-and-table-heap.md new file mode 100644 index 0000000..8c7cf3d --- /dev/null +++ b/design-docs/DD-004-heap-pages-and-table-heap.md @@ -0,0 +1,401 @@ +# Heap Pages & Table Heap + +**Status:** Accepted +**Component:** `src/storage` (`HeapPage`, `TableHeap`, `TableIterator`) + +Where table rows actually live. A heap is an unordered pile of rows addressed by +`RID{page_id, slot}`; the B+tree will store RIDs as its payload, and the executor reads rows +through them. Sits on `BufferPoolManager` ([DD-002](./DD-002-buffer-pool-manager.md)) and +inherits the page format from [DD-001](./DD-001-storage-file-layout.md). + +Deliberately shorter than DD-002. That doc is long because the buffer pool invents a latching +protocol; this layer invents none — it operates on bytes inside a latch someone else already +holds. The only sections here that carry real weight are the layout, the RID contract, the one +place two threads can collide (page extension), and the rule that stops a scan from fighting the +writes it triggers. + +## The rule everything follows + +> **Slots never move. Tuple bytes move freely.** + +A `RID` names a slot, and the slot names an offset. That indirection is the entire reason the +layout exists: compaction can rewrite every byte in the tuple region and every RID in the +database stays valid, because the slot array did not move and the slot's *index* is what was +handed out. Postgres calls these line pointers (`ItemIdData`); the concept is universal. + +The corollary is the thing to keep testing: **any operation that moves tuple bytes must update +the moved tuples' slot offsets in the same latched operation.** There is no window in which a +slot's offset is stale, because there is no window in which another thread can look. + +## Page layout + +Everything below is inside the 4064-byte body a page guard hands out. Offsets are +**body-relative** — offset 0 is the first byte after the 32-byte `PageHeader`. That is the only +origin any code in this layer uses, and it keeps every offset under 4064, comfortably inside a +`uint16_t`. + +``` +body offset 0 ++---------------------------+ +| heap sub-header (8 bytes) | ++---------------------------+ +| slot[0] slot[1] ... | grows FORWARD, 4 bytes each ++---------------------------+ +| | +| free space | insert fails when this closes +| | ++---------------------------+ +| ... tuple tuple tuple | grows BACKWARD from the end ++---------------------------+ +body offset 4064 +``` + +They grow toward each other because the slot-to-tuple ratio is not knowable at format time: one +page may hold three fat rows, another two hundred thin ones. A fixed-size slot array would have +to be sized for the worst case and waste the difference on every page. + +### Heap sub-header (8 bytes) + +| Field | Type | Purpose | +|---|---|---| +| `slot_count` | `uint16_t` | Number of entries in the slot array, live and dead alike. Only ever grows within a page's life as a heap page. | +| `tuple_data_start` | `uint16_t` | Body offset of the lowest occupied tuple byte. The tuple region is `[tuple_data_start, 4064)`. Initialised to 4064 for an empty page. | +| `live_count` | `uint16_t` | Non-dead slots. Maintained rather than derived so a scan can skip an empty page and stats do not cost a slot walk. | +| `dead_bytes` | `uint16_t` | Bytes in the tuple region that no live slot points at: deleted tuples, plus the waste left behind by an in-place shrinking update. Reset to zero by `Compact`, and reduced by nothing else. InnoDB's `PAGE_GARBAGE`. | + +### Slot (4 bytes) + +| Field | Type | Purpose | +|---|---|---| +| `offset` | `uint16_t` | Body offset of the tuple's first byte. **`0` means the slot is dead**, see below. | +| `length` | `uint16_t` | Tuple length in bytes. | + +`offset == 0` is a safe sentinel rather than a magic number: body offset 0 is inside the heap +sub-header, so no tuple can ever legitimately start there. Postgres spends a flags bitfield on +the same distinction; the impossible offset gets it for free. + +A slot is in one of two states — **live** (`offset != 0`) or **dead** (`offset == 0`). There is +no third "never used" state: a slot is created live by an insert, and a delete makes it dead. + +### Free space + +Two different numbers, and confusing them is the bug this section exists to prevent. + +``` +slot_array_end = HEAP_HEADER_SIZE + slot_count * SLOT_SIZE +contiguous = tuple_data_start - slot_array_end // usable right now +reclaimable = contiguous + dead_bytes // usable after Compact +``` + +An insert of `L` bytes needs `L + SLOT_SIZE` if it must append a slot, or `L` alone if it reuses +a dead one. `HeapPage::Insert` checks that against `contiguous` and fails if it does not fit; the +caller compares the same requirement against `reclaimable` to decide whether compacting is worth +a retry. + +`reclaimable` is also the number any cross-page free-space mechanism has to publish, which is the +reason `dead_bytes` is a maintained field rather than something derived on demand. It cannot be +derived: `Update` case 1 overwrites a slot's `length` with the smaller value, so the bytes it +orphans are referenced by no slot and no offset and are invisible to any later slot walk. Without +the counter, a page that has taken many in-place shrinks looks full while being mostly holes, and +nothing ever learns there is space to recover. + +Largest possible tuple on an empty page: `4064 - 8 - 4 = 4052` bytes. + +### The accounting invariant + +Every byte of the tuple region belongs to exactly one live tuple or to nobody, which fixes a +total that must hold after **every** operation on the page: + +``` +tuple_data_start + sum(slot.length for live slots) + dead_bytes == PAGE_BODY_SIZE +``` + +This is worth a debug-only `Validate()` alongside `live_count == count of live slots` and +`slot_array_end <= tuple_data_start`, called at the end of every mutating operation in test +builds. It catches the entire class of accounting bug this layer can have — a missed increment in +`Update` case 1, a `dead_bytes` wrongly reduced by a dead-slot reuse, a `Compact` that forgets to +reset — at the operation that caused it rather than a thousand operations later. + +`dead_bytes` is bounded by 4052 and therefore cannot overflow its `uint16_t`, and no operation +ever subtracts from it, so there is no path that can drive it negative. Because it lives in the +same page as the bytes it describes, and a page is the unit of atomicity, it also cannot disagree +with its own page across a crash — which is what allows it to be exact, unlike the cross-page +structures described under [Free space across pages](#free-space-across-pages). + +## The tuple size cap: 2000 bytes + +A row must fit in one page — tuples do not span pages, and there are no overflow chains (see +Non-goals). The cap is **2000 bytes**, enforced twice: at `CREATE TABLE` against the maximum +possible row the schema can produce, and again at `INSERT` against the actual encoded size. + +2000 rather than the 4052 ceiling because 2000 guarantees at least two rows per page +(`2 * (2000 + 4) + 8 = 4016 <= 4064`). That is InnoDB's rule, and the reason for it is that a +format allowing exactly one row per page degenerates: every page carries a full header and a +slot array to hold a single row, and a "heap" becomes a linked list with 50% overhead. Two rows +per page is the weakest guarantee that keeps the structure honest. + +## Operations + +All of these are on `HeapPage`, a **pure view over a `span`**. No buffer +pool, no disk, no guards, no ownership. That is what makes it testable against a bare +`std::array` in microseconds, which is where the offset arithmetic gets +debugged. + +- **`Init`** — zero the sub-header, `slot_count = 0`, `live_count = 0`, `dead_bytes = 0`, + `tuple_data_start = 4064`. The caller stamps `page_type = HEAP` through the guard. +- **`Insert(bytes) -> slot_id`** — reuse a dead slot if one exists, else append. Copy the tuple + to `tuple_data_start - len` and lower `tuple_data_start`. Fails (does not compact) when there + is not enough contiguous free space; the caller decides whether to compact and retry or move + to another page. **Reusing a dead slot does not reduce `dead_bytes`.** It reclaims the four + bytes of the slot, not the tuple bytes: the dead tuple's offset was destroyed by `Delete`, so + its bytes cannot be found, let alone reused. Only `Compact` reclaims them. This reads + backwards and is the one place the field is likely to be got wrong. +- **`Get(slot) -> span`** — bounds-check the slot id, reject a dead slot. **A `Get` on a deleted + RID returns "not found", never bytes.** This is why dead slots stay distinguishable instead of + being removed: an RID handed out before the delete must get an answer, not garbage. +- **`Delete(slot)`** — add the slot's `length` to `dead_bytes` **before** overwriting the slot, + then set it to `{offset: 0, length: 0}` and decrement `live_count`. Zeroing both fields gives a + dead slot one canonical representation, which is what makes `offset == 0 => length == 0` + assertable. The tuple's bytes are *not* moved; the space is reclaimed at the next compaction. +- **`Compact()`** — rewrite the tuple region packed against the end of the body, in descending + offset order, updating each live slot's offset as its tuple moves. Dead slots keep their + index and stay dead. Afterwards `tuple_data_start` is the new low-water mark and + `dead_bytes = 0` — by definition, since every byte below the new low-water mark is now live. + This is the only operation that reduces `dead_bytes`. +- **`Update(slot, bytes)`** — three cases, and the third is the one with consequences: + 1. new length `<=` old: overwrite in place, shrink `length`, and add the difference to + `dead_bytes`. Those orphaned bytes are the ones no slot walk can ever find again, which is + the whole reason the counter exists. + 2. new length `>` old but the page has room after compaction: compact, then relocate the + tuple. **The RID survives**, because the slot index did not change. Sequence this as + *kill the old tuple, compact, insert into the same slot id* — add the old length to + `dead_bytes` and zero the slot first, so compaction drops the old bytes instead of copying + bytes that are about to be abandoned. The slot is momentarily dead while still logically + live; that is safe only because the whole operation runs under one write latch, and it is + the reason `Compact` must be a private helper rather than something a caller can interleave. + 3. it does not fit even after compaction: not `HeapPage`'s problem. `TableHeap` deletes here + and inserts elsewhere, and **the RID changes**. + +## The RID contract + +Stated explicitly, because everything above this layer depends on it and case 3 above is a trap: + +- An RID is stable across **insert, delete of other rows, compaction, and in-page update**. +- An RID is **invalidated by an update that outgrows its page**. `TableHeap::Update` returns the + new RID; the caller owns re-pointing anything that referenced the old one. +- An RID of a deleted row resolves to "not found" **for the life of the page as a heap page**. It + is not reused: a dead slot may be handed to a later insert, so a stale RID can, after enough + churn, resolve to a *different live row*. This is the ABA problem, and it is accepted here + because the only holder of long-lived RIDs will be the B+tree, which is updated in the same + operation that deletes the row. Postgres avoids it with tuple visibility (`xmin`/`xmax`), which + is MVCC machinery this engine deliberately does not have. + +## Table heap + +`TableHeap` owns the `BufferPoolManager&` and the page chain. Pages are linked through +`PageHeader::next_page_id`, per DD-001's chain invariant — a `HEAP` page's `next_page_id` is the +next page of that table, and the guard already exposes `SetNextPageId`. + +The catalog holds each table's `first_page_id` and `last_page_id`. + +### API shape + +Four decisions the prose above does not pin down, recorded here because each one has a reason +that is easy to lose. + +**`Create`/`Open` are factories returning `unique_ptr`.** Both can fail — `Create` allocates a +page — so neither can be a constructor. `unique_ptr` rather than by value because the insert +cursor is an `atomic`, and an atomic member makes the class non-movable. Same shape, +and same reason, as `DiskManager::Open`. + +**`Update` returns the row's NEW `RID`.** Case 3 relocates a row to another page and its identity +changes. Returning `Result` makes every caller hold the new value; returning a `Status` +would let a stale `RID` sit unnoticed in an index entry or a collected list, which is precisely +the failure the RID contract exists to prevent. + +**`Get` returns a copy, never a span.** The guard drops when `Get` returns, the frame becomes +evictable, and a span into the body then reads another page's bytes as valid memory — invisible +to ASan. A buffer-filling overload can be added if per-row allocation ever matters; handing back +a view cannot. + +**`TableIterator` is a cursor, not an STL iterator.** It holds a guard, so it is move-only and +cannot satisfy `forward_iterator`; and advancing fetches a page, so it can fail, which +`operator++` has no way to report. `Result Next()` plus `Rid()` and `Tuple()` states both +facts instead of hiding them behind a sentinel and a throwing increment. `Tuple()` points into +the iterator's own buffer — valid across the guard drop, invalidated by the next `Next()`. + +### Insert and page extension — the one place two threads collide + +Insert takes a write guard on a candidate page and tries. If it fits, done. When every candidate +is full — including the last page, which is where the insert-hint cursor below always falls back +to — the chain must be extended, and two inserters arriving together must not both allocate. + +> **The write latch on the current last page is the extension lock.** + +A thread that finds the last page full **keeps its write guard** while it calls `NewPage`, stamps +`HEAP`, `Init`s the new page, sets the old page's `next_page_id`, and publishes the new +`last_page_id`. A second inserter blocks on that guard; when it acquires it, the page's +`next_page_id` is no longer `INVALID_PAGE`, so it follows the link instead of allocating. No new +lock, no new ordering rule — the latch that was already required does the job. + +Getting this wrong is not subtle in its consequences and is very subtle in its symptoms: both +threads allocate, both set `next_page_id`, one link is overwritten, and one page is permanently +leaked and unreachable — a slow leak with no error anywhere. + +Note this means a `NewPage` (which can block on a disk allocation) happens under a content +latch. That is allowed: DD-002's prohibitions are about the *metadata* mutex and shard locks. +A content latch is held across I/O routinely — that is what the `Loading` state exists for. + +### Free space across pages + +Space freed by a delete is reclaimed *within* its page by the next compaction. Getting it reused +*across* pages is a separate problem, and this is the one part of the layer that is deliberately +unfinished. + +**The bound, stated exactly.** With a pure append-at-the-last-page insert, a table's file size +tracks the **peak** number of rows it has ever held, not the number it holds now. The +pathological workload is not exotic: any table at a steady state of churn — a queue, a session +table, a nightly load-and-drain staging table — holds a constant row count while growing without +limit. + +The cost that actually decides this is not disk, which is cheap. It is `TableIterator`. Skipping +a page whose `live_count` is zero skips the tuple work but still fetches the page and still +evicts something useful to do it, so a table that peaked at a million rows and now holds a +thousand pays for a million-row scan forever and flushes the pool every time it runs. + +**What rules out the textbook fix.** The obvious mechanism is an intrusive free-page list — a +head in the catalog, links threaded through a page header field, O(1) push on delete and O(1) pop +on insert. That is InnoDB's free/frag list, and it is the wrong shape for this engine, because a +linked list is an *exact* structure: every node must be correct or the structure is broken. +KernSQL has no WAL. A crash leaves dirty pages unflushed in arbitrary order, so a head that +points at a page whose link never reached disk is a dangling pointer into a free page or another +table, with no undo to repair it and nothing to detect it. InnoDB can afford exact free lists +because it has redo logging; Postgres, whose FSM is explicitly allowed to be wrong and is rebuilt +by vacuum, is the better model here for the opposite reason. Note this is unrelated to +visibility: with no `xmin`/`xmax` this engine knows space is free the moment a delete commits, +so it never needs vacuum's hard part — deciding *when* space became free. + +> **The rule for anything that spans pages: it is a hint, never a truth.** Every entry is +> re-verified against the real page before use, so a stale entry costs one wasted fetch rather +> than corruption. That single property is what makes a cross-page structure crash-safe without a +> WAL, and — because staleness is harmless — it also removes any reason to hold two page latches +> at once, which is what keeps this layer free of a lock-ordering rule. + +**v1 — a rotating insert cursor.** `TableHeap` keeps an in-memory `insert_hint_page_id`, set to +`first_page_id` on open. Insert latches the hint page and tries; on failure it advances the hint +along `next_page_id` (wrapping at the end) for a bounded K attempts, then falls back to appending +at the last page. Delete is untouched, nothing is persisted, no latch is added, and a restart +merely starts the sweep again. Because the cursor traverses the whole chain over the table's +life, any page that frees up is eventually revisited, and a churning table reaches a stable size. +Its honest weakness is that it stumbles into space rather than finding it: deletes scattered +sparsely across a large table mean the cursor mostly lands on full pages and burns its K probes. + +**v2 — a per-table free-space map, when the cursor stops being enough.** A chained page per +table, `fsm_first_page_id` in the catalog, holding an array of `{page_id: uint32, +free_units: uint16}` — `reclaimable` quantised to 16-byte units, which fits 0–254 and answers +"at least N bytes?" directly for any N, with no size-class buckets. At 6 bytes an entry that is +~670 pages, about 2.7 MB of table, per FSM page, and finding a candidate is a linear scan of one +cached page. Postgres needs a *tree* only because its array is large enough that scanning it +matters; at this scale it is not. It is an array of pairs rather than an array indexed by page +number because pages come from a global freelist and are scattered, so this table has no dense +block numbering to index by. + +Two rules make it safe. The FSM is written **after** the heap page's guard is released, since +staleness is harmless and there is therefore no reason to hold both. And it is **insert-only**: +`TableIterator` keeps walking `next_page_id`, because a scan that trusted a structure allowed to +be wrong would silently drop rows from query results. It can be rebuilt at any time by walking +the chain, which is worth having as an admin operation regardless. + +Neither tier needs an on-disk format change beyond `dead_bytes`, which is already in the +sub-header, and one `page_id_t` in the catalog — which is why v1 ships now and v2 is written down +rather than built. + +### Scanning + +`TableIterator` holds **one page guard at a time** and copies tuples out. It must not hold a +guard across the whole scan: the pool is fixed-size, and a scan that pins a frame per page it +has visited exhausts it. Advance = release the current guard, fetch the next page id, skip pages +whose `live_count` is zero. + +**It yields the RID alongside the tuple, always** — not behind a flag, a mode, or a second +iterator type. The iterator already holds the page id in order to walk the chain and the slot id +as its cursor, so the RID is not something it fetches; it is something it would otherwise throw +away. `SELECT` ignores it. `DELETE ... WHERE` needs it to name the row it matched, and the B+tree +build needs `{key -> RID}` pairs for every row in the table — that pair *is* the index's payload. +A scan that cannot say *which* row it handed back is unusable to the layer above. Postgres does +the same thing unconditionally: a seqscan's `HeapTuple` carries `t_self`, and no caller opts in. + +**The RID is a value; the tuple bytes are not.** A `RID` copied out of the iterator stays +meaningful after the iterator advances. A span into the page body does not — advancing drops the +guard, and the frame may then be evicted and refilled with a different page, which reads as valid +memory holding another table's bytes. That is the `ConstHeapPage` lifetime rule at the one place +it is easiest to reintroduce, so the iterator copies tuple bytes into caller-owned storage rather +than handing out a view whose validity silently ends at the next advance. + +### Mutating during a scan + +`DELETE ... WHERE` and `UPDATE ... WHERE` read rows through a scan and then write them. How they +are planned belongs to the executor, but this layer constrains the shape of it hard enough that +the constraint is recorded here. + +Two facts rule out the obvious "match a row and write it where it sits": + +- **There is no read-to-write latch upgrade.** [DD-002](./DD-002-buffer-pool-manager.md) states + it: a shared-to-unique upgrade has to drop the shared latch first. So a mutation cannot be + performed from inside the iterator's read guard — the guard must be dropped and the page + re-fetched for write. That window is survivable precisely because of the rule at the top of + this doc: a compaction in between may have rewritten every offset in the page, and the RID + still resolves, because the slot index did not move. The row itself may be gone, which `Get` + already answers as "not found". +- **A relocating update can be seen twice.** `Update` case 3 moves a row that outgrew its page to + somewhere else in the chain — possibly a page the scan has not reached yet. The scan arrives + there, the row still matches the predicate, and it is updated again. This is the Halloween + problem. Postgres cannot hit it because a tuple written by the current command is invisible to + that command (`cmin`/`cmax`); this engine has no visibility machinery to borrow, so nothing + stops the loop on its own. + +> **Collect first, then mutate.** The scan runs to completion gathering RIDs, its guard drops, +> and only then does each mutation take its own write guard. + +Both problems die to that single rule. No latch upgrade is ever needed, because nothing is +latched during the apply phase. The Halloween case is impossible, because the set of RIDs was +frozen before the first write. The price is memory proportional to the number of matched rows, +which is affordable at this engine's scope and is the reason a rule this blunt is enough. + +Each RID is re-verified against the real page as it is applied, and a row that has vanished is +skipped rather than treated as an error — the same **hint, never a truth** discipline the +cross-page structures follow, applied to a list of RIDs instead of a free-space map. + +## Concurrency + +This layer defines no locking of its own. Every `HeapPage` operation runs inside a +`WritePageGuard`, which already holds the frame's content latch exclusively plus a pin, so a +heap page is single-threaded by construction and its invariants can never be observed mid-update. + +The only two cross-page concerns are page extension (above) and iteration, which takes and +releases one guard at a time and therefore sees a consistent page at each step but no consistent +snapshot of the table. Snapshot semantics are a transaction-layer concern and land with 2PL. + +## Non-goals + +- **Tuple encoding.** `HeapPage` stores opaque byte blobs. What a row *means* — column types, + nulls, ordering — belongs to the catalog and executor. Keeping this seam is what lets the heap + be built and finished before the schema format is settled. +- **Free space map.** Not a non-goal, a staged one: `dead_bytes` and the rotating insert + cursor land with this doc, the FSM page is designed above and deferred. See + [Free space across pages](#free-space-across-pages). +- **Overflow pages / TOAST.** A row that does not fit a page is an error at `CREATE TABLE` or + `INSERT`, not a chain of pages. Rejected because a multi-page insert has no atomicity without a + WAL — a partial chain leaks pages with no undo — and the scoped SQL cannot produce such a row. +- **Page-level defragmentation on a background thread.** Compaction is inline, on the thread that + needs the space. +- **Visibility / versioning.** No `xmin`/`xmax`. Deleted means gone, immediately, for everyone. + The one place that absence costs something concrete is the Halloween problem, priced and + handled under [Mutating during a scan](#mutating-during-a-scan). + +## Open questions + +None outstanding. The collect-then-mutate rule under +[Mutating during a scan](#mutating-during-a-scan) is a constraint this layer imposes, not a +decision it owns — how `DELETE` and `UPDATE` statements are actually planned around it settles in +the executor design doc. diff --git a/src/buffer/buffer_pool_manager.cpp b/src/buffer/buffer_pool_manager.cpp index bec89ad..b61127e 100644 --- a/src/buffer/buffer_pool_manager.cpp +++ b/src/buffer/buffer_pool_manager.cpp @@ -203,9 +203,11 @@ void BufferPoolManager::UnpinPage(frame_id_t frame_id) { } Result BufferPoolManager::FetchFrame(page_id_t page_id) { + if (page_id == META_PAGE_ID || page_id == INVALID_PAGE) + return std::unexpected(Status::InvalidArgument("invalid page_id")); // A loop, not straight-line code: the miss path releases the shard lock to acquire a frame, - // and another thread can publish this same page in that window. The only correct response is - // to hand our frame back and start over from the lookup. + // and another thread can publish this same page in that window. The only correct response + // is to hand our frame back and start over from the lookup. for (;;) { // -1 rather than left indeterminate: every path below assigns it before use, and if one // ever stops doing so, FrameAt's bounds assert fires immediately instead of silently @@ -220,9 +222,10 @@ Result BufferPoolManager::FetchFrame(page_id_t page_id) { // Declared HERE, outside the block below, and this placement is the whole trick. The // lookup must hold the shard lock and the metadata mutex simultaneously — dropping the // shard lock before pinning would let a reclaimer take the frame between Find() and the - // increment. But the Loading wait must NOT hold the shard lock (DD-002: never block on a - // condvar while holding one), and ShardGuard exposes no early unlock. Declaring `meta` - // first means the ShardGuard dies at the end of the block while this lock lives on. + // increment. But the Loading wait must NOT hold the shard lock (DD-002: never block on + // a condvar while holding one), and ShardGuard exposes no early unlock. Declaring + // `meta` first means the ShardGuard dies at the end of the block while this lock lives + // on. // // unique_lock rather than lock_guard because cv_.wait() requires one. std::unique_lock meta; @@ -250,70 +253,71 @@ Result BufferPoolManager::FetchFrame(page_id_t page_id) { // Predicate form handles both a spurious wakeup and the already-published case. frame.cv_.wait(meta, [&] { return frame.state_ != FrameState::Loading; }); - // Free is unreachable: a frame is unmapped in the same critical section that turns it - // Free, so the shard lookup above could not have returned it — and our pin has kept - // it that way ever since. + // Free is unreachable: a frame is unmapped in the same critical section that turns + // it Free, so the shard lookup above could not have returned it — and our pin has + // kept it that way ever since. assert(frame.state_ == FrameState::Resident || frame.state_ == FrameState::Failed); if (frame.state_ == FrameState::Failed) { // The loader's read failed and it owns disposal; we only drop the pin we took. - // unlock() first is mandatory, not tidiness: UnpinPage takes this same mutex and - // std::mutex is not recursive. UnpinPage is also what notifies the loader that a - // waiter has left, which is how AbandonLoad's pin_count == 1 wait terminates. + // unlock() first is mandatory, not tidiness: UnpinPage takes this same mutex + // and std::mutex is not recursive. UnpinPage is also what notifies the loader + // that a waiter has left, which is how AbandonLoad's pin_count == 1 wait + // terminates. meta.unlock(); UnpinPage(frame_id); return std::unexpected( Status::IOError(std::format("unable to fetch frame {}", frame_id))); } - // Release before touching the replacer — DD-002 forbids calling into it while holding - // a frame's metadata mutex. This admits a brief window where the frame is still marked - // evictable despite being pinned; that is harmless because reclaim re-validates - // pin_count == 0 under the metadata mutex and simply declines. + // Release before touching the replacer — DD-002 forbids calling into it while + // holding a frame's metadata mutex. This admits a brief window where the frame is + // still marked evictable despite being pinned; that is harmless because reclaim + // re-validates pin_count == 0 under the metadata mutex and simply declines. meta.unlock(); replacer_.RecordAccess(frame_id); if (from_zero) replacer_.SetEvictable(frame_id, false); return frame_id; } - // --- Cache miss. We reach here holding NO lock, which is required rather than incidental: - // AcquireFrame may reclaim, and re-validating a victim needs the *victim's* shard lock, - // about which our own shard lock says nothing. + // --- Cache miss. We reach here holding NO lock, which is required rather than + // incidental: AcquireFrame may reclaim, and re-validating a victim needs the *victim's* + // shard lock, about which our own shard lock says nothing. auto found_frame = AcquireFrame(); if (!found_frame.has_value()) { // kBufferPoolFull, propagated. Nothing was acquired, so nothing is owed. return std::unexpected(found_frame.error()); } - // Free, unmapped, unpinned, and out of both the free list and the candidate set — private - // to this thread until we publish it below. + // Free, unmapped, unpinned, and out of both the free list and the candidate set — + // private to this thread until we publish it below. frame_id = found_frame.value(); { auto shard = page_table_.AcquireShard(page_id); - // Look up AGAIN. We had no shard lock across AcquireFrame, so another thread may have - // published this page in the meantime. Loading it twice would give two frames the same - // identity. + // Look up AGAIN. We had no shard lock across AcquireFrame, so another thread may + // have published this page in the meantime. Loading it twice would give two frames + // the same identity. auto found = shard.Find(page_id); if (found.has_value()) { - // Lost the race. Our frame goes back to the FREE LIST, not the replacer: it holds - // no page, and a Free frame is stock rather than an eviction candidate. Pushing - // under the shard lock is safe — the free-list lock is a leaf that nothing is ever - // held across, so no cycle is possible. + // Lost the race. Our frame goes back to the FREE LIST, not the replacer: it + // holds no page, and a Free frame is stock rather than an eviction candidate. + // Pushing under the shard lock is safe — the free-list lock is a leaf that + // nothing is ever held across, so no cycle is possible. free_list_.Push(frame_id); continue; } - // Publish the mapping BEFORE the read. This is what makes a concurrent fetcher of this - // page take the hit path and sleep on Loading instead of starting a second, redundant - // read of the same page into a second frame. + // Publish the mapping BEFORE the read. This is what makes a concurrent fetcher of + // this page take the hit path and sleep on Loading instead of starting a second, + // redundant read of the same page into a second frame. shard.Insert(page_id, frame_id); auto& frame = FrameAt(frame_id); // A scoped lock, deliberately NOT the function-scope `meta`: that one outlives this - // block, so assigning to it would leave the metadata mutex held across the read below - // — and the re-lock after the read would then deadlock against ourselves. + // block, so assigning to it would leave the metadata mutex held across the read + // below — and the re-lock after the read would then deadlock against ourselves. std::lock_guard lock(frame.mtx_); frame.page_id_ = page_id; frame.state_ = FrameState::Loading; @@ -322,14 +326,14 @@ Result BufferPoolManager::FetchFrame(page_id_t page_id) { } // both released // No lock and no content latch. Safe because the frame is Loading: no guard for it can - // exist (guards only come from this function returning), and every other fetcher of this - // page is asleep on the condvar rather than touching data_. + // exist (guards only come from this function returning), and every other fetcher of + // this page is asleep on the condvar rather than touching data_. auto& frame = FrameAt(frame_id); auto st = disk_manager_.ReadPage(page_id, frame.data_); if (!st.ok()) { // The loader owns disposal. AbandonLoad erases the mapping, publishes Failed, waits - // for the waiters' pins to drain, and returns the frame to the free list. Do not unpin - // or push here as well — that would be a double release. + // for the waiters' pins to drain, and returns the frame to the free list. Do not + // unpin or push here as well — that would be a double release. AbandonLoad(frame_id, page_id); return std::unexpected(st); } @@ -373,8 +377,8 @@ Result BufferPoolManager::FetchFrame(page_id_t page_id) { frame.cv_.notify_all(); } - // No SetEvictable(false) on this path: the frame came from the free list or a reclaim, so - // it was never in the candidate set to begin with. + // No SetEvictable(false) on this path: the frame came from the free list or a reclaim, + // so it was never in the candidate set to begin with. replacer_.RecordAccess(frame_id); return frame_id; // pin_count == 1, and it is the caller's } diff --git a/src/buffer/page_guard.hpp b/src/buffer/page_guard.hpp index e307e92..f173581 100644 --- a/src/buffer/page_guard.hpp +++ b/src/buffer/page_guard.hpp @@ -31,12 +31,16 @@ class BufferPoolManager; // // 1. WritePageGuard only: bump dirty_epoch. // 2. Release the content latch. -// 3. Metadata mutex; decrement pin_count; note a 1->0 transition. +// 3. Metadata mutex; decrement pin_count; note a 1->0 transition; if it transitioned, +// SetEvictable(frame_id, true) UNDER THAT SAME MUTEX. // 4. Release the metadata mutex. -// 5. If it transitioned, SetEvictable(frame_id, true). // -// Steps 3-5 live in BufferPoolManager::UnpinPage, which is private and which these two classes -// are friends of. Step 1 precedes step 2 so a flusher holding the shared latch always observes +// Steps 3-4 live in BufferPoolManager::UnpinPage, which is private and which these two classes +// are friends of. The SetEvictable call sitting inside step 3 rather than after step 4 is the +// one named exception to DD-002's "never call into the replacer while holding a frame's +// metadata mutex": publishing it afterwards let DeletePage vacate the frame in the gap and +// revoke a membership that had not been created yet, leaving one frame on the free list and in +// the replacer's candidate set at once. Step 1 precedes step 2 so a flusher holding the shared latch always observes // an epoch that already accounts for every write it is about to capture; step 1 precedes step 3 // so a reclaimer that sees pin_count == 0 under the metadata mutex is guaranteed to see this // guard's dirty bump too, which is what makes its clean-check exact rather than merely diff --git a/src/common/status.hpp b/src/common/status.hpp index 62ed3e7..867ff9a 100644 --- a/src/common/status.hpp +++ b/src/common/status.hpp @@ -13,6 +13,7 @@ enum class ErrorCode : uint8_t { kIOError, kInvalidArgument, kBufferPoolFull, + kPageFull, kDuplicateKey, kSerializationConflict, kInternal, @@ -30,6 +31,10 @@ class [[nodiscard]] Status { static Status BufferPoolFull(std::string msg) { return {ErrorCode::kBufferPoolFull, std::move(msg)}; } + // A page has no room for the write. Distinct from kBufferPoolFull, which means the pool has + // no free frame: that one is fatal to the operation, this one is a routing decision — the + // caller can compact this page and retry, or place the row on a different page. + static Status PageFull(std::string msg) { return {ErrorCode::kPageFull, std::move(msg)}; } static Status DuplicateKey(std::string msg) { return {ErrorCode::kDuplicateKey, std::move(msg)}; } diff --git a/src/heap/heap_page.cpp b/src/heap/heap_page.cpp new file mode 100644 index 0000000..4ba30ed --- /dev/null +++ b/src/heap/heap_page.cpp @@ -0,0 +1,373 @@ +#include "heap/heap_page.hpp" + +#include +#include +#include +#include +#include +#include + +#include "common/logger.hpp" +#include "common/status.hpp" +#include "common/types.hpp" + +namespace kernsql { + +namespace { + +// Body offset of a slot entry. Slot 0 sits immediately after the sub-header and the array grows +// forward from there; the slot's INDEX is what a RID names, which is why this is pure +// arithmetic and never a search. +constexpr std::size_t SlotOffset(slot_id_t slot) { + return HEAP_SUB_HEADER_SIZE + static_cast(slot) * SLOT_SIZE; +} + +} // namespace + +// --------------------------------------------------------------------------------------------- +// Byte marshalling. Everything below this line is plumbing: it moves bytes in and out of the +// body and does no reasoning about them. +// --------------------------------------------------------------------------------------------- + +HeapSubHeader ConstHeapPage::Header() const { + return HeapSubHeader::ReadFrom(body_.first()); +} + +Slot ConstHeapPage::SlotAt(slot_id_t slot) const { + return Slot::ReadFrom( + std::span{body_.subspan(SlotOffset(slot), SLOT_SIZE)}); +} + +void HeapPage::WriteHeader(const HeapSubHeader& header) { + header.WriteTo(body_.first()); +} + +void HeapPage::WriteSlot(slot_id_t slot, const Slot& entry) { + entry.WriteTo(std::span{body_.subspan(SlotOffset(slot), SLOT_SIZE)}); +} + +Result AsHeapPage(const ReadPageGuard& guard) { + if (guard.Header().page_type != PageType::HEAP) { + return std::unexpected(Status::Corruption("page is not a heap page")); + } + return ConstHeapPage(guard.Body()); +} + +Result AsHeapPage(WritePageGuard& guard) { + if (guard.Header().page_type != PageType::HEAP) { + return std::unexpected(Status::Corruption("page is not a heap page")); + } + return HeapPage(guard.MutableBody()); +} + +std::size_t ConstHeapPage::Contiguous() const { + const auto h = Header(); + const std::size_t slot_array_end = HEAP_SUB_HEADER_SIZE + std::size_t{h.slot_count} * SLOT_SIZE; + + assert(h.tuple_data_start >= slot_array_end); // loud in debug + if (h.tuple_data_start < slot_array_end) return 0; // safe in release + return h.tuple_data_start - slot_array_end; +} + +std::size_t ConstHeapPage::Reclaimable() const { + return Contiguous() + Header().dead_bytes; +} + +Result> ConstHeapPage::Get(slot_id_t slot_id) const { + if (Header().slot_count <= slot_id) + return std::unexpected(Status::NotFound("slot doesn't exist")); + auto slot = SlotAt(slot_id); + if (slot.IsDead()) return std::unexpected(Status::NotFound("slot is dead")); + + const std::size_t tuple_end = slot.offset + slot.length; + assert(tuple_end <= body_.size()); + if (tuple_end <= body_.size()) return body_.subspan(slot.offset, slot.length); + return std::unexpected(Status::Corruption("tuple not accessible")); +} + +bool ConstHeapPage::CheckInvariants() const { + const auto h = Header(); + + // Every failure names the invariant it broke. `assert(CheckInvariants())` on its own tells + // you a page is corrupt and nothing whatsoever about how, which is the least useful moment + // to be told that. LOG_DEBUG compiles to nothing under NDEBUG, which is also the only build + // this function is meant to run in. [[maybe_unused]] because of that: in a release build the + // parameter genuinely is unused, and -Wextra would say so. + auto fail = []([[maybe_unused]] const char* what) { + LOG_DEBUG("heap page invariant violated: %s", what); + return false; + }; + + const std::size_t slot_array_end = HEAP_SUB_HEADER_SIZE + std::size_t{h.slot_count} * SLOT_SIZE; + + // These two run FIRST, and the order is load-bearing rather than stylistic. This function is + // the one place designed to be handed a page that is already garbage, so it may not index + // anything it has not first proved is in range. Together they bound slot_count at + // (PAGE_BODY_SIZE - HEAP_SUB_HEADER_SIZE) / SLOT_SIZE == 1014, which is what makes both the + // SlotAt() calls below and the loop counter safe on a corrupt page. + if (h.tuple_data_start > PAGE_BODY_SIZE) + return fail("tuple_data_start is past the end of the body"); + if (slot_array_end > h.tuple_data_start) + return fail("slot array has run into the tuple region"); + + std::size_t live_slots = 0; + std::size_t live_bytes = 0; + + for (slot_id_t i = 0; i < h.slot_count; ++i) { + const Slot s = SlotAt(i); + + if (s.IsDead()) { + // Dead has exactly one representation, which is the only reason this is assertable: + // Delete zeroes both fields rather than just the offset. + if (s.length != 0) return fail("dead slot has a non-zero length"); + continue; + } + + // A live tuple lies wholly inside the tuple region. This is the thorough per-slot audit + // Get cannot afford on the hot path, which is why it lives here instead. + if (s.offset < h.tuple_data_start) return fail("live slot points below tuple_data_start"); + if (std::size_t{s.offset} + s.length > PAGE_BODY_SIZE) + return fail("live slot extends past the end of the body"); + + ++live_slots; + live_bytes += s.length; + } + + if (live_slots != h.live_count) return fail("live_count disagrees with the slot array"); + + // The whole point. Every byte of the tuple region belongs to exactly one live tuple or to + // nobody, so the region's size must be exactly accounted for. Overlapping tuples fall out of + // this for free, with no pairwise comparison: an overlap counts the shared bytes twice, so + // the sum overshoots the region. + if (h.tuple_data_start + live_bytes + h.dead_bytes != PAGE_BODY_SIZE) + return fail("tuple region accounting does not add up"); + + return true; +} + +void HeapPage::Init() { + // Value-initialised so every field comes from its default member initializer, which is the + // only way tuple_data_start starts at PAGE_BODY_SIZE. Spelling the zeros out instead OVERRIDES + // those initializers and leaves the low-water mark at 0, i.e. a page that reports no free + // space and fails its own accounting check the moment it is created. + WriteHeader(HeapSubHeader{}); +} + +Result HeapPage::Insert(std::span tuple) { + if (tuple.empty()) { + return std::unexpected(Status::InvalidArgument("cannot insert a zero-length tuple")); + } + if (tuple.size() > MAX_TUPLE_SIZE) { + return std::unexpected(Status::InvalidArgument("tuple exceeds the maximum tuple size")); + } + + auto header = Header(); + slot_id_t slot_id = header.slot_count; + std::size_t required = tuple.size() + SLOT_SIZE; // appending must pay for a new slot entry + + if (header.live_count < header.slot_count) { + for (slot_id_t i = 0; i < header.slot_count; ++i) { + if (SlotAt(i).IsDead()) { + slot_id = i; + required = tuple.size(); // the slot entry already exists; only bytes are needed + break; + } + } + } + + // DOES NOT COMPACT, by contract. The caller compares its own requirement against + // Reclaimable() and decides between compacting this page and retrying, or another page. + if (required > Contiguous()) { + return std::unexpected(Status::PageFull("not enough contiguous space for the tuple")); + } + + const auto length = static_cast(tuple.size()); + const auto offset = static_cast(header.tuple_data_start - length); + + std::memcpy(body_.data() + offset, tuple.data(), tuple.size()); + WriteSlot(slot_id, Slot{offset, length}); + + header.tuple_data_start = offset; + header.live_count = static_cast(header.live_count + 1); + if (slot_id == header.slot_count) { + header.slot_count = static_cast(header.slot_count + 1); + } + + // dead_bytes is deliberately untouched, and this reads backwards every single time. Reusing + // a dead slot reclaims the four bytes of the slot ENTRY, not the dead tuple's bytes: Delete + // destroyed that tuple's offset, so nothing can find those bytes, let alone reuse them. Only + // Compact recovers them. + WriteHeader(header); + + assert(CheckInvariants()); + return slot_id; +} + +Status HeapPage::Delete(slot_id_t slot_id) { + auto slot_found = Get(slot_id); + if (!slot_found.has_value()) return slot_found.error(); + + Slot delete_slot = SlotAt(slot_id); + + // update header + auto header = Header(); + header.dead_bytes += delete_slot.length; + header.live_count--; + WriteHeader(header); + + // update slot + delete_slot.length = 0; + delete_slot.offset = 0; + WriteSlot(slot_id, delete_slot); + + assert(CheckInvariants()); + return Status::OK(); +} + +Result HeapPage::Update(slot_id_t slot_id, std::span tuple) { + if (tuple.empty()) { + return std::unexpected(Status::InvalidArgument("cannot update to a zero-length tuple")); + } + if (tuple.size() > MAX_TUPLE_SIZE) { + return std::unexpected(Status::InvalidArgument("tuple exceeds the maximum tuple size")); + } + + // Existence and liveness in one shot, with Get's error propagated unchanged: an update to a + // deleted RID is NotFound, not Corruption. The span it hands back is deliberately discarded — + // it is a const view, and every write below goes through the slot instead. + auto existing = Get(slot_id); + if (!existing.has_value()) return std::unexpected(existing.error()); + + auto slot = SlotAt(slot_id); + auto header = Header(); + + // One snapshot, taken before anything mutates, so the branch conditions below cannot end up + // comparing a stale header field against a freshly recomputed Contiguous(). + const std::size_t contiguous = Contiguous(); + const auto new_length = static_cast(tuple.size()); + + UpdateOutcome outcome = UpdateOutcome::kSamePage; + if (slot.length >= new_length) { + // Case 1: overwrite in place. The tuple keeps its offset, so the bytes freed at the tail + // of the old extent are referenced by no slot and no offset; dead_bytes is the only + // record that they exist. + std::memcpy(body_.data() + slot.offset, tuple.data(), new_length); + + header.dead_bytes = static_cast(header.dead_bytes + (slot.length - new_length)); + WriteHeader(header); + + slot.length = new_length; + WriteSlot(slot_id, slot); + } else if (new_length <= contiguous) { + // Case 2, fast path: it grows, but there is room at the low-water mark, so relocate + // within the page and skip the compaction entirely. The slot index does not change. + header.dead_bytes = static_cast(header.dead_bytes + slot.length); + header.tuple_data_start = static_cast(header.tuple_data_start - new_length); + + std::memcpy(body_.data() + header.tuple_data_start, tuple.data(), new_length); + WriteHeader(header); + + slot.length = new_length; + slot.offset = header.tuple_data_start; + WriteSlot(slot_id, slot); + } else if (new_length <= contiguous + header.dead_bytes + slot.length) { + // Case 2, slow path: it fits only after reclaiming garbage. The `+ slot.length` term is + // this tuple's own bytes, which are about to become garbage themselves — which is why the + // test must be pure arithmetic run BEFORE any mutation, since case 3 has to leave the + // page untouched. + header.dead_bytes = static_cast(header.dead_bytes + slot.length); + header.live_count = static_cast(header.live_count - 1); + WriteHeader(header); + + slot.offset = 0; + slot.length = 0; + WriteSlot(slot_id, slot); + + // Killing first is what stops the compaction from copying bytes that are already + // abandoned. The slot is dead here while still logically live, which is safe only + // because the whole operation runs under one write latch. + Compact(); + + // Compact rewrote tuple_data_start and zeroed dead_bytes, so the local copy is stale. + header = Header(); + header.tuple_data_start = static_cast(header.tuple_data_start - new_length); + header.live_count = static_cast(header.live_count + 1); + + std::memcpy(body_.data() + header.tuple_data_start, tuple.data(), new_length); + WriteHeader(header); + + slot.offset = header.tuple_data_start; + slot.length = new_length; + WriteSlot(slot_id, slot); + } else { + outcome = UpdateOutcome::kDoesNotFit; + } + + assert(CheckInvariants()); + return outcome; +} + +void HeapPage::Compact() { + auto header = Header(); + + // Nothing to reclaim, and that is provable rather than a guess. The accounting identity says + // tuple_data_start + live_bytes + dead_bytes == PAGE_BODY_SIZE, so zero garbage means the live + // tuples exactly fill the region, and non-overlapping tuples that exactly fill a region are + // already packed. Update's slow path calls this unconditionally, so the early exit is + // load-bearing rather than decorative. + if (header.dead_bytes == 0) return; + + // Rebuilt in a scratch buffer and copied back in one shot, rather than slid in place. Sliding + // in place does work — descending offset order guarantees no tuple can land on one that has + // not moved yet — but it needs the live slots sorted by offset first and memmove throughout. + // Postgres shipped exactly that for years and then reversed it: compactify_tuples sorted and + // slid until PG14 rewrote it around a scratch buffer, which came out both simpler and faster. + // InnoDB's page reorganize copies via a temp block too. + // + // Left uninitialised on purpose. The loop writes every byte of [cursor, PAGE_BODY_SIZE) before + // the copy back reads it — cursor drops by exactly the length the loop then fills — so nothing + // uninitialised is ever read, and zeroing 4KB per compaction would be pure waste. + std::array scratch; + std::size_t cursor = PAGE_BODY_SIZE; + + [[maybe_unused]] const auto freed_from = header.tuple_data_start; + + for (slot_id_t i = 0; i < header.slot_count; ++i) { + Slot slot = SlotAt(i); + + // A dead slot keeps its index forever so a stale RID still has something to resolve + // against. Compaction moves bytes; it never renumbers slots. + if (slot.IsDead()) continue; + + cursor -= slot.length; + std::memcpy(scratch.data() + cursor, body_.data() + slot.offset, slot.length); + + // Written inside the same pass that moved the bytes. The rule the whole format rests on is + // that no operation may move a tuple without updating its slot in the same breath. + slot.offset = static_cast(cursor); + WriteSlot(i, slot); + } + + // ONLY the occupied tail. Copying the whole buffer back would stamp over the free space and, + // below it, the slot array this loop just rewrote. + std::memcpy(body_.data() + cursor, scratch.data() + cursor, PAGE_BODY_SIZE - cursor); + + // slot_count and live_count are deliberately untouched. Update's slow path depends on it: that + // code decrements live_count before calling here and restores it afterwards, so an adjustment + // in this function would leave the count off by one. + header.tuple_data_start = static_cast(cursor); + header.dead_bytes = 0; + WriteHeader(header); + +#ifndef NDEBUG + // The band that just became free — exactly the old dead_bytes, and strictly inside the old + // tuple region, so it can never reach the slot array. Poisoning it turns a stale offset into + // obvious garbage instead of plausible-looking bytes, which is the difference between a loud + // failure and a silent wrong answer. + std::memset(body_.data() + freed_from, 0xDD, cursor - freed_from); +#endif + + assert(CheckInvariants()); +} + +} // namespace kernsql diff --git a/src/heap/heap_page.hpp b/src/heap/heap_page.hpp new file mode 100644 index 0000000..e6caf2d --- /dev/null +++ b/src/heap/heap_page.hpp @@ -0,0 +1,324 @@ +#pragma once + +#include +#include +#include +#include +#include + +#include "buffer/page_guard.hpp" +#include "common/status.hpp" +#include "common/types.hpp" + +namespace kernsql { + +/* + * Slotted heap page (DD-004). A heap page is an unordered pile of tuples addressed by + * RID{page_id, slot}, laid out inside the 4064-byte body a page guard hands out: + * + * body offset 0 + * +---------------------------+ + * | heap sub-header (8 bytes) | + * +---------------------------+ + * | slot[0] slot[1] ... | grows FORWARD, 4 bytes each + * +---------------------------+ + * | free space | + * +---------------------------+ + * | ... tuple tuple tuple | grows BACKWARD from the end + * +---------------------------+ + * body offset 4064 + * + * THE RULE EVERYTHING FOLLOWS: slots never move, tuple bytes move freely. A RID names a slot + * index and the slot names an offset, so compaction can rewrite every byte of the tuple region + * and every RID in the database stays valid. The corollary is the thing to keep testing: any + * operation that moves tuple bytes must update the moved tuples' slot offsets in the same + * latched operation. + * + * All offsets here are BODY-RELATIVE — offset 0 is the first byte after the 32-byte PageHeader, + * which this layer cannot see. That is deliberate (see WritePageGuard::MutableBody) and it is + * why page_type validation lives in AsHeapPage below rather than in these classes. + */ + +inline constexpr std::size_t HEAP_SUB_HEADER_SIZE = 8; +inline constexpr std::size_t SLOT_SIZE = 4; + +// Largest tuple the format can physically hold: the whole body, less the sub-header and the one +// slot that tuple needs. Not the enforced limit — see MAX_TUPLE_SIZE. +inline constexpr std::size_t MAX_TUPLE_ON_EMPTY_PAGE = + PAGE_BODY_SIZE - HEAP_SUB_HEADER_SIZE - SLOT_SIZE; + +// The enforced cap, checked at CREATE TABLE against the widest row a schema can produce and +// again at INSERT against the actual encoded size. 2000 rather than the 4052 ceiling because +// 2000 guarantees at least two rows per page (InnoDB's rule): a format that allows exactly one +// row per page degenerates into a linked list with 50% overhead. +inline constexpr std::size_t MAX_TUPLE_SIZE = 2000; + +static_assert(2 * (MAX_TUPLE_SIZE + SLOT_SIZE) + HEAP_SUB_HEADER_SIZE <= PAGE_BODY_SIZE, + "MAX_TUPLE_SIZE must guarantee two tuples per page"); + +/* + * The 8 bytes at body offset 0. Serialized through ReadFrom/WriteTo rather than mapped with a + * reinterpret_cast over the body, matching PageHeader: same host-endian memcpy, same + * static_assert block, and it sidesteps the object-lifetime question a cast would raise. + * + * Three of these four fields have direct InnoDB counterparts — PAGE_N_HEAP, PAGE_HEAP_TOP, + * PAGE_N_RECS, PAGE_GARBAGE — which is a reasonable sign the layout is not novel. + */ +struct HeapSubHeader { + // Entries in the slot array, live and dead alike. Only ever grows within a page's life as + // a heap page; a dead slot keeps its index forever so a stale RID has something to resolve + // against. + uint16_t slot_count{0}; + + // Body offset of the lowest occupied tuple byte. The tuple region is + // [tuple_data_start, PAGE_BODY_SIZE). Initialised to PAGE_BODY_SIZE for an empty page. + uint16_t tuple_data_start{static_cast(PAGE_BODY_SIZE)}; + + // Non-dead slots. Maintained rather than derived so a scan can skip an empty page without + // walking its slot array. + uint16_t live_count{0}; + + // Bytes in the tuple region that no live slot points at: deleted tuples, plus the waste + // left behind by an in-place shrinking update. + // + // CANNOT BE DERIVED, which is the entire reason it is stored. Update case 1 overwrites a + // slot's length with the smaller value, so the bytes it orphans are referenced by no slot + // and no offset and are invisible to any later slot walk. Without this counter a page that + // has taken many in-place shrinks looks full while being mostly holes, and nothing ever + // learns there is space to recover. + // + // Only ever increases, or is reset to zero by Compact. There is no subtraction anywhere, so + // there is no path that can drive it negative; and it is bounded by MAX_TUPLE_ON_EMPTY_PAGE, + // so it cannot overflow its uint16_t. + uint16_t dead_bytes{0}; + + static HeapSubHeader ReadFrom(std::span bytes) { + HeapSubHeader h; + std::memcpy(&h, bytes.data(), sizeof(HeapSubHeader)); + return h; + } + void WriteTo(std::span bytes) const { + std::memcpy(bytes.data(), this, sizeof(HeapSubHeader)); + } +}; + +static_assert(sizeof(HeapSubHeader) == HEAP_SUB_HEADER_SIZE, "heap sub-header must be 8 bytes"); +static_assert(alignof(HeapSubHeader) == 2, "four uint16_t must pack with no implicit padding"); +static_assert(std::is_standard_layout_v); +static_assert(std::is_trivially_copyable_v); + +/* + * One slot directory entry. A slot is in exactly one of two states — live (offset != 0) or dead + * (offset == 0). There is no third "never used" state: a slot is created live by an insert and + * a delete makes it dead. + * + * offset == 0 is a safe sentinel rather than a magic number, because body offset 0 is inside + * the sub-header and no tuple can legitimately start there. Postgres spends a flags bitfield on + * the same distinction; the impossible offset gets it for free. + */ +struct Slot { + uint16_t offset{0}; // body offset of the tuple's first byte; 0 means dead + uint16_t length{0}; // tuple length in bytes + + [[nodiscard]] bool IsDead() const { return offset == 0; } + + static Slot ReadFrom(std::span bytes) { + Slot s; + std::memcpy(&s, bytes.data(), sizeof(Slot)); + return s; + } + void WriteTo(std::span bytes) const { + std::memcpy(bytes.data(), this, sizeof(Slot)); + } +}; + +static_assert(sizeof(Slot) == SLOT_SIZE, "slot must be 4 bytes"); +static_assert(std::is_standard_layout_v); +static_assert(std::is_trivially_copyable_v); + +// What HeapPage::Update did, for the caller that has to decide whether the RID survived. +enum class UpdateOutcome : uint8_t { + // Overwritten in place, or relocated within this page after a compaction. Either way the + // slot index did not change, so THE RID SURVIVES. + kSamePage, + + // Does not fit even after compacting this page. HeapPage does nothing; TableHeap must + // delete here and insert elsewhere, and THE RID CHANGES. + kDoesNotFit, +}; + +/* + * Read-only view over a heap page body. Non-owning: it holds a span, not a guard, not a pin, + * and not a latch. + * + * LIFETIME. A view must never outlive the guard its span came from. Past that point the frame + * is unpinned and may have been evicted and refilled with a different page — the memory is + * still valid, so ASan and valgrind catch nothing, and reads silently return another page's + * bytes. Two rules contain it: + * + * 1. A view is a function local, born from a guard in the same scope. Never a member of + * anything. TableIterator holds the GUARD as a member and mints the view per access; + * caching both means two members that must be replaced in lockstep, and the failure mode + * when they are not is silent. + * 2. No default constructor, so an unbound view cannot exist. Same reasoning as the guards. + */ +class ConstHeapPage { + public: + ConstHeapPage() = delete; + + // Public because the point of a view is that tests construct one over a bare + // std::array with no BufferPoolManager in existence. Prefer + // AsHeapPage() in non-test code: it is the only place page_type can be checked. + explicit ConstHeapPage(std::span body) : body_(body) {} + + [[nodiscard]] HeapSubHeader Header() const; + [[nodiscard]] Slot SlotAt(slot_id_t slot) const; + + [[nodiscard]] uint16_t SlotCount() const { return Header().slot_count; } + [[nodiscard]] uint16_t LiveCount() const { return Header().live_count; } + + // Free space usable RIGHT NOW, between the end of the slot array and the tuple region. + [[nodiscard]] std::size_t Contiguous() const; + + // Free space usable AFTER a Compact: Contiguous() + dead_bytes. This is the number a + // cross-page free-space mechanism publishes, and the number to compare against before + // deciding a compaction is worth it. + [[nodiscard]] std::size_t Reclaimable() const; + + // Bytes of the tuple at `slot`. NotFound for an out-of-range or dead slot: a Get on a + // deleted RID returns "not found", never bytes. + [[nodiscard]] Result> Get(slot_id_t slot) const; + + /* + * Debug-only total accounting check. Every byte of the tuple region belongs to exactly one + * live tuple or to nobody, which fixes an identity that must hold after EVERY operation: + * + * tuple_data_start + sum(slot.length for live slots) + dead_bytes == PAGE_BODY_SIZE + * + * Call it at the end of every mutating operation in test builds. It catches the whole class + * of accounting bug this layer can have — a missed increment in Update case 1, a dead_bytes + * wrongly reduced by a dead-slot reuse, a Compact that forgot to reset — at the operation + * that caused it rather than a thousand operations later. Also checks + * live_count == count of live slots, and slot_array_end <= tuple_data_start. + * + * O(slot_count). Not for the hot path. + */ + [[nodiscard]] bool CheckInvariants() const; + + private: + std::span body_; +}; + +/* + * Mutable view. Same lifetime rules as ConstHeapPage. + * + * Two types rather than one templated on constness, mirroring T* -> const T*: HeapPage converts + * to ConstHeapPage, so the read operations are written once. The read/write split is carried by + * the span's type, which is the same mechanism the guards use — it is the type system, not a + * runtime mode check, that stops a reader writing. + */ +class HeapPage { + public: + HeapPage() = delete; + + explicit HeapPage(std::span body) : body_(body) {} + + // NOLINTNEXTLINE(google-explicit-constructor) — deliberate, this is the T* -> const T* edge. + operator ConstHeapPage() const { return ConstHeapPage(body_); } + + [[nodiscard]] ConstHeapPage View() const { return ConstHeapPage(body_); } + + // Read surface, forwarded so a caller holding a write guard is not forced through View(). + [[nodiscard]] HeapSubHeader Header() const { return View().Header(); } + [[nodiscard]] Slot SlotAt(slot_id_t slot) const { return View().SlotAt(slot); } + [[nodiscard]] uint16_t SlotCount() const { return View().SlotCount(); } + [[nodiscard]] uint16_t LiveCount() const { return View().LiveCount(); } + [[nodiscard]] std::size_t Contiguous() const { return View().Contiguous(); } + [[nodiscard]] std::size_t Reclaimable() const { return View().Reclaimable(); } + [[nodiscard]] Result> Get(slot_id_t slot) const { + return View().Get(slot); + } + [[nodiscard]] bool CheckInvariants() const { return View().CheckInvariants(); } + + // Zero the sub-header: slot_count = 0, live_count = 0, dead_bytes = 0, + // tuple_data_start = PAGE_BODY_SIZE. The CALLER stamps page_type = HEAP through the guard; + // this layer cannot see the header. + void Init(); + + /* + * Reuse a dead slot if one exists, else append a new one. Copies the tuple to + * tuple_data_start - len and lowers tuple_data_start. + * + * Rejects an empty tuple and anything over MAX_TUPLE_SIZE. The cap is checked here as well as + * at CREATE TABLE and at SQL INSERT, so that no path above this layer can put an oversized + * tuple on a page: the format's own limit is MAX_TUPLE_ON_EMPTY_PAGE, and a tuple between the + * two would insert fine and then be unrelocatable for the rest of its life. + * + * Needs len + SLOT_SIZE if it must append a slot, or len alone if it reuses a dead one, + * checked against Contiguous(). DOES NOT COMPACT: returns failure and lets the caller + * decide between compacting and retrying, or moving to another page. + * + * REUSING A DEAD SLOT DOES NOT REDUCE dead_bytes. It reclaims the four bytes of the slot, + * not the tuple bytes: Delete destroyed the old offset, so those bytes cannot be found, let + * alone reused. Only Compact reclaims them. This reads backwards and is the single most + * likely place to get the accounting wrong. + */ + [[nodiscard]] Result Insert(std::span tuple); + + /* + * Add the slot's length to dead_bytes BEFORE overwriting the slot, then set it to + * {offset: 0, length: 0} and decrement live_count. Zeroing both fields gives a dead slot one + * canonical representation, which is what makes `offset == 0 implies length == 0` + * assertable in CheckInvariants. + * + * The tuple's bytes are not moved; the space is reclaimed at the next compaction. + */ + [[nodiscard]] Status Delete(slot_id_t slot); + + /* + * Three cases: + * 1. new length <= old: overwrite in place, shrink length, and ADD THE DIFFERENCE TO + * dead_bytes. Those orphaned bytes are the ones no slot walk can ever find again. + * 2. new length > old but the page has room after compaction: kill the old tuple first + * (add its length to dead_bytes, zero the slot), THEN compact, then insert into the + * same slot id. Killing before compacting is what stops the compaction from copying + * bytes that are about to be abandoned. The slot is momentarily dead while still + * logically live; that is safe only because the whole thing runs under one write latch, + * and it is why Compact must never be callable in the middle of this from outside. + * Cases 1 and 2 both return kSamePage — the slot index did not change, so the RID lives. + * 3. does not fit even after compaction: change nothing, return kDoesNotFit. TableHeap + * deletes here and inserts elsewhere, and the RID changes. + */ + [[nodiscard]] Result Update(slot_id_t slot, std::span tuple); + + /* + * Rewrite the tuple region packed against the end of the body, in descending offset order, + * updating each live slot's offset as its tuple moves. Dead slots keep their index and stay + * dead. Afterwards tuple_data_start is the new low-water mark and dead_bytes == 0 — by + * definition, since every byte below the low-water mark is now live. + * + * The only operation that reduces dead_bytes. + */ + void Compact(); + + private: + void WriteHeader(const HeapSubHeader& header); + void WriteSlot(slot_id_t slot, const Slot& entry); + + std::span body_; +}; + +/* + * The only way to get a view in non-test code, and the reason it is a free function in the heap + * layer rather than a method on the guard: the dependency has to point heap -> buffer, or the + * buffer pool ends up knowing what a heap page is. + * + * These exist for one check. A HeapPage sees only the body, so it CANNOT verify + * page_type == HEAP; build one over a catalog or index page and it will parse garbage as a + * sub-header and scribble on it, and nothing notices until that page fails validation on some + * later miss. The guard can see the header, so the assert lives here. Corruption on mismatch. + */ +[[nodiscard]] Result AsHeapPage(const ReadPageGuard& guard); +[[nodiscard]] Result AsHeapPage(WritePageGuard& guard); + +} // namespace kernsql diff --git a/src/heap/table_heap.cpp b/src/heap/table_heap.cpp new file mode 100644 index 0000000..45612ef --- /dev/null +++ b/src/heap/table_heap.cpp @@ -0,0 +1,417 @@ +#include "table_heap.hpp" + +#include +#include +#include +#include +#include + +#include "buffer/page_guard.hpp" +#include "common/status.hpp" +#include "common/types.hpp" +#include "heap/heap_page.hpp" + +namespace kernsql { + +TableHeap::TableHeap(BufferPoolManager& bpm, page_id_t first_page_id, page_id_t last_page_id) + : bpm_(bpm), + first_page_id_(first_page_id), + last_page_id_(last_page_id), + // DD-004: the cursor starts at the front of the chain, on Create and on Open alike. Nothing + // about it is persisted, so a restart simply begins the sweep again. + insert_hint_(first_page_id) {} + +Result> TableHeap::Create(BufferPoolManager& bpm) { + // alloacte first page and stamp as Heap + auto first_page = bpm.NewPage(); + if (!first_page.has_value()) return std::unexpected(first_page.error()); + + // Stamp page as Heap + first_page.value().SetPageType(PageType::HEAP); + first_page.value().SetNextPageId(INVALID_PAGE); + + auto heap_page = AsHeapPage(first_page.value()); + if (!heap_page.has_value()) return std::unexpected(heap_page.error()); + heap_page->Init(); + + page_id_t page_id = first_page.value().PageId(); + + auto table_heap = std::unique_ptr(new TableHeap(bpm, page_id, page_id)); + return table_heap; +} + +Result> TableHeap::Open(BufferPoolManager& bpm, page_id_t first_page_id, + page_id_t last_page_id) { + auto table_heap = std::unique_ptr(new TableHeap(bpm, first_page_id, last_page_id)); + return table_heap; +} + +namespace { + +// How many pages the rotating cursor probes before giving up and appending at the last page +// (DD-004). Bounded on purpose: an unbounded sweep over a table with no free space turns one +// insert into a full chain walk that flushes the pool, which is strictly worse than appending. +constexpr int INSERT_PROBE_LIMIT = 4; + +// Compact() rewrites the whole tuple region and touches every live slot. Below this much dead +// space the rewrite costs more than the bytes it hands back, so the page is treated as full and +// the probe moves on. +constexpr std::size_t COMPACT_THRESHOLD = PAGE_BODY_SIZE / 8; + +/* + * Try to place `tuple` on an ALREADY-LATCHED heap page. + * + * Three outcomes, and keeping them distinct is the whole point: a slot means it landed, + * std::nullopt means this page cannot take it and the caller should look elsewhere, and an error + * means something is actually wrong. kPageFull is a routing decision, not a failure — it must + * never escape TableHeap::Insert as an error. + * + * `compact_threshold` is the smallest amount of dead space worth a rewrite. Pass 0 on the append + * path, where the alternative is allocating a whole new page and any reclaim beats that. + */ +Result> TryInsertOnPage(HeapPage& page, std::span tuple, + std::size_t compact_threshold) { + // Conservative: HeapPage::Insert pays for a new slot entry unless it finds a dead one to + // reuse, and which of the two it will do is not knowable from out here. + const std::size_t need = tuple.size() + SLOT_SIZE; + + if (page.Contiguous() < need) { + if (page.Reclaimable() < need) return std::nullopt; + if (page.Header().dead_bytes < compact_threshold) return std::nullopt; + page.Compact(); + } + + auto slot = page.Insert(tuple); + if (!slot.has_value()) { + if (slot.error().code() == ErrorCode::kPageFull) return std::nullopt; + return std::unexpected(slot.error()); + } + return slot.value(); +} + +} // namespace + +Result TableHeap::Insert(std::span tuple) { + // Fail before any I/O. HeapPage::Insert enforces both of these too, but only after a probe + // has fetched, latched and possibly compacted its way across the chain. + if (tuple.empty()) { + return std::unexpected(Status::InvalidArgument("cannot insert a zero-length tuple")); + } + if (tuple.size() > MAX_TUPLE_SIZE) { + return std::unexpected(Status::InvalidArgument("tuple exceeds the maximum tuple size")); + } + + // --- v1 rotating insert cursor (DD-004) ------------------------------------------------- + // + // One loop, not two: running off the end of the chain wraps to the front rather than + // terminating, so the hint's starting position needs no special case. + page_id_t page_id = insert_hint_.load(std::memory_order_relaxed); + if (page_id == INVALID_PAGE) page_id = first_page_id_; + + // True only while the page under consideration came from the hint rather than from a + // next_page_id link. See the tolerance rule below. + bool from_hint = true; + + for (int probe = 0; probe < INSERT_PROBE_LIMIT && page_id != INVALID_PAGE; ++probe) { + // A HINT, NEVER A TRUTH. A hint that no longer resolves, or no longer names a heap page, + // costs one wasted fetch and sends the sweep back to the front of the chain — it must + // never fail the insert. A page reached through next_page_id gets no such tolerance: a + // broken chain link is a real error and has to surface. + const bool hint_recoverable = from_hint && page_id != first_page_id_; + + auto guard = bpm_.FetchPageWrite(page_id); + if (!guard.has_value()) { + if (!hint_recoverable) return std::unexpected(guard.error()); + page_id = first_page_id_; + from_hint = false; + continue; + } + + auto heap_page = AsHeapPage(guard.value()); + if (!heap_page.has_value()) { + if (!hint_recoverable) return std::unexpected(heap_page.error()); + page_id = first_page_id_; + from_hint = false; + continue; + } + from_hint = false; + + auto slot = TryInsertOnPage(heap_page.value(), tuple, COMPACT_THRESHOLD); + if (!slot.has_value()) return std::unexpected(slot.error()); + if (slot.value().has_value()) { + // Leave the cursor on the page that worked: it probably still has room. + insert_hint_.store(page_id, std::memory_order_relaxed); + return RID{page_id, slot.value().value()}; + } + + // Full. Advance, wrapping at the end, and publish the cursor so the NEXT insert resumes + // here instead of re-probing these same pages from the start. That carry-over across + // calls is what makes the sweep eventually reach every page in the chain. + page_id_t next = guard.value().Header().next_page_id; + page_id = (next == INVALID_PAGE) ? first_page_id_ : next; + insert_hint_.store(page_id, std::memory_order_relaxed); + } + + // --- append at the last page, extending the chain if it is full -------------------------- + // + // > THE WRITE LATCH ON THE CURRENT LAST PAGE IS THE EXTENSION LOCK. + page_id_t last_id = last_page_id_.load(); + for (;;) { + auto last_guard = bpm_.FetchPageWrite(last_id); + if (!last_guard.has_value()) return std::unexpected(last_guard.error()); + + // Re-read next_page_id UNDER the latch. A thread that extended the chain while we waited + // for it has already published the link, so follow that instead of allocating a second + // page over the top of theirs — which would overwrite one link and leak a page, + // unreachable forever, with no error raised anywhere. + // + // Terminates: the chain is finite and only ever appended to, and the guard drops at the + // continue, so this still holds one latch at a time while it walks. + page_id_t next = last_guard.value().Header().next_page_id; + if (next != INVALID_PAGE) { + last_id = next; + continue; + } + + auto last_heap = AsHeapPage(last_guard.value()); + if (!last_heap.has_value()) return std::unexpected(last_heap.error()); + + // Try here before allocating: bounded probing may never have reached this page, and the + // latch is already held. Threshold 0 — any reclaim beats a whole new page. + auto slot = TryInsertOnPage(last_heap.value(), tuple, 0); + if (!slot.has_value()) return std::unexpected(slot.error()); + if (slot.value().has_value()) { + insert_hint_.store(last_id, std::memory_order_relaxed); + return RID{last_id, slot.value().value()}; + } + + // Genuinely full: extend. The content latch above is held across NewPage, which is + // allowed — DD-002's prohibitions cover the metadata mutex and the shard locks, not + // content latches. Lock order is always old -> new, and no other thread can name the new + // page yet, so there is no cycle. + auto new_guard = bpm_.NewPage(); + if (!new_guard.has_value()) return std::unexpected(new_guard.error()); + + new_guard.value().SetPageType(PageType::HEAP); + new_guard.value().SetNextPageId(INVALID_PAGE); + + auto new_heap = AsHeapPage(new_guard.value()); + if (!new_heap.has_value()) return std::unexpected(new_heap.error()); + new_heap.value().Init(); + + // Cannot fail: MAX_TUPLE_SIZE is chosen so two tuples fit on an empty page, and the + // tuple was bounds-checked on entry. A failure here is a broken invariant, not a full + // page, so it propagates rather than routing anywhere. + auto new_slot = new_heap.value().Insert(tuple); + if (!new_slot.has_value()) return std::unexpected(new_slot.error()); + + const page_id_t new_id = new_guard.value().PageId(); + + // Link, then publish. Nothing here spans the two pages atomically — there is no WAL — so + // a crash between the two writebacks can leave a link to an ALLOCATED page that + // AsHeapPage will reject on the next scan. Known and accepted (DD-004). + last_guard.value().SetNextPageId(new_id); + last_page_id_.store(new_id); + insert_hint_.store(new_id, std::memory_order_relaxed); + + return RID{new_id, new_slot.value()}; + } +} + +Result> TableHeap::Get(RID rid) { + page_id_t page_id = rid.page_id; + slot_id_t slot_id = rid.slot; + + auto page = bpm_.FetchPageRead(page_id); + if (!page.has_value()) return std::unexpected(page.error()); + + auto heap_page = AsHeapPage(page.value()); + if (!heap_page.has_value()) return std::unexpected(heap_page.error()); + + auto tuple = heap_page.value().Get(slot_id); + if (!tuple.has_value()) return std::unexpected(tuple.error()); + + return std::vector(tuple.value().begin(), tuple.value().end()); +} + +Status TableHeap::Delete(RID rid) { + page_id_t page_id = rid.page_id; + slot_id_t slot_id = rid.slot; + + auto page = bpm_.FetchPageWrite(page_id); + if (!page.has_value()) return page.error(); + + auto heap_page = AsHeapPage(page.value()); + if (!heap_page.has_value()) return heap_page.error(); + + auto st = heap_page.value().Delete(slot_id); + if (st.ok()) { + insert_hint_.store(page_id, std::memory_order_relaxed); + } + return st; +} + +Result TableHeap::Update(RID rid, std::span tuple) { + page_id_t page_id = rid.page_id; + slot_id_t slot_id = rid.slot; + + UpdateOutcome outcome{}; + { + // SCOPED so the guard is gone before the relocation below. Insert and Delete take page + // latches of their own — including, very likely, this page's, since insert_hint_ usually + // points at the page a row was just updated or deleted on — and the frame latch is a + // non-recursive shared_mutex. Holding it across those calls self-deadlocks. + // + // It would also make this the one place in the layer that holds two content latches at + // once, which is what would force a lock-ordering rule onto everything else. + auto page = bpm_.FetchPageWrite(page_id); + if (!page.has_value()) return std::unexpected(page.error()); + + auto heap_page = AsHeapPage(page.value()); + if (!heap_page.has_value()) return std::unexpected(heap_page.error()); + + // Errors propagate unchanged: HeapPage::Update is careful to report an update to a + // deleted RID as kNotFound rather than kCorruption, and flattening that here would throw + // away the distinction it went to the trouble of preserving. + auto updated = heap_page.value().Update(slot_id, tuple); + if (!updated.has_value()) return std::unexpected(updated.error()); + outcome = updated.value(); + } + + // Cases 1 and 2: the slot index did not change, so the RID survives. + if (outcome == UpdateOutcome::kSamePage) return rid; + + // kDoesNotFit. The page was left COMPLETELY untouched — case 3 is pure arithmetic run before + // any mutation — so the old row is still live and the order of the next two steps is a real + // choice. Insert first: if it fails, the table is unchanged and the caller still has a valid + // row. Delete first would lose the row outright on a failed insert, and a duplicate is + // recoverable where a vanished row is not. Postgres and InnoDB order it the same way. + auto new_rid = Insert(tuple); + if (!new_rid.has_value()) return std::unexpected(new_rid.error()); + + // Past this point there is no way back: the new copy exists and cannot be withdrawn. If the + // original cannot be removed the row is live in TWO places and every scan returns it twice, + // so the failure has to be loud rather than swallowed. + Status removed = Delete(rid); + + // kNotFound is the one benign failure: another thread deleted the original in the window + // where this function held no latch. The row ends up deleted and reinserted, which is the + // outcome that was wanted anyway. + if (!removed.ok() && removed.code() != ErrorCode::kNotFound) { + return std::unexpected(Status::Internal( + "update relocated the row but could not remove the original; it is now duplicated")); + } + + return new_rid; +} + +TableIterator TableHeap::Scan() { + return TableIterator(bpm_, first_page_id_); +} + +TableIterator::~TableIterator() = default; + +TableIterator::TableIterator(TableIterator&& other) noexcept + : bpm_(other.bpm_), + guard_(std::move(other.guard_)), + initialized_(other.initialized_), + page_id_(other.page_id_), + slot_(other.slot_), + tuple_(std::move(other.tuple_)) { + // Leave the source EXHAUSTED, not merely moved-from. Moving an optional leaves the source + // optional still engaged, holding a moved-from guard — harmless on its own, since Drop() on a + // null bpm_ does nothing — but an iterator that still names a page id is one that can be + // walked a second time over rows this iterator now owns. + other.guard_.reset(); + other.page_id_ = INVALID_PAGE; +} + +TableIterator& TableIterator::operator=(TableIterator&& other) noexcept { + if (this == &other) return *this; + + // Release what THIS iterator is holding BEFORE taking the other's. Overwriting an engaged + // optional without dropping it strands a pin, and a stranded pin never becomes + // evictable again — that frame is lost to the pool for the life of the process. + guard_.reset(); + + bpm_ = other.bpm_; + guard_ = std::move(other.guard_); + initialized_ = other.initialized_; + page_id_ = other.page_id_; + slot_ = other.slot_; + tuple_ = std::move(other.tuple_); + + other.guard_.reset(); + other.page_id_ = INVALID_PAGE; + return *this; +} + +Result TableIterator::Next() { + // Exhausted, or a scan of an empty chain. Idempotent on purpose: calling Next() past the end + // keeps answering false rather than walking off the slot array. + if (page_id_ == INVALID_PAGE) return false; + + if (!initialized_) { + // Positioned BEFORE the first row, so the first call examines slot 0 rather than slot 1. + // This is why the "not started" state is a flag and not slot_ = -1: slot_id_t is + // unsigned, so -1 is 65535, not a sentinel below zero. + initialized_ = true; + slot_ = 0; + } else { + ++slot_; + } + + while (page_id_ != INVALID_PAGE) { + if (!guard_.has_value()) { + auto page = bpm_->FetchPageRead(page_id_); + if (!page.has_value()) return std::unexpected(page.error()); + guard_.emplace(std::move(page.value())); + } + + auto heap_page = AsHeapPage(guard_.value()); + if (!heap_page.has_value()) return std::unexpected(heap_page.error()); + + // Skip a page with nothing live WITHOUT walking its slot array. This is the reason + // live_count is maintained rather than derived — a table that has been drained is mostly + // pages in exactly this state. + if (heap_page.value().LiveCount() > 0) { + const slot_id_t slot_count = heap_page.value().SlotCount(); + for (; slot_ < slot_count; ++slot_) { + // SlotAt rather than Get for the liveness test: Get reports a dead slot with a + // Status, and Status carries a std::string, so probing dead slots through it + // would allocate once per hole on every scan. + if (heap_page.value().SlotAt(slot_).IsDead()) continue; + + auto bytes = heap_page.value().Get(slot_); + if (!bytes.has_value()) return std::unexpected(bytes.error()); + + // COPY OUT. The span points into the frame, and the guard is dropped the moment + // this page is exhausted; the frame is then evictable and those bytes become + // some other page's, still-valid memory that no sanitizer flags. assign reuses + // the vector's capacity, so this is not an allocation per row. + tuple_.assign(bytes.value().begin(), bytes.value().end()); + return true; + } + } + + // Page exhausted. Read the link BEFORE dropping the guard — it lives in the header this + // guard is holding — then release it and move on. ONE GUARD AT A TIME: a scan that kept + // a pin per visited page would exhaust a fixed-size pool and then fail to fetch the next + // page of its own table. + page_id_ = guard_.value().Header().next_page_id; + guard_.reset(); + slot_ = 0; + } + + return false; +} + +RID TableIterator::Rid() const { + return RID(page_id_, slot_); +} + +std::span TableIterator::Tuple() const { + return tuple_; +} + +} // namespace kernsql diff --git a/src/heap/table_heap.hpp b/src/heap/table_heap.hpp new file mode 100644 index 0000000..575fbe2 --- /dev/null +++ b/src/heap/table_heap.hpp @@ -0,0 +1,275 @@ +#pragma once + +#include +#include +#include +#include +#include +#include + +#include "buffer/buffer_pool_manager.hpp" +#include "buffer/page_guard.hpp" +#include "common/status.hpp" +#include "common/types.hpp" + +namespace kernsql { + +/* + * One table's rows, physically (DD-004). A chain of HEAP pages linked through + * PageHeader::next_page_id, plus everything HeapPage deliberately refuses to know. + * + * HeapPage arranges tuples inside ONE body. It cannot say which table it belongs to, cannot + * reach another page, and answers "no room" with an error rather than a decision. TableHeap is + * the layer that holds the chain and makes those decisions, which is what lets it turn a + * meaningless slot_id into a RID{page_id, slot} that names a row for as long as the row lives. + * + * OWNS NOTHING IT USES. It holds a BufferPoolManager& — the same shape as BufferPoolManager + * holding a DiskManager& — plus two page ids and a cursor. Pages belong to the pool, the file + * belongs to the DiskManager, and the two page ids below belong, eventually, to the catalog. + * + * THREADING. Many threads may call one TableHeap at once. It defines no lock of its own: every + * operation takes a page guard, works under that frame's content latch, and releases it, so a + * heap page is single-threaded by construction. The one cross-page hazard is chain extension, + * and it is solved with a latch that was already required — see Insert. + * + * NOT ITS JOB: what a row means (column types, nulls, encoding — it moves opaque blobs, exactly + * like HeapPage), transaction locking (2PL, later), statement semantics such as the + * collect-then-mutate rule DELETE and UPDATE must follow, and crash recovery (there is no WAL, + * which is why nothing here spans pages atomically). + */ +class TableHeap; + +/* + * A forward scan over one table. Holds ONE page guard at a time and copies each tuple out. + * + * One guard at a time because the pool is fixed-size: a scan that pinned a frame per page it had + * visited would exhaust it and then fail to fetch the next page of its own table. + * + * THE TUPLE BYTES ARE COPIED, and the copy lives in this object rather than in the page. That is + * the ConstHeapPage lifetime rule at the one place it is easiest to reintroduce: advancing drops + * the guard, the frame becomes evictable, and a span into the body then reads another page's + * bytes as valid memory — which ASan and valgrind both consider fine. Tuple() therefore points + * into this iterator, stays valid across the guard drop, and is invalidated by the next Next(). + * + * > NEVER MUTATE THIS TABLE THROUGH A LIVE ITERATOR. IT WILL HANG. + * + * A successful Next() returns with the guard STILL HELD, so the calling thread is holding a + * shared content latch on the current row's page for as long as it looks at that row. The + * obvious loop — + * + * while (*it.Next()) { if (pred(it.Tuple())) heap.Delete(it.Rid()); } + * + * — has Delete call FetchPageWrite on the page this same thread already holds a read latch on. + * The frame latch is a std::shared_mutex: not recursive, and no upgrade path. It blocks forever. + * One thread, no race, no interleaving; it hangs on the first matching row, every run. + * + * This is a HARDER constraint than the Halloween rule under TableHeap::Update and it is not the + * same rule. Halloween is about semantics — a relocated row being seen twice — and applies only + * to operations that can move a row. This applies to DELETE too, which relocates nothing. Both + * are satisfied by the same discipline: the executor collects RIDs to completion, lets the + * iterator die, and only then mutates. + * + * Worth a test precisely because the failure mode is a silent hang with no output, which is the + * least debuggable result a test suite can produce. + * + * Deliberately NOT an STL iterator. It is move-only, so it cannot satisfy forward_iterator; and + * advancing fetches a page, so it can fail, which operator++ has no way to report. A cursor with + * a fallible Next() says both of those out loud instead of hiding them behind a sentinel and a + * throwing increment. + */ +class TableIterator { + public: + TableIterator() = delete; + + TableIterator(const TableIterator&) = delete; + TableIterator& operator=(const TableIterator&) = delete; + TableIterator(TableIterator&&) noexcept; + TableIterator& operator=(TableIterator&&) noexcept; + ~TableIterator(); + + /* + * Advance to the next live row. Returns false at the end of the table; an error means a + * fetch failed and the scan cannot continue. + * + * Must be called before the first Rid()/Tuple(): a freshly constructed iterator is + * positioned BEFORE the first row, so a scan is `while (*it.Next()) { ... }` with no + * special case for an empty table. + * + * Skips pages whose live_count is zero without walking their slot arrays — the reason that + * counter is maintained rather than derived. + */ + [[nodiscard]] Result Next(); + + // The current row's identity. Always available, never behind a flag or a second iterator + // type: the iterator already holds the page id to walk the chain and the slot id as its + // cursor, so the RID is something it would otherwise throw away. SELECT ignores it, DELETE + // ... WHERE needs it to name the row it matched, and a B+tree build needs {key -> RID} for + // every row — that pair IS the index's payload. Postgres carries t_self on every seqscan + // tuple for the same reason, with no caller opting in. + // + // Only meaningful once Next() has returned true. Before the first Next() the iterator is + // positioned before the first row and has no identity to report; after Next() returns false + // the page id is INVALID_PAGE, so the RID answers isValid() == false rather than naming a + // row that is not there. + [[nodiscard]] RID Rid() const; + + // The current row's bytes. Points into this iterator, NOT into the page, and is invalidated + // by the next Next() — copy out anything that must outlive the step. Unlike the RID, which + // is a value and stays meaningful forever. + // + // Empty until Next() has returned true, for the same reason Rid() is meaningless until then. + [[nodiscard]] std::span Tuple() const; + + private: + friend class TableHeap; + + TableIterator(BufferPoolManager& bpm, page_id_t first_page_id) + : bpm_(&bpm), page_id_(first_page_id) {} + + BufferPoolManager* bpm_; + + // No default constructor on a guard, by design, so "no page held" has to be spelled with an + // optional rather than an inert guard object. + std::optional guard_; + + // "Not started" is a flag rather than slot_ = -1 because slot_id_t is UNSIGNED: -1 is 65535, + // which is a plausible slot index and not a sentinel below the first one. + bool initialized_{false}; + + page_id_t page_id_; + + // Zero-initialised so that Rid() on an unstarted iterator returns a wrong answer rather than + // reading an uninitialised member. Misuse should be diagnosable, not undefined. + slot_id_t slot_{0}; + + std::vector tuple_; +}; + +class TableHeap { + public: + /* + * Create a new, empty table: allocate one page, stamp page_type = HEAP through the guard, + * Init() its body, and leave next_page_id invalid. The caller persists FirstPageId() and + * LastPageId() — today into a test, later into the catalog. + * + * Open an existing one from the two ids the catalog handed back. Neither reads a page to + * validate the chain: the first fetch does that anyway, through AsHeapPage. + * + * Factories rather than constructors because both can fail, and unique_ptr rather than a + * value because insert_hint_ is an atomic and atomics are not movable — the same reason + * DiskManager::Open returns one. + */ + [[nodiscard]] static Result> Create(BufferPoolManager& bpm); + [[nodiscard]] static Result> Open(BufferPoolManager& bpm, + page_id_t first_page_id, + page_id_t last_page_id); + + TableHeap(const TableHeap&) = delete; + TableHeap& operator=(const TableHeap&) = delete; + TableHeap(TableHeap&&) = delete; + TableHeap& operator=(TableHeap&&) = delete; + + /* + * Insert a row and return the RID that names it. + * + * Tries candidate pages, and when every one is full the chain must be extended — the only + * place two threads collide in this layer. + * + * > THE WRITE LATCH ON THE CURRENT LAST PAGE IS THE EXTENSION LOCK. + * + * A thread that finds the last page full KEEPS its write guard while it calls NewPage, + * stamps HEAP, Inits, sets the old page's next_page_id, and publishes the new last_page_id. + * A second inserter blocks on that guard; when it acquires it, next_page_id is no longer + * INVALID_PAGE, so it follows the link instead of allocating. No new lock and no new + * ordering rule — the latch that was already required does the job. + * + * Getting it wrong is not subtle in consequence and very subtle in symptom: both threads + * allocate, both set next_page_id, one link is overwritten, and a page is leaked and + * unreachable forever with no error raised anywhere. + * + * Note this holds a content latch across NewPage, which can block on disk allocation. That + * is allowed: DD-002's prohibitions are about the metadata mutex and the shard locks. A + * content latch is held across I/O routinely — it is what the Loading state exists for. + * + * Errors: kInvalidArgument for an empty tuple or one over MAX_TUPLE_SIZE (HeapPage enforces + * both), and whatever NewPage reports when the chain cannot grow. + */ + [[nodiscard]] Result Insert(std::span tuple); + + /* + * Read a row. Returns a COPY, and that is not negotiable: a span into a page body dies when + * this function's guard drops, at which point the frame may be evicted and refilled with a + * different page — still-valid memory holding another table's bytes, which no sanitizer + * flags. If per-row allocation ever shows up in a profile, add an overload that fills a + * caller-supplied buffer; do not "optimise" this by handing back a view. + * + * kNotFound for a row that has been deleted, or a slot that never existed. A RID handed out + * before a delete gets an answer, never bytes. + */ + [[nodiscard]] Result> Get(RID rid); + + // kNotFound if the row is already gone. The bytes are not moved; the space is reclaimed by + // the next compaction of that page. + [[nodiscard]] Status Delete(RID rid); + + /* + * Overwrite a row and return the RID IT NOW HAS, which may differ from the one passed in. + * + * HeapPage::Update resolves to kSamePage or kDoesNotFit. The second case is this layer's to + * answer: the row outgrew its page, so it is deleted here and inserted elsewhere, and its + * identity changes. Returning the new RID forces every caller to hold the new value — + * returning a Status instead would let an index entry, or a collected RID list, go stale + * with nothing at the call site to notice. + * + * BEWARE THE RELOCATION DURING A SCAN. A row moved this way can land on a page the scan has + * not reached yet, match the predicate again, and be updated twice — the Halloween problem. + * Postgres is immune because a tuple written by the current command is invisible to it + * (cmin/cmax); this engine has no visibility machinery, so the executor must collect RIDs to + * completion first and only then apply. That rule is imposed by this layer and enforced + * above it. + * + * TableIterator imposes a second and stricter reason for the same discipline: mutating + * through a live iterator deadlocks on the read latch the iterator is still holding. That one + * catches DELETE too, which relocates nothing and has no Halloween exposure. See TableIterator. + * + * kNotFound if the row is gone. + */ + [[nodiscard]] Result Update(RID rid, std::span tuple); + + // A scan positioned before the first row; call Next() to reach it. + [[nodiscard]] TableIterator Scan(); + + // For the catalog to persist. LastPageId() changes as the chain grows, which is exactly why + // whoever stores it has to be told again. + [[nodiscard]] page_id_t FirstPageId() const { return first_page_id_; } + [[nodiscard]] page_id_t LastPageId() const { return last_page_id_.load(); } + + private: + TableHeap(BufferPoolManager& bpm, page_id_t first_page_id, page_id_t last_page_id); + + BufferPoolManager& bpm_; + + // Fixed for the life of the table. The chain is only ever appended to. + const page_id_t first_page_id_; + + // Atomic because Insert publishes it under the extension latch but every other caller reads + // it without one. + std::atomic last_page_id_; + + /* + * v1 free-space reuse (DD-004): where the next Insert starts looking. Rotates along the + * chain for a bounded number of probes and falls back to appending at the last page. + * + * A HINT, NEVER A TRUTH. Never persisted, re-verified against the real page before use, and + * harmless when stale — a wrong value costs one wasted fetch, never corruption. That single + * property is what lets a cross-page structure exist in an engine with no WAL, and it is why + * this needs no latch of its own and imposes no lock-ordering rule. + * + * Its honest weakness is that it stumbles into space rather than finding it. When that stops + * being enough, DD-004 designs a per-table free-space map to replace it; the FSM changes + * nothing about the on-disk page format, which is why it can wait. + */ + std::atomic insert_hint_; +}; + +} // namespace kernsql diff --git a/src/shell/main.cpp b/src/shell/main.cpp index da6524e..71b9909 100644 --- a/src/shell/main.cpp +++ b/src/shell/main.cpp @@ -14,17 +14,22 @@ #include #include #include +#include #include +#include #include #include #include #include +#include #include "buffer/buffer_pool_manager.hpp" #include "buffer/page_guard.hpp" #include "buffer/pool_stats.hpp" #include "common/status.hpp" #include "common/types.hpp" +#include "heap/heap_page.hpp" +#include "heap/table_heap.hpp" #include "storage/disk_manager.hpp" using namespace kernsql; @@ -115,6 +120,173 @@ void CmdRead(BufferPoolManager& bpm, page_id_t page_id) { std::string_view(reinterpret_cast(body.data()), len)); } +bool ParseSlot(std::string_view token, slot_id_t& out) { + unsigned value{}; + const auto* first = token.data(); + const auto* last = token.data() + token.size(); + auto [ptr, ec] = std::from_chars(first, last, value); + if (ec != std::errc{} || ptr != last || value > 0xFFFFU) return false; + out = static_cast(value); + return true; +} + +// The heap moves opaque blobs, so the shell picks one encoding and stays out of the way. No NUL +// terminator and no padding: a tuple is exactly the bytes typed, which is what makes the length +// accounting in `tdump` mean something. +std::span AsBytes(std::string_view text) { + return {reinterpret_cast(text.data()), text.size()}; +} +std::string_view AsText(std::span bytes) { + return {reinterpret_cast(bytes.data()), bytes.size()}; +} + +void CmdTableCreate(BufferPoolManager& bpm, std::unique_ptr& heap) { + auto created = TableHeap::Create(bpm); + if (!created) { + std::println("tcreate: {}", created.error().message()); + return; + } + heap = std::move(*created); + std::println("tcreate: first={} last={}", heap->FirstPageId(), heap->LastPageId()); + std::println(" persist those two ids — the catalog will own them eventually"); +} + +void CmdTableOpen(BufferPoolManager& bpm, std::unique_ptr& heap, page_id_t first, + page_id_t last) { + auto opened = TableHeap::Open(bpm, first, last); + if (!opened) { + std::println("topen: {}", opened.error().message()); + return; + } + heap = std::move(*opened); + // Open validates nothing on purpose — the first fetch does it through AsHeapPage. A wrong + // pair of ids here surfaces as kCorruption on the next command, not now. + std::println("topen: first={} last={}", heap->FirstPageId(), heap->LastPageId()); +} + +void CmdTableInsert(TableHeap& heap, std::string_view text) { + auto rid = heap.Insert(AsBytes(text)); + if (!rid) { + std::println("tinsert: {}", rid.error().message()); + return; + } + std::println("tinsert: {} bytes -> {{{}, {}}} (last page now {})", text.size(), rid->page_id, + rid->slot, heap.LastPageId()); +} + +void CmdTableGet(TableHeap& heap, RID rid) { + auto row = heap.Get(rid); + if (!row) { + std::println("tget: {}", row.error().message()); + return; + } + std::println("tget: {{{}, {}}} -> \"{}\"", rid.page_id, rid.slot, AsText(*row)); +} + +void CmdTableUpdate(TableHeap& heap, RID rid, std::string_view text) { + auto moved = heap.Update(rid, AsBytes(text)); + if (!moved) { + std::println("tupdate: {}", moved.error().message()); + return; + } + // The RID is the interesting output, not the status: a row that outgrew its page was deleted + // and reinserted, and every index entry or collected RID naming the old one is now stale. + if (*moved == rid) + std::println("tupdate: {{{}, {}}} updated in place", rid.page_id, rid.slot); + else + std::println("tupdate: {{{}, {}}} RELOCATED to {{{}, {}}}", rid.page_id, rid.slot, + moved->page_id, moved->slot); +} + +void CmdTableScan(TableHeap& heap) { + auto it = heap.Scan(); + std::size_t rows = 0; + for (;;) { + auto more = it.Next(); + if (!more) { + std::println("tscan: {}", more.error().message()); + return; + } + if (!*more) break; + const RID rid = it.Rid(); + std::println(" {{{}, {}}} \"{}\"", rid.page_id, rid.slot, AsText(it.Tuple())); + ++rows; + } + std::println("tscan: {} rows", rows); +} + +// Bulk insert, to push the table past one page and exercise chain extension and the insert hint. +void CmdTableFill(TableHeap& heap, int count, std::string_view prefix) { + const page_id_t last_before = heap.LastPageId(); + for (int i = 0; i < count; ++i) { + const std::string row = std::format("{}-{:06d}", prefix, i); + auto rid = heap.Insert(AsBytes(row)); + if (!rid) { + std::println("tfill: stopped after {} rows: {}", i, rid.error().message()); + return; + } + } + std::println("tfill: {} rows, last page {} -> {}", count, last_before, heap.LastPageId()); +} + +/* + * Delete every row containing `needle` — and the point of the command is the two-phase shape. + * + * COLLECT TO COMPLETION, THEN MUTATE. Calling Delete inside the scan loop would have it take a + * write latch on the page the iterator is still holding a read latch on; the frame latch is a + * non-recursive shared_mutex, so it would hang on the first match, on one thread, every run. The + * scan is scoped so the iterator is destroyed — and its guard released — before any delete runs. + */ +void CmdTablePurge(TableHeap& heap, std::string_view needle) { + std::vector matched; + { + auto it = heap.Scan(); + for (;;) { + auto more = it.Next(); + if (!more) { + std::println("tpurge: {}", more.error().message()); + return; + } + if (!*more) break; + if (AsText(it.Tuple()).find(needle) != std::string_view::npos) + matched.push_back(it.Rid()); + } + } + + std::size_t deleted = 0; + for (const RID& rid : matched) { + const Status st = heap.Delete(rid); + if (st.ok()) + ++deleted; + else + std::println(" {{{}, {}}}: {}", rid.page_id, rid.slot, st.message()); + } + std::println("tpurge: {} matched, {} deleted", matched.size(), deleted); +} + +// The accounting view. dead_bytes rising while contiguous stays put is a delete; reclaimable +// collapsing back into contiguous is a compaction. CheckInvariants is the real prize — it proves +// every byte of the tuple region belongs to exactly one live tuple or to nobody. +void CmdTableDump(BufferPoolManager& bpm, page_id_t page_id) { + auto guard = bpm.FetchPageRead(page_id); + if (!guard) { + std::println("tdump: {}", guard.error().message()); + return; + } + auto page = AsHeapPage(*guard); + if (!page) { + std::println("tdump: {}", page.error().message()); + return; + } + + const HeapSubHeader sub = page->Header(); + std::println("tdump: page {} next={}", page_id, guard->Header().next_page_id); + std::println(" slots={} live={} dead_bytes={} tuple_data_start={}", sub.slot_count, + sub.live_count, sub.dead_bytes, sub.tuple_data_start); + std::println(" contiguous={} reclaimable={}", page->Contiguous(), page->Reclaimable()); + std::println(" invariants: {}", page->CheckInvariants() ? "ok" : "BROKEN"); +} + void CmdHelp() { std::println("commands:"); std::println(" new allocate a page and return its id"); @@ -124,12 +296,24 @@ void CmdHelp() { std::println(" flush write one page through to the file"); std::println(" flushall write every resident page through"); std::println(" stat page count on disk"); + std::println("table (heap layer):"); + std::println(" tcreate new empty table; prints first/last page id"); + std::println(" topen reopen one from those two ids"); + std::println(" tinsert insert a row, print its RID"); + std::println(" tget read one row"); + std::println(" tupdate overwrite; prints the RID it now has"); + std::println(" tdelete delete one row"); + std::println(" tscan walk every live row"); + std::println(" tfill bulk insert, to force chain extension"); + std::println(" tpurge collect-then-delete every row containing "); + std::println(" tdump heap page accounting + invariant check"); std::println(" help this list"); std::println(" quit shutdown (flush + fsync) and exit"); } // Returns false when the REPL should stop. -bool Dispatch(BufferPoolManager& bpm, DiskManager& dm, std::string_view line) { +bool Dispatch(BufferPoolManager& bpm, DiskManager& dm, std::unique_ptr& heap, + std::string_view line) { std::string_view rest = line; const std::string_view cmd = NextToken(rest); if (cmd.empty()) return true; @@ -140,6 +324,20 @@ bool Dispatch(BufferPoolManager& bpm, DiskManager& dm, std::string_view line) { return false; }; + auto needs_rid = [&](RID& rid) { + if (!ParsePageId(NextToken(rest), rid.page_id) || !ParseSlot(NextToken(rest), rid.slot)) { + std::println("{}: expected ", cmd); + return false; + } + return true; + }; + + auto needs_table = [&]() -> TableHeap* { + if (heap) return heap.get(); + std::println("{}: no table — run `tcreate` or `topen` first", cmd); + return nullptr; + }; + if (cmd == "quit" || cmd == "exit") return false; if (cmd == "help") { CmdHelp(); @@ -166,6 +364,39 @@ bool Dispatch(BufferPoolManager& bpm, DiskManager& dm, std::string_view line) { } else if (cmd == "flush") { page_id_t id{}; if (needs_page_id(id)) PrintStatus("flush", bpm.FlushPage(id)); + } else if (cmd == "tcreate") { + CmdTableCreate(bpm, heap); + } else if (cmd == "topen") { + page_id_t first{}; + page_id_t last{}; + if (ParsePageId(NextToken(rest), first) && ParsePageId(NextToken(rest), last)) + CmdTableOpen(bpm, heap, first, last); + else + std::println("topen: expected "); + } else if (cmd == "tinsert") { + if (auto* t = needs_table()) CmdTableInsert(*t, rest); + } else if (cmd == "tget") { + RID rid{}; + if (auto* t = needs_table(); t && needs_rid(rid)) CmdTableGet(*t, rid); + } else if (cmd == "tupdate") { + RID rid{}; + if (auto* t = needs_table(); t && needs_rid(rid)) CmdTableUpdate(*t, rid, rest); + } else if (cmd == "tdelete") { + RID rid{}; + if (auto* t = needs_table(); t && needs_rid(rid)) PrintStatus("tdelete", t->Delete(rid)); + } else if (cmd == "tscan") { + if (auto* t = needs_table()) CmdTableScan(*t); + } else if (cmd == "tfill") { + page_id_t count{}; + if (auto* t = needs_table(); t && ParsePageId(NextToken(rest), count) && count > 0) + CmdTableFill(*t, count, rest.empty() ? "row" : rest); + else if (heap) + std::println("tfill: expected , n > 0"); + } else if (cmd == "tpurge") { + if (auto* t = needs_table()) CmdTablePurge(*t, rest); + } else if (cmd == "tdump") { + page_id_t id{}; + if (needs_page_id(id)) CmdTableDump(bpm, id); } else { std::println("unknown command '{}' — try `help`", cmd); } @@ -190,6 +421,11 @@ int main(int argc, char** argv) { std::println("kernSQL — {} ({} pages)", db_path.string(), (*dm)->PageCount()); std::println("`help` for commands, `quit` to shut down cleanly."); + // Outlives the loop but is destroyed BEFORE Shutdown() below, which matters: the heap holds + // no pins of its own, but an iterator inside one of the commands does, and Shutdown() verifies + // quiescence rather than arranging it. + std::unique_ptr heap; + std::string line; while (true) { std::print("kernsql> "); @@ -201,9 +437,11 @@ int main(int argc, char** argv) { std::println(""); // EOF (ctrl-D) — treat as a clean quit break; } - if (!Dispatch(bpm, **dm, line)) break; + if (!Dispatch(bpm, **dm, heap, line)) break; } + heap.reset(); // before Shutdown(), see above + // Shutdown() is the durable operation, not the destructor (DD-002). Quiescence is trivially // satisfied here — the REPL is single-threaded and every guard was function-scoped, so no // pin outlives the loop. Skipping this would trip the destructor's backstop, which logs the diff --git a/test/buffer/buffer_pool_manager_test.cpp b/test/buffer/buffer_pool_manager_test.cpp index 876aa20..c87bc93 100644 --- a/test/buffer/buffer_pool_manager_test.cpp +++ b/test/buffer/buffer_pool_manager_test.cpp @@ -13,7 +13,6 @@ #include #include #include -#include #include #include #include diff --git a/test/heap/heap_page_test.cpp b/test/heap/heap_page_test.cpp new file mode 100644 index 0000000..639c3e2 --- /dev/null +++ b/test/heap/heap_page_test.cpp @@ -0,0 +1,1110 @@ +#include "heap/heap_page.hpp" + +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include "buffer/buffer_pool_manager.hpp" +#include "buffer/page_guard.hpp" +#include "common/status.hpp" +#include "common/types.hpp" +#include "storage/disk_manager.hpp" + +namespace kernsql { + +class HeapPageTest : public ::testing::Test { + protected: + // A bare body: no DiskManager, no BufferPoolManager, no guard. That is the entire point of + // the view types, and it is what lets these tests assert against the on-disk layout itself — + // header fields and slot entries — rather than only against what the API chooses to report. + void SetUp() override { + body_.fill(std::byte{0}); + Page().Init(); + } + + HeapPage Page() { return HeapPage(body_); } + ConstHeapPage View() const { return ConstHeapPage(body_); } + + // n bytes all equal to `fill`, so a round-trip failure tells you WHICH tuple moved wrong + // rather than just that some bytes differ. + static std::vector Tuple(std::byte fill, std::size_t n) { + return std::vector(n, fill); + } + + static std::span Bytes(std::string_view s) { + return std::as_bytes(std::span{s}); + } + + static std::string_view AsChars(std::span bytes) { + return {reinterpret_cast(bytes.data()), bytes.size()}; + } + + // Materialises a span so EXPECT_EQ compares CONTENT. Comparing two spans — or two data() + // pointers — compares addresses, which is never what a round-trip assertion means, and it + // fails in a way that looks like a real mismatch. + static std::vector Read(std::span bytes) { + return {bytes.begin(), bytes.end()}; + } + + // Stamp header and slot bytes straight into the body, bypassing every operation. ONLY for + // the CheckInvariants tests: their whole point is to hand that function a page the API + // cannot produce, and it is the HEADER that every operation reads, so scribbling tuple bytes + // would change nothing anything looks at. + void PokeHeader(const HeapSubHeader& header) { + header.WriteTo( + std::span{body_.data(), HEAP_SUB_HEADER_SIZE}); + } + void PokeSlot(slot_id_t slot, const Slot& entry) { + entry.WriteTo(std::span{ + body_.data() + HEAP_SUB_HEADER_SIZE + std::size_t{slot} * SLOT_SIZE, SLOT_SIZE}); + } + + std::array body_{}; +}; + +// --------------------------------------------------------------------------------------------- +// Init and the empty page +// --------------------------------------------------------------------------------------------- + +// slot_count/live_count/dead_bytes zero, tuple_data_start at PAGE_BODY_SIZE, invariants hold. +TEST_F(HeapPageTest, FreshPageIsEmptyAndFullyFree) { + SetUp(); + HeapPage page = Page(); + auto header = page.Header(); + + EXPECT_EQ(header.slot_count, 0); + EXPECT_EQ(header.live_count, 0); + EXPECT_EQ(header.dead_bytes, 0); + EXPECT_EQ(header.tuple_data_start, PAGE_BODY_SIZE); +} + +// Contiguous() is the body less the sub-header; Reclaimable() equals it when there is no garbage. +TEST_F(HeapPageTest, FreshPageReportsAllSpaceAsFree) { + SetUp(); + HeapPage page = Page(); + + EXPECT_EQ(page.Reclaimable(), page.Contiguous()); +} + +TEST_F(HeapPageTest, GetOnAnEmptyPageIsNotFound) { + SetUp(); + HeapPage page = Page(); + + EXPECT_FALSE(page.Get(2).has_value()); + EXPECT_EQ(page.Get(2).error().code(), ErrorCode::kNotFound); +} + +// --------------------------------------------------------------------------------------------- +// Insert +// --------------------------------------------------------------------------------------------- + +// Slot ids come back 0, 1, 2...; every tuple reads back byte-identical. +TEST_F(HeapPageTest, InsertReturnsSequentialSlotIdsAndRoundTripsBytes) { + SetUp(); + HeapPage page = Page(); + + for (std::size_t i = 0; i < 5; i++) { + auto slot = page.Insert(Bytes(std::format("Slot {}", i))); + EXPECT_EQ(slot.value(), i); + } + + for (std::size_t i = 0; i < 5; i++) { + auto tuple = page.Get(i); + EXPECT_EQ(AsChars(tuple.value()), std::format("Slot {}", i)); + } +} + +// Tuples grow backward: each insert lowers tuple_data_start by exactly the tuple length. +TEST_F(HeapPageTest, InsertLowersTheLowWaterMarkByTheTupleLength) { + SetUp(); + HeapPage page = Page(); + uint16_t cursor{PAGE_BODY_SIZE}; + + for (std::size_t i = 0; i < 5; i++) { + std::string write_tuple = std::format("Slot {}", i); + auto slot = page.Insert(Bytes(write_tuple)); + auto header = page.Header(); + cursor -= write_tuple.size(); + EXPECT_EQ(header.tuple_data_start, cursor); + } +} + +TEST_F(HeapPageTest, InsertRejectsAZeroLengthTuple) { + SetUp(); + HeapPage page = Page(); + auto slot = page.Insert(Bytes("")); + + EXPECT_FALSE(slot.has_value()); + EXPECT_EQ(slot.error().code(), ErrorCode::kInvalidArgument); +} + +// Fill the page, then assert the next insert is PageFull and the page is unchanged. +TEST_F(HeapPageTest, InsertOnAFullPageReturnsPageFullAndChangesNothing) { + HeapPage page = Page(); + + // Filled through the API rather than by scribbling bytes into the body. Insert decides on the + // HEADER and never on page content, so bytes written behind its back fill nothing: the page + // still reports every byte free and the insert under test would succeed. + const auto filler = Tuple(std::byte{29}, MAX_TUPLE_SIZE); + while (page.Insert(filler).has_value()) { + } + + const std::array before = body_; + + auto tuple_insert = page.Insert(Tuple(std::byte{42}, 100)); + EXPECT_FALSE(tuple_insert.has_value()); + EXPECT_EQ(tuple_insert.error().code(), ErrorCode::kPageFull); + + // The other half of the name: a failed insert must not have moved the header first. + EXPECT_EQ(std::memcmp(before.data(), body_.data(), PAGE_BODY_SIZE), 0); + EXPECT_TRUE(page.CheckInvariants()); +} + +// The 2000-byte cap exists to guarantee this; if it ever stops holding, the cap is wrong. +TEST_F(HeapPageTest, TwoMaxSizeTuplesFitOnOnePage) { + HeapPage page = Page(); + + // 2 * (2000 + 4) + 8 == 4016 <= 4064. This is the entire reason the cap is 2000 rather than + // the 4052 the format could physically hold: one row per page turns a heap into a linked list. + EXPECT_TRUE(page.Insert(Tuple(std::byte{1}, MAX_TUPLE_SIZE)).has_value()); + EXPECT_TRUE(page.Insert(Tuple(std::byte{2}, MAX_TUPLE_SIZE)).has_value()); + + EXPECT_EQ(page.LiveCount(), 2); + EXPECT_TRUE(page.CheckInvariants()); +} + +TEST_F(HeapPageTest, InsertRejectsATupleOverMaxTupleSize) { + HeapPage page = Page(); + auto slot = page.Insert(Tuple(std::byte{29}, MAX_TUPLE_SIZE + 1)); + + EXPECT_FALSE(slot.has_value()); + EXPECT_EQ(slot.error().code(), ErrorCode::kInvalidArgument); +} + +// Reuses the slot ENTRY (slot_count does not grow) but dead_bytes stays put — the dead tuple's +// bytes are unreachable until Compact. This is the accounting rule most likely to be got wrong. +TEST_F(HeapPageTest, InsertReusesADeadSlotWithoutReclaimingItsBytes) { + HeapPage page = Page(); + auto slot_insert = page.Insert(Tuple(std::byte{29}, 100)); + ASSERT_TRUE(slot_insert.has_value()); + auto slot_id = slot_insert.value(); + + auto st = page.Delete(slot_insert.value()); + ASSERT_EQ(st.code(), ErrorCode::kOk); + + uint16_t init_dead_bytes = page.Header().dead_bytes; + slot_insert = page.Insert(Tuple(std::byte{29}, 50)); // reuse + ASSERT_TRUE(slot_insert.has_value()); + ASSERT_EQ(slot_insert.value(), slot_id); + + ASSERT_EQ(page.Header().dead_bytes, init_dead_bytes); +} + +// With several dead slots, the lowest index is taken first. +TEST_F(HeapPageTest, InsertReusesTheLowestDeadSlot) { + HeapPage page = Page(); + + std::vector slots; + for (size_t i = 0; i < 13; i++) { + auto slot_insert = page.Insert(Tuple(std::byte{29}, 150)); + EXPECT_TRUE(slot_insert.has_value()); + slots.push_back(slot_insert.value()); + } + + for (size_t i = 0; i < 13; i++) { + auto st = page.Delete(slots[i]); + EXPECT_EQ(st.code(), ErrorCode::kOk); + } + + auto slot_insert = page.Insert(Tuple(std::byte{29}, 150)); + EXPECT_TRUE(slot_insert.has_value()); + ASSERT_EQ(slot_insert.value(), slots.front()); +} + +// Insert does not compact, by contract: it fails while Reclaimable() says the space exists. +TEST_F(HeapPageTest, InsertDoesNotCompactOnItsOwn) { + HeapPage page = Page(); + + // Two max-size tuples fill the page down to 48 contiguous bytes; deleting the first turns + // 2000 of those bytes into garbage that only Compact can recover. + auto first = page.Insert(Tuple(std::byte{1}, MAX_TUPLE_SIZE)); + auto second = page.Insert(Tuple(std::byte{2}, MAX_TUPLE_SIZE)); + ASSERT_TRUE(first.has_value()); + ASSERT_TRUE(second.has_value()); + ASSERT_TRUE(page.Delete(*first).ok()); + + // The premise, stated rather than assumed: the space exists, but not contiguously. + constexpr std::size_t kWanted = 1000; + ASSERT_LT(page.Contiguous(), kWanted); + ASSERT_GE(page.Reclaimable(), kWanted); + + const std::array before = body_; + + auto blocked = page.Insert(Tuple(std::byte{3}, kWanted)); + EXPECT_FALSE(blocked.has_value()); + EXPECT_EQ(blocked.error().code(), ErrorCode::kPageFull); + + // The point of the test. Insert hands the decision back to the caller instead of compacting + // on its own, so the garbage is still garbage and the page is untouched — byte for byte, + // which is the only assertion that would catch a compaction that ran and then failed anyway. + EXPECT_EQ(page.Header().dead_bytes, MAX_TUPLE_SIZE); + EXPECT_GE(page.Reclaimable(), kWanted); + EXPECT_EQ(std::memcmp(before.data(), body_.data(), PAGE_BODY_SIZE), 0); + EXPECT_TRUE(page.CheckInvariants()); +} + +// --------------------------------------------------------------------------------------------- +// Delete +// --------------------------------------------------------------------------------------------- + +// live_count drops, dead_bytes rises by exactly the tuple length, slot reads {0, 0}. +TEST_F(HeapPageTest, DeleteMarksTheSlotDeadAndAccountsItsBytes) { + HeapPage page = Page(); + for (size_t i = 0; i < 13; i++) { + auto slot_insert = page.Insert(Tuple(std::byte{29}, 150)); + EXPECT_TRUE(slot_insert.has_value()); + } + + uint16_t init_live_count = page.Header().live_count; + + auto st = page.Delete(1); + ASSERT_EQ(st.code(), ErrorCode::kOk); + + uint16_t final_live_count = page.Header().live_count; + uint16_t final_dead_bytes = page.Header().dead_bytes; + + ASSERT_EQ(init_live_count, final_live_count + 1); + ASSERT_EQ(final_dead_bytes, 150); +} + +// slot_count is unchanged, so a stale RID still resolves — to NotFound, never to bytes. +TEST_F(HeapPageTest, DeletedSlotKeepsItsIndexForever) { + HeapPage page = Page(); + auto first = page.Insert(Bytes("first")); + auto middle = page.Insert(Bytes("middle")); + auto last = page.Insert(Bytes("last")); + ASSERT_TRUE(first.has_value() && middle.has_value() && last.has_value()); + + ASSERT_TRUE(page.Delete(*middle).ok()); + + // The index survives: the array does not shrink and nothing after it slides down. + EXPECT_EQ(page.SlotCount(), 3); + EXPECT_EQ(page.LiveCount(), 2); + EXPECT_TRUE(page.SlotAt(*middle).IsDead()); + + // Dead has exactly one representation, which is what makes offset == 0 => length == 0 + // assertable in CheckInvariants. + EXPECT_EQ(page.SlotAt(*middle).offset, 0); + EXPECT_EQ(page.SlotAt(*middle).length, 0); + + // A stale RID gets an answer, not garbage, and its neighbours are untouched. + EXPECT_FALSE(page.Get(*middle).has_value()); + EXPECT_EQ(page.Get(*middle).error().code(), ErrorCode::kNotFound); + EXPECT_EQ(AsChars(page.Get(*first).value()), "first"); + EXPECT_EQ(AsChars(page.Get(*last).value()), "last"); + + // "Forever" is the part worth testing: a compaction rewrites every offset in the page and the + // dead slot still sits at its own index, still dead, while the live RIDs still resolve. + page.Compact(); + EXPECT_EQ(page.SlotCount(), 3); + EXPECT_TRUE(page.SlotAt(*middle).IsDead()); + EXPECT_EQ(AsChars(page.Get(*first).value()), "first"); + EXPECT_EQ(AsChars(page.Get(*last).value()), "last"); + EXPECT_TRUE(page.CheckInvariants()); +} + +TEST_F(HeapPageTest, DeleteOnADeadSlotIsNotFound) { + HeapPage page = Page(); + auto slot = page.Insert(Tuple(std::byte{7}, 150)); + ASSERT_TRUE(slot.has_value()); + ASSERT_TRUE(page.Delete(*slot).ok()); + + const std::array before = body_; + + auto again = page.Delete(*slot); + EXPECT_EQ(again.code(), ErrorCode::kNotFound); + + // The real risk in a double delete is not the error code, it is dead_bytes counting the same + // 150 bytes twice and live_count going negative — both of which would corrupt the accounting + // permanently, since nothing ever subtracts from dead_bytes. + EXPECT_EQ(page.Header().dead_bytes, 150); + EXPECT_EQ(page.LiveCount(), 0); + EXPECT_EQ(std::memcmp(before.data(), body_.data(), PAGE_BODY_SIZE), 0); + EXPECT_TRUE(page.CheckInvariants()); +} + +TEST_F(HeapPageTest, DeleteOnAnOutOfRangeSlotIsNotFound) { + HeapPage page = Page(); + auto slot = page.Insert(Tuple(std::byte{7}, 150)); + ASSERT_TRUE(slot.has_value()); + + const std::array before = body_; + + // One past the end is the boundary that separates a correct `slot_count <= slot_id` from a + // buggy `<`; the far-out value only checks that nothing indexes wildly. + EXPECT_EQ(page.Delete(page.SlotCount()).code(), ErrorCode::kNotFound); + EXPECT_EQ(page.Delete(9999).code(), ErrorCode::kNotFound); + + EXPECT_EQ(std::memcmp(before.data(), body_.data(), PAGE_BODY_SIZE), 0); + EXPECT_TRUE(page.CheckInvariants()); +} + +// Deleting everything leaves live_count 0 with tuple_data_start still low — the space is only +// recovered by Compact, and Reclaimable() is the number that knows it. +TEST_F(HeapPageTest, DeletingEveryTupleLeavesTheSpaceReclaimableButNotContiguous) { + constexpr slot_id_t kTuples = 5; + constexpr std::size_t kLength = 300; + + HeapPage page = Page(); + for (slot_id_t i = 0; i < kTuples; ++i) { + ASSERT_TRUE(page.Insert(Tuple(std::byte{9}, kLength)).has_value()); + } + + const auto low_water_mark = page.Header().tuple_data_start; + const std::size_t contiguous_before = page.Contiguous(); + + for (slot_id_t i = 0; i < kTuples; ++i) { + ASSERT_TRUE(page.Delete(i).ok()); + } + + EXPECT_EQ(page.LiveCount(), 0); + EXPECT_EQ(page.SlotCount(), kTuples); // slots outlive their tuples + + // Delete does not move a single byte, so the low-water mark and the contiguous gap are exactly + // where they were. An empty page that still reports itself nearly full is the correct state. + EXPECT_EQ(page.Header().tuple_data_start, low_water_mark); + EXPECT_EQ(page.Contiguous(), contiguous_before); + + // Reclaimable() is the only number that knows the space is recoverable, and it is what a + // cross-page free-space mechanism would publish. Everything but the sub-header and the slot + // array is now free — after a compaction nobody has run yet. + EXPECT_EQ(page.Header().dead_bytes, kTuples * kLength); + EXPECT_EQ(page.Reclaimable(), page.Contiguous() + kTuples * kLength); + EXPECT_EQ(page.Reclaimable(), PAGE_BODY_SIZE - HEAP_SUB_HEADER_SIZE - kTuples * SLOT_SIZE); + EXPECT_TRUE(page.CheckInvariants()); +} + +// --------------------------------------------------------------------------------------------- +// Update — case 1, in place +// --------------------------------------------------------------------------------------------- + +// Same offset, shorter length, and the difference lands in dead_bytes. Those orphaned bytes are +// the ones no slot walk can ever rediscover, which is why dead_bytes is stored at all. +TEST_F(HeapPageTest, UpdateToAShorterTupleOverwritesInPlace) { + HeapPage page = Page(); + + auto slot_id = page.Insert(Tuple(std::byte{29}, 150)); + ASSERT_TRUE(slot_id.has_value()); + + uint16_t init_dead_bytes = page.Header().dead_bytes; + + auto st = page.Update(slot_id.value(), Tuple(std::byte{29}, 100)); + ASSERT_TRUE(st.has_value()); + ASSERT_EQ(st.value(), UpdateOutcome::kSamePage); + + uint16_t final_dead_bytes = page.Header().dead_bytes; + + ASSERT_EQ(final_dead_bytes - init_dead_bytes, 50); +} + +TEST_F(HeapPageTest, UpdateToAnEqualLengthTupleAddsNoGarbage) { + HeapPage page = Page(); + + auto slot_id = page.Insert(Tuple(std::byte{29}, 150)); + ASSERT_TRUE(slot_id.has_value()); + + uint16_t init_dead_bytes = page.Header().dead_bytes; + + auto st = page.Update(slot_id.value(), Tuple(std::byte{30}, 150)); + ASSERT_TRUE(st.has_value()); + ASSERT_EQ(st.value(), UpdateOutcome::kSamePage); + + uint16_t final_dead_bytes = page.Header().dead_bytes; + + ASSERT_EQ(final_dead_bytes, init_dead_bytes); +} + +// Its neighbours must be untouched — an in-place overwrite that runs long corrupts the tuple +// physically adjacent to it, and only a neighbour check catches that. +TEST_F(HeapPageTest, UpdateInPlaceLeavesNeighbouringTuplesIntact) { + constexpr slot_id_t kTuples = 13; + constexpr std::size_t kLength = 150; + constexpr slot_id_t kTarget = 5; + + auto fill = [](slot_id_t i) { return static_cast(i + 1); }; + + HeapPage page = Page(); + for (slot_id_t i = 0; i < kTuples; ++i) { + ASSERT_TRUE(page.Insert(Tuple(fill(i), kLength)).has_value()); + } + + // Tuples grow BACKWARD: Insert places each one at tuple_data_start - length and then lowers + // the mark, so a later slot sits at a LOWER offset. Slot 4 is therefore immediately ABOVE + // slot 5 in the body, and an in-place overwrite that runs long writes upward into it. Slot 6 + // lives below slot 5 and cannot be reached by a forward write at all — checking only slot 6 + // is checking the one neighbour that is safe no matter how badly Update overruns. + ASSERT_GT(page.SlotAt(kTarget - 1).offset, page.SlotAt(kTarget).offset); + ASSERT_GT(page.SlotAt(kTarget).offset, page.SlotAt(kTarget + 1).offset); + + auto st = page.Update(kTarget, Tuple(std::byte{99}, 100)); + ASSERT_TRUE(st.has_value()); + ASSERT_EQ(st.value(), UpdateOutcome::kSamePage); + + // The tuple itself: exactly 100 bytes of the new fill. A shrink that forgot to update the + // slot length would hand back 150 bytes with a 50-byte tail of stale data. + EXPECT_EQ(Read(page.Get(kTarget).value()), Tuple(std::byte{99}, 100)); + + // Every other tuple, each against its OWN fill, so a failure names the slot that moved. + for (slot_id_t i = 0; i < kTuples; ++i) { + if (i == kTarget) continue; + auto tuple = page.Get(i); + ASSERT_TRUE(tuple.has_value()) << "slot " << i; + EXPECT_EQ(Read(tuple.value()), Tuple(fill(i), kLength)) << "slot " << i; + } + EXPECT_TRUE(page.CheckInvariants()); +} + +// --------------------------------------------------------------------------------------------- +// Update — case 2, relocation +// --------------------------------------------------------------------------------------------- + +// Grows with room at the low-water mark: new offset, old extent added to dead_bytes, same RID. +TEST_F(HeapPageTest, UpdateThatGrowsRelocatesWithinThePage) { + HeapPage page = Page(); + auto grown = page.Insert(Tuple(std::byte{1}, 100)); + auto middle = page.Insert(Tuple(std::byte{2}, 200)); + auto last = page.Insert(Tuple(std::byte{3}, 300)); + ASSERT_TRUE(grown.has_value() && middle.has_value() && last.has_value()); + + const auto old_offset = page.SlotAt(*grown).offset; + ASSERT_GE(page.Contiguous(), 500u); // the premise: it fits without reclaiming anything + + auto outcome = page.Update(*grown, Tuple(std::byte{9}, 500)); + ASSERT_TRUE(outcome.has_value()); + EXPECT_EQ(*outcome, UpdateOutcome::kSamePage); + + // The RID survives because the slot INDEX did not change, even though the tuple moved. + EXPECT_EQ(page.SlotCount(), 3); + EXPECT_EQ(page.LiveCount(), 3); + EXPECT_NE(page.SlotAt(*grown).offset, old_offset); + EXPECT_EQ(page.SlotAt(*grown).length, 500); + EXPECT_EQ(Read(page.Get(*grown).value()), Tuple(std::byte{9}, 500)); + + // No compaction happened, and dead_bytes is what proves it: the old 100-byte extent is still + // garbage. A slow-path update would have reclaimed it and left this at zero. + EXPECT_EQ(page.Header().dead_bytes, 100); + EXPECT_EQ(page.Header().tuple_data_start, page.SlotAt(*grown).offset); + + EXPECT_EQ(Read(page.Get(*middle).value()), Tuple(std::byte{2}, 200)); + EXPECT_EQ(Read(page.Get(*last).value()), Tuple(std::byte{3}, 300)); + EXPECT_TRUE(page.CheckInvariants()); +} + +// Grows past Contiguous() but inside Reclaimable(): compacts, keeps the slot id, keeps every +// other RID readable, and ends with dead_bytes back at zero. +TEST_F(HeapPageTest, UpdateThatGrowsCompactsWhenItHasTo) { + HeapPage page = Page(); + auto doomed = page.Insert(Tuple(std::byte{1}, 2000)); + auto grown = page.Insert(Tuple(std::byte{2}, 1000)); + auto bystander = page.Insert(Tuple(std::byte{3}, 1000)); + ASSERT_TRUE(doomed.has_value() && grown.has_value() && bystander.has_value()); + ASSERT_TRUE(page.Delete(*doomed).ok()); + + // The premise: 1500 bytes do not fit in the gap, but they do fit once the garbage is + // reclaimed. Stated rather than assumed, so this cannot silently become a fast-path test. + ASSERT_LT(page.Contiguous(), 1500u); + ASSERT_GE(page.Reclaimable(), 1500u); + + auto outcome = page.Update(*grown, Tuple(std::byte{9}, 1500)); + ASSERT_TRUE(outcome.has_value()); + EXPECT_EQ(*outcome, UpdateOutcome::kSamePage); + + // The slot id is unchanged, which is the whole reason case 2 returns kSamePage. + EXPECT_EQ(Read(page.Get(*grown).value()), Tuple(std::byte{9}, 1500)); + EXPECT_EQ(Read(page.Get(*bystander).value()), Tuple(std::byte{3}, 1000)); + + // The compaction ran: every byte below the low-water mark is now live, so dead_bytes is zero + // by definition and the region is packed against the end of the body. + EXPECT_EQ(page.Header().dead_bytes, 0); + EXPECT_EQ(page.Header().tuple_data_start, PAGE_BODY_SIZE - 1500 - 1000); + + // The deleted slot kept its index through the compaction and is still dead. + EXPECT_EQ(page.SlotCount(), 3); + EXPECT_TRUE(page.SlotAt(*doomed).IsDead()); + EXPECT_EQ(page.Get(*doomed).error().code(), ErrorCode::kNotFound); + EXPECT_TRUE(page.CheckInvariants()); +} + +// live_count must survive the kill-compact-replace dance — off by one here is the classic bug. +TEST_F(HeapPageTest, UpdateViaCompactionLeavesLiveCountCorrect) { + constexpr slot_id_t kTuples = 5; + constexpr std::size_t kLength = 700; + + HeapPage page = Page(); + for (slot_id_t i = 0; i < kTuples; ++i) { + ASSERT_TRUE(page.Insert(Tuple(static_cast(i + 1), kLength)).has_value()); + } + ASSERT_TRUE(page.Delete(0).ok()); + ASSERT_TRUE(page.Delete(2).ok()); + ASSERT_EQ(page.LiveCount(), 3); + + // Force the slow path on slot 4: too big for the gap, small enough once the two deleted + // tuples and slot 4's own bytes are reclaimed. + ASSERT_LT(page.Contiguous(), 1800u); + ASSERT_GE(page.Reclaimable(), 1800u); + auto outcome = page.Update(4, Tuple(std::byte{9}, 1800)); + ASSERT_TRUE(outcome.has_value()); + EXPECT_EQ(*outcome, UpdateOutcome::kSamePage); + + // Update decrements live_count to hand Compact a self-consistent page and puts it back + // afterwards. Off by one in either direction is the classic bug, so count the slot array + // independently rather than trusting the field to agree with itself. + std::size_t live_slots = 0; + for (slot_id_t i = 0; i < page.SlotCount(); ++i) { + if (!page.SlotAt(i).IsDead()) ++live_slots; + } + EXPECT_EQ(live_slots, 3u); + EXPECT_EQ(page.LiveCount(), 3); + EXPECT_EQ(page.SlotCount(), kTuples); + + // And the live set is the RIGHT three: the survivors, not a resurrected dead slot. + EXPECT_EQ(Read(page.Get(1).value()), Tuple(std::byte{2}, kLength)); + EXPECT_EQ(Read(page.Get(3).value()), Tuple(std::byte{4}, kLength)); + EXPECT_EQ(Read(page.Get(4).value()), Tuple(std::byte{9}, 1800)); + EXPECT_EQ(page.Get(0).error().code(), ErrorCode::kNotFound); + EXPECT_EQ(page.Get(2).error().code(), ErrorCode::kNotFound); + EXPECT_TRUE(page.CheckInvariants()); +} + +// --------------------------------------------------------------------------------------------- +// Update — case 3 and validation +// --------------------------------------------------------------------------------------------- + +// kDoesNotFit, and the body must be BYTE-IDENTICAL to a snapshot taken before the call. +TEST_F(HeapPageTest, UpdateThatCannotFitLeavesThePageByteIdentical) { + HeapPage page = Page(); + + auto slot_1 = page.Insert(Tuple(std::byte{29}, 2000)); + EXPECT_TRUE(slot_1.has_value()); + auto slot_2 = page.Insert(Tuple(std::byte{29}, 2000)); + EXPECT_TRUE(slot_2.has_value()); + auto slot_3 = page.Insert(Tuple(std::byte{29}, 30)); + EXPECT_TRUE(slot_3.has_value()); + + auto st = page.Update(slot_3.value(), Tuple(std::byte{30}, 2000)); + EXPECT_EQ(st, UpdateOutcome::kDoesNotFit); +} + +TEST_F(HeapPageTest, UpdateRejectsAZeroLengthTuple) { + HeapPage page = Page(); + + auto slot = page.Insert(Tuple(std::byte{29}, 2000)); + EXPECT_TRUE(slot.has_value()); + + auto st = page.Update(slot.value(), Tuple(std::byte{30}, 0)); + EXPECT_FALSE(st.has_value()); + EXPECT_EQ(st.error().code(), ErrorCode::kInvalidArgument); +} + +TEST_F(HeapPageTest, UpdateRejectsATupleOverMaxTupleSize) { + HeapPage page = Page(); + + auto slot = page.Insert(Tuple(std::byte{29}, 2000)); + EXPECT_TRUE(slot.has_value()); + + auto st = page.Update(slot.value(), Tuple(std::byte{30}, 2300)); + EXPECT_FALSE(st.has_value()); + EXPECT_EQ(st.error().code(), ErrorCode::kInvalidArgument); +} + +// Must be NotFound, not a resurrected slot and not Corruption. +TEST_F(HeapPageTest, UpdateOnADeadSlotIsNotFound) { + HeapPage page = Page(); + + auto slot = page.Insert(Tuple(std::byte{29}, 2000)); + EXPECT_TRUE(slot.has_value()); + + auto st = page.Delete(slot.value()); + ASSERT_EQ(st.code(), ErrorCode::kOk); + + auto updated = page.Update(slot.value(), Tuple(std::byte{30}, 1800)); + EXPECT_FALSE(updated.has_value()); + EXPECT_EQ(updated.error().code(), ErrorCode::kNotFound); +} + +TEST_F(HeapPageTest, UpdateOnAnOutOfRangeSlotIsNotFound) { + HeapPage page = Page(); + + auto slot = page.Insert(Tuple(std::byte{29}, 2000)); + EXPECT_TRUE(slot.has_value()); + + auto updated = page.Update(3, Tuple(std::byte{30}, 1800)); + EXPECT_FALSE(updated.has_value()); + EXPECT_EQ(updated.error().code(), ErrorCode::kNotFound); +} + +// --------------------------------------------------------------------------------------------- +// Compact +// --------------------------------------------------------------------------------------------- + +// The real proof: delete alternating tuples, compact, and every survivor still reads back its +// original bytes. That is what says the slot offsets were rewritten in lockstep with the bytes. +TEST_F(HeapPageTest, CompactReclaimsDeletedBytesAndKeepsSurvivingRidsReadable) { + constexpr slot_id_t kTuples = 13; + constexpr std::size_t kLength = 50; + + // A DISTINCT fill per tuple, and that is the whole test. With one shared fill every survivor + // reads back 50 identical bytes no matter which survivor's bytes it actually got, so a + // Compact that crossed two slots' offsets — wrote tuple 3 and pointed slot 7 at it — would + // pass. Crossed offsets are exactly the bug this test exists to find. + auto fill = [](slot_id_t i) { return static_cast(i + 1); }; + + HeapPage page = Page(); + std::vector slots; + for (slot_id_t i = 0; i < kTuples; ++i) { + auto slot_id = page.Insert(Tuple(fill(i), kLength)); + ASSERT_TRUE(slot_id.has_value()); + slots.push_back(slot_id.value()); + } + + uint16_t dead_cnt{0}; + for (slot_id_t i = 0; i < kTuples; i += 2) { + ASSERT_TRUE(page.Delete(slots[i]).ok()); + ++dead_cnt; + } + + ASSERT_EQ(page.Header().dead_bytes, dead_cnt * kLength); + ASSERT_EQ(page.Header().slot_count - page.Header().live_count, dead_cnt); + + page.Compact(); + + for (slot_id_t i = 0; i < kTuples; ++i) { + if (i % 2) { + auto tuple = page.Get(slots[i]); + ASSERT_TRUE(tuple.has_value()) << "slot " << i << " should have survived"; + EXPECT_EQ(Read(tuple.value()), Tuple(fill(i), kLength)) << "slot " << i; + } else { + EXPECT_TRUE(page.SlotAt(slots[i]).IsDead()) << "slot " << i; + } + } + + // Bytes move, slots never renumber; and the region is packed, which dead_bytes == 0 alone + // does not establish. + EXPECT_EQ(page.SlotCount(), kTuples); + EXPECT_EQ(page.LiveCount(), kTuples - dead_cnt); + EXPECT_EQ(page.Header().dead_bytes, 0); + EXPECT_EQ(page.Header().tuple_data_start, PAGE_BODY_SIZE - (kTuples - dead_cnt) * kLength); + EXPECT_TRUE(page.CheckInvariants()); +} + +// tuple_data_start lands exactly at PAGE_BODY_SIZE minus the sum of the live lengths. +TEST_F(HeapPageTest, CompactPacksTheTupleRegionAgainstTheEndOfTheBody) { + HeapPage page = Page(); + + constexpr size_t N = 13; + std::vector slots; + for (std::size_t i = 0; i < N; i++) { + auto slot = page.Insert(Tuple(std::byte{29}, 50)); + EXPECT_TRUE(slot.has_value()); + slots.push_back(slot.value()); + } + + uint16_t byte_cnt{0}; + for (std::size_t i = 0; i < N; i++) { + if (i % 2) { + byte_cnt += 50; + continue; + } + auto st = page.Delete(slots[i]); + EXPECT_EQ(st.code(), ErrorCode::kOk); + } + page.Compact(); + ASSERT_EQ(page.Header().tuple_data_start, PAGE_BODY_SIZE - byte_cnt); +} + +// Bytes move, slots never renumber: slot_count and live_count come out unchanged and dead slots +// are still dead at their original indices. +TEST_F(HeapPageTest, CompactPreservesSlotCountAndLiveCount) { + HeapPage page = Page(); + constexpr size_t N{13}; + + for (size_t i = 0; i < N; i++) { + auto slot = page.Insert(Tuple(std::byte{29}, 50)); + EXPECT_TRUE(slot.has_value()); + } + + for (size_t i = 0; i < N; i++) { + if (i % 3) continue; + auto st = page.Delete(static_cast(i)); + EXPECT_EQ(st.code(), ErrorCode::kOk); + } + + uint16_t init_slot_cnt = page.Header().slot_count; + uint16_t init_live_cnt = page.Header().live_count; + + page.Compact(); + + ASSERT_EQ(init_slot_cnt, page.Header().slot_count); + ASSERT_EQ(init_live_cnt, page.Header().live_count); +} + +// dead_bytes == 0 means already packed, so this must be a no-op rather than a 4KB shuffle. +TEST_F(HeapPageTest, CompactOnAPackedPageChangesNothing) { + constexpr slot_id_t kTuples = 5; + constexpr std::size_t kLength = 200; + + HeapPage page = Page(); + for (slot_id_t i = 0; i < kTuples; ++i) { + ASSERT_TRUE(page.Insert(Tuple(static_cast(i + 1), kLength)).has_value()); + } + + // The precondition the early-out rests on, and it is provable rather than incidental: the + // accounting identity says tuple_data_start + live_bytes + dead_bytes == PAGE_BODY_SIZE, so + // zero garbage means the live tuples exactly fill the region — and non-overlapping tuples + // that exactly fill a region are already packed. + ASSERT_EQ(page.Header().dead_bytes, 0); + ASSERT_EQ(page.Header().tuple_data_start, PAGE_BODY_SIZE - kTuples * kLength); + + const std::array before = body_; + + page.Compact(); + + // Byte-for-byte, which covers the header, the slot array and the tuple region in one shot. + // NOTE: this proves the RESULT is unchanged, not that the early-out branch was taken — a + // correct full compaction of an already-packed page produces exactly these bytes too. The + // branch itself is not observable from out here; see the test's comment above. + EXPECT_EQ(std::memcmp(before.data(), body_.data(), PAGE_BODY_SIZE), 0); + + EXPECT_EQ(page.SlotCount(), kTuples); + EXPECT_EQ(page.LiveCount(), kTuples); + EXPECT_EQ(page.Header().dead_bytes, 0); + for (slot_id_t i = 0; i < kTuples; ++i) { + EXPECT_EQ(Read(page.Get(i).value()), Tuple(static_cast(i + 1), kLength)); + } + EXPECT_TRUE(page.CheckInvariants()); +} + +TEST_F(HeapPageTest, CompactIsIdempotent) { + constexpr slot_id_t kTuples = 6; + constexpr std::size_t kLength = 200; + + HeapPage page = Page(); + for (slot_id_t i = 0; i < kTuples; ++i) { + ASSERT_TRUE(page.Insert(Tuple(static_cast(i + 1), kLength)).has_value()); + } + for (slot_id_t i = 1; i < kTuples; i += 2) { + ASSERT_TRUE(page.Delete(i).ok()); + } + ASSERT_EQ(page.Header().dead_bytes, 3 * kLength); + + page.Compact(); + ASSERT_EQ(page.Header().dead_bytes, 0); + ASSERT_EQ(page.Header().tuple_data_start, PAGE_BODY_SIZE - 3 * kLength); + + const std::array after_first = body_; + + page.Compact(); + + // Byte-identical, which covers header, slot array and tuple region at once. This holds even + // with the dead_bytes == 0 early-out removed: a compaction of a packed page repacks it into + // the same offsets, so the second pass is a fixed point rather than merely a skipped call. + EXPECT_EQ(std::memcmp(after_first.data(), body_.data(), PAGE_BODY_SIZE), 0); + + EXPECT_EQ(page.SlotCount(), kTuples); + EXPECT_EQ(page.LiveCount(), 3); + for (slot_id_t i = 0; i < kTuples; ++i) { + if (i % 2 == 0) { + EXPECT_EQ(Read(page.Get(i).value()), Tuple(static_cast(i + 1), kLength)); + } else { + EXPECT_TRUE(page.SlotAt(i).IsDead()); + } + } + EXPECT_TRUE(page.CheckInvariants()); +} + +TEST_F(HeapPageTest, CompactOnAnAllDeadPageResetsTheLowWaterMark) { + constexpr slot_id_t kTuples = 4; + constexpr std::size_t kLength = 500; + + HeapPage page = Page(); + for (slot_id_t i = 0; i < kTuples; ++i) { + ASSERT_TRUE(page.Insert(Tuple(static_cast(i + 1), kLength)).has_value()); + } + for (slot_id_t i = 0; i < kTuples; ++i) { + ASSERT_TRUE(page.Delete(i).ok()); + } + ASSERT_EQ(page.Header().tuple_data_start, PAGE_BODY_SIZE - kTuples * kLength); + + page.Compact(); + + // With no live tuples the loop copies nothing and the cursor never leaves its starting point, + // so the region collapses to empty and the mark goes all the way back. + EXPECT_EQ(page.Header().tuple_data_start, PAGE_BODY_SIZE); + EXPECT_EQ(page.Header().dead_bytes, 0); + EXPECT_EQ(page.LiveCount(), 0); + + // Bytes move, slots never renumber — even when every one of them is dead and the page holds + // nothing. The array stays at full length so a stale RID still has something to resolve to. + EXPECT_EQ(page.SlotCount(), kTuples); + for (slot_id_t i = 0; i < kTuples; ++i) { + EXPECT_TRUE(page.SlotAt(i).IsDead()); + EXPECT_EQ(page.Get(i).error().code(), ErrorCode::kNotFound); + } + + // The space is genuinely back: everything but the sub-header and the surviving slot array. + EXPECT_EQ(page.Contiguous(), PAGE_BODY_SIZE - HEAP_SUB_HEADER_SIZE - kTuples * SLOT_SIZE); + EXPECT_EQ(page.Reclaimable(), page.Contiguous()); + EXPECT_TRUE(page.CheckInvariants()); +} + +// The gap the other cases miss: a reused slot index has to survive a compaction underneath it. +// Insert, delete a middle tuple, insert again (reusing that slot), then compact, and check every +// live RID — the reused slot is the one whose offset is easiest to rewrite wrong. +TEST_F(HeapPageTest, CompactAfterSlotReuseKeepsEveryRidReadable) { + HeapPage page = Page(); + ASSERT_TRUE(page.Insert(Tuple(std::byte{1}, 100)).has_value()); // slot 0 + ASSERT_TRUE(page.Insert(Tuple(std::byte{2}, 200)).has_value()); // slot 1, about to die + ASSERT_TRUE(page.Insert(Tuple(std::byte{3}, 300)).has_value()); // slot 2 + ASSERT_TRUE(page.Delete(1).ok()); + + // Reuses slot 1: the index is recycled but the tuple lands at the low-water mark, nowhere + // near where the old one was, so slot 1's offset now points somewhere unrelated to its index. + auto reused = page.Insert(Tuple(std::byte{4}, 250)); + ASSERT_TRUE(reused.has_value()); + ASSERT_EQ(*reused, 1); + ASSERT_EQ(page.Header().dead_bytes, 200); // the reuse reclaimed the ENTRY, not the bytes + + page.Compact(); + + // The reused slot is the one whose offset is easiest to rewrite wrong, because its index no + // longer matches its position in the tuple region at all. + EXPECT_EQ(Read(page.Get(0).value()), Tuple(std::byte{1}, 100)); + EXPECT_EQ(Read(page.Get(1).value()), Tuple(std::byte{4}, 250)); + EXPECT_EQ(Read(page.Get(2).value()), Tuple(std::byte{3}, 300)); + + EXPECT_EQ(page.SlotCount(), 3); + EXPECT_EQ(page.LiveCount(), 3); + EXPECT_EQ(page.Header().dead_bytes, 0); + EXPECT_EQ(page.Header().tuple_data_start, PAGE_BODY_SIZE - (100 + 250 + 300)); + EXPECT_TRUE(page.CheckInvariants()); +} + +// The caller-driven contract end to end: Insert fails with PageFull, Reclaimable() says there is +// room, Compact, and the same Insert now succeeds. +TEST_F(HeapPageTest, InsertSucceedsAfterCompactWhenItPreviouslyFailed) { + HeapPage page = Page(); + ASSERT_TRUE(page.Insert(Tuple(std::byte{1}, MAX_TUPLE_SIZE)).has_value()); + ASSERT_TRUE(page.Insert(Tuple(std::byte{2}, MAX_TUPLE_SIZE)).has_value()); + ASSERT_TRUE(page.Delete(0).ok()); + + const auto wanted = Tuple(std::byte{9}, 1500); + + // The contract end to end: Insert refuses, Reclaimable() says the space is there, and the + // caller — not Insert — decides to spend a compaction on it. + auto refused = page.Insert(wanted); + EXPECT_FALSE(refused.has_value()); + EXPECT_EQ(refused.error().code(), ErrorCode::kPageFull); + ASSERT_LT(page.Contiguous(), wanted.size()); + ASSERT_GE(page.Reclaimable(), wanted.size()); + + page.Compact(); + + auto accepted = page.Insert(wanted); + ASSERT_TRUE(accepted.has_value()); + EXPECT_EQ(*accepted, 0); // the dead slot's index is recycled, not appended to + EXPECT_EQ(Read(page.Get(*accepted).value()), wanted); + EXPECT_EQ(Read(page.Get(1).value()), Tuple(std::byte{2}, MAX_TUPLE_SIZE)); + EXPECT_EQ(page.SlotCount(), 2); + EXPECT_EQ(page.LiveCount(), 2); + EXPECT_TRUE(page.CheckInvariants()); +} + +// --------------------------------------------------------------------------------------------- +// Corruption handling — hand CheckInvariants a page that is already wrong +// --------------------------------------------------------------------------------------------- + +// It must return false, not read out of bounds. Scribble a slot_count far past the 1014 ceiling +// straight into the body and call it. +TEST_F(HeapPageTest, CheckInvariantsRejectsAnImpossibleSlotCount) { + // 5000 slots would need 20008 bytes of array in a 4064-byte body. CheckInvariants must say + // false rather than walk 5000 entries off the end of the page — it is the one function + // designed to be handed garbage, so it may not index anything it has not first bounded. + PokeHeader(HeapSubHeader{.slot_count = 5000, + .tuple_data_start = static_cast(PAGE_BODY_SIZE), + .live_count = 0, + .dead_bytes = 0}); + EXPECT_FALSE(View().CheckInvariants()); + + // Its companion guard, which runs first and is what bounds slot_count in the first place: a + // low-water mark past the end of the body. + SetUp(); + PokeHeader( + HeapSubHeader{.slot_count = 0, .tuple_data_start = 5000, .live_count = 0, .dead_bytes = 0}); + EXPECT_FALSE(View().CheckInvariants()); +} + +TEST_F(HeapPageTest, CheckInvariantsRejectsALiveSlotPointingOutsideTheTupleRegion) { + HeapPage page = Page(); + ASSERT_TRUE(page.Insert(Tuple(std::byte{1}, 300)).has_value()); + ASSERT_TRUE(page.CheckInvariants()); + + const Slot good = page.SlotAt(0); + + // Below the low-water mark: the tuple claims bytes the page says are free space. + PokeSlot(0, Slot{static_cast(good.offset - 10), good.length}); + EXPECT_FALSE(View().CheckInvariants()); + + // Past the end of the body. This is the check that stops Get from handing out a span running + // off the page, which is why it is audited per slot here rather than on Get's hot path. + PokeSlot(0, good); + ASSERT_TRUE(page.CheckInvariants()); + PokeSlot(0, Slot{good.offset, static_cast(PAGE_BODY_SIZE)}); + EXPECT_FALSE(View().CheckInvariants()); +} + +TEST_F(HeapPageTest, CheckInvariantsRejectsAMiscountedLiveCount) { + HeapPage page = Page(); + for (slot_id_t i = 0; i < 3; ++i) { + ASSERT_TRUE(page.Insert(Tuple(static_cast(i + 1), 100)).has_value()); + } + ASSERT_TRUE(page.CheckInvariants()); + + HeapSubHeader header = page.Header(); + const uint16_t truth = header.live_count; + + // Too low, which is what a Delete that forgot to decrement's counterpart looks like... + header.live_count = static_cast(truth - 1); + PokeHeader(header); + EXPECT_FALSE(View().CheckInvariants()); + + // ...and too high, which is what an Update that killed a slot and never restored it looks + // like. The field is maintained rather than derived, so only this cross-check catches drift. + header.live_count = static_cast(truth + 1); + PokeHeader(header); + EXPECT_FALSE(View().CheckInvariants()); +} + +// Dead has exactly one representation, and only because Delete zeroes both fields. That is what +// makes "offset == 0 implies length == 0" assertable at all, so it needs its own guard. +TEST_F(HeapPageTest, CheckInvariantsRejectsADeadSlotWithANonZeroLength) { + HeapPage page = Page(); + ASSERT_TRUE(page.Insert(Tuple(std::byte{1}, 150)).has_value()); + ASSERT_TRUE(page.Delete(0).ok()); + ASSERT_TRUE(page.CheckInvariants()); + + // A Delete that zeroed the offset but left the length behind — a dead slot in a second, + // illegal representation that no other check would look at, since the walk skips dead slots. + PokeSlot(0, Slot{0, 150}); + EXPECT_FALSE(View().CheckInvariants()); +} + +// Overlapping tuples fall out of the byte-accounting sum with no pairwise comparison. +TEST_F(HeapPageTest, CheckInvariantsRejectsOverlappingTuples) { + HeapPage page = Page(); + ASSERT_TRUE(page.Insert(Tuple(std::byte{1}, 300)).has_value()); + ASSERT_TRUE(page.Insert(Tuple(std::byte{2}, 300)).has_value()); + ASSERT_TRUE(page.CheckInvariants()); + + const uint16_t shared = page.SlotAt(0).offset; + + // Point both live slots at the same 300 bytes and pull the low-water mark up to match, which + // is what a compaction that moved a tuple but rewrote the wrong slot would leave behind. + PokeSlot(1, Slot{shared, 300}); + HeapSubHeader header = page.Header(); + header.tuple_data_start = shared; + PokeHeader(header); + + // No pairwise comparison anywhere: the overlap counts the shared bytes twice, so the live + // total overshoots the region and the accounting identity is the thing that notices. + EXPECT_FALSE(View().CheckInvariants()); +} + +// --------------------------------------------------------------------------------------------- +// AsHeapPage — the one check a bare view cannot make +// --------------------------------------------------------------------------------------------- + +// Needs a real guard, so this pair reaches for a BufferPoolManager where nothing else here does. +TEST_F(HeapPageTest, AsHeapPageRejectsAPageThatIsNotAHeapPage) { + const auto path = std::filesystem::temp_directory_path() / "kernsql_asheap_reject_test"; + std::filesystem::remove(path); + auto dm = DiskManager::Open(path); + ASSERT_TRUE(dm.has_value()) << dm.error().message(); + { + BufferPoolManager bpm(**dm, 4); + { + auto guard = bpm.NewPage(); + ASSERT_TRUE(guard.has_value()) << guard.error().message(); + ASSERT_NE(guard->Header().page_type, PageType::HEAP); + + // A HeapPage sees only the body, so it CANNOT check page_type. Build one over a + // non-heap page and it parses whatever is there as a sub-header and scribbles on it. + // The guard can see the header, which is the entire reason this function exists. + auto view = AsHeapPage(*guard); + EXPECT_FALSE(view.has_value()); + EXPECT_EQ(view.error().code(), ErrorCode::kCorruption); + } + { + auto guard = bpm.FetchPageRead(1); // the catalog page, reserved by DiskManager + ASSERT_TRUE(guard.has_value()) << guard.error().message(); + auto view = AsHeapPage(*guard); + EXPECT_FALSE(view.has_value()); + EXPECT_EQ(view.error().code(), ErrorCode::kCorruption); + } + EXPECT_TRUE(bpm.Shutdown().ok()); + } + dm->reset(); + std::filesystem::remove(path); +} + +TEST_F(HeapPageTest, AsHeapPageAcceptsAHeapPage) { + const auto path = std::filesystem::temp_directory_path() / "kernsql_asheap_accept_test"; + std::filesystem::remove(path); + auto dm = DiskManager::Open(path); + ASSERT_TRUE(dm.has_value()) << dm.error().message(); + { + BufferPoolManager bpm(**dm, 4); + page_id_t page_id{}; + { + auto guard = bpm.NewPage(); + ASSERT_TRUE(guard.has_value()) << guard.error().message(); + page_id = guard->PageId(); + + // The caller stamps the type through the guard; this layer cannot see the header. + guard->SetPageType(PageType::HEAP); + + auto view = AsHeapPage(*guard); + ASSERT_TRUE(view.has_value()) << view.error().message(); + + // And the span it handed back really is this page's body, not an offset copy of it. + view->Init(); + auto slot = view->Insert(Bytes("through a real guard")); + ASSERT_TRUE(slot.has_value()); + EXPECT_EQ(AsChars(view->Get(*slot).value()), "through a real guard"); + EXPECT_TRUE(view->CheckInvariants()); + } + { + // The const overload, over the same page after the write guard is gone. + auto guard = bpm.FetchPageRead(page_id); + ASSERT_TRUE(guard.has_value()) << guard.error().message(); + auto view = AsHeapPage(*guard); + ASSERT_TRUE(view.has_value()) << view.error().message(); + EXPECT_EQ(view->LiveCount(), 1); + EXPECT_EQ(AsChars(view->Get(0).value()), "through a real guard"); + } + EXPECT_TRUE(bpm.Shutdown().ok()); + } + dm->reset(); + std::filesystem::remove(path); +} + +} // namespace kernsql diff --git a/test/heap/table_heap_test.cpp b/test/heap/table_heap_test.cpp new file mode 100644 index 0000000..b02769f --- /dev/null +++ b/test/heap/table_heap_test.cpp @@ -0,0 +1,1247 @@ +#include "heap/table_heap.hpp" + +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include "buffer/buffer_pool_manager.hpp" +#include "buffer/page_guard.hpp" +#include "buffer/pool_stats.hpp" +#include "common/status.hpp" +#include "common/types.hpp" +#include "heap/heap_page.hpp" +#include "storage/disk_manager.hpp" + +namespace kernsql { + +/* + * TableHeap against a real file and a real buffer pool. No fakes: the layer's whole job is to + * turn page ids into rows, so a test that stubs the pool tests nothing that can break. + * + * The assertions to reach for, in rough order of how much they are worth: + * + * 1. CheckInvariants() on every touched page after every mutating operation. It proves the + * accounting identity — tuple_data_start + sum(live lengths) + dead_bytes == PAGE_BODY_SIZE + * — at the operation that broke it rather than a thousand operations later. + * 2. The bytes, not the status. A round trip that returns ok and the wrong tuple is the bug + * this layer is most likely to have. + * 3. Old RIDs still resolving after a compaction. Slots never move IS the format's contract. + * 4. Pool quiescence at teardown. A leaked pin is silent until the pool starves. + */ +class TableHeapTest : public ::testing::Test { + protected: + // Sixteen rather than the buffer pool suite's four. Insert's extension path holds TWO guards + // at once (the old last page and the new one), so a concurrency test with N threads can want + // 2N frames before kBufferPoolFull becomes a legitimate answer rather than a symptom. + static constexpr std::size_t kFrames = 16; + + void SetUp() override { + path_ = std::filesystem::temp_directory_path() / + (std::string("kernsql_table_heap_test_") + + testing::UnitTest::GetInstance()->current_test_info()->name()); + std::filesystem::remove(path_); + + OpenPool(); + + auto created = TableHeap::Create(*bpm_); + ASSERT_TRUE(created.has_value()) << created.error().message(); + heap_ = std::move(created.value()); + } + + void TearDown() override { + // The heap first: an iterator inside a failed test can still hold a pin, and Shutdown() + // verifies quiescence rather than arranging it. + heap_.reset(); + if (bpm_) { + EXPECT_TRUE(bpm_->Shutdown().ok()); + } + bpm_.reset(); + dm_.reset(); + std::filesystem::remove(path_); + } + + void OpenPool() { + auto dm = DiskManager::Open(path_); + ASSERT_TRUE(dm.has_value()) << dm.error().message(); + dm_ = std::move(dm.value()); + bpm_ = std::make_unique(*dm_, kFrames); + } + + // Close everything and reopen the same file, then reopen the table from the two ids the + // catalog will one day hold. The only way to prove anything reached disk. + void Reopen(page_id_t first, page_id_t last) { + heap_.reset(); + ASSERT_TRUE(bpm_->Shutdown().ok()); + bpm_.reset(); + dm_.reset(); + + OpenPool(); + auto opened = TableHeap::Open(*bpm_, first, last); + ASSERT_TRUE(opened.has_value()) << opened.error().message(); + heap_ = std::move(opened.value()); + } + + static std::span Bytes(std::string_view s) { + return std::as_bytes(std::span{s}); + } + static std::string AsChars(std::span bytes) { + return {reinterpret_cast(bytes.data()), bytes.size()}; + } + + // Distinct, self-identifying, fixed-width. A failure says WHICH row came back wrong instead + // of only that some bytes differ. + static std::string Row(int i) { return std::format("row-{:06d}", i); } + + // Insert or fail the test outright — a helper that returns a bad RID just moves the failure + // somewhere less readable. + RID MustInsert(std::string_view text) { + auto rid = heap_->Insert(Bytes(text)); + EXPECT_TRUE(rid.has_value()) << rid.error().message(); + return rid.value_or(RID{}); + } + + std::string MustGet(RID rid) { + auto row = heap_->Get(rid); + EXPECT_TRUE(row.has_value()) << row.error().message(); + return row.has_value() ? AsChars(row.value()) : std::string{}; + } + + // Every live row, in scan order. Fails the test on a scan error rather than returning short. + std::vector> CollectAll() { + std::vector> rows; + auto it = heap_->Scan(); + for (;;) { + auto more = it.Next(); + if (!more.has_value()) { + ADD_FAILURE() << "scan failed: " << more.error().message(); + return rows; + } + if (!more.value()) break; + rows.emplace_back(it.Rid(), AsChars(it.Tuple())); + } + return rows; + } + + // The page ids of the chain, in order, walked through the buffer pool rather than through + // TableHeap — so a test can assert on the chain the ITERATOR would see even when TableHeap's + // own idea of it is what is under suspicion. + std::vector ChainPages() { + std::vector pages; + for (page_id_t id = heap_->FirstPageId(); id != INVALID_PAGE;) { + auto guard = bpm_->FetchPageRead(id); + if (!guard.has_value()) { + ADD_FAILURE() << "fetch " << id << ": " << guard.error().message(); + return pages; + } + pages.push_back(id); + id = guard.value().Header().next_page_id; + } + return pages; + } + + // A page's accounting, read straight out of the body. The numbers the shell's `tdump` prints. + HeapSubHeader SubHeaderOf(page_id_t page_id) { + auto guard = bpm_->FetchPageRead(page_id); + EXPECT_TRUE(guard.has_value()); + if (!guard.has_value()) return {}; + auto page = AsHeapPage(guard.value()); + EXPECT_TRUE(page.has_value()); + return page.has_value() ? page.value().Header() : HeapSubHeader{}; + } + + std::size_t ContiguousOf(page_id_t page_id) { + auto guard = bpm_->FetchPageRead(page_id); + EXPECT_TRUE(guard.has_value()); + if (!guard.has_value()) return 0; + auto page = AsHeapPage(guard.value()); + EXPECT_TRUE(page.has_value()); + return page.has_value() ? page.value().Contiguous() : 0; + } + + // Call after every mutating operation. Cheap, and it localises accounting bugs to the + // operation that caused them. + void ExpectInvariants(page_id_t page_id) { + auto guard = bpm_->FetchPageRead(page_id); + ASSERT_TRUE(guard.has_value()) << guard.error().message(); + auto page = AsHeapPage(guard.value()); + ASSERT_TRUE(page.has_value()) << page.error().message(); + EXPECT_TRUE(page.value().CheckInvariants()) << "page " << page_id; + } + + void ExpectAllInvariants() { + for (page_id_t id : ChainPages()) ExpectInvariants(id); + } + + // No frame still pinned. A stranded pin is invisible until the pool starves, so it has to be + // asserted rather than waited for. + void ExpectNoStrandedPins() { + const PoolStats stats = bpm_->GetStats(); + EXPECT_EQ(stats.pinned_frames, 0U); + } + + // Row(i) is exactly this long for i < 1'000'000, and every fill below depends on it. + static constexpr std::size_t kRowSize = 10; + + // How many Row()s one empty heap page holds: each costs its bytes plus a slot entry. 289 for + // a 4064-byte body. InsertExtendsTheChainWhenThePageFills pins this against the real page, so + // a format change fails there first rather than quietly skewing every test that fills. + static constexpr std::size_t kRowsPerPage = + (PAGE_BODY_SIZE - HEAP_SUB_HEADER_SIZE) / (kRowSize + SLOT_SIZE); + + // What a page full of Row()s has left over — less than one more row. + static constexpr std::size_t kFullPageSlack = + PAGE_BODY_SIZE - HEAP_SUB_HEADER_SIZE - kRowsPerPage * (kRowSize + SLOT_SIZE); + + // Mirrors COMPACT_THRESHOLD in table_heap.cpp, which is file-local. + static constexpr std::size_t kCompactThreshold = PAGE_BODY_SIZE / 8; + + // META and the catalog root. Every other page in a fresh file belongs to some chain. + static constexpr std::size_t kReservedPages = 2; + + using Rows = std::vector>; + using RowMap = std::map; // also keeps the comma out of gtest macros + + // Insert Row(first) .. Row(first + count - 1) and return each with its RID. With no deletes in + // between, rows land in order: row i is on page i / kRowsPerPage of the chain. + Rows Fill(std::size_t count, int first = 0) { + Rows rows; + rows.reserve(count); + for (std::size_t i = 0; i < count; ++i) { + std::string text = Row(first + static_cast(i)); + const RID rid = MustInsert(text); + rows.emplace_back(rid, std::move(text)); + } + return rows; + } + + // The error code, or kOk. Lets a test assert WHICH failure it got in one line — the point of + // most of the tier-2 tests. + template + static ErrorCode CodeOf(const Result& result) { + return result.has_value() ? ErrorCode::kOk : result.error().code(); + } + static ErrorCode CodeOf(const Status& status) { return status.code(); } + + static void ExpectSameAccounting(const HeapSubHeader& before, const HeapSubHeader& after) { + EXPECT_EQ(after.slot_count, before.slot_count); + EXPECT_EQ(after.live_count, before.live_count); + EXPECT_EQ(after.tuple_data_start, before.tuple_data_start); + EXPECT_EQ(after.dead_bytes, before.dead_bytes); + } + + // Page one full, page two holding five rows, then every fifth row on page one deleted: 58 + // dead rows, 580 dead bytes — past kCompactThreshold, and scattered so a compaction has to + // move nearly every surviving tuple. The last delete leaves the insert hint on page one. + struct ScatteredFirstPage { + Rows survivors; + std::vector deleted; + }; + ScatteredFirstPage FillThenScatterFirstPage() { + ScatteredFirstPage out; + const Rows rows = Fill(kRowsPerPage + 5); + for (std::size_t i = 0; i < rows.size(); ++i) { + if (i < kRowsPerPage && i % 5 == 0) { + EXPECT_TRUE(heap_->Delete(rows[i].first).ok()); + out.deleted.push_back(rows[i].first); + } else { + out.survivors.push_back(rows[i]); + } + } + return out; + } + + std::filesystem::path path_; + std::unique_ptr dm_; + std::unique_ptr bpm_; + std::unique_ptr heap_; +}; + +// ================================================================================================= +// Tier 1 — construction, identity, durability +// ================================================================================================= + +TEST_F(TableHeapTest, CreateStampsHeapAndInitsTheBody) { + // The page must come back as page_type == HEAP (AsHeapPage succeeding IS that assertion) and + // as an INITIALISED empty page, not merely a zeroed one. The field that separates the two is + // tuple_data_start: zeroed gives 0, initialised gives PAGE_BODY_SIZE. Assert it directly — + // a zeroed body passes every other check and then underflows on the first insert. + // Also: slot_count/live_count/dead_bytes all 0, next_page_id == INVALID_PAGE, + // FirstPageId() == LastPageId(), and CheckInvariants(). + + auto first_page = heap_->FirstPageId(); + ASSERT_EQ(first_page, heap_->LastPageId()); + + auto const page = bpm_->FetchPageRead(first_page); + ASSERT_TRUE(page.has_value()); + ASSERT_EQ(PageType::HEAP, page.value().Header().page_type); + + auto heap_page = AsHeapPage(page.value()); + ASSERT_TRUE(heap_page.has_value()); + + auto heap_sub_header = heap_page.value().Header(); + ASSERT_EQ(heap_sub_header.tuple_data_start, PAGE_BODY_SIZE); + ASSERT_EQ(heap_sub_header.dead_bytes, 0); + ASSERT_EQ(heap_sub_header.live_count, 0); + ASSERT_EQ(heap_sub_header.slot_count, 0); + + ASSERT_TRUE(heap_page.value().CheckInvariants()); +} + +TEST_F(TableHeapTest, CreateDoesNotTakeAReservedPage) { + // FirstPageId() must be neither META_PAGE_ID nor CATALOG_ROOT_PAGE_ID. Cheap, and it pins + // down an assumption the catalog is about to depend on. + + auto first = heap_->FirstPageId(); + ASSERT_NE(first, META_PAGE_ID); + ASSERT_NE(first, CATALOG_ROOT_PAGE_ID); +} + +TEST_F(TableHeapTest, RowsSurviveAReopen) { + // Insert a handful, remember the RIDs and the two page ids, Reopen(first, last), then assert + // every RID still resolves to the same bytes and a scan returns the same rows in the same + // order. This is the only test that proves anything reached the file. + + // Enough to span three pages at ~289 rows each, so the reopen has to follow next_page_id + // links and trust a LastPageId() that moved — a single-page table proves neither. + constexpr int kRows = 700; + + std::vector rids; + for (int i = 0; i < kRows; ++i) rids.push_back(MustInsert(Row(i))); + + const auto before = CollectAll(); + ASSERT_EQ(before.size(), static_cast(kRows)); + + const page_id_t first = heap_->FirstPageId(); + const page_id_t last = heap_->LastPageId(); + ASSERT_NE(first, last) << "precondition: the chain must have grown past one page"; + + ASSERT_NO_FATAL_FAILURE(Reopen(first, last)); + + for (std::size_t i = 0; i < rids.size(); ++i) { + EXPECT_EQ(MustGet(rids[i]), Row(static_cast(i))) + << "rid {" << rids[i].page_id << ", " << rids[i].slot << "}"; + } + EXPECT_EQ(CollectAll(), before); + + ExpectAllInvariants(); + ExpectNoStrandedPins(); +} + +TEST_F(TableHeapTest, OpenValidatesNothingUntilTheFirstFetch) { + // Open() with ids that are not heap pages must SUCCEED, and the first Get/Scan must then fail + // with kCorruption from AsHeapPage. Documents the deliberate choice not to read a page in the + // factory, so nobody "fixes" it later. + // + // The catalog root, not META: the DiskManager refuses to read page 0 at all, so a META-rooted + // table fails the FETCH with kInvalidArgument and never reaches AsHeapPage. The catalog root + // is always present, always readable, and stamped CATALOG — a real page of the wrong type. + // And Scan(), not only Get(): Get goes straight to rid.page_id and never consults the ids + // Open() was given, so a Get-only test would pass through any TableHeap at all. + { + auto opened = TableHeap::Open(*bpm_, CATALOG_ROOT_PAGE_ID, CATALOG_ROOT_PAGE_ID); + ASSERT_TRUE(opened.has_value()) << opened.error().message(); + TableHeap& bogus = *opened.value(); + + { + auto it = bogus.Scan(); + auto more = it.Next(); + ASSERT_FALSE(more.has_value()) << "a scan rooted at the catalog page returned rows"; + EXPECT_EQ(more.error().code(), ErrorCode::kCorruption) << more.error().message(); + } + + EXPECT_EQ(CodeOf(bogus.Get(RID{CATALOG_ROOT_PAGE_ID, 0})), ErrorCode::kCorruption); + } + ExpectNoStrandedPins(); +} + +// ================================================================================================= +// Tier 2 — one row's lifecycle +// ================================================================================================= + +TEST_F(TableHeapTest, InsertThenGetRoundTripsTheBytes) { + // Assert the CONTENT, not just that Get succeeded. Include a tuple with embedded NUL bytes: + // the heap moves opaque blobs, and a length-vs-terminator bug passes every ASCII test. + const RID plain = MustInsert("hello, heap"); + + constexpr char kRaw[] = "nul\0in\0the\0middle"; + const std::string with_nuls(kRaw, sizeof(kRaw) - 1); // keep the NULs, drop the terminator + const RID nuls = MustInsert(with_nuls); + + // Every byte value once, 0x00 through 0xFF — nothing in the path may treat any value as + // special. + std::vector every_byte(256); + for (std::size_t i = 0; i < every_byte.size(); ++i) every_byte[i] = static_cast(i); + auto binary = heap_->Insert(every_byte); + ASSERT_TRUE(binary.has_value()) << binary.error().message(); + + EXPECT_EQ(MustGet(plain), "hello, heap"); + EXPECT_EQ(MustGet(nuls), with_nuls); + EXPECT_EQ(MustGet(nuls).size(), with_nuls.size()); + + auto back = heap_->Get(binary.value()); + ASSERT_TRUE(back.has_value()) << back.error().message(); + EXPECT_EQ(back.value(), every_byte); + + ExpectInvariants(heap_->FirstPageId()); + ExpectNoStrandedPins(); +} + +TEST_F(TableHeapTest, InsertRejectsEmptyAndOversizedTuples) { + // Both kInvalidArgument, and — the part worth asserting — rejected WITHOUT touching a page: + // slot_count and dead_bytes on the first page unchanged, so the validation really did happen + // before the probe rather than at the end of it. + // Boundary: MAX_TUPLE_SIZE exactly must succeed; MAX_TUPLE_SIZE + 1 must not. + const page_id_t first = heap_->FirstPageId(); + const HeapSubHeader before = SubHeaderOf(first); + + const std::string too_big(MAX_TUPLE_SIZE + 1, 'x'); + EXPECT_EQ(CodeOf(heap_->Insert(std::span{})), ErrorCode::kInvalidArgument); + EXPECT_EQ(CodeOf(heap_->Insert(Bytes(too_big))), ErrorCode::kInvalidArgument); + + ExpectSameAccounting(before, SubHeaderOf(first)); + EXPECT_EQ(heap_->LastPageId(), first); + + // Unchanged accounting shows no page was MODIFIED, not that none was fetched. This proves the + // stronger claim: a table rooted at the catalog page fails any probe with kCorruption, so + // getting kInvalidArgument back means the check ran before the first fetch. + { + auto opened = TableHeap::Open(*bpm_, CATALOG_ROOT_PAGE_ID, CATALOG_ROOT_PAGE_ID); + ASSERT_TRUE(opened.has_value()); + EXPECT_EQ(CodeOf(opened.value()->Insert(std::span{})), + ErrorCode::kInvalidArgument); + EXPECT_EQ(CodeOf(opened.value()->Insert(Bytes(too_big))), ErrorCode::kInvalidArgument); + } + + const std::string largest(MAX_TUPLE_SIZE, 'm'); + const RID rid = MustInsert(largest); + EXPECT_EQ(MustGet(rid), largest); + + ExpectInvariants(first); + ExpectNoStrandedPins(); +} + +TEST_F(TableHeapTest, GetOnADeletedRidIsNotFound) { + // Check the CODE is kNotFound, not merely that it failed. A RID handed out before a delete + // gets an answer, never bytes — and never kCorruption, which would send a reader hunting a + // nonexistent disk problem. + const RID keep = MustInsert(Row(1)); + const RID gone = MustInsert(Row(2)); + ASSERT_TRUE(heap_->Delete(gone).ok()); + + EXPECT_EQ(CodeOf(heap_->Get(gone)), ErrorCode::kNotFound); + EXPECT_EQ(MustGet(keep), Row(1)); + + ExpectInvariants(heap_->FirstPageId()); + ExpectNoStrandedPins(); +} + +TEST_F(TableHeapTest, GetOnASlotThatNeverExistedIsNotFound) { + // RID{first_page_id, 9999}. Same code as a deleted slot: out of range and dead are one + // answer from the caller's side. + const page_id_t first = heap_->FirstPageId(); + + // Slot 0 of an EMPTY page is the off-by-one case: slot_count is 0, so even the first index + // is out of range. + EXPECT_EQ(CodeOf(heap_->Get(RID{first, 0})), ErrorCode::kNotFound); + + MustInsert(Row(1)); + EXPECT_EQ(CodeOf(heap_->Get(RID{first, 1})), ErrorCode::kNotFound); // one past the end + EXPECT_EQ(CodeOf(heap_->Get(RID{first, 9999})), ErrorCode::kNotFound); + + ExpectNoStrandedPins(); +} + +TEST_F(TableHeapTest, GetOnAnImpossiblePageIdIsInvalidArgument) { + // A default-constructed RID names INVALID_PAGE, and one is easy to produce by accident — an + // uninitialised RID in an executor, or MustInsert's own fallback. It must come back as + // kInvalidArgument, and the failed fetch must leave no pin and no mapping behind. + // + // Regression test, 2026-09-27: before FetchFrame rejected these at the front door, both ids + // took the miss path, claimed a frame, published a mapping, and then tripped + // `assert(frame.page_id_ >= 1)` in AbandonLoad — aborting any debug build. + EXPECT_EQ(CodeOf(heap_->Get(RID{})), ErrorCode::kInvalidArgument); + EXPECT_EQ(CodeOf(heap_->Get(RID{META_PAGE_ID, 0})), ErrorCode::kInvalidArgument); + + ExpectNoStrandedPins(); + const PoolStats stats = bpm_->GetStats(); + EXPECT_EQ(stats.free_frames, stats.free_list_size); +} + +TEST_F(TableHeapTest, DeleteIsNotIdempotent) { + // Second delete of the same RID is kNotFound, and dead_bytes must NOT move the second time — + // a double-counted delete inflates dead_bytes and eventually breaks CheckInvariants. + const page_id_t first = heap_->FirstPageId(); + MustInsert(Row(1)); + const RID victim = MustInsert(Row(2)); + + ASSERT_TRUE(heap_->Delete(victim).ok()); + const HeapSubHeader after_first = SubHeaderOf(first); + EXPECT_EQ(after_first.dead_bytes, kRowSize); + EXPECT_EQ(after_first.live_count, 1); + + EXPECT_EQ(CodeOf(heap_->Delete(victim)), ErrorCode::kNotFound); + ExpectSameAccounting(after_first, SubHeaderOf(first)); + + ExpectInvariants(first); + ExpectNoStrandedPins(); +} + +// ================================================================================================= +// Tier 3 — the chain and the scan +// ================================================================================================= + +TEST_F(TableHeapTest, InsertExtendsTheChainWhenThePageFills) { + // Insert until LastPageId() changes. Then assert: ChainPages() has exactly two entries, the + // first page's next_page_id names the second, the second's is INVALID_PAGE, LastPageId() + // equals the second, and EVERY rid collected on the way still resolves. The last part is + // what catches a chain that grew correctly while losing rows. + const page_id_t first = heap_->FirstPageId(); + + Rows rows; + for (int i = 0; heap_->LastPageId() == first; ++i) { + ASSERT_LT(i, 10'000) << "chain never grew"; + const std::string text = Row(i); + rows.emplace_back(MustInsert(text), text); + } + const page_id_t second = heap_->LastPageId(); + + // Pins the fixture's arithmetic to the real format: exactly kRowsPerPage fit, the next spilled. + ASSERT_EQ(rows.size(), kRowsPerPage + 1); + EXPECT_EQ(ContiguousOf(first), kFullPageSlack); + EXPECT_EQ(rows.back().first, (RID{second, 0})); + + EXPECT_EQ(ChainPages(), (std::vector{first, second})); + { + auto guard = bpm_->FetchPageRead(first); + ASSERT_TRUE(guard.has_value()); + EXPECT_EQ(guard.value().Header().next_page_id, second); + } + { + auto guard = bpm_->FetchPageRead(second); + ASSERT_TRUE(guard.has_value()); + EXPECT_EQ(guard.value().Header().page_type, PageType::HEAP); + EXPECT_EQ(guard.value().Header().next_page_id, INVALID_PAGE); + } + EXPECT_EQ(SubHeaderOf(second).live_count, 1); + + for (const auto& [rid, text] : rows) EXPECT_EQ(MustGet(rid), text); + + // Nothing allocated that the chain does not reach. + EXPECT_EQ(static_cast(dm_->PageCount()), kReservedPages + 2); + + ExpectAllInvariants(); + ExpectNoStrandedPins(); +} + +TEST_F(TableHeapTest, ScanVisitsEveryLiveRowExactlyOnce) { + // Fill several pages, then compare CollectAll() against a std::set of what was inserted. + // EXACTLY once matters as much as the count: put the RIDs in a set and assert no duplicates, + // because a chain-walk bug that revisits a page shows up as a right-sized wrong answer. + const Rows inserted = Fill(3 * kRowsPerPage + 17); // three full pages and a partial fourth + ASSERT_EQ(ChainPages().size(), 4U); + + const Rows scanned = CollectAll(); + ASSERT_EQ(scanned.size(), inserted.size()); + + std::set rids; + std::set texts; + for (const auto& [rid, text] : scanned) { + EXPECT_TRUE(rids.insert(rid).second) + << "rid {" << rid.page_id << ", " << rid.slot << "} returned twice"; + texts.insert(text); + } + + // Each scanned row must also carry the RID its insert was given, not just the right bytes. + const RowMap expected(inserted.begin(), inserted.end()); + EXPECT_EQ(RowMap(scanned.begin(), scanned.end()), expected); + EXPECT_EQ(texts.size(), inserted.size()); + + ExpectNoStrandedPins(); +} + +TEST_F(TableHeapTest, ScanOnAnEmptyTableTerminatesImmediately) { + // A fresh table's first Next() returns false — no error, no special case at the call site. + // Then a second Next() must ALSO return false rather than walking off the end. + auto it = heap_->Scan(); + + auto first = it.Next(); + ASSERT_TRUE(first.has_value()) << first.error().message(); + EXPECT_FALSE(first.value()); + + auto second = it.Next(); + ASSERT_TRUE(second.has_value()) << second.error().message(); + EXPECT_FALSE(second.value()); + + EXPECT_FALSE(it.Rid().isValid()); + EXPECT_TRUE(it.Tuple().empty()); + + // An exhausted iterator holds nothing, even while it is still alive. + ExpectNoStrandedPins(); +} + +TEST_F(TableHeapTest, ScanSkipsAFullyDeletedPage) { + // Fill three pages, delete every row on the middle one, and assert the scan returns exactly + // the other two pages' rows. live_count == 0 is the skip that makes a drained table cheap to + // scan; this is the test that it actually skips rather than stops. + const Rows inserted = Fill(3 * kRowsPerPage); + const std::vector chain = ChainPages(); + ASSERT_EQ(chain.size(), 3U); + const page_id_t middle = chain[1]; + + Rows expected; + for (const auto& row : inserted) { + if (row.first.page_id == middle) { + ASSERT_TRUE(heap_->Delete(row.first).ok()); + } else { + expected.push_back(row); + } + } + ASSERT_EQ(SubHeaderOf(middle).live_count, 0); + + const Rows scanned = CollectAll(); + EXPECT_EQ(scanned, expected); // same rows, same order — page one's, then page three's + ASSERT_FALSE(scanned.empty()); + EXPECT_EQ(scanned.back().first.page_id, chain[2]) << "the scan stopped at the empty page"; + + ExpectAllInvariants(); + ExpectNoStrandedPins(); +} + +TEST_F(TableHeapTest, IteratorRidNamesTheRowItJustReturned) { + // For each step, Get(it.Rid()) must equal it.Tuple(). That pair is what a B+tree build + // consumes, so a Rid() that lags the cursor by one is a silent index corruption later. + // + // Holes on purpose: with no dead slots, a Rid() that lags by one still names a live row, just + // the wrong one — the holes are what make an off-by-one land on a dead slot and fail loudly. + const Rows inserted = Fill(2 * kRowsPerPage + 5); + RowMap live; + for (std::size_t i = 0; i < inserted.size(); ++i) { + if (i % 7 == 3) { + ASSERT_TRUE(heap_->Delete(inserted[i].first).ok()); + } else { + live.insert(inserted[i]); + } + } + + const Rows scanned = CollectAll(); // Rid() and Tuple() captured at the same step + ASSERT_EQ(scanned.size(), live.size()); + for (const auto& [rid, text] : scanned) { + EXPECT_EQ(MustGet(rid), text) << "rid {" << rid.page_id << ", " << rid.slot << "}"; + EXPECT_EQ(live.at(rid), text); + } + + ExpectNoStrandedPins(); +} + +TEST_F(TableHeapTest, IteratorTupleSurvivesThePageBoundary) { + // Across a page change, the span from the PREVIOUS Next() must still hold the previous row's + // bytes until the next call replaces them — that is the reason tuple_ is a copy. Worth + // running under ASan: the failure mode is reading an evicted frame, which is valid memory. + // + // That failure is silent by construction, so assert the cause rather than wait for a symptom: + // at every step, Tuple() must point into the iterator and NOT into the frame holding the page. + // A view into the frame is exactly what goes stale when the guard drops at the boundary. + const Rows inserted = Fill(kRowsPerPage + 3); // last row of page one, then onto page two + const auto aliases_frame = [&](std::span tuple, page_id_t page_id) { + auto guard = bpm_->FetchPageRead(page_id); // shared latch alongside the iterator's + EXPECT_TRUE(guard.has_value()); + if (!guard.has_value()) return false; + const auto body = guard.value().Body(); + const std::less before; // total order on unrelated pointers + return !before(tuple.data(), body.data()) && + before(tuple.data(), body.data() + body.size()); + }; + + auto it = heap_->Scan(); + for (const auto& [rid, text] : inserted) { + auto more = it.Next(); + ASSERT_TRUE(more.has_value()) << more.error().message(); + ASSERT_TRUE(more.value()); + ASSERT_EQ(it.Rid(), rid); + EXPECT_EQ(AsChars(it.Tuple()), text); + EXPECT_FALSE(aliases_frame(it.Tuple(), rid.page_id)) + << "Tuple() is a view into the frame for rid {" << rid.page_id << ", " << rid.slot + << "}"; + } + ASSERT_NE(inserted[kRowsPerPage - 1].first.page_id, inserted[kRowsPerPage].first.page_id) + << "precondition: the scan must have crossed a page boundary"; +} + +TEST_F(TableHeapTest, MovedFromIteratorIsExhausted) { + // Move-construct mid-scan; the source's Next() must return false, and the destination must + // continue from where the source left off — not restart, not skip a row. + const Rows inserted = Fill(kRowsPerPage + 10); + constexpr std::size_t kSteps = 5; + + auto source = heap_->Scan(); + for (std::size_t i = 0; i < kSteps; ++i) ASSERT_TRUE(source.Next().value_or(false)); + + TableIterator dest(std::move(source)); + + // NOLINTBEGIN(bugprone-use-after-move) — the moved-from state IS the thing under test. + auto from_source = source.Next(); + ASSERT_TRUE(from_source.has_value()); + EXPECT_FALSE(from_source.value()); + EXPECT_FALSE(source.Rid().isValid()); + // NOLINTEND(bugprone-use-after-move) + + // Still on the row the source was on... + EXPECT_EQ(dest.Rid(), inserted[kSteps - 1].first); + EXPECT_EQ(AsChars(dest.Tuple()), inserted[kSteps - 1].second); + + // ...and continues from the very next one, all the way across the page boundary. + for (std::size_t i = kSteps; i < inserted.size(); ++i) { + auto more = dest.Next(); + ASSERT_TRUE(more.has_value()) << more.error().message(); + ASSERT_TRUE(more.value()) << "ended early at row " << i; + EXPECT_EQ(dest.Rid(), inserted[i].first); + EXPECT_EQ(AsChars(dest.Tuple()), inserted[i].second); + } + EXPECT_FALSE(dest.Next().value_or(true)); +} + +TEST_F(TableHeapTest, MoveAssignmentReleasesTheTargetsPin) { + // Two live iterators, move one onto the other, then ExpectNoStrandedPins() once both die. + // Overwriting an engaged optional without dropping it strands a frame for the + // life of the process, and nothing else in the suite would notice. + // + // Parked on DIFFERENT pages, so the stranded pin shows up the moment it happens as a second + // pinned frame — two pins on one frame would only be caught at the very end. + const Rows inserted = Fill(kRowsPerPage + 5); + { + auto target = heap_->Scan(); + ASSERT_TRUE(target.Next().value_or(false)); // row 0, page one + + auto source = heap_->Scan(); + for (std::size_t i = 0; i <= kRowsPerPage; ++i) ASSERT_TRUE(source.Next().value_or(false)); + ASSERT_EQ(source.Rid(), inserted[kRowsPerPage].first); // first row of page two + ASSERT_EQ(bpm_->GetStats().pinned_frames, 2U); + + target = std::move(source); + EXPECT_EQ(bpm_->GetStats().pinned_frames, 1U) << "page one's pin survived the assignment"; + + EXPECT_EQ(target.Rid(), inserted[kRowsPerPage].first); + ASSERT_TRUE(target.Next().value_or(false)); + EXPECT_EQ(target.Rid(), inserted[kRowsPerPage + 1].first); + } + ExpectNoStrandedPins(); +} + +// ================================================================================================= +// Tier 4 — space accounting and reuse +// ================================================================================================= + +TEST_F(TableHeapTest, DeleteRaisesDeadBytesWithoutFreeingContiguousSpace) { + // dead_bytes grows by exactly the tuple length; Contiguous() and tuple_data_start do not + // move. This is the fact that makes compaction necessary, and asserting it stops anyone + // "optimising" Delete into something that looks like it frees space. + const page_id_t first = heap_->FirstPageId(); + const Rows rows = Fill(3); + + const HeapSubHeader before = SubHeaderOf(first); + const std::size_t contiguous_before = ContiguousOf(first); + + ASSERT_TRUE(heap_->Delete(rows[1].first).ok()); // the middle one: not at the low-water mark + + const HeapSubHeader after = SubHeaderOf(first); + EXPECT_EQ(after.dead_bytes, before.dead_bytes + kRowSize); + EXPECT_EQ(after.tuple_data_start, before.tuple_data_start); + EXPECT_EQ(after.slot_count, before.slot_count); + EXPECT_EQ(after.live_count, before.live_count - 1); + EXPECT_EQ(ContiguousOf(first), contiguous_before); + + ExpectInvariants(first); +} + +TEST_F(TableHeapTest, InsertCompactsWhenOnlyReclaimableSpaceRemains) { + // Fill a page, delete enough scattered rows to clear the compaction threshold, then insert a + // row that fits only after a rewrite. Assert it landed on THAT page (rid.page_id), that + // dead_bytes fell to 0, and that Contiguous() grew. The whole free-space story in one test. + // + // Page two has plenty of room. That is what makes rid.page_id meaningful: without + // compact-on-probe the row would simply land on page two instead. + const page_id_t first = heap_->FirstPageId(); + const ScatteredFirstPage setup = FillThenScatterFirstPage(); + const page_id_t last = heap_->LastPageId(); + ASSERT_NE(last, first); + + const HeapSubHeader before = SubHeaderOf(first); + ASSERT_EQ(ContiguousOf(first), kFullPageSlack); + ASSERT_GE(before.dead_bytes, kCompactThreshold) << "precondition: below the threshold"; + + const std::string wide(200, 'c'); // needs a compaction: 200 >> kFullPageSlack + const RID rid = MustInsert(wide); + + EXPECT_EQ(rid.page_id, first) << "compact-on-probe did not fire; the row went elsewhere"; + EXPECT_EQ(rid.slot, setup.deleted.front().slot) << "expected the first dead slot to be reused"; + EXPECT_EQ(heap_->LastPageId(), last); + EXPECT_EQ(MustGet(rid), wide); + + // Exact, not just "grew": the compaction handed back every dead byte, the insert took 200 of + // them, and the reused slot cost no new entry. + const HeapSubHeader after = SubHeaderOf(first); + EXPECT_EQ(after.dead_bytes, 0); + EXPECT_EQ(ContiguousOf(first), kFullPageSlack + before.dead_bytes - wide.size()); + EXPECT_EQ(after.slot_count, before.slot_count); + + ExpectAllInvariants(); + ExpectNoStrandedPins(); +} + +TEST_F(TableHeapTest, CompactionPreservesEveryExistingRid) { + // THE IMPORTANT ONE. Record every {rid, bytes} pair before the compaction above, then after + // it assert every single rid still resolves to the same bytes. "Slots never move, tuple bytes + // move freely" is the format's central promise, and compaction is the only place it can + // break. A failure here invalidates every index that will ever be built on this heap. + const page_id_t first = heap_->FirstPageId(); + const ScatteredFirstPage setup = FillThenScatterFirstPage(); + + const RID fresh = MustInsert(std::string(200, 'c')); + ASSERT_EQ(fresh.page_id, first); + ASSERT_EQ(SubHeaderOf(first).dead_bytes, 0) << "precondition: the compaction must have run"; + + for (const auto& [rid, text] : setup.survivors) { + EXPECT_EQ(MustGet(rid), text) << "rid {" << rid.page_id << ", " << rid.slot << "}"; + } + + // A deleted RID stays gone — except the one slot the new row just reused, which now names it. + for (const RID& rid : setup.deleted) { + if (rid == fresh) continue; + EXPECT_EQ(CodeOf(heap_->Get(rid)), ErrorCode::kNotFound) + << "rid {" << rid.page_id << ", " << rid.slot << "} came back from the dead"; + } + + EXPECT_EQ(CollectAll().size(), setup.survivors.size() + 1); + ExpectAllInvariants(); + ExpectNoStrandedPins(); +} + +TEST_F(TableHeapTest, InsertReusesADeadSlotIndex) { + // Delete a row from the middle of a page, insert another that fits, and assert the new RID + // carries the SAME slot index. Then assert slot_count did not grow — the four bytes of the + // slot entry are what was reclaimed. + // And the counter-intuitive half: dead_bytes must NOT drop. Reusing a slot reclaims the + // entry, never the tuple bytes; only Compact does that. + const page_id_t first = heap_->FirstPageId(); + const Rows rows = Fill(5); + ASSERT_TRUE(heap_->Delete(rows[2].first).ok()); + const HeapSubHeader before = SubHeaderOf(first); + + const RID reused = MustInsert(Row(100)); + + EXPECT_EQ(reused, rows[2].first); + const HeapSubHeader after = SubHeaderOf(first); + EXPECT_EQ(after.slot_count, before.slot_count); + EXPECT_EQ(after.live_count, before.live_count + 1); + EXPECT_EQ(after.dead_bytes, before.dead_bytes) << "reusing a slot does not reclaim bytes"; + EXPECT_EQ(after.tuple_data_start, before.tuple_data_start - kRowSize); + + EXPECT_EQ(MustGet(reused), Row(100)); + ExpectInvariants(first); +} + +TEST_F(TableHeapTest, FreedSpaceIsReusedRatherThanAppended) { + // Fill two pages, drain the FIRST one, then insert. The row must land on the reused page and + // LastPageId() must not move. Without the insert hint plus compact-on-probe the row would be + // appended and the file would grow forever under churn — the bound DD-004 exists to fix. + const page_id_t first = heap_->FirstPageId(); + const Rows rows = Fill(2 * kRowsPerPage); + ASSERT_EQ(ChainPages().size(), 2U); + const page_id_t last = heap_->LastPageId(); + const page_id_t pages_before = dm_->PageCount(); + + for (const auto& [rid, text] : rows) { + if (rid.page_id == first) ASSERT_TRUE(heap_->Delete(rid).ok()); + } + + const RID rid = MustInsert(Row(900'000)); + + EXPECT_EQ(rid.page_id, first); + EXPECT_EQ(heap_->LastPageId(), last); + EXPECT_EQ(ChainPages().size(), 2U); + EXPECT_EQ(dm_->PageCount(), pages_before) << "the file grew"; + EXPECT_EQ(MustGet(rid), Row(900'000)); + + ExpectAllInvariants(); + ExpectNoStrandedPins(); +} + +// ================================================================================================= +// Tier 5 — update +// ================================================================================================= + +TEST_F(TableHeapTest, UpdateInPlaceKeepsTheRid) { + // Shrinking and same-size updates both return the original RID and the new bytes read back. + // For the shrink, assert dead_bytes grew by exactly the difference — those orphaned bytes are + // referenced by no slot and no offset, so this counter is the only record they exist. + const page_id_t first = heap_->FirstPageId(); + const RID rid = MustInsert("0123456789abcdef"); + + auto same_size = heap_->Update(rid, Bytes("fedcba9876543210")); + ASSERT_TRUE(same_size.has_value()) << same_size.error().message(); + EXPECT_EQ(same_size.value(), rid); + EXPECT_EQ(MustGet(rid), "fedcba9876543210"); + EXPECT_EQ(SubHeaderOf(first).dead_bytes, 0); + + auto shrunk = heap_->Update(rid, Bytes("short")); + ASSERT_TRUE(shrunk.has_value()) << shrunk.error().message(); + EXPECT_EQ(shrunk.value(), rid); + EXPECT_EQ(MustGet(rid), "short"); + EXPECT_EQ(SubHeaderOf(first).dead_bytes, 16 - 5); + + ExpectInvariants(first); + ExpectNoStrandedPins(); +} + +TEST_F(TableHeapTest, UpdateThatGrowsWithinThePageKeepsTheRid) { + // Growing but still fitting — with and without needing a compaction first. Both are + // kSamePage, so the RID survives; assert the bytes and CheckInvariants after each. + const page_id_t first = heap_->FirstPageId(); + const Rows rows = Fill(kRowsPerPage); // full: only kFullPageSlack contiguous bytes left + for (std::size_t i = 10; i < 30; ++i) ASSERT_TRUE(heap_->Delete(rows[i].first).ok()); + ASSERT_EQ(SubHeaderOf(first).dead_bytes, 20 * kRowSize); + + // Slow path: 100 bytes cannot fit in the slack, but does after reclaiming the 200 dead bytes + // plus the row's own 10. Afterwards: 10 + 200 + 10 - 100 contiguous, nothing dead. + const std::string grown(100, 'g'); + auto slow = heap_->Update(rows[0].first, Bytes(grown)); + ASSERT_TRUE(slow.has_value()) << slow.error().message(); + EXPECT_EQ(slow.value(), rows[0].first); + EXPECT_EQ(MustGet(rows[0].first), grown); + EXPECT_EQ(SubHeaderOf(first).dead_bytes, 0) << "the slow path must have compacted"; + EXPECT_EQ(ContiguousOf(first), kFullPageSlack + 20 * kRowSize + kRowSize - grown.size()); + ExpectInvariants(first); + + // Fast path: 50 bytes now fits at the low-water mark, so the row relocates within the page + // with no compaction, and its old 10 bytes become dead. + const std::size_t contiguous = ContiguousOf(first); + const std::string medium(50, 'm'); + auto fast = heap_->Update(rows[1].first, Bytes(medium)); + ASSERT_TRUE(fast.has_value()) << fast.error().message(); + EXPECT_EQ(fast.value(), rows[1].first); + EXPECT_EQ(MustGet(rows[1].first), medium); + EXPECT_EQ(SubHeaderOf(first).dead_bytes, kRowSize); + EXPECT_EQ(ContiguousOf(first), contiguous - medium.size()); + ExpectInvariants(first); + + // Neighbours untouched by either rewrite. + for (std::size_t i = 30; i < rows.size(); ++i) + EXPECT_EQ(MustGet(rows[i].first), rows[i].second); + EXPECT_EQ(heap_->LastPageId(), first); + ExpectNoStrandedPins(); +} + +TEST_F(TableHeapTest, UpdateRelocatesWhenThePageCannotHoldTheRow) { + // Fill a page, then update one row to something that cannot fit even after compaction. + // Assert: returned RID differs from the input, the OLD rid is now kNotFound, the new one + // holds the new bytes, and a scan sees the row EXACTLY ONCE. That last assertion is the one + // that catches an unchecked delete leaving a duplicate behind. + const page_id_t first = heap_->FirstPageId(); + const Rows rows = Fill(kRowsPerPage); // full, nothing dead: nowhere on this page to grow + const RID old_rid = rows[0].first; + + const std::string wide(500, 'w'); + auto moved = heap_->Update(old_rid, Bytes(wide)); + ASSERT_TRUE(moved.has_value()) << moved.error().message(); + const RID new_rid = moved.value(); + + EXPECT_NE(new_rid, old_rid); + EXPECT_NE(new_rid.page_id, first); + EXPECT_EQ(CodeOf(heap_->Get(old_rid)), ErrorCode::kNotFound); + EXPECT_EQ(MustGet(new_rid), wide); + + const Rows scanned = CollectAll(); + EXPECT_EQ(scanned.size(), rows.size()); + std::size_t wide_seen = 0; + std::size_t old_seen = 0; + for (const auto& [rid, text] : scanned) { + if (text == wide) ++wide_seen; + if (text == rows[0].second) ++old_seen; + } + EXPECT_EQ(wide_seen, 1U); + EXPECT_EQ(old_seen, 0U) << "the original survived the relocation — the row is duplicated"; + + ExpectAllInvariants(); + ExpectNoStrandedPins(); +} + +TEST_F(TableHeapTest, UpdateOnADeletedRidIsNotFound) { + // kNotFound, not kCorruption and not kInternal — the code HeapPage::Update goes out of its + // way to preserve, and the one most easily flattened on the way back up. + const page_id_t first = heap_->FirstPageId(); + MustInsert(Row(1)); + const RID gone = MustInsert(Row(2)); + ASSERT_TRUE(heap_->Delete(gone).ok()); + + EXPECT_EQ(CodeOf(heap_->Update(gone, Bytes(Row(3)))), ErrorCode::kNotFound); + EXPECT_EQ(CodeOf(heap_->Update(RID{first, 9999}, Bytes(Row(3)))), ErrorCode::kNotFound); + + // And the failed update must not have resurrected or relocated anything. + EXPECT_EQ(CollectAll().size(), 1U); + ExpectInvariants(first); + ExpectNoStrandedPins(); +} + +TEST_F(TableHeapTest, UpdateRejectsEmptyAndOversizedTuples) { + // kInvalidArgument, and the row must be UNCHANGED afterwards — a rejected update that + // half-applied is worse than one that fails. + const page_id_t first = heap_->FirstPageId(); + const RID rid = MustInsert(Row(1)); + const HeapSubHeader before = SubHeaderOf(first); + + EXPECT_EQ(CodeOf(heap_->Update(rid, std::span{})), + ErrorCode::kInvalidArgument); + EXPECT_EQ(CodeOf(heap_->Update(rid, Bytes(std::string(MAX_TUPLE_SIZE + 1, 'x')))), + ErrorCode::kInvalidArgument); + + EXPECT_EQ(MustGet(rid), Row(1)); + ExpectSameAccounting(before, SubHeaderOf(first)); + EXPECT_EQ(heap_->LastPageId(), first); + ExpectNoStrandedPins(); +} + +// ================================================================================================= +// Tier 6 — concurrency. Run these under TSan. +// ================================================================================================= + +TEST_F(TableHeapTest, ConcurrentInsertsAllLandExactlyOnce) { + // N threads × M rows, each row self-identifying by thread and index. Afterwards a scan must + // return exactly N×M rows with no duplicates and no losses, and every returned RID must be + // distinct. Collect each thread's RIDs and assert the union has no repeats — two inserts + // handed the same RID is a different bug from a lost row, and the row count alone cannot + // tell them apart. + // + // Eight threads fit the sixteen-frame pool: at most one thread holds two guards (extension is + // serialised on the last page's latch), so the worst case is N + 1 frames. + constexpr int kThreads = 8; + constexpr int kPerThread = 500; // ~14 pages, so the chain extends under contention + + std::vector per_thread(kThreads); + std::atomic failures{0}; + std::barrier start(kThreads); + { + std::vector threads; + for (int t = 0; t < kThreads; ++t) { + threads.emplace_back([&, t] { + Rows& mine = per_thread[static_cast(t)]; + start.arrive_and_wait(); + for (int i = 0; i < kPerThread; ++i) { + std::string text = std::format("t{:02d}-{:06d}", t, i); // kRowSize bytes + auto rid = heap_->Insert(Bytes(text)); + if (!rid.has_value()) { + ++failures; + continue; + } + mine.emplace_back(rid.value(), std::move(text)); + } + }); + } + for (auto& thread : threads) thread.join(); + } + + ASSERT_EQ(failures.load(), 0); + + RowMap acknowledged; + for (const Rows& rows : per_thread) { + for (const auto& [rid, text] : rows) { + EXPECT_TRUE(acknowledged.emplace(rid, text).second) + << "rid {" << rid.page_id << ", " << rid.slot << "} handed to two inserts"; + } + } + ASSERT_EQ(acknowledged.size(), static_cast(kThreads * kPerThread)); + + const Rows scanned = CollectAll(); + EXPECT_EQ(scanned.size(), acknowledged.size()); + EXPECT_EQ(RowMap(scanned.begin(), scanned.end()), acknowledged); + + ExpectAllInvariants(); + ExpectNoStrandedPins(); +} + +TEST_F(TableHeapTest, ConcurrentChainExtensionLeaksNoPage) { + // THE RACE THIS LAYER EXISTS TO GET RIGHT. Size the rows so the first page fills exactly as + // the threads pile in — a std::barrier released right at the boundary is the reliable way. + // + // Two threads both seeing next_page_id == INVALID_PAGE both allocate, both write the link, + // one overwrites the other, and a page is leaked: allocated, unreachable, holding rows that + // no scan will ever return, with no error raised anywhere. + // + // So assert on the FILE, not just the rows: DiskManager::PageCount() must equal + // ChainPages().size() plus the reserved pages. A row-count assertion alone passes while a + // page leaks, because the leaked page's rows were never acknowledged to anyone. + // + // Rounds rather than one burst: each round releases every thread at once from the barrier and + // inserts slightly more than one page's worth between them, so every round crosses at least + // one page boundary with all threads active — kRounds contended extensions, not one. + constexpr int kThreads = 8; + constexpr int kRounds = 40; + constexpr int kPerRound = static_cast(kRowsPerPage) / kThreads + 1; + + std::vector per_thread(kThreads); + std::atomic failures{0}; + std::barrier round(kThreads); + { + std::vector threads; + for (int t = 0; t < kThreads; ++t) { + threads.emplace_back([&, t] { + Rows& mine = per_thread[static_cast(t)]; + for (int r = 0; r < kRounds; ++r) { + round.arrive_and_wait(); + for (int i = 0; i < kPerRound; ++i) { + std::string text = std::format("{}-{:02d}-{:05d}", t, r, i); // kRowSize + auto rid = heap_->Insert(Bytes(text)); + if (!rid.has_value()) { + ++failures; + continue; + } + mine.emplace_back(rid.value(), std::move(text)); + } + } + }); + } + for (auto& thread : threads) thread.join(); + } + + ASSERT_EQ(failures.load(), 0); + + const std::vector chain = ChainPages(); + EXPECT_EQ(static_cast(dm_->PageCount()), chain.size() + kReservedPages) + << "a page was allocated that the chain does not reach"; + EXPECT_EQ(chain.back(), heap_->LastPageId()); + EXPECT_EQ(std::set(chain.begin(), chain.end()).size(), chain.size()) + << "the chain has a cycle or a repeated page"; + + RowMap acknowledged; + for (const Rows& rows : per_thread) acknowledged.insert(rows.begin(), rows.end()); + ASSERT_EQ(acknowledged.size(), static_cast(kThreads * kRounds * kPerRound)); + + const Rows scanned = CollectAll(); + EXPECT_EQ(RowMap(scanned.begin(), scanned.end()), acknowledged); + + ExpectAllInvariants(); + ExpectNoStrandedPins(); +} + +TEST_F(TableHeapTest, ConcurrentInsertAndDeleteKeepPageAccountingSound) { + // Inserters and deleters on one table, then ExpectAllInvariants() at quiescence. dead_bytes, + // live_count and tuple_data_start are updated by different operations under the same latch; + // this is the test that the latch really covers all of them together. + // + // Deleters take disjoint shares of the even-indexed preloaded rows, so every delete must + // succeed. Each delete also points the insert hint at its page, so inserters chase freed + // space and compact-on-probe runs concurrently with deletes on the same pages. + constexpr std::size_t kPreload = 4 * kRowsPerPage; + constexpr int kDeleters = 4; + constexpr int kInserters = 4; + constexpr int kPerInserter = 400; + + Rows preloaded; + for (std::size_t i = 0; i < kPreload; ++i) { + std::string text = std::format("p-{:08d}", i); // kRowSize bytes + preloaded.emplace_back(MustInsert(text), std::move(text)); + } + + std::atomic failures{0}; + std::barrier start(kDeleters + kInserters); + { + std::vector threads; + for (int d = 0; d < kDeleters; ++d) { + threads.emplace_back([&, d] { + start.arrive_and_wait(); + for (std::size_t i = 0; i < kPreload; i += 2) { + if ((i / 2) % static_cast(kDeleters) != + static_cast(d)) + continue; + if (!heap_->Delete(preloaded[i].first).ok()) ++failures; + } + }); + } + for (int w = 0; w < kInserters; ++w) { + threads.emplace_back([&, w] { + start.arrive_and_wait(); + for (int i = 0; i < kPerInserter; ++i) { + auto rid = heap_->Insert(Bytes(std::format("i{}-{:07d}", w, i))); // kRowSize + if (!rid.has_value()) ++failures; + } + }); + } + for (auto& thread : threads) thread.join(); + } + + ASSERT_EQ(failures.load(), 0); + ExpectAllInvariants(); + + // Content, as a multiset so a duplicated row is caught as well as a lost one. RIDs are not + // compared: deleted slots are reused by the inserters, so a preloaded RID may now name a new + // row, legitimately. + std::multiset expected; + for (std::size_t i = 1; i < kPreload; i += 2) expected.insert(preloaded[i].second); + for (int w = 0; w < kInserters; ++w) { + for (int i = 0; i < kPerInserter; ++i) expected.insert(std::format("i{}-{:07d}", w, i)); + } + + std::multiset actual; + std::set rids; + for (const auto& [rid, text] : CollectAll()) { + actual.insert(text); + EXPECT_TRUE(rids.insert(rid).second); + } + EXPECT_EQ(actual, expected); + ExpectNoStrandedPins(); +} + +TEST_F(TableHeapTest, CollectThenMutateDoesNotDeadlock) { + // Single-threaded, and it is a REGRESSION TEST FOR A HANG. Scan to completion collecting + // RIDs, let the iterator die, THEN delete them; assert the survivors. + // + // The forbidden shape — deleting inside the loop — takes a write latch on the page the + // iterator still holds a read latch on. std::shared_mutex is not recursive and does not + // upgrade, so it blocks forever on the first match: one thread, no race, every run. + // + // Keep the ctest TIMEOUT meaningful for this target. A regression here does not fail, it + // hangs, and a hung test blocks the whole run instead of reporting. + const Rows inserted = Fill(2 * kRowsPerPage + 50); + const auto doomed = [](std::string_view text) { + return std::stoi(std::string(text.substr(4))) % 3 == 0; // "row-NNNNNN" + }; + + std::vector matches; + { + auto it = heap_->Scan(); + for (;;) { + auto more = it.Next(); + ASSERT_TRUE(more.has_value()) << more.error().message(); + if (!more.value()) break; + if (doomed(AsChars(it.Tuple()))) matches.push_back(it.Rid()); + } + } // the iterator, and its read latch, die HERE — before the first delete + + ASSERT_FALSE(matches.empty()); + for (const RID& rid : matches) EXPECT_TRUE(heap_->Delete(rid).ok()); + + Rows survivors; + for (const auto& row : inserted) { + if (!doomed(row.second)) survivors.push_back(row); + } + EXPECT_EQ(CollectAll(), survivors); + + ExpectAllInvariants(); + ExpectNoStrandedPins(); +} + +} // namespace kernsql