Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 2 additions & 1 deletion families/llama/runtime/CMakeLists.txt
Original file line number Diff line number Diff line change
Expand Up @@ -53,7 +53,8 @@ if(TRTMC_BUILD_TESTS)
trtmc_core nlohmann_json::nlohmann_json ${TRTMC_CUDART_LIBRARY})
endforeach()
add_test(NAME test_llama_speculative_contract COMMAND test_llama_speculative_contract)
foreach(test_name IN ITEMS test_llama_chat_template test_llama_native_kv_cache)
foreach(test_name IN ITEMS test_llama_chat_template test_llama_native_kv_cache
test_llama_pretokenizer_variant)
add_executable(${test_name}
${PROJECT_SOURCE_DIR}/families/llama/tests/cpp/${test_name}.cpp
)
Expand Down
39 changes: 36 additions & 3 deletions families/llama/runtime/bpe_tokenizer.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -1272,19 +1272,52 @@ class BpeTokenizer final : public ITokenizer {
return pretok::Variant::kLlama;
}

// Detect variant from the Split inside a Sequence pre_tokenizer.
// Detect variant from the Split step(s) inside a Sequence pre_tokenizer.
// Some checkpoints isolate digit-grouping (\p{N}{1,3}) as its own Split
// step, ahead of the step whose regex actually identifies the variant
// (e.g. Qwen3's `[^\r\n...`). That earlier step's regex never matches
// any known variant signature and falls through to kLlama, so scanning
// stops before reaching the real classifier. Scan every Split step for
// classification instead of only the first; take the digit-group size
// independently, from whichever step's own regex encodes one, since the
// classifying step and the digit-grouping step are not guaranteed to be
// the same step.
static pretok::Variant detect_split_variant(const nlohmann::json& pt, int& digit_group_out) {
digit_group_out = 0;
if (!pt.contains("pretokenizers"))
return pretok::Variant::kLlama;

pretok::Variant variant = pretok::Variant::kLlama;
bool variant_found = false;
for (auto& sub : pt["pretokenizers"]) {
if (sub.value("type", "") != "Split")
continue;
if (!sub.contains("pattern") || !sub["pattern"].contains("Regex"))
continue;
return classify_split(sub, digit_group_out);
const auto regex = sub["pattern"]["Regex"].get<std::string>();

if (digit_group_out == 0) {
if (int group = parse_digit_group(regex); group > 0)
digit_group_out = group;
}

int step_digit_group = 0; // discarded; digit_group_out is tracked separately above
pretok::Variant step_variant = classify_split(sub, step_digit_group);
if (step_variant == pretok::Variant::kLlama)
continue;
if (variant_found && step_variant != variant) {
throw std::runtime_error(
"Ambiguous Sequence pre_tokenizer: Split steps classify to different "
"variants (" +
std::to_string(static_cast<int>(variant)) + " and " +
std::to_string(static_cast<int>(step_variant)) + ")");
}
if (!variant_found) {
variant = step_variant;
variant_found = true;
}
}
return pretok::Variant::kLlama;
return variant;
}

static bool is_space_split_pre_tokenizer(const nlohmann::json& pt, const std::string& pt_type) {
Expand Down
65 changes: 65 additions & 0 deletions families/llama/tests/MINICPM5_1B.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,65 @@
# MiniCPM5-1B checkpoint coverage

This adds a pinned E2E selection for `openbmb/MiniCPM5-1B` through the existing
Llama family. It does not add a new architecture or claim GPU qualification.

The configuration fixture is copied from the Apache-2.0 publisher checkpoint:
https://huggingface.co/openbmb/MiniCPM5-1B/blob/87179e5c1f455ef22e6223592d2d61351b525bfc/config.json

Unlike the existing 2B checkpoint, the 1B model has hidden width 1536 but
16 attention heads of width 128 (attention width 2048), with two KV heads and
24 layers. CPU tests preserve that explicit width, full-context KV byte
geometry and the two stop IDs. The E2E uses the existing family build/runtime
and reference comparison with thinking disabled, FP16 candidate, FP32 reference,
a 256-token build limit and ten generated tokens. These limits are a small
parity experiment, not a long-context or performance claim.

Validation remaining: build the pinned checkpoint on authorized GPU hardware
and run `families/llama/tests/test_e2e.py` with `--e2e-model minicpm5-1b`, using
the normal family E2E runtime environment. No large weights were downloaded locally.

The Llama tokenizer's Sequence/Split classification fix is owned by upstream
PR #1423. Its original commit is included as a dependency for combined testing;
maintainers should review and integrate #1423 first. This checkpoint coverage
must not be treated as complete runtime qualification until target-GPU parity
is verified.

## Tokenizer-sensitive premerge coverage

The second premerge case uses raw Chinese punctuation, a ten-digit sequence and
an English continuation prompt. Its `expected_prompt_token_ids` come from the
pinned publisher tokenizer, including its `<s>` post-processor token. It uses the
same FP16/FP32 generation comparison and limits as the chat case; no acceptance
threshold is changed. A separate native prefill receipt assertion requires
16 tokens, including BOS, so the observed 18-token baseline cannot pass merely
by generating similar text. This assertion retains the subsequent Hugging Face
generation comparison; it does not turn the case into a KV-contract-only test.
It checks token count, not equality of every native input ID. Selecting `--e2e-model minicpm5-1b` selects both cases.

CPU comparison against Hugging Face Tokenizers 0.22.2 found six mismatches out
of 28 text probes on the initial PR source (f0c70eac). Replacing only the Llama
tokenizer source with PR #1423 at f509329e735f89acb05a24da96e1fa72461a35de made
all 28 probes match. That comparison disabled special-token insertion on both
sides and covered digits, Chinese punctuation, code, whitespace and multilingual
text. The new manifest prompt was also checked separately against the publisher
post-processor. These are tokenizer-only CPU results, not model inference proof.

For example, `1234567890` encodes as `[5645, 12740, 17371, 37]` without special
tokens in the pinned tokenizer. The initial PR runtime instead produced
`[5645, 12740, 1877, 1609]`. The dependency fixes that discrepancy. The branch now includes the original #1423 commit as a history-preserving
dependency merge. The tokenizer implementation retains its original author and
commit; this change does not reimplement it. The combined branch still requires
target-GPU validation before the checkpoint is qualified.

## Staged checkpoint prerequisite

In offline mode, the Llama E2E lookup consumes a pre-populated Hugging Face
snapshot cache without refreshing Hub metadata. Community GPU CI stages that cache
before entering its offline test container. For a manual run, populate the cache
with the pinned checkpoint first and set `HF_HUB_OFFLINE=1`; an absent revision or missing `config.json`
still fails rather than selecting another checkpoint.

This avoids a Hub 1.32 repository-tree lookup that otherwise raises
`OfflineModeIsEnabled` even when the pinned snapshot is already cached. The
regression tests use temporary local snapshots without any tree index or model
weights. No manifest revisions, comparisons or acceptance thresholds are changed.
191 changes: 191 additions & 0 deletions families/llama/tests/cpp/test_llama_pretokenizer_variant.cpp
Original file line number Diff line number Diff line change
@@ -0,0 +1,191 @@
/*
* SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
* SPDX-License-Identifier: Apache-2.0
*/

// detect_split_variant() previously returned on the first Split step found
// in a Sequence pre_tokenizer, even when that step's regex classifies as
// the generic kLlama default. Checkpoints that isolate digit-grouping
// (\p{N}{1,3}) as its own Split step ahead of the step that actually
// identifies the variant (e.g. Qwen3's `[^\r\n...` signature) were
// misclassified as kLlama, silently disabling variant-specific
// pre-tokenization behavior (trailing-newline attachment, grouped digit
// runs) for the whole checkpoint.
//
// The first two checks below use the pre_tokenizer field exactly as
// published in openbmb/MiniCPM5-2B's tokenizer.json (a real, public
// checkpoint that triggers this misclassification), so a passing result
// here reflects the real checkpoint's own regex shapes, not a synthetic
// stand-in. The third check is a synthetic fixture verifying the
// ambiguous-classification case (two Split steps that each classify to a
// different, non-kLlama variant) throws rather than silently picking one.

#include "families/llama/runtime/tokenizer.h"

#include <cstdint>
#include <iostream>
#include <stdexcept>
#include <string>
#include <vector>

namespace {

int failures = 0;

void check_ids(const std::vector<int32_t>& actual, const std::vector<int32_t>& expected,
const char* name) {
if (actual == expected)
return;
std::cerr << "FAIL: " << name << " expected";
for (const int32_t token : expected)
std::cerr << ' ' << token;
std::cerr << " got";
for (const int32_t token : actual)
std::cerr << ' ' << token;
std::cerr << '\n';
++failures;
}

// MiniCPM5-2B's actual pre_tokenizer field (openbmb/MiniCPM5-2B,
// tokenizer.json): a digit-grouping Split step first, then the real
// classifying (Qwen3-shaped) Split step second, then ByteLevel.
constexpr const char* kMiniCPM5PreTokenizer = R"(
"pre_tokenizer": {
"type": "Sequence",
"pretokenizers": [
{
"type": "Split",
"pattern": {"Regex": "\\p{N}{1,3}"},
"behavior": "Isolated",
"invert": false
},
{
"type": "Split",
"pattern": {
"Regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?\\p{L}+|\\p{N}+| ?[^\\s\\p{L}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+"
},
"behavior": "Isolated",
"invert": false
},
{
"type": "ByteLevel",
"add_prefix_space": false,
"trim_offsets": true,
"use_regex": false
}
]
})";

// Trailing newlines after punctuation should stay attached to the same
// pre-token (Qwen3-variant behavior) rather than split off as their own
// whitespace token. Only reachable if the second Split step's regex is
// actually used to classify the variant, not the first.
void check_newline_attachment() {
const std::string tokenizer_json = std::string(R"({
"model": {
"type": "BPE",
"vocab": {
".": 0,
"Ċ": 1,
"ĊĊ": 2,
".ĊĊ": 3
},
"merges": ["Ċ Ċ", ". ĊĊ"]
},)") + kMiniCPM5PreTokenizer + R"(
})";

auto tokenizer = trtmc::CreateBpeTokenizer(tokenizer_json.data(), tokenizer_json.size(), false);
if (!tokenizer) {
std::cerr << "FAIL: check_newline_attachment: tokenizer was not created\n";
++failures;
return;
}
check_ids(tokenizer->encode(".\n\n"), {3}, "check_newline_attachment");
}

// Multi-digit runs should be grouped into chunks of up to 3 digits
// (\p{N}{1,3}), each its own pre-token, not left as one unbounded digit
// run (the kLlama-misclassified behavior) and not split into individual
// one-digit pre-tokens (the failure mode of a fix that scans for the
// variant but couples digit-group size to the same winning step, which
// does not carry the {1,3} grouping in this checkpoint's real regex).
void check_digit_grouping() {
const std::string tokenizer_json = std::string(R"({
"model": {
"type": "BPE",
"vocab": {
"1": 0,
"5": 1,
"0": 2,
"15": 3,
"150": 4,
"1500": 5
},
"merges": ["1 5", "15 0", "150 0"]
},)") + kMiniCPM5PreTokenizer + R"(
})";

auto tokenizer = trtmc::CreateBpeTokenizer(tokenizer_json.data(), tokenizer_json.size(), false);
if (!tokenizer) {
std::cerr << "FAIL: check_digit_grouping: tokenizer was not created\n";
++failures;
return;
}
// Correct (grouped by 3): ["150", "0"] -> ids {4, 2}.
// kLlama misclassification (unpatched bug): one ungrouped word "1500" -> {5}.
// Coupled/naive fix (digit_group taken from the same step as the
// variant, which for this checkpoint's real second step is 0): four
// one-digit words -> {0, 1, 2, 2}.
check_ids(tokenizer->encode("1500"), {4, 2}, "check_digit_grouping");
}

// Synthetic fixture: two Split steps that each classify to a different,
// non-kLlama variant. Not a real checkpoint shape -- this exercises the
// ambiguous-classification guard rather than reproducing a reported bug.
void check_ambiguous_variants_throw() {
const std::string tokenizer_json = R"({
"model": {
"type": "BPE",
"vocab": {"a": 0},
"merges": []
},
"pre_tokenizer": {
"type": "Sequence",
"pretokenizers": [
{
"type": "Split",
"pattern": {"Regex": "[^(\\s)]+"},
"behavior": "Isolated",
"invert": false
},
{
"type": "Split",
"pattern": {"Regex": "[^\\r\\n\\p{L}\\p{N}]?\\p{L}+"},
"behavior": "Isolated",
"invert": false
}
]
}
})";

bool threw = false;
try {
trtmc::CreateBpeTokenizer(tokenizer_json.data(), tokenizer_json.size(), false);
} catch (const std::runtime_error&) {
threw = true;
}
if (!threw) {
std::cerr
<< "FAIL: check_ambiguous_variants_throw: expected std::runtime_error, none thrown\n";
++failures;
}
}

} // namespace

int main() {
check_newline_attachment();
check_digit_grouping();
check_ambiguous_variants_throw();
return failures == 0 ? 0 : 1;
}
30 changes: 30 additions & 0 deletions families/llama/tests/fixtures/minicpm5-1b-config.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,30 @@
{
"_name_or_path": "openbmb/MiniCPM5-1B",
"architectures": [
"LlamaForCausalLM"
],
"bos_token_id": 0,
"eos_token_id": [
1,
130073
],
"pad_token_id": 1,
"hidden_act": "silu",
"hidden_size": 1536,
"initializer_range": 0.02,
"intermediate_size": 4608,
"max_position_embeddings": 131072,
"model_type": "llama",
"num_attention_heads": 16,
"num_hidden_layers": 24,
"num_key_value_heads": 2,
"head_dim": 128,
"rms_norm_eps": 1e-06,
"rope_theta": 5000000,
"rope_scaling": null,
"tie_word_embeddings": false,
"torch_dtype": "bfloat16",
"transformers_version": "5.6.2",
"use_cache": true,
"vocab_size": 130560
}
Loading
Loading