Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 2 additions & 1 deletion .github/workflows/native-dev.yml
Original file line number Diff line number Diff line change
Expand Up @@ -84,12 +84,13 @@ jobs:
- name: Build
run: cmake --build --preset release --parallel 2
- name: Build non-default fixed-block pilot
run: cmake --build --preset release --target uchardet-fixed-block uchardet-fixed-block-test --parallel 2
run: cmake --build --preset release --target uchardet-fixed-block uchardet-fixed-block-test uchardet-fixed-block-timing --parallel 2
- name: Fixed-block contract comparisons
run: timeout 10s build/release/benchmark/uchardet-fixed-block-test
- name: Tool tests
env:
UCHARDET_FIXED_BLOCK: build/release/benchmark/uchardet-fixed-block
UCHARDET_FIXED_BLOCK_TIMING: build/release/benchmark/uchardet-fixed-block-timing
UCHARDET_TRACE: build/release/benchmark/uchardet-trace
UCHARDET_FILTER_PROFILE: build/release/benchmark/uchardet-filter-profile
run: uv run --no-project --python 3.11 python -m unittest discover -s benchmark -p 'test_*.py'
Expand Down
2 changes: 2 additions & 0 deletions benchmark/CMakeLists.txt
Original file line number Diff line number Diff line change
Expand Up @@ -14,6 +14,8 @@ add_executable(uchardet-fixed-block EXCLUDE_FROM_ALL uchardet-conformance.cpp)
target_compile_definitions(uchardet-fixed-block PRIVATE
UCHARDET_FIXED_BLOCK_PILOT=1 UCHARDET_EXPERIMENTAL_INPUT_LIMIT=4096)
target_link_libraries(uchardet-fixed-block ${UCHARDET_LIBRARY})
add_executable(uchardet-fixed-block-timing EXCLUDE_FROM_ALL fixed-block-timing.cpp)
target_link_libraries(uchardet-fixed-block-timing ${UCHARDET_LIBRARY})

if(TARGET libuchardet_experimental)
add_executable(uchardet-conformance-experimental EXCLUDE_FROM_ALL uchardet-conformance.cpp)
Expand Down
103 changes: 103 additions & 0 deletions benchmark/fixed-block-timing.cpp
Original file line number Diff line number Diff line change
@@ -0,0 +1,103 @@
// SPDX-License-Identifier: MIT
// Small warmed reuse timing; no filesystem I/O or detector creation in timed region.
#include "fixed-block-detector.h"
#include <chrono>
#include <fstream>
#include <iomanip>
#include <iostream>
#include <string>

static size_t number(const char* text, size_t maximum) {
const std::string value(text);
if (value.empty() || value.find_first_not_of("0123456789") != std::string::npos)
throw std::invalid_argument("invalid numeric parameter");
const unsigned long parsed = std::stoul(value);
if (parsed > maximum) throw std::invalid_argument("parameter exceeds pilot bound");
return parsed;
}

int main(int argc, char** argv) {
try {
if (argc != 6) throw std::invalid_argument("usage: fixed-block-timing BLOCK EXTERNAL_CHUNK ITERATIONS REPEATS FILE");
const size_t block = number(argv[1], 4096), chunk = number(argv[2], 4096);
const size_t iterations = number(argv[3], 10000), repeats = number(argv[4], 31);
if (!block || !iterations || !repeats) throw std::invalid_argument("zero parameter");
std::ifstream input(argv[5], std::ios::binary);
if (!input) throw std::runtime_error("cannot open input");
std::vector<char> bytes(4097);
input.read(bytes.data(), static_cast<std::streamsize>(bytes.size()));
bytes.resize(static_cast<size_t>(input.gcount()));
if (input.bad() || bytes.size() > 4096) throw std::runtime_error("input exceeds bounded pilot");
const size_t external = chunk ? chunk : std::max(size_t(1), bytes.size());
experimental::FixedBlockDetector adapter(block, 4096);
typedef std::unique_ptr<struct uchardet, decltype(&uchardet_delete)> Detector;
Detector whole(uchardet_new(), &uchardet_delete), fixed(uchardet_new(), &uchardet_delete);
if (!whole || !fixed) throw std::runtime_error("allocation failed");
const auto run = [&](size_t mode) -> size_t {
uchardet_t handle;
if (mode == 2) {
adapter.reset();
for (size_t offset = 0; offset < bytes.size(); offset += external)
adapter.feed(bytes.data() + offset, std::min(external, bytes.size() - offset));
adapter.finish();
handle = adapter.handle();
} else {
handle = mode == 0 ? whole.get() : fixed.get();
uchardet_reset(handle);
const size_t step = mode == 0 ? std::max(size_t(1), bytes.size()) : block;
for (size_t offset = 0; offset < bytes.size() && !uchardet_is_done(handle); offset += step)
if (uchardet_handle_data(handle, bytes.data() + offset,
std::min(step, bytes.size() - offset)))
throw std::runtime_error("feed failed");
uchardet_data_end(handle);
}
// Match the established benchmark's small result consumption, not Python conversion.
const size_t count = uchardet_get_n_candidates(handle);
return count + (count ? static_cast<unsigned char>(uchardet_get_encoding(handle, 0)[0]) : 0);
};
size_t expected[3];
for (size_t mode = 0; mode < 3; ++mode) {
expected[mode] = run(mode);
for (size_t warm = 0; warm < 16; ++warm)
if (run(mode) != expected[mode]) throw std::runtime_error("unstable warmup result");
}
std::vector<double> timings[3];
std::vector<size_t> checksums[3];
for (size_t trial = 0; trial < repeats; ++trial) {
for (size_t order = 0; order < 3; ++order) {
const size_t mode = (trial + order) % 3;
size_t checksum = 0;
const auto start = std::chrono::steady_clock::now();
for (size_t iteration = 0; iteration < iterations; ++iteration) checksum += run(mode);
const auto end = std::chrono::steady_clock::now();
if (checksum != expected[mode] * iterations) throw std::runtime_error("timed checksum differs");
timings[mode].push_back(std::chrono::duration<double, std::nano>(end - start).count() / iterations);
checksums[mode].push_back(checksum);
}
}
const char* names[] = {"whole", "direct_fixed", "adapter"};
std::cout << std::setprecision(17) << "{\"byte_length\":" << bytes.size()
<< ",\"block\":" << block << ",\"external_chunk\":" << chunk
<< ",\"iterations\":" << iterations << ",\"repeats\":" << repeats
<< ",\"modes\":{";
for (size_t mode = 0; mode < 3; ++mode) {
if (mode) std::cout << ',';
std::cout << '"' << names[mode] << "\":{\"trial_mean_ns\":[";
for (size_t i = 0; i < repeats; ++i) {
if (i) std::cout << ',';
std::cout << timings[mode][i];
}
std::cout << "],\"checksums\":[";
for (size_t i = 0; i < repeats; ++i) {
if (i) std::cout << ',';
std::cout << checksums[mode][i];
}
std::cout << "]}";
}
std::cout << "}}\n";
return 0;
} catch (const std::exception& error) {
std::cerr << error.what() << '\n';
return 1;
}
}
93 changes: 93 additions & 0 deletions benchmark/fixed-block-timing.ja.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,93 @@
<!-- SPDX-License-Identifier: MIT -->
# 固定block adapterのnative処理コスト

## 範囲と比較方式

第3試作のwarm reuse計測。Python処理・process起動・入力I/O・初期detector/buffer確保を
計測外にし、reset・feed・finalize・候補数/先頭encodingの最小限の結果取得を計測する。
標準modelの同じstatic libraryへlinkした一つのbinaryで以下を比較する。

1. `whole`: 入力全体を直接nativeへ渡す。
2. `direct_fixed`: adapterと同じ固定blockを直接nativeへ渡す。
3. `adapter`: 外部chunkをadapterで正規化してnativeへ渡す。

wholeとの比較は候補/confidence等の動作差を含む。direct_fixedとの差は主にadapterの
buffer copy/外部feed処理に対応するが、microbenchmarkの揺らぎを因果証明とはしない。
CIは短い入力のschema・checksum・引数検証だけを実行し、速度閾値をassertしない。

## 固定条件

- corpus: 固定tuningのcp1252/UTF-8各8入力、計16入力、各4096 bytes以下。
- 内部block 1/7/64/1024、外部chunk whole/1/64。結果による候補追加なし。
- CPU2へaffinity固定。各入力100反復×7試行、各方式17 warmup。
- 3方式の順番は試行ごとに循環。各native processは10秒timeout。
- C++ steady_clockで計測。checksumは候補数と先頭encodingの先頭byteに基づく。
checksum一致は候補全体の同等性証明ではなく、完全比較は別のconformance評価で行う。
- GCC 16.2.1、CMake Release、C++11、`-msse2 -mfpmath=sse -O3 -DNDEBUG`。
- 計測中に別のbuild/testを実行しない。hostの他process・SMT相方・周波数は固定しない。
確認時のgovernorはpowersave。affinityをCPU占有と称しない。

## 初回結果(2026-09-22)

数値は文書別trial平均を合計した7試行のmedian、単位ms。
同じ文書のwarm cache測定の合計であり、異なる文書を順次処理するcorpus passではない。

| 内部block / 外部chunk | whole | direct_fixed | adapter |
| --- | ---: | ---: | ---: |
| 64 / whole | 10.4270 | 11.1460 | 11.1323 |
| 64 / 1 | 10.4268 | 11.1544 | 11.3727 |
| 64 / 64 | 10.4269 | 11.1194 | 11.1388 |
| 1024 / whole | 10.4053 | 10.4694 | 10.4829 |
| 1024 / 1 | 10.4064 | 10.4893 | 10.7285 |
| 1024 / 64 | 10.4039 | 10.4877 | 10.4781 |

64のadapterはwhole比で約6.8〜9.1%遅く、開発計画の5%調査閾値を超えた。
1024は約0.7〜3.1%。「1%以内」は外部whole/64に限り、外部1-byteへ一般化しない。
初回結果だけでblock採用・性能gate通過を決めない。追加runで再現性を確認する。
requestごとのp95、初回確保、allocation数、peak/live memoryは未測定。

文書別medianでも確認した。外部whole時、64のadapterは16入力中14件でwhole比5%超、
範囲は+4.76〜9.30%。1024は16入力とも5%以下(+0.28〜1.24%)だった。
sumのmedianと文書別medianは異なる集計であり、両方を保存結果から確認する。

## 同条件の2回目

同じbinary・driver・入力・affinity・反復数で再計測した。終了後のgovernorもpowersave。

| 内部block / 外部chunk | whole | direct_fixed | adapter |
| --- | ---: | ---: | ---: |
| 64 / whole | 10.4233 | 11.1432 | 11.1145 |
| 64 / 1 | 10.4375 | 11.1443 | 11.3852 |
| 64 / 64 | 10.4195 | 11.1211 | 11.1264 |
| 1024 / whole | 10.4047 | 10.4735 | 10.4649 |
| 1024 / 1 | 10.4144 | 10.4823 | 10.7153 |
| 1024 / 64 | 10.4153 | 10.4980 | 10.4942 |

外部wholeの文書別medianは、64で16入力中15件が5%超(+4.84〜8.55%)、
1024で5%超なし(+0.19〜1.15%)。64の5%超悪化は再現したため、
今回のまま標準採用する根拠にはしない。direct_fixedでも増分があるため、
adapterのcopyを削るだけで差を解消できると仮定しない。
1024は追加評価候補に残せるが、これはmemory・request p95・全言語での採用gate通過ではない。
速度で有利な1-byteの品質後退を無視して採用することもしない。

run2 content hash: `228857d6ec5f75488c3a66fd0b8faee90862b44456e9c4b727b3f62b1ba30b6c`。
全12条件の生データは各reportに保存しており、表から省略した1/7-byte条件も削除していない。

## 再現・保存

```sh
cmake --build /tmp/uchardet-fixed-block-build --target uchardet-fixed-block-timing
UV_CACHE_DIR=/tmp/cchardet-v3-uv taskset -c 2 uv run --no-project python benchmark/fixed_block_timing.py \
/workspace/archives/v3-corpus/paris-training-tuning-generated-v1/manifest.json \
/workspace/archives/v3-corpus/fixed-block-tuning-v3.json \
/tmp/uchardet-fixed-block-build/benchmark/uchardet-fixed-block-timing \
/workspace/archives/v3-corpus/fixed-block-timing-run1.json
```

`/workspace`は実際の保存先へ置き換える。時間の全byte一致は要求せず、全試行を保存する。
reportは入力・binary・driver hash、uname、affinity、試行平均とchecksumを含む。
run1 content hash: `bd9e9141081efa6d36f7b34e7494ae74115be9359b650826c12136f8e4526cc8`。
binary hash: `ee9b4632384c653e5ade9542d2e884508421db60855e4c51669d8cf625ac32a6`。
driver hash: `beec60ff4c8852e496856351b7c455adafae5ebbb0c97c0a4a113189df92933d`。

非デフォルト実験のみ。P01の大入力・追加安全性検証や独立holdout開封は行っていない。
39 changes: 39 additions & 0 deletions benchmark/fixed-block.ja.md
Original file line number Diff line number Diff line change
Expand Up @@ -89,3 +89,42 @@ baseline binary: `2944d943c50af9a76222617e9591bb6294526b5e18d6390b978501187d6750
範囲外pathの拒否、CLI引数・入力上限、fresh/resetとevidence上限を検証する。
ローカルのbenchmark suiteは36件中31成功・5 skip(各追加toolの環境指定条件)。
初回PR CIは11件成功。追加testを含む最終headのCIは別途確認する。

## 既存validationでの固定block比較(2026-09-22)

`--split validation`を明示すると、保存済みfull-engine reportのhash
`7b4695e8ff78effdeca177f71cf761081b46c64427ae8ad5c1d58de02e7e7265`と
対応manifestへ固定する。tuningとvalidationは同じ入力として扱わない。
独立holdoutの指定は受け付けない。結果を見たblock追加やモデル変更は行っていない。

Paris validation 16録音のcp1252/UTF-8各16入力(計32)のすべてを評価した。
32×4内部block×5外部chunkの640観測で、同じblock長の候補/done/core位置が一致した。
再実行のreport全byte一致。

| 内部block | cp1252 exact / 16 | cp1252 decode-equivalent / 16 | UTF-8 exact / 16 | wholeと候補全体が異なる入力 / 32 |
| --- | --- | --- | --- | --- |
| 1 | 0 | 0 | 16 | 32 |
| 7 | 3 | 4 | 16 | 32 |
| 64 | 11 | 16 | 16 | 32 |
| 1024 | 11 | 16 | 16 | 8 |

legacy whole-inputはcp1252 exact 11/16、decode-equivalent 16/16。
64/1024がこのvalidationのtop-1件数を維持したことと、旧候補/confidence完全互換は別。
既に分析に使ったvalidationであり、未参照の独立評価と称しない。
全言語・実Webへの一般化、性能/memory、block長採用は依然として未確定。

```sh
uv run --no-project python benchmark/fixed_block_compare.py \
/workspace/archives/v3-corpus/paris-stories-generated-1/manifest.json \
/workspace/archives/v3-corpus/paris-full-engine-comparison-v1.json \
/tmp/uchardet-fixed-block-build/benchmark/uchardet-fixed-block \
/tmp/uchardet-fixed-block-build/benchmark/uchardet-conformance \
/workspace/archives/v3-corpus/fixed-block-validation-v1.json --split validation
```

report content hash: `07945ca56191edfab5afad23c492e708849d2e64b941b5c1338d5e2c89f63c19`。
3件の追加testで、validationへのtuning/独立sample混入、保存reportの改変、
独立split指定を拒否する。新旧合わせて10件成功。

後続の[native処理コスト測定](fixed-block-timing.ja.md)では、64-byteの性能悪化が
2 runで再現した。品質件数だけで64/1024のどちらも採用可能とは判断しない。
52 changes: 37 additions & 15 deletions benchmark/fixed_block_compare.py
Original file line number Diff line number Diff line change
@@ -1,5 +1,5 @@
# SPDX-License-Identifier: MIT
"""Frozen small tuning pilot; not independent evaluation or block selection."""
"""Frozen small tuning/validation pilot; not independent evaluation or selection."""

import argparse
import hashlib
Expand All @@ -16,6 +16,7 @@
from sequence_contract import content_hash

FROZEN = "5d28f112f1ed472e1a438df9790f9e2f50c9aa0e216943059e2bb51621d0c8c1"
VALIDATION = "7b4695e8ff78effdeca177f71cf761081b46c64427ae8ad5c1d58de02e7e7265"
BLOCKS = (1, 7, 64, 1024)
CHUNKS = (0, 1, 7, 64, 1024)

Expand All @@ -24,27 +25,44 @@ def sha(path):
return hashlib.sha256(path.read_bytes()).hexdigest()


def run(manifest_path, previous_path, binary, baseline):
def run(manifest_path, previous_path, binary, baseline, split="tuning"):
if split not in ("tuning", "validation"):
raise ValueError("independent/unknown split is not permitted")
manifest = json.loads(manifest_path.read_text())
previous = json.loads(previous_path.read_text())
if previous.get("content_hash") != FROZEN or content_hash(previous) != FROZEN:
raise ValueError("requires frozen tuning chunk report")
if manifest_hash(manifest) != previous["manifest_hash"]:
raise ValueError("manifest differs from frozen tuning run")
frozen = FROZEN if split == "tuning" else VALIDATION
expected_count = 16 if split == "tuning" else 32
if previous.get("content_hash") != frozen or content_hash(previous) != frozen:
raise ValueError("requires frozen report for selected split")
corpus_hash = previous["manifest_hash" if split == "tuning" else "corpus_content_hash"]
if manifest_hash(manifest) != corpus_hash:
raise ValueError("manifest differs from frozen run")
records = previous["documents"]
if split == "validation":
if previous["split"] != "validation":
raise ValueError("prior report is not validation")
records = [
dict(
row,
encoding=row["sample_encoding"],
observations={"0": {"legacy": row["observations"]["legacy"]}},
)
for row in records
]
samples = {s["id"]: s for s in manifest["samples"]}
if len(samples) != len(manifest["samples"]):
raise ValueError("duplicate sample IDs")
hashes = {"adapter": sha(binary), "baseline": sha(baseline)}
documents = []
for old in previous["documents"]:
for old in records:
sample = samples[old["sample_id"]]
if (
sample["split"] != "tuning"
sample["split"] != split
or sample["boundary"] != "complete"
or sample["encoding"] != old["encoding"]
or sample["sha256"] != old["sample_sha256"]
):
raise ValueError("only complete tuning samples are allowed")
raise ValueError("only complete samples from selected split are allowed")
root = manifest_path.parent.resolve()
path = (root / sample["path"]).resolve()
if not path.is_relative_to(root) or path.stat().st_size > 4096:
Expand Down Expand Up @@ -101,8 +119,8 @@ def observe(executable, args):
baseline_score=score(normal, data, sample["encoding"], "fr"),
)
)
if len(documents) != 16:
raise ValueError("expected sixteen frozen tuning inputs")
if len(documents) != expected_count:
raise ValueError("unexpected number of frozen inputs")
if hashes != {"adapter": sha(binary), "baseline": sha(baseline)}:
raise ValueError("binary changed during evaluation")
summary = {}
Expand All @@ -121,9 +139,10 @@ def observe(executable, args):
),
)
result = dict(
schema="fixed-block-tuning-pilot-v1",
previous_hash=FROZEN,
manifest_hash=previous["manifest_hash"],
schema="fixed-block-" + split + "-pilot-v1",
split=split,
previous_hash=frozen,
manifest_hash=corpus_hash,
binary_hashes=hashes,
driver_hash=sha(Path(__file__)),
blocks=list(BLOCKS),
Expand All @@ -143,7 +162,10 @@ def observe(executable, args):
parser = argparse.ArgumentParser(description=__doc__)
for name in ("manifest", "previous", "binary", "baseline", "output"):
parser.add_argument(name, type=Path)
parser.add_argument("--split", choices=("tuning", "validation"), default="tuning")
args = parser.parse_args()
result = run(args.manifest, args.previous, args.binary.resolve(), args.baseline.resolve())
result = run(
args.manifest, args.previous, args.binary.resolve(), args.baseline.resolve(), args.split
)
write_idempotent(args.output, canonical(result))
print(json.dumps(result["summary"], indent=2))
Loading
Loading