diff --git a/.github/workflows/native-dev.yml b/.github/workflows/native-dev.yml index e5ef7bd..73b0d28 100644 --- a/.github/workflows/native-dev.yml +++ b/.github/workflows/native-dev.yml @@ -83,8 +83,13 @@ jobs: run: cmake --preset release -DBUILD_INTROSPECTION=ON - name: Build run: cmake --build --preset release --parallel 2 + - name: Build non-default fixed-block pilot + run: cmake --build --preset release --target uchardet-fixed-block uchardet-fixed-block-test --parallel 2 + - name: Fixed-block contract comparisons + run: timeout 10s build/release/benchmark/uchardet-fixed-block-test - name: Tool tests env: + UCHARDET_FIXED_BLOCK: build/release/benchmark/uchardet-fixed-block UCHARDET_TRACE: build/release/benchmark/uchardet-trace UCHARDET_FILTER_PROFILE: build/release/benchmark/uchardet-filter-profile run: uv run --no-project --python 3.11 python -m unittest discover -s benchmark -p 'test_*.py' diff --git a/benchmark/CMakeLists.txt b/benchmark/CMakeLists.txt index a955394..7cc64a4 100644 --- a/benchmark/CMakeLists.txt +++ b/benchmark/CMakeLists.txt @@ -7,6 +7,14 @@ target_link_libraries(uchardet-output ${UCHARDET_LIBRARY}) add_executable(uchardet-conformance uchardet-conformance.cpp) target_link_libraries(uchardet-conformance ${UCHARDET_LIBRARY}) +# Explicit-only pilot; never installed or part of the default build. +add_executable(uchardet-fixed-block-test EXCLUDE_FROM_ALL test-fixed-block.cpp) +target_link_libraries(uchardet-fixed-block-test ${UCHARDET_LIBRARY}) +add_executable(uchardet-fixed-block EXCLUDE_FROM_ALL uchardet-conformance.cpp) +target_compile_definitions(uchardet-fixed-block PRIVATE + UCHARDET_FIXED_BLOCK_PILOT=1 UCHARDET_EXPERIMENTAL_INPUT_LIMIT=4096) +target_link_libraries(uchardet-fixed-block ${UCHARDET_LIBRARY}) + if(TARGET libuchardet_experimental) add_executable(uchardet-conformance-experimental EXCLUDE_FROM_ALL uchardet-conformance.cpp) target_compile_definitions(uchardet-conformance-experimental PRIVATE UCHARDET_EXPERIMENTAL_INPUT_LIMIT=4096) diff --git a/benchmark/fixed-block-detector.h b/benchmark/fixed-block-detector.h new file mode 100644 index 0000000..2847f21 --- /dev/null +++ b/benchmark/fixed-block-detector.h @@ -0,0 +1,84 @@ +// SPDX-License-Identifier: MIT +// Non-default experiment; not an installed or supported public API. +#ifndef UCHARDET_EXPERIMENTAL_FIXED_BLOCK_DETECTOR_H +#define UCHARDET_EXPERIMENTAL_FIXED_BLOCK_DETECTOR_H + +#include "uchardet.h" +#include +#include +#include +#include +#include + +namespace experimental { +class FixedBlockDetector { + public: + // Both bounds are explicit experiment parameters, never library defaults. + FixedBlockDetector(size_t block_size, size_t evidence_limit) + : detector_(uchardet_new(), &uchardet_delete), block_(checked(block_size)), + limit_(evidence_limit) { + if (!detector_) throw std::runtime_error("detector allocation failed"); + if (limit_ > 4096) throw std::invalid_argument("pilot evidence exceeds 4096 bytes"); + } + + void feed(const char* data, size_t length) { + if (closed_) throw std::logic_error("feed after finish requires reset"); + // Empty feed is not EOF and does not flush pending evidence. + if (!length || done()) return; + length = std::min(length, limit_ - accepted_); + while (length && !done()) { + const size_t take = std::min(length, block_.size() - pending_); + std::memcpy(block_.data() + pending_, data, take); + data += take; + pending_ += take; + accepted_ += take; + length -= take; + if (pending_ == block_.size()) flush(); + } + } + + void finish() { + if (closed_) return; + if (pending_ && !done()) flush(); + uchardet_data_end(detector_.get()); + closed_ = true; + } + + void reset() { + uchardet_reset(detector_.get()); + pending_ = accepted_ = processed_ = calls_ = first_done_ = 0; + observed_done_ = closed_ = false; + } + + uchardet_t handle() const { return detector_.get(); } + bool done() const { return uchardet_is_done(detector_.get()) != 0; } + size_t processed() const { return processed_; } + size_t calls() const { return calls_; } + size_t pending() const { return pending_; } + bool observed_done() const { return observed_done_; } + size_t first_done() const { return first_done_; } + + private: + static size_t checked(size_t size) { + if (!size || size > 4096) throw std::invalid_argument("block must be 1..4096 bytes"); + return size; + } + void flush() { + if (uchardet_handle_data(detector_.get(), block_.data(), pending_)) + throw std::runtime_error("handle_data failed"); + processed_ += pending_; + pending_ = 0; + ++calls_; + if (!observed_done_ && done()) { + observed_done_ = true; + first_done_ = processed_; + } + } + std::unique_ptr detector_; + std::vector block_; + size_t limit_; + size_t pending_ = 0, accepted_ = 0, processed_ = 0, calls_ = 0, first_done_ = 0; + bool observed_done_ = false, closed_ = false; +}; +} // namespace experimental +#endif diff --git a/benchmark/fixed-block.ja.md b/benchmark/fixed-block.ja.md new file mode 100644 index 0000000..7afb4b9 --- /dev/null +++ b/benchmark/fixed-block.ja.md @@ -0,0 +1,91 @@ + +# 固定 block 入力 adapter の非デフォルト試作 + +2026-09-22、maintainerの再開指示を受け、第3試作として着手。 +既存のBOM/model generator試作は保存する。P01は再開しない。 +この実験用classを公開C API、Python API、既定engineへ採用する変更ではない。 + +## 実験契約 + +- block長とevidence上限は明示指定。初期pilotはいずれも最大4096 bytes。 +- 同じ入力prefixを、外部feedの境界に関係なく固定blockでcoreへ渡す。 +- evidence上限0は空prefix。上限に達してもEOF扱いにしない。 +- 空feedではflushしない。finishで端数を一度だけflushし、native finalizeを一度呼ぶ。 +- nativeがdoneなら追加入力を処理しない。doneの初回観測はcoreへ渡したbyte位置で記録する。 + finishだけでdoneになる場合とfeed中にdoneになる場合を混同しない。 +- resetでpending buffer、counter、終了状態とnative文書状態を初期化する。 +- finishはadapterとしてべき等。finish後feedはlogic_error。native API自体の保証は変更しない。 +- adapterの保持bufferはblock長分のみ。入力全体をadapterに保存しない。 + これはnative内部のmemory上限やallocation失敗時の保証を意味しない。 + +## 初期検証 + +`test-fixed-block.cpp`は通常の空/ASCII/UTF-8/cp1252/BOM入力とblock前後長を使用する。 +外部chunkは1/7/64/1024/4096 bytes、内部blockは1/7/64/1024 bytes、 +evidence上限は0/1/7/64/4096 bytes。 +別のfresh C API detectorへ同じprefixを固定blockで渡した結果をoracleとする。 +候補数・順序・encoding・languageのnull区別・confidence bits・done、 +処理byte数・core feed回数・最初のdone位置を比較する。 +途中reset、空feed、繰返しfinish、finish後feed拒否も確認する。 + +```sh +cmake -S . -B /tmp/uchardet-fixed-block-build -DBUILD_BENCHMARK=ON -DBUILD_SHARED_LIBS=OFF +cmake --build /tmp/uchardet-fixed-block-build --target uchardet-fixed-block-test +timeout 10s /tmp/uchardet-fixed-block-build/benchmark/uchardet-fixed-block-test +``` + +targetはEXCLUDE_FROM_ALLでinstall対象外。既存libraryを変更しない。 +旧whole-inputの候補と一致することは要求しない。corpusの品質・性能・memory評価、 +block長の選定、Python/wheel統合、標準採用は未実施。 +独立holdoutやP01対象の大入力/追加fuzzは扱わない。 + +## 固定tuning pilot(2026-09-22) + +`fixed_block_compare.py`は保存済みratio chunk reportのcontent hashを固定し、 +同じmanifestのcomplete tuning 16入力だけを読み、各入力のhashを照合する。 +legacy whole-inputが保存観測と完全一致することを先に確認する。 +各processに10秒のtimeoutを設け、入力は4096 bytes以下に限定する。 +block候補は事前に1/7/64/1024、外部chunkはwhole/1/7/64/1024と固定した。 +生成modelは使わず、既存の標準modelを同じlibraryから呼び出す。 + +初回結果: 16入力×4 block×5 chunkの320観測で、外部chunk間の候補・最終done・ +core feed回数・処理byte数・coreの最初のdone位置が一致した。 +callerのfeed回数やdoneを返す外部境界は一致要件に含めない。 + +| 内部block | cp1252 exact / 8 | cp1252 decode-equivalent / 8 | UTF-8 exact / 8 | +| --- | --- | --- | --- | +| 1 | 0 | 0 | 8 | +| 7 | 2 | 6 | 8 | +| 64 | 4 | 8 | 8 | +| 1024 | 4 | 8 | 8 | + +legacy whole-inputはcp1252 exact 4/8、decode-equivalent 8/8。 +whole-inputと候補が完全一致しない入力はblock 1/7/64で各16/16、1024で4/16。 +64/1024のtop-1品質が同じでもconfidenceや下位候補の同等性を意味しない。 +このtuning結果からblock長を標準採用しない。独立holdoutは未開封。 + +```sh +uv run --no-project python benchmark/fixed_block_compare.py \ + /workspace/archives/v3-corpus/paris-training-tuning-generated-v1/manifest.json \ + /workspace/archives/v3-corpus/paris24-ratio-chunks-v1.json \ + /tmp/uchardet-fixed-block-build/benchmark/uchardet-fixed-block \ + /tmp/uchardet-fixed-block-build/benchmark/uchardet-conformance \ + /workspace/archives/v3-corpus/fixed-block-tuning-v3.json +``` + +`/workspace`は保存済みprivate artifactの実際の配置へ置き換える。 +reportはbinary hash・driver hash・入力hash・全観測を含み、同じ出力への上書きは +内容が同一の場合のみ許可する。性能・memory測定や一般化した精度保証ではない。 + +最終driverで再実行したreportのcontent hash: +`d198e153f5167a561d5809f011717def074fb3ab3991d84fa0f438bad816c63d`。 +driver hash: `ff20a322644fabf6778dcea38ab44145ff06bb00ed765e339c1aee92e36efc67`。 +adapter binary: `b3f82885dcc9dab19b08dd30e26add6daf3a9b83599c62cea5eb8faf03a473fb`、 +baseline binary: `2944d943c50af9a76222617e9591bb6294526b5e18d6390b978501187d67501e`。 +再実行のreport全byte一致。v1/v2はdriverのmetadata検証・記録拡充前の保存物であり、 +結果の選別ではない。各blockの観測結果は同じだった。 + +追加の7 unittestは、固定report改変、用途/境界/encoding/hash metadata、重複ID、 +範囲外pathの拒否、CLI引数・入力上限、fresh/resetとevidence上限を検証する。 +ローカルのbenchmark suiteは36件中31成功・5 skip(各追加toolの環境指定条件)。 +初回PR CIは11件成功。追加testを含む最終headのCIは別途確認する。 diff --git a/benchmark/fixed_block_compare.py b/benchmark/fixed_block_compare.py new file mode 100644 index 0000000..df94310 --- /dev/null +++ b/benchmark/fixed_block_compare.py @@ -0,0 +1,149 @@ +# SPDX-License-Identifier: MIT +"""Frozen small tuning pilot; not independent evaluation or block selection.""" + +import argparse +import hashlib +import json +import subprocess +import sys +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "models/experimental")) +import engine_comparison +from engine_comparison import score +from framework import content_hash as manifest_hash +from model import canonical, write_idempotent +from sequence_contract import content_hash + +FROZEN = "5d28f112f1ed472e1a438df9790f9e2f50c9aa0e216943059e2bb51621d0c8c1" +BLOCKS = (1, 7, 64, 1024) +CHUNKS = (0, 1, 7, 64, 1024) + + +def sha(path): + return hashlib.sha256(path.read_bytes()).hexdigest() + + +def run(manifest_path, previous_path, binary, baseline): + manifest = json.loads(manifest_path.read_text()) + previous = json.loads(previous_path.read_text()) + if previous.get("content_hash") != FROZEN or content_hash(previous) != FROZEN: + raise ValueError("requires frozen tuning chunk report") + if manifest_hash(manifest) != previous["manifest_hash"]: + raise ValueError("manifest differs from frozen tuning run") + samples = {s["id"]: s for s in manifest["samples"]} + if len(samples) != len(manifest["samples"]): + raise ValueError("duplicate sample IDs") + hashes = {"adapter": sha(binary), "baseline": sha(baseline)} + documents = [] + for old in previous["documents"]: + sample = samples[old["sample_id"]] + if ( + sample["split"] != "tuning" + or sample["boundary"] != "complete" + or sample["encoding"] != old["encoding"] + or sample["sha256"] != old["sample_sha256"] + ): + raise ValueError("only complete tuning samples are allowed") + root = manifest_path.parent.resolve() + path = (root / sample["path"]).resolve() + if not path.is_relative_to(root) or path.stat().st_size > 4096: + raise ValueError("input outside bounded pilot") + data = path.read_bytes() + if sha(path) != old["sample_sha256"] or len(data) != old["byte_length"]: + raise ValueError("input changed") + + def observe(executable, args): + return json.loads( + subprocess.run( + [str(executable), *map(str, args), str(path)], + capture_output=True, + check=True, + text=True, + timeout=10, + ).stdout + ) + + normal = observe(baseline, ["fresh", 0]) + if normal != old["observations"]["0"]["legacy"]: + raise ValueError("baseline differs from frozen observation") + blocks = {} + for block in BLOCKS: + observations = { + str(chunk): observe(binary, [block, 4096, "fresh", chunk]) for chunk in CHUNKS + } + stable = ( + "candidates", + "candidate_count", + "initial_done", + "final_done", + "core_processed_bytes", + "core_feed_calls", + "core_first_done_offset", + ) + reference = observations["0"] + for observation in observations.values(): + if any(observation[key] != reference[key] for key in stable): + raise ValueError("external chunk changed canonical result") + blocks[str(block)] = dict( + observations=observations, + score=score(reference, data, sample["encoding"], "fr"), + vs_legacy_whole_candidates=reference["candidates"] != normal["candidates"], + vs_legacy_whole_done=reference["final_done"] != normal["final_done"], + ) + documents.append( + dict( + sample_id=sample["id"], + sha256=sha(path), + encoding=sample["encoding"], + byte_length=len(data), + blocks=blocks, + baseline_score=score(normal, data, sample["encoding"], "fr"), + ) + ) + if len(documents) != 16: + raise ValueError("expected sixteen frozen tuning inputs") + if hashes != {"adapter": sha(binary), "baseline": sha(baseline)}: + raise ValueError("binary changed during evaluation") + summary = {} + for block in map(str, BLOCKS): + summary[block] = {} + for encoding in ("cp1252", "utf-8"): + rows = [d for d in documents if d["encoding"] == encoding] + summary[block][encoding] = dict( + samples=len(rows), + exact=sum(d["blocks"][block]["score"]["top1_exact_codec"] for d in rows), + decode_equal=sum( + d["blocks"][block]["score"]["top1_decode_status"] == "equal" for d in rows + ), + changed_candidates=sum( + d["blocks"][block]["vs_legacy_whole_candidates"] for d in rows + ), + ) + result = dict( + schema="fixed-block-tuning-pilot-v1", + previous_hash=FROZEN, + manifest_hash=previous["manifest_hash"], + binary_hashes=hashes, + driver_hash=sha(Path(__file__)), + blocks=list(BLOCKS), + chunks=list(CHUNKS), + score_driver_hash=sha(Path(engine_comparison.__file__)), + evidence_limit=4096, + documents=documents, + summary=summary, + compatible_superset_metric="NOT_EVALUATED", + independent_holdout="NOT_OPENED", + ) + result["content_hash"] = content_hash(result) + return result + + +if __name__ == "__main__": + parser = argparse.ArgumentParser(description=__doc__) + for name in ("manifest", "previous", "binary", "baseline", "output"): + parser.add_argument(name, type=Path) + args = parser.parse_args() + result = run(args.manifest, args.previous, args.binary.resolve(), args.baseline.resolve()) + write_idempotent(args.output, canonical(result)) + print(json.dumps(result["summary"], indent=2)) diff --git a/benchmark/test-fixed-block.cpp b/benchmark/test-fixed-block.cpp new file mode 100644 index 0000000..f0c8aa6 --- /dev/null +++ b/benchmark/test-fixed-block.cpp @@ -0,0 +1,95 @@ +// SPDX-License-Identifier: MIT +#include "fixed-block-detector.h" +#include +#include +#include +#include + +static void require(bool ok, const char* message) { + if (!ok) throw std::runtime_error(message); +} + +static std::string snapshot(uchardet_t detector) { + std::ostringstream out; + out << uchardet_is_done(detector) << ':' << uchardet_get_n_candidates(detector); + for (size_t i = 0; i < uchardet_get_n_candidates(detector); ++i) { + const char* language = uchardet_get_language(detector, i); + const float confidence = uchardet_get_confidence(detector, i); + uint32_t bits; + static_assert(sizeof(bits) == sizeof(confidence), "binary32 required"); + std::memcpy(&bits, &confidence, sizeof(bits)); + out << '|' << uchardet_get_encoding(detector, i) << ':' + << (language ? "present:" : "null:") << (language ? language : "") << ':' << bits; + } + return out.str(); +} + +int main() { + try { + const std::vector inputs = { + "", "a", "plain ASCII text", "caf\xc3\xa9 et th\xc3\xa9", + "caf\xe9 et th\xe9", std::string(135, 'a'), + std::string(63, 'a'), std::string(64, 'a'), std::string(65, 'a'), + "\xef\xbb\xbf" "UTF-8 text" + }; + size_t comparisons = 0; + for (size_t block : {size_t(1), size_t(7), size_t(64), size_t(1024)}) { + for (size_t limit : {size_t(0), size_t(1), size_t(7), size_t(64), size_t(4096)}) { + experimental::FixedBlockDetector reused(block, limit); + for (const auto& input : inputs) { + // Direct C API oracle, fed the canonical prefix in fixed blocks. + std::unique_ptr + direct(uchardet_new(), &uchardet_delete); + require(bool(direct), "allocation failed"); + const size_t length = std::min(input.size(), limit); + size_t processed = 0, calls = 0, first_done = 0; + bool observed = false; + while (processed < length && !uchardet_is_done(direct.get())) { + const size_t step = std::min(block, length - processed); + require(!uchardet_handle_data(direct.get(), input.data() + processed, step), "feed failed"); + processed += step; + ++calls; + if (uchardet_is_done(direct.get())) { observed = true; first_done = processed; } + } + uchardet_data_end(direct.get()); + const std::string expected = snapshot(direct.get()); + for (size_t chunk : {size_t(1), size_t(7), size_t(64), size_t(1024), size_t(4096)}) { + reused.reset(); + reused.feed("discarded", 9); + reused.reset(); + for (size_t offset = 0; offset < input.size(); offset += chunk) { + reused.feed(input.data() + offset, std::min(chunk, input.size() - offset)); + const size_t pending = reused.pending(); + reused.feed("", 0); + require(reused.pending() == pending, "empty feed flushed evidence"); + } + reused.finish(); + require(snapshot(reused.handle()) == expected, "candidate/done mismatch"); + require(reused.processed() == processed && reused.calls() == calls, + "canonical feed counters differ"); + require(reused.observed_done() == observed && reused.first_done() == first_done, + "core completion offset differs"); + reused.finish(); + require(snapshot(reused.handle()) == expected, "finish is not idempotent"); + bool rejected = false; + try { reused.feed("x", 1); } + catch (const std::logic_error&) { rejected = true; } + require(rejected, "feed after finish was not rejected"); + ++comparisons; + } + } + } + } + for (size_t block : {size_t(0), size_t(4097)}) { + bool rejected = false; + try { experimental::FixedBlockDetector invalid(block, 4096); } + catch (const std::invalid_argument&) { rejected = true; } + require(rejected, "invalid block accepted"); + } + std::cout << "fixed-block comparisons: " << comparisons << '\n'; + return 0; + } catch (const std::exception& error) { + std::cerr << error.what() << '\n'; + return 1; + } +} diff --git a/benchmark/test_fixed_block.py b/benchmark/test_fixed_block.py new file mode 100644 index 0000000..e221544 --- /dev/null +++ b/benchmark/test_fixed_block.py @@ -0,0 +1,134 @@ +# SPDX-License-Identifier: MIT +"""Small valid-input pilot contracts and metadata rejection tests.""" + +import copy +import json +import os +import subprocess +import tempfile +import unittest +from pathlib import Path +from unittest.mock import patch + +import fixed_block_compare as comparison + + +class FrozenInputTests(unittest.TestCase): + def setUp(self): + self.temporary = tempfile.TemporaryDirectory() + self.addCleanup(self.temporary.cleanup) + self.root = Path(self.temporary.name) + self.sample = self.root / "sample.bin" + self.sample.write_bytes(b"plain ASCII") + self.manifest = dict( + samples=[ + dict( + id="example", + path="sample.bin", + split="tuning", + boundary="complete", + encoding="cp1252", + sha256=comparison.sha(self.sample), + ) + ] + ) + + def run_rejected(self, manifest=None, mutate_previous=False): + manifest = copy.deepcopy(manifest or self.manifest) + previous = dict( + manifest_hash=comparison.manifest_hash(manifest), + documents=[ + dict( + sample_id="example", + sample_sha256=comparison.sha(self.sample), + byte_length=11, + encoding="cp1252", + ) + ], + ) + frozen = comparison.content_hash(previous) + previous["content_hash"] = frozen + if mutate_previous: + previous["documents"][0]["byte_length"] += 1 + manifest_path = self.root / "manifest.json" + previous_path = self.root / "previous.json" + manifest_path.write_text(json.dumps(manifest), encoding="utf-8") + previous_path.write_text(json.dumps(previous), encoding="utf-8") + # Synthetic frozen fixture only; the production constant is never changed. + with ( + patch.object(comparison, "FROZEN", frozen), + patch.object(comparison.subprocess, "run") as run, + ): + with self.assertRaises(ValueError): + comparison.run(manifest_path, previous_path, self.sample, self.sample) + run.assert_not_called() + + def test_changed_frozen_report_rejected_before_execution(self): + self.run_rejected(mutate_previous=True) + + def test_non_tuning_and_incomplete_metadata_rejected(self): + for field, value in ( + ("split", "independent"), + ("boundary", "byte-truncated"), + ("encoding", "utf-8"), + ("sha256", "0" * 64), + ): + manifest = copy.deepcopy(self.manifest) + manifest["samples"][0][field] = value + with self.subTest(field=field): + self.run_rejected(manifest) + + def test_duplicate_id_rejected(self): + manifest = copy.deepcopy(self.manifest) + manifest["samples"] *= 2 + self.run_rejected(manifest) + + def test_path_outside_corpus_rejected(self): + manifest = copy.deepcopy(self.manifest) + manifest["samples"][0]["path"] = "../not-a-corpus-sample" + self.run_rejected(manifest) + + +@unittest.skipUnless(os.environ.get("UCHARDET_FIXED_BLOCK"), "set UCHARDET_FIXED_BLOCK") +class FixedBlockCliTests(unittest.TestCase): + def invoke(self, args, inputs): + with tempfile.TemporaryDirectory() as directory: + paths = [] + for index, data in enumerate(inputs): + path = Path(directory) / str(index) + path.write_bytes(data) + paths.append(str(path)) + return subprocess.run( + [os.environ["UCHARDET_FIXED_BLOCK"], *map(str, args), *paths], + capture_output=True, + text=True, + timeout=10, + ) + + def test_invalid_parameters(self): + for block, limit in ((0, 4096), (4097, 4096), (64, 4097), (-1, 1), ("7x", 1)): + with self.subTest(block=block, limit=limit): + result = self.invoke([block, limit, "fresh", 0], [b"ASCII"]) + self.assertNotEqual(result.returncode, 0) + self.assertEqual(result.stdout, "") + + def test_limit_and_fresh_reuse(self): + inputs = [b"", b"plain ASCII", "café et thé".encode("cp1252"), b"a" * 65] + for limit in (0, 1, 63, 64, 65, 4096): + fresh = self.invoke([64, limit, "fresh", 1], inputs) + reused = self.invoke([64, limit, "reuse", 1], inputs) + self.assertEqual(fresh.returncode, 0, fresh.stderr) + self.assertEqual(reused.returncode, 0, reused.stderr) + self.assertEqual(fresh.stdout, reused.stdout) + for line, data in zip(fresh.stdout.splitlines(), inputs): + record = json.loads(line) + self.assertEqual(record["core_processed_bytes"], min(len(data), limit)) + + def test_input_cap(self): + result = self.invoke([64, 4096, "fresh", 0], [b"a" * 4097]) + self.assertNotEqual(result.returncode, 0) + self.assertIn("exceeds 4096", result.stderr) + + +if __name__ == "__main__": + unittest.main() diff --git a/benchmark/uchardet-conformance.cpp b/benchmark/uchardet-conformance.cpp index 1edeb1a..e88f0d5 100644 --- a/benchmark/uchardet-conformance.cpp +++ b/benchmark/uchardet-conformance.cpp @@ -13,6 +13,9 @@ #include #include #include +#ifdef UCHARDET_FIXED_BLOCK_PILOT +#include "fixed-block-detector.h" +#endif static void quoted(const char* value) { if (!value) { std::cout << "null"; return; } @@ -29,6 +32,20 @@ static void quoted(const char* value) { int main(int argc, char** argv) { try { +#ifdef UCHARDET_FIXED_BLOCK_PILOT + if (argc < 6) throw std::runtime_error("usage: uchardet-fixed-block BLOCK LIMIT fresh|reuse CHUNK FILE..."); + const auto parameter = [](const char* text) -> size_t { + const std::string value(text); + if (value.empty() || value.find_first_not_of("0123456789") != std::string::npos) + throw std::runtime_error("invalid block/limit parameter"); + const unsigned long number = std::stoul(value); + if (number > 4096) throw std::runtime_error("pilot parameter exceeds 4096"); + return static_cast(number); + }; + const size_t block_size = parameter(argv[1]), evidence_limit = parameter(argv[2]); + argc -= 2; + argv += 2; +#endif if (argc < 4) throw std::runtime_error("usage: uchardet-conformance fresh|reuse 0|1|7|64|1024|random FILE..."); const std::string mode(argv[1]), chunk(argv[2]); if (mode != "fresh" && mode != "reuse") throw std::runtime_error("invalid lifecycle mode"); @@ -36,8 +53,12 @@ int main(int argc, char** argv) { throw std::runtime_error("invalid chunk schedule"); static_assert(sizeof(float) == sizeof(uint32_t) && std::numeric_limits::is_iec559, "exact output requires IEEE-754 binary32"); +#ifdef UCHARDET_FIXED_BLOCK_PILOT + std::unique_ptr adapter; +#else typedef std::unique_ptr Detector; Detector detector(nullptr, &uchardet_delete); +#endif for (int file = 3; file < argc; ++file) { std::ifstream stream(argv[file], std::ios::binary); if (!stream) throw std::runtime_error("cannot open input"); @@ -51,10 +72,18 @@ int main(int argc, char** argv) { const std::vector bytes((std::istreambuf_iterator(stream)), std::istreambuf_iterator()); #endif if (stream.bad()) throw std::runtime_error("cannot read input"); +#ifdef UCHARDET_FIXED_BLOCK_PILOT + if (!adapter || mode == "fresh") + adapter.reset(new experimental::FixedBlockDetector(block_size, evidence_limit)); + else adapter->reset(); + const uchardet_t handle = adapter->handle(); +#else if (!detector || mode == "fresh") detector.reset(uchardet_new()); else uchardet_reset(detector.get()); if (!detector) throw std::runtime_error("detector allocation failed"); - const bool initial_done = uchardet_is_done(detector.get()) != 0; + const uchardet_t handle = detector.get(); +#endif + const bool initial_done = uchardet_is_done(handle) != 0; size_t offset = 0, calls = 0, first_done = 0; bool observed_done = false; uint32_t seed = 0x12345678; @@ -63,24 +92,39 @@ int main(int argc, char** argv) { const size_t step = chunk == "random" ? 1 + seed % 1024 : chunk == "0" ? bytes.size() : static_cast(std::stoul(chunk)); const size_t length = std::min(step, bytes.size() - offset); - if (uchardet_handle_data(detector.get(), bytes.empty() ? "" : &bytes[offset], length)) +#ifdef UCHARDET_FIXED_BLOCK_PILOT + adapter->feed(bytes.empty() ? "" : &bytes[offset], length); +#else + if (uchardet_handle_data(handle, bytes.empty() ? "" : &bytes[offset], length)) throw std::runtime_error("handle_data failed"); +#endif offset += length; ++calls; - if (!observed_done && uchardet_is_done(detector.get())) { observed_done = true; first_done = offset; } + if (!observed_done && uchardet_is_done(handle)) { observed_done = true; first_done = offset; } } while (offset < bytes.size()); - uchardet_data_end(detector.get()); +#ifdef UCHARDET_FIXED_BLOCK_PILOT + adapter->finish(); +#else + uchardet_data_end(handle); +#endif std::cout << "{\"schema_version\":1,\"input_index\":" << file - 3 << ",\"byte_length\":" << bytes.size() << ",\"initial_done\":" << (initial_done ? "true" : "false") << ",\"feed_calls\":" << calls << ",\"first_done_offset\":"; if (observed_done) std::cout << first_done; else std::cout << "null"; - const size_t count = uchardet_get_n_candidates(detector.get()); - std::cout << ",\"final_done\":" << (uchardet_is_done(detector.get()) ? "true" : "false") +#ifdef UCHARDET_FIXED_BLOCK_PILOT + std::cout << ",\"block_size\":" << block_size << ",\"evidence_limit\":" << evidence_limit + << ",\"core_processed_bytes\":" << adapter->processed() + << ",\"core_feed_calls\":" << adapter->calls() + << ",\"core_first_done_offset\":"; + if (adapter->observed_done()) std::cout << adapter->first_done(); else std::cout << "null"; +#endif + const size_t count = uchardet_get_n_candidates(handle); + std::cout << ",\"final_done\":" << (uchardet_is_done(handle) ? "true" : "false") << ",\"candidate_count\":" << count << ",\"candidates\":["; for (size_t i = 0; i < count; ++i) { if (i) std::cout << ','; - std::cout << "{\"encoding\":"; quoted(uchardet_get_encoding(detector.get(), i)); - std::cout << ",\"language\":"; quoted(uchardet_get_language(detector.get(), i)); - const float confidence = uchardet_get_confidence(detector.get(), i); + std::cout << "{\"encoding\":"; quoted(uchardet_get_encoding(handle, i)); + std::cout << ",\"language\":"; quoted(uchardet_get_language(handle, i)); + const float confidence = uchardet_get_confidence(handle, i); uint32_t bits; std::memcpy(&bits, &confidence, sizeof(bits)); std::cout << ",\"confidence_bits\":\"" << std::hex << std::setw(8) << std::setfill('0') << bits << std::dec << "\"}"; }