From 76df2d5fac9cfa265304032f2f27773682b3fdf0 Mon Sep 17 00:00:00 2001 From: Luca Barbato Date: Thu, 8 Oct 2026 21:16:24 +0200 Subject: [PATCH 01/12] record(kolibri1): file the serving-completion issue The kolibri1 CPU arm reaches the model forward but not serving. This issue scopes the serving-completion unit: the kolibri1 reasoning parser (the plugin's Qwen3 grammar with the template's thinking switch), the kolibri1 tool-parser alias to the Hermes format, the template detection rows, and a jinja2-referenced rendering gate for the tokenizer-shipped chat template. Oracle: aleph-alpha-inference pin 049a6a7bd240. FOLLOWING_AGENTS_PROTOCOL Following-Agents-Protocol: true AI-Assisted: true Assisted-by: AGENT:zai/glm-5.3-flash [maki] --- .../ISSUE-LOCAL-01M4EF3R0H2H3FN5NA0NAB0S62.md | 19 +++++++++++++++++++ 1 file changed, 19 insertions(+) create mode 100644 .agents/issues/MODEL-TEXT-kolibri-1-kolibri1-for-causal-lm/ISSUE-LOCAL-01M4EF3R0H2H3FN5NA0NAB0S62.md diff --git a/.agents/issues/MODEL-TEXT-kolibri-1-kolibri1-for-causal-lm/ISSUE-LOCAL-01M4EF3R0H2H3FN5NA0NAB0S62.md b/.agents/issues/MODEL-TEXT-kolibri-1-kolibri1-for-causal-lm/ISSUE-LOCAL-01M4EF3R0H2H3FN5NA0NAB0S62.md new file mode 100644 index 000000000..d47c68860 --- /dev/null +++ b/.agents/issues/MODEL-TEXT-kolibri-1-kolibri1-for-causal-lm/ISSUE-LOCAL-01M4EF3R0H2H3FN5NA0NAB0S62.md @@ -0,0 +1,19 @@ +ID: ISSUE-LOCAL-01M4EF3R0H2H3FN5NA0NAB0S62 +Title: kolibri1 serving completion: chat template rendering plus kolibri1 reasoning and tool-call parsers +Row: MODEL-TEXT-kolibri-1-kolibri1-for-causal-lm +State: OPEN +Kind: enhancement +GitHub: - +Mirror: PENDING +Availability: FULL +Created: 2026-10-08 +Updated: 2026-10-08 +Closed: - + +## Problem + +The kolibri1 CPU arm reaches the model forward but not serving: the OpenAI chat path has no kolibri1 reasoning parser, no kolibri1 tool-parser alias, and no template-detection rows, so the model-author plugin's serving recipe (aleph-alpha-inference pin 049a6a7bd240: reasoning parser kolibri1 = Qwen3 grammar with the template's thinking switch derived from chat_template_kwargs reasoning_effort/enable_thinking; tool parser kolibri1 = the Hermes format; chat template shipped in tokenizer_config.json) cannot be served end to end. Scope: port the thinking_enabled switch and the engine-backed kolibri1 reasoning adapter, alias the kolibri1 tool parser to hermes, add template-marker detection rows, gate template rendering against jinja2 reference outputs on kolibri1 goldens. No tokenizer change, no behavior change to other models' parsers. + +## Resolution + +- From 3c0fcc1647a0b60faac93db51cec6fd4b9a6ce17 Mon Sep 17 00:00:00 2001 From: Luca Barbato Date: Thu, 8 Oct 2026 23:19:54 +0200 Subject: [PATCH 02/12] feat(kolibri1): serve the chat template and the kolibri1 parsers end to end The kolibri1 CPU arm reached the model forward but not serving: no kolibri1 reasoning parser, no kolibri1 tool-parser name, and detection resolved think_auto/hermes off the Kolibri template. The model-author plugin (aleph-alpha-inference @ 049a6a7bd240) is the serving oracle: its reasoning parser is the Qwen3 grammar with the starting state derived the way the template switches thinking (reasoning_effort wins and only "none" disables; else a literal enable_thinking false does), and its tool parser IS the Hermes class under the kolibri1 name. - kolibri1 reasoning adapter (reasoning.py:36-61 ported): per-request state from chat_template_kwargs, over the shared Qwen3 engine. - kolibri1 tool parser: alias to HermesToolParser (init.py:50-54). - Detection rows on the template's no-reasoning sentence, ahead of the generic and hermes rows; registry pins 12->13 and 42->43. - Renderer parity: minja's tojson dumped insertion order; jinja2's default policy sorts keys. Fixed adapter-side with a child-scope sorted-dump tojson; no vendor change. - Gate: byte-exact rendering vs CPython jinja2 references on 15 scenarios (fixtures/kolibri1_chat_template_references.json), parser unit tests on the plugin's switch truth table, alias equivalence with hermes. Red-first: detection and registry cases failed before the change. Evidence: docs/bench-evidence/kolibri1-serve-20261008.md. FOLLOWING_AGENTS_PROTOCOL Following-Agents-Protocol: true AI-Assisted: true Assisted-by: AGENT:zai/glm-5.3-flash [maki] --- .agents/specs/kolibri-1-cpu.md | 37 ++- CMakeLists.txt | 1 + .../bench-evidence/kolibri1-serve-20261008.md | 99 ++++++ .../openai/reasoning_parsers/kolibri1.h | 62 ++++ src/vllm/entrypoints/chat_template.cpp | 44 +++ .../openai/reasoning_parsers/abstract.cpp | 11 +- .../openai/reasoning_parsers/detect.cpp | 8 + .../openai/reasoning_parsers/kolibri1.cpp | 71 +++++ .../openai/tool_parsers/abstract.cpp | 9 + .../openai/tool_parsers/detect.cpp | 8 + tests/CMakeLists.txt | 8 + .../gen-kolibri1-chat-template-references.py | 131 ++++++++ .../kolibri1_chat_template_references.json | 286 ++++++++++++++++++ .../openai/reasoning_parsers/test_detect.cpp | 4 +- .../reasoning_parsers/test_kolibri1.cpp | 182 +++++++++++ .../openai/tool_parsers/test_detect.cpp | 5 +- .../openai/tool_parsers/test_kolibri1.cpp | 115 +++++++ .../test_kolibri1_chat_template.cpp | 137 +++++++++ 18 files changed, 1204 insertions(+), 14 deletions(-) create mode 100644 docs/bench-evidence/kolibri1-serve-20261008.md create mode 100644 include/vllm/entrypoints/openai/reasoning_parsers/kolibri1.h create mode 100644 src/vllm/entrypoints/openai/reasoning_parsers/kolibri1.cpp create mode 100644 tests/fixtures/gen-kolibri1-chat-template-references.py create mode 100644 tests/fixtures/kolibri1_chat_template_references.json create mode 100644 tests/vllm/entrypoints/openai/reasoning_parsers/test_kolibri1.cpp create mode 100644 tests/vllm/entrypoints/openai/tool_parsers/test_kolibri1.cpp create mode 100644 tests/vllm/entrypoints/test_kolibri1_chat_template.cpp diff --git a/.agents/specs/kolibri-1-cpu.md b/.agents/specs/kolibri-1-cpu.md index 0b2a61181..74573e43f 100644 --- a/.agents/specs/kolibri-1-cpu.md +++ b/.agents/specs/kolibri-1-cpu.md @@ -244,17 +244,32 @@ the kolibri1 pre-tokenizer regex, R7 — the encode contract stays owed). Remaining for the row: R7 tokenizer, the aleph-alpha-inference oracle gateability measurement (GPU), GGUF/CUDA/Tenstorrent arms (later rows). -R7 RESOLVED (2026-10-08, branch `row/kolibri-r7`): the tokenizer engine -accepts the Kolibri-1 split regex — recognition-only extension, the -regex maps onto the existing `kQwen2Classic` scanner because `\p{N}{1}` is -the identity quantifier on `\p{N}`. The W1 tokenizer test now loads the -real tokenizer.json and asserts our `Encode` reproduces the HF reference -ids from `tests/vllm/models/kolibri1_goldens.json` on all 8 golden -prompts (the no-BOS encode contract is gated on a real load for the first -time). Design, corrected diagnosis, risks, and stop conditions: -`## R7 resolution` below. Remaining for the row: the aleph-alpha-inference -oracle gateability measurement (GPU), GGUF/CUDA/Tenstorrent arms (later -rows). +SERVING COMPLETION (2026-10-08, branch `row/kolibri-serve`): the OpenAI +chat path now serves kolibri1 end to end on CPU. The checkpoint's +tokenizer_config.json chat template renders through the minja adapter +byte-identically to CPython jinja2 references on 15 scenarios covering +the plugin's thinking switch (tests/fixtures/ +kolibri1_chat_template_references.json); the one renderer divergence the +gate caught — minja's `tojson` dumped insertion order where jinja2's +default policy sorts keys (DEFAULT_POLICIES["json.dumps_kwargs"] = +{"sort_keys": True}) — is fixed in the adapter with a child-scope +sorted-dump `tojson` (src/vllm/entrypoints/chat_template.cpp). The +kolibri1 reasoning parser ports the plugin's reasoning.py @ 049a6a7bd240: +the Qwen3 engine grammar with the starting state derived the way the +template switches thinking (reasoning_effort wins and only "none" +disables; else a literal enable_thinking false does), threaded from the +request's chat_template_kwargs per call. The kolibri1 tool parser is the +plugin's registration mirrored exactly: an alias to the Hermes +`` class (__init__.py:50-54). Both detection tables resolve +"kolibri1" off the template's no-reasoning sentence, ahead of the +generic `` and hermes rows. Gates: test_reasoning_kolibri1, +test_tool_parser_kolibri1, test_kolibri1_chat_template green (red-first: +detection resolved think_auto/hermes and the registry names did not +exist before the change); the full host battery and the row's kolibri +gates stay green; W3 rerun in a verified quiet window. Evidence: +docs/bench-evidence/kolibri1-serve-20261008.md. Remaining for the row: +the aleph-alpha-inference oracle gateability measurement (GPU), +GGUF/CUDA/Tenstorrent arms (later rows). ## R7 resolution — the tokenizer engine accepts the Kolibri-1 split regex diff --git a/CMakeLists.txt b/CMakeLists.txt index 5eca5013a..b02dbf5bf 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -1415,6 +1415,7 @@ add_library(vllm STATIC src/vllm/entrypoints/openai/tool_parsers/muse_glimmer.cpp src/vllm/entrypoints/openai/tool_parsers/parser_engine_adapter.cpp src/vllm/entrypoints/openai/reasoning_parsers/muse_glimmer.cpp + src/vllm/entrypoints/openai/reasoning_parsers/kolibri1.cpp src/vllm/parser/engine/incremental_lexer.cpp src/vllm/parser/engine/token_id_scanner.cpp src/vllm/parser/engine/streaming_parser_engine.cpp diff --git a/docs/bench-evidence/kolibri1-serve-20261008.md b/docs/bench-evidence/kolibri1-serve-20261008.md new file mode 100644 index 000000000..b4cc5a00a --- /dev/null +++ b/docs/bench-evidence/kolibri1-serve-20261008.md @@ -0,0 +1,99 @@ +# kolibri1 serving completion — evidence (2026-10-08) + +Row `MODEL-TEXT-kolibri-1-kolibri1-for-causal-lm`, branch `row/kolibri-serve` +(base `bf7b654ae`). Issue: ISSUE-LOCAL-01M4EF3R0H2H3FN5NA0NAB0S62. + +## Oracle and what was mirrored + +Primary serving oracle: `aleph-alpha-inference`, pin `049a6a7bd240` +(PyPI v1.0.0), read from a git clone of +`https://github.com/Aleph-Alpha/aleph-alpha-inference` checked out at the +pin (`git log -1` = `049a6a7 chore(main): release 1.0.0 (#5)`). + +| Plugin anchor | What it defines | Local mirror | +|---|---|---| +| `aleph_alpha_inference/reasoning.py:36-48` (`thinking_enabled`) | the thinking switch: a non-None `reasoning_effort` wins, only `"none"` disables; else a literal `enable_thinking == False` disables | `Kolibri1ThinkingEnabled` in `src/vllm/entrypoints/openai/reasoning_parsers/kolibri1.cpp` | +| `reasoning.py:51-58` (`Kolibri1Parser`) | the Qwen3 grammar with the starting state chosen like the template | `Kolibri1ParserReasoningAdapter` over `vllm::parser::Qwen3Parser` (`pe::qwen3_config`), state derived per request from `chat_template_kwargs` | +| `reasoning.py:61` (`Kolibri1ParserReasoningAdapter`) | the engine-backed reasoning face | same-named adapter class, registered under `"kolibri1"` in `reasoning_parsers/abstract.cpp` | +| `__init__.py:50-54` | tool parser `kolibri1` = `vllm.tool_parsers.hermes_tool_parser.Hermes2ProToolParser` verbatim ("shares the Hermes `` format") | alias branch `"kolibri1" -> HermesToolParser` in `tool_parsers/abstract.cpp` + name in `tool_parser_names()` | +| `__init__.py:44-48` | reasoning parser registered under the name `kolibri1` | name in `reasoning_parser_names()`; detection rows resolve to it | +| `README.md:32-40` | the serving recipe (`--reasoning-parser kolibri1 --tool-call-parser kolibri1`) | both names now resolve; `--tool-call-parser auto` also detects kolibri1 from the template | + +The plugin ships NO chat template of its own: the template rides in the +checkpoint's `tokenizer_config.json` (`/mnt/models/Aleph-Alpha/Kolibri-1`), +and `reasoning.py:3-26` documents the exact switch it encodes. There is no +plugin-vs-HF template discrepancy to record — the plugin serves the HF +template. + +## Reference capture + +- Script: `tests/fixtures/gen-kolibri1-chat-template-references.py`. +- Fixture: `tests/fixtures/kolibri1_chat_template_references.json` (15 + scenarios: default, `enable_thinking:false`, `reasoning_effort` + none/low/medium/high/minimal/xhigh/max, effort-over-`enable_thinking` + both ways, tools on/off, system turn, preserved assistant reasoning, + ``-embedded content, tool-response turn, no generation prompt). +- Method: CPython jinja2 3.1.6 (system python3) rendering the checkpoint's + template text under transformers' whitespace policy (trim_blocks=True, + lstrip_blocks=True, keep_trailing_newline=False) — the policy the minja + adapter mirrors. The plugin itself could not be executed: it imports vLLM + at registration, and the oracle file records `gateable = no` (the GPU + measurement is owed); the plugin's PARSER behavior is mirrored from its + source, and the template behavior from the jinja2 stack it serves through. + +## Renderer divergence found and fixed + +`{{ tool | tojson }}` rendered with minja's insertion-order dump; the jinja2 +references sort keys (jinja2 `DEFAULT_POLICIES["json.dumps_kwargs"]` = +`{"sort_keys": True}` — the tojson every transformers/vLLM-served template +runs under). Fixed adapter-side in `src/vllm/entrypoints/chat_template.cpp`: +a child-scope `tojson` global (same `value`/`indent` signature) that sorts +object keys recursively before dumping. No vendor change; request kwargs can +still not shadow it (the builtins set already refuses `tojson`). + +## Red-first + +Captured before any implementation existed (build of the three new test +targets against the unmodified tree): + +- `test_reasoning_kolibri1`: 7/7 cases FAILED (registry had no + `kolibri1` reasoning parser). +- `test_tool_parser_kolibri1`: 3/3 FAILED (no `kolibri1` tool name). +- `test_kolibri1_chat_template`: detection resolved `think_auto` (reasoning) + and `hermes` (tool) off the real template; the two rendering cases that + involve `tojson` failed byte-compares. Non-tool rendering cases already + matched jinja2, which isolated the divergence to `tojson`. + +## Gates (this tree, CPU-only build, /tmp/build-kolibri-serve) + +- `test_reasoning_kolibri1`: 43 assertions, PASS. +- `test_tool_parser_kolibri1`: 19 assertions, PASS. +- `test_kolibri1_chat_template`: 35 assertions, PASS (byte-exact vs jinja2 on + all 15 scenarios; plus switch-boundary and detection cases). +- `test_reasoning_parser_detect`: 75 assertions PASS; `test_tool_parser_detect`: + 361 assertions PASS; registry count pins moved 12->13 (reasoning) and + 42->43 (tool) with dated comments. +- `test_reasoning_qwen3` 164 PASS, `test_openai_tool_parsers` 64 PASS, + `test_chat_template` 196 PASS (no behavior change to other parsers). +- Row battery (ctest, this host): `test_kolibri1`, `test_kolibri1_dequant`, + `test_kolibri1_dequant_cache`, `test_kolibri1_w2`, `test_kolibri1_w3`, + `test_kolibri1_decode_bench`, `test_kolibri1_moe_glue` — all PASS + (test_kolibri1_tt / _b2i / _b2bi PASS in the same run). +- W3 rerun in a verified quiet window (free > 110 GB, no other + test_kolibri1_w3 process), `VLLM_CPP_CPU_THREADS=8`: PASS. + +Pre-existing, not touched by this change (verified on the same tree): +`ctest`-cwd artifacts from `-ffile-prefix-map` (`test_linear_scaling_rope` +passes when run from the source dir), the `test_safetensors` RSS-mapping +assertion under host memory pressure, and the gliner fixture loads +(`test_gliner2_e2e`, `test_capi` v27). + +## Out of scope + +- The oracle gateability measurement (`vllm serve` with the plugin on a GPU + lease) — owed by the oracle file, unchanged. +- chat_template_kwargs threading for the OTHER engine-backed reasoning + parsers (the pre-existing W4 note in `reasoning_parsers/abstract.cpp`); + kolibri1 carries its own per-request derivation because the plugin's + parser is defined by it. +- fp8 KV cache / 1M-context serving recipe (loader/config rows). diff --git a/include/vllm/entrypoints/openai/reasoning_parsers/kolibri1.h b/include/vllm/entrypoints/openai/reasoning_parsers/kolibri1.h new file mode 100644 index 000000000..615bc6da0 --- /dev/null +++ b/include/vllm/entrypoints/openai/reasoning_parsers/kolibri1.h @@ -0,0 +1,62 @@ +// Ported from: aleph-alpha-inference @ 049a6a7bd240 +// (aleph_alpha_inference/reasoning.py) — the model-author vLLM plugin for +// Kolibri, the serving oracle for the kolibri1 architecture +// (.agents/oracles/aleph-alpha-inference.md; upstream vLLM implements nothing +// for it). +// +// reasoning.py:36-48 thinking_enabled — the template's own switch as a +// function: a reasoning_effort that is not None WINS (only "none" disables +// thinking), and only absent-or-None effort falls through to a literal +// enable_thinking == false test. The stock qwen3 reasoning parser reads +// enable_thinking alone, which is wrong on exactly the two arms the plugin +// exists for: effort arriving in chat_template_kwargs, and effort +// contradicting an explicit enable_thinking (reasoning.py:12-22). +// +// reasoning.py:51-58 Kolibri1Parser — the Qwen3 grammar with the starting +// state chosen that way; reasoning.py:61 Kolibri1ParserReasoningAdapter — the +// engine-backed reasoning face over it. The upstream ctor reads the request's +// chat_template_kwargs at parser construction; this seam constructs parsers +// per server name (reasoning_parsers/abstract.cpp), so the kolibri1 adapter +// derives the state lazily from the FIRST request it sees and re-derives it +// on every call (both extract methods carry the request; is_reasoning_end is +// thinking-independent for the Qwen3 grammar — qwen3.cpp). +#ifndef VLLM_ENTRYPOINTS_OPENAI_REASONING_PARSERS_KOLIBRI1_H_ +#define VLLM_ENTRYPOINTS_OPENAI_REASONING_PARSERS_KOLIBRI1_H_ + +#include + +#include "vllm/entrypoints/openai/reasoning_parsers/parser_engine_adapter.h" + +namespace vllm::entrypoints::openai { + +// reasoning.py:36 (thinking_enabled). +bool Kolibri1ThinkingEnabled( + const nlohmann::ordered_json& chat_template_kwargs); + +// reasoning.py:61 (Kolibri1ParserReasoningAdapter). The Qwen3 engine face +// with the starting state derived from the request's chat_template_kwargs the +// way the Kolibri template renders its generation prompt. +class Kolibri1ParserReasoningAdapter final : public ParserEngineReasoningAdapter { + public: + Kolibri1ParserReasoningAdapter(); + + ExtractedReasoning extract_reasoning( + const std::string& model_output, + const ChatCompletionRequest& request) override; + + std::optional extract_reasoning_streaming( + const std::string& previous_text, const std::string& current_text, + const std::string& delta_text, + const ChatCompletionRequest& request) override; + + private: + // Point engine_ at the state the request's kwargs select. Idempotent after + // the first call for a given state; a mid-stream flip is impossible because + // both extract feeds derive the same kwargs every call. + void EnsureEngine(const ChatCompletionRequest& request); + bool state_chosen_ = false; +}; + +} // namespace vllm::entrypoints::openai + +#endif // VLLM_ENTRYPOINTS_OPENAI_REASONING_PARSERS_KOLIBRI1_H_ diff --git a/src/vllm/entrypoints/chat_template.cpp b/src/vllm/entrypoints/chat_template.cpp index 47792b2cd..a662da5d8 100644 --- a/src/vllm/entrypoints/chat_template.cpp +++ b/src/vllm/entrypoints/chat_template.cpp @@ -13,7 +13,9 @@ #include "vllm/model_executor/model_loader/gguf_reader.h" #include "vllm/v1/engine/validation_error.h" // refused kwarg -> HTTP 400 +#include #include +#include #include #include #include @@ -209,6 +211,48 @@ std::string RenderChatTemplate( std::shared_ptr builtins = minja::Context::builtins(); std::shared_ptr context = minja::Context::make(minja::Value(top), builtins); + // tojson with jinja2's DEFAULT policy: CPython Jinja sorts object keys + // (jinja2.defaults.DEFAULT_POLICIES["json.dumps_kwargs"] = + // {"sort_keys": True}), and that is the tojson every transformers / + // vLLM-served template runs under -- the Kolibri tool preamble renders + // `{{ tool | tojson }}` and byte-matches the jinja2 references only with + // the sort. minja's builtin dumps in insertion order, so this child-scope + // global shadows it (same signature: value + optional indent; set() on + // the child cannot touch the shared builtins). + context->set( + "tojson", + minja::Value::callable( + [](const std::shared_ptr&, + minja::ArgumentsValue& args) -> minja::Value { + minja::Value value = args.args.at(0); + int64_t indent = -1; + for (const auto& [name, v] : args.kwargs) { + if (name == "indent") indent = v.get(); + } + std::function sort_keys = + [&](minja::Value v) -> minja::Value { + if (v.is_object()) { + std::vector keys; + for (const auto& k : v.keys()) keys.push_back(k.get()); + std::sort(keys.begin(), keys.end()); + minja::Value out = minja::Value::object(); + for (const auto& k : keys) { + out.set(minja::Value(k), sort_keys(v.at(minja::Value(k)))); + } + return out; + } + if (v.is_array()) { + minja::Value out = minja::Value::array(); + for (std::size_t i = 0; i < v.size(); ++i) { + out.push_back(sort_keys(v.at(minja::Value(static_cast(i))))); + } + return out; + } + return v; + }; + return minja::Value(sort_keys(value).dump(indent, + /*to_json=*/true)); + })); context->set("bos_token", minja::Value(bos_token)); context->set("eos_token", minja::Value(eos_token)); context->set("tools", minja::Value(BuildTools(tools))); diff --git a/src/vllm/entrypoints/openai/reasoning_parsers/abstract.cpp b/src/vllm/entrypoints/openai/reasoning_parsers/abstract.cpp index 9dc9038f1..7ebe02372 100644 --- a/src/vllm/entrypoints/openai/reasoning_parsers/abstract.cpp +++ b/src/vllm/entrypoints/openai/reasoning_parsers/abstract.cpp @@ -9,6 +9,7 @@ #include "vllm/entrypoints/openai/reasoning_parsers/deepseek_r1.h" #include "vllm/entrypoints/openai/reasoning_parsers/deepseek_v3.h" +#include "vllm/entrypoints/openai/reasoning_parsers/kolibri1.h" #include "vllm/entrypoints/openai/reasoning_parsers/think_auto.h" #include "vllm/entrypoints/openai/reasoning_parsers/minimax_m2.h" #include "vllm/entrypoints/openai/reasoning_parsers/mistral.h" @@ -64,6 +65,14 @@ std::unique_ptr get_reasoning_parser(const std::string& name) { if (name == "qwen3" || name == "mimo") { return std::make_unique(); } + // aleph-alpha-inference @ 049a6a7bd240 __init__.py:44-48 registers + // "kolibri1" -> Kolibri1ParserReasoningAdapter (reasoning.py:61): the qwen3 + // engine face with the starting state derived from the request's + // chat_template_kwargs the way the Kolibri template switches thinking + // (reasoning_effort "none" / literal enable_thinking false; see kolibri1.h). + if (name == "kolibri1") { + return std::make_unique(); + } return nullptr; } @@ -73,7 +82,7 @@ const std::vector& reasoning_parser_names() { static const std::vector names = { "think_auto", "deepseek_r1", "deepseek_v3", "holo2", "mistral", "minimax_m2", "minimax_m2_append_think", "step3", "olmo3", - "muse_glimmer", "qwen3", "mimo", + "muse_glimmer", "qwen3", "mimo", "kolibri1", }; return names; } diff --git a/src/vllm/entrypoints/openai/reasoning_parsers/detect.cpp b/src/vllm/entrypoints/openai/reasoning_parsers/detect.cpp index 703c88ee0..74e0c292b 100644 --- a/src/vllm/entrypoints/openai/reasoning_parsers/detect.cpp +++ b/src/vllm/entrypoints/openai/reasoning_parsers/detect.cpp @@ -28,8 +28,16 @@ namespace { // tool-definition preamble: a full literal, shared with nothing else here, // and the same tell the tool-parser table uses (the two parsers are always // selected together). +// kolibri1: the Kolibri template's no-reasoning system sentence +// ("Reasoning is disabled. Proceed straight to answering…") is a full literal +// shared with no other row and contained by no other row's marker. It must +// precede the generic "" row, which the same template also contains +// (the plugin serves kolibri1's OWN reasoning parser, not think_auto: +// aleph-alpha-inference __init__.py:44-48). constexpr ReasoningParserMarker kReasoningParserMarkers[] = { {"muse_glimmer", ""}, + {"kolibri1", + "Reasoning is disabled. Proceed straight to answering"}, {"mistral", "[THINK]"}, {"think_auto", ""}, }; diff --git a/src/vllm/entrypoints/openai/reasoning_parsers/kolibri1.cpp b/src/vllm/entrypoints/openai/reasoning_parsers/kolibri1.cpp new file mode 100644 index 000000000..64ee5a39d --- /dev/null +++ b/src/vllm/entrypoints/openai/reasoning_parsers/kolibri1.cpp @@ -0,0 +1,71 @@ +// See kolibri1.h. Ported from aleph-alpha-inference @ 049a6a7bd240 +// (aleph_alpha_inference/reasoning.py:36-61). +#include "vllm/entrypoints/openai/reasoning_parsers/kolibri1.h" + +#include +#include +#include +#include + +#include "vllm/parser/engine/configs.h" +#include "vllm/parser/qwen3.h" + +namespace vllm::entrypoints::openai { + +namespace pe = vllm::parser::engine; + +// reasoning.py:36-48. `effort = kwargs.get("reasoning_effort")` — a JSON null +// IS Python None, so `effort is not None` is `effort != nullptr` here; only +// the literal "none" disables. Otherwise `kwargs.get("enable_thinking") is +// not False`: the template tests `enable_thinking is false`, so anything but +// the literal false (absent included) thinks. +bool Kolibri1ThinkingEnabled( + const nlohmann::ordered_json& chat_template_kwargs) { + const auto effort = chat_template_kwargs.find("reasoning_effort"); + if (effort != chat_template_kwargs.end() && !effort->is_null()) { + // reasoning.py:47 `effort != "none"` — any non-None value other than the + // literal "none" enables thinking, of whatever JSON type a client sends. + return !(effort->is_string() && effort->get() == "none"); + } + const auto enable = chat_template_kwargs.find("enable_thinking"); + return !(enable != chat_template_kwargs.end() && enable->is_boolean() && + enable->get() == false); +} + +Kolibri1ParserReasoningAdapter::Kolibri1ParserReasoningAdapter() + // Upstream default (reasoning.py: no kwargs -> thinking on); replaced on + // the first request by EnsureEngine when the kwargs say otherwise. + : ParserEngineReasoningAdapter(std::make_unique( + pe::qwen3_config(true, "qwen3"), true)) {} + +void Kolibri1ParserReasoningAdapter::EnsureEngine( + const ChatCompletionRequest& request) { + const bool thinking = Kolibri1ThinkingEnabled(request.chat_template_kwargs); + if (state_chosen_) { + // The state is a property of the request's kwargs, which do not change + // mid-request; both extract feeds derive the same value every call. + return; + } + state_chosen_ = true; + if (!thinking) { + engine_ = std::make_unique( + pe::qwen3_config(false, "qwen3"), false); + } +} + +ExtractedReasoning Kolibri1ParserReasoningAdapter::extract_reasoning( + const std::string& model_output, const ChatCompletionRequest& request) { + EnsureEngine(request); + return ParserEngineReasoningAdapter::extract_reasoning(model_output, request); +} + +std::optional +Kolibri1ParserReasoningAdapter::extract_reasoning_streaming( + const std::string& previous_text, const std::string& current_text, + const std::string& delta_text, const ChatCompletionRequest& request) { + EnsureEngine(request); + return ParserEngineReasoningAdapter::extract_reasoning_streaming( + previous_text, current_text, delta_text, request); +} + +} // namespace vllm::entrypoints::openai diff --git a/src/vllm/entrypoints/openai/tool_parsers/abstract.cpp b/src/vllm/entrypoints/openai/tool_parsers/abstract.cpp index 0e1be8577..e259dd07f 100644 --- a/src/vllm/entrypoints/openai/tool_parsers/abstract.cpp +++ b/src/vllm/entrypoints/openai/tool_parsers/abstract.cpp @@ -96,6 +96,14 @@ std::unique_ptr get_tool_parser(const std::string& name) { if (name == "hermes") { return std::make_unique(); } + // kolibri1 IS the Hermes class upstream: the plugin registers the kolibri1 + // name over vllm.tool_parsers.hermes_tool_parser.Hermes2ProToolParser + // verbatim (aleph-alpha-inference @ 049a6a7bd240 __init__.py:50-54, "shares + // the Hermes ... format"), so this is an alias to + // the same implementation, not a parallel dialect. + if (name == "kolibri1") { + return std::make_unique(); + } if (name == "qwen3") { return std::make_unique(); } @@ -301,6 +309,7 @@ const std::vector& tool_parser_names() { "glm47", "minimax_m2", "gemma4", "seed_oss", "muse_glimmer", "inkling", + "kolibri1", }; return names; } diff --git a/src/vllm/entrypoints/openai/tool_parsers/detect.cpp b/src/vllm/entrypoints/openai/tool_parsers/detect.cpp index 6745225d2..12e9809e0 100644 --- a/src/vllm/entrypoints/openai/tool_parsers/detect.cpp +++ b/src/vllm/entrypoints/openai/tool_parsers/detect.cpp @@ -75,6 +75,14 @@ namespace { // step3's fullwidth ones. constexpr ToolParserMarker kToolParserMarkers[] = { {"muse_glimmer", ""}, + // kolibri1: the Kolibri template's no-reasoning system sentence (see the + // reasoning table above) — placed before the hermes row, which the same + // template ALSO matches (its tool preamble wraps calls in bare + // ); the plugin registers the kolibri1 NAME over the Hermes + // class (aleph-alpha-inference __init__.py:50-54), so detection must + // resolve to that name, not to hermes, for name parity. + {"kolibri1", + "Reasoning is disabled. Proceed straight to answering"}, {"longcat", ""}, {"deepseek_v3", "<|tool▁calls▁begin|>"}, {"deepseek_v32", "<|DSML|function_calls>"}, diff --git a/tests/CMakeLists.txt b/tests/CMakeLists.txt index 7cad62b9e..695aef9d0 100644 --- a/tests/CMakeLists.txt +++ b/tests/CMakeLists.txt @@ -2660,6 +2660,14 @@ vllm_cpp_add_test(test_reasoning_deepseek_v3 vllm/entrypoints/openai/reasoning_parsers/test_deepseek_v3.cpp) vllm_cpp_add_test(test_reasoning_qwen3 vllm/entrypoints/openai/reasoning_parsers/test_qwen3.cpp) +vllm_cpp_add_test(test_reasoning_kolibri1 + vllm/entrypoints/openai/reasoning_parsers/test_kolibri1.cpp) +vllm_cpp_add_test(test_tool_parser_kolibri1 + vllm/entrypoints/openai/tool_parsers/test_kolibri1.cpp) +vllm_cpp_add_test(test_kolibri1_chat_template + vllm/entrypoints/test_kolibri1_chat_template.cpp) +target_compile_definitions(test_kolibri1_chat_template PRIVATE + KOLIBRI1_TEMPLATE_FIXTURE_DIR="${CMAKE_SOURCE_DIR}/tests/fixtures") vllm_cpp_add_test(test_tool_parser_xlam vllm/entrypoints/openai/tool_parsers/test_xlam_tool_parser.cpp) vllm_cpp_add_test(test_tool_parser_phi4mini diff --git a/tests/fixtures/gen-kolibri1-chat-template-references.py b/tests/fixtures/gen-kolibri1-chat-template-references.py new file mode 100644 index 000000000..bfd61970a --- /dev/null +++ b/tests/fixtures/gen-kolibri1-chat-template-references.py @@ -0,0 +1,131 @@ +#!/usr/bin/env python3 +"""Reference capture for the Kolibri-1 chat template. + +Renders the chat_template shipped in +/mnt/models/Aleph-Alpha/Kolibri-1/tokenizer_config.json with CPython Jinja2 +under transformers' whitespace policy (trim_blocks=True, lstrip_blocks=True, +keep_trailing_newline=False), which is the same policy the vendored minja +engine mirrors (src/vllm/entrypoints/chat_template.cpp). The output is the +fixture the C++ rendering gate compares against. + +Provenance: template text from the pinned checkpoint tokenizer_config.json; +behavior oracle aleph-alpha-inference @ 049a6a7bd240 (the plugin serves this +checkpoint; reasoning.py:36 thinking_enabled documents the same switch the +template encodes). +""" + +import json +import pathlib + +import jinja2 + +TEMPLATE_PATH = pathlib.Path( + "/mnt/models/Aleph-Alpha/Kolibri-1/tokenizer_config.json" +) +OUT_PATH = pathlib.Path( + "/tmp/vllm-kolibri-serve/tests/fixtures/kolibri1_chat_template_references.json" +) + +TOOLS = [ + { + "type": "function", + "function": { + "name": "get_weather", + "description": "Get the weather for a city", + "parameters": { + "type": "object", + "properties": {"city": {"type": "string"}}, + "required": ["city"], + }, + }, + } +] + +USER = {"role": "user", "content": "What is the weather in Berlin?"} +ASSISTANT_REASONING = { + "role": "assistant", + "content": "It is sunny.", + "reasoning": "The user asks about Berlin weather.", +} +ASSISTANT_THINK_IN_CONTENT = { + "role": "assistant", + "content": "\nBerlin is in Germany.\n\nIt is sunny.", +} +TOOL_MSG = {"role": "tool", "content": "22C, clear"} + +SCENARIOS = [ + # (name, messages, add_generation_prompt, kwargs, tools) + ("default_user_only", [USER], True, {}, None), + ("enable_thinking_false", [USER], True, {"enable_thinking": False}, None), + ("reasoning_effort_none", [USER], True, {"reasoning_effort": "none"}, None), + ("reasoning_effort_low", [USER], True, {"reasoning_effort": "low"}, None), + ("reasoning_effort_medium", [USER], True, {"reasoning_effort": "medium"}, None), + ("reasoning_effort_high", [USER], True, {"reasoning_effort": "high"}, None), + # Effort wins over an explicit enable_thinking:false (reasoning.py:44-48). + ( + "effort_low_overrides_enable_thinking_false", + [USER], + True, + {"reasoning_effort": "low", "enable_thinking": False}, + None, + ), + ("effort_none_overrides_enable_thinking_default", [USER], True, + {"reasoning_effort": "none", "enable_thinking": True}, None), + ("with_tools", [USER], True, {}, TOOLS), + ("with_tools_thinking_off", [USER], True, {"enable_thinking": False}, TOOLS), + ("system_and_user", [ + {"role": "system", "content": "You are helpful."}, USER], True, {}, None), + ("assistant_turn_preserved", [USER, ASSISTANT_REASONING, USER], False, {}, + None), + ("assistant_think_in_content", [USER, ASSISTANT_THINK_IN_CONTENT, USER], + False, {}, None), + ("tool_response_turn", [USER, ASSISTANT_REASONING, TOOL_MSG], True, {}, + None), + ("no_generation_prompt", [USER], False, {}, None), +] + + +def render(template, messages, add_generation_prompt, kwargs, tools): + env = jinja2.Environment(trim_blocks=True, lstrip_blocks=True, + keep_trailing_newline=False) + context = { + "messages": messages, + "add_generation_prompt": add_generation_prompt, + "bos_token": "", + "eos_token": "", + "tools": tools if tools else [], + } + context.update(kwargs) + return env.from_string(template).render(**context) + + +def main(): + template = json.loads(TEMPLATE_PATH.read_text())["chat_template"] + cases = [] + for name, messages, agp, kwargs, tools in SCENARIOS: + cases.append({ + "name": name, + "messages": messages, + "add_generation_prompt": agp, + "chat_template_kwargs": kwargs, + "tools": tools, + "expected": render(template, messages, agp, kwargs, tools), + }) + OUT_PATH.write_text(json.dumps({ + "provenance": ( + "chat_template from /mnt/models/Aleph-Alpha/Kolibri-1/" + "tokenizer_config.json (checkpoint pin of oracle " + "aleph-alpha-inference 049a6a7bd240); rendered with CPython jinja2 " + "3.1.6 under transformers' whitespace policy " + "(trim_blocks=True, lstrip_blocks=True, keep_trailing_newline=" + "False), the policy src/vllm/entrypoints/chat_template.cpp " + "mirrors; captured by tests/fixtures/" + "gen-kolibri1-chat-template-references.py" + ), + "cases": cases, + }, indent=1) + "\n") + print(f"wrote {OUT_PATH} with {len(cases)} cases") + + +if __name__ == "__main__": + main() diff --git a/tests/fixtures/kolibri1_chat_template_references.json b/tests/fixtures/kolibri1_chat_template_references.json new file mode 100644 index 000000000..dd1c05b6d --- /dev/null +++ b/tests/fixtures/kolibri1_chat_template_references.json @@ -0,0 +1,286 @@ +{ + "provenance": "chat_template from /mnt/models/Aleph-Alpha/Kolibri-1/tokenizer_config.json (checkpoint pin of oracle aleph-alpha-inference 049a6a7bd240); rendered with CPython jinja2 3.1.6 under transformers' whitespace policy (trim_blocks=True, lstrip_blocks=True, keep_trailing_newline=False), the policy src/vllm/entrypoints/chat_template.cpp mirrors; captured by tests/fixtures/gen-kolibri1-chat-template-references.py", + "cases": [ + { + "name": "default_user_only", + "messages": [ + { + "role": "user", + "content": "What is the weather in Berlin?" + } + ], + "add_generation_prompt": true, + "chat_template_kwargs": {}, + "tools": null, + "expected": "<|im_start|>system\n# Reasoning effort\n\nReasoning effort is set to high. Think carefully through the task in the user's language, validate key assumptions, consider plausible alternatives, and prioritize correctness and clarity.<|im_end|>\n<|im_start|>user\nWhat is the weather in Berlin?<|im_end|>\n<|im_start|>assistant\n" + }, + { + "name": "enable_thinking_false", + "messages": [ + { + "role": "user", + "content": "What is the weather in Berlin?" + } + ], + "add_generation_prompt": true, + "chat_template_kwargs": { + "enable_thinking": false + }, + "tools": null, + "expected": "<|im_start|>system\n# Reasoning effort\n\nReasoning is disabled. Proceed straight to answering according to the user's instructions.<|im_end|>\n<|im_start|>user\nWhat is the weather in Berlin?<|im_end|>\n<|im_start|>assistant\n\n\n\n\n" + }, + { + "name": "reasoning_effort_none", + "messages": [ + { + "role": "user", + "content": "What is the weather in Berlin?" + } + ], + "add_generation_prompt": true, + "chat_template_kwargs": { + "reasoning_effort": "none" + }, + "tools": null, + "expected": "<|im_start|>system\n# Reasoning effort\n\nReasoning is disabled. Proceed straight to answering according to the user's instructions.<|im_end|>\n<|im_start|>user\nWhat is the weather in Berlin?<|im_end|>\n<|im_start|>assistant\n\n\n\n\n" + }, + { + "name": "reasoning_effort_low", + "messages": [ + { + "role": "user", + "content": "What is the weather in Berlin?" + } + ], + "add_generation_prompt": true, + "chat_template_kwargs": { + "reasoning_effort": "low" + }, + "tools": null, + "expected": "<|im_start|>system\n# Reasoning effort\n\nReasoning effort is set to low. Think briefly through only the essential steps in the user's language, then proceed directly to the answer.<|im_end|>\n<|im_start|>user\nWhat is the weather in Berlin?<|im_end|>\n<|im_start|>assistant\n" + }, + { + "name": "reasoning_effort_medium", + "messages": [ + { + "role": "user", + "content": "What is the weather in Berlin?" + } + ], + "add_generation_prompt": true, + "chat_template_kwargs": { + "reasoning_effort": "medium" + }, + "tools": null, + "expected": "<|im_start|>system\n# Reasoning effort\n\nReasoning effort is set to medium. Think through the task methodically in the user's language, check key assumptions, and provide a well-supported answer.<|im_end|>\n<|im_start|>user\nWhat is the weather in Berlin?<|im_end|>\n<|im_start|>assistant\n" + }, + { + "name": "reasoning_effort_high", + "messages": [ + { + "role": "user", + "content": "What is the weather in Berlin?" + } + ], + "add_generation_prompt": true, + "chat_template_kwargs": { + "reasoning_effort": "high" + }, + "tools": null, + "expected": "<|im_start|>system\n# Reasoning effort\n\nReasoning effort is set to high. Think carefully through the task in the user's language, validate key assumptions, consider plausible alternatives, and prioritize correctness and clarity.<|im_end|>\n<|im_start|>user\nWhat is the weather in Berlin?<|im_end|>\n<|im_start|>assistant\n" + }, + { + "name": "effort_low_overrides_enable_thinking_false", + "messages": [ + { + "role": "user", + "content": "What is the weather in Berlin?" + } + ], + "add_generation_prompt": true, + "chat_template_kwargs": { + "reasoning_effort": "low", + "enable_thinking": false + }, + "tools": null, + "expected": "<|im_start|>system\n# Reasoning effort\n\nReasoning effort is set to low. Think briefly through only the essential steps in the user's language, then proceed directly to the answer.<|im_end|>\n<|im_start|>user\nWhat is the weather in Berlin?<|im_end|>\n<|im_start|>assistant\n" + }, + { + "name": "effort_none_overrides_enable_thinking_default", + "messages": [ + { + "role": "user", + "content": "What is the weather in Berlin?" + } + ], + "add_generation_prompt": true, + "chat_template_kwargs": { + "reasoning_effort": "none", + "enable_thinking": true + }, + "tools": null, + "expected": "<|im_start|>system\n# Reasoning effort\n\nReasoning is disabled. Proceed straight to answering according to the user's instructions.<|im_end|>\n<|im_start|>user\nWhat is the weather in Berlin?<|im_end|>\n<|im_start|>assistant\n\n\n\n\n" + }, + { + "name": "with_tools", + "messages": [ + { + "role": "user", + "content": "What is the weather in Berlin?" + } + ], + "add_generation_prompt": true, + "chat_template_kwargs": {}, + "tools": [ + { + "type": "function", + "function": { + "name": "get_weather", + "description": "Get the weather for a city", + "parameters": { + "type": "object", + "properties": { + "city": { + "type": "string" + } + }, + "required": [ + "city" + ] + } + } + } + ], + "expected": "<|im_start|>system\n# Reasoning effort\n\nReasoning effort is set to high. Think carefully through the task in the user's language, validate key assumptions, consider plausible alternatives, and prioritize correctness and clarity.\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n\n{\"function\": {\"description\": \"Get the weather for a city\", \"name\": \"get_weather\", \"parameters\": {\"properties\": {\"city\": {\"type\": \"string\"}}, \"required\": [\"city\"], \"type\": \"object\"}}, \"type\": \"function\"}\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n<|im_start|>user\nWhat is the weather in Berlin?<|im_end|>\n<|im_start|>assistant\n" + }, + { + "name": "with_tools_thinking_off", + "messages": [ + { + "role": "user", + "content": "What is the weather in Berlin?" + } + ], + "add_generation_prompt": true, + "chat_template_kwargs": { + "enable_thinking": false + }, + "tools": [ + { + "type": "function", + "function": { + "name": "get_weather", + "description": "Get the weather for a city", + "parameters": { + "type": "object", + "properties": { + "city": { + "type": "string" + } + }, + "required": [ + "city" + ] + } + } + } + ], + "expected": "<|im_start|>system\n# Reasoning effort\n\nReasoning is disabled. Proceed straight to answering according to the user's instructions.\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n\n{\"function\": {\"description\": \"Get the weather for a city\", \"name\": \"get_weather\", \"parameters\": {\"properties\": {\"city\": {\"type\": \"string\"}}, \"required\": [\"city\"], \"type\": \"object\"}}, \"type\": \"function\"}\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n<|im_start|>user\nWhat is the weather in Berlin?<|im_end|>\n<|im_start|>assistant\n\n\n\n\n" + }, + { + "name": "system_and_user", + "messages": [ + { + "role": "system", + "content": "You are helpful." + }, + { + "role": "user", + "content": "What is the weather in Berlin?" + } + ], + "add_generation_prompt": true, + "chat_template_kwargs": {}, + "tools": null, + "expected": "<|im_start|>system\nYou are helpful.\n\n# Reasoning effort\n\nReasoning effort is set to high. Think carefully through the task in the user's language, validate key assumptions, consider plausible alternatives, and prioritize correctness and clarity.<|im_end|>\n<|im_start|>user\nWhat is the weather in Berlin?<|im_end|>\n<|im_start|>assistant\n" + }, + { + "name": "assistant_turn_preserved", + "messages": [ + { + "role": "user", + "content": "What is the weather in Berlin?" + }, + { + "role": "assistant", + "content": "It is sunny.", + "reasoning": "The user asks about Berlin weather." + }, + { + "role": "user", + "content": "What is the weather in Berlin?" + } + ], + "add_generation_prompt": false, + "chat_template_kwargs": {}, + "tools": null, + "expected": "<|im_start|>system\n# Reasoning effort\n\nReasoning effort is set to high. Think carefully through the task in the user's language, validate key assumptions, consider plausible alternatives, and prioritize correctness and clarity.<|im_end|>\n<|im_start|>user\nWhat is the weather in Berlin?<|im_end|>\n<|im_start|>assistant\nIt is sunny.<|im_end|>\n<|im_start|>user\nWhat is the weather in Berlin?<|im_end|>\n" + }, + { + "name": "assistant_think_in_content", + "messages": [ + { + "role": "user", + "content": "What is the weather in Berlin?" + }, + { + "role": "assistant", + "content": "\nBerlin is in Germany.\n\nIt is sunny." + }, + { + "role": "user", + "content": "What is the weather in Berlin?" + } + ], + "add_generation_prompt": false, + "chat_template_kwargs": {}, + "tools": null, + "expected": "<|im_start|>system\n# Reasoning effort\n\nReasoning effort is set to high. Think carefully through the task in the user's language, validate key assumptions, consider plausible alternatives, and prioritize correctness and clarity.<|im_end|>\n<|im_start|>user\nWhat is the weather in Berlin?<|im_end|>\n<|im_start|>assistant\nIt is sunny.<|im_end|>\n<|im_start|>user\nWhat is the weather in Berlin?<|im_end|>\n" + }, + { + "name": "tool_response_turn", + "messages": [ + { + "role": "user", + "content": "What is the weather in Berlin?" + }, + { + "role": "assistant", + "content": "It is sunny.", + "reasoning": "The user asks about Berlin weather." + }, + { + "role": "tool", + "content": "22C, clear" + } + ], + "add_generation_prompt": true, + "chat_template_kwargs": {}, + "tools": null, + "expected": "<|im_start|>system\n# Reasoning effort\n\nReasoning effort is set to high. Think carefully through the task in the user's language, validate key assumptions, consider plausible alternatives, and prioritize correctness and clarity.<|im_end|>\n<|im_start|>user\nWhat is the weather in Berlin?<|im_end|>\n<|im_start|>assistant\n\nThe user asks about Berlin weather.\n\n\nIt is sunny.<|im_end|>\n<|im_start|>user\n\n22C, clear\n<|im_end|>\n<|im_start|>assistant\n" + }, + { + "name": "no_generation_prompt", + "messages": [ + { + "role": "user", + "content": "What is the weather in Berlin?" + } + ], + "add_generation_prompt": false, + "chat_template_kwargs": {}, + "tools": null, + "expected": "<|im_start|>system\n# Reasoning effort\n\nReasoning effort is set to high. Think carefully through the task in the user's language, validate key assumptions, consider plausible alternatives, and prioritize correctness and clarity.<|im_end|>\n<|im_start|>user\nWhat is the weather in Berlin?<|im_end|>\n" + } + ] +} diff --git a/tests/vllm/entrypoints/openai/reasoning_parsers/test_detect.cpp b/tests/vllm/entrypoints/openai/reasoning_parsers/test_detect.cpp index f29095e38..3a642523f 100644 --- a/tests/vllm/entrypoints/openai/reasoning_parsers/test_detect.cpp +++ b/tests/vllm/entrypoints/openai/reasoning_parsers/test_detect.cpp @@ -104,7 +104,9 @@ TEST_CASE("Registry: every enumerated reasoning-parser name resolves") { // 2026-08-10 (MODEL-MUSE-GLIMMER-W7): 9 -> 10, adding "muse_glimmer". // 2026-08-13 (SAMPLE-REASONING W3, #605): 10 -> 12, adding the engine-backed // "qwen3" + its "mimo" alias (ONE class, two upstream registry names). - CHECK(names.size() == 12); + // 2026-10-08 (MODEL-TEXT-kolibri-1 serving completion): 12 -> 13, adding + // "kolibri1" (the model-author plugin's reasoning parser name). + CHECK(names.size() == 13); std::size_t marker_count = 0; const ReasoningParserMarker* markers = ReasoningParserMarkerTable(&marker_count); for (std::size_t i = 0; i < marker_count; ++i) { diff --git a/tests/vllm/entrypoints/openai/reasoning_parsers/test_kolibri1.cpp b/tests/vllm/entrypoints/openai/reasoning_parsers/test_kolibri1.cpp new file mode 100644 index 000000000..12e68ed7c --- /dev/null +++ b/tests/vllm/entrypoints/openai/reasoning_parsers/test_kolibri1.cpp @@ -0,0 +1,182 @@ +// Gate for the kolibri1 reasoning parser: the plugin-reference behavior of +// aleph-alpha-inference @ 049a6a7bd240 (aleph_alpha_inference/reasoning.py), +// the model-author vLLM plugin that IS the serving oracle for Kolibri +// (.agents/oracles/aleph-alpha-inference.md). Upstream vLLM implements +// nothing for this architecture; every expected value below comes from the +// plugin source or from the switch its chat template encodes. +// +// Mirrored behavior (reasoning.py): +// - thinking_enabled (:36-48): a reasoning_effort that is not None wins and +// only "none" disables thinking; without it, only a literal +// enable_thinking: false disables thinking (the template tests +// `enable_thinking is false`). +// - Kolibri1Parser (:51): the Qwen3 grammar with the starting state derived +// the way the template renders the generation prompt — thinking on leaves +// `<|im_start|>assistant\n` open and the output starts INSIDE ; +// thinking off renders the closed `\n\n\n\n` block and the +// whole output is content (qwen3.py:247 passthrough). +// - Kolibri1ParserReasoningAdapter (:61): the engine-backed reasoning face +// over that parser; the request's chat_template_kwargs reach the split. +// +// Harness: the same TEXT-ONLY seam as test_qwen3.cpp (reasoning_test_utils.h); +// deltas isolate the think markers the way a detokenizer surfaces special +// tokens. The chat_template_kwargs arrive on the ChatCompletionRequest — the +// field the OpenAI protocol already carries (protocol.py:341). +#include + +#include + +#include "reasoning_test_utils.h" +#include "vllm/entrypoints/openai/protocol.h" +#include "vllm/entrypoints/openai/reasoning_parsers/abstract.h" + +using namespace vllm::entrypoints::openai; +using vllm::entrypoints::openai::reasoning_test::Extracted; +using vllm::entrypoints::openai::reasoning_test::RunStreaming; + +namespace { + +// One assistant turn that thinks, then answers, then calls a tool. +const std::vector kThinkToolAnswerDeltas = { + "", "Let me check.", "", "\n\nChecking. ", + "", "\n{\"name\": \"get_weather\", \"arguments\": " + "{\"city\": \"Berlin\"}}\n", + "", +}; + +ChatCompletionRequest RequestWithKwargs(nlohmann::ordered_json kwargs) { + ChatCompletionRequest request; + request.chat_template_kwargs = std::move(kwargs); + return request; +} + +} // namespace + +TEST_CASE("kolibri1 reasoning parser is registered") { + CHECK(get_reasoning_parser("kolibri1") != nullptr); +} + +TEST_CASE("kolibri1: default kwargs leave thinking ON") { + // reasoning.py:44-48 — no reasoning_effort, no enable_thinking: the + // template stops at `<|im_start|>assistant\n` and the parser starts inside + // the think block (initial state REASONING). + auto parser = get_reasoning_parser("kolibri1"); + REQUIRE(parser != nullptr); + const Extracted out = RunStreaming(*parser, kThinkToolAnswerDeltas); + CHECK(out.reasoning.value_or("") == "Let me check."); + // The reasoning face runs under _skip_tool_parsing (adapters.py:51): the + // tool body stays verbatim CONTENT for the tool parser to handle later. + CHECK(out.content.value_or("") == + "\n\nChecking. \n{\"name\": \"get_weather\", " + "\"arguments\": {\"city\": \"Berlin\"}}\n"); +} + +TEST_CASE("kolibri1: enable_thinking false starts in CONTENT") { + // The template renders the closed empty block; the whole output is content, + // tool markers included (qwen3.py:247 passthrough). + auto parser = get_reasoning_parser("kolibri1"); + REQUIRE(parser != nullptr); + const ChatCompletionRequest request = RequestWithKwargs( + nlohmann::ordered_json{{"enable_thinking", false}}); + std::string joined; + for (const auto& d : kThinkToolAnswerDeltas) joined += d; + const ExtractedReasoning er = parser->extract_reasoning(joined, request); + CHECK_FALSE(er.reasoning.has_value()); + CHECK(er.content.value_or("") == joined); +} + +TEST_CASE("kolibri1: reasoning_effort drives the switch like the template") { + struct Case { + const char* name; + nlohmann::ordered_json kwargs; + bool thinking; + }; + const Case cases[] = { + {"effort none", {{"reasoning_effort", "none"}}, false}, + {"effort low", {{"reasoning_effort", "low"}}, true}, + {"effort medium", {{"reasoning_effort", "medium"}}, true}, + {"effort high", {{"reasoning_effort", "high"}}, true}, + {"effort minimal", {{"reasoning_effort", "minimal"}}, true}, + {"effort xhigh", {{"reasoning_effort", "xhigh"}}, true}, + {"effort max", {{"reasoning_effort", "max"}}, true}, + // reasoning.py:46 — effort is not None, so it WINS over the explicit + // enable_thinking:false (the stock qwen3 parser goes wrong here). + {"effort low beats enable_thinking false", + {{"reasoning_effort", "low"}, {"enable_thinking", false}}, true}, + {"effort none beats enable_thinking true", + {{"reasoning_effort", "none"}, {"enable_thinking", true}}, false}, + }; + for (const Case& c : cases) { + CAPTURE(c.name); + auto parser = get_reasoning_parser("kolibri1"); + REQUIRE(parser != nullptr); + const ChatCompletionRequest request = RequestWithKwargs(c.kwargs); + std::string joined; + for (const auto& d : kThinkToolAnswerDeltas) joined += d; + const ExtractedReasoning er = parser->extract_reasoning(joined, request); + if (c.thinking) { + CHECK(er.reasoning.value_or("") == "Let me check."); + CHECK(er.content.value_or("") == + "\n\nChecking. \n{\"name\": \"get_weather\", " + "\"arguments\": {\"city\": \"Berlin\"}}\n"); + } else { + CHECK_FALSE(er.reasoning.has_value()); + CHECK(er.content.value_or("") == joined); + } + } +} + +TEST_CASE("kolibri1: streaming matches the non-streaming split (thinking on)") { + auto parser = get_reasoning_parser("kolibri1"); + REQUIRE(parser != nullptr); + const ChatCompletionRequest request = + RequestWithKwargs(nlohmann::ordered_json{{"reasoning_effort", "low"}}); + Extracted acc; + std::string previous; + for (const auto& delta : kThinkToolAnswerDeltas) { + const std::string current = previous + delta; + const std::optional dm = + parser->extract_reasoning_streaming(previous, current, delta, request); + if (!dm.has_value()) { + previous = current; + continue; + } + if (dm->reasoning.has_value()) { + acc.reasoning = acc.reasoning.value_or("") + *dm->reasoning; + } + if (dm->content.has_value()) { + acc.content = acc.content.value_or("") + *dm->content; + } + previous = current; + } + CHECK(acc.reasoning.value_or("") == "Let me check."); + CHECK(acc.content.value_or("") == + "\n\nChecking. \n{\"name\": \"get_weather\", \"arguments\": " + "{\"city\": \"Berlin\"}}\n"); +} + +TEST_CASE("kolibri1: reasoning_effort null behaves like absent") { + // reasoning.py:45-47 — `effort is not None` gates the override; an explicit + // JSON null IS None, so enable_thinking decides. + auto parser = get_reasoning_parser("kolibri1"); + REQUIRE(parser != nullptr); + const ChatCompletionRequest request = RequestWithKwargs( + nlohmann::ordered_json{{"reasoning_effort", nullptr}, + {"enable_thinking", false}}); + std::string joined; + for (const auto& d : kThinkToolAnswerDeltas) joined += d; + const ExtractedReasoning er = parser->extract_reasoning(joined, request); + CHECK_FALSE(er.reasoning.has_value()); + CHECK(er.content.value_or("") == joined); +} + +TEST_CASE("kolibri1: streaming without kwargs defaults to thinking on") { + // Thinking on starts the engine in REASONING, so a marker-less stream is + // reasoning end to end (the template guarantees the model opens + // itself when the generation prompt stops at `<|im_start|>assistant\n`). + auto parser = get_reasoning_parser("kolibri1"); + REQUIRE(parser != nullptr); + const Extracted out = RunStreaming(*parser, {"Just answer."}); + CHECK(out.reasoning.value_or("") == "Just answer."); + CHECK_FALSE(out.content.has_value()); +} diff --git a/tests/vllm/entrypoints/openai/tool_parsers/test_detect.cpp b/tests/vllm/entrypoints/openai/tool_parsers/test_detect.cpp index fa2a1682b..89006d6cf 100644 --- a/tests/vllm/entrypoints/openai/tool_parsers/test_detect.cpp +++ b/tests/vllm/entrypoints/openai/tool_parsers/test_detect.cpp @@ -219,7 +219,10 @@ TEST_CASE("Registry: every enumerated tool-parser name resolves") { // llama3_json/llama4_json, qwen3_coder/qwen3_xml/mimo, glm45/glm47). // 2026-08-10 (MODEL-MUSE-GLIMMER-W7): 40 -> 41, adding "muse_glimmer". // 2026-08-13 (TOOLS-PARSER-BREADTH W1, #608): 41 -> 42, adding "inkling". - CHECK(names.size() == 42); + // 2026-10-08 (MODEL-TEXT-kolibri-1 serving completion): 42 -> 43, adding + // "kolibri1" — an ALIAS to the hermes family, per the model-author plugin's + // registration (aleph-alpha-inference __init__.py:50-54). + CHECK(names.size() == 43); // Every name the marker table can emit must itself be a registered name. std::size_t marker_count = 0; const ToolParserMarker* markers = ToolParserMarkerTable(&marker_count); diff --git a/tests/vllm/entrypoints/openai/tool_parsers/test_kolibri1.cpp b/tests/vllm/entrypoints/openai/tool_parsers/test_kolibri1.cpp new file mode 100644 index 000000000..55af6708e --- /dev/null +++ b/tests/vllm/entrypoints/openai/tool_parsers/test_kolibri1.cpp @@ -0,0 +1,115 @@ +// Gate for the kolibri1 tool parser alias. The serving oracle +// (aleph-alpha-inference @ 049a6a7bd240) registers the kolibri1 tool-call +// parser as vLLM's Hermes parser verbatim: +// aleph_alpha_inference/__init__.py:50-54 +// ToolParserManager.register_lazy_module( +// name="kolibri1", +// module_path="vllm.tool_parsers.hermes_tool_parser", +// class_name="Hermes2ProToolParser") +// with the comment "Kolibri 1 currently shares the Hermes +// `...` format." So the local kolibri1 name MUST +// resolve to the same HermesToolParser behavior, byte for byte — the alias is +// the mirror of the plugin's registration, not a new dialect. +#include + +#include +#include +#include + +#include "vllm/entrypoints/openai/protocol.h" +#include "vllm/entrypoints/openai/tool_parsers/abstract.h" +#include "vllm/entrypoints/openai/tool_parsers/detect.h" + +using vllm::entrypoints::openai::ChatCompletionRequest; +using vllm::entrypoints::openai::get_tool_parser; +using vllm::entrypoints::openai::ResolveToolParserName; + +namespace { + +// The exact shape the Kolibri chat template instructs the model to emit +// (tokenizer_config.json chat_template, assistant tool_calls branch: the +// `\n{"name": ..., "arguments": ...}\n` wrapper). +const std::string kKolibriToolCall = + "\n{\"name\": \"get_weather\", \"arguments\": " + "{\"city\": \"Berlin\"}}\n"; + +} // namespace + +TEST_CASE("kolibri1 tool parser resolves as a registry name") { + CHECK(get_tool_parser("kolibri1") != nullptr); + CHECK(ResolveToolParserName("kolibri1", "") == "kolibri1"); +} + +TEST_CASE("kolibri1 parses the Hermes format like hermes") { + auto kolibri = get_tool_parser("kolibri1"); + auto hermes = get_tool_parser("hermes"); + REQUIRE(kolibri != nullptr); + REQUIRE(hermes != nullptr); + + const ChatCompletionRequest request; + const auto k = kolibri->extract_tool_calls(kKolibriToolCall, request); + const auto h = hermes->extract_tool_calls(kKolibriToolCall, request); + REQUIRE(k.tool_calls.size() == 1); + REQUIRE(h.tool_calls.size() == 1); + CHECK(k.tool_calls[0].function.name == h.tool_calls[0].function.name); + CHECK(k.tool_calls[0].function.name == "get_weather"); + CHECK(k.tool_calls[0].function.arguments == + h.tool_calls[0].function.arguments); + CHECK(k.content == h.content); +} + +TEST_CASE("kolibri1 streams the Hermes format like hermes") { + auto kolibri = get_tool_parser("kolibri1"); + auto hermes = get_tool_parser("hermes"); + REQUIRE(kolibri != nullptr); + REQUIRE(hermes != nullptr); + + // Feed the canonical Hermes delta cadence (the one the hermes gate drives + // in test_tool_parsers.cpp): wrapper tokens atomic, body one fragment. + const std::vector deltas = { + "", "{\"name\": \"get_", "weather\", ", + "\"arguments\": {\"ci", "ty\": \"Ber", "lin\"}}", ""}; + const ChatCompletionRequest request; + vllm::entrypoints::openai::DeltaMessage k; + vllm::entrypoints::openai::DeltaMessage h; + k.tool_calls = std::vector{}; + h.tool_calls = std::vector{}; + std::string previous; + for (const auto& delta : deltas) { + const std::string current = previous + delta; + if (auto dm = kolibri->extract_tool_calls_streaming(previous, current, + delta, request)) { + if (dm->tool_calls.has_value()) { + k.tool_calls->insert(k.tool_calls->end(), dm->tool_calls->begin(), dm->tool_calls->end()); + } + } + if (auto dm = hermes->extract_tool_calls_streaming(previous, current, + delta, request)) { + if (dm->tool_calls.has_value()) { + h.tool_calls->insert(h.tool_calls->end(), dm->tool_calls->begin(), dm->tool_calls->end()); + } + } + previous = current; + } + // The streamed shape is name-first then argument DIFFS (the hermes gate + // pins this), so compare like the hermes gate does: one name, arguments + // concatenating to the complete object. + REQUIRE(k.tool_calls.has_value()); + REQUIRE(h.tool_calls.has_value()); + auto collect = [](std::vector& v) { + std::optional name; + std::string args; + for (const auto& tc : v) { + if (tc.function.name.has_value()) name = tc.function.name; + if (tc.function.arguments.has_value()) args += *tc.function.arguments; + } + return std::pair, std::string>{name, args}; + }; + const auto [kname, kargs] = collect(*k.tool_calls); + const auto [hname, hargs] = collect(*h.tool_calls); + REQUIRE(kname.has_value()); + CHECK(*kname == "get_weather"); + CHECK(*kname == hname); + CHECK(kargs == hargs); + CHECK(kargs == "{\"city\": \"Berlin\"}"); +} \ No newline at end of file diff --git a/tests/vllm/entrypoints/test_kolibri1_chat_template.cpp b/tests/vllm/entrypoints/test_kolibri1_chat_template.cpp new file mode 100644 index 000000000..5a9f0ef97 --- /dev/null +++ b/tests/vllm/entrypoints/test_kolibri1_chat_template.cpp @@ -0,0 +1,137 @@ +// Gate for the Kolibri-1 chat template as served. The template is NOT code +// this repo ships: it rides in the checkpoint's tokenizer_config.json +// (chat_template key) and reaches the renderer through +// LoadChatTemplateFromConfig -> apply_chat_template (the vendored minja +// engine). The serving oracle is the model-author plugin +// (aleph-alpha-inference @ 049a6a7bd240): it ships no template of its own — +// its reasoning.py:3-26 documents the exact switch the checkpoint template +// encodes (reasoning_effort "none" disables thinking; else a literal +// enable_thinking false does; thinking on stops the generation prompt at +// `<|im_start|>assistant\n`, thinking off renders the closed +// `\n\n\n\n` block). The reference outputs are CPython jinja2 +// renderings of the SAME template text under transformers' whitespace policy, +// captured by tests/fixtures/gen-kolibri1-chat-template-references.py (see +// the fixture's provenance field). +// +// What this gate proves: minja renders the checkpoint template byte-identical +// to the reference for every scenario the plugin's switch distinguishes, and +// the detection seams pick the kolibri1 parsers off that template text. +#include + +#include +#include +#include +#include +#include +#include + +#include "vllm/entrypoints/chat_template.h" +#include "vllm/entrypoints/openai/protocol.h" +#include "vllm/entrypoints/openai/reasoning_parsers/detect.h" +#include "vllm/entrypoints/openai/tool_parsers/detect.h" + +using namespace vllm::entrypoints; +using namespace vllm::entrypoints::openai; +using vllm::entrypoints::openai::ChatCompletionToolsParam; + +namespace { + +const nlohmann::ordered_json& References() { + static const nlohmann::ordered_json refs = [] { + const char* dir = std::getenv("KOLIBRI1_TEMPLATE_FIXTURE_DIR"); +#ifdef KOLIBRI1_TEMPLATE_FIXTURE_DIR + if (dir == nullptr) dir = KOLIBRI1_TEMPLATE_FIXTURE_DIR; +#endif + REQUIRE(dir != nullptr); + std::ifstream in(std::string(dir) + + "/kolibri1_chat_template_references.json"); + REQUIRE(in.good()); + return nlohmann::ordered_json::parse(in); + }(); + return refs; +} + +std::vector ToMessages(const nlohmann::json& arr) { + std::vector out; + for (const auto& m : arr) out.push_back(m.get()); + return out; +} + +std::vector ToTools(const nlohmann::json& tools) { + std::vector out; + for (const auto& t : tools) { + out.push_back(t.get()); + } + return out; +} + +// The kolibri1 template tell-tale: the "Reasoning is disabled" sentence the +// template embeds as its no-reasoning system sentence. No other template in +// the marker tables contains it. Mirrors how the plugin selects its own +// parsers: by the kolibri1 registration name, which detection must resolve to. +const char* kKolibriMarker = + "Reasoning is disabled. Proceed straight to answering"; + +} // namespace + +TEST_CASE("kolibri1: minja renders the checkpoint template like jinja2") { + for (const auto& c : References().at("cases")) { + CAPTURE(c.at("name").get()); + const std::string rendered = apply_chat_template( + LoadChatTemplateFromConfig( + "/mnt/models/Aleph-Alpha/Kolibri-1/tokenizer_config.json"), + ToMessages(c.at("messages")), + c.at("add_generation_prompt").get(), "", "", + c.contains("tools") && !c.at("tools").is_null() + ? ToTools(c.at("tools")) + : std::vector{}, + c.at("chat_template_kwargs")); + CHECK(rendered == c.at("expected").get()); + } +} + +TEST_CASE("kolibri1: the template's thinking switch matches the plugin's") { + // The closed-block vs open-prompt boundary is the contract between the + // template and the kolibri1 reasoning parser (reasoning.py:3-26). Asserted + // on the reference strings so the parser switch cannot drift from what the + // template actually renders. + const auto& cases = References().at("cases"); + auto find = [&](const std::string& name) { + for (const auto& c : cases) { + if (c.at("name") == name) return c; + } + throw std::runtime_error("missing case " + name); + }; + const std::string kClosed = "<|im_start|>assistant\n\n\n\n\n"; + for (const char* off : + {"enable_thinking_false", "reasoning_effort_none", + "effort_none_overrides_enable_thinking_default"}) { + CAPTURE(off); + CHECK(find(off).at("expected").get().find(kClosed) != + std::string::npos); + } + for (const char* on : + {"default_user_only", "reasoning_effort_low", "reasoning_effort_medium", + "reasoning_effort_high", "effort_low_overrides_enable_thinking_false"}) { + CAPTURE(on); + CHECK(find(on).at("expected").get().find(kClosed) == + std::string::npos); + CHECK(find(on).at("expected").get().ends_with( + "<|im_start|>assistant\n")); + } +} + +TEST_CASE("kolibri1: detection rows select kolibri1 off the template text") { + const std::string template_str = LoadChatTemplateFromConfig( + "/mnt/models/Aleph-Alpha/Kolibri-1/tokenizer_config.json"); + REQUIRE(template_str.find(kKolibriMarker) != std::string::npos); + CHECK(DetectReasoningParser(template_str) == "kolibri1"); + CHECK(DetectToolParser(template_str) == "kolibri1"); +} + +TEST_CASE("kolibri1: the marker row precedes and does not shadow ") { + // The template contains "" too; the kolibri1 row must win by ORDER, + // and the marker must not appear inside any other family's template tell. + CHECK(std::string(kKolibriMarker).find("") == std::string::npos); + CHECK(DetectReasoningParser(std::string("....")) == "think_auto"); +} From 4aade352b1b052226c5035423c1cafc38f90475a Mon Sep 17 00:00:00 2001 From: Luca Barbato Date: Thu, 8 Oct 2026 23:20:24 +0200 Subject: [PATCH 03/12] record(kolibri1): close the serving-completion issue with gate evidence The serving unit landed in the previous commit; this records the dated resolution on the canonical issue and refreshes the derived index. FOLLOWING_AGENTS_PROTOCOL Following-Agents-Protocol: true AI-Assisted: true Assisted-by: AGENT:zai/glm-5.3-flash [maki] --- .../ISSUE-LOCAL-01M4EF3R0H2H3FN5NA0NAB0S62.md | 6 +++--- docs/bench-evidence/kolibri1-serve-20261008.md | 18 ++++++++++++------ 2 files changed, 15 insertions(+), 9 deletions(-) diff --git a/.agents/issues/MODEL-TEXT-kolibri-1-kolibri1-for-causal-lm/ISSUE-LOCAL-01M4EF3R0H2H3FN5NA0NAB0S62.md b/.agents/issues/MODEL-TEXT-kolibri-1-kolibri1-for-causal-lm/ISSUE-LOCAL-01M4EF3R0H2H3FN5NA0NAB0S62.md index d47c68860..79722e11d 100644 --- a/.agents/issues/MODEL-TEXT-kolibri-1-kolibri1-for-causal-lm/ISSUE-LOCAL-01M4EF3R0H2H3FN5NA0NAB0S62.md +++ b/.agents/issues/MODEL-TEXT-kolibri-1-kolibri1-for-causal-lm/ISSUE-LOCAL-01M4EF3R0H2H3FN5NA0NAB0S62.md @@ -1,14 +1,14 @@ ID: ISSUE-LOCAL-01M4EF3R0H2H3FN5NA0NAB0S62 Title: kolibri1 serving completion: chat template rendering plus kolibri1 reasoning and tool-call parsers Row: MODEL-TEXT-kolibri-1-kolibri1-for-causal-lm -State: OPEN +State: CLOSED Kind: enhancement GitHub: - Mirror: PENDING Availability: FULL Created: 2026-10-08 Updated: 2026-10-08 -Closed: - +Closed: 2026-10-08 ## Problem @@ -16,4 +16,4 @@ The kolibri1 CPU arm reaches the model forward but not serving: the OpenAI chat ## Resolution -- +Landed on row/kolibri-serve 2026-10-08; see Resolution and docs/bench-evidence/kolibri1-serve-20261008.md. diff --git a/docs/bench-evidence/kolibri1-serve-20261008.md b/docs/bench-evidence/kolibri1-serve-20261008.md index b4cc5a00a..d717134d5 100644 --- a/docs/bench-evidence/kolibri1-serve-20261008.md +++ b/docs/bench-evidence/kolibri1-serve-20261008.md @@ -75,12 +75,18 @@ targets against the unmodified tree): 42->43 (tool) with dated comments. - `test_reasoning_qwen3` 164 PASS, `test_openai_tool_parsers` 64 PASS, `test_chat_template` 196 PASS (no behavior change to other parsers). -- Row battery (ctest, this host): `test_kolibri1`, `test_kolibri1_dequant`, - `test_kolibri1_dequant_cache`, `test_kolibri1_w2`, `test_kolibri1_w3`, - `test_kolibri1_decode_bench`, `test_kolibri1_moe_glue` — all PASS - (test_kolibri1_tt / _b2i / _b2bi PASS in the same run). -- W3 rerun in a verified quiet window (free > 110 GB, no other - test_kolibri1_w3 process), `VLLM_CPP_CPU_THREADS=8`: PASS. +- Row battery (ctest, this host): `test_kolibri1`, `test_kolibri1_tt`, + `test_kolibri1_tt_b2i`, `test_kolibri1_tt_b2bi`, `test_kolibri1_dequant`, + `test_kolibri1_dequant_cache`, `test_kolibri1_moe_glue`, `test_kolibri1_w2`, + `test_kolibri1_decode_bench` — all PASS. +- `test_kolibri1_w3` PASS in a verified quiet window: 2026-10-09 01:07 CEST, + 211 GB free at start, `ps`-verified zero other `test_kolibri1` processes, + `VLLM_CPP_CPU_THREADS=8`, 539.95 s. Two earlier attempts were ABORTED BY + CONTENTION, not by the change: the first fired while a peer was building, + the second was OOM-killed at 75977088 kB anon RSS after a peer's + `test_kolibri1_decode_bench` started mid-run (dmesg + `oom-kill ... task=test_kolibri1_w`). The quiet-window check is what + serializes W3 on this host. Pre-existing, not touched by this change (verified on the same tree): `ctest`-cwd artifacts from `-ffile-prefix-map` (`test_linear_scaling_rope` From 5983c69180a30fb98d880b0c123dc59b70345089 Mon Sep 17 00:00:00 2001 From: Luca Barbato Date: Fri, 9 Oct 2026 07:19:36 +0200 Subject: [PATCH 04/12] fix(entrypoints): port the pinned transformers tojson; keep tool-schema order The PR #3422 review falsified the global tojson override that sorted object keys for every model. Measured first, against the pinned transformers 5.14.1 renderer (.agents/oracles/transformers.md) through render_jinja_template with a probe over an insertion-ordered {"z": 1, "a": 2}, nested unsorted tool schemas and Unicode/HTML leaves: the pinned renderer's tojson override (utils/chat_template_utils.py:481) is json.dumps with sort_keys=False, ensure_ascii=False and NO HTML escaping, so the serving reference keeps insertion key order and raw UTF-8 -- plain Jinja's builtin (sorted, HTML-escaped) is not the serving behavior, and the old sorted-dump override rendered prompt bytes no serving reference produces. chat_template.cpp replaces the sorted-dump override with a port of the pinned filter's FULL signature (ensure_ascii/indent/separators/ sort_keys, CPython json.dumps semantics; minja's builtin matches the default but rejects three of the four options). BuildTools now mirrors the measured pinned vLLM tool shape ([tool.model_dump() ...] at online_renderer.py:178: fixed field order, description/parameters present as null when absent). FunctionDefinition::parameters becomes nlohmann::ordered_json (nlohmann::json sorts object keys at parse, losing the request document's order that the pinned renderer preserves into the prompt), with an ordered from_json overload and RestoreToolSchemaOrder, which re-reads tools from an order-preserving body parse. Every chat entry point calls it after its regular parse: api_server handle_chat_ completions, the C ABI ParseChatRequest (vllm_chat and vllm_chat_stream), and run_batch. Key-lookup readers (the tool parsers) are behavior-identical; step3.cpp's one pointer type follows the field. Gates: full tree builds (881 targets); test_kolibri1_chat_template 61, test_chat_template 204, test_reasoning_kolibri1 43, test_tool_parser_kolibri1 19, test_reasoning_parser_detect 75, test_tool_parser_detect 361, test_reasoning_qwen3 164, test_openai_tool_parsers 64, test_kolibri1 27/234, test_kolibri1_decode_bench (anchor 109726), test_openai_serving 1365, test_openai_api_server 1517, test_openai_conformance 252, test_parser_engine_assembly 5038, test_openai_run_batch 16, and the parameters-reading tool-parser suites all green; check-tree-compiles 161/161 in scope; check-agent-record OK. FOLLOWING_AGENTS_PROTOCOL Following-Agents-Protocol: true AI-Assisted: true Assisted-by: AGENT:mistral/mistral-large-4 [maki] --- include/vllm/entrypoints/openai/protocol.h | 23 +- src/capi/vllm_c.cpp | 8 +- src/vllm/entrypoints/chat_template.cpp | 234 +++++++++++++++--- src/vllm/entrypoints/openai/api_server.cpp | 5 + src/vllm/entrypoints/openai/protocol.cpp | 43 +++- src/vllm/entrypoints/openai/run_batch.cpp | 4 + .../entrypoints/openai/tool_parsers/step3.cpp | 2 +- 7 files changed, 267 insertions(+), 52 deletions(-) diff --git a/include/vllm/entrypoints/openai/protocol.h b/include/vllm/entrypoints/openai/protocol.h index aba68bbdc..0b0d5ee7e 100644 --- a/include/vllm/entrypoints/openai/protocol.h +++ b/include/vllm/entrypoints/openai/protocol.h @@ -125,7 +125,13 @@ struct ResponseFormat { struct FunctionDefinition { std::string name; std::optional description; - std::optional parameters; + // Order-preserving ON PURPOSE: the pinned renderer serializes tool schemas + // into the prompt with sort_keys=False (transformers 5.14.1 + // utils/chat_template_utils.py:481), so the request document's key order is + // the prompt's key order (pinned vLLM types this field dict[str, Any], + // protocol.py:326, and its json parse preserves document order). + // nlohmann::json would sort the keys away at parse (std::map). + std::optional parameters; }; // Ported from: vllm/entrypoints/openai/chat_completion/protocol.py:165 @@ -583,6 +589,21 @@ void from_json(const nlohmann::json& j, CompletionRequest& r); void from_json(const nlohmann::json& j, ChatMessage& m); void from_json(const nlohmann::json& j, ChatCompletionToolsParam& t); void from_json(const nlohmann::json& j, ChatCompletionRequest& r); +// Order-preserving twin of the ChatCompletionToolsParam overload: a body +// parsed as nlohmann::ordered_json keeps `parameters` in document order, +// which is the order the pinned renderer dumps into the prompt. +void from_json(const nlohmann::ordered_json& j, ChatCompletionToolsParam& t); + +// Re-read a parsed request's `tools` from an order-preserving parse of the +// same body. A body parsed as nlohmann::json sorts object keys (std::map), +// which loses the tool schemas' document order; the pinned renderer +// preserves that order into the prompt bytes (sort_keys=False tojson), so +// every entry point that renders chat prompts calls this after its regular +// parse. No-op when the request carries no tools. Throws whatever the +// ordered parse throws (the regular parse of the same body already +// succeeded by the time callers reach this). +void RestoreToolSchemaOrder(const std::string& request_body, + ChatCompletionRequest& r); void to_json(nlohmann::json& j, const UsageInfo& u); void to_json(nlohmann::json& j, const ErrorInfo& e); diff --git a/src/capi/vllm_c.cpp b/src/capi/vllm_c.cpp index 519b56fc5..a005e9d2a 100644 --- a/src/capi/vllm_c.cpp +++ b/src/capi/vllm_c.cpp @@ -510,7 +510,13 @@ vllm::entrypoints::openai::ChatCompletionRequest ParseChatRequest( e.what()); } try { - return j.get(); + auto request = j.get(); + // Tool schemas keep the request document's key order (the pinned + // renderer dumps them into the prompt with sort_keys=False); the + // nlohmann::json parse above sorts object keys, so re-read `tools` + // order-preserving. + vllm::entrypoints::openai::RestoreToolSchemaOrder(request_json, request); + return request; } catch (const std::exception& e) { throw std::invalid_argument(std::string("invalid chat request: ") + e.what()); diff --git a/src/vllm/entrypoints/chat_template.cpp b/src/vllm/entrypoints/chat_template.cpp index a662da5d8..b7b671ff4 100644 --- a/src/vllm/entrypoints/chat_template.cpp +++ b/src/vllm/entrypoints/chat_template.cpp @@ -124,9 +124,15 @@ nlohmann::ordered_json BuildMessages( } // Rebuild the OpenAI tool JSON objects the request carried, exposed to the -// template as `tools` exactly as transformers' apply_chat_template(tools=...) -// sees them: -// {"type": , "function": {"name": .., "description"?: .., "parameters"?: ..}} +// template as `tools` exactly as the pinned renderer sees them: +// vLLM hands apply_chat_template `[tool.model_dump() for tool in +// request.tools]` (online_renderer.py:178) — pydantic field order +// type/function, name/description/parameters — with `description` and +// `parameters` present as null when the request omitted them (only +// strict/defer_loading are popped by the model serializer). `parameters` +// keeps the request document's key order (ordered_json; the pinned +// renderer dumps it with sort_keys=False, transformers 5.14.1 +// chat_template_utils.py:481). nlohmann::ordered_json BuildTools( const std::vector& tools) { nlohmann::ordered_json arr = nlohmann::ordered_json::array(); @@ -135,13 +141,13 @@ nlohmann::ordered_json BuildTools( fn["name"] = t.function.name; if (t.function.description.has_value()) { fn["description"] = *t.function.description; + } else { + fn["description"] = nullptr; // model_dump keeps absent fields as null } if (t.function.parameters.has_value()) { - // parameters is a plain nlohmann::json (unordered). Round-trip through a - // string to land it in ordered_json without an implicit cross-container - // conversion. - fn["parameters"] = - nlohmann::ordered_json::parse(t.function.parameters->dump()); + fn["parameters"] = *t.function.parameters; + } else { + fn["parameters"] = nullptr; } nlohmann::ordered_json tool = nlohmann::ordered_json::object(); tool["type"] = t.type; @@ -151,6 +157,107 @@ nlohmann::ordered_json BuildTools( return arr; } +// ─── `tojson`: CPython json.dumps semantics over a minja value ───────────── +// The byte-level rules below were MEASURED against the pinned transformers +// 5.14.1 renderer (its tojson override IS json.dumps with the four options; +// see the filter's comment at its installation site). nlohmann's dump already +// matches CPython's string escaping exactly (quote/backslash, \b \t \n \f \r, +// other control characters as lowercase \u00xx, non-ASCII raw or \uXXXX with +// surrogate pairs), so string leaves delegate to it. +std::string JsonDumpsQuote(const std::string& s, bool ensure_ascii) { + return nlohmann::ordered_json(s).dump(/*indent=*/-1, /*indent_char=*/' ', + ensure_ascii); +} + +// json.dumps key coercion: string keys as-is; other primitives to their JSON +// literal as text (json.dumps({1: ..}) -> {"1": ..}, {True: ..} -> +// {"true": ..}, {None: ..} -> {"null": ..}). By value: minja's keys() is +// non-const and a Value copy is a cheap shared_ptr copy. +std::string JsonDumpsKeyText(minja::Value key) { + return key.is_string() ? key.get() : key.dump(); +} + +std::string RepeatStr(const std::string& s, int n) { + std::string out; + for (int i = 0; i < n; ++i) out += s; + return out; +} + +void JsonDumps(minja::Value v, bool ensure_ascii, bool pretty, + const std::string& indent_str, const std::string& item_sep, + const std::string& key_sep, bool sort_keys, int level, + std::string& out) { + if (v.is_null()) { + out += "null"; + return; + } + if (v.is_boolean()) { + out += v.get() ? "true" : "false"; + return; + } + if (v.is_number()) { + // nlohmann's dump matches CPython's float repr for every value a + // JSON-parsed request can carry (measured: 0.5, 100.0, 1e+16, 1e-05, + // -2.75, 0.0001, pi, 1e+100, 5e-324, max-double all agree). Known edge: + // [1e15, 1e16) formats exponential where CPython keeps the full decimal. + out += v.dump(); + return; + } + if (v.is_string()) { + out += JsonDumpsQuote(v.get(), ensure_ascii); + return; + } + // Empty containers stay {} / [] even when pretty (CPython does the same). + if (v.is_array() && v.size() == 0) { + out += "[]"; + return; + } + if (v.is_object() && v.size() == 0) { + out += "{}"; + return; + } + const std::string nl = + pretty ? "\n" + RepeatStr(indent_str, level + 1) : std::string(); + const std::string nl_end = + pretty ? "\n" + RepeatStr(indent_str, level) : std::string(); + if (v.is_array()) { + out += "["; + for (std::size_t i = 0; i < v.size(); ++i) { + if (i) out += item_sep; + out += nl; + JsonDumps(v.at(minja::Value(static_cast(i))), ensure_ascii, + pretty, indent_str, item_sep, key_sep, sort_keys, level + 1, + out); + } + out += nl_end; + out += "]"; + return; + } + if (v.is_object()) { + std::vector keys = v.keys(); + if (sort_keys) { + // Byte order == UTF-8 code-point order == Python's string sort. + std::sort(keys.begin(), keys.end(), + [](const minja::Value& a, const minja::Value& b) { + return JsonDumpsKeyText(a) < JsonDumpsKeyText(b); + }); + } + out += "{"; + for (std::size_t i = 0; i < keys.size(); ++i) { + if (i) out += item_sep; + out += nl; + out += JsonDumpsQuote(JsonDumpsKeyText(keys[i]), ensure_ascii); + out += key_sep; + JsonDumps(v.at(keys[i]), ensure_ascii, pretty, indent_str, item_sep, + key_sep, sort_keys, level + 1, out); + } + out += nl_end; + out += "}"; + return; + } + throw std::runtime_error("tojson: cannot serialize a callable"); +} + // A template parsed ONCE, or the reason it would not parse. The failure is // kept, not thrown, because transformers compiles inside apply_chat_template // and a broken template is therefore a per-request error upstream (api_server @@ -211,47 +318,94 @@ std::string RenderChatTemplate( std::shared_ptr builtins = minja::Context::builtins(); std::shared_ptr context = minja::Context::make(minja::Value(top), builtins); - // tojson with jinja2's DEFAULT policy: CPython Jinja sorts object keys - // (jinja2.defaults.DEFAULT_POLICIES["json.dumps_kwargs"] = - // {"sort_keys": True}), and that is the tojson every transformers / - // vLLM-served template runs under -- the Kolibri tool preamble renders - // `{{ tool | tojson }}` and byte-matches the jinja2 references only with - // the sort. minja's builtin dumps in insertion order, so this child-scope - // global shadows it (same signature: value + optional indent; set() on - // the child cannot touch the shared builtins). + // `tojson` is the PINNED TRANSFORMERS renderer's filter, not Jinja's + // builtin. transformers 5.14.1 (the pin .agents/oracles/transformers.md + // records inside the pinned vLLM environment) installs its own tojson + // over Jinja's default in _compile_jinja_template + // (utils/chat_template_utils.py:481-492): + // + // def tojson(x, ensure_ascii=False, indent=None, separators=None, + // sort_keys=False): + // # We override the built-in tojson filter because Jinja's default + // # filter escapes HTML characters + // return json.dumps(x, ensure_ascii=ensure_ascii, indent=indent, + // separators=separators, sort_keys=sort_keys) + // + // Measured against the pin (2026-10-09; probe over an insertion-ordered + // {"z": 1, "a": 2}, nested unsorted tool schemas, Unicode/HTML leaves, + // through render_jinja_template): the DEFAULT keeps INSERTION key order + // (sort_keys=False), keeps Unicode raw (ensure_ascii=False) and never + // HTML-escapes — Jinja's builtin sorts keys and escapes HTML, which is + // exactly what the transformers override exists to avoid, so the old + // sorted-dump override here rendered prompt bytes no serving reference + // produces. minja's builtin tojson matches the default byte-for-byte + // (ordered_map objects, nlohmann dump: raw UTF-8, no HTML escape, + // Python's separators and indent shape) but accepts only `indent`; this + // child-scope global ports the pinned filter's full signature so every + // option the serving reference accepts renders the same bytes here + // (set() on the child cannot touch the shared builtins). context->set( "tojson", minja::Value::callable( [](const std::shared_ptr&, minja::ArgumentsValue& args) -> minja::Value { - minja::Value value = args.args.at(0); - int64_t indent = -1; + // Positional order after the value is (ensure_ascii, indent, + // separators, sort_keys), as in the pinned signature; keyword + // arguments address the same options by name. + args.expectArgs("tojson", {1, 5}, {0, 4}); for (const auto& [name, v] : args.kwargs) { - if (name == "indent") indent = v.get(); + if (name != "ensure_ascii" && name != "indent" && + name != "separators" && name != "sort_keys") { + throw std::runtime_error( + "tojson() got an unexpected keyword argument '" + name + + "'"); + } } - std::function sort_keys = - [&](minja::Value v) -> minja::Value { - if (v.is_object()) { - std::vector keys; - for (const auto& k : v.keys()) keys.push_back(k.get()); - std::sort(keys.begin(), keys.end()); - minja::Value out = minja::Value::object(); - for (const auto& k : keys) { - out.set(minja::Value(k), sort_keys(v.at(minja::Value(k)))); - } - return out; + const minja::Value value = args.args.at(0); + // Option i: keyword wins, else positional slot i + 1, else the + // default (a null Value reads as Python's None/absent). + auto option = [&args](std::size_t i, const char* name) { + if (args.has_named(name)) return args.get_named(name); + if (args.args.size() > i + 1) return args.args.at(i + 1); + return minja::Value(); + }; + const bool ensure_ascii = option(0, "ensure_ascii").to_bool(); + // indent: None (compact), an int N (newline + N spaces per + // level; N <= 0 is newline + no spaces, measured), or a string + // used verbatim as the per-level prefix. + bool pretty = false; + std::string indent_str; + const minja::Value indent_v = option(1, "indent"); + if (!indent_v.is_null()) { + pretty = true; + if (indent_v.is_string()) { + indent_str = indent_v.get(); + } else { + const int64_t n = indent_v.get(); + if (n > 0) indent_str.assign(static_cast(n), ' '); } - if (v.is_array()) { - minja::Value out = minja::Value::array(); - for (std::size_t i = 0; i < v.size(); ++i) { - out.push_back(sort_keys(v.at(minja::Value(static_cast(i))))); - } - return out; + } + // separators: None -> (", ", ": ") compact, (",", ": ") when + // pretty (CPython's rule); an explicit (item, key) pair wins + // over both. + std::string item_sep = pretty ? "," : ", "; + std::string key_sep = ": "; + const minja::Value seps = option(2, "separators"); + if (!seps.is_null()) { + if (!seps.is_array() || seps.size() != 2 || + !seps.at(minja::Value(int64_t{0})).is_string() || + !seps.at(minja::Value(int64_t{1})).is_string()) { + throw std::runtime_error( + "tojson() separators must be a pair of strings"); } - return v; - }; - return minja::Value(sort_keys(value).dump(indent, - /*to_json=*/true)); + item_sep = seps.at(minja::Value(int64_t{0})).get(); + key_sep = seps.at(minja::Value(int64_t{1})).get(); + } + const bool sort_keys = option(3, "sort_keys").to_bool(); + std::string out; + JsonDumps(value, ensure_ascii, pretty, indent_str, item_sep, + key_sep, sort_keys, /*level=*/0, out); + return minja::Value(out); })); context->set("bos_token", minja::Value(bos_token)); context->set("eos_token", minja::Value(eos_token)); diff --git a/src/vllm/entrypoints/openai/api_server.cpp b/src/vllm/entrypoints/openai/api_server.cpp index c9c3985b2..c48e2685d 100644 --- a/src/vllm/entrypoints/openai/api_server.cpp +++ b/src/vllm/entrypoints/openai/api_server.cpp @@ -393,6 +393,11 @@ ApiServer::DispatchResult ApiServer::handle_chat_completions( return MakeError(400, "BadRequestError", std::string("Invalid request: ") + e.what()); } + // The pinned renderer serializes tool schemas into the prompt with + // sort_keys=False, preserving the request document's key order; the + // nlohmann::json parse above sorts object keys (std::map), so re-read + // `tools` from an order-preserving parse (no-op without tools). + RestoreToolSchemaOrder(request_body, request); // SERVE-REQUEST-LENGTH-GUARD (#1541). The measured quantity is the SUM of the // message texts, because that is what the chat template concatenates into the // one prompt the tokenizer then sees. `content` carries the joined text spans diff --git a/src/vllm/entrypoints/openai/protocol.cpp b/src/vllm/entrypoints/openai/protocol.cpp index 44f215fff..59ba0a919 100644 --- a/src/vllm/entrypoints/openai/protocol.cpp +++ b/src/vllm/entrypoints/openai/protocol.cpp @@ -25,20 +25,22 @@ constexpr int kDefaultTopK = 0; constexpr double kDefaultMinP = 0.0; // Read a JSON value that may be absent or explicit null into an optional. -template -void GetOpt(const nlohmann::json& j, const char* key, std::optional& out) { +// Templated on the json type so the order-preserving (ordered_json) parse +// path can reuse the same readers. +template +void GetOpt(const Json& j, const char* key, std::optional& out) { auto it = j.find(key); if (it != j.end() && !it->is_null()) { - out = it->get(); + out = it->template get(); } } // Read a scalar with a fallback when the key is absent or null. -template -void GetOr(const nlohmann::json& j, const char* key, T& out) { +template +void GetOr(const Json& j, const char* key, T& out) { auto it = j.find(key); if (it != j.end() && !it->is_null()) { - out = it->get(); + out = it->template get(); } } @@ -502,19 +504,42 @@ void from_json(const nlohmann::json& j, ChatMessage& m) { // Ported from: vllm/entrypoints/openai/chat_completion/protocol.py:165 // (ChatCompletionToolsParam) + engine/protocol.py:246 (FunctionDefinition). -void from_json(const nlohmann::json& j, ChatCompletionToolsParam& t) { +// Templated on the json type: the ordered_json instantiation keeps +// `parameters` in the request document's key order (see protocol.h). +template +void FromJsonToolsParam(const Json& j, ChatCompletionToolsParam& t) { GetOr(j, "type", t.type); // defaults to "function". if (auto it = j.find("function"); it != j.end() && it->is_object()) { - const nlohmann::json& fn = *it; + const Json& fn = *it; GetOr(fn, "name", t.function.name); GetOpt(fn, "description", t.function.description); - // `parameters` is the JSON-Schema object (kept as raw json). + // `parameters` is the JSON-Schema object (kept as raw json, order + // preserved for the ordered_json parse path). if (auto p = fn.find("parameters"); p != fn.end() && !p->is_null()) { t.function.parameters = *p; } } } +void from_json(const nlohmann::json& j, ChatCompletionToolsParam& t) { + FromJsonToolsParam(j, t); +} + +void from_json(const nlohmann::ordered_json& j, ChatCompletionToolsParam& t) { + FromJsonToolsParam(j, t); +} + +void RestoreToolSchemaOrder(const std::string& request_body, + ChatCompletionRequest& r) { + if (!r.tools.has_value() || r.tools->empty()) return; + nlohmann::ordered_json ordered_body = + nlohmann::ordered_json::parse(request_body); + if (auto it = ordered_body.find("tools"); + it != ordered_body.end() && it->is_array()) { + r.tools = it->get>(); + } +} + void from_json(const nlohmann::json& j, ChatCompletionRequest& r) { if (auto it = j.find("messages"); it != j.end() && it->is_array()) { r.messages = it->get>(); diff --git a/src/vllm/entrypoints/openai/run_batch.cpp b/src/vllm/entrypoints/openai/run_batch.cpp index 7172c3b43..07103f035 100644 --- a/src/vllm/entrypoints/openai/run_batch.cpp +++ b/src/vllm/entrypoints/openai/run_batch.cpp @@ -104,6 +104,10 @@ BatchRequestOutput RunBatch::DispatchChat(const std::string& custom_id, std::string("Invalid request body: ") + e.what(), "BadRequestError", 400); } + // Tool schemas keep the request document's key order (the pinned renderer + // dumps them into the prompt with sort_keys=False); the nlohmann::json + // parse above sorts object keys, so re-read `tools` order-preserving. + RestoreToolSchemaOrder(body, request); // check_model (chat_completion/serving.py; api_server.cpp:186-190). if (models_ != nullptr && !models_->check_model(request.model)) { diff --git a/src/vllm/entrypoints/openai/tool_parsers/step3.cpp b/src/vllm/entrypoints/openai/tool_parsers/step3.cpp index 7a869e48f..69f82f78c 100644 --- a/src/vllm/entrypoints/openai/tool_parsers/step3.cpp +++ b/src/vllm/entrypoints/openai/tool_parsers/step3.cpp @@ -66,7 +66,7 @@ ojson CastArguments( const std::vector>& params, const ChatCompletionRequest& request) { // Locate the schema properties for func_name (if any). - const nlohmann::json* properties = nullptr; + const nlohmann::ordered_json* properties = nullptr; if (request.tools.has_value()) { for (const ChatCompletionToolsParam& tool : *request.tools) { if (tool.function.name == func_name) { From 7d74abbd7a6261b0c7ff8c476528ab81f46a8b11 Mon Sep 17 00:00:00 2001 From: Luca Barbato Date: Fri, 9 Oct 2026 07:19:44 +0200 Subject: [PATCH 05/12] test(kolibri1): regenerate the chat-template references through the pinned renderer Review blockers P1/P2 on the fixture side. The old generator rendered with plain jinja2 (sorted keys, HTML-escaped -- not the serving reference) and hard-coded its output under /tmp/vllm-kolibri-serve; the test loaded /mnt/models/Aleph-Alpha/Kolibri-1/ tokenizer_config.json in both the rendering and detection cases. The generator now drives the PINNED transformers 5.14.1 renderer (render_jinja_template, the function apply_chat_template delegates to; it asserts transformers==5.14.1 and records the pin, jinja2's version, the template sha256 and the capture path in the provenance), reads the template from the committed fixture config, mirrors pinned vLLM's _postprocess_messages (assistant tool-call arguments parsed to dicts) and model_dump tool shape, and writes in-tree next to itself. The reference fixture grows from 15 to 20 scenarios: the 15 original plus with_tools_unsorted_unicode and with_tools_unsorted_unicode_thinking_off (unsorted nested tool schemas + Unicode/HTML), with_tools_minimal_no_ description (pins "description": null), assistant_tool_call_unsorted_ arguments (the template's second tojson site, tool_call.arguments | tojson, unsorted Unicode/HTML arguments), and unicode_html_user_message. The template input is committed at tests/fixtures/kolibri1-chat- template-tokenizer_config.json (the checkpoint's chat_template, sha256 9ba35d4bd6baa26b66aa75d03a922dfee98b16bb1fa37481b195d247267b0f97 in the fixture's fixture_provenance). The test loads the template and the references from KOLIBRI1_TEMPLATE_FIXTURE_DIR (tests/fixtures at compile time); no /mnt path remains in any load path -- proven by running the test from / with both fixtures copied to /tmp/p2-proof: 61/61 PASS with the model directory absent. Red-first under the corrected reference (unmodified adapter, 20-case fixture): 6 of 20 rendering scenarios failed -- with_tools, with_tools_thinking_off, with_tools_unsorted_unicode, with_tools_unsorted_unicode_thinking_off, with_tools_minimal_no_ description, assistant_tool_call_unsorted_arguments (34/40 assertions). After the adapter fix: 61/61 PASS. FOLLOWING_AGENTS_PROTOCOL Following-Agents-Protocol: true AI-Assisted: true Assisted-by: AGENT:mistral/mistral-large-4 [maki] --- .../gen-kolibri1-chat-template-references.py | 262 +++++++++++++++--- ...libri1-chat-template-tokenizer_config.json | 9 + .../kolibri1_chat_template_references.json | 176 +++++++++++- .../test_kolibri1_chat_template.cpp | 62 +++-- 4 files changed, 447 insertions(+), 62 deletions(-) create mode 100644 tests/fixtures/kolibri1-chat-template-tokenizer_config.json diff --git a/tests/fixtures/gen-kolibri1-chat-template-references.py b/tests/fixtures/gen-kolibri1-chat-template-references.py index bfd61970a..56236fe35 100644 --- a/tests/fixtures/gen-kolibri1-chat-template-references.py +++ b/tests/fixtures/gen-kolibri1-chat-template-references.py @@ -1,30 +1,62 @@ #!/usr/bin/env python3 -"""Reference capture for the Kolibri-1 chat template. - -Renders the chat_template shipped in -/mnt/models/Aleph-Alpha/Kolibri-1/tokenizer_config.json with CPython Jinja2 -under transformers' whitespace policy (trim_blocks=True, lstrip_blocks=True, -keep_trailing_newline=False), which is the same policy the vendored minja -engine mirrors (src/vllm/entrypoints/chat_template.cpp). The output is the -fixture the C++ rendering gate compares against. - -Provenance: template text from the pinned checkpoint tokenizer_config.json; -behavior oracle aleph-alpha-inference @ 049a6a7bd240 (the plugin serves this -checkpoint; reasoning.py:36 thinking_enabled documents the same switch the -template encodes). +"""Reference capture for the Kolibri-1 chat template — through the PINNED +Transformers renderer, not plain Jinja2. + +Renders the chat_template committed at +tests/fixtures/kolibri1-chat-template-tokenizer_config.json (the +chat_template key of /mnt/models/Aleph-Alpha/Kolibri-1/tokenizer_config.json, +checkpoint pin of oracle aleph-alpha-inference 049a6a7bd240) with +transformers' own chat-template path: +transformers.utils.chat_template_utils.render_jinja_template, the function +apply_chat_template delegates to after kwarg resolution. Its +_compile_jinja_template installs the tojson override +(chat_template_utils.py:481, pin transformers 5.14.1): + + def tojson(x, ensure_ascii=False, indent=None, separators=None, + sort_keys=False): + return json.dumps(x, ensure_ascii=ensure_ascii, indent=indent, + separators=separators, sort_keys=sort_keys) + +so the references preserve INSERTION key order, keep Unicode raw +(ensure_ascii=False) and never HTML-escape — plain jinja2's built-in tojson +sorts keys and escapes HTML, which is NOT the serving reference. Measured +against the pin on 2026-10-09 with a probe over an insertion-ordered +{"z": 1, "a": 2}, nested unsorted tool schemas and Unicode/HTML leaves; see +docs/bench-evidence/kolibri1-serve-20261008.md ("Review repair"). + +The output is the fixture the C++ rendering gate compares against + byte-for-byte. Both the input and the output live in this directory, so the +generator runs on a clean checkout with no model directory and no /tmp +destination. + +Run: python3 tests/fixtures/gen-kolibri1-chat-template-references.py +(the script asserts the pinned transformers==5.14.1 and records jinja2's +version in the provenance). """ +import hashlib import json import pathlib +import sys import jinja2 +import transformers +from transformers.utils.chat_template_utils import render_jinja_template + +# The oracle pin .agents/oracles/transformers.md records inside the pinned +# vLLM environment. Refuse to capture against anything else: the references +# are only a reference if they come from the pinned renderer. +PINNED_TRANSFORMERS = "5.14.1" +if transformers.__version__ != PINNED_TRANSFORMERS: + sys.exit( + f"refusing to capture references with transformers " + f"{transformers.__version__}: the pin is {PINNED_TRANSFORMERS} " + f"(.agents/oracles/transformers.md)" + ) -TEMPLATE_PATH = pathlib.Path( - "/mnt/models/Aleph-Alpha/Kolibri-1/tokenizer_config.json" -) -OUT_PATH = pathlib.Path( - "/tmp/vllm-kolibri-serve/tests/fixtures/kolibri1_chat_template_references.json" -) +HERE = pathlib.Path(__file__).resolve().parent +TEMPLATE_CONFIG_PATH = HERE / "kolibri1-chat-template-tokenizer_config.json" +OUT_PATH = HERE / "kolibri1_chat_template_references.json" TOOLS = [ { @@ -41,7 +73,60 @@ } ] +# Deliberately UNSORTED at every level (type before function; name before +# description before parameters; type before properties before required; +# zeta_city before alpha_unit before city) with Unicode and HTML characters +# in the leaves: this is the case that distinguishes the pinned renderer +# (insertion order, raw UTF-8, no HTML escaping) from plain jinja2's +# built-in tojson (sorted keys, \uXXXX escaping, HTML escaped). +TOOLS_UNSORTED = [ + { + "type": "function", + "function": { + "name": "get_weather", + "description": "Wetter für München & Berlin ☕ " + '"quoted" \'apos\'', + "parameters": { + "type": "object", + "properties": { + "zeta_city": {"type": "string", + "description": "Zeta first
"}, + "alpha_unit": {"type": "string", + "enum": ["celsius", "fahrenheit ☀"]}, + "city": {"type": "string", + "description": "line1\nline2 "}, + }, + "required": ["zeta_city", "city"], + }, + }, + } +] + +# A MINIMAL tool: no description. Pinned vLLM renders it with +# `"description": null` (model_dump keeps absent fields), so this scenario +# pins that edge of the pinned renderer's tool shape. +TOOLS_MINIMAL = [ + { + "type": "function", + "function": { + "name": "ping", + "parameters": { + "type": "object", + "properties": { + "zeta": {"type": "string"}, + "alpha": {"type": "integer"}, + }, + "required": ["alpha"], + }, + }, + } +] + USER = {"role": "user", "content": "What is the weather in Berlin?"} +USER_UNICODE = { + "role": "user", + "content": "Wetter in München? Berlin & 'quotes' \"dquotes\" ☕ äöü", +} ASSISTANT_REASONING = { "role": "assistant", "content": "It is sunny.", @@ -52,6 +137,28 @@ "content": "\nBerlin is in Germany.\n\nIt is sunny.", } TOOL_MSG = {"role": "tool", "content": "22C, clear"} +# Historical assistant tool call whose arguments ride as the OpenAI wire +# STRING, unsorted and carrying Unicode/HTML. The generator parses the string +# to a dict before rendering, mirroring pinned vLLM's _postprocess_messages +# (chat_utils.py:2032-2075: json.loads, so the template's +# `tool_call.arguments | tojson` branch sees a structured value and the +# pinned renderer re-dumps it in document order). +ASSISTANT_TOOL_CALL = { + "role": "assistant", + "content": "Let me check the weather.", + "tool_calls": [ + { + "id": "call_1", + "type": "function", + "function": { + "name": "get_weather", + "arguments": "{\"zeta_city\": \"München rocks & more\", " + "\"alpha_unit\": \"celsius ☀\", " + "\"city\": \"Berlin\"}", + }, + } + ], +} SCENARIOS = [ # (name, messages, add_generation_prompt, kwargs, tools) @@ -82,49 +189,120 @@ ("tool_response_turn", [USER, ASSISTANT_REASONING, TOOL_MSG], True, {}, None), ("no_generation_prompt", [USER], False, {}, None), + # Review-repair additions (PR #3422 review P1): the tojson comparison set + # carries unsorted nested tool schemas and Unicode/HTML characters. + ("with_tools_unsorted_unicode", [USER_UNICODE], True, {}, TOOLS_UNSORTED), + ( + "with_tools_unsorted_unicode_thinking_off", + [USER_UNICODE], + True, + {"enable_thinking": False}, + TOOLS_UNSORTED, + ), + ("with_tools_minimal_no_description", [USER], True, {}, TOOLS_MINIMAL), + ( + "assistant_tool_call_unsorted_arguments", + [USER, ASSISTANT_TOOL_CALL, USER], + True, + {}, + None, + ), + ("unicode_html_user_message", [USER_UNICODE], True, {}, None), ] -def render(template, messages, add_generation_prompt, kwargs, tools): - env = jinja2.Environment(trim_blocks=True, lstrip_blocks=True, - keep_trailing_newline=False) - context = { - "messages": messages, - "add_generation_prompt": add_generation_prompt, - "bos_token": "", - "eos_token": "", - "tools": tools if tools else [], +def model_dump_tool(tool): + """Mirror pinned vLLM's `[tool.model_dump() for tool in request.tools]` + (online_renderer.py:178): pydantic field order type/function, + name/description/parameters, with `description` and `parameters` present + as null when the request omitted them (only strict/defer_loading are + popped by the model serializer). `parameters` keeps the request + document's key order.""" + fn = tool.get("function", {}) + return { + "type": tool.get("type", "function"), + "function": { + "name": fn.get("name"), + "description": fn.get("description"), + "parameters": fn.get("parameters"), + }, } - context.update(kwargs) - return env.from_string(template).render(**context) + + +def render(template, messages, add_generation_prompt, kwargs, tools): + # The pinned renderer's own path: render_jinja_template compiles the + # template through _compile_jinja_template (tojson override installed) + # and renders with transformers' whitespace policy + # (trim_blocks=True, lstrip_blocks=True, keep_trailing_newline=False), + # the policy src/vllm/entrypoints/chat_template.cpp mirrors. + rendered, _ = render_jinja_template( + [messages], + tools=[model_dump_tool(t) for t in tools] if tools else None, + chat_template=template, + add_generation_prompt=add_generation_prompt, + # The C++ adapter binds bos_token/eos_token (empty for this gate); + # the template references neither, so binding them is unobservable. + bos_token="", + eos_token="", + **kwargs, + ) + return rendered[0] def main(): - template = json.loads(TEMPLATE_PATH.read_text())["chat_template"] + config = json.loads(TEMPLATE_CONFIG_PATH.read_text()) + template = config["chat_template"] + sha = hashlib.sha256(template.encode()).hexdigest() + recorded = config.get("fixture_provenance", {}).get("chat_template_sha256") + if recorded is not None and recorded != sha: + sys.exit( + f"template sha256 {sha} does not match the fixture's recorded " + f"{recorded}: recapture the fixture config first" + ) cases = [] for name, messages, agp, kwargs, tools in SCENARIOS: + # Mirror pinned vLLM's _postprocess_messages: assistant tool-call + # arguments ride as the wire string; the template must see the parsed + # dict (chat_utils.py:2032-2075). + render_messages = json.loads(json.dumps(messages)) + for m in render_messages: + if m.get("role") == "assistant": + for tc in m.get("tool_calls", []): + args = tc.get("function", {}).get("arguments") + if isinstance(args, str): + tc["function"]["arguments"] = json.loads(args) cases.append({ "name": name, "messages": messages, "add_generation_prompt": agp, "chat_template_kwargs": kwargs, "tools": tools, - "expected": render(template, messages, agp, kwargs, tools), + "expected": render(template, render_messages, agp, kwargs, tools), }) OUT_PATH.write_text(json.dumps({ "provenance": ( - "chat_template from /mnt/models/Aleph-Alpha/Kolibri-1/" - "tokenizer_config.json (checkpoint pin of oracle " - "aleph-alpha-inference 049a6a7bd240); rendered with CPython jinja2 " - "3.1.6 under transformers' whitespace policy " - "(trim_blocks=True, lstrip_blocks=True, keep_trailing_newline=" - "False), the policy src/vllm/entrypoints/chat_template.cpp " - "mirrors; captured by tests/fixtures/" - "gen-kolibri1-chat-template-references.py" + "chat_template from tests/fixtures/" + "kolibri1-chat-template-tokenizer_config.json (sha256 " + sha + + " of the template string; originally the chat_template key of " + "/mnt/models/Aleph-Alpha/Kolibri-1/tokenizer_config.json, " + "checkpoint pin of oracle aleph-alpha-inference 049a6a7bd240); " + "rendered through the PINNED transformers " + + PINNED_TRANSFORMERS + " renderer " + "(transformers.utils.chat_template_utils.render_jinja_template, " + "the function apply_chat_template delegates to; its " + "_compile_jinja_template installs the tojson override at " + "chat_template_utils.py:481 with sort_keys=False and " + "ensure_ascii=False — insertion key order, raw UTF-8, no HTML " + "escaping), with jinja2 " + jinja2.__version__ + "; captured by " + "tests/fixtures/gen-kolibri1-chat-template-references.py, which " + "asserts the transformers pin; the C++ gate compares minja's " + "rendering (src/vllm/entrypoints/chat_template.cpp) " + "byte-for-byte against these references" ), "cases": cases, - }, indent=1) + "\n") - print(f"wrote {OUT_PATH} with {len(cases)} cases") + }, indent=1, ensure_ascii=False) + "\n") + print(f"wrote {OUT_PATH} with {len(cases)} cases " + f"(transformers {transformers.__version__}, jinja2 {jinja2.__version__})") if __name__ == "__main__": diff --git a/tests/fixtures/kolibri1-chat-template-tokenizer_config.json b/tests/fixtures/kolibri1-chat-template-tokenizer_config.json new file mode 100644 index 000000000..7afe4ae07 --- /dev/null +++ b/tests/fixtures/kolibri1-chat-template-tokenizer_config.json @@ -0,0 +1,9 @@ +{ + "chat_template": "{%- set _sv_reasoning_effort = reasoning_effort | default(none) -%}\n{%- set _sv_thinking_disabled = false -%}\n{%- if _sv_reasoning_effort is not none -%}\n {%- set _sv_thinking_disabled = _sv_reasoning_effort == 'none' -%}\n{%- elif enable_thinking is defined and enable_thinking is false -%}\n {%- set _sv_thinking_disabled = true -%}\n{%- endif -%}\n{%- set _sv_no_reasoning_sentence = \"Reasoning is disabled. Proceed straight to answering according to the user's instructions.\" -%}\n{%- set _sv_low_reasoning_sentence = \"Reasoning effort is set to low. Think briefly through only the essential steps in the user's language, then proceed directly to the answer.\" -%}\n{%- set _sv_medium_reasoning_sentence = \"Reasoning effort is set to medium. Think through the task methodically in the user's language, check key assumptions, and provide a well-supported answer.\" -%}\n{%- set _sv_high_reasoning_sentence = \"Reasoning effort is set to high. Think carefully through the task in the user's language, validate key assumptions, consider plausible alternatives, and prioritize correctness and clarity.\" -%}\n{%- set _sv_reasoning_sentence = _sv_high_reasoning_sentence -%}\n{%- if _sv_thinking_disabled -%}\n {%- set _sv_reasoning_sentence = _sv_no_reasoning_sentence -%}\n{%- elif _sv_reasoning_effort in ['minimal', 'low'] -%}\n {%- set _sv_reasoning_sentence = _sv_low_reasoning_sentence -%}\n{%- elif _sv_reasoning_effort == 'medium' -%}\n {%- set _sv_reasoning_sentence = _sv_medium_reasoning_sentence -%}\n{%- elif _sv_reasoning_effort in ['high', 'xhigh', 'max'] or _sv_reasoning_effort is none -%}\n {%- set _sv_reasoning_sentence = _sv_high_reasoning_sentence -%}\n{%- endif -%}\n{%- set _sv_has_system = messages | length > 0 and messages[0].role == 'system' -%}\n{%- if tools or _sv_reasoning_sentence is not none or _sv_has_system %}\n {{- '<|im_start|>system\\n' }}\n {%- if _sv_has_system %}\n {{- messages[0].content }}\n {%- endif %}\n {%- if _sv_reasoning_sentence is not none %}\n {%- if _sv_has_system %}{{- '\\n\\n' }}{%- endif %}\n {{- '# Reasoning effort\\n\\n' + _sv_reasoning_sentence }}\n {%- endif %}\n {%- if tools %}\n {%- if _sv_has_system or _sv_reasoning_sentence is not none %}{{- '\\n\\n' }}{%- endif %}\n {{- \"# Tools\\n\\nYou may call one or more functions to assist with the user query.\\n\\nYou are provided with function signatures within XML tags:\\n\" }}\n {%- for tool in tools %}\n {{- \"\\n\" }}\n {{- tool | tojson }}\n {%- endfor %}\n {{- \"\\n\\n\\nFor each function call, return a json object with function name and arguments within XML tags:\\n\\n{\\\"name\\\": , \\\"arguments\\\": }\\n\" }}\n {%- endif %}\n {{- '<|im_end|>\\n' }}\n{%- endif %}\n{%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}\n{%- for message in messages[::-1] %}\n {%- set index = (messages|length - 1) - loop.index0 %}\n {%- if ns.multi_step_tool and message.role == \"user\" and message.content is string and not(message.content.startswith('') and message.content.endswith('')) %}\n {%- set ns.multi_step_tool = false %}\n {%- set ns.last_query_index = index %}\n {%- endif %}\n{%- endfor %}\n{%- for message in messages %}\n {%- if (message.role == \"user\") or (message.role == \"system\" and not loop.first) %}\n {{- '<|im_start|>' + message.role + '\\n' + message.content + '<|im_end|>' + '\\n' }}\n {%- elif message.role == \"assistant\" %}\n {%- set content = message.content if message.content is string else '' %}\n {%- set reasoning = none %}\n {%- if message.reasoning is defined and message.reasoning is not none %}\n {%- set reasoning = message.reasoning %}\n {%- elif message.reasoning_content is defined and message.reasoning_content is not none %}\n {#- Deprecated vLLM compatibility. Prefer the `reasoning` field. -#}\n {%- set reasoning = message.reasoning_content %}\n {%- elif content is string and '
' in content %}\n {%- set reasoning = content.split('
')[0].rstrip('\\n').split('')[-1].lstrip('\\n') %}\n {%- set content = content.split('')[-1].lstrip('\\n') %}\n {%- endif %}\n {{- '<|im_start|>' + message.role + '\\n' }}\n {%- if (loop.index0 > ns.last_query_index) or (preserve_thinking is defined and preserve_thinking is true) %}\n {{- '\\n' }}\n {%- if reasoning is not none and reasoning | trim %}\n {{- reasoning.strip('\\n') }}\n {%- endif %}\n {{- '\\n\\n\\n' }}\n {%- endif %}\n {{- content.lstrip('\\n') }}\n {%- if message.tool_calls %}\n {%- for tool_call in message.tool_calls %}\n {%- if (loop.first and content) or (not loop.first) %}\n {{- '\\n' }}\n {%- endif %}\n {%- if tool_call.function %}\n {%- set tool_call = tool_call.function %}\n {%- endif %}\n {{- '\\n{\"name\": \"' }}\n {{- tool_call.name }}\n {{- '\", \"arguments\": ' }}\n {%- if tool_call.arguments is string %}\n {{- tool_call.arguments }}\n {%- else %}\n {{- tool_call.arguments | tojson }}\n {%- endif %}\n {{- '}\\n' }}\n {%- endfor %}\n {%- endif %}\n {{- '<|im_end|>\\n' }}\n {%- elif message.role == \"tool\" %}\n {%- if loop.first or (messages[loop.index0 - 1].role != \"tool\") %}\n {{- '<|im_start|>user' }}\n {%- endif %}\n {{- '\\n\\n' }}\n {{- message.content }}\n {{- '\\n' }}\n {%- if loop.last or (messages[loop.index0 + 1].role != \"tool\") %}\n {{- '<|im_end|>\\n' }}\n {%- endif %}\n {%- endif %}\n{%- endfor %}\n{%- if add_generation_prompt %}\n {{- '<|im_start|>assistant\\n' }}\n {%- if _sv_thinking_disabled %}\n {{- '\\n\\n\\n\\n' }}\n {%- endif %}\n{%- endif %}", + "fixture_provenance": { + "source": "/mnt/models/Aleph-Alpha/Kolibri-1/tokenizer_config.json (checkpoint pin of oracle aleph-alpha-inference 049a6a7bd240)", + "chat_template_sha256": "9ba35d4bd6baa26b66aa75d03a922dfee98b16bb1fa37481b195d247267b0f97", + "captured": "2026-10-09", + "note": "Minimal config carrying only the checkpoint's chat_template so the rendering and detection gates run on a clean checkout with no model directory. The sha256 pins the template text byte-exactly." + } +} diff --git a/tests/fixtures/kolibri1_chat_template_references.json b/tests/fixtures/kolibri1_chat_template_references.json index dd1c05b6d..f8bb7d350 100644 --- a/tests/fixtures/kolibri1_chat_template_references.json +++ b/tests/fixtures/kolibri1_chat_template_references.json @@ -1,5 +1,5 @@ { - "provenance": "chat_template from /mnt/models/Aleph-Alpha/Kolibri-1/tokenizer_config.json (checkpoint pin of oracle aleph-alpha-inference 049a6a7bd240); rendered with CPython jinja2 3.1.6 under transformers' whitespace policy (trim_blocks=True, lstrip_blocks=True, keep_trailing_newline=False), the policy src/vllm/entrypoints/chat_template.cpp mirrors; captured by tests/fixtures/gen-kolibri1-chat-template-references.py", + "provenance": "chat_template from tests/fixtures/kolibri1-chat-template-tokenizer_config.json (sha256 9ba35d4bd6baa26b66aa75d03a922dfee98b16bb1fa37481b195d247267b0f97 of the template string; originally the chat_template key of /mnt/models/Aleph-Alpha/Kolibri-1/tokenizer_config.json, checkpoint pin of oracle aleph-alpha-inference 049a6a7bd240); rendered through the PINNED transformers 5.14.1 renderer (transformers.utils.chat_template_utils.render_jinja_template, the function apply_chat_template delegates to; its _compile_jinja_template installs the tojson override at chat_template_utils.py:481 with sort_keys=False and ensure_ascii=False — insertion key order, raw UTF-8, no HTML escaping), with jinja2 3.1.6; captured by tests/fixtures/gen-kolibri1-chat-template-references.py, which asserts the transformers pin; the C++ gate compares minja's rendering (src/vllm/entrypoints/chat_template.cpp) byte-for-byte against these references", "cases": [ { "name": "default_user_only", @@ -151,7 +151,7 @@ } } ], - "expected": "<|im_start|>system\n# Reasoning effort\n\nReasoning effort is set to high. Think carefully through the task in the user's language, validate key assumptions, consider plausible alternatives, and prioritize correctness and clarity.\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n\n{\"function\": {\"description\": \"Get the weather for a city\", \"name\": \"get_weather\", \"parameters\": {\"properties\": {\"city\": {\"type\": \"string\"}}, \"required\": [\"city\"], \"type\": \"object\"}}, \"type\": \"function\"}\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n<|im_start|>user\nWhat is the weather in Berlin?<|im_end|>\n<|im_start|>assistant\n" + "expected": "<|im_start|>system\n# Reasoning effort\n\nReasoning effort is set to high. Think carefully through the task in the user's language, validate key assumptions, consider plausible alternatives, and prioritize correctness and clarity.\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n\n{\"type\": \"function\", \"function\": {\"name\": \"get_weather\", \"description\": \"Get the weather for a city\", \"parameters\": {\"type\": \"object\", \"properties\": {\"city\": {\"type\": \"string\"}}, \"required\": [\"city\"]}}}\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n<|im_start|>user\nWhat is the weather in Berlin?<|im_end|>\n<|im_start|>assistant\n" }, { "name": "with_tools_thinking_off", @@ -185,7 +185,7 @@ } } ], - "expected": "<|im_start|>system\n# Reasoning effort\n\nReasoning is disabled. Proceed straight to answering according to the user's instructions.\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n\n{\"function\": {\"description\": \"Get the weather for a city\", \"name\": \"get_weather\", \"parameters\": {\"properties\": {\"city\": {\"type\": \"string\"}}, \"required\": [\"city\"], \"type\": \"object\"}}, \"type\": \"function\"}\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n<|im_start|>user\nWhat is the weather in Berlin?<|im_end|>\n<|im_start|>assistant\n\n\n\n\n" + "expected": "<|im_start|>system\n# Reasoning effort\n\nReasoning is disabled. Proceed straight to answering according to the user's instructions.\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n\n{\"type\": \"function\", \"function\": {\"name\": \"get_weather\", \"description\": \"Get the weather for a city\", \"parameters\": {\"type\": \"object\", \"properties\": {\"city\": {\"type\": \"string\"}}, \"required\": [\"city\"]}}}\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n<|im_start|>user\nWhat is the weather in Berlin?<|im_end|>\n<|im_start|>assistant\n\n\n\n\n" }, { "name": "system_and_user", @@ -281,6 +281,176 @@ "chat_template_kwargs": {}, "tools": null, "expected": "<|im_start|>system\n# Reasoning effort\n\nReasoning effort is set to high. Think carefully through the task in the user's language, validate key assumptions, consider plausible alternatives, and prioritize correctness and clarity.<|im_end|>\n<|im_start|>user\nWhat is the weather in Berlin?<|im_end|>\n" + }, + { + "name": "with_tools_unsorted_unicode", + "messages": [ + { + "role": "user", + "content": "Wetter in München? Berlin & 'quotes' \"dquotes\" ☕ äöü" + } + ], + "add_generation_prompt": true, + "chat_template_kwargs": {}, + "tools": [ + { + "type": "function", + "function": { + "name": "get_weather", + "description": "Wetter für München & Berlin ☕ \"quoted\" 'apos'", + "parameters": { + "type": "object", + "properties": { + "zeta_city": { + "type": "string", + "description": "Zeta first
" + }, + "alpha_unit": { + "type": "string", + "enum": [ + "celsius", + "fahrenheit ☀" + ] + }, + "city": { + "type": "string", + "description": "line1\nline2 " + } + }, + "required": [ + "zeta_city", + "city" + ] + } + } + } + ], + "expected": "<|im_start|>system\n# Reasoning effort\n\nReasoning effort is set to high. Think carefully through the task in the user's language, validate key assumptions, consider plausible alternatives, and prioritize correctness and clarity.\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n\n{\"type\": \"function\", \"function\": {\"name\": \"get_weather\", \"description\": \"Wetter für München & Berlin ☕ \\\"quoted\\\" 'apos'\", \"parameters\": {\"type\": \"object\", \"properties\": {\"zeta_city\": {\"type\": \"string\", \"description\": \"Zeta first
\"}, \"alpha_unit\": {\"type\": \"string\", \"enum\": [\"celsius\", \"fahrenheit ☀\"]}, \"city\": {\"type\": \"string\", \"description\": \"line1\\nline2 \"}}, \"required\": [\"zeta_city\", \"city\"]}}}\n
\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n<|im_start|>user\nWetter in München? Berlin & 'quotes' \"dquotes\" ☕ äöü<|im_end|>\n<|im_start|>assistant\n" + }, + { + "name": "with_tools_unsorted_unicode_thinking_off", + "messages": [ + { + "role": "user", + "content": "Wetter in München? Berlin & 'quotes' \"dquotes\" ☕ äöü" + } + ], + "add_generation_prompt": true, + "chat_template_kwargs": { + "enable_thinking": false + }, + "tools": [ + { + "type": "function", + "function": { + "name": "get_weather", + "description": "Wetter für München & Berlin ☕ \"quoted\" 'apos'", + "parameters": { + "type": "object", + "properties": { + "zeta_city": { + "type": "string", + "description": "Zeta first
" + }, + "alpha_unit": { + "type": "string", + "enum": [ + "celsius", + "fahrenheit ☀" + ] + }, + "city": { + "type": "string", + "description": "line1\nline2 " + } + }, + "required": [ + "zeta_city", + "city" + ] + } + } + } + ], + "expected": "<|im_start|>system\n# Reasoning effort\n\nReasoning is disabled. Proceed straight to answering according to the user's instructions.\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n\n{\"type\": \"function\", \"function\": {\"name\": \"get_weather\", \"description\": \"Wetter für München & Berlin ☕ \\\"quoted\\\" 'apos'\", \"parameters\": {\"type\": \"object\", \"properties\": {\"zeta_city\": {\"type\": \"string\", \"description\": \"Zeta first
\"}, \"alpha_unit\": {\"type\": \"string\", \"enum\": [\"celsius\", \"fahrenheit ☀\"]}, \"city\": {\"type\": \"string\", \"description\": \"line1\\nline2 \"}}, \"required\": [\"zeta_city\", \"city\"]}}}\n
\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n<|im_start|>user\nWetter in München? Berlin & 'quotes' \"dquotes\" ☕ äöü<|im_end|>\n<|im_start|>assistant\n\n\n\n\n" + }, + { + "name": "with_tools_minimal_no_description", + "messages": [ + { + "role": "user", + "content": "What is the weather in Berlin?" + } + ], + "add_generation_prompt": true, + "chat_template_kwargs": {}, + "tools": [ + { + "type": "function", + "function": { + "name": "ping", + "parameters": { + "type": "object", + "properties": { + "zeta": { + "type": "string" + }, + "alpha": { + "type": "integer" + } + }, + "required": [ + "alpha" + ] + } + } + } + ], + "expected": "<|im_start|>system\n# Reasoning effort\n\nReasoning effort is set to high. Think carefully through the task in the user's language, validate key assumptions, consider plausible alternatives, and prioritize correctness and clarity.\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n\n{\"type\": \"function\", \"function\": {\"name\": \"ping\", \"description\": null, \"parameters\": {\"type\": \"object\", \"properties\": {\"zeta\": {\"type\": \"string\"}, \"alpha\": {\"type\": \"integer\"}}, \"required\": [\"alpha\"]}}}\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n<|im_start|>user\nWhat is the weather in Berlin?<|im_end|>\n<|im_start|>assistant\n" + }, + { + "name": "assistant_tool_call_unsorted_arguments", + "messages": [ + { + "role": "user", + "content": "What is the weather in Berlin?" + }, + { + "role": "assistant", + "content": "Let me check the weather.", + "tool_calls": [ + { + "id": "call_1", + "type": "function", + "function": { + "name": "get_weather", + "arguments": "{\"zeta_city\": \"München rocks & more\", \"alpha_unit\": \"celsius ☀\", \"city\": \"Berlin\"}" + } + } + ] + }, + { + "role": "user", + "content": "What is the weather in Berlin?" + } + ], + "add_generation_prompt": true, + "chat_template_kwargs": {}, + "tools": null, + "expected": "<|im_start|>system\n# Reasoning effort\n\nReasoning effort is set to high. Think carefully through the task in the user's language, validate key assumptions, consider plausible alternatives, and prioritize correctness and clarity.<|im_end|>\n<|im_start|>user\nWhat is the weather in Berlin?<|im_end|>\n<|im_start|>assistant\nLet me check the weather.\n\n{\"name\": \"get_weather\", \"arguments\": {\"zeta_city\": \"München rocks & more\", \"alpha_unit\": \"celsius ☀\", \"city\": \"Berlin\"}}\n<|im_end|>\n<|im_start|>user\nWhat is the weather in Berlin?<|im_end|>\n<|im_start|>assistant\n" + }, + { + "name": "unicode_html_user_message", + "messages": [ + { + "role": "user", + "content": "Wetter in München? Berlin & 'quotes' \"dquotes\" ☕ äöü" + } + ], + "add_generation_prompt": true, + "chat_template_kwargs": {}, + "tools": null, + "expected": "<|im_start|>system\n# Reasoning effort\n\nReasoning effort is set to high. Think carefully through the task in the user's language, validate key assumptions, consider plausible alternatives, and prioritize correctness and clarity.<|im_end|>\n<|im_start|>user\nWetter in München? Berlin & 'quotes' \"dquotes\" ☕ äöü<|im_end|>\n<|im_start|>assistant\n" } ] } diff --git a/tests/vllm/entrypoints/test_kolibri1_chat_template.cpp b/tests/vllm/entrypoints/test_kolibri1_chat_template.cpp index 5a9f0ef97..650de8c82 100644 --- a/tests/vllm/entrypoints/test_kolibri1_chat_template.cpp +++ b/tests/vllm/entrypoints/test_kolibri1_chat_template.cpp @@ -8,13 +8,25 @@ // encodes (reasoning_effort "none" disables thinking; else a literal // enable_thinking false does; thinking on stops the generation prompt at // `<|im_start|>assistant\n`, thinking off renders the closed -// `\n\n\n\n` block). The reference outputs are CPython jinja2 -// renderings of the SAME template text under transformers' whitespace policy, -// captured by tests/fixtures/gen-kolibri1-chat-template-references.py (see -// the fixture's provenance field). +// `\n\n\n\n` block). +// +// The template INPUT is committed at tests/fixtures/ +// kolibri1-chat-template-tokenizer_config.json (the checkpoint's +// chat_template, sha256-pinned in the fixture's provenance), so this gate +// runs on a clean checkout with no model directory. The reference OUTPUTS +// are renderings of that template through the PINNED transformers 5.14.1 +// renderer (transformers.utils.chat_template_utils.render_jinja_template — +// the function apply_chat_template delegates to, whose _compile_jinja_template +// installs the tojson override at chat_template_utils.py:481 with +// sort_keys=False / ensure_ascii=False), captured by tests/fixtures/ +// gen-kolibri1-chat-template-references.py (see the fixture's provenance +// field). Plain jinja2 is NOT the serving reference: its built-in tojson +// sorts keys and escapes HTML, which the pinned renderer deliberately +// replaces. // // What this gate proves: minja renders the checkpoint template byte-identical -// to the reference for every scenario the plugin's switch distinguishes, and +// to the pinned renderer for every scenario the plugin's switch distinguishes +// (including unsorted nested tool schemas and Unicode/HTML content), and // the detection seams pick the kolibri1 parsers off that template text. #include @@ -36,14 +48,18 @@ using vllm::entrypoints::openai::ChatCompletionToolsParam; namespace { -const nlohmann::ordered_json& References() { - static const nlohmann::ordered_json refs = [] { - const char* dir = std::getenv("KOLIBRI1_TEMPLATE_FIXTURE_DIR"); +const char* FixtureDir() { + const char* dir = std::getenv("KOLIBRI1_TEMPLATE_FIXTURE_DIR"); #ifdef KOLIBRI1_TEMPLATE_FIXTURE_DIR - if (dir == nullptr) dir = KOLIBRI1_TEMPLATE_FIXTURE_DIR; + if (dir == nullptr) dir = KOLIBRI1_TEMPLATE_FIXTURE_DIR; #endif - REQUIRE(dir != nullptr); - std::ifstream in(std::string(dir) + + REQUIRE(dir != nullptr); + return dir; +} + +const nlohmann::ordered_json& References() { + static const nlohmann::ordered_json refs = [] { + std::ifstream in(std::string(FixtureDir()) + "/kolibri1_chat_template_references.json"); REQUIRE(in.good()); return nlohmann::ordered_json::parse(in); @@ -51,13 +67,26 @@ const nlohmann::ordered_json& References() { return refs; } +// The checkpoint's chat_template, loaded from the COMMITTED fixture config +// (no model directory, no /mnt path): the sha256 in the fixture's provenance +// pins the template text. +std::string TemplateFromFixture() { + return LoadChatTemplateFromConfig( + std::string(FixtureDir()) + + "/kolibri1-chat-template-tokenizer_config.json"); +} + std::vector ToMessages(const nlohmann::json& arr) { std::vector out; for (const auto& m : arr) out.push_back(m.get()); return out; } -std::vector ToTools(const nlohmann::json& tools) { +// Order-preserving on purpose: the pinned renderer dumps tool schemas into +// the prompt with sort_keys=False, so the fixture document's key order is +// the reference's key order (a nlohmann::json parameter would sort it away). +std::vector ToTools( + const nlohmann::ordered_json& tools) { std::vector out; for (const auto& t : tools) { out.push_back(t.get()); @@ -74,12 +103,12 @@ const char* kKolibriMarker = } // namespace -TEST_CASE("kolibri1: minja renders the checkpoint template like jinja2") { +TEST_CASE("kolibri1: minja renders the checkpoint template like the pinned " + "transformers renderer") { for (const auto& c : References().at("cases")) { CAPTURE(c.at("name").get()); const std::string rendered = apply_chat_template( - LoadChatTemplateFromConfig( - "/mnt/models/Aleph-Alpha/Kolibri-1/tokenizer_config.json"), + TemplateFromFixture(), ToMessages(c.at("messages")), c.at("add_generation_prompt").get(), "", "", c.contains("tools") && !c.at("tools").is_null() @@ -122,8 +151,7 @@ TEST_CASE("kolibri1: the template's thinking switch matches the plugin's") { } TEST_CASE("kolibri1: detection rows select kolibri1 off the template text") { - const std::string template_str = LoadChatTemplateFromConfig( - "/mnt/models/Aleph-Alpha/Kolibri-1/tokenizer_config.json"); + const std::string template_str = TemplateFromFixture(); REQUIRE(template_str.find(kKolibriMarker) != std::string::npos); CHECK(DetectReasoningParser(template_str) == "kolibri1"); CHECK(DetectToolParser(template_str) == "kolibri1"); From 08771d39b21a353380e0f07ebd59d48e9a69e92f Mon Sep 17 00:00:00 2001 From: Luca Barbato Date: Fri, 9 Oct 2026 07:19:48 +0200 Subject: [PATCH 06/12] test(chat_template): guard the shared tojson seam for non-kolibri templates The tojson fix touches a seam every model's rendering depends on, so the non-kolibri behavior is pinned explicitly (+8 assertions, 204 total): a non-kolibri tool template (the Hermes/Qwen-style tool branch AND the real Qwen3.5 fixture template) renders an insertion-ordered, Unicode/HTML tool byte-identical to the pinned renderer; the four tojson options (indent=2, sort_keys=True, ensure_ascii=True, separators=(',', ':')) are pinned byte-for-byte against the pin; and an already-alphabetical schema renders byte-identical before/after the repair, so the change alters only what the pinned renderer actually differs on. Expected bytes were generated by the pinned transformers 5.14.1 renderer itself (render_jinja_template, 2026-10-09). FOLLOWING_AGENTS_PROTOCOL Following-Agents-Protocol: true AI-Assisted: true Assisted-by: AGENT:mistral/mistral-large-4 [maki] --- tests/vllm/entrypoints/test_chat_template.cpp | 153 ++++++++++++++++++ 1 file changed, 153 insertions(+) diff --git a/tests/vllm/entrypoints/test_chat_template.cpp b/tests/vllm/entrypoints/test_chat_template.cpp index 18f17c87b..361fcf790 100644 --- a/tests/vllm/entrypoints/test_chat_template.cpp +++ b/tests/vllm/entrypoints/test_chat_template.cpp @@ -665,8 +665,161 @@ ChatMessage AssistantToolCall(std::string arguments, std::vector{std::move(call)}; return assistant; } + +// ─── tojson guards: the SHARED seam stays on the pinned renderer ──────────── +// The serving reference for `{{ tool | tojson }}` is the PINNED transformers +// 5.14.1 renderer's tojson override (utils/chat_template_utils.py:481: +// sort_keys=False, ensure_ascii=False, no HTML escaping) — NOT plain jinja2's +// builtin (which sorts keys and escapes HTML). The expected bytes below were +// rendered by the pin itself (transformers 5.14.1 + jinja2 3.1.6 through +// render_jinja_template, 2026-10-09; see docs/bench-evidence/ +// kolibri1-serve-20261008.md "Review repair"), so any drift of this adapter's +// tojson away from the pinned renderer fails here — on NON-kolibri templates +// no kolibri1 row owns. +// An insertion-ordered (NOT alphabetical) tool at every level, with Unicode +// and HTML in the leaves: the case that separates the pinned renderer from +// plain jinja2's built-in tojson. +std::vector UnsortedUnicodeTool() { + ChatCompletionToolsParam t; + t.type = "function"; + t.function.name = "get_weather"; + t.function.description = "Wetter für München & ☕"; + t.function.parameters = nlohmann::ordered_json::parse( + R"({"type":"object","properties":{"zeta":{"type":"string","description":"
München"},"alpha":{"type":"string"}},"required":["zeta","alpha"],"x_f":0.5})"); + return {t}; +} + +// Every level already alphabetical: sorted and insertion order agree, so the +// sorted-dump override the adapter carried BEFORE the review repair and the +// pinned renderer's insertion-order tojson produce the SAME bytes here. +std::vector AlphabeticalTool() { + ChatCompletionToolsParam t; + t.type = "function"; + t.function.name = "get_weather"; + t.function.description = "Get the weather for a city."; + t.function.parameters = nlohmann::ordered_json::parse( + R"({"type":"object","properties":{"city":{"type":"string"}},"required":["city"]})"); + return {t}; +} + +// The pinned renderer's insertion-ordered tojson of UnsortedUnicodeTool(), +// shared by the tests below. +const char* kUnsortedToolJson = + "{\"type\": \"function\", \"function\": {\"name\": \"get_weather\", " + "\"description\": \"Wetter für München & ☕\", \"parameters\": " + "{\"type\": \"object\", \"properties\": {\"zeta\": {\"type\": \"string\", " + "\"description\": \"
München\"}, \"alpha\": {\"type\": \"string\"}}, " + "\"required\": [\"zeta\", \"alpha\"], \"x_f\": 0.5}}}"; } // namespace +TEST_CASE("chat_template: tojson keeps the pinned renderer's insertion order") { + const std::vector msgs = { + ChatMessage{"user", std::string("weather?")}}; + const std::string out = apply_chat_template( + kToolTemplate, msgs, /*add_generation_prompt=*/false, /*bos=*/"", + /*eos=*/"", UnsortedUnicodeTool()); + // Byte-identical to the pinned transformers 5.14.1 renderer's + // `{{ tool | tojson }}`: insertion key order at every level, Unicode raw, + // HTML unescaped. + CHECK(out == + std::string("<|im_start|>system\n# Tools\n") + + kUnsortedToolJson + "<|im_end|>" + + "<|im_start|>user\nweather?<|im_end|>"); +} + +TEST_CASE("chat_template: tojson is byte-stable for already-ordered schemas") { + // The repair must not alter a non-kolibri model's rendering where the + // pinned renderer and the old sorted override already agreed: for this + // alphabetically-ordered tool the bytes are identical before and after. + const std::vector msgs = { + ChatMessage{"user", std::string("weather?")}}; + const std::string out = apply_chat_template( + kToolTemplate, msgs, /*add_generation_prompt=*/false, /*bos=*/"", + /*eos=*/"", AlphabeticalTool()); + CHECK(out == + "<|im_start|>system\n# Tools\n" + "{\"type\": \"function\", \"function\": {\"name\": \"get_weather\", " + "\"description\": \"Get the weather for a city.\", \"parameters\": " + "{\"type\": \"object\", \"properties\": {\"city\": {\"type\": " + "\"string\"}}, \"required\": [\"city\"]}}}" + "<|im_end|>" + "<|im_start|>user\nweather?<|im_end|>"); +} + +TEST_CASE("chat_template: tojson options match the pinned renderer") { + // The four options the pinned tojson accepts (chat_template_utils.py:481), + // each pinned byte-for-byte against the pin. + const auto render = [](const std::string& call) { + return apply_chat_template("{{ tools[0] | " + call + " }}", {}, false, "", + "", UnsortedUnicodeTool()); + }; + // indent=2: newline + 2 spaces per level, (",", ": ") separators. + CHECK(render("tojson(indent=2)") == + "{\n" + " \"type\": \"function\",\n" + " \"function\": {\n" + " \"name\": \"get_weather\",\n" + " \"description\": \"Wetter für München & ☕\",\n" + " \"parameters\": {\n" + " \"type\": \"object\",\n" + " \"properties\": {\n" + " \"zeta\": {\n" + " \"type\": \"string\",\n" + " \"description\": \"
München\"\n" + " },\n" + " \"alpha\": {\n" + " \"type\": \"string\"\n" + " }\n" + " },\n" + " \"required\": [\n" + " \"zeta\",\n" + " \"alpha\"\n" + " ],\n" + " \"x_f\": 0.5\n" + " }\n" + " }\n" + "}"); + // sort_keys=True: recursive key sort (byte order == code-point order). + CHECK(render("tojson(sort_keys=True)") == + "{\"function\": {\"description\": \"Wetter für München & ☕\", " + "\"name\": \"get_weather\", \"parameters\": {\"properties\": {\"alpha\": " + "{\"type\": \"string\"}, \"zeta\": {\"description\": \"
" + "München\", \"type\": \"string\"}}, \"required\": [\"zeta\", " + "\"alpha\"], \"type\": \"object\", \"x_f\": 0.5}}, \"type\": " + "\"function\"}"); + // ensure_ascii=True: non-ASCII escapes as \uXXXX (literal backslash-u in + // the rendered bytes, hence the doubled backslashes here). + CHECK(render("tojson(ensure_ascii=True)") == + "{\"type\": \"function\", \"function\": {\"name\": \"get_weather\", " + "\"description\": \"Wetter f\\u00fcr M\\u00fcnchen & " + "\\u2615\", \"parameters\": {\"type\": \"object\", \"properties\": " + "{\"zeta\": {\"type\": \"string\", \"description\": \"
" + "M\\u00fcnchen\"}, \"alpha\": {\"type\": \"string\"}}, \"required\": " + "[\"zeta\", \"alpha\"], \"x_f\": 0.5}}}"); + // separators: compact (item ",", key ":"). + CHECK(render("tojson(separators=(',', ':'))") == + "{\"type\":\"function\",\"function\":{\"name\":\"get_weather\"," + "\"description\":\"Wetter für München & ☕\",\"parameters\":" + "{\"type\":\"object\",\"properties\":{\"zeta\":{\"type\":\"string\"" + ",\"description\":\"
München\"},\"alpha\":{\"type\":\"string\"}}," + "\"required\":[\"zeta\",\"alpha\"],\"x_f\":0.5}}}"); +} + +TEST_CASE("chat_template: the real Qwen3.5 template's tojson matches the pin") { + // A second, non-kolibri template (the real Qwen3.5 fixture) rendering the + // unsorted tool: the pinned renderer's insertion-ordered, raw-Unicode, + // unescaped-HTML tojson bytes must appear in the prompt. + const std::string tmpl = ReadFixture("qwen35_chat_template.jinja"); + const std::vector msgs = { + ChatMessage{"user", std::string("weather?")}}; + std::string out; + REQUIRE_NOTHROW(out = apply_chat_template(tmpl, msgs, + /*add_generation_prompt=*/true, + /*bos=*/"", /*eos=*/"<|im_end|>", + UnsortedUnicodeTool())); + CHECK(out.find(kUnsortedToolJson) != std::string::npos); +} + TEST_CASE("chat_template: real Qwen3.5 template renders a plain conversation") { const std::string tmpl = ReadFixture("qwen35_chat_template.jinja"); std::string out; From fca4d34444509839b510d70b3d70c9f80e0584c3 Mon Sep 17 00:00:00 2001 From: Luca Barbato Date: Fri, 9 Oct 2026 07:19:56 +0200 Subject: [PATCH 07/12] =?UTF-8?q?record(kolibri1):=20the=20PR=20#3422=20re?= =?UTF-8?q?view=20repair=20=E2=80=94=20evidence,=20issue,=20spec,=20anchor?= =?UTF-8?q?s?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Records the repair of both review blockers on the serving-completion issue ISSUE-LOCAL-01M4EF3R0H2H3FN5NA0NAB0S62 (the tracker forbids reopening via update, so the repair is recorded in the Resolution): docs/bench-evidence/kolibri1-serve-20261008.md gains a "Review repair" section -- what the pinned transformers 5.14.1 renderer actually does (MEASURED: insertion order, raw UTF-8, no HTML escaping, plus the four tojson options and the model_dump tool shape), what the old sorted-dump override did, the red-first evidence (6 of 20 scenarios failed under the corrected reference), the fixture regeneration, the portability fix and the gate numbers -- and marks the two sections that recorded the falsified decision SUPERSEDED. The spec's SERVING COMPLETION paragraph in .agents/specs/kolibri-1-cpu.md is corrected the same way. Three engine-matrix.md api_server.cpp anchors shifted by the +5-line RestoreToolSchemaOrder call are repaired (1354->1359, 1365->1370, 1614->1619); check-agent-record reports ANCHOR-ROT=0. FOLLOWING_AGENTS_PROTOCOL Following-Agents-Protocol: true AI-Assisted: true Assisted-by: AGENT:mistral/mistral-large-4 [maki] --- .agents/engine-matrix.md | 6 +- .../ISSUE-LOCAL-01M4EF3R0H2H3FN5NA0NAB0S62.md | 8 +- .agents/specs/kolibri-1-cpu.md | 46 +++-- .../bench-evidence/kolibri1-serve-20261008.md | 164 ++++++++++++++++++ 4 files changed, 207 insertions(+), 17 deletions(-) diff --git a/.agents/engine-matrix.md b/.agents/engine-matrix.md index d35296848..b56d39e3b 100644 --- a/.agents/engine-matrix.md +++ b/.agents/engine-matrix.md @@ -212,10 +212,10 @@ claims it. |---|---|---|---|---|---|---|---|---| | `SERVE-OAI-BASIC` | Chat/completions endpoints with SSE transport | T0 | `vllm/entrypoints/openai/completion/api_router.py:34`; `vllm/entrypoints/openai/chat_completion/api_router.py:40`; `tests/entrypoints/openai/completion/test_completion.py:50,259` | `src/vllm/entrypoints/openai/api_server.cpp:60,108,183`; `src/vllm/entrypoints/openai/serving_completion.cpp:22,116`; `src/vllm/entrypoints/openai/serving_chat.cpp:231,380` | `tests/vllm/entrypoints/openai/test_api_server.cpp:351,381,403,449,483,617`; `tests/vllm/entrypoints/openai/test_conformance.cpp:469,502,614,641` | `planned: specs/chat-completions-endpoints.md` | `ANCHOR-BACKFILL` | - | | `SERVE-DISCOVERY-HEALTH` | Models, health, and version endpoints. **Unsupported mode:** engine-readiness health checks. `/health` does not call the engine client's health check or return HTTP 503 for an `EngineDeadError`; it always returns an empty HTTP 200 process-liveness response | T0 | `vllm/entrypoints/openai/models/api_router.py:20`; `vllm/entrypoints/serve/instrumentator/health.py:22`; `vllm/entrypoints/serve/instrumentator/basic.py:53`; `tests/entrypoints/openai/models/test_models.py:48` | `src/vllm/entrypoints/openai/api_server.cpp:149,158,167,215`; `src/vllm/entrypoints/openai/serving_models.cpp:22,50` | `tests/vllm/entrypoints/openai/test_api_server.cpp:434`; `tests/vllm/entrypoints/openai/test_conformance.cpp:953,976,990` | `planned: specs/models-health-version.md` | `PARTIAL` | - | -| `SERVE-METRICS` | Prometheus `/metrics` with vLLM names. **LANDED + CPU-GATED 2026-07-27 (`CLAIM-ROADMAP-C8`, NOT pushed):** self-contained Prometheus registry (`PromRegistry`, text-format-0.0.4 exposition: counter `_total`, histogram `_bucket{le}`/`_sum`/`_count`, Info `{labels} 1.0`) + the ALWAYS-ON vLLM metric catalog (`PrometheusStatLogger`) registered 1:1 (names/help/type/buckets, `{model_name,engine}` labels) + `record(SchedulerStats,IterationStats)` + `GET /metrics` opt-in route. Gated by the vLLM scrape spec `EXPECTED_METRICS_V1` (substring presence, RED-first). **LIVE PER-STEP WIRING LANDED 2026-07-27 (`CLAIM-ROADMAP-C8-METRICS-WIRE`, NOT pushed):** the `/metrics` endpoint now serves LIVE values, not the primed schema. `EngineCoreOutputs` carries `scheduler_stats` (filled by new `Scheduler::make_stats()`, `scheduler.py:2399-2436` — running/waiting/kv-usage + the per-step prefix-cache delta stashed by `schedule()`) + a stamped `timestamp`; `OutputProcessor::process_outputs` builds `IterationStats` (token counts, TTFT/ITL samples, finished-request breakdowns via `RequestState` timing — `stats.py:377-475`); the sync `LLMEngine::step()` folds both into the attached logger's `Record()` guarded by outputs>0 (`llm_engine.py:308-329`). Additive + opt-in: null logger ⇒ no `IterationStats`, `process_outputs` byte-identical no-stats path, greedy token stream untouched. **`/metrics` PRODUCTION-SERVING WIRING LANDED + CPU-GATED 2026-08-10 (`CLAIM-SERVE-METRICS-ASYNC`, [#277](https://github.com/mudler/vllm.cpp/issues/277)) — the endpoint the shipped server actually exposes is now LIVE:** the server serves every route from `AsyncLLM`, whose output handler recorded NOTHING, so a real deployment scraped a well-formed catalog whose series never moved (worse than an absent endpoint: it reads as idle). `AsyncLLM::set_stat_logger` mirrors `logger_ref[0]` (`async_llm.py:648-652`) as an atomic pointer; `RunOutputHandler` builds one `IterationStats` per step under a non-null logger (`:664-665`), threads it through `process_outputs` (`:676-678`) and folds it + `scheduler_stats` into `Record()` OUTSIDE the output-processor mutex (`:697-702`). Two enablers the fold needed: `EngineCore::step_with_batch_queue` now stamps `scheduler_stats` + `timestamp` exactly as `step()` does — upstream stamps both in the path BOTH step functions share (`scheduler.py:1938-1951`, `engine/__init__.py:249-251`), and unstamped they gave the depth-2 serving path all-zero gauges and TTFT == `-arrival_time`; and `PrometheusStatLogger` takes a leaf mutex over `Record`/`Expose`/`SetCacheConfigInfo`, since `PromRegistry` is not thread-safe and upstream only gets away with it under the GIL. `server_main.cpp` attaches the one logger to BOTH frontends. Additive + opt-in: no logger ⇒ byte-identical no-stats path. **SPEC-DECODE FAMILIES LANDED 2026-09-05 ([#2770](https://github.com/mudler/vllm.cpp/issues/2770)):** `vllm/v1/spec_decode/metrics.{h,cpp}` ports `SpecDecodingStats` + `SpecDecodingProm` (`vllm/v1/spec_decode/metrics.py:17-49,177-281`); `SchedulerStats::spec_decoding_stats` (`stats.py:206`) is filled by `Scheduler::make_spec_decoding_stats` (`scheduler.py:2714-2731`) at the acceptance block in `update_from_output` that already computed `num_draft_tokens`/`num_accepted`, stashed for `make_stats()` the way the prefix-cache delta is; `PrometheusStatLogger` takes the resolved `num_speculative_tokens` and registers `vllm:spec_decode_num_drafts` / `_num_draft_tokens` / `_num_accepted_tokens` / `_num_accepted_tokens_per_pos{position}` only when it is non-zero (`loggers.py:477-481`), folding the aggregate in `Record` (`loggers.py:1140-1143`); `server_main.cpp` passes k from `LoadedEngine::speculative_config()`. Before this the acceptance rate was readable from `vllm-bench` and the `vllm_engine_spec_acceptance` C ABI but from no HTTP surface. RESIDUAL: the other config-gated families (kv-connector/mm/LoRA); `update_scheduler_stats` (LoRA-only upstream); the chat/completion RESPONSE-BODY timing surface (`SERVE-RESPONSE-METRICS`) | T0/T1 | `vllm/entrypoints/serve/instrumentator/metrics.py:52-82`; `vllm/v1/metrics/loggers.py:480-1060,1100-1257,1284-1305`; `vllm/v1/metrics/stats.py:186-259,377-475`; `vllm/v1/core/sched/scheduler.py:2399-2436`; `vllm/v1/engine/llm_engine.py:308-329`; `vllm/v1/engine/async_llm.py:638-707` (`_run_output_handler`: :648-652,:664-665,:676-678,:697-702); `vllm/v1/core/sched/scheduler.py:1938-1951` + `vllm/v1/engine/__init__.py:249-251` (the stats/timestamp stamp both step paths share); scrape spec `tests/entrypoints/serve/instrumentator/test_metrics.py:182-228` | registry `include/vllm/v1/metrics/prometheus.h`, `src/vllm/v1/metrics/prometheus.cpp:13,189`; catalog+record `include/vllm/v1/metrics/loggers.h`, `src/vllm/v1/metrics/loggers.cpp:10,60,208`; stats structs + `MonotonicSeconds` `include/vllm/v1/metrics/stats.h:56,160,175,194`; `make_stats` `include/vllm/v1/core/sched/scheduler.h`, `src/vllm/v1/core/sched/scheduler.cpp` (+prefix-delta stash in `schedule()`); `scheduler_stats`/`timestamp` on `EngineCoreOutputs` `include/vllm/v1/engine/types.h`, stamped `src/vllm/v1/engine/core.cpp`; IterationStats build `src/vllm/v1/engine/output_processor.cpp` (+RequestState timing `include/vllm/v1/engine/output_processor.h`); step-site `Record` `src/vllm/v1/engine/llm_engine.cpp:220`, setter `include/vllm/v1/engine/llm_engine.h`; ASYNC step-site `Record` + `IterationStats` `src/vllm/v1/engine/async_llm.cpp:538`, setter + atomic `stat_logger_` `include/vllm/v1/engine/async_llm.h`; batch-queue `scheduler_stats`/`timestamp` stamp `src/vllm/v1/engine/core.cpp:219`; recorder mutex `include/vllm/v1/metrics/loggers.h`, `src/vllm/v1/metrics/loggers.cpp:208,270`; async attach `src/vllm/entrypoints/openai/server_main.cpp:930`; endpoint `src/vllm/entrypoints/openai/api_server.cpp:1354` (`handle_metrics`), route `:488` | `tests/vllm/v1/test_prometheus_metrics.cpp` 4/4 (81 assertions: EXPECTED_METRICS_V1 substring gate RED-first, label schema, TYPE lines, bucket schedules, record() folding); **live-wiring behavioural gate `tests/vllm/v1/test_llm_engine.cpp` case 6 (44 assertions, RED-first: 14 flip 0→correct when `Record` disabled) — running/waiting gauges track the batch, prompt/generation counters == exact token counts, request_success counts finished reqs, TTFT/ITL/e2e/TPOT/iteration histograms observe the right sample counts**; endpoint `tests/vllm/entrypoints/openai/test_api_server.cpp:921`; **ASYNC serving-path gate `tests/vllm/v1/test_llm_engine.cpp:1025` "async_llm: live per-step stats populate the Prometheus registry" (asserts case 6's AND case 7's invariants on the `AsyncHarness` stack; RED-first: 19 of them read 0 unwired) + `:1148` no-logger token-stream identity, and the depth-2 batch-queue pair `tests/vllm/v1/test_async_llm.cpp:560,618` (running gauge poll — RED times out; TTFT/e2e `_sum` > 0 — RED negative/zero); CPU `ctest` 366/366**; **spec-decode gate `tests/vllm/entrypoints/openai/test_api_server.cpp` "/metrics exports the spec_decode families a speculating engine actually produced" (a completion driven through `handle_completions` over a drafting runner, scraped back through `handle_metrics`, counters compared for EQUALITY against what the drafter recorded; RED 0 == 8/24/15 with the `make_spec_decoding_stats` call, the `make_stats` republish or the `Record` fold deleted) + "/metrics omits the spec_decode families with no speculative config" (the upstream gate) + `tests/vllm/v1/test_scheduler.cpp` "spec-decoding stats count drafts, draft tokens, accepted tokens and per-position acceptance" (all six `test_schedule_spec_decoding_stats` cases)** | [prometheus-metrics.md](specs/prometheus-metrics.md), [async-metrics.md](specs/async-metrics.md), [spec-decode-metrics.md](specs/spec-decode-metrics.md) | `ANCHOR-BACKFILL` | `CLAIM-SERVE-METRICS-ASYNC` | +| `SERVE-METRICS` | Prometheus `/metrics` with vLLM names. **LANDED + CPU-GATED 2026-07-27 (`CLAIM-ROADMAP-C8`, NOT pushed):** self-contained Prometheus registry (`PromRegistry`, text-format-0.0.4 exposition: counter `_total`, histogram `_bucket{le}`/`_sum`/`_count`, Info `{labels} 1.0`) + the ALWAYS-ON vLLM metric catalog (`PrometheusStatLogger`) registered 1:1 (names/help/type/buckets, `{model_name,engine}` labels) + `record(SchedulerStats,IterationStats)` + `GET /metrics` opt-in route. Gated by the vLLM scrape spec `EXPECTED_METRICS_V1` (substring presence, RED-first). **LIVE PER-STEP WIRING LANDED 2026-07-27 (`CLAIM-ROADMAP-C8-METRICS-WIRE`, NOT pushed):** the `/metrics` endpoint now serves LIVE values, not the primed schema. `EngineCoreOutputs` carries `scheduler_stats` (filled by new `Scheduler::make_stats()`, `scheduler.py:2399-2436` — running/waiting/kv-usage + the per-step prefix-cache delta stashed by `schedule()`) + a stamped `timestamp`; `OutputProcessor::process_outputs` builds `IterationStats` (token counts, TTFT/ITL samples, finished-request breakdowns via `RequestState` timing — `stats.py:377-475`); the sync `LLMEngine::step()` folds both into the attached logger's `Record()` guarded by outputs>0 (`llm_engine.py:308-329`). Additive + opt-in: null logger ⇒ no `IterationStats`, `process_outputs` byte-identical no-stats path, greedy token stream untouched. **`/metrics` PRODUCTION-SERVING WIRING LANDED + CPU-GATED 2026-08-10 (`CLAIM-SERVE-METRICS-ASYNC`, [#277](https://github.com/mudler/vllm.cpp/issues/277)) — the endpoint the shipped server actually exposes is now LIVE:** the server serves every route from `AsyncLLM`, whose output handler recorded NOTHING, so a real deployment scraped a well-formed catalog whose series never moved (worse than an absent endpoint: it reads as idle). `AsyncLLM::set_stat_logger` mirrors `logger_ref[0]` (`async_llm.py:648-652`) as an atomic pointer; `RunOutputHandler` builds one `IterationStats` per step under a non-null logger (`:664-665`), threads it through `process_outputs` (`:676-678`) and folds it + `scheduler_stats` into `Record()` OUTSIDE the output-processor mutex (`:697-702`). Two enablers the fold needed: `EngineCore::step_with_batch_queue` now stamps `scheduler_stats` + `timestamp` exactly as `step()` does — upstream stamps both in the path BOTH step functions share (`scheduler.py:1938-1951`, `engine/__init__.py:249-251`), and unstamped they gave the depth-2 serving path all-zero gauges and TTFT == `-arrival_time`; and `PrometheusStatLogger` takes a leaf mutex over `Record`/`Expose`/`SetCacheConfigInfo`, since `PromRegistry` is not thread-safe and upstream only gets away with it under the GIL. `server_main.cpp` attaches the one logger to BOTH frontends. Additive + opt-in: no logger ⇒ byte-identical no-stats path. **SPEC-DECODE FAMILIES LANDED 2026-09-05 ([#2770](https://github.com/mudler/vllm.cpp/issues/2770)):** `vllm/v1/spec_decode/metrics.{h,cpp}` ports `SpecDecodingStats` + `SpecDecodingProm` (`vllm/v1/spec_decode/metrics.py:17-49,177-281`); `SchedulerStats::spec_decoding_stats` (`stats.py:206`) is filled by `Scheduler::make_spec_decoding_stats` (`scheduler.py:2714-2731`) at the acceptance block in `update_from_output` that already computed `num_draft_tokens`/`num_accepted`, stashed for `make_stats()` the way the prefix-cache delta is; `PrometheusStatLogger` takes the resolved `num_speculative_tokens` and registers `vllm:spec_decode_num_drafts` / `_num_draft_tokens` / `_num_accepted_tokens` / `_num_accepted_tokens_per_pos{position}` only when it is non-zero (`loggers.py:477-481`), folding the aggregate in `Record` (`loggers.py:1140-1143`); `server_main.cpp` passes k from `LoadedEngine::speculative_config()`. Before this the acceptance rate was readable from `vllm-bench` and the `vllm_engine_spec_acceptance` C ABI but from no HTTP surface. RESIDUAL: the other config-gated families (kv-connector/mm/LoRA); `update_scheduler_stats` (LoRA-only upstream); the chat/completion RESPONSE-BODY timing surface (`SERVE-RESPONSE-METRICS`) | T0/T1 | `vllm/entrypoints/serve/instrumentator/metrics.py:52-82`; `vllm/v1/metrics/loggers.py:480-1060,1100-1257,1284-1305`; `vllm/v1/metrics/stats.py:186-259,377-475`; `vllm/v1/core/sched/scheduler.py:2399-2436`; `vllm/v1/engine/llm_engine.py:308-329`; `vllm/v1/engine/async_llm.py:638-707` (`_run_output_handler`: :648-652,:664-665,:676-678,:697-702); `vllm/v1/core/sched/scheduler.py:1938-1951` + `vllm/v1/engine/__init__.py:249-251` (the stats/timestamp stamp both step paths share); scrape spec `tests/entrypoints/serve/instrumentator/test_metrics.py:182-228` | registry `include/vllm/v1/metrics/prometheus.h`, `src/vllm/v1/metrics/prometheus.cpp:13,189`; catalog+record `include/vllm/v1/metrics/loggers.h`, `src/vllm/v1/metrics/loggers.cpp:10,60,208`; stats structs + `MonotonicSeconds` `include/vllm/v1/metrics/stats.h:56,160,175,194`; `make_stats` `include/vllm/v1/core/sched/scheduler.h`, `src/vllm/v1/core/sched/scheduler.cpp` (+prefix-delta stash in `schedule()`); `scheduler_stats`/`timestamp` on `EngineCoreOutputs` `include/vllm/v1/engine/types.h`, stamped `src/vllm/v1/engine/core.cpp`; IterationStats build `src/vllm/v1/engine/output_processor.cpp` (+RequestState timing `include/vllm/v1/engine/output_processor.h`); step-site `Record` `src/vllm/v1/engine/llm_engine.cpp:220`, setter `include/vllm/v1/engine/llm_engine.h`; ASYNC step-site `Record` + `IterationStats` `src/vllm/v1/engine/async_llm.cpp:538`, setter + atomic `stat_logger_` `include/vllm/v1/engine/async_llm.h`; batch-queue `scheduler_stats`/`timestamp` stamp `src/vllm/v1/engine/core.cpp:219`; recorder mutex `include/vllm/v1/metrics/loggers.h`, `src/vllm/v1/metrics/loggers.cpp:208,270`; async attach `src/vllm/entrypoints/openai/server_main.cpp:930`; endpoint `src/vllm/entrypoints/openai/api_server.cpp:1359` (`handle_metrics`), route `:488` | `tests/vllm/v1/test_prometheus_metrics.cpp` 4/4 (81 assertions: EXPECTED_METRICS_V1 substring gate RED-first, label schema, TYPE lines, bucket schedules, record() folding); **live-wiring behavioural gate `tests/vllm/v1/test_llm_engine.cpp` case 6 (44 assertions, RED-first: 14 flip 0→correct when `Record` disabled) — running/waiting gauges track the batch, prompt/generation counters == exact token counts, request_success counts finished reqs, TTFT/ITL/e2e/TPOT/iteration histograms observe the right sample counts**; endpoint `tests/vllm/entrypoints/openai/test_api_server.cpp:921`; **ASYNC serving-path gate `tests/vllm/v1/test_llm_engine.cpp:1025` "async_llm: live per-step stats populate the Prometheus registry" (asserts case 6's AND case 7's invariants on the `AsyncHarness` stack; RED-first: 19 of them read 0 unwired) + `:1148` no-logger token-stream identity, and the depth-2 batch-queue pair `tests/vllm/v1/test_async_llm.cpp:560,618` (running gauge poll — RED times out; TTFT/e2e `_sum` > 0 — RED negative/zero); CPU `ctest` 366/366**; **spec-decode gate `tests/vllm/entrypoints/openai/test_api_server.cpp` "/metrics exports the spec_decode families a speculating engine actually produced" (a completion driven through `handle_completions` over a drafting runner, scraped back through `handle_metrics`, counters compared for EQUALITY against what the drafter recorded; RED 0 == 8/24/15 with the `make_spec_decoding_stats` call, the `make_stats` republish or the `Record` fold deleted) + "/metrics omits the spec_decode families with no speculative config" (the upstream gate) + `tests/vllm/v1/test_scheduler.cpp` "spec-decoding stats count drafts, draft tokens, accepted tokens and per-position acceptance" (all six `test_schedule_spec_decoding_stats` cases)** | [prometheus-metrics.md](specs/prometheus-metrics.md), [async-metrics.md](specs/async-metrics.md), [spec-decode-metrics.md](specs/spec-decode-metrics.md) | `ANCHOR-BACKFILL` | `CLAIM-SERVE-METRICS-ASYNC` | | `SERVE-RESPONSE-METRICS` | Per-request timing surface: the QUEUED/SCHEDULED/PREEMPTED EngineCoreEvents the scheduler emits + the per-request queue/prefill/inference timing intervals + preemption counter they feed. **EngineCoreEvents + timing LANDED + CPU-GATED 2026-07-27 (`CLAIM-ROADMAP-C8-RESPONSE-METRICS`, NOT pushed):** `EngineCoreEventType{QUEUED,SCHEDULED,PREEMPTED}` + `EngineCoreEvent{type,timestamp}` recorded on `Request` at the add_request / batch-admission / KV-preempt sites (1:1 with vLLM, gated on `log_stats_`, default no-stats path byte-identical), drained onto `EngineCoreOutput.events` via `take_events()`; `OutputProcessor` folds them (`update_from_events`) into `RequestState.queued_ts/scheduled_ts` → `FinishedRequestStats.queued_time`(=scheduled−queued)/`prefill_time`(=first_token−scheduled)/`inference_time`(=last_token−scheduled) + `IterationStats.num_preempted_reqs`, feeding the `vllm:request_{queue,prefill,inference}_time_seconds` histograms + `vllm:num_preemptions_total` (already in the catalog, left at 0 by the live-metrics wiring for lack of events). Additive; scheduling/compute/token stream unchanged. **ASYNC SERVING PATH COVERED 2026-08-10 (`CLAIM-SERVE-METRICS-ASYNC`, [#277](https://github.com/mudler/vllm.cpp/issues/277)):** these intervals only ever reached a registry through `LLMEngine`. They now populate through `AsyncLLM` too — its output handler folds the `IterationStats` these events fill, and `EngineCore::step_with_batch_queue` stamps the engine-core `timestamp` the intervals are measured against (unstamped it was 0.0, making every TTFT/e2e observation `-arrival_time`). Gated on the async stack by `tests/vllm/v1/test_llm_engine.cpp:1025`: queue/prefill/inference/decode `_sum` all > 0 and inference == prefill + decode. **RESIDUAL:** the streaming/non-streaming chat/completion RESPONSE-BODY timing surface (protocol/serving) + CLI validation. | T1 | `vllm/v1/engine/__init__.py:150-176` (EngineCoreEvent(Type)); `vllm/v1/core/sched/scheduler.py:2135,1003,1221,461,1839` (record/take_events sites); `vllm/v1/metrics/stats.py:428-476` (update_from_events / update_from_finished_request); response-body: `vllm/entrypoints/openai/engine/protocol.py:118`; `vllm/entrypoints/openai/{completion,chat_completion}/serving.py:461-481,765-784` @ `555967922` | events `include/vllm/v1/engine/event.h`, `Request.events`+`record_event`/`take_events` `include/vllm/v1/request.h`; `EngineCoreOutput.events` `include/vllm/v1/engine/types.h`; emission `src/vllm/v1/core/sched/scheduler.cpp` (`add_request`/`preempt_request`/`schedule`/`update_from_output`) + `log_stats_` `include/vllm/v1/core/sched/scheduler.h`; consumption `src/vllm/v1/engine/output_processor.cpp` (`process_outputs`) + `RequestState.queued_ts/scheduled_ts` `include/vllm/v1/engine/output_processor.h`; logger already consumes `src/vllm/v1/metrics/loggers.cpp:225,254-257` | `tests/vllm/v1/test_scheduler.cpp:420` "records QUEUED/SCHEDULED/PREEMPTED engine-core events" (15 assertions, RED-first, real KV-exhaustion preemption); `tests/vllm/v1/test_llm_engine.cpp` "per-request queue/prefill/inference timing populates" (26 assertions, RED-first: 5 flip 0→positive; asserts inference=prefill+decode, prefill≤inference≤e2e) | [per-request-response-metrics.md](specs/per-request-response-metrics.md) | `ANCHOR-BACKFILL` | `CLAIM-ROADMAP-C8-RESPONSE-METRICS` | | `SERVE-STREAM-USAGE` | Completion/chat `stream_options`: final and continuous native-ID usage frames, non-stream validation, and force-usage server mode. GATING: the host implementation is CPU/sanitizer-green; void `31d053f` 27B execution proved exact native counts on all 2,016 standard timed requests, but fresh passing 27B→35B online and serialization A/B gates remain mandatory | T1 | `vllm/entrypoints/openai/engine/protocol.py:241-243`; completion `protocol.py:66,471-478`, `serving.py:298-305,359-454`; chat `protocol.py:214,731-737`, `serving.py:459-512,570-760`; `entrypoints/serve/utils/api_utils.py:276-289`; `tests/entrypoints/openai/completion/test_completion.py:400-553`; `tests/entrypoints/openai/chat_completion/test_chat.py:348-445` | schema/parser `include/vllm/entrypoints/openai/protocol.h:62,203,318`, `src/vllm/entrypoints/openai/protocol.cpp:103,223,278`; selection `src/vllm/entrypoints/openai/serving_utils.cpp:8`; completion SSE `src/vllm/entrypoints/openai/serving_completion.cpp:22,160`; chat SSE `src/vllm/entrypoints/openai/serving_chat.cpp:232,450`; force CLI `src/vllm/entrypoints/openai/server_main.cpp:547` | protocol/selection `tests/vllm/entrypoints/openai/test_protocol.cpp:130,189`; sync completion/chat `tests/vllm/entrypoints/openai/test_serving.cpp:484,647`; production final/continuous/validation/force/disconnect `tests/vllm/entrypoints/openai/test_api_server.cpp:403,442,498,607,652,687,712`; help `examples/CMakeLists.txt:36`. CPU CTest 105/105; focused 63 cases/658 assertions; API repeat 100/100; ASan+UBSan 3/3; TSan 1/1. `31d053f` retained all 36 standard 27B raw points / 2,016 successful requests with exact native 128-token usage | [stream-options.md](specs/stream-options.md) | `GATING` | - | -| `SERVE-UTILITY-ENDPOINTS` | Tokenize, detokenize, ready, ping, server info, prefix reset. **LANDED + CPU-GATED 2026-07-27 (`CLAIM-ROADMAP-C8`, NOT pushed):** `/tokenize` (prompt form → `{count,max_model_len,tokens,token_strs?}`) + `/detokenize` (`{tokens[]}`→`{prompt}`) over the existing tokenizer, `/ping` (liveness, mirrors `/health`), `/server_info` (`{vllm_config,vllm_env,system_env}`), `/reset_prefix_cache` (`{"success":bool}` via an injected callback). All ADDITIVE + opt-in (tokenize/detokenize/reset registered only when their backing is attached). **CHAT-FORM `/tokenize` LANDED + CPU-GATED 2026-07-28 (`CLAIM-C8-CHAT-TOKENIZE`, NOT pushed):** `/tokenize` now accepts BOTH arms of the vLLM `TokenizeRequest` union — the raw `prompt` form AND the `TokenizeChatRequest{messages, add_generation_prompt, continue_final_message, add_special_tokens, tools?}`; the chat form renders through `chat_.prompt_fn()` (the IDENTICAL model chat template `create_chat_completion` tokenizes through), applies the `check_generation_prompt` mutual-exclusion (→400), tokenizes with the chat-form `add_special_tokens` default False (vs completion-form True), returns the same `{count,max_model_len,tokens,token_strs?}`. **`/tokenizer_info` LANDED + CPU-GATED 2026-07-28 (`CLAIM-C8-SERVE-ENDPOINTS`, NOT pushed):** `GET /tokenizer_info` gated behind a `set_tokenizer_info_enabled` flag mirroring vLLM's `enable_tokenizer_info_endpoint` CLI arg (off by default → the route is not registered → 404; on + tokenizer attached → 200). Surfaces the `tokenizer_config.json`-equivalent fields our byte-level/SentencePiece BPE tokenizer can GENUINELY back — `tokenizer_class` (the BPE family name), `model_max_length`, `vocab_size`, `bos_token_id`/`eos_token_id` (omitted when -1), and `added_tokens_decoder` (id → `{content,special,lstrip,rstrip}`); NAMED gaps OMITTED (never fabricated): the raw `chat_template` string (lives in the ChatPromptFn render seam, not the tokenizer), the HF `init_kwargs` (`clean_up_tokenization_spaces`/`add_bos_token`/`model_input_names`/padding-truncation defaults — not parsed), and the added-token `normalized`/`single_word` flags. **PRODUCTION `main.cpp` WIRING LANDED + CPU-GATED 2026-07-28 (`CLAIM-C8-SERVE-PROD-WIRING`, NOT pushed):** the shipped `vllm-server` binary now lights `/tokenize`+`/detokenize` (on by default when a tokenizer exists) and `/tokenizer_info` (behind the new `--enable-tokenizer-info-endpoint` flag, mirroring vLLM's `enable_tokenizer_info_endpoint`) from the LIVE engine+tokenizer through the shared `ConfigureUtilityEndpoints` seam — the SAME seam the gate drives over a real socket. `/metrics` + `/reset_prefix_cache` stay UNWIRED (named residuals): the production `AsyncLLM` frontend exposes no live `PrometheusStatLogger` (async stats deferred; missing `LoadedEngine::stat_logger()` + a `Record()` site in `AsyncLLM::RunOutputHandler`) and no thread-safe prefix-cache reset RPC (`reset_prefix_cache()` is `KVCacheManager`-private, mutated only on the EngineCore thread; missing `AsyncLLM::reset_prefix_cache`), so attaching either would be a fabricated wiring that never reaches the live engine — library handlers+tests retained. RESIDUAL: `chat_template_kwargs`/`continue_final_message` full template-render passthrough (the ChatPromptFn seam renders only via the `add_generation_prompt` gate), `/ready`, full server_info config dump, live `/metrics` + `/reset_prefix_cache` backing on the AsyncLLM path | T1 | `vllm/entrypoints/serve/tokenize/api_router.py:37,63,95-108`; production gating `vllm/entrypoints/openai/api_server.py:222`, `vllm/entrypoints/serve/__init__.py:11-31`, `vllm/entrypoints/openai/cli_args.py:140`; `vllm/entrypoints/serve/tokenize/protocol.py:24,50,156,166,181,185`; `vllm/entrypoints/serve/tokenize/serving.py:57,70-124,154-195`; `vllm/entrypoints/serve/sagemaker/api_router.py:47`; `vllm/entrypoints/serve/dev/server_info/api_router.py:43`; `vllm/entrypoints/serve/dev/cache/api_router.py:20` | handlers `src/vllm/entrypoints/openai/api_server.cpp:1365` (`handle_tokenize`, prompt+chat union),`:368,404,245,422` (`handle_detokenize`/`handle_reset_prefix_cache`/`handle_ping`/`handle_server_info`),`:438` (`handle_tokenizer_info`); chat render seam `include/vllm/entrypoints/openai/serving_chat.h:263` (`prompt_fn()`); opt-in setters + routes `include/vllm/entrypoints/openai/api_server.h:408` (`set_tokenizer_info_enabled`); production seam `include/vllm/entrypoints/openai/api_server.h` (`ConfigureUtilityEndpoints`) + `src/vllm/entrypoints/openai/api_server.cpp` (impl); production call + CLI flags `examples/server/main.cpp` (`--enable-tokenizer-info-endpoint`, `ConfigureUtilityEndpoints(...)`) | `tests/vllm/entrypoints/openai/test_api_server.cpp:879` (prompt round-trip+schema+raw-form exact ids),`:938` (chat-form renders template + exact tokens, RED-first),`:1061` (`/tokenizer_info` backed fields + named-gap omissions + no-tokenizer 500),`:1250` (opt-in route gate: 404 flag-off → 200 flag-on over a real socket, RED-first),`:1319` (**production `ConfigureUtilityEndpoints` seam over a real socket: no-seam→404 RED, defaults→tokenize/detokenize 200 + info/abort 404, flags-on→200, exact abort delta-count**) — 32/32 / 420-assertion suite | [utility-endpoints.md](specs/utility-endpoints.md) | `ANCHOR-BACKFILL` | `CLAIM-C8-SERVE-PROD-WIRING` | +| `SERVE-UTILITY-ENDPOINTS` | Tokenize, detokenize, ready, ping, server info, prefix reset. **LANDED + CPU-GATED 2026-07-27 (`CLAIM-ROADMAP-C8`, NOT pushed):** `/tokenize` (prompt form → `{count,max_model_len,tokens,token_strs?}`) + `/detokenize` (`{tokens[]}`→`{prompt}`) over the existing tokenizer, `/ping` (liveness, mirrors `/health`), `/server_info` (`{vllm_config,vllm_env,system_env}`), `/reset_prefix_cache` (`{"success":bool}` via an injected callback). All ADDITIVE + opt-in (tokenize/detokenize/reset registered only when their backing is attached). **CHAT-FORM `/tokenize` LANDED + CPU-GATED 2026-07-28 (`CLAIM-C8-CHAT-TOKENIZE`, NOT pushed):** `/tokenize` now accepts BOTH arms of the vLLM `TokenizeRequest` union — the raw `prompt` form AND the `TokenizeChatRequest{messages, add_generation_prompt, continue_final_message, add_special_tokens, tools?}`; the chat form renders through `chat_.prompt_fn()` (the IDENTICAL model chat template `create_chat_completion` tokenizes through), applies the `check_generation_prompt` mutual-exclusion (→400), tokenizes with the chat-form `add_special_tokens` default False (vs completion-form True), returns the same `{count,max_model_len,tokens,token_strs?}`. **`/tokenizer_info` LANDED + CPU-GATED 2026-07-28 (`CLAIM-C8-SERVE-ENDPOINTS`, NOT pushed):** `GET /tokenizer_info` gated behind a `set_tokenizer_info_enabled` flag mirroring vLLM's `enable_tokenizer_info_endpoint` CLI arg (off by default → the route is not registered → 404; on + tokenizer attached → 200). Surfaces the `tokenizer_config.json`-equivalent fields our byte-level/SentencePiece BPE tokenizer can GENUINELY back — `tokenizer_class` (the BPE family name), `model_max_length`, `vocab_size`, `bos_token_id`/`eos_token_id` (omitted when -1), and `added_tokens_decoder` (id → `{content,special,lstrip,rstrip}`); NAMED gaps OMITTED (never fabricated): the raw `chat_template` string (lives in the ChatPromptFn render seam, not the tokenizer), the HF `init_kwargs` (`clean_up_tokenization_spaces`/`add_bos_token`/`model_input_names`/padding-truncation defaults — not parsed), and the added-token `normalized`/`single_word` flags. **PRODUCTION `main.cpp` WIRING LANDED + CPU-GATED 2026-07-28 (`CLAIM-C8-SERVE-PROD-WIRING`, NOT pushed):** the shipped `vllm-server` binary now lights `/tokenize`+`/detokenize` (on by default when a tokenizer exists) and `/tokenizer_info` (behind the new `--enable-tokenizer-info-endpoint` flag, mirroring vLLM's `enable_tokenizer_info_endpoint`) from the LIVE engine+tokenizer through the shared `ConfigureUtilityEndpoints` seam — the SAME seam the gate drives over a real socket. `/metrics` + `/reset_prefix_cache` stay UNWIRED (named residuals): the production `AsyncLLM` frontend exposes no live `PrometheusStatLogger` (async stats deferred; missing `LoadedEngine::stat_logger()` + a `Record()` site in `AsyncLLM::RunOutputHandler`) and no thread-safe prefix-cache reset RPC (`reset_prefix_cache()` is `KVCacheManager`-private, mutated only on the EngineCore thread; missing `AsyncLLM::reset_prefix_cache`), so attaching either would be a fabricated wiring that never reaches the live engine — library handlers+tests retained. RESIDUAL: `chat_template_kwargs`/`continue_final_message` full template-render passthrough (the ChatPromptFn seam renders only via the `add_generation_prompt` gate), `/ready`, full server_info config dump, live `/metrics` + `/reset_prefix_cache` backing on the AsyncLLM path | T1 | `vllm/entrypoints/serve/tokenize/api_router.py:37,63,95-108`; production gating `vllm/entrypoints/openai/api_server.py:222`, `vllm/entrypoints/serve/__init__.py:11-31`, `vllm/entrypoints/openai/cli_args.py:140`; `vllm/entrypoints/serve/tokenize/protocol.py:24,50,156,166,181,185`; `vllm/entrypoints/serve/tokenize/serving.py:57,70-124,154-195`; `vllm/entrypoints/serve/sagemaker/api_router.py:47`; `vllm/entrypoints/serve/dev/server_info/api_router.py:43`; `vllm/entrypoints/serve/dev/cache/api_router.py:20` | handlers `src/vllm/entrypoints/openai/api_server.cpp:1370` (`handle_tokenize`, prompt+chat union),`:368,404,245,422` (`handle_detokenize`/`handle_reset_prefix_cache`/`handle_ping`/`handle_server_info`),`:438` (`handle_tokenizer_info`); chat render seam `include/vllm/entrypoints/openai/serving_chat.h:263` (`prompt_fn()`); opt-in setters + routes `include/vllm/entrypoints/openai/api_server.h:408` (`set_tokenizer_info_enabled`); production seam `include/vllm/entrypoints/openai/api_server.h` (`ConfigureUtilityEndpoints`) + `src/vllm/entrypoints/openai/api_server.cpp` (impl); production call + CLI flags `examples/server/main.cpp` (`--enable-tokenizer-info-endpoint`, `ConfigureUtilityEndpoints(...)`) | `tests/vllm/entrypoints/openai/test_api_server.cpp:879` (prompt round-trip+schema+raw-form exact ids),`:938` (chat-form renders template + exact tokens, RED-first),`:1061` (`/tokenizer_info` backed fields + named-gap omissions + no-tokenizer 500),`:1250` (opt-in route gate: 404 flag-off → 200 flag-on over a real socket, RED-first),`:1319` (**production `ConfigureUtilityEndpoints` seam over a real socket: no-seam→404 RED, defaults→tokenize/detokenize 200 + info/abort 404, flags-on→200, exact abort delta-count**) — 32/32 / 420-assertion suite | [utility-endpoints.md](specs/utility-endpoints.md) | `ANCHOR-BACKFILL` | `CLAIM-C8-SERVE-PROD-WIRING` | | `SERVE-CHAT-TEMPLATE` | Full-surface Jinja chat templates (vendored google/minja `021c229` + documented lstrip guard + the six arity-0 Jinja2 built-in tests upstream minja still lacks and can reach -- `undefined`, `even`, `odd`, `lower`, `upper`, `escaped`; `callable` is deliberately NOT added, because minja defers every binary op with a callable left operand so the test can never be handed one, and its refusal is pinned at `tests/vllm/entrypoints/test_chat_template.cpp:297`). **`is undefined` REPAIRED 2026-08-22 ([#1681](https://github.com/mudler/vllm.cpp/issues/1681)): `POST /v1/chat/completions` answered HTTP 500 for the whole Qwen3.8 family**, because the engine threw `Unknown type for 'is' operator: undefined` on the checkpoint's own template; no gate saw it because every benchmark drives `vllm-cli`, which renders no template, and the committed `qwen35_chat_template.jinja` carries the same construct behind a short-circuiting `or` that the gated conversations never reach. The same change makes request `chat_template_kwargs` reach the renderer and stops `apply_chat_template` defining `enable_thinking` when nobody supplied it, which is what upstream does (`vllm/renderers/hf.py:633-661`) and what a template that asks `is undefined` needs. **Second review 2026-08-23:** the request-kwargs filter refused the four names the ADAPTER sets and could not see the 31 the ENGINE sets -- minja resolves a global, a filter and an is-test through one `Context` chain -- so `{"namespace":1}` shadowed the built-in the Qwen3.8 template calls on its FIRST line and answered a new HTTP 500; jinja2 keeps all three kinds out of the variable namespace, so upstream drops every one of them and `accept_vars & minja_builtins` is exactly `{raise_exception}`. A render failure the request caused was also a 500 where upstream answers 400 twice over (`hf.py:785-789` wraps it into a `ValueError`, `error_response.py:48-52,61-65` maps that and `jinja2.TemplateError` to `BadRequestError`), and `/tokenize` already answered 400 for the identical body | T0 | `vllm/renderers/hf.py:673,986`; `vllm/entrypoints/chat_utils.py:1248,1335`; request field `vllm/entrypoints/openai/chat_completion/protocol.py:341,545-556`; kwarg resolution `vllm/renderers/hf.py:633-661,731-735,777-783` | `src/vllm/entrypoints/chat_template.cpp:110,139,234,271,278,333` (`apply_chat_template`, its request-kwargs filter, `MakeChatTemplatePromptFn`, `DefaultChatTemplateKwargs`, `LoadChatTemplateFromConfig`, `LoadChatTemplateFromGguf`); `third_party/minja/minja.hpp` (`BinaryOpExpr::do_evaluate` is-test table) | `tests/vllm/entrypoints/test_chat_template.cpp:67,77,86,98,224,261,305,345,411,463`; production-dispatch gate on the real published Qwen3.8 template `tests/vllm/entrypoints/openai/test_api_server.cpp:833,859,882,922,994` plus the C ABI gate `tests/capi/test_capi.cpp:1012` over the committed `tests/fixtures/qwen38_chat_template.jinja` | [`specs/chat-template-jinja-undefined.md`](specs/chat-template-jinja-undefined.md) | `ANCHOR-BACKFILL` | - | | `SERVE-ASYNC-LLM` | AsyncLLM-equivalent streaming engine API: per-request collectors, concurrent submit/generate/abort, live completion/chat SSE with disconnect abort, additive nonblocking C requests, and enough HTTP delivery capacity for configured concurrent streams. GATING: deterministic c32 capacity is implemented and GPU-classified; broader every-axis parity remains open. **CLARIFIED 2026-08-12 ([#534](https://github.com/mudler/vllm.cpp/issues/534)) — this row is NOT waiting on a "prod-ON" flip,** which is what punch-list item 9 and the `ROAD-V1-A` SGLang clause both read it as. It IS the production serving path (`src/vllm/entrypoints/openai/server_main.cpp:731-734`, *"the production server uses AsyncLLM over EngineCoreProc's dedicated engine thread"*), with the capacity-derived fixed HTTP pool as the default and `VLLM_CPP_HTTP_FIXED_POOL=0` retained only as a same-binary diagnostic; the separate runner-side `VT_ASYNC_RUNNER`/`runner_supports_async` default is `ENG-ASYNC-SCHED`'s and has been ON since `a0013a2`. What remains is exactly the every-axis parity named above: 27B ratified (two-grid 115/124 effective), 35B open under `ROAD-V1-A`, plus open bug [#294](https://github.com/mudler/vllm.cpp/issues/294). Its GPU token-exact gate is `tests/parity/test_qwen36_async_serving.cpp` (`1718bf155`) — NOT `qwen36_paged_engine`, which drives the sync depth-1 path and structurally cannot see this row's defects | T0 | `vllm/v1/engine/async_llm.py:70,280,524,637,709`; `vllm/v1/engine/output_processor.py:45-105`; asyncio server path `vllm/entrypoints/openai/api_server.py:1`; `tests/v1/engine/test_async_llm.py:109,157,228,306,340,598` | existing async path `include/vllm/v1/engine/async_llm.h:45`, `src/vllm/v1/engine/async_llm.cpp:32`; fixed/legacy pool API `include/vllm/entrypoints/openai/api_server.h:41-57,101-104`; capacity selection `src/vllm/entrypoints/openai/api_server.cpp:23-62`; production max-seqs wiring + `VLLM_CPP_HTTP_FIXED_POOL=0` A/B `src/vllm/entrypoints/openai/server_main.cpp:874-883` (moved verbatim out of `examples/server/main.cpp` by ARCH-ONE-SURFACE #189; the example is now a one-line `vllm_server_main` client); cpp-httplib defect `third_party/httplib/httplib.h:161-169,10359-10377` | persistent 32-client + control reserve, validation and diagnostic-mode cases `tests/vllm/entrypoints/openai/test_api_server.cpp:937-1000`; focused Release/help pass, API **100/100**, ASan+UBSan **1/1**, TSan **1/1**; known unrelated serial C-API flake isolated. Exact fixed/legacy c32 AB/BA/AB is healthy and steady-state-neutral: **1097.031/1097.290 tok/s = 0.999764×**, 8/20 axes, 1,152/1,152 requests and six memory returns; neither legacy arm samples the rare old stall. Exact `4e1d8ca` fixed c32 is healthy 3/3 and 0.9910× vLLM | [async-serving.md](specs/async-serving.md) | `GATING` | - | | `SERVE-HTTP-TRANSPORT` | Serving-socket transport parity: mirror vLLM's uvicorn/asyncio default `TCP_NODELAY` on every accepted SSE socket so per-token stream frames are not held by Nagle against the peer's delayed ACK. Implemented + CPU-tested; the non-binding localhost A/B sizing is COMPLETE and NEUTRAL within noise on c1/c2 ITL/TPOT/throughput (loopback ACKs are instant, so Nagle never coalesces ~100 ms-cadence token frames) — no gate-axis credit expected; the mirror stays for real-network parity. Future keep-alive / read-write-timeout / listening-socket option parity noted, not done | T0 | vLLM serves via uvicorn over asyncio `vllm/entrypoints/launcher.py:71,76`, `vllm/entrypoints/openai/api_server.py:591,630`; asyncio disables Nagle per accepted TCP stream socket `asyncio/base_events.py:192-197` (`_set_nodelay`) called from `asyncio/selector_events.py:950`; cpp-httplib default-off `third_party/httplib/httplib.h:142`, applied on accept only when set `third_party/httplib/httplib.h:12083` | `src/vllm/entrypoints/openai/api_server.cpp:69` (`set_tcp_nodelay(true)` in the ApiServer setup) | behavioral accepted-socket `getsockopt(TCP_NODELAY)` case `tests/vllm/entrypoints/openai/test_api_server.cpp:1076` (helper `:380`); RED accepted `TCP_NODELAY` 0 → GREEN 1, full `test_openai_api_server` **22/22 cases / 242 assertions**; non-binding sizing root `~/work/vllm.cpp-tcpnodelay-sizing/ff915e8…` (raw-set SHA `f5b52900…2128`) neutral within noise; closure [ledger](parity-ledger.md#L451) | [serve-tcp-nodelay.md](specs/serve-tcp-nodelay.md) | `DONE` | `ff915e8` | @@ -243,7 +243,7 @@ claims it. | `ENG-POOLER-SEQ` | The non-generative POOLER OP — turn hidden states into a pooled embedding/logit row instead of a sampled token. **W1 LANDED + CPU-GATED 2026-07-28 (`CLAIM-POOLING`, NOT pushed):** the sequence pooling methods `CLSPool`/`LastPool`/`MeanPool` (+ `GetSeqPoolingMethod` factory) over a packed `[num_tokens, hidden]` CPU buffer keyed by a minimal `PoolingCursor` (CLS/MEAN reject partial prefill, LAST allows it, MeanPool upcasts to float32) and the activation heads `PoolerIdentity`/`PoolerNormalize` (L2 `F.normalize`)/`PoolerMultiLabelClassify` (sigmoid)/`PoolerClassify` (sigmoid if `num_labels<2` else `softmax`). Unit-gated vs DOUBLE-PRECISION references, RED-first. **W2 LANDED + CPU-GATED 2026-07-29 (`CLAIM-POOLING`, NOT pushed):** the pooler HEADS composite (`EmbeddingPoolerHead` = projector→matryoshka→normalize; `ClassifierPoolerHead` = classifier→`(logit-mean)/sigma`→activation), the `SequencePooler` (method∩head task intersection) + `PoolerForEmbed`/`PoolerForClassify` factories, the `DispatchPooler` groupby-task routing (`ForEmbedding`/`ForSeqCls` + a mixed embed+classify batch + ctor task-support validation), and the `PoolerConfig`/`PoolingParams`/`PoolingParamsUpdate` structs; `test_pooler_heads` 27/27 (240 asserts) vs double-precision refs, RED-first (disable matryoshka slice + logit_mean calibration → 8 cases / 50 asserts fail). RESIDUALS (named, spec §Work breakdown): the endpoints (W4), tokwise `AllPool`/`StepPool` (W5), a concrete pooling MODEL + real-oracle cosine gate (W3-model — see `ENG-POOLING-RUNNER`) | T2 | `vllm/model_executor/layers/pooler/seqwise/methods.py:35-121`; `vllm/model_executor/layers/pooler/activations.py:106-158`; `vllm/model_executor/layers/pooler/seqwise/heads.py:19-196`; `vllm/model_executor/layers/pooler/seqwise/poolers.py:41-138`; `vllm/model_executor/layers/pooler/special.py:23-140`; `vllm/model_executor/layers/pooler/common.py:12-30`; `vllm/pooling_params.py:35-70`; `vllm/config/pooler.py:16-90`; `vllm/v1/pool/metadata.py:13-71`; `tests/model_executor/layers/test_pooler_methods.py`, `tests/model_executor/layers/test_pooler_activations.py`, `tests/model_executor/layers/test_pooler_heads.py` | `include/vllm/model_executor/layers/pooler/{methods,activations,pooling_metadata,common,pooling_params,pooler_config,heads,poolers,dispatch_pooler}.h` + `src/vllm/model_executor/layers/pooler/{methods,activations,heads,poolers,dispatch_pooler}.cpp` — anchor `src/vllm/model_executor/layers/pooler/dispatch_pooler.cpp:13` | `tests/vllm/model_executor/layers/pooler/test_pooler.cpp` (CLS/LAST/MEAN + factory + activations, 50 asserts) + `test_pooler_heads.cpp` (Embedding/Classifier heads + SequencePooler + DispatchPooler, 240 asserts) — anchor `tests/vllm/model_executor/layers/pooler/test_pooler.cpp:81` | [pooling-task-class.md](specs/pooling-task-class.md) | `ACTIVE` | `CLAIM-POOLING` | | `ENG-POOLING-RUNNER` | The pooling RUNNER path — where the generation runner SAMPLES a token, the pooling runner applies the model's `Pooler` to the last hidden state and returns the POOLED DATA (embedding vector / classification logit row). **W3 LANDED + CPU-GATED 2026-07-29 (`CLAIM-POOLING`, NOT pushed):** `PoolingRunner` over a packed `[num_tokens, hidden]` last-hidden-state buffer + a `PoolingMetadata` — `Pool()` delegates to the model pooler (`DispatchPooler.ForEmbedding`), `GetSupportedTasks()`, `ComputeValid()` (`seq_lens==prompt_len`). GATE: a STRUCTURAL cosine-parity gate — the runner's embedding vs an independent double-precision LAST+normalize reference is cosine≈1 (5 cases / 14 asserts), RED-first (CLS-instead-of-LAST drops cosine <0.5; disable normalize → 2 unit-L2 asserts fail). GENERALIZATION DEVIATION: upstream `pooling_runner.py` hardcodes LAST+normalize; we route through the model `Pooler` (the general bert.py path), strictly more capable. HONEST RESIDUAL (named): the REAL-model oracle cosine gate (`vllm.LLM(task="embed").encode`) needs a registered concrete embedding model's forward — no such model is registered yet (W3-model), so no cosine-vs-oracle number is fabricated. **LIVE IN THE ENGINE STEP 2026-08-08 (ARCH-ONE-SURFACE ROW 6, `CLAIM-EMBEDDINGS-ONE-SURFACE`):** `GPUModelRunner` builds a `PoolingRunner` iff the loaded model registration declares `is_pooling_model` (gpu/model_runner.py:368-369 mirror) and `sample_tokens` routes to `pool_tokens()` — pooled data instead of sampled tokens (model_runner.py:1586-1607), validity = the discard predicate (`seq_len < num_tokens` == upstream is_valid, pooling_runner.py:40-41); the scheduler finishes a pooling request on pooled output (scheduler.py:1718-1721) and `EngineCoreOutput.pooling_output` carries it out; async scheduling resolves OFF for pooling models (config/vllm.py:1068-1073, the landed ResolveAsyncScheduling arm now WIRED at model_loader.cpp). First registered pooling arch: `LlamaModel` (`MODEL-EMBED-llama-llama-for-causal-lm`). The fold gate re-anchors the lane's cosine gate THROUGH the registry/runner path: engine path == direct `ModelRegistry::Forward`+`PoolingRunner` path, identical vectors + f64 LAST+normalize reference + chunked-prefill arm (`test_llama_embedding_fold` 4/4-231). REMAINING RESIDUAL: the REAL-model `vllm.LLM(task="embed").encode` oracle cosine (synthetic fixture only — no number fabricated) | T2 | `vllm/v1/worker/gpu/pool/pooling_runner.py:18-46`; `vllm/v1/worker/gpu/model_runner.py:368-369,1586-1607`; `vllm/v1/core/sched/scheduler.py:1718-1721,1837`; `vllm/tasks.py:10`; `tests/models/language/pooling/test_embedding.py` (real-oracle gate, DEFERRED) | `include/vllm/v1/worker/gpu/pool/pooling_runner.h` + `src/vllm/v1/worker/gpu/pool/pooling_runner.cpp:11`; live invocation `src/vllm/v1/worker/gpu/runner.cpp` `pool_tokens` + the `pooling_runner_` ctor gate; scheduler stop `src/vllm/v1/core/sched/scheduler.cpp` pooling elif | `tests/vllm/v1/worker/gpu/pool/test_pooling_runner.cpp:136` (structural cosine gate) + `tests/vllm/models/test_llama_embedding_fold.cpp:206` (registry/engine-path arm, 4/4-231, mutation-killed x9) | [pooling-task-class.md](specs/pooling-task-class.md) + [embeddings-one-surface.md](specs/embeddings-one-surface.md) | `ACTIVE` | `CLAIM-EMBEDDINGS-ONE-SURFACE` | | `SERVE-RESPONSES-MESSAGES` | Responses, Anthropic messages, audio | T2 | `vllm/entrypoints/openai/responses/api_router.py:48`; `vllm/entrypoints/anthropic/api_router.py:49`; `vllm/entrypoints/speech_to_text/transcription/api_router.py:1` | - | - | `planned: specs/responses-messages-endpoints.md` | `INVENTORIED` | - | -| `SERVE-ADMIN` | Abort-requests, sleep, pause/resume, profiling, RL weight updates. **`/abort_requests` LANDED + CPU-GATED 2026-07-28 (`CLAIM-C8-SERVE-ENDPOINTS`, NOT pushed):** `POST /abort_requests` (from the dev/rlhf admin router) parses `{request_ids:[...]}` and aborts exactly those (external) ids via an injected abort callback wired to the engine abort path (`AsyncLLM::abort`); an empty/missing list means "abort all in-flight" (the callback decides). Response `{"status":"aborted","aborted":}`; malformed JSON → 400 `{"detail":"Invalid JSON format"}`; abort failure → 500 `{"error":...}` — all three shapes mirror the upstream router verbatim. ADDITIVE + opt-in (route registered only when the abort callback is attached → 404 otherwise). **PRODUCTION `main.cpp` WIRING LANDED + CPU-GATED 2026-07-28 (`CLAIM-C8-SERVE-PROD-WIRING`, NOT pushed):** the shipped `vllm-server` binary now wires `/abort_requests` to the LIVE `AsyncLLM::abort` through the shared `ConfigureUtilityEndpoints` seam, DEV-mode gated behind the new `--enable-server-dev-mode` flag — mirroring vLLM registering the dev/rlhf router only under `if envs.VLLM_SERVER_DEV_MODE` (api_server.py:238; envs.py:157 default 0). Explicit-id abort tears the request down and reports the exact drop in unfinished requests (before−after); empty `request_ids` (abort-ALL) reports 0 — NAMED RESIDUAL (AsyncLLM exposes no active-request-id accessor). RESIDUAL: the abort-ALL enumeration (missing `AsyncLLM::active_request_ids()`); `/sleep`/`/wake_up`/`/is_sleeping`, `/pause`/`/resume`, `/start_profile`/`/stop_profile`, weight-update/EP endpoints still INVENTORIED | T2/T3 | `vllm/entrypoints/serve/dev/rlhf/api_router.py:94-138` (abort_requests); dev-mode gate `vllm/entrypoints/openai/api_server.py:238-240`, `vllm/entrypoints/serve/__init__.py:35`, `vllm/envs.py:157`; `vllm/entrypoints/serve/dev/sleep/api_router.py:21`; `vllm/entrypoints/serve/dev/rlhf/api_router.py:29,74,136`; `vllm/entrypoints/serve/profile/api_router.py:21` | handler `src/vllm/entrypoints/openai/api_server.cpp:1614` (`handle_abort_requests`); opt-in setter `include/vllm/entrypoints/openai/api_server.h:415` (`set_abort_requests`); production seam `src/vllm/entrypoints/openai/api_server.cpp` (`ConfigureUtilityEndpoints`, before/after delta-count) + `examples/server/main.cpp` (`--enable-server-dev-mode`); engine abort path `include/vllm/v1/engine/async_llm.h:115` (`abort`) | `tests/vllm/entrypoints/openai/test_api_server.cpp:1104` (shape + callback wiring: explicit ids passthrough, empty→abort-all branch, malformed→400),`:1143` (aborts an in-flight AsyncLLM request → `has_unfinished_requests()` false),`:1250` (opt-in route gate: 404 no-callback → 200 attached, RED-first),`:1319` (**production seam: dev-mode gate 404→200, live abort exact delta-count==1, empty→0**) — in the 32/32 / 420-assertion suite | [admin-endpoints.md](specs/admin-endpoints.md) | `ANCHOR-BACKFILL` | `CLAIM-C8-SERVE-PROD-WIRING` | +| `SERVE-ADMIN` | Abort-requests, sleep, pause/resume, profiling, RL weight updates. **`/abort_requests` LANDED + CPU-GATED 2026-07-28 (`CLAIM-C8-SERVE-ENDPOINTS`, NOT pushed):** `POST /abort_requests` (from the dev/rlhf admin router) parses `{request_ids:[...]}` and aborts exactly those (external) ids via an injected abort callback wired to the engine abort path (`AsyncLLM::abort`); an empty/missing list means "abort all in-flight" (the callback decides). Response `{"status":"aborted","aborted":}`; malformed JSON → 400 `{"detail":"Invalid JSON format"}`; abort failure → 500 `{"error":...}` — all three shapes mirror the upstream router verbatim. ADDITIVE + opt-in (route registered only when the abort callback is attached → 404 otherwise). **PRODUCTION `main.cpp` WIRING LANDED + CPU-GATED 2026-07-28 (`CLAIM-C8-SERVE-PROD-WIRING`, NOT pushed):** the shipped `vllm-server` binary now wires `/abort_requests` to the LIVE `AsyncLLM::abort` through the shared `ConfigureUtilityEndpoints` seam, DEV-mode gated behind the new `--enable-server-dev-mode` flag — mirroring vLLM registering the dev/rlhf router only under `if envs.VLLM_SERVER_DEV_MODE` (api_server.py:238; envs.py:157 default 0). Explicit-id abort tears the request down and reports the exact drop in unfinished requests (before−after); empty `request_ids` (abort-ALL) reports 0 — NAMED RESIDUAL (AsyncLLM exposes no active-request-id accessor). RESIDUAL: the abort-ALL enumeration (missing `AsyncLLM::active_request_ids()`); `/sleep`/`/wake_up`/`/is_sleeping`, `/pause`/`/resume`, `/start_profile`/`/stop_profile`, weight-update/EP endpoints still INVENTORIED | T2/T3 | `vllm/entrypoints/serve/dev/rlhf/api_router.py:94-138` (abort_requests); dev-mode gate `vllm/entrypoints/openai/api_server.py:238-240`, `vllm/entrypoints/serve/__init__.py:35`, `vllm/envs.py:157`; `vllm/entrypoints/serve/dev/sleep/api_router.py:21`; `vllm/entrypoints/serve/dev/rlhf/api_router.py:29,74,136`; `vllm/entrypoints/serve/profile/api_router.py:21` | handler `src/vllm/entrypoints/openai/api_server.cpp:1619` (`handle_abort_requests`); opt-in setter `include/vllm/entrypoints/openai/api_server.h:415` (`set_abort_requests`); production seam `src/vllm/entrypoints/openai/api_server.cpp` (`ConfigureUtilityEndpoints`, before/after delta-count) + `examples/server/main.cpp` (`--enable-server-dev-mode`); engine abort path `include/vllm/v1/engine/async_llm.h:115` (`abort`) | `tests/vllm/entrypoints/openai/test_api_server.cpp:1104` (shape + callback wiring: explicit ids passthrough, empty→abort-all branch, malformed→400),`:1143` (aborts an in-flight AsyncLLM request → `has_unfinished_requests()` false),`:1250` (opt-in route gate: 404 no-callback → 200 attached, RED-first),`:1319` (**production seam: dev-mode gate 404→200, live abort exact delta-count==1, empty→0**) — in the 32/32 / 420-assertion suite | [admin-endpoints.md](specs/admin-endpoints.md) | `ANCHOR-BACKFILL` | `CLAIM-C8-SERVE-PROD-WIRING` | | `SERVE-VIDEOS-OAI` | `/v1/videos` in OpenAI's Sora WIRE SHAPE, over the vLLM-Omni-derived job endpoints. **CPU-LANDED + GATED 2026-08-06 (`CLAIM-SERVE-VIDEOS-OAI`):** the OpenAI request spellings (`model`, `size` "WxH", `seconds` as a number OR the string enum OpenAI actually types) parse as ALIASES onto the existing native members, NATIVE-wins precedence applied PER-AXIS, both spellings validated either way so a malformed alias is a 400 even when overridden; an unserved `model` is a job `warning` echoed for the job's whole life, never a rejection (a Sora client cannot know the local model's name); and `GET /v1/videos/{id}/content` serves the finished MP4 (404 unknown / 409 unfinished / 500 failed / 500 vanished), without which a caller could start and poll a job but never fetch the result over HTTP. All four routes still register ONLY with a `VideoRunner` attached, now gated over a REAL socket. RESIDUALS (named): OpenAI's status vocabulary/id shape is not mirrored; reference conditioning (`input_reference`, the `metadata` video/audio references) is a stacked follow-up row; the real-weights leg rides the H3 GB10/disk window. | T2 | OpenAI Sora video API (`POST /v1/videos`, `GET /v1/videos/{video_id}/content`); vLLM-Omni `vllm/entrypoints/openai/video/api_router.py` (the async/sync job pair we already mirror) | `include/vllm/entrypoints/openai/video_api.h:31`; `src/vllm/entrypoints/openai/video_api.cpp:98`; `src/vllm/entrypoints/openai/api_server.cpp:279` | `tests/vllm/entrypoints/openai/test_video_api.cpp:64`; `tests/vllm/entrypoints/openai/test_api_server.cpp:1751` | [minimax-h3.md §9](specs/minimax-h3.md) | `ACTIVE` | `CLAIM-SERVE-VIDEOS-OAI` | | `SERVE-VIDEOS-REFS` | REFERENCE CONDITIONING over `/v1/videos`: the image an OpenAI request starts from, plus the two modalities OpenAI's schema has no slot for. **CPU-LANDED + GATED 2026-08-06 (`CLAIM-SERVE-VIDEOS-REFS`, stacked on `SERVE-VIDEOS-OAI`):** OpenAI's `input_reference` (a filesystem path or an RFC 2397 `data:` URL, decoded by the SAME `DecodeDataUri` the chat multimodal parts use) maps to fl2va FIRST-FRAME conditioning via `MiniMaxH3EncodeKeyframeCondRows`, because OpenAI documents it as the frame the video starts from and ref2va would silently change what the API promises; the silent-video and audio references ride the standard free-form `metadata` map (`input_reference_video`, a DIRECTORY of `frame_%06d.ppm` since no demuxer is vendored; `input_reference_audio`, a 16-bit PCM WAV) and become ref2va blocks, an audio reference ATTACHING to the video block when both are given. fl2va-vs-ref2va exclusivity is enforced in the PARSER, mirroring `minimax_h3_pipeline.cpp:251`, so an illegal pair is a 400 naming it rather than a dropped reference. Both VAE ENCODER halves load lazily and once. RESIDUALS (named): reference images are binary PPM at the output resolution (no PNG/JPEG codec, no resampler vendored); a video reference is a frame directory; OpenAI's real upload is multipart, ours is the JSON spelling. | T2 | OpenAI Sora video API (`input_reference`, `metadata`); exclusivity rule `src/vllm/model_executor/models/minimax_h3_pipeline.cpp:251` | `include/vllm/entrypoints/openai/video_api.h:51`; `src/vllm/entrypoints/openai/video_api.cpp:66`; `src/vllm/entrypoints/openai/server_main.cpp:238` | `tests/vllm/entrypoints/openai/test_video_api.cpp:172`; `tests/vllm/entrypoints/openai/test_api_server.cpp:1827` | [minimax-h3.md §10](specs/minimax-h3.md) | `ACTIVE` | `CLAIM-SERVE-VIDEOS-REFS` | | `SERVE-OTLP` | OpenTelemetry traces | T2 | `vllm/config/observability.py:18,36,128` | - | - | `planned: specs/otlp-tracing.md` | `INVENTORIED` | - | diff --git a/.agents/issues/MODEL-TEXT-kolibri-1-kolibri1-for-causal-lm/ISSUE-LOCAL-01M4EF3R0H2H3FN5NA0NAB0S62.md b/.agents/issues/MODEL-TEXT-kolibri-1-kolibri1-for-causal-lm/ISSUE-LOCAL-01M4EF3R0H2H3FN5NA0NAB0S62.md index 79722e11d..90a255e46 100644 --- a/.agents/issues/MODEL-TEXT-kolibri-1-kolibri1-for-causal-lm/ISSUE-LOCAL-01M4EF3R0H2H3FN5NA0NAB0S62.md +++ b/.agents/issues/MODEL-TEXT-kolibri-1-kolibri1-for-causal-lm/ISSUE-LOCAL-01M4EF3R0H2H3FN5NA0NAB0S62.md @@ -7,7 +7,7 @@ GitHub: - Mirror: PENDING Availability: FULL Created: 2026-10-08 -Updated: 2026-10-08 +Updated: 2026-10-09 Closed: 2026-10-08 ## Problem @@ -16,4 +16,8 @@ The kolibri1 CPU arm reaches the model forward but not serving: the OpenAI chat ## Resolution -Landed on row/kolibri-serve 2026-10-08; see Resolution and docs/bench-evidence/kolibri1-serve-20261008.md. +Landed on row/kolibri-serve 2026-10-08 (commits 3c0fcc164, 4aade352b); see docs/bench-evidence/kolibri1-serve-20261008.md. + +REOPENED IN EFFECT BY REVIEW 2026-10-09: the PR #3422 review (localai-org-maint-bot) blocked the landing with two findings against the serving-completion change. P1: the adapter's global tojson override sorted object keys for every model, but the pinned transformers 5.14.1 renderer overrides Jinja's tojson with sort_keys=False / ensure_ascii=False / no HTML escaping (utils/chat_template_utils.py:481), so the override rendered prompt bytes no serving reference produces; the reference fixtures had been captured with plain jinja2, which certifies the same incorrect reference (sorted keys, HTML escaped). P2: tests/vllm/entrypoints/test_kolibri1_chat_template.cpp loaded /mnt/models/Aleph-Alpha/Kolibri-1/tokenizer_config.json in both the rendering and detection cases, so a clean checkout could not run the gate, and the fixture generator hard-coded its output under /tmp/vllm-kolibri-serve. + +REPAIR LANDED on row/kolibri-serve 2026-10-09 (review-repair commits): the pinned renderer was MEASURED first (probe through transformers 5.14.1 render_jinja_template over an insertion-ordered {"z":1,"a":2}, nested unsorted tool schemas, Unicode/HTML leaves): the default keeps insertion key order, raw UTF-8, no HTML escaping. chat_template.cpp's tojson is now a port of the pinned filter's full signature (ensure_ascii/indent/separators/sort_keys, CPython json.dumps semantics) instead of the sorted-dump override; FunctionDefinition::parameters is order-preserving (ordered_json) with RestoreToolSchemaOrder re-reading tools from an order-preserving body parse at every chat entry point (api_server, C ABI, run_batch), and BuildTools mirrors pinned vLLM's model_dump tool shape (description/parameters present as null when absent). The kolibri1 references were REGENERATED through the pinned renderer (20 scenarios, including unsorted nested tool schemas, Unicode/HTML content, a description-less tool and unsorted historical tool-call arguments) by a generator that asserts transformers==5.14.1 and writes in-tree; the template input is committed at tests/fixtures/kolibri1-chat-template-tokenizer_config.json (sha256 9ba35d4bd6baa26b66aa75d03a922dfee98b16bb1fa37481b195d247267b0f97) and the test loads it from the fixture directory with no /mnt path. Red-first: 6 of 20 scenarios failed under the corrected reference before the fix (with_tools, with_tools_thinking_off, with_tools_unsorted_unicode, with_tools_unsorted_unicode_thinking_off, with_tools_minimal_no_description, assistant_tool_call_unsorted_arguments). Gates after the repair: test_reasoning_kolibri1 43, test_tool_parser_kolibri1 19, test_kolibri1_chat_template 61, test_chat_template 204 (196 pre-existing plus 8 new non-kolibri tojson guard assertions), test_reasoning_parser_detect 75, test_tool_parser_detect 361, test_reasoning_qwen3 164, test_openai_tool_parsers 64, test_kolibri1 27/234, test_kolibri1_decode_bench anchor 109726, all green; serving/protocol suites (test_openai_serving 1365, test_openai_api_server 1517, test_openai_conformance 252, test_parser_engine_assembly 5038, test_openai_run_batch 16) and the parameters-reading tool-parser suites green; test_capi v27 (gliner fixture load) remains the pre-existing host failure recorded in the evidence doc. W3 not rerun: no forward change. Evidence: docs/bench-evidence/kolibri1-serve-20261008.md "Review repair". diff --git a/.agents/specs/kolibri-1-cpu.md b/.agents/specs/kolibri-1-cpu.md index 74573e43f..714198701 100644 --- a/.agents/specs/kolibri-1-cpu.md +++ b/.agents/specs/kolibri-1-cpu.md @@ -244,17 +244,32 @@ the kolibri1 pre-tokenizer regex, R7 — the encode contract stays owed). Remaining for the row: R7 tokenizer, the aleph-alpha-inference oracle gateability measurement (GPU), GGUF/CUDA/Tenstorrent arms (later rows). -SERVING COMPLETION (2026-10-08, branch `row/kolibri-serve`): the OpenAI -chat path now serves kolibri1 end to end on CPU. The checkpoint's -tokenizer_config.json chat template renders through the minja adapter -byte-identically to CPython jinja2 references on 15 scenarios covering -the plugin's thinking switch (tests/fixtures/ -kolibri1_chat_template_references.json); the one renderer divergence the -gate caught — minja's `tojson` dumped insertion order where jinja2's -default policy sorts keys (DEFAULT_POLICIES["json.dumps_kwargs"] = -{"sort_keys": True}) — is fixed in the adapter with a child-scope -sorted-dump `tojson` (src/vllm/entrypoints/chat_template.cpp). The -kolibri1 reasoning parser ports the plugin's reasoning.py @ 049a6a7bd240: +SERVING COMPLETION (2026-10-08, branch `row/kolibri-serve`; REVIEW-REPAIRED +2026-10-09, PR #3422): the OpenAI chat path now serves kolibri1 end to end +on CPU. The checkpoint's tokenizer_config.json chat template renders through +the minja adapter byte-identically to the PINNED transformers 5.14.1 +renderer on 20 scenarios covering the plugin's thinking switch (tests/ +fixtures/kolibri1_chat_template_references.json, regenerated through +`render_jinja_template` by tests/fixtures/ +gen-kolibri1-chat-template-references.py). The 2026-10-08 version of this +paragraph recorded the opposite tojson decision — a child-scope sorted-dump +`tojson` overriding minja's insertion order, because the references had +been captured with plain jinja2 — and the PR #3422 review falsified it: +the pinned renderer installs its OWN tojson over Jinja's builtin with +sort_keys=False / ensure_ascii=False / no HTML escaping (transformers 5.14.1 +utils/chat_template_utils.py:481), so plain Jinja's default is not the +serving behavior and the sorted override rendered prompt bytes no serving +reference produces. The repair replaced the override with a port of the +pinned filter's full signature (ensure_ascii/indent/separators/sort_keys, +CPython json.dumps semantics) in src/vllm/entrypoints/chat_template.cpp, +made `FunctionDefinition::parameters` order-preserving (ordered_json, with +RestoreToolSchemaOrder re-reading `tools` from an order-preserving body +parse at every chat entry point — api_server, the C ABI, run_batch), and +made BuildTools mirror pinned vLLM's measured `model_dump` tool shape +(description/parameters present as null when absent). The template input is +committed at tests/fixtures/kolibri1-chat-template-tokenizer_config.json so +the gate runs on a clean checkout (no /mnt path). The kolibri1 reasoning +parser ports the plugin's reasoning.py @ 049a6a7bd240: the Qwen3 engine grammar with the starting state derived the way the template switches thinking (reasoning_effort wins and only "none" disables; else a literal enable_thinking false does), threaded from the @@ -267,7 +282,14 @@ test_tool_parser_kolibri1, test_kolibri1_chat_template green (red-first: detection resolved think_auto/hermes and the registry names did not exist before the change); the full host battery and the row's kolibri gates stay green; W3 rerun in a verified quiet window. Evidence: -docs/bench-evidence/kolibri1-serve-20261008.md. Remaining for the row: +docs/bench-evidence/kolibri1-serve-20261008.md. REVIEW-REPAIR GATES +(2026-10-09): test_kolibri1_chat_template 61, test_chat_template 204 +(196 pre-existing + 8 non-kolibri tojson guard assertions), +test_reasoning_parser_detect 75, test_tool_parser_detect 361, +test_reasoning_qwen3 164, test_openai_tool_parsers 64, +test_kolibri1 27/234, test_kolibri1_decode_bench anchor 109726 — all +green; serving/protocol suites and the parameters-reading tool-parser +suites green; W3 not rerun (no forward change). Remaining for the row: the aleph-alpha-inference oracle gateability measurement (GPU), GGUF/CUDA/Tenstorrent arms (later rows). diff --git a/docs/bench-evidence/kolibri1-serve-20261008.md b/docs/bench-evidence/kolibri1-serve-20261008.md index d717134d5..622bf58ff 100644 --- a/docs/bench-evidence/kolibri1-serve-20261008.md +++ b/docs/bench-evidence/kolibri1-serve-20261008.md @@ -27,6 +27,13 @@ template. ## Reference capture +**SUPERSEDED 2026-10-09 by the "Review repair" section below:** the references +were captured with plain jinja2, whose built-in tojson sorts keys and escapes +HTML — NOT the pinned Transformers renderer's tojson, which preserves +insertion order and never HTML-escapes. The regenerated references (20 +scenarios, through the pinned renderer) are the ones the gate compares +against. + - Script: `tests/fixtures/gen-kolibri1-chat-template-references.py`. - Fixture: `tests/fixtures/kolibri1_chat_template_references.json` (15 scenarios: default, `enable_thinking:false`, `reasoning_effort` @@ -43,6 +50,15 @@ template. ## Renderer divergence found and fixed +**SUPERSEDED 2026-10-09 by the "Review repair" section below.** The claim that +"the tojson every transformers/vLLM-served template runs under" sorts keys was +measured against plain jinja2, not the pinned Transformers renderer — and the +pinned renderer overrides Jinja's tojson with `sort_keys=False` +(transformers 5.14.1 `utils/chat_template_utils.py:481`), so the override +below rendered prompt bytes no serving reference produces. See "Review +repair" for the measurement, the corrected adapter and the regenerated +fixtures. + `{{ tool | tojson }}` rendered with minja's insertion-order dump; the jinja2 references sort keys (jinja2 `DEFAULT_POLICIES["json.dumps_kwargs"]` = `{"sort_keys": True}` — the tojson every transformers/vLLM-served template @@ -94,6 +110,154 @@ passes when run from the source dir), the `test_safetensors` RSS-mapping assertion under host memory pressure, and the gliner fixture loads (`test_gliner2_e2e`, `test_capi` v27). +## Review repair (2026-10-09, PR #3422 review blockers P1/P2) + +The review (localai-org-maint-bot) blocked the landing with two findings: + +- **P1:** the adapter's global `tojson` override sorted object keys for every + model, but the pinned Transformers renderer overrides Jinja's filter with + `tojson(..., sort_keys=False)` and installs it in `_compile_jinja_template` + (transformers 5.14.1 `utils/chat_template_utils.py:481`), so plain Jinja's + default is not the serving behavior — and the fixture generator used plain + `jinja2.Environment`, certifying the same incorrect reference. +- **P2:** `tests/vllm/entrypoints/test_kolibri1_chat_template.cpp` loaded + `/mnt/models/Aleph-Alpha/Kolibri-1/tokenizer_config.json` in both the + rendering and detection cases, and the generator hard-coded its output under + `/tmp/vllm-kolibri-serve` — a clean checkout could not run the gate. + +### What the pinned renderer actually does (MEASURED) + +The pinned oracle is `transformers` 5.14.1 (`.agents/oracles/transformers.md`, +the version the pinned vLLM environment resolves), run in a venv with +`jinja2` 3.1.6. Probe: `transformers.utils.chat_template_utils. +render_jinja_template` (the function `apply_chat_template` delegates to, whose +`_compile_jinja_template` installs the override) over an insertion-ordered +`{"z": 1, "a": 2}`, nested unsorted tool schemas, and Unicode/HTML leaves. +Command: `/tmp/tfprobe-venv/bin/python /tmp/probe_tojson.py` (and +`/tmp/probe_tojson2.py` for the option/edge semantics). + +- **Default:** `{"z": 1, "a": 2}` — INSERTION order (`sort_keys=False`), raw + UTF-8 (`ensure_ascii=False`), NO HTML escaping (``, `&` stay raw — Jinja's + builtin escapes them as `\u003c`/`\u0026`, which is exactly what the + transformers override exists to avoid), Python `json.dumps` separators + (`", "` / `": "`), control characters as `\u00xx`, `"`/`\` escaped, `/` raw. +- **Options (all four accepted):** `indent=2` pretty-prints (`{\n "z": 1,\n + "a": 2\n}`, empty containers stay `{}`/`[]`, `indent=0`/`-1` newline with no + spaces, a string indent is the per-level prefix); `sort_keys=True` sorts + recursively; `ensure_ascii=True` escapes non-ASCII as `\uXXXX` with surrogate + pairs; `separators=(',', ':')` replaces the defaults (still between the + newlines when indent is set). +- Plain jinja2's built-in tojson (the OLD reference): sorted keys, `\uXXXX`, + HTML-escaped — different bytes on every multi-key object — and it rejects the + four options (`TypeError: do_tojson() got an unexpected keyword argument`). +- minja's builtin tojson matches the pinned DEFAULT byte-for-byte (ordered_map + objects, nlohmann dump: raw UTF-8, no HTML escape, Python's separators and + indent shape) but accepts only `indent`; the other three options raise + "Unknown argument". + +Pinned vLLM's tool shape was also measured (replication of +`online_renderer.py:178` `[tool.model_dump() for tool in request.tools]` with +the pinned `ChatCompletionToolsParam`/`FunctionDefinition` pydantic models): +fixed field order `type`/`function`, `name`/`description`/`parameters`, with +`description` and `parameters` present as **null** when the request omitted +them, and `parameters` keeping the request document's key order. + +### What changed + +- `src/vllm/entrypoints/chat_template.cpp`: the sorted-dump `tojson` override + is replaced by a port of the pinned filter's FULL signature + (`ensure_ascii`/`indent`/`separators`/`sort_keys`, CPython `json.dumps` + semantics — see the `JsonDumps` helpers and the filter's comment). The + default now keeps insertion order; the four options render the same bytes + as the pin. `BuildTools` mirrors the measured `model_dump` shape (null + `description`/`parameters` when absent) and assigns `parameters` directly. +- `include/vllm/entrypoints/openai/protocol.h` + `protocol.cpp`: + `FunctionDefinition::parameters` is `nlohmann::ordered_json` + (order-preserving; `nlohmann::json` sorts at parse), with an ordered + `from_json` overload and `RestoreToolSchemaOrder`, which re-reads `tools` + from an order-preserving body parse. +- Entry points call it after their regular parse: `api_server.cpp` + (`handle_chat_completions`), `src/capi/vllm_c.cpp` (`ParseChatRequest`, + both `vllm_chat` and `vllm_chat_stream`), `run_batch.cpp`. Tool-less requests + pay nothing; the parsers that only look keys up are untouched (behavior + identical). +- `src/vllm/entrypoints/openai/tool_parsers/step3.cpp`: one pointer type + follows the field (`const nlohmann::ordered_json*`). +- Fixtures REGENERATED through the pinned renderer: + `tests/fixtures/gen-kolibri1-chat-template-references.py` now drives + `render_jinja_template` (asserting `transformers==5.14.1`), reads the + template from the committed fixture config, mirrors vLLM's + `_postprocess_messages` (assistant tool-call arguments parsed to dicts) and + `model_dump` tool shape, and writes in-tree next to itself. The reference + fixture has 20 scenarios (was 15): the 15 original plus + `with_tools_unsorted_unicode`, `with_tools_unsorted_unicode_thinking_off` + (unsorted nested tool schemas + Unicode/HTML), `with_tools_minimal_no_description` + (pins `"description": null`), `assistant_tool_call_unsorted_arguments` + (the template's second tojson site, `tool_call.arguments | tojson`, with + unsorted Unicode/HTML arguments), and `unicode_html_user_message`. +- P2 portability: the template input is committed at + `tests/fixtures/kolibri1-chat-template-tokenizer_config.json` (the + checkpoint's `chat_template`, sha256 + `9ba35d4bd6baa26b66aa75d03a922dfee98b16bb1fa37481b195d247267b0f97` recorded in + the fixture's `fixture_provenance`). The test loads it (and the references) + from `KOLIBRI1_TEMPLATE_FIXTURE_DIR` (= `tests/fixtures` at compile time); + no `/mnt` path remains in any load path — proven by running the test from + `/` with both fixtures copied to `/tmp/p2-proof`: 61/61 PASS with the model + directory absent. +- Non-kolibri guard (`tests/vllm/entrypoints/test_chat_template.cpp`, +8 + assertions): a non-kolibri tool template (the Hermes/Qwen-style tool branch + AND the real Qwen3.5 fixture template) renders an insertion-ordered, + Unicode/HTML tool byte-identical to the pinned renderer; the four tojson + options are pinned byte-for-byte against the pin; an already-alphabetical + schema renders byte-identical before/after the repair (the change alters + only what the pinned renderer actually differs on). + +### Red-first (under the corrected reference, before the fix) + +`test_kolibri1_chat_template` against the regenerated 20-scenario fixture with +the UNMODIFIED adapter: 6 of 20 rendering scenarios FAILED — `with_tools`, +`with_tools_thinking_off`, `with_tools_unsorted_unicode`, +`with_tools_unsorted_unicode_thinking_off`, `with_tools_minimal_no_description`, +`assistant_tool_call_unsorted_arguments` (40 assertions, 34 passed / 6 +failed). The rendered bytes were fully sorted +(`{"function": {"description": ..., "name": ..., "parameters": {"properties": +..., "required": ..., "type": ...}}, "type": "function"}`) where the pinned +renderer emits insertion order. Non-tool scenarios already matched, which +isolated the divergence to `tojson` (and, for the minimal tool, the +model_dump null shape). + +### Gates (this tree, after the repair) + +- `test_reasoning_kolibri1` 43, `test_tool_parser_kolibri1` 19, + `test_kolibri1_chat_template` 61 (20 scenarios + switch-boundary + + detection + fixture REQUIREs), `test_chat_template` 204 (196 pre-existing + + 8 new guard assertions), `test_reasoning_parser_detect` 75, + `test_tool_parser_detect` 361, `test_reasoning_qwen3` 164, + `test_openai_tool_parsers` 64 — all PASS. +- `test_kolibri1` 27/234 PASS; `test_kolibri1_decode_bench` default config + PASS (anchor chain `101807, 109726, …`, last token 109726); + `test_kolibri1_dequant` 10, `test_kolibri1_dequant_cache` 52, + `test_kolibri1_moe_glue` 21, `test_kolibri1_w2` 1608 — all PASS. +- Protocol/serving suites green: `test_openai_serving` 1365, + `test_openai_api_server` 1517, `test_openai_conformance` 252, + `test_parser_engine_assembly` 5038, + `test_openai_api_server_dots3_mm_forward` 16499, `test_openai_run_batch` 16, + `test_input_batch` 232; the parameters-reading tool-parser suites + (`test_deepseek_v32` 143, `test_glm47` 60, `test_minimax_m2_tool` 56, + `test_tool_parser_step3` 31, `test_tool_parser_step3p5` 155, + `test_tool_parser_qwen3_coder` 166, `test_tool_parser_minicpm5` 119, + `test_tool_parser_hy_v3` 40, `test_tool_parser_poolside_v1` 25, + `test_tool_parser_gemma4` 161) all PASS — the ordered field is + behavior-preserving for key-lookup readers. +- `test_capi` 710/711: the one failure is `capi v27` (gliner fixture load), + the pre-existing host failure recorded above, untouched by this change. +- W3 (`test_kolibri1_w3`) NOT rerun: no forward change; the serving path + renders prompts that feed the bench, and the bench's own gate is green. +- Full tree builds clean (881 targets) with the `protocol.h` change; + `scripts/check-agent-record.py` OK (three `engine-matrix.md` api_server.cpp + anchors shifted by the +5-line `RestoreToolSchemaOrder` call were repaired: + 1354→1359, 1365→1370, 1614→1619). + ## Out of scope - The oracle gateability measurement (`vllm serve` with the plugin on a GPU From d93d9d49225dbabf6214aeb4ca2f0609f9d12fa2 Mon Sep 17 00:00:00 2001 From: Luca Barbato Date: Fri, 9 Oct 2026 07:47:41 +0200 Subject: [PATCH 08/12] record(kolibri1): retrigger CI for the updated PR #3422 body The review-repair commits are pushed; the PR body gained the "Review repair" section after that push, and a body edit alone does not rerun the guard jobs. This empty commit retriggers CI so the trailer/style guards read the body that will become the squash commit message. FOLLOWING_AGENTS_PROTOCOL Following-Agents-Protocol: true AI-Assisted: true Assisted-by: AGENT:mistral/mistral-large-4 [maki] From a6679a82a3e6edc7bbc71a09585a5944bcd91810 Mon Sep 17 00:00:00 2001 From: Luca Barbato Date: Fri, 9 Oct 2026 10:34:05 +0200 Subject: [PATCH 09/12] fix(entrypoints): thread the original body text into run_batch's RestoreToolSchemaOrder MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The PR #3422 repair wired RestoreToolSchemaOrder into RunBatch::DispatchChat as RestoreToolSchemaOrder(body, request), but the seam's first parameter is the body TEXT (const std::string&, protocol.h:605) and body is the parsed nlohmann::json object. The call compiles only through nlohmann's implicit get() conversion, which throws json.exception.type_error.302 on every object body, and DispatchChat has no try/catch around it — so every object-body /v1/chat/completions line through RunBatch throws. Clean-head evidence at d93d9d492: test_openai_run_batch 3 of 7 cases throw (test_run_batch.cpp:451/:507/:601); the repair's recorded 'run_batch 16/16' gate counted assertions and missed the throwing cases (stale binary), so the PR's CI lane would be red. ISSUE-LOCAL-01M4FR20MES4HQVBRWJBJ2AVCN. A first in-flow attempt (body.dump()) is wrong and was reverted: body is the key-sorted nlohmann::json, so its dump is already sorted and the order restoration reads sorted text — the crash goes away but the seam no-ops and the request document's tool-schema key order is lost on the batch path. The original request text IS available: RunLine parses request_json and drops it. RunLine now re-serializes the chat body (the line's top-level "body" member) from an order-preserving nlohmann::ordered_json parse of the original line text and threads it through DispatchChat(custom_id, body, body_json); DispatchChat calls RestoreToolSchemaOrder(body_json, request). ordered_json::dump keeps insertion order, so the re-read restores the request document's schema key order into the prompt, matching the api_server and C-ABI entry points, which already pass their raw body strings. The shared seam's signature is unchanged. Red-first: at the committed throwing call, 4 of 8 test_openai_run_batch cases throw type_error.302 (the 3 pre-existing plus the new F1 case) while assertions read 16/16; under the body.dump() mutation the 3 old cases pass but the new F1 case fails both key-order assertions (seam no-op); at this fix 8/8 cases, 89/89 assertions pass. FOLLOWING_AGENTS_PROTOCOL Following-Agents-Protocol: true AI-Assisted: true Assisted-by: AGENT:mistral/mistral-large-4 [maki] --- include/vllm/entrypoints/openai/run_batch.h | 7 ++++++- src/vllm/entrypoints/openai/run_batch.cpp | 21 +++++++++++++++++---- 2 files changed, 23 insertions(+), 5 deletions(-) diff --git a/include/vllm/entrypoints/openai/run_batch.h b/include/vllm/entrypoints/openai/run_batch.h index 9ecc61868..6f8ca05ac 100644 --- a/include/vllm/entrypoints/openai/run_batch.h +++ b/include/vllm/entrypoints/openai/run_batch.h @@ -99,8 +99,13 @@ class RunBatch { BatchRequestOutput RunLine(const std::string& request_json); private: + // `body_json` is the chat body's serialization from an ORDER-PRESERVING + // parse of the original line text: the nlohmann::json `body` sorts object + // keys, and RestoreToolSchemaOrder re-reads `tools` from this text so the + // request document's schema key order survives into the prompt. BatchRequestOutput DispatchChat(const std::string& custom_id, - const nlohmann::json& body); + const nlohmann::json& body, + const std::string& body_json); OpenAIServingChat* chat_; OpenAIServingModels* models_; diff --git a/src/vllm/entrypoints/openai/run_batch.cpp b/src/vllm/entrypoints/openai/run_batch.cpp index 07103f035..cf13ed6d5 100644 --- a/src/vllm/entrypoints/openai/run_batch.cpp +++ b/src/vllm/entrypoints/openai/run_batch.cpp @@ -85,7 +85,8 @@ RunBatch::RunBatch(OpenAIServingChat* chat, OpenAIServingModels* models) : chat_(chat), models_(models) {} BatchRequestOutput RunBatch::DispatchChat(const std::string& custom_id, - const nlohmann::json& body) { + const nlohmann::json& body, + const std::string& body_json) { // Mirrors run_request(openai_serving_chat.create_chat_completion, ...): // check_type_for_url validates the body as a ChatCompletionRequest // (run_batch.py:175-176); a validation failure surfaces through @@ -106,8 +107,11 @@ BatchRequestOutput RunBatch::DispatchChat(const std::string& custom_id, } // Tool schemas keep the request document's key order (the pinned renderer // dumps them into the prompt with sort_keys=False); the nlohmann::json - // parse above sorts object keys, so re-read `tools` order-preserving. - RestoreToolSchemaOrder(body, request); + // parse above sorts object keys, so re-read `tools` order-preserving from + // the body's ORIGINAL text. The argument is a string: passing the json + // object itself would implicitly convert through get() and + // throw type_error.302 on every object body (ISSUE-LOCAL-01M4FR20MES4HQVBRWJBJ2AVCN). + RestoreToolSchemaOrder(body_json, request); // check_model (chat_completion/serving.py; api_server.cpp:186-190). if (models_ != nullptr && !models_->check_model(request.model)) { @@ -184,7 +188,16 @@ BatchRequestOutput RunBatch::RunLine(const std::string& request_json) { s.compare(s.size() - suf.size(), suf.size(), suf) == 0; }; if (url == "/v1/chat/completions") { - return DispatchChat(custom_id, body); + // The chat body is the line's top-level "body" member. RestoreToolSchemaOrder + // re-reads `tools` from the body's ORIGINAL text order (the pinned + // renderer dumps schemas into the prompt with sort_keys=False), and the + // nlohmann::json parse above sorted its keys — so re-serialize the body + // from an order-preserving parse of the original line text. + // ordered_json::dump keeps insertion order, so this text carries the + // request document's key order (ISSUE-LOCAL-01M4FR20MES4HQVBRWJBJ2AVCN). + const nlohmann::ordered_json ordered_line = + nlohmann::ordered_json::parse(request_json); + return DispatchChat(custom_id, body, ordered_line.at("body").dump()); } // Registered endpoint keys (run_batch.py:732-777) whose serving handler is not // wired here yet -> the "does not support endpoint" error (handler_getter -> From 2f535c002a1b15c8717a53a5e8103258ad5fd229 Mon Sep 17 00:00:00 2001 From: Luca Barbato Date: Fri, 9 Oct 2026 10:34:14 +0200 Subject: [PATCH 10/12] test(entrypoints): pin the ordered-parameters seam at the run_batch and api_server entry points MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Scoped re-review F1 of PR #3422 (ISSUE-LOCAL-01M4FR1D7RH7CN5X2R2CAR6N61): RestoreToolSchemaOrder was detected by NO committed test — a no-op passed every gate, and the batch call site was not merely unguarded but crashing (ISSUE-LOCAL-01M4FR20MES4HQVBRWJBJ2AVCN). Two entry-point tests now close the gap, one per affected dispatch: - tests/vllm/entrypoints/openai/test_run_batch.cpp: a new case drives RunBatch::RunLine (the batch entry point) with a RAW JSONL line whose nested chat body is non-alphabetical at every level (tool wrapper, function, and parameters schema with zeta before alpha). The raw string is load-bearing: BatchLine() re-serializes through the key-sorting nlohmann::json and would destroy the order under test. The case asserts (a) no exception and a 200 row — pinning the crash fix — and (b) the prompt seam receives the schema in the request document's key order and not the key-sorted form, captured via a CapturingToolsPrompt seam over the parsed tools (the capture pattern the api_server harness uses). Mutation-proven in both directions: with the committed throwing call restored the case throws type_error.302; with the call's argument mutated to body.dump() it fails both key-order assertions (the sorted dump no-ops the seam). - tests/vllm/entrypoints/openai/test_api_server.cpp: a new case posts an unsorted-schema tools request through the PRODUCTION /v1/chat/completions dispatch (handle_chat_completions) with the real Qwen3.8 fixture template and asserts the rendered prompt bytes carry the document order. Mutation-proven: with RestoreToolSchemaOrder no-op'd at api_server.cpp:400 the case fails both key-order assertions. Gates at this head: test_openai_run_batch 8/8 cases / 89/89 assertions; test_openai_api_server 104/104 cases / 1521/1521 assertions (1517 pre-existing + 4 new). No existing assertion weakened: both additions are new TEST_CASEs; the run_batch file's pre-existing cases are untouched. FOLLOWING_AGENTS_PROTOCOL Following-Agents-Protocol: true AI-Assisted: true Assisted-by: AGENT:mistral/mistral-large-4 [maki] --- .../entrypoints/openai/test_api_server.cpp | 50 ++++++++++ .../entrypoints/openai/test_run_batch.cpp | 98 +++++++++++++++++++ 2 files changed, 148 insertions(+) diff --git a/tests/vllm/entrypoints/openai/test_api_server.cpp b/tests/vllm/entrypoints/openai/test_api_server.cpp index 450abd458..516d2557e 100644 --- a/tests/vllm/entrypoints/openai/test_api_server.cpp +++ b/tests/vllm/entrypoints/openai/test_api_server.cpp @@ -1087,6 +1087,56 @@ TEST_CASE("api_server: a request's chat_template_kwargs reach the renderer " std::string::npos); } +// The ordered-parameters seam (ISSUE-LOCAL-01M4FR1D7RH7CN5X2R2CAR6N61, F1): +// handle_chat_completions parses the body with nlohmann::json, which SORTS +// object keys, and RestoreToolSchemaOrder (api_server.cpp:400) re-reads +// `tools` from an order-preserving parse of the raw body — the pinned +// renderer dumps schemas into the prompt with sort_keys=False. No committed +// test crossed that boundary (the kolibri1 gate drives apply_chat_template +// over C++-constructed tools), so a no-op RestoreToolSchemaOrder passed every +// gate. This case posts an unsorted-schema tools request through the +// PRODUCTION /v1/chat/completions dispatch and asserts the rendered prompt +// carries the schema in the REQUEST DOCUMENT's key order. RED evidence: with +// RestoreToolSchemaOrder mutated to a no-op, the first CHECK fails — the +// prompt carries the key-sorted form instead. +TEST_CASE("api_server: an unsorted tool schema keeps its document key order " + "in the rendered prompt") { + const HfConfig c = MakeConfig(); + const Qwen3_5MoeWeights w = MakeWeights(c); + CapturingTemplatePrompt prompt(ReadTestFixture("qwen38_chat_template.jinja")); + ServerHarness h(c, w, Fixture(), /*enable_force_include_usage=*/false, + ApiServer::kDefaultMaxConcurrentStreams, prompt.fn); + + // The schema keys are deliberately NON-alphabetical at every level of the + // `parameters` object: document order type/properties/required, and + // zeta/alpha inside properties. + const std::string body = + R"({"messages":[{"role":"user","content":"hi"}],)" + R"("max_completion_tokens":4,"temperature":0.0,)" + R"("tools":[{"type":"function","function":{"name":"get_weather",)" + R"("description":"Get the weather for a city.",)" + R"("parameters":{"type":"object","properties":{)" + R"("zeta":{"type":"string"},"alpha":{"type":"string"}},)" + R"("required":["zeta","alpha"]}}}]})"; + ApiServer::DispatchResult r = h.server.handle_chat_completions(body); + + INFO("dispatch body: " << r.body); + REQUIRE(r.status == 200); + // The schema reaches the prompt in the request document's key order + // (CPython json.dumps separators: ", " and ": "). + CHECK(prompt.rendered->find( + "\"parameters\": {\"type\": \"object\", \"properties\": {\"zeta\": " + "{\"type\": \"string\"}, \"alpha\": {\"type\": \"string\"}}, " + "\"required\": [\"zeta\", \"alpha\"]}") != std::string::npos); + // ...and not in the key-sorted form the nlohmann::json parse leaves behind + // without the order-preserving re-read. + CHECK(prompt.rendered->find( + "\"parameters\": {\"properties\": {\"alpha\": {\"type\": " + "\"string\"}, \"zeta\": {\"type\": \"string\"}}, \"required\": " + "[\"zeta\", \"alpha\"], \"type\": \"object\"}") == + std::string::npos); +} + // #1681 review F1. `chat_template_kwargs` is the first request-controlled key // that can reach the render context at all, and the seam it opens is the // conversation itself: bound unfiltered, a request key REPLACED `messages`, so diff --git a/tests/vllm/entrypoints/openai/test_run_batch.cpp b/tests/vllm/entrypoints/openai/test_run_batch.cpp index d93469430..3247d7298 100644 --- a/tests/vllm/entrypoints/openai/test_run_batch.cpp +++ b/tests/vllm/entrypoints/openai/test_run_batch.cpp @@ -419,6 +419,39 @@ json ChatBody() { {"temperature", 0.0}}; } +// The ORDER-ASSERTING prompt seam (ISSUE-LOCAL-01M4FR1D7RH7CN5X2R2CAR6N61 +// F1 / ISSUE-LOCAL-01M4FR20MES4HQVBRWJBJ2AVCN): the serving layer hands the +// prompt fn the parsed `tools`, and the pinned renderer dumps their schemas +// into the prompt with sort_keys=False — so the schema's key order AT THIS +// SEAM is the order the prompt carries. The fn records the tools' +// `parameters` serialized order-preserving (ordered_json::dump keeps +// insertion order) and returns an in-vocab string for the engine, the same +// split the api_server capturing seam uses. What it captures is readable; +// what the engine tokenizes stays encodable. +struct CapturingToolsPrompt { + std::shared_ptr rendered = std::make_shared(); + vllm::entrypoints::openai::ChatPromptFn fn; + + CapturingToolsPrompt() + : fn([out = rendered]( + const std::vector& messages, bool, + const std::vector& tools, + const nlohmann::ordered_json&) { + std::string captured; + for (const ChatMessage& m : messages) { + if (m.content.has_value()) captured += *m.content; + } + for (const ChatCompletionToolsParam& t : tools) { + if (t.function.parameters.has_value()) { + captured += "|"; + captured += t.function.parameters->dump(); + } + } + *out = captured; + return std::string("hello"); // in-vocab for the fixture tokenizer + }) {} +}; + // One BatchRequestInput JSONL line. std::string BatchLine(const std::string& custom_id, const std::string& url, const json& body) { @@ -617,3 +650,68 @@ TEST_CASE("run_batch: an unknown model yields a 404 ErrorResponse row") { CHECK(row.error->is_object()); // ErrorResponse object, not a bare string CHECK(row.error->at("error").at("code") == 404); } + +// ─── The batch entry point keeps the request document's tool-schema key order +// (ISSUE-LOCAL-01M4FR1D7RH7CN5X2R2CAR6N61 F1 + +// ISSUE-LOCAL-01M4FR20MES4HQVBRWJBJ2AVCN). RunLine parses the line with +// nlohmann::json, which SORTS object keys, so DispatchChat must re-read +// `tools` from the body's ORIGINAL text (RestoreToolSchemaOrder) for the +// schema order to survive into the prompt — the pinned renderer dumps it +// with sort_keys=False. This case drives RunLine with a RAW line whose nested +// chat body is NON-alphabetical at every level (built as a raw string: +// BatchLine() would re-serialize through nlohmann::json and sort the keys, +// destroying the very order under test). It pins BOTH halves of the seam: +// (a) an object body dispatches to a 200 row — the committed +// RestoreToolSchemaOrder(body, request) call passed the json OBJECT where a +// string is expected and threw json.exception.type_error.302 on every object +// body; (b) the prompt seam receives the schema in the request document's +// key order, not the key-sorted form. RED evidence: (a) fails with the +// throwing call restored; (b) fails under a body.dump() argument — the dump is +// already key-sorted, so the order restoration reads sorted text and the +// seam no-ops. ────────────────────────────────────────────────────────────── +TEST_CASE("run_batch: a tools line keeps the document's schema key order in " + "the prompt (ISSUE-LOCAL-01M4FR1D7RH7CN5X2R2CAR6N61," + " ISSUE-LOCAL-01M4FR20MES4HQVBRWJBJ2AVCN)") { + const HfConfig c = MakeConfig(); + const Qwen3_5MoeWeights w = MakeWeights(c); // named: runner holds a reference + Harness h(c, w, Fixture()); + CapturingToolsPrompt prompt; + OpenAIServingChat serving(h.engine, "test-model", prompt.fn); + RunBatch runner(&serving); + + // Non-alphabetical at every level: the tool wrapper (type before + // function), the function (name/description/parameters), and the parameters + // schema (type/properties/required, zeta before alpha). + const std::string line = + R"({"custom_id":"order-1","method":"POST","url":"/v1/chat/completions",)" + R"("body":{"messages":[{"role":"user","content":"hi"}],)" + R"("max_completion_tokens":4,"temperature":0.0,)" + R"("tools":[{"type":"function","function":{"name":"get_weather",)" + R"("description":"Get the weather for a city.",)" + R"("parameters":{"type":"object","properties":{)" + R"("zeta":{"type":"string"},"alpha":{"type":"string"}},)" + R"("required":["zeta","alpha"]}}}]}})"; + + // (a) No exception; the object body dispatches to a 200 row. + const BatchRequestOutput row = runner.RunLine(line); + CHECK(row.custom_id == "order-1"); + REQUIRE(row.response.has_value()); + CHECK(row.response->status_code == 200); + CHECK_FALSE(row.error.has_value()); + REQUIRE(row.response->body.has_value()); + CHECK(row.response->body->contains("choices")); + + // (b) The prompt seam received the schema in the request document's key + // order (compact ordered_json dump), not the key-sorted form the + // nlohmann::json parse leaves behind without the order-preserving re-read. + REQUIRE(prompt.rendered != nullptr); + CHECK(prompt.rendered->find( + "|{\"type\":\"object\",\"properties\":{\"zeta\":{\"type\":\"string\"}," + "\"alpha\":{\"type\":\"string\"}}," + "\"required\":[\"zeta\",\"alpha\"]}") != std::string::npos); + CHECK(prompt.rendered->find( + "|{\"properties\":{\"alpha\":{\"type\":\"string\"}," + "\"zeta\":{\"type\":\"string\"}}," + "\"required\":[\"zeta\",\"alpha\"],\"type\":\"object\"}") == + std::string::npos); +} From 10ea0f6c5758bffafb0ec1a4902bc5e74daf9ee3 Mon Sep 17 00:00:00 2001 From: Luca Barbato Date: Fri, 9 Oct 2026 11:37:17 +0200 Subject: [PATCH 11/12] =?UTF-8?q?record(kolibri1):=20the=20re-review=20rep?= =?UTF-8?q?air=20=E2=80=94=20F2=20guard=20fixture,=20issues,=20spec,=20evi?= =?UTF-8?q?dence?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The scoped re-review of PR #3422 (ISSUE-LOCAL-01M4FR1D7RH7CN5X2R2CAR6N61) filed two findings; the F1 half landed with the F1 tests in the previous commit. This commit lands F2 and the records the repair invalidates. F2 (mislabeled guard case): the 'byte-stable for already-ordered schemas' case in tests/vllm/entrypoints/test_chat_template.cpp used an AlphabeticalTool fixture that was NOT fully alphabetical (the tool wrapper is insertion-ordered non-alphabetically at type/function and name/description/parameters, and the schema was type/properties/required), so under a sorted tojson mutation the case went red too and the 'alphabetical-schema byte-identity before/after' claim was not demonstrated. The fixture is now genuinely alphabetical at every level (schema properties < required < type; single-property schema needs no order) and the case renders the schema alone ({{ tools[0].function.parameters | tojson }}) because the tool wrapper's field order is the pinned model_dump order, which is NOT alphabetical — so the byte-identity claim is now real and the comment says so. Mutation-proven: with the tojson default mutated to sorted keys (the override the review repair removed) the insertion-order case goes RED while this byte-stable case stays GREEN. No existing assertion was weakened: the full-template exact-bytes rendering remains pinned by the insertion-order case (UnsortedUnicodeTool) and the WeatherTool cases; test_chat_template 45/45 cases, 204/204 assertions. Records: ISSUE-LOCAL-01M4FR20MES4HQVBRWJBJ2AVCN (the confirmed run_batch crash, rewritten — the first in-flow draft prescribed body.dump(), which silences the crash but no-ops the order restoration because the dump is already key-sorted; the landed fix threads the original body text through) and ISSUE-LOCAL-01M4FR1D7RH7CN5X2R2CAR6N61 (F1+F2) are closed with dated resolution evidence. The row spec's ## Now records the re-review repair and its gates; docs/bench-evidence/kolibri1-serve-20261008.md gains a 'Re-review repair' section that corrects the falsified 'test_openai_run_batch 16' gate line (a stale binary: 3 of 7 cases threw type_error.302 at the repair head while assertions read 16/16). Gates at this head: test_openai_run_batch 8/8 cases / 89/89 assertions, test_openai_api_server 104/104 / 1521/1521, test_openai_serving 48/48 / 1365/1365, test_chat_template 45/45 / 204/204, test_kolibri1_chat_template 4/4 / 61/61, test_reasoning_kolibri1 7/7 / 43/43, test_tool_parser_kolibri1 3/3 / 19/19, test_reasoning_parser_detect 8/8 / 75/75, test_tool_parser_detect 16/16 / 361/361, test_kolibri1 27/27 / 234/234, test_kolibri1_decode_bench 1/1 / 2/2 (anchor chain ends 109726) — all PASS; W3 not rerun (no forward change). check-agent-record OK (ANCHOR-ROT=0); agent-issue-index refreshed; agent-preflight.sh --staged exits 0 (five argument-requiring checkers SKIP in this environment, unchanged by this diff). FOLLOWING_AGENTS_PROTOCOL Following-Agents-Protocol: true AI-Assisted: true Assisted-by: AGENT:mistral/mistral-large-4 [maki] --- .../ISSUE-LOCAL-01M4FR1D7RH7CN5X2R2CAR6N61.md | 29 ++++++ .../ISSUE-LOCAL-01M4FR20MES4HQVBRWJBJ2AVCN.md | 81 +++++++++++++++++ .agents/specs/kolibri-1-cpu.md | 33 +++++++ .../bench-evidence/kolibri1-serve-20261008.md | 90 +++++++++++++++++++ tests/vllm/entrypoints/test_chat_template.cpp | 39 ++++---- 5 files changed, 254 insertions(+), 18 deletions(-) create mode 100644 .agents/issues/MODEL-TEXT-kolibri-1-kolibri1-for-causal-lm/ISSUE-LOCAL-01M4FR1D7RH7CN5X2R2CAR6N61.md create mode 100644 .agents/issues/MODEL-TEXT-kolibri-1-kolibri1-for-causal-lm/ISSUE-LOCAL-01M4FR20MES4HQVBRWJBJ2AVCN.md diff --git a/.agents/issues/MODEL-TEXT-kolibri-1-kolibri1-for-causal-lm/ISSUE-LOCAL-01M4FR1D7RH7CN5X2R2CAR6N61.md b/.agents/issues/MODEL-TEXT-kolibri-1-kolibri1-for-causal-lm/ISSUE-LOCAL-01M4FR1D7RH7CN5X2R2CAR6N61.md new file mode 100644 index 000000000..72584f223 --- /dev/null +++ b/.agents/issues/MODEL-TEXT-kolibri-1-kolibri1-for-causal-lm/ISSUE-LOCAL-01M4FR1D7RH7CN5X2R2CAR6N61.md @@ -0,0 +1,29 @@ +ID: ISSUE-LOCAL-01M4FR1D7RH7CN5X2R2CAR6N61 +Title: Scoped re-review of PR #3422's repair: F1 the ordered-parameters seam (RestoreToolSchemaOrder) is detected by no committed test; F2 the 'byte-stable for already-ordered schemas' guard uses a non-alphabetical fixture +Row: MODEL-TEXT-kolibri-1-kolibri1-for-causal-lm +State: CLOSED +Kind: bug +GitHub: - +Mirror: PENDING +Availability: FULL +Created: 2026-10-09 +Updated: 2026-10-09 +Closed: 2026-10-09 + +## Problem + +Two findings from the scoped re-review of the PR #3422 review repair (branch row/kolibri-serve, head d93d9d492). F1 (coverage gap): RestoreToolSchemaOrder — the ordered-parameters seam called at src/vllm/entrypoints/openai/api_server.cpp:400 (handle_chat_completions), src/capi/vllm_c.cpp:518 (ParseChatRequest) and src/vllm/entrypoints/openai/run_batch.cpp:110 (DispatchChat) — is detected by NO committed test: a no-op passes every gate. The kolibri1 chat-template gate drives apply_chat_template directly over C++-constructed tools and never crosses the entry-point parse, so the seam that re-reads tools order-preserving after the key-sorting nlohmann::json parse is unguarded. OWED: a test that posts an unsorted-schema tools request through a chat entry point and asserts the rendered prompt bytes keep the request document's key order. F2 (mislabeled guard case): the 'byte-stable for already-ordered schemas' case in tests/vllm/entrypoints/test_chat_template.cpp uses an AlphabeticalTool fixture that is NOT fully alphabetical — the rendered tool is insertion-ordered non-alphabetically at type/function, name/description/parameters and type/properties/required — so under a sorted mutation the case goes red too and the 'alphabetical-schema byte-identity before/after' sub-claim is not demonstrated. OWED: make the fixture genuinely alphabetical at every level (or relabel the case to what it actually pins; prefer genuinely alphabetical so the byte-identity-before/after claim is real) and correct the comment. + +The same re-review surfaced a CONFIRMED crash the repair introduced on the batch path (RestoreToolSchemaOrder(body, request) throws type_error.302 on every object body) — filed separately as ISSUE-LOCAL-01M4FR20MES4HQVBRWJBJ2AVCN, since the crash and the coverage gap are distinct defects with distinct fixes. + +## Resolution + +Both findings landed on row/kolibri-serve 2026-10-09 (test commit 2f535c002a1b15c8717a53a5e8103258ad5fd229, landed with the F2 guard and +these records in the follow-up records commit — see commit bodies). + +F1 — two entry-point tests now detect a no-op RestoreToolSchemaOrder, one per affected entry point: +- tests/vllm/entrypoints/openai/test_run_batch.cpp, new case 'run_batch: a tools line keeps the document's schema key order in the prompt': drives RunBatch::RunLine (the batch entry point) with a RAW JSONL line whose nested chat body is non-alphabetical at every level (built as a raw string — BatchLine() would re-serialize through the key-sorting nlohmann::json and destroy the order under test). It asserts (a) no exception and a 200 row — the crash fix, ISSUE-LOCAL-01M4FR20MES4HQVBRWJBJ2AVCN — and (b) the prompt seam receives the schema in the request document's key order (captured via a CapturingToolsPrompt seam over the parsed tools, the same capture pattern the api_server harness uses), and NOT the key-sorted form. Mutation-proven in both directions: with the committed throwing call restored the case throws type_error.302 (RED); with the call's argument mutated to body.dump() — the reverted first repair attempt — the case fails both key-order assertions (the sorted dump no-ops the seam); at the fix it passes. +- tests/vllm/entrypoints/openai/test_api_server.cpp, new case 'api_server: an unsorted tool schema keeps its document key order in the rendered prompt': posts an unsorted-schema tools request through the PRODUCTION /v1/chat/completions dispatch (handle_chat_completions) with the real Qwen3.8 fixture template and asserts the rendered prompt bytes carry the document order. Mutation-proven: with RestoreToolSchemaOrder no-op'd at api_server.cpp:400 the case fails both key-order assertions (RED); restored, green. +Gates: test_openai_run_batch 8/8 cases / 89/89 assertions; test_openai_api_server 104/104 cases / 1521/1521 assertions (1517 pre-existing + 4 new). + +F2 — the 'byte-stable for already-ordered schemas' case (tests/vllm/entrypoints/test_chat_template.cpp) now uses a genuinely alphabetical fixture at every level: the schema is {"properties": {"city": ...}, "required": [...], "type": "object"} (properties < required < type; the single-property schema needs no order), and the case renders the SCHEMA alone ({{ tools[0].function.parameters | tojson }}) because the tool wrapper's field order (type/function, name/description/parameters) is the pinned model_dump order, which is not alphabetical — so the byte-identity claim is now real and the comment says so. Mutation-proven: with the tojson default mutated to sorted keys (the override the repair removed), the insertion-order case goes RED while this byte-stable case stays GREEN — exactly the before/after agreement it claims to pin. Gate: test_chat_template 45/45 cases / 204/204 assertions. diff --git a/.agents/issues/MODEL-TEXT-kolibri-1-kolibri1-for-causal-lm/ISSUE-LOCAL-01M4FR20MES4HQVBRWJBJ2AVCN.md b/.agents/issues/MODEL-TEXT-kolibri-1-kolibri1-for-causal-lm/ISSUE-LOCAL-01M4FR20MES4HQVBRWJBJ2AVCN.md new file mode 100644 index 000000000..b1f2c56a1 --- /dev/null +++ b/.agents/issues/MODEL-TEXT-kolibri-1-kolibri1-for-causal-lm/ISSUE-LOCAL-01M4FR20MES4HQVBRWJBJ2AVCN.md @@ -0,0 +1,81 @@ +ID: ISSUE-LOCAL-01M4FR20MES4HQVBRWJBJ2AVCN +Title: run_batch: RestoreToolSchemaOrder(body, request) passes a nlohmann::json where a string is expected — every chat batch line throws json.exception.type_error.302 (test_openai_run_batch red at the PR #3422 repair head) +Row: MODEL-TEXT-kolibri-1-kolibri1-for-causal-lm +State: CLOSED +Kind: bug +GitHub: - +Mirror: PENDING +Availability: FULL +Created: 2026-10-09 +Updated: 2026-10-09 +Closed: 2026-10-09 + +## Problem + +Found by the operator's clean-head re-verification of PR #3422 (branch +row/kolibri-serve, head d93d9d492), after the scoped re-review F1 finding +(ISSUE-LOCAL-01M4FR1D7RH7CN5X2R2CAR6N61) showed the ordered-parameters seam +was detected by no committed test. The repair commit 5983c6918 wired +RestoreToolSchemaOrder into RunBatch::DispatchChat as +RestoreToolSchemaOrder(body, request) (src/vllm/entrypoints/openai/run_batch.cpp:110), +but the function's first parameter is const std::string& +(include/vllm/entrypoints/openai/protocol.h:605) and body is a +const nlohmann::json&. The call compiles only through nlohmann's implicit +operator ValueType() -> get(), which throws +json.exception.type_error.302 ('type must be string, but is object') for every +object body — and DispatchChat has no try/catch around the call, so EVERY +/v1/chat/completions line dispatched through RunBatch throws an uncaught +exception. Operator's clean-head evidence at d93d9d492: +/tmp/build-kolibri-serve/tests/test_openai_run_batch reports 'test cases: 7 | +4 passed | 3 failed', 'assertions: 16 | 16 passed', Status: FAILURE — the three +chat-dispatch cases (test_run_batch.cpp:451, :507, :601 at that head) all throw +type_error.302. The repair's recorded gate 'test_openai_run_batch 16' +(docs/bench-evidence/kolibri1-serve-20261008.md "Review repair"; the PR #3422 +body; commit 5983c6918's message) counted the ASSERTION total (16/16) and +missed the three throwing test cases — a stale binary, so the PR's CI lane +(build-test-cpu runs the full ctest) would be red. The other two call sites +pass the raw body string and are correct (api_server.cpp:400 +handle_chat_completions(request_body); vllm_c.cpp:518 +ParseChatRequest(request_json)). + +A first in-flow repair attempt (passing body.dump()) is WRONG and was +reverted: body is the key-sorted nlohmann::json (std::map), so its dump is +already alphabetically sorted and the order restoration re-reads sorted text — +the crash goes away but the seam no-ops, and the request document's tool-schema +key order is lost on the batch path (the pinned renderer dumps schemas into +the prompt with sort_keys=False, so the prompt bytes would silently change +order). Mutation-proven below. + +## Resolution + +Landed on row/kolibri-serve 2026-10-09 (fix commit a6679a82a3e6edc7bbc71a09585a5944bcd91810; test commit +2f535c002a1b15c8717a53a5e8103258ad5fd229 — see commit bodies). The original request text IS available: +RunBatch::RunLine(const std::string& request_json) parses it at run_batch.cpp:159 +and drops the text. The fix threads it through: RunLine re-serializes the chat +body — the line's top-level "body" member — from an ORDER-PRESERVING +(nlohmann::ordered_json) parse of the original line text, and passes it to +DispatchChat(custom_id, body, body_json); DispatchChat calls +RestoreToolSchemaOrder(body_json, request). ordered_json::dump keeps insertion +order, so body_json carries the request document's key order and the re-read +restores it into request.tools, exactly as the api_server and C-ABI entry +points already did with their raw body strings. The shared seam's signature is +unchanged. + +Red-first and mutation evidence (all at /tmp/build-kolibri-serve, this host): +- Committed throwing call (head d93d9d492 plus the new F1 test): 4 of 8 cases + throw type_error.302 — the 3 pre-existing cases (test_run_batch.cpp:484/:540/:634 + in the edited file) plus the new F1 case; assertions read '16 | 16 passed' + while 4 cases fail — the stale-binary signature. +- body.dump() mutation (the reverted first attempt): the 3 pre-existing cases + go GREEN (crash silenced) but the new F1 case fails BOTH key-order assertions + — the prompt carries the key-sorted schema, proving the seam no-ops. +- The fix: test_openai_run_batch 8/8 cases, 89/89 assertions, Status: SUCCESS; + the F1 case's prompt carries the document-ordered schema. +Full battery re-verified at the fixed head: test_openai_api_server 1521, +test_openai_serving 1365, test_chat_template 204, test_kolibri1_chat_template +61, test_reasoning_kolibri1 43, test_tool_parser_kolibri1 19, +test_reasoning_parser_detect 75, test_tool_parser_detect 361, test_kolibri1 +27/234, test_kolibri1_decode_bench anchor 109726 — all PASS; W3 not rerun (no +forward change). The repair's 'test_openai_run_batch 16' gate line in +docs/bench-evidence/kolibri1-serve-20261008.md is falsified by this evidence and +corrected in that document's re-review section. diff --git a/.agents/specs/kolibri-1-cpu.md b/.agents/specs/kolibri-1-cpu.md index 714198701..cd375281e 100644 --- a/.agents/specs/kolibri-1-cpu.md +++ b/.agents/specs/kolibri-1-cpu.md @@ -293,6 +293,39 @@ suites green; W3 not rerun (no forward change). Remaining for the row: the aleph-alpha-inference oracle gateability measurement (GPU), GGUF/CUDA/Tenstorrent arms (later rows). +RE-REVIEW REPAIR (2026-10-09, PR #3422, branch row/kolibri-serve): the +operator's clean-head re-verification at d93d9d492 found the review repair +itself RED — test_openai_run_batch failed 3 of 7 cases, all throwing +json.exception.type_error.302, because DispatchChat called +RestoreToolSchemaOrder(body, request) with the parsed json OBJECT where the +seam takes the body TEXT (the repair's recorded 'run_batch 16/16' gate counted +assertions and missed the throwing cases — a stale binary). +ISSUE-LOCAL-01M4FR20MES4HQVBRWJBJ2AVCN. Fixed by threading the original +request text through: RunLine re-serializes the chat body from an +order-preserving ordered_json parse of the original line and DispatchChat +calls RestoreToolSchemaOrder(body_json, request) — a first body.dump() +attempt was reverted because the sorted dump no-ops the order restoration. +The scoped re-review's two findings also landed +(ISSUE-LOCAL-01M4FR1D7RH7CN5X2R2CAR6N61): F1, the ordered-parameters seam +is now detected at two entry points (a new test_run_batch case drives +RunLine with a raw non-alphabetical tools line and asserts the 200 row and +the document-ordered schema at the prompt seam; a new test_api_server case +posts an unsorted-schema request through the production dispatch — both +mutation-proven against a no-op seam and against the throwing call); F2, +the 'byte-stable for already-ordered schemas' guard's AlphabeticalTool +fixture is now genuinely alphabetical at every level and renders the schema +alone, so its before/after byte-identity claim is real (mutation-proven: +stays green under a sorted tojson while the insertion-order case goes red). +RE-REVIEW GATES (2026-10-09, fixed head): test_openai_run_batch 8/8 cases / +89 assertions (7 pre-existing + the new F1 case), test_openai_api_server +1521, test_openai_serving 1365, test_chat_template 204, +test_kolibri1_chat_template 61, test_reasoning_kolibri1 43, +test_tool_parser_kolibri1 19, test_reasoning_parser_detect 75, +test_tool_parser_detect 361, test_kolibri1 27/234, +test_kolibri1_decode_bench anchor 109726 — all green; W3 not rerun (no +forward change). Evidence: docs/bench-evidence/kolibri1-serve-20261008.md +"Re-review repair". + ## R7 resolution — the tokenizer engine accepts the Kolibri-1 split regex ### Scope — what is actually in the file diff --git a/docs/bench-evidence/kolibri1-serve-20261008.md b/docs/bench-evidence/kolibri1-serve-20261008.md index 622bf58ff..89186b84e 100644 --- a/docs/bench-evidence/kolibri1-serve-20261008.md +++ b/docs/bench-evidence/kolibri1-serve-20261008.md @@ -258,6 +258,96 @@ model_dump null shape). anchors shifted by the +5-line `RestoreToolSchemaOrder` call were repaired: 1354→1359, 1365→1370, 1614→1619). +## Re-review repair (2026-10-09, PR #3422 scoped re-review + the operator's clean-head check) + +The operator's clean-head re-verification at the review-repair head +`d93d9d492` found the repair itself RED, and the "Gates (this tree, after +the repair)" list above is corrected by this section: **`test_openai_run_batch` +16 was a STALE BINARY.** At `d93d9d492` the suite fails 3 of 7 cases — +`test_run_batch.cpp:451`, `:507`, `:601` all throw +`[json.exception.type_error.302] type must be string, but is object` — while +the doctest summary still reads `assertions: 16 | 16 passed`, which is what +the repair's gate report counted. The PR's CI lane (build-test-cpu runs the +full ctest) would be red. ISSUE-LOCAL-01M4FR20MES4HQVBRWJBJ2AVCN. + +**The crash.** The repair wired `RestoreToolSchemaOrder(body, request)` into +`RunBatch::DispatchChat` (run_batch.cpp:110), but the seam's first parameter +is the body TEXT (`const std::string&`, protocol.h:605) and `body` is the +parsed `nlohmann::json` object; the implicit `get()` conversion +throws on every object body, and `DispatchChat` has no try/catch around the +call, so every object-body chat line through `RunBatch` throws. The other two +call sites pass the raw body string and were already correct +(api_server.cpp:400, vllm_c.cpp:518). + +**The fix** (commit `a6679a82a`). The original request text IS available: +`RunLine(const std::string& request_json)` parses it and dropped the text. +`RunLine` now re-serializes the chat body (the line's top-level `body` +member) from an ORDER-PRESERVING `nlohmann::ordered_json` parse of the +original line text and threads it through +`DispatchChat(custom_id, body, body_json)`; `DispatchChat` calls +`RestoreToolSchemaOrder(body_json, request)`. `ordered_json::dump` keeps +insertion order, so the re-read restores the request document's schema key +order into the prompt, matching the other entry points. A first in-flow +attempt (`body.dump()`) was REVERTED: `body` is the key-sorted +`nlohmann::json`, so its dump is already sorted and the order restoration +reads sorted text — the crash goes away but the seam no-ops and the batch +path loses the document's schema key order. + +**Red-first / mutation evidence** (`/tmp/build-kolibri-serve`, this host): +- Committed throwing call + the new F1 test: 4 of 8 cases throw + type_error.302 (the 3 pre-existing + the new case); assertions read + `16 | 16 passed` while 4 cases fail — the stale-binary signature. +- `body.dump()` mutation: the 3 pre-existing cases go GREEN but the new F1 + case fails BOTH key-order assertions (the prompt carries the key-sorted + schema) — the seam no-op is detected. +- The fix: 8/8 cases, 89/89 assertions, SUCCESS; the F1 case's prompt carries + the document-ordered schema. + +**F1 (coverage gap)** — ISSUE-LOCAL-01M4FR1D7RH7CN5X2R2CAR6N61: the +ordered-parameters seam was detected by NO committed test. Two entry-point +tests now pin it (commit `2f535c002a`): +- `test_run_batch.cpp`, new case "run_batch: a tools line keeps the + document's schema key order in the prompt": drives `RunBatch::RunLine` with + a RAW JSONL line whose nested chat body is non-alphabetical at every level + (the raw string is load-bearing — `BatchLine()` would re-serialize through + the key-sorting `nlohmann::json` and destroy the order under test). Asserts + (a) no exception + a 200 row and (b) the prompt seam receives the schema in + the request document's key order, not the sorted form, captured via a + `CapturingToolsPrompt` seam (the capture pattern the api_server harness + uses). +- `test_api_server.cpp`, new case "api_server: an unsorted tool schema keeps + its document key order in the rendered prompt": posts an unsorted-schema + tools request through the PRODUCTION `/v1/chat/completions` dispatch with + the real Qwen3.8 fixture template and asserts the rendered prompt bytes. + Mutation-proven: with `RestoreToolSchemaOrder` no-op'd at api_server.cpp:400 + the case fails both key-order assertions (RED); restored, green. + +**F2 (mislabeled guard)** — same issue: the "byte-stable for already-ordered +schemas" case's `AlphabeticalTool` fixture was NOT fully alphabetical, so +its before/after byte-identity claim was not demonstrated. The fixture is now +genuinely alphabetical at every level (`properties` < `required` < `type`; +single-property schema) and the case renders the schema alone +(`{{ tools[0].function.parameters | tojson }}`) because the tool wrapper's +`model_dump` field order is not alphabetical. Mutation-proven: with the +`tojson` default mutated to sorted keys (the override the repair removed) the +insertion-order case goes RED while this byte-stable case stays GREEN — the +agreement it claims to pin. `test_chat_template` 45/45 cases, 204/204 +assertions. + +**Gates (this tree, at the FIXED head, CPU-only build, +/tmp/build-kolibri-serve, 2026-10-09):** `test_openai_run_batch` 8/8 cases / +89/89 assertions (7 pre-existing + the new F1 case); `test_openai_api_server` +104/104 / 1521/1521 (1517 pre-existing + 4 new); `test_openai_serving` +48/48 / 1365/1365; `test_chat_template` 45/45 / 204/204; +`test_kolibri1_chat_template` 4/4 / 61/61; `test_reasoning_kolibri1` 7/7 / +43/43; `test_tool_parser_kolibri1` 3/3 / 19/19; +`test_reasoning_parser_detect` 8/8 / 75/75; `test_tool_parser_detect` 16/16 / +361/361; `test_kolibri1` 27/27 / 234/234; `test_kolibri1_decode_bench` 1/1 / +2/2 (anchor chain `101807, 109726, …`, last token 109726) — all PASS. +W3 (`test_kolibri1_w3`) NOT rerun: no forward change. +`scripts/check-agent-record.py` OK (ANCHOR-ROT=0); +`scripts/agent-preflight.sh --staged` exits 0. + ## Out of scope - The oracle gateability measurement (`vllm serve` with the plugin on a GPU diff --git a/tests/vllm/entrypoints/test_chat_template.cpp b/tests/vllm/entrypoints/test_chat_template.cpp index 361fcf790..92979af86 100644 --- a/tests/vllm/entrypoints/test_chat_template.cpp +++ b/tests/vllm/entrypoints/test_chat_template.cpp @@ -689,16 +689,21 @@ std::vector UnsortedUnicodeTool() { return {t}; } -// Every level already alphabetical: sorted and insertion order agree, so the -// sorted-dump override the adapter carried BEFORE the review repair and the -// pinned renderer's insertion-order tojson produce the SAME bytes here. +// The schema is alphabetical at EVERY level (properties < required < type; +// the single-property schema needs no order), so the sorted-dump override the +// adapter carried BEFORE the review repair and the pinned renderer's +// insertion-order tojson produce the SAME bytes for it. The tool WRAPPER's +// field order (type/function, name/description/parameters) is the pinned +// renderer's fixed model_dump order — NOT alphabetical — so the byte-identity +// claim can only be demonstrated on the schema itself, which is what the +// case below renders. std::vector AlphabeticalTool() { ChatCompletionToolsParam t; t.type = "function"; t.function.name = "get_weather"; t.function.description = "Get the weather for a city."; t.function.parameters = nlohmann::ordered_json::parse( - R"({"type":"object","properties":{"city":{"type":"string"}},"required":["city"]})"); + R"({"properties":{"city":{"type":"string"}},"required":["city"],"type":"object"})"); return {t}; } @@ -728,22 +733,20 @@ TEST_CASE("chat_template: tojson keeps the pinned renderer's insertion order") { } TEST_CASE("chat_template: tojson is byte-stable for already-ordered schemas") { - // The repair must not alter a non-kolibri model's rendering where the - // pinned renderer and the old sorted override already agreed: for this - // alphabetically-ordered tool the bytes are identical before and after. - const std::vector msgs = { - ChatMessage{"user", std::string("weather?")}}; + // The repair must not alter a rendering where the pinned renderer and the + // old sorted override already agreed: for this alphabetically-ordered + // schema the sorted and insertion-order dumps are the same bytes, so this + // case stays green even under a sorted tojson — the byte-identity-before/ + // after claim is real. The schema is rendered alone because the tool + // wrapper's field order (type/function, name/description/parameters) is the + // pinned model_dump order, which is NOT alphabetical. const std::string out = apply_chat_template( - kToolTemplate, msgs, /*add_generation_prompt=*/false, /*bos=*/"", - /*eos=*/"", AlphabeticalTool()); + "{{ tools[0].function.parameters | tojson }}", {}, + /*add_generation_prompt=*/false, /*bos=*/"", /*eos=*/"", + AlphabeticalTool()); CHECK(out == - "<|im_start|>system\n# Tools\n" - "{\"type\": \"function\", \"function\": {\"name\": \"get_weather\", " - "\"description\": \"Get the weather for a city.\", \"parameters\": " - "{\"type\": \"object\", \"properties\": {\"city\": {\"type\": " - "\"string\"}}, \"required\": [\"city\"]}}}" - "<|im_end|>" - "<|im_start|>user\nweather?<|im_end|>"); + "{\"properties\": {\"city\": {\"type\": \"string\"}}, \"required\": " + "[\"city\"], \"type\": \"object\"}"); } TEST_CASE("chat_template: tojson options match the pinned renderer") { From e8ff143eab2cb1d2f31134ce43e0250ae46042b7 Mon Sep 17 00:00:00 2001 From: Luca Barbato Date: Fri, 9 Oct 2026 11:38:10 +0200 Subject: [PATCH 12/12] record(kolibri1): retrigger CI for the updated PR #3422 body The re-review repair commits are pushed; the PR body gained the "Re-review repair" section after that push, and a body edit alone does not rerun the guard jobs. This empty commit retriggers CI so the trailer/style guards read the body that will become the squash commit message. FOLLOWING_AGENTS_PROTOCOL Following-Agents-Protocol: true AI-Assisted: true Assisted-by: AGENT:mistral/mistral-large-4 [maki]