From 6cd5048c4b1e83ce558755f545da007f043f4cf7 Mon Sep 17 00:00:00 2001 From: Tianyao Wu Date: Mon, 5 Oct 2026 17:58:15 +0800 Subject: [PATCH] open_jev: support the Open-Jev-9B checkpoint Open-Jev-9B uses the same method, prompt format and head design as Open-Jev-27B-v1.1. Select the pinned revisions, request alias and backbone dimensions by the export's model_id so one worker serves either checkpoint, and key the export script's pins and head width the same way. Add the 9B recipe steps and its reference comparison. 9B runs one forward pass per candidate: candidate packing is validated on 27B only. --- .../references/repository-contracts.md | 2 +- README.md | 15 +-- docs/architecture.md | 12 +- docs/supported-models.md | 1 + mkdocs.yml | 3 +- recipe/README.md | 6 +- recipe/open_jev/export_merged.py | 34 ++++-- recipe/open_jev/native.md | 65 ++++++++--- recipe/open_jev/validation-9b.md | 91 ++++++++++++++++ recipe/open_jev/validation.md | 4 +- src/models/open_jev/README.md | 20 ++-- src/models/open_jev/native/Cargo.toml | 2 +- src/models/open_jev/native/src/contract.rs | 73 +++++++++++-- src/models/open_jev/native/src/engine.rs | 20 ++-- src/models/open_jev/native/src/executor.rs | 33 ++++-- src/models/open_jev/native/src/lib.rs | 2 +- src/models/open_jev/native/src/main.rs | 5 +- src/models/open_jev/native/src/processing.rs | 17 ++- tests/open_jev/contract.rs | 87 +++++++++++++-- tests/open_jev/data/qwen3_5_9b/config.json | 103 ++++++++++++++++++ tests/open_jev/head.rs | 26 +++++ tests/open_jev/processing.rs | 26 ++++- tests/open_jev/tokenization.rs | 8 +- 23 files changed, 555 insertions(+), 100 deletions(-) create mode 100644 recipe/open_jev/validation-9b.md create mode 100644 tests/open_jev/data/qwen3_5_9b/config.json diff --git a/.agents/skills/system1-omni-review/references/repository-contracts.md b/.agents/skills/system1-omni-review/references/repository-contracts.md index 833ab2c5..b6ace00b 100644 --- a/.agents/skills/system1-omni-review/references/repository-contracts.md +++ b/.agents/skills/system1-omni-review/references/repository-contracts.md @@ -29,7 +29,7 @@ Read `src/models/open_jev/README.md`, `src/models/open_jev/native/src/contract.r - The reference compiler/readout is pinned to Open-Jev `3308a15ccd7eea1df7a37d6ddc39b023b801ba16`; backbone and checkpoint identities are checked in the exported manifest. Cua-S1's prompt, option limit, JSON ordering, and confidence formula do not apply to this model. - Each candidate has an independent prompt and scalar head score. Preserve candidate order and sorted structured-input rendering. Normalize across each complete question using the saved temperature; `noul` uses logits `[0, score]`. Candidate regrouping must preserve question identity and token usage. -- The worker validates and tokenizes every candidate before inference, warms up before binding, and uses the shared Qwen executor/CUDA ABI. Current scoring is independent single-prompt execution with a CPU head. Use `recipe/open_jev/validation.md` for the scope and limitations of full-checkpoint comparisons. +- The worker validates and tokenizes every candidate before inference, warms up before binding, and uses the shared Qwen executor/CUDA ABI. Current scoring is independent single-prompt execution with a CPU head. Use `recipe/open_jev/validation.md` (27B on H200) and `recipe/open_jev/validation-9b.md` (9B) for the scope and limitations of full-checkpoint comparisons. ## CUDA, graph and compatibility evidence diff --git a/README.md b/README.md index cbcb6333..62c30ed0 100644 --- a/README.md +++ b/README.md @@ -43,8 +43,8 @@ scheduling from model execution; a native Metal backend is planned, while LAYA already has a Python worker for Apple GPUs through PyTorch MPS. The Rust frontend forwards requests to a separately running model worker. The -Cua-S1 4B 0.2 `text` adapter and Open-Jev-27B-v1.1 have native workers using -shared CUDA kernels in this repository. +Cua-S1 4B 0.2 `text` adapter, Open-Jev-27B-v1.1 and Open-Jev-9B have native +workers using shared CUDA kernels in this repository. ## News @@ -64,9 +64,9 @@ shared CUDA kernels in this repository. learned heads, device state, and kernel selection. Native workers have separate processing and executor modules, with shared FIFO admission and blocking dispatch in the [native runtime](src/runtime/README.md). -- **Native CUDA workers.** The Cua-S1 4B 0.2 `text` adapter and - Open-Jev-27B-v1.1 run as native workers with shared CUDA kernels. Cua-S1 - also has a Python worker that serves as the correctness reference. +- **Native CUDA workers.** The Cua-S1 4B 0.2 `text` adapter, + Open-Jev-27B-v1.1 and Open-Jev-9B run as native workers with shared CUDA + kernels. Cua-S1 also has a Python worker that serves as the correctness reference. - **LAYA text serving.** LAYA runs as an external CPU Python worker, the in-repository Python MPS/CPU worker, or a native Rust/CUDA worker on Hopper. - **CUDA backend and planned Metal backend.** High-performance GPU operations @@ -158,8 +158,8 @@ The [frontend documentation](src/frontend/README.md) describes transport and con LAYA text serving uses the upstream CPU worker, the in-repository Python MPS/CPU worker, or a native Rust/CUDA worker on Hopper. The Cua-S1 4B 0.2 `text` adapter runs as a Python worker or as a native worker on CUDA. Open-Jev-27B-v1.1 -runs as a native Rust/CUDA worker. Cua-S1 also has a Python screenshot worker, -and CLM has a stub-encoder contract recipe: +and Open-Jev-9B run on the same native Rust/CUDA worker. Cua-S1 also has a +Python screenshot worker, and CLM has a stub-encoder contract recipe: | Model | Status | | --- | --- | @@ -167,6 +167,7 @@ and CLM has a stub-encoder contract recipe: | Cua-S1 4B 0.2 (`text` adapter) | [Python worker](recipe/cua_s1/text.md); [native worker](recipe/cua_s1/native.md), CUDA, run on sm_89 | | Cua-S1 4B 0.2 (`multimodal` adapter) | [Python CUDA worker](src/frontend/cua_s1.py); one PNG/JPEG screenshot, `choice`; native screenshot execution remains in progress | | Open-Jev-27B-v1.1 | [Native Rust/CUDA worker](recipe/open_jev/native.md); eager independent text candidates; [H200 validation](recipe/open_jev/validation.md) | +| Open-Jev-9B | The same [native Rust/CUDA worker](recipe/open_jev/native.md); eager independent text candidates; [reference comparison on sm_89](recipe/open_jev/validation-9b.md) | | CLM-v0.1-8B | [External worker with a CPU stub encoder](recipe/clm/README.md); contract checks only, real Qwen3-8B decisions unverified by this recipe | [Supported models and hardware](docs/supported-models.md) lists the devices diff --git a/docs/architecture.md b/docs/architecture.md index e4612f7a..7710c734 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -9,9 +9,10 @@ design. Concrete input/output types follow each executor's supported layout. The [Rust frontend](../src/frontend/README.md) currently forwards HTTP requests to separately running workers. Cua-S1 and Open-Jev have native Rust/CUDA workers that share the [Qwen3.5/3.8 executor](../src/models/qwen3_5/native/), which accepts -single prompts and bounded packed prefill. Cua-S1 uses single-prompt calls; -Open-Jev packs candidates within one request for input and gate/up GEMMs while -preserving per-sequence mixers and output/down GEMM shapes. +single prompts and bounded packed prefill. Cua-S1 and Open-Jev-9B use +single-prompt calls; Open-Jev-27B-v1.1 packs candidates within one request for +input and gate/up GEMMs while preserving per-sequence mixers and output/down GEMM +shapes. [Laya's native worker](../src/models/laya/README.md) uses a separate Hopper CUDA backend for one complete padded request. All three coordinate independent processors and executors through @@ -40,8 +41,9 @@ and candidate identity, usage, and response metadata outside the executor. | Open-Jev | Token-ID vectors grouped by question, then independent candidate, in request order. | One FP32 learned scalar per candidate in the same grouping. | Add the `noul` false logit of zero, calibrate across each complete question, and restore typed answers, usage, and metadata. | | Laya | One padded request: token IDs, true lengths, question types and ordered option markers; at most 16 questions, 512 tokens per row and 2048 markers. | Per-question FP32 option logits and two action logits copied back after GPU heads. | Calibrate and decode ordered `choice`, `score` and `noul` answers, usage and metadata. | -Cua-S1 input collections are serial work. Open-Jev's model-specific batch adapter -packs up to 16 independent candidates and 4096 tokens per group; longer prompts +Cua-S1 and Open-Jev-9B input collections are serial work. For Open-Jev-27B-v1.1, +Open-Jev's model-specific batch adapter packs up to 16 independent candidates and +4096 tokens per group; longer prompts execute alone. It restores question/candidate grouping before normalization. Laya batches questions within one request. Shared runtime admission precedes blocking dispatch: Cua-S1 admits one question forward at a diff --git a/docs/supported-models.md b/docs/supported-models.md index f9a8539c..9aba8d39 100644 --- a/docs/supported-models.md +++ b/docs/supported-models.md @@ -15,6 +15,7 @@ Models that are being added are also tracked in issues labeled [new model](https | Cua-S1 4B 0.2, `text` adapter | [Native Rust worker](../recipe/cua_s1/native.md) on the [Qwen3.5 CUDA kernels](../src/backends/cuda/qwen3_5/README.md) | Not supported | Validated on compute capability 8.9 ([#19](https://github.com/ThinkFlowLab/system1-omni/pull/19), [#52](https://github.com/ThinkFlowLab/system1-omni/pull/52)) | Not supported | Compute capability 8.0 or newer, the CUDA toolkit to build, weights merged with `export_text_merged.py` | | Cua-S1 4B 0.2, `multimodal` adapter | Reference worker on Transformers and PEFT, [`src/frontend/cua_s1.py`](../src/frontend/cua_s1.py); no recipe yet | Not supported | Validated ([#17](https://github.com/ThinkFlowLab/system1-omni/pull/17), [#18](https://github.com/ThinkFlowLab/system1-omni/pull/18)) | Not supported | The state is one PNG or JPEG image; upstream's `weights.lock.json` next to the base weights | | Open-Jev-27B-v1.1 | [Native Rust/CUDA worker](../recipe/open_jev/native.md) on the shared Qwen3.5/3.8 executor | Not supported | Validated on H200 (sm_90) for the [74 single-candidate workload](../recipe/open_jev/validation.md) | Not supported | Compute capability 8.0 or newer, CUDA toolkit to build, exported merged weights and trained head | +| Open-Jev-9B | The same [native Rust/CUDA worker](../recipe/open_jev/native.md) | Not supported | Validated on compute capability 8.9 against the reference for [253 requests](../recipe/open_jev/validation-9b.md) | Not supported | Compute capability 8.0 or newer, CUDA toolkit to build, exported merged weights and trained head | | CLM-v0.1-8B | [External `clm-serve` recipe](../recipe/clm/README.md) with a CPU stub embeddings server | **Stub-encoder contract checks only** ([#23](https://github.com/ThinkFlowLab/system1-omni/pull/23)); not real Qwen3-8B decisions | Real encoder unverified by the merged recipe | Unverified | Python, upstream CLM and head checkpoint; a real encoder requires a separate embeddings server | - **Validated:** covered by the recipe on `main` or by the checks in the linked merged pull request. diff --git a/mkdocs.yml b/mkdocs.yml index 39e0d744..e276a583 100644 --- a/mkdocs.yml +++ b/mkdocs.yml @@ -66,6 +66,7 @@ nav: - Cua-S1 text worker: recipe/cua_s1/text.md - Cua-S1 native text worker: recipe/cua_s1/native.md - Open-Jev native text worker: recipe/open_jev/native.md - - Open-Jev H200 validation: recipe/open_jev/validation.md + - Open-Jev-27B H200 validation: recipe/open_jev/validation.md + - Open-Jev-9B validation: recipe/open_jev/validation-9b.md - CLM stub-encoder contract: recipe/clm/README.md - Contributing: CONTRIBUTING.md diff --git a/recipe/README.md b/recipe/README.md index 7d64dcf2..8909eb94 100644 --- a/recipe/README.md +++ b/recipe/README.md @@ -13,8 +13,10 @@ For a first real decision, follow the [complete CPU walkthrough](../docs/getting the worker and connect the Rust frontend. - [Cua-S1 4B 0.2 native text worker](cua_s1/native.md): build the CUDA library and the Rust worker, export the merged weights and start the worker. -- [Open-Jev-27B-v1.1 native text worker](open_jev/native.md): export the merged - text backbone and trained decision head, then serve with Rust and CUDA. +- [Open-Jev native text worker](open_jev/native.md): export the merged text + backbone and trained decision head of Open-Jev-27B-v1.1 or Open-Jev-9B, then + serve with Rust and CUDA. [Open-Jev-9B validation](open_jev/validation-9b.md) + compares the 9B worker with the reference. - [CLM behind the frontend](clm/README.md): run CLM's own server behind the frontend on CPU with a stub encoder, and what the response comparison has to allow for. diff --git a/recipe/open_jev/export_merged.py b/recipe/open_jev/export_merged.py index d64cb64f..ef891b68 100644 --- a/recipe/open_jev/export_merged.py +++ b/recipe/open_jev/export_merged.py @@ -1,7 +1,8 @@ -"""Export the pinned Open-Jev-27B-v1.1 text backbone and scalar head on CPU. +"""Export a pinned Open-Jev text backbone and scalar head on CPU. Use the reference environment documented in native.md. This preparation step -needs about 110 GB of host RAM and 52 GB of output storage, without a GPU. +runs without a GPU. Open-Jev-27B-v1.1 needs about 110 GB of host RAM and 52 GB +of output storage; Open-Jev-9B needs about 20 GB of RAM and 16 GB of storage. """ import argparse @@ -12,8 +13,19 @@ from peft import PeftModel from transformers import AutoModelForImageTextToText, AutoTokenizer -BASE_REVISION = "1d4bf0f2ff6012fd82039f2fa52739d0dd7c60c0" -CHECKPOINT_REVISION = "28cf73067d5b337860bbef3c85b8b82ba8730956" +# Base model -> (base revision, checkpoint revision, scalar head width). +CHECKPOINTS = { + "Qwen/Qwen3.8-27B": ( + "1d4bf0f2ff6012fd82039f2fa52739d0dd7c60c0", + "28cf73067d5b337860bbef3c85b8b82ba8730956", + 5120, + ), + "Qwen/Qwen3.5-9B": ( + "c202236235762e1c871ad0ccb60c8ee5ba337b9a", + "47e966881e489511c0c7f5633a9e1960a676a551", + 4096, + ), +} def main(): @@ -24,8 +36,10 @@ def main(): parser.add_argument("--max-length", type=int, default=4096) args = parser.parse_args() config = json.loads((args.checkpoint / "model.json").read_text()) - if config["model_id"] != "Qwen/Qwen3.8-27B" or config["revision"] != BASE_REVISION: - raise ValueError("expected Open-Jev-27B-v1.1's pinned base") + pins = CHECKPOINTS.get(config["model_id"]) + if pins is None or config["revision"] != pins[0]: + raise ValueError("expected Open-Jev-27B-v1.1's or Open-Jev-9B's pinned base") + base_revision, checkpoint_revision, width = pins if args.out.exists(): raise ValueError("output already exists; choose a new export directory") if not 1 <= args.max_length <= 16384: @@ -41,8 +55,8 @@ def main(): raise ValueError("expected a single-user text chat template") prefix, suffix = chat.split(marker) head = torch.load(args.checkpoint / "head.pt", map_location="cpu", weights_only=True) - if head["weight"].shape != (1, 5120) or head["bias"].shape != (1,): - raise ValueError("expected a 5120-wide trained scalar head") + if head["weight"].shape != (1, width) or head["bias"].shape != (1,): + raise ValueError(f"expected a {width}-wide trained scalar head") if not all(torch.isfinite(v).all() for v in head.values()): raise ValueError("non-finite scalar head") full = AutoModelForImageTextToText.from_pretrained( @@ -58,8 +72,8 @@ def main(): # Written last: the native worker refuses incomplete exports or plain base weights. (args.out / "open_jev_export.json").write_text(json.dumps({ "format": "open-jev-text-merged/1", - "model_id": config["model_id"], "base_revision": BASE_REVISION, - "checkpoint_revision": CHECKPOINT_REVISION, "temperature": temperature, + "model_id": config["model_id"], "base_revision": base_revision, + "checkpoint_revision": checkpoint_revision, "temperature": temperature, "max_length": args.max_length, "chat_prefix": prefix, "chat_suffix": suffix, "head_weight": head["weight"].float().reshape(-1).tolist(), "head_bias": head["bias"].float().item(), diff --git a/recipe/open_jev/native.md b/recipe/open_jev/native.md index 245841e5..8b28b37a 100644 --- a/recipe/open_jev/native.md +++ b/recipe/open_jev/native.md @@ -1,4 +1,4 @@ -# Open-Jev-27B-v1.1 native text worker +# Open-Jev native text worker The worker owns request compilation, tokenization, candidate scoring and typed responses in Rust. It uses the native CUDA prefill implementation introduced in @@ -6,6 +6,11 @@ responses in Rust. It uses the native CUDA prefill implementation introduced in under [`src/models/qwen3_5/native/`](../../src/models/qwen3_5/native/). Python is required only to prepare the merged checkpoint. +The worker serves Open-Jev-27B-v1.1 or Open-Jev-9B. Both checkpoints use the +same method, prompt format and head design, each with its own trained head; the +export's `model_id` selects the pinned revisions, the accepted request model +names and the expected backbone dimensions. + It supports `choice` (1–255 candidates), `score` (2–10 levels), and `noul` (yes/no). Each candidate has an independent prompt; the last hidden state goes through Open-Jev's trained FP32 scalar head. Noul uses logits `[0, score]`. @@ -34,9 +39,26 @@ CUDA_VISIBLE_DEVICES='' python recipe/open_jev/export_merged.py \ --out weights/open-jev-27b-merged ``` -CPU export needs roughly 110 GB of RAM and 52 GB of output storage. It merges -LoRA in BF16 and saves the trained head, temperature, and single-user chat -template in `open_jev_export.json`. The worker refuses a plain base checkpoint +For Open-Jev-9B, download its pinned base and checkpoint and export them the +same way: + +```sh +hf download Qwen/Qwen3.5-9B \ + --revision c202236235762e1c871ad0ccb60c8ee5ba337b9a \ + --local-dir weights/Qwen3.5-9B +hf download ZefanCai/Open-Jev-9B \ + --revision 47e966881e489511c0c7f5633a9e1960a676a551 \ + --include 'package/checkpoint/*' --local-dir weights/Open-Jev-9B +CUDA_VISIBLE_DEVICES='' python recipe/open_jev/export_merged.py \ + --base weights/Qwen3.5-9B \ + --checkpoint weights/Open-Jev-9B/package/checkpoint \ + --out weights/open-jev-9b-merged +``` + +CPU export needs roughly 110 GB of RAM and 52 GB of output storage for 27B, +and about 20 GB of RAM and 16 GB of storage for 9B. It merges LoRA in BF16 and +saves the trained head, temperature, and single-user chat template in +`open_jev_export.json`. The worker refuses a plain base checkpoint or an incomplete export. The saved limit defaults to 4096 tokens per candidate; `--max-length` may raise it to 16384. Oversize prompts fail before inference. @@ -54,6 +76,8 @@ OPEN_JEV_MODEL=weights/open-jev-27b-merged \ target/release/omni-open-jev-native ``` +For 9B, set `OPEN_JEV_MODEL=weights/open-jev-9b-merged`. + `OPEN_JEV_HOST` and `OPEN_JEV_PORT` default to `127.0.0.1` and `8000`. `OPEN_JEV_CUDA_LIB` overrides the default library next to the executable. The worker loads all text weights onto visible CUDA device 0, performs a real @@ -69,9 +93,11 @@ curl http://127.0.0.1:8080/v1/systemone \ -H 'Content-Type: application/json' --data-binary @recipe/open_jev/example-request.json ``` -The worker accepts the model's base name `Qwen/Qwen3.8-27B`, `open-jev`, -`jev-latest`, and `open-jev-27b-v1.1`; the response model is the base name, -matching Open-Jev. Error wording and metadata differ from the reference service. +The worker accepts the loaded checkpoint's base name (`Qwen/Qwen3.8-27B` or +`Qwen/Qwen3.5-9B`), `open-jev`, `jev-latest`, and its own alias +(`open-jev-27b-v1.1` or `open-jev-9b`); other names get HTTP 422. The response +and `/health` model is the base name, matching Open-Jev. The checkpoint aliases, +error wording and metadata differ from the reference service. Requests are bounded to 4 MiB, 4096 questions, and 65536 candidate sequences. ## Validation and optimization scope @@ -93,6 +119,9 @@ The frontend mock-worker API coverage is tracked in and CUDA kernel tests are opt-in; the latter require a GPU reservation: ```sh +# Tokenizer cases against either merged export, on CPU: +OPEN_JEV_MODEL=weights/open-jev-9b-merged \ + cargo test --locked -p omni-open-jev-native --lib -- --ignored # Inside a GPU reservation, after building the library: CUA_S1_CUDA_LIB=$PWD/target/release/libqwen3_5_cuda.so \ cargo test --release --locked -p omni-qwen3-5-native --test kernels -- --ignored @@ -115,32 +144,38 @@ kernel tests retain PR #19's The worker reuses PR #19's fused norm, activation, QK/RoPE and chunked Gated DeltaNet operations. Attention's sigmoid gate is fused into its output epilogue, preserving both BF16 rounding points and removing one launch and one output -read/write pass per full-attention layer (16 layers for this model). Residual -RMSNorm keeps thread values in registers at widths 2560/5120. MLP SiLU uses +read/write pass per full-attention layer (16 layers on 27B, 8 on 9B). Residual +RMSNorm keeps thread values in registers at widths 2560/5120; 9B's width 4096 +uses the general kernel. MLP SiLU uses 16-byte BF16 loads/stores when width, stride and pointers permit it, retaining both BF16 rounding points; other layouts use the scalar path. -This recipe leaves `CUA_S1_GRAPH` unset and uses eager prefill. Candidates within -a request are packed in prepared order, up to 16 sequences and 4096 total tokens -per group; longer prompts execute alone without truncation. Input and gate/up +This recipe leaves `CUA_S1_GRAPH` unset and uses eager prefill. For +Open-Jev-27B-v1.1, candidates within a request are packed in prepared order, up +to 16 sequences and 4096 total tokens per group; longer prompts execute alone +without truncation. Input and gate/up projections share GEMMs. Output/down projections preserve their per-prompt shapes and reduction order, and each sequence retains independent attention, positions, convolution and GDN state. Calibration still uses every candidate in its question. The [H200 packing comparison](../../benchmarks/prefill_batching/README.md) records latency, exact output checks and frozen controls. Packing validation covers H200 (sm_90); other CUDA architectures remain unverified. +Open-Jev-9B runs one forward pass per candidate: packing is not validated for it. Set `CUA_S1_GRAPH=1` on the worker to enable CUDA Graph replay. The shared backend retains at most 64 graphs, keyed by ordered sequence token lengths; growing the scratch buffer clears them. A new shape first runs an eager forward to initialize its plans and captures the layer loop for later replay. This adds cost for new lengths, so graph mode remains opt-in. The earlier -single-prompt graph comparison on the 74-case H200 workload measured mean HTTP -latency 2.03% below eager execution after all lengths were warmed. Combined +single-prompt graph comparison with Open-Jev-27B-v1.1 on the 74-case H200 workload +measured mean HTTP latency 2.03% below eager execution after all lengths were +warmed. Combined packing and graph performance remains unmeasured. Tokenization, transfers and the CPU scalar head remain outside the graph. Prefix sharing, GEMM autotuning, quantization and multimodal inference are not implemented. The -[H200 validation](validation.md) reports full-checkpoint results for 74 +[H200 validation](validation.md) reports Open-Jev-27B-v1.1 results for 74 single-candidate requests, including probability differences and timing variability. It does not establish general accuracy parity or a speedup over OpenJev-Fast; the author's B300 results use different hardware and workloads. +The [Open-Jev-9B validation](validation-9b.md) compares the 9B worker with the +reference on one RTX 6000 Ada; it makes no claim about 27B. diff --git a/recipe/open_jev/validation-9b.md b/recipe/open_jev/validation-9b.md new file mode 100644 index 00000000..cc03ac73 --- /dev/null +++ b/recipe/open_jev/validation-9b.md @@ -0,0 +1,91 @@ +# Open-Jev-9B validation + +## Reference comparison, 2026-10-05 + +The native worker serving the Open-Jev-9B export was compared with the +[Open-Jev reference](https://github.com/Zefan-Cai/Open-Jev/tree/3308a15ccd7eea1df7a37d6ddc39b023b801ba16) +at `3308a15` (uncached) on 253 requests, both with 9B's trained head and saved +temperature (1.8969118766347646). Every request had the same input token count +on both sides. + +| Requests | Questions | Same decision | Largest probability difference | Questions above 0.01 | +| --- | ---: | ---: | ---: | ---: | +| M1 to M3: generated `choice` and `score` requests and Open-Jev's examples (22) | 87 | 87 | 0.041 | 5 | +| M4: JevBench single-candidate `noul` tasks (74) | 74 | 74 | 0.068 | 4 | +| M5: the other public JevBench tasks, 3 to 6 candidates (157) | 157 | 155 | 0.104 | 25 | + +Both changed decisions are close calls on both sides. One is a four-way choice +whose top two options are at 0.403/0.390 native and 0.397/0.400 in the +reference; the other is a three-way choice split almost evenly between two +options, 0.52/0.48 native and 0.44/0.56 in the reference. The largest difference +is on a four-way choice of about 2,240 tokens per candidate. Differences grow +with prompt length: across JevBench, the mean of each question's largest +difference is 0.012 above 2,000 tokens per candidate and 0.003 below 500. + +The final-norm hidden state at each candidate's last token, the input to the +scalar head, was also compared for all 2,157 candidates: + +| Requests | Candidates | Relative L2: median | p99 | largest | Head output: median | p99 | largest | +| --- | ---: | ---: | ---: | ---: | ---: | ---: | ---: | +| M1 to M3 | 1,382 | 0.0093 | 0.020 | 0.027 | 0.024 | 0.18 | 0.36 | +| M4 | 74 | 0.014 | 0.036 | 0.050 | 0.051 | 0.43 | 0.87 | +| M5 | 701 | 0.011 | 0.028 | 0.063 | 0.038 | 0.39 | 1.49 | +| All | 2,157 | 0.0096 | 0.023 | 0.063 | 0.028 | 0.21 | 1.49 | + +The relative L2 difference is ‖native − reference‖ / ‖reference‖ per candidate. +The head output is the scalar before temperature, computed from each side's +hidden state with the same FP64 head. These differences come from two different +implementations: the native worker runs the LoRA adapter merged into BF16 +weights on its CUDA kernels, while the reference applies the adapter unmerged +through PEFT, uses the PyTorch Gated DeltaNet path and pads candidates into +batches of 8. The median relative L2 difference is 0.009 to 0.010 in each of +three prompt-length ranges (below 500 tokens per candidate, 500 to 2,000, and +above 2,000), while the tail of the head output difference grows with length +(p99 0.18 below 500 tokens, 0.38 above 2,000), as do the probability +differences above. + +These results are for 9B on one GPU. They make no claim about 27B. + +### Workload + +- **M1:** one `choice` question with 2, 8, 32, 128 or 255 candidates, over + states of about 256, 1,024 or 3,072 tokens built from the text of Open-Jev's + example states (15 requests). +- **M2:** the four `examples/workflows` requests and `examples/community/support_28` + at the reference revision, 7 to 50 candidates each (5 requests). +- **M3:** one `score` question with 5 or 10 levels over a 1,024-token state + (2 requests). +- **M4 and M5:** the 231 public tasks of + [JevBench](https://github.com/fstandhartinger/jevbench) at + `f8ce71361165846101d02ebc83ad44e47ae44fc3`, one question each: the 74 `noul` + tasks used in the [H200 validation](validation.md), and the 157 tasks with 3 to + 6 candidates. JevBench's license allows publishing aggregate results only, so + requests and responses are not included. + +### Controls and reproduction + +- One NVIDIA RTX 6000 Ada (compute capability 8.9, 48 GB, 300 W limit), CUDA 13.2, + BF16, one request at a time. The card ran near 1 GHz at its power limit during + the long requests; timing is not part of this comparison. +- Native: this change's Rust sources and `Cargo.lock` on `main` `7f39ac4` + (unchanged since the measurement), with the CUDA library built by + `src/backends/cuda/qwen3_5/build.sh 89`, eager execution + (`CUA_S1_GRAPH` unset), requests over HTTP to the worker. The worker used + 15,874 MiB of device memory after warmup and 16,672 MiB after all requests. +- Export: [`export_merged.py`](export_merged.py) on the pinned base and checkpoint + in the [native recipe](native.md), with the default 4096-token limit. It + produced the same bytes as the export used for these measurements, in all 10 + files. +- Reference: `jev.serving.load_predictor` on the 9B checkpoint package, batch + size 8, prefix cache off, with Torch 2.14.0+cu130, Transformers 5.10.2, + PEFT 0.19.1 and Accelerate 1.13.0. Flash Linear Attention is not installed, + so Gated DeltaNet runs on the PyTorch path, as in the H200 HF baseline. +- Hidden states: on the native side, `Model::forward` with the worker's own + prompt rendering and tokenization; on the reference side, the input to its + scalar head, recorded in a separate run with the same predictor settings. + Through each side's head, the native hidden states reproduce the worker's + probabilities and the recorded reference ones reproduce the compared reference + probabilities, both to within 1e-15. + +Request manifests, raw responses, hidden-state dumps and scripts are kept +locally, outside this repository. diff --git a/recipe/open_jev/validation.md b/recipe/open_jev/validation.md index 76232a81..ab48f3cf 100644 --- a/recipe/open_jev/validation.md +++ b/recipe/open_jev/validation.md @@ -1,8 +1,8 @@ -# Open-Jev H200 validation +# Open-Jev-27B-v1.1 H200 validation ## Raw HF Transformers comparison, 2026-10-03 -The native Rust/CUDA worker delivers a **7.47× speedup over raw HF Transformers** +With Open-Jev-27B-v1.1, the native Rust/CUDA worker delivers a **7.47× speedup over raw HF Transformers** by mean warm HTTP latency: **362.21→48.50 ms (86.61% lower)** on one H200. This comparison covers 74 real JevBench `noul` requests, one candidate each, 80–3399 tokens, BF16, max length 16384 and concurrency 1. Each backend reuses diff --git a/src/models/open_jev/README.md b/src/models/open_jev/README.md index 01fe199a..48fdbe56 100644 --- a/src/models/open_jev/README.md +++ b/src/models/open_jev/README.md @@ -1,9 +1,13 @@ -# Open-Jev-27B-v1.1 +# Open-Jev -The [native Rust/CUDA worker](../../../recipe/open_jev/native.md) uses the -pinned Qwen3.8-27B backbone, merged LoRA adapter, trained scalar decision head, -and saved calibration temperature. It supports choice, ordinal score, and yes/no -text decisions through the existing Rust frontend. +The [native Rust/CUDA worker](../../../recipe/open_jev/native.md) serves +Open-Jev-27B-v1.1 on its pinned Qwen3.8-27B backbone or Open-Jev-9B on its +pinned Qwen3.5-9B backbone, with the merged LoRA adapter, trained scalar decision +head, and saved calibration temperature. Both checkpoints use the same method, +prompt format and head design, each with its own trained head; the export's +`model_id` selects the checkpoint's pinned revisions, request model names and +backbone dimensions. It supports choice, ordinal score, and yes/no text +decisions through the existing Rust frontend. The request compiler and response formulas follow [Open-Jev @ 3308a15](https://github.com/Zefan-Cai/Open-Jev/tree/3308a15ccd7eea1df7a37d6ddc39b023b801ba16). @@ -25,12 +29,14 @@ calibration, usage, and metadata. The executor owns the trained head and returns one FP32 scalar per candidate. Response finishing adds the `noul` false baseline and applies calibrated normalization across each complete question. The worker retains request-wide model locking and independent sequence semantics, with -the scalar head on the CPU after CUDA prefill. Its [batch adapter](native/src/batching.rs) -packs at most 16 candidates and 4096 tokens per group, in prepared order; +the scalar head on the CPU after CUDA prefill. For Open-Jev-27B-v1.1, its +[batch adapter](native/src/batching.rs) packs at most 16 candidates and 4096 +tokens per group, in prepared order; longer individual prompts execute alone without truncation. Input and gate/up projections share packed GEMMs. Output/down projections retain their original per-prompt GEMM shapes; attention, positions, convolution and GDN state reset at each sequence boundary. Results are regrouped before question normalization. +Open-Jev-9B runs candidates one at a time until packing is validated for it. The engine owns a [shared serial scheduler](../../runtime/README.md) that admits the complete request before blocking dispatch. Cross-request batching and shared queue budgets remain planned. diff --git a/src/models/open_jev/native/Cargo.toml b/src/models/open_jev/native/Cargo.toml index 52c15634..35566211 100644 --- a/src/models/open_jev/native/Cargo.toml +++ b/src/models/open_jev/native/Cargo.toml @@ -3,7 +3,7 @@ name = "omni-open-jev-native" version = "0.1.0" edition = "2024" publish = false -description = "Native Rust/CUDA worker for Open-Jev-27B-v1.1" +description = "Native Rust/CUDA worker for Open-Jev-27B-v1.1 and Open-Jev-9B" [dependencies] anyhow = "1.0.100" diff --git a/src/models/open_jev/native/src/contract.rs b/src/models/open_jev/native/src/contract.rs index 5464d5c7..009aa75a 100644 --- a/src/models/open_jev/native/src/contract.rs +++ b/src/models/open_jev/native/src/contract.rs @@ -5,9 +5,67 @@ use anyhow::{Context, Result, bail, ensure}; use omni_qwen3_5_native::json; use serde_json::{Map, Value, json}; -pub const MODEL_ID: &str = "Qwen/Qwen3.8-27B"; -pub const BASE_REVISION: &str = "1d4bf0f2ff6012fd82039f2fa52739d0dd7c60c0"; -pub const CHECKPOINT_REVISION: &str = "28cf73067d5b337860bbef3c85b8b82ba8730956"; +/// A pinned Open-Jev checkpoint, selected by its export's `model_id`. +pub struct Checkpoint { + pub name: &'static str, + /// The base model, which Open-Jev also returns as the response model. + pub model_id: &'static str, + pub base_revision: &'static str, + pub checkpoint_revision: &'static str, + /// Accepted as the request model, besides `model_id`, `open-jev` and `jev-latest`. + pub alias: &'static str, + /// Hidden, intermediate, layers, attention, KV, linear key and linear value heads. + pub backbone: (usize, usize, usize, usize, usize, usize, usize), + /// Pack a request's candidates for shared GEMMs. Packing was validated against + /// per-candidate execution on 27B only, so 9B runs candidates one at a time. + pub pack_candidates: bool, +} + +pub static CHECKPOINTS: [Checkpoint; 2] = [ + Checkpoint { + name: "Open-Jev-27B-v1.1", + model_id: "Qwen/Qwen3.8-27B", + base_revision: "1d4bf0f2ff6012fd82039f2fa52739d0dd7c60c0", + checkpoint_revision: "28cf73067d5b337860bbef3c85b8b82ba8730956", + alias: "open-jev-27b-v1.1", + backbone: (5120, 17408, 64, 24, 4, 16, 48), + pack_candidates: true, + }, + Checkpoint { + name: "Open-Jev-9B", + model_id: "Qwen/Qwen3.5-9B", + base_revision: "c202236235762e1c871ad0ccb60c8ee5ba337b9a", + checkpoint_revision: "47e966881e489511c0c7f5633a9e1960a676a551", + alias: "open-jev-9b", + backbone: (4096, 12288, 32, 16, 4, 16, 32), + pack_candidates: false, + }, +]; + +impl Checkpoint { + /// The pinned checkpoint that produced a merged export manifest. + pub fn from_export(manifest: &Value) -> Result<&'static Self> { + ensure!( + manifest["format"] == "open-jev-text-merged/1", + "expected an Open-Jev merged text export" + ); + let checkpoint = CHECKPOINTS + .iter() + .find(|c| manifest["model_id"] == c.model_id) + .context("expected an Open-Jev-27B-v1.1 or Open-Jev-9B export")?; + ensure!( + manifest["base_revision"] == checkpoint.base_revision + && manifest["checkpoint_revision"] == checkpoint.checkpoint_revision, + "expected a pinned {} export", + checkpoint.name + ); + Ok(checkpoint) + } + + fn serves(&self, model: &str) -> bool { + [self.model_id, "open-jev", "jev-latest", self.alias].contains(&model) + } +} #[derive(Debug, PartialEq)] pub enum Kind { @@ -42,14 +100,13 @@ fn description(value: &Value) -> Result { Ok(render(value)) } -pub fn compile(raw: &[u8]) -> Result> { +pub fn compile(raw: &[u8], checkpoint: &Checkpoint) -> Result> { let request = json::parse(raw).map_err(anyhow::Error::msg)?; ensure!( matches!(request.get("model"), None | Some(Value::Null)) - || matches!( - request["model"].as_str(), - Some(MODEL_ID | "open-jev" | "jev-latest" | "open-jev-27b-v1.1") - ), + || request["model"] + .as_str() + .is_some_and(|model| checkpoint.serves(model)), "requested model is not loaded" ); let state = request.get("state").context("request requires state")?; diff --git a/src/models/open_jev/native/src/engine.rs b/src/models/open_jev/native/src/engine.rs index 5e5d3d66..d6175601 100644 --- a/src/models/open_jev/native/src/engine.rs +++ b/src/models/open_jev/native/src/engine.rs @@ -6,13 +6,12 @@ use anyhow::{Context, Result, ensure}; use omni_runtime::SerialScheduler; use serde_json::Value; -use crate::contract::MODEL_ID; +use crate::contract::Checkpoint; use crate::executor::{DecisionHead, Executor}; use crate::processing::Processor; -pub use crate::contract::{BASE_REVISION, CHECKPOINT_REVISION}; - pub struct Engine { + pub checkpoint: &'static Checkpoint, pub processor: Processor, pub scheduler: SerialScheduler, pub executor: Executor, @@ -24,13 +23,7 @@ impl Engine { &std::fs::read(dir.join("open_jev_export.json")) .context("export the merged checkpoint; see recipe/open_jev/native.md")?, )?; - ensure!( - manifest["format"] == "open-jev-text-merged/1" - && manifest["model_id"] == MODEL_ID - && manifest["base_revision"] == BASE_REVISION - && manifest["checkpoint_revision"] == CHECKPOINT_REVISION, - "expected a pinned Open-Jev-27B-v1.1 export" - ); + let checkpoint = Checkpoint::from_export(&manifest)?; let temperature = manifest["temperature"].as_f64().context("temperature")?; ensure!( temperature.is_finite() && temperature > 0.0, @@ -49,10 +42,11 @@ impl Engine { .as_str() .context("chat_suffix")? .to_owned(); - let head = DecisionHead::load(dir, &manifest)?; - let processor = Processor::load(dir, prefix, suffix, temperature, max_length)?; - let executor = Executor::load(dir, library, head).await?; + let head = DecisionHead::load(dir, &manifest, checkpoint)?; + let processor = Processor::load(dir, checkpoint, prefix, suffix, temperature, max_length)?; + let executor = Executor::load(dir, library, head, checkpoint).await?; Ok(Self { + checkpoint, processor, scheduler: SerialScheduler::default(), executor, diff --git a/src/models/open_jev/native/src/executor.rs b/src/models/open_jev/native/src/executor.rs index cdccb641..b06f085f 100644 --- a/src/models/open_jev/native/src/executor.rs +++ b/src/models/open_jev/native/src/executor.rs @@ -9,6 +9,7 @@ use omni_runtime::SerialScheduler; use serde_json::Value; use crate::batching; +use crate::contract::Checkpoint; #[derive(Clone)] pub(crate) struct DecisionHead { @@ -17,7 +18,7 @@ pub(crate) struct DecisionHead { } impl DecisionHead { - pub(crate) fn load(dir: &Path, manifest: &Value) -> Result { + pub(crate) fn load(dir: &Path, manifest: &Value, checkpoint: &Checkpoint) -> Result { let cfg = Config::load(dir)?; ensure!( ( @@ -28,8 +29,9 @@ impl DecisionHead { cfg.kv_heads, cfg.lin_k_heads, cfg.lin_v_heads - ) == (5120, 17408, 64, 24, 4, 16, 48), - "expected the Qwen3.8-27B backbone dimensions" + ) == checkpoint.backbone, + "expected the {} backbone dimensions", + checkpoint.model_id ); let weights: Vec = manifest["head_weight"] .as_array() @@ -69,20 +71,28 @@ impl DecisionHead { pub struct Executor { model: Arc>, head: DecisionHead, + pack: bool, } impl Executor { - pub(crate) async fn load(dir: &Path, library: &Path, head: DecisionHead) -> Result { + pub(crate) async fn load( + dir: &Path, + library: &Path, + head: DecisionHead, + checkpoint: &Checkpoint, + ) -> Result { let (d, lib) = (dir.to_path_buf(), library.to_path_buf()); let model = tokio::task::spawn_blocking(move || Model::load(&d, &lib)).await??; Ok(Self { model: Arc::new(Mutex::new(model)), head, + pack: checkpoint.pack_candidates, }) } /// Inputs and outputs are grouped by question, then candidate, in prepared order. - /// Admit one whole request and pack independent candidates for shared GEMMs. + /// Admit one whole request; where the checkpoint allows it, pack independent + /// candidates for shared GEMMs, otherwise run them one at a time. pub async fn execute( &self, scheduler: &SerialScheduler, @@ -90,6 +100,7 @@ impl Executor { ) -> Result>> { let model = self.model.clone(); let head = self.head.clone(); + let pack = self.pack; scheduler .run(move || { let mut model = model @@ -97,9 +108,15 @@ impl Executor { .map_err(|_| anyhow::anyhow!("poisoned model"))?; let inputs: Vec<&[u32]> = ids.iter().flatten().map(Vec::as_slice).collect(); let mut scores = Vec::with_capacity(inputs.len()); - for range in batching::ranges(&inputs) { - for last in model.forward_batch(&inputs[range])? { - scores.push(head.score(last)?); + if pack { + for range in batching::ranges(&inputs) { + for last in model.forward_batch(&inputs[range])? { + scores.push(head.score(last)?); + } + } + } else { + for ids in &inputs { + scores.push(head.score(model.forward(ids)?)?); } } let mut scores = scores.into_iter(); diff --git a/src/models/open_jev/native/src/lib.rs b/src/models/open_jev/native/src/lib.rs index 268ac289..aff57e5b 100644 --- a/src/models/open_jev/native/src/lib.rs +++ b/src/models/open_jev/native/src/lib.rs @@ -1,4 +1,4 @@ -//! Open-Jev-27B-v1.1 request compilation, candidate scoring and typed responses. +//! Open-Jev-27B-v1.1 and Open-Jev-9B request compilation, candidate scoring and typed responses. mod batching; pub mod contract; pub mod engine; diff --git a/src/models/open_jev/native/src/main.rs b/src/models/open_jev/native/src/main.rs index 7352bff5..5f712f16 100644 --- a/src/models/open_jev/native/src/main.rs +++ b/src/models/open_jev/native/src/main.rs @@ -11,7 +11,7 @@ use axum::{ response::{IntoResponse, Response}, routing::{get, post}, }; -use omni_open_jev_native::{contract::MODEL_ID, engine::Engine}; +use omni_open_jev_native::engine::Engine; use omni_qwen3_5_native::cuda; use serde_json::json; @@ -83,10 +83,11 @@ async fn main() -> Result<()> { let port: u16 = std::env::var("OPEN_JEV_PORT") .map_or(Ok(8000), |v| v.parse()) .context("OPEN_JEV_PORT")?; + let model = engine.checkpoint.model_id; let app = Router::new() .route( "/health", - get(|| async { Json(json!({"status": "ready", "model": MODEL_ID})) }), + get(move || async move { Json(json!({"status": "ready", "model": model})) }), ) .route("/v1/systemone", post(systemone)) .layer(DefaultBodyLimit::max(4 << 20)) diff --git a/src/models/open_jev/native/src/processing.rs b/src/models/open_jev/native/src/processing.rs index 5e0c564e..8d77f701 100644 --- a/src/models/open_jev/native/src/processing.rs +++ b/src/models/open_jev/native/src/processing.rs @@ -7,9 +7,10 @@ use anyhow::{Result, ensure}; use serde_json::{Map, Value, json}; use tokenizers::Tokenizer; -use crate::contract::{self, BASE_REVISION, Kind, Question}; +use crate::contract::{self, Checkpoint, Kind, Question}; pub struct Processor { + checkpoint: &'static Checkpoint, tokenizer: Tokenizer, prefix: String, suffix: String, @@ -24,6 +25,7 @@ pub struct PreparedRequest { /// The original question/candidate mapping, calibration, usage, and timing. pub struct ResponseContext { + checkpoint: &'static Checkpoint, questions: Vec, input_tokens: usize, candidates: usize, @@ -35,6 +37,7 @@ pub struct ResponseContext { impl Processor { pub(crate) fn load( dir: &Path, + checkpoint: &'static Checkpoint, prefix: String, suffix: String, temperature: f64, @@ -43,6 +46,7 @@ impl Processor { let tokenizer = Tokenizer::from_file(dir.join("tokenizer.json")).map_err(anyhow::Error::msg)?; Ok(Self { + checkpoint, tokenizer, prefix, suffix, @@ -53,7 +57,7 @@ impl Processor { /// Validate every candidate length before inference; never truncate. pub fn prepare(&self, raw: &[u8]) -> Result { - let questions = contract::compile(raw)?; + let questions = contract::compile(raw, self.checkpoint)?; let start = Instant::now(); let inputs = encode_questions( &self.tokenizer, @@ -67,6 +71,7 @@ impl Processor { Ok(PreparedRequest { inputs, context: ResponseContext { + checkpoint: self.checkpoint, questions, input_tokens, candidates, @@ -93,12 +98,14 @@ impl ResponseContext { contract::answer(q, &logits, self.temperature)?, ); } - Ok(json!({"model": contract::MODEL_ID, "answers": answers, + Ok( + json!({"model": self.checkpoint.model_id, "answers": answers, "usage": {"input_tokens": self.input_tokens, "output_tokens": 0}, "metadata": {"method": "native_merged_lora_decision_head", "temperature": self.temperature, "candidate_sequences": self.candidates, "inference_seconds": self.start.elapsed().as_secs_f64(), - "base_revision": BASE_REVISION, "max_length": self.max_length, - "prefix_cache": {"enabled": false, "mode": "independent_candidates"}}})) + "base_revision": self.checkpoint.base_revision, "max_length": self.max_length, + "prefix_cache": {"enabled": false, "mode": "independent_candidates"}}}), + ) } } diff --git a/tests/open_jev/contract.rs b/tests/open_jev/contract.rs index 4c63afa9..7a6d2ab0 100644 --- a/tests/open_jev/contract.rs +++ b/tests/open_jev/contract.rs @@ -1,6 +1,9 @@ -use omni_open_jev_native::contract::{answer, compile}; +use omni_open_jev_native::contract::{CHECKPOINTS, Checkpoint, answer, compile}; use serde_json::{Value, json}; +static OPEN_JEV_27B: &Checkpoint = &CHECKPOINTS[0]; +static OPEN_JEV_9B: &Checkpoint = &CHECKPOINTS[1]; + fn close(actual: &Value, expected: &Value) { match (actual, expected) { (Value::Number(a), Value::Number(b)) => { @@ -25,7 +28,7 @@ fn prompts_and_typed_answers_match_open_jev_reference() { let cases: Value = serde_json::from_str(include_str!("data/contract.json")).unwrap(); for case in cases.as_array().unwrap() { let raw = serde_json::to_vec(&case["request"]).unwrap(); - let questions = compile(&raw).unwrap(); + let questions = compile(&raw, OPEN_JEV_27B).unwrap(); let prompts: Vec<_> = questions.iter().map(|q| &q.prompts).collect(); assert_eq!(json!(prompts), case["prompts"]); let mut answers = serde_json::Map::new(); @@ -59,10 +62,16 @@ fn rejects_malformed_and_unsupported_requests() { r#"{"state":"x","questions":{"q":{"type":"score","instructions":"x","criteria":["one"]}}}"#, r#"{"state":"x","questions":{"q":{"type":"noul","instructions":"x","criteria":{"true":"yes"}}}}"#, ] { - assert!(compile(raw.as_bytes()).is_err(), "accepted {raw}"); + assert!( + compile(raw.as_bytes(), OPEN_JEV_27B).is_err(), + "accepted {raw}" + ); } - let q = - compile(br#"{"state":"x","questions":{"q":{"type":"noul","instructions":"x"}}}"#).unwrap(); + let q = compile( + br#"{"state":"x","questions":{"q":{"type":"noul","instructions":"x"}}}"#, + OPEN_JEV_27B, + ) + .unwrap(); for temperature in [0.0, -1.0, f64::NAN, f64::INFINITY] { assert!(answer(&q[0], &[0.0, 1.0], temperature).is_err()); } @@ -79,18 +88,78 @@ fn choice_and_score_limits_match_reference() { let mut body = json!({"state": "x", "questions": {"q": { "type": "choice", "instructions": "Pick", "criteria": candidates }}}); - let questions = compile(&serde_json::to_vec(&body).unwrap()).unwrap(); + let questions = compile(&serde_json::to_vec(&body).unwrap(), OPEN_JEV_27B).unwrap(); assert_eq!(questions[0].prompts.len(), 255); body["questions"]["q"]["criteria"]["extra"] = Value::Null; - assert!(compile(&serde_json::to_vec(&body).unwrap()).is_err()); + assert!(compile(&serde_json::to_vec(&body).unwrap(), OPEN_JEV_27B).is_err()); body["questions"]["q"] = json!({"type": "score", "instructions": "Rate", "criteria": vec!["level"; 10]}); assert_eq!( - compile(&serde_json::to_vec(&body).unwrap()).unwrap()[0] + compile(&serde_json::to_vec(&body).unwrap(), OPEN_JEV_27B).unwrap()[0] .prompts .len(), 10 ); body["questions"]["q"]["criteria"] = json!(vec!["level"; 11]); - assert!(compile(&serde_json::to_vec(&body).unwrap()).is_err()); + assert!(compile(&serde_json::to_vec(&body).unwrap(), OPEN_JEV_27B).is_err()); +} + +fn export(checkpoint: &Checkpoint) -> Value { + json!({"format": "open-jev-text-merged/1", "model_id": checkpoint.model_id, + "base_revision": checkpoint.base_revision, + "checkpoint_revision": checkpoint.checkpoint_revision}) +} + +#[test] +fn exports_select_their_pinned_checkpoint() { + for checkpoint in [OPEN_JEV_27B, OPEN_JEV_9B] { + let selected = Checkpoint::from_export(&export(checkpoint)).unwrap(); + assert_eq!(selected.model_id, checkpoint.model_id); + } + let mut manifest = export(OPEN_JEV_9B); + manifest["checkpoint_revision"] = json!(OPEN_JEV_27B.checkpoint_revision); + assert!(Checkpoint::from_export(&manifest).is_err()); + let mut manifest = export(OPEN_JEV_9B); + manifest["base_revision"] = json!(OPEN_JEV_27B.base_revision); + assert!(Checkpoint::from_export(&manifest).is_err()); + let mut manifest = export(OPEN_JEV_27B); + manifest["model_id"] = json!("Qwen/Qwen3.5-2B"); + assert!(Checkpoint::from_export(&manifest).is_err()); + let mut manifest = export(OPEN_JEV_27B); + manifest["format"] = json!("open-jev-text-merged/2"); + assert!(Checkpoint::from_export(&manifest).is_err()); +} + +#[test] +fn only_validated_checkpoints_pack_candidates() { + assert!(OPEN_JEV_27B.pack_candidates); + assert!(!OPEN_JEV_9B.pack_candidates); +} + +#[test] +fn request_models_follow_the_loaded_checkpoint() { + let request = |model: &str| { + serde_json::to_vec(&json!({"model": model, "state": "x", + "questions": {"q": {"type": "noul", "instructions": "x"}}})) + .unwrap() + }; + for (checkpoint, other) in [(OPEN_JEV_27B, OPEN_JEV_9B), (OPEN_JEV_9B, OPEN_JEV_27B)] { + for model in [ + checkpoint.model_id, + checkpoint.alias, + "open-jev", + "jev-latest", + ] { + assert!( + compile(&request(model), checkpoint).is_ok(), + "rejected {model}" + ); + } + for model in [other.model_id, other.alias] { + assert!( + compile(&request(model), checkpoint).is_err(), + "accepted {model}" + ); + } + } } diff --git a/tests/open_jev/data/qwen3_5_9b/config.json b/tests/open_jev/data/qwen3_5_9b/config.json new file mode 100644 index 00000000..273ce437 --- /dev/null +++ b/tests/open_jev/data/qwen3_5_9b/config.json @@ -0,0 +1,103 @@ +{ + "architectures": [ + "Qwen3_5ForConditionalGeneration" + ], + "image_token_id": 248056, + "model_type": "qwen3_5", + "text_config": { + "attention_bias": false, + "attention_dropout": 0.0, + "attn_output_gate": true, + "dtype": "bfloat16", + "eos_token_id": 248044, + "full_attention_interval": 4, + "head_dim": 256, + "hidden_act": "silu", + "hidden_size": 4096, + "initializer_range": 0.02, + "intermediate_size": 12288, + "layer_types": [ + "linear_attention", + "linear_attention", + "linear_attention", + "full_attention", + "linear_attention", + "linear_attention", + "linear_attention", + "full_attention", + "linear_attention", + "linear_attention", + "linear_attention", + "full_attention", + "linear_attention", + "linear_attention", + "linear_attention", + "full_attention", + "linear_attention", + "linear_attention", + "linear_attention", + "full_attention", + "linear_attention", + "linear_attention", + "linear_attention", + "full_attention", + "linear_attention", + "linear_attention", + "linear_attention", + "full_attention", + "linear_attention", + "linear_attention", + "linear_attention", + "full_attention" + ], + "linear_conv_kernel_dim": 4, + "linear_key_head_dim": 128, + "linear_num_key_heads": 16, + "linear_num_value_heads": 32, + "linear_value_head_dim": 128, + "max_position_embeddings": 262144, + "mlp_only_layers": [], + "model_type": "qwen3_5_text", + "mtp_num_hidden_layers": 1, + "mtp_use_dedicated_embeddings": false, + "num_attention_heads": 16, + "num_hidden_layers": 32, + "num_key_value_heads": 4, + "rms_norm_eps": 1e-06, + "use_cache": true, + "vocab_size": 248320, + "mamba_ssm_dtype": "float32", + "rope_parameters": { + "mrope_interleaved": true, + "mrope_section": [ + 11, + 11, + 10 + ], + "rope_type": "default", + "rope_theta": 10000000, + "partial_rotary_factor": 0.25 + } + }, + "tie_word_embeddings": false, + "transformers_version": "4.57.0.dev0", + "video_token_id": 248057, + "vision_config": { + "deepstack_visual_indexes": [], + "depth": 27, + "hidden_act": "gelu_pytorch_tanh", + "hidden_size": 1152, + "in_channels": 3, + "initializer_range": 0.02, + "intermediate_size": 4304, + "model_type": "qwen3_5", + "num_heads": 16, + "num_position_embeddings": 2304, + "out_hidden_size": 4096, + "patch_size": 16, + "spatial_merge_size": 2, + "temporal_patch_size": 2 + }, + "vision_end_token_id": 248054, + "vision_start_token_id": 248053 +} \ No newline at end of file diff --git a/tests/open_jev/head.rs b/tests/open_jev/head.rs index f4bf823e..50f0bbcd 100644 --- a/tests/open_jev/head.rs +++ b/tests/open_jev/head.rs @@ -1,4 +1,5 @@ use super::*; +use crate::contract::CHECKPOINTS; #[test] fn scalar_projection_preserves_accumulation_rounding_and_bias_order() { @@ -24,3 +25,28 @@ fn scalar_projection_rejects_nonfinite_scores() { assert!(head.score(vec![2.0]).is_err()); assert!(head.score(vec![f32::NAN]).is_err()); } + +#[test] +fn head_loader_checks_the_selected_checkpoints_backbone() { + // Qwen/Qwen3.8-27B @ 1d4bf0f2 and Qwen/Qwen3.5-9B @ c2022362 configurations. + let tests = Path::new(concat!(env!("CARGO_MANIFEST_DIR"), "/../../../../tests")); + let (qwen_27b, qwen_9b) = ( + tests.join("qwen3_5/data"), + tests.join("open_jev/data/qwen3_5_9b"), + ); + let (open_jev_27b, open_jev_9b) = (&CHECKPOINTS[0], &CHECKPOINTS[1]); + let manifest = + |width: usize| serde_json::json!({"head_weight": vec![0.0; width], "head_bias": 0.0}); + assert!(DecisionHead::load(&qwen_27b, &manifest(5120), open_jev_27b).is_ok()); + assert!(DecisionHead::load(&qwen_9b, &manifest(4096), open_jev_9b).is_ok()); + assert_eq!( + DecisionHead::load(&qwen_9b, &manifest(4096), open_jev_27b) + .err() + .unwrap() + .to_string(), + "expected the Qwen/Qwen3.8-27B backbone dimensions" + ); + assert!(DecisionHead::load(&qwen_27b, &manifest(5120), open_jev_9b).is_err()); + // The trained head must match the selected backbone's width. + assert!(DecisionHead::load(&qwen_9b, &manifest(5120), open_jev_9b).is_err()); +} diff --git a/tests/open_jev/processing.rs b/tests/open_jev/processing.rs index 9f1a5779..95742530 100644 --- a/tests/open_jev/processing.rs +++ b/tests/open_jev/processing.rs @@ -1,4 +1,5 @@ use super::*; +use crate::contract::CHECKPOINTS; use tokenizers::{ models::wordlevel::WordLevel, pre_tokenizers::whitespace::WhitespaceSplit, processors::template::TemplateProcessing, @@ -54,6 +55,7 @@ fn processor(max_length: usize) -> Processor { .unwrap(), )); Processor { + checkpoint: &CHECKPOINTS[0], tokenizer, prefix: "".into(), suffix: "".into(), @@ -123,13 +125,33 @@ fn prepared_candidates_have_exact_ids_and_finished_responses_keep_mapping() { body["metadata"]["method"], "native_merged_lora_decision_head" ); - assert_eq!(body["metadata"]["base_revision"], BASE_REVISION); + assert_eq!( + body["metadata"]["base_revision"], + CHECKPOINTS[0].base_revision + ); assert_eq!( body["metadata"]["prefix_cache"], json!({"enabled":false,"mode":"independent_candidates"}) ); assert!(body["metadata"]["inference_seconds"].as_f64().unwrap() >= 0.0); - assert_eq!(body["model"], contract::MODEL_ID); + assert_eq!(body["model"], CHECKPOINTS[0].model_id); +} + +#[test] +fn responses_name_the_loaded_checkpoint() { + let mut processor = processor(4096); + processor.checkpoint = &CHECKPOINTS[1]; + let body = processor + .prepare(REQUEST) + .unwrap() + .context + .finish(vec![vec![3.0, 3.0], vec![0.0]]) + .unwrap(); + assert_eq!(body["model"], "Qwen/Qwen3.5-9B"); + assert_eq!( + body["metadata"]["base_revision"], + "c202236235762e1c871ad0ccb60c8ee5ba337b9a" + ); } #[test] diff --git a/tests/open_jev/tokenization.rs b/tests/open_jev/tokenization.rs index 43f9a6d7..6e9cf9c6 100644 --- a/tests/open_jev/tokenization.rs +++ b/tests/open_jev/tokenization.rs @@ -9,6 +9,7 @@ fn tokenization_matches_reference_and_rejects_oversize_prompts() { serde_json::from_slice(&std::fs::read(dir.join("open_jev_export.json")).unwrap()).unwrap(); let mut processor = Processor::load( dir, + Checkpoint::from_export(&manifest).unwrap(), manifest["chat_prefix"].as_str().unwrap().to_owned(), manifest["chat_suffix"].as_str().unwrap().to_owned(), manifest["temperature"].as_f64().unwrap(), @@ -17,7 +18,12 @@ fn tokenization_matches_reference_and_rejects_oversize_prompts() { .unwrap(); let cases: Value = serde_json::from_str(include_str!("data/tokenization.json")).unwrap(); for case in cases.as_array().unwrap() { - let raw = serde_json::to_vec(&case["request"]).unwrap(); + let mut request = case["request"].clone(); + if request.get("model").is_some() { + // The fixture names the 27B checkpoint; each export accepts its own alias. + request["model"] = processor.checkpoint.alias.into(); + } + let raw = serde_json::to_vec(&request).unwrap(); processor.max_length = 4096; let prepared = processor.prepare(&raw).unwrap(); assert_eq!(serde_json::json!(prepared.inputs), case["ids"]);