diff --git a/docs/api/build_from_gguf.md b/docs/api/build_from_gguf.md
index e4a3d381e..7ac39b573 100644
--- a/docs/api/build_from_gguf.md
+++ b/docs/api/build_from_gguf.md
@@ -9,7 +9,7 @@ Support is capability-specific: graph import does not imply runtime packaging.
| Census | Total | Closure |
|---|---:|---|
-| Architectures | 148 | graph verdicts: {'deferred': 41, 'rejected': 2, 'supported': 105}; importable: 105; quantized import: {'rejected': 47, 'supported': 101}; runtime: {'deferred': 136, 'rejected': 2, 'supported': 10} |
+| Architectures | 148 | graph verdicts: {'deferred': 41, 'rejected': 2, 'supported': 105}; importable: 105; quantized import: {'rejected': 47, 'supported': 101}; runtime: {'deferred': 135, 'rejected': 2, 'supported': 11} |
| Active stored qtypes | 25 | 24 have an import route; 1 are explicitly deferred with no route |
| Serialized projector strings | 60 | {'graph-importable': 56, 'runtime-evidenced': 0} |
| Tokenizer pre identifiers | 87 | 56 semantic groups; route dispositions: {'deferred-compiled-semantics': 45, 'deferred-pinned-artifact-evidence': 4, 'deferred-pinned-artifact-mismatch': 7, 'validated-pinned-source': 31} |
@@ -88,6 +88,7 @@ network-free selection, budget, exclusions, and fail-closed candidate reasons ar
| Evidence ID | GGUF identity | Config identity | Tokenizer identity | Runtime proof |
|---|---|---|---|---|
+| `apertus-v1.1-1.5b-instruct-bf16-ort-genai-0.15.2` | `MrMeOrYou/Apertus-v1.1-1.5B-Instruct-GGUF@88c75ad49566d3c2157d03709bf772262c3241ed`
`Apertus-v1.1-1.5B-Instruct-BF16.gguf`
3,028,052,608 B
`f9ec154d0ec29dad1f6465b458b7f27bd25ad7b9a3899233ae98ca6d358501c2` | `swiss-ai/Apertus-v1.1-1.5B-Instruct@9e9d01154446a645d30f04174cf1515a38058be7` | `swiss-ai/Apertus-v1.1-1.5B-Instruct@9e9d01154446a645d30f04174cf1515a38058be7`
`chat_template.jinja` 5,250 B `4afab8361a4bd0c2994404e0b0851dbaad461a06e56bdfdb635aae3977473c19`, `special_tokens_map.json` 560 B `9f69883bd70fc5d8b55822799837a216d3ac4fb565e05256a0d4f9850404bbc5`, `tokenizer.json` 17,078,368 B `be12f4375d655cc740864e3a9041bcddd8477942f209d9e7f27f6c8767162638`, `tokenizer_config.json` 177,274 B `77b14a0664585c26065f07d7a4c852a4615c83348d9378e23def01957bbd3f57`
metadata `3097bd9f22efd32db9045c4705d978539dc8adeeae763324062a5e1a73fc24a5` | ONNX Runtime 1.29.0 `CPUExecutionProvider`; ort-genai 0.15.2; result=passed; full-logit; dynamic KV cache prefill, full-sequence replay, rollback, reorder, and 20 decode steps |
| `gpt2-q2-k-ort-genai-0.15.2` | `tensorblock/gpt2-GGUF@5b01870b15c4b2e43695d7f3f3bfb5b26106f23b`
`gpt2-Q2_K.gguf`
81,196,544 B
`4234545f917ec1df10dab4d926796a83422b68e9010d85a4c111b8b541f32892` | `openai-community/gpt2@607a30d783dfa663caf39e06633721c8d4cfcd7e` | `Xenova/gpt2@bf2c7f02e0b826c60d03af341171bde20893da66`
`special_tokens_map.json` 99 B `6f50ab5a5a509a1c309d6171f339b196a900dc9c99ad0408ff23bb615fdae7ad`, `tokenizer.json` 2,107,653 B `cda20b8ca044949aa07ac4078420c80d1a57139d5f9f33700e46fb2d891e7c66`, `tokenizer_config.json` 234 B `551e26ec611d8d0c8edc3ef72e518a38418cb71f40de1347dd486a595e1557d7`
metadata `b2417176025f8500d864004b0bf93b1403dc3c52238f6628f82fb0e3c498977e` | ONNX Runtime 1.29.0 `CPUExecutionProvider`; ort-genai 0.15.2; result=passed; full-logit; dynamic KV cache prefill, full-sequence replay, rollback, reorder, and 20 decode steps; Explicit-float correctness route; source GGUF blocks are dequantized. |
| `lfm2-350m-f16-ort-genai-0.15.2` | `LiquidAI/LFM2-350M-GGUF@8fdc9d526b7ed346b19257551b05816c7912ecc2`
`LFM2-350M-F16.gguf`
711,482,304 B
`379ffdcbf08147c0313f6f1ce7ff558a2bc935eda633f4b46c52347032419c42` | `LiquidAI/LFM2-350M@f37d3f5c8c5484bc01dad379a595cf4c68c4e70e` | `LiquidAI/LFM2-350M@73e3c253078a3b97c2e14b4c4665679f4d9b6d56`
`chat_template.jinja` 209 B `a805e50fed68938a076b07e2e602639611b50b1ced0e50f11eb92f1ba25be4dc`, `special_tokens_map.json` 434 B `742aefe2b7dec496e8caffdba03a75d0c1a9925d53bd3f3e0d388c96b591b6f4`, `tokenizer.json` 4,732,426 B `98cff83b4f6d7e9d8929bebc62b07e92cf1b3f99c80d16bafe8b84a75448f40b`, `tokenizer_config.json` 91,509 B `36f511115e9d8952cbc9d15d9a20dfa7ce7d1444940e5c1dc42a762020c99bf5`
metadata `e5626d605bb50bc53fdb0fbfcf374fb33dfbaa0cc698d9746ba1e9b0b7e6d07d` | ONNX Runtime 1.29.0 `CPUExecutionProvider`; ort-genai 0.15.2; result=passed; full-logit; hybrid convolution and KV state prefill, replay, rollback, reorder, and 20 decode steps |
| `pythia-70m-q2-k-ort-genai-0.15.2` | `mradermacher/pythia-70m-GGUF@52d6f045404c9f93418df2a0144d20c9de34316b`
`pythia-70m.Q2_K.gguf`
38,508,192 B
`8e331c8c8016bed8ff1863b78fafe51e86b2364f32c2d7f3e201687e081cf7f7` | `EleutherAI/pythia-70m@a39f36b100fe8a5377810d56c3f4789b9c53ac42` | `EleutherAI/pythia-70m@a39f36b100fe8a5377810d56c3f4789b9c53ac42`
`special_tokens_map.json` 99 B `6f50ab5a5a509a1c309d6171f339b196a900dc9c99ad0408ff23bb615fdae7ad`, `tokenizer.json` 2,113,710 B `c24618a1b3e6a38167beff1c72cffd126c3a66254347304b50547d12c5f25624`, `tokenizer_config.json` 396 B `70e38394e494931c6f773ba41e19460dd4436526b852207367f04341b4066d3f`
metadata `d5c722f646ff6462ac217da5e3514d1fa8b4a7b33aedced3daa8e7f00cc74f78` | ONNX Runtime 1.29.0 `CPUExecutionProvider`; ort-genai 0.15.2; result=passed; full-logit; dynamic KV cache prefill, full-sequence replay, rollback, reorder, and 20 decode steps; Explicit-float correctness route on the portable graph; source GGUF blocks are dequantized and the pinned tokenizer vocabulary is extended only with deterministic padding IDs present in the GGUF. |
@@ -129,7 +130,7 @@ remain machine-readable in `_route_census.py`; this table groups only shared nex
| `dependency-or-mobius-abi-blocked` | `projector-package-abi` | `projector:resampler` | Mobius dynamic processor-to-graph media shape contract |
| `dependency-or-mobius-abi-blocked` | `tokenizer-compiled-semantics` | `tokenizer:afmoe`, `tokenizer:bloom`, `tokenizer:chameleon`, `tokenizer:codeshell`, `tokenizer:command-r`, `tokenizer:dbrx`, `tokenizer:deepseek-coder`, `tokenizer:deepseek-llm`, `tokenizer:deepseek-v3`, `tokenizer:default`, `tokenizer:exaone`, `tokenizer:exaone-moe`, `tokenizer:falcon`, `tokenizer:gpt3-finnish`, `tokenizer:granite-docling`, `tokenizer:granite-embed-multi-97m`, `tokenizer:grok-2`, `tokenizer:hunyuan`, `tokenizer:hunyuan-dense`, `tokenizer:jais`, `tokenizer:jais-2`, `tokenizer:joyai-llm`, `tokenizer:kimi-k2`, `tokenizer:laguna`, `tokenizer:megrez`, `tokenizer:mellum2`, `tokenizer:minerva-7b`, `tokenizer:minicpm5`, `tokenizer:minimax-m2`, `tokenizer:mpt`, `tokenizer:olmo`, `tokenizer:poro-chat`, `tokenizer:refact`, `tokenizer:sarvam-moe`, `tokenizer:seed-coder`, `tokenizer:smaug-bpe`, `tokenizer:solar-open`, `tokenizer:stablelm2`, `tokenizer:starcoder`, `tokenizer:superbpe`, `tokenizer:tekken`, `tokenizer:trillion`, `tokenizer:viking`, `tokenizer:whitespace`, `tokenizer:youtu` | compiled pinned llama.cpp oracle; dispatch-equivalence fixture |
| `dependency-or-mobius-abi-blocked` | `tokenizer-compiled-semantics` | `tokenizer:bailingmoe`, `tokenizer:bailingmoe2`, `tokenizer:chatglm-bpe`, `tokenizer:cohere2moe`, `tokenizer:glm4`, `tokenizer:llada-moe`, `tokenizer:tiny_aya` | upstream tokenizer semantic parity; independently proven replacement reconstruction |
-| `evidence-only` | `architecture-runtime-evidence` | `architecture:apertus`, `architecture:arcee`, `architecture:arctic`, `architecture:baichuan`, `architecture:bailingmoe`, `architecture:bert`, `architecture:bitnet`, `architecture:bloom`, `architecture:chatglm`, `architecture:codeshell`, `architecture:cohere2`, `architecture:command-r`, `architecture:dbrx`, `architecture:deci`, `architecture:deepseek`, `architecture:dots1`, `architecture:dream`, `architecture:ernie4_5`, `architecture:ernie4_5-moe`, `architecture:eurobert`, `architecture:exaone`, `architecture:falcon`, `architecture:falcon-h1`, `architecture:gemma`, `architecture:gemma-embedding`, `architecture:gemma2`, `architecture:gemma3`, `architecture:gemma4`, `architecture:glm-dsa`, `architecture:granite`, `architecture:granitehybrid`, `architecture:granitemoe`, `architecture:grok`, `architecture:grovemoe`, `architecture:hunyuan-dense`, `architecture:hunyuan-moe`, `architecture:hy_v3`, `architecture:internlm2`, `architecture:jais`, `architecture:jais2`, `architecture:jamba`, `architecture:jina-bert-v2`, `architecture:jina-bert-v3`, `architecture:kimi-k3`, `architecture:kimi-linear`, `architecture:lfm2moe`, `architecture:llada`, `architecture:llada-moe`, `architecture:llama-embed`, `architecture:maincoder`, `architecture:mamba`, `architecture:mamba2`, `architecture:minicpm`, `architecture:minicpm3`, `architecture:minimax-01`, `architecture:minimax-m2`, `architecture:mistral4`, `architecture:modern-bert`, `architecture:muse-glimmer`, `architecture:nemotron`, `architecture:nemotron_h`, `architecture:nemotron_h_moe`, `architecture:neo-bert`, `architecture:nomic-bert`, `architecture:nomic-bert-moe`, `architecture:olmo2`, `architecture:olmoe`, `architecture:openelm`, `architecture:orion`, `architecture:pangu-embedded`, `architecture:phi2`, `architecture:phi3`, `architecture:phimoe`, `architecture:plamo`, `architecture:plamo2`, `architecture:plm`, `architecture:qwen`, `architecture:qwen2moe`, `architecture:qwen2vl`, `architecture:qwen3`, `architecture:qwen35`, `architecture:qwen3moe`, `architecture:qwen3next`, `architecture:refact`, `architecture:rnd1`, `architecture:seed_oss`, `architecture:smallthinker`, `architecture:smollm3`, `architecture:stablelm`, `architecture:t5`, `architecture:t5encoder`, `architecture:talkie`, `architecture:xverse` | immutable representative GGUF; full-logit prefill and cached-decode parity; deterministic generation/state evidence |
+| `evidence-only` | `architecture-runtime-evidence` | `architecture:arcee`, `architecture:arctic`, `architecture:baichuan`, `architecture:bailingmoe`, `architecture:bert`, `architecture:bitnet`, `architecture:bloom`, `architecture:chatglm`, `architecture:codeshell`, `architecture:cohere2`, `architecture:command-r`, `architecture:dbrx`, `architecture:deci`, `architecture:deepseek`, `architecture:dots1`, `architecture:dream`, `architecture:ernie4_5`, `architecture:ernie4_5-moe`, `architecture:eurobert`, `architecture:exaone`, `architecture:falcon`, `architecture:falcon-h1`, `architecture:gemma`, `architecture:gemma-embedding`, `architecture:gemma2`, `architecture:gemma3`, `architecture:gemma4`, `architecture:glm-dsa`, `architecture:granite`, `architecture:granitehybrid`, `architecture:granitemoe`, `architecture:grok`, `architecture:grovemoe`, `architecture:hunyuan-dense`, `architecture:hunyuan-moe`, `architecture:hy_v3`, `architecture:internlm2`, `architecture:jais`, `architecture:jais2`, `architecture:jamba`, `architecture:jina-bert-v2`, `architecture:jina-bert-v3`, `architecture:kimi-k3`, `architecture:kimi-linear`, `architecture:lfm2moe`, `architecture:llada`, `architecture:llada-moe`, `architecture:llama-embed`, `architecture:maincoder`, `architecture:mamba`, `architecture:mamba2`, `architecture:minicpm`, `architecture:minicpm3`, `architecture:minimax-01`, `architecture:minimax-m2`, `architecture:mistral4`, `architecture:modern-bert`, `architecture:muse-glimmer`, `architecture:nemotron`, `architecture:nemotron_h`, `architecture:nemotron_h_moe`, `architecture:neo-bert`, `architecture:nomic-bert`, `architecture:nomic-bert-moe`, `architecture:olmo2`, `architecture:olmoe`, `architecture:openelm`, `architecture:orion`, `architecture:pangu-embedded`, `architecture:phi2`, `architecture:phi3`, `architecture:phimoe`, `architecture:plamo`, `architecture:plamo2`, `architecture:plm`, `architecture:qwen`, `architecture:qwen2moe`, `architecture:qwen2vl`, `architecture:qwen3`, `architecture:qwen35`, `architecture:qwen3moe`, `architecture:qwen3next`, `architecture:refact`, `architecture:rnd1`, `architecture:seed_oss`, `architecture:smallthinker`, `architecture:smollm3`, `architecture:stablelm`, `architecture:t5`, `architecture:t5encoder`, `architecture:talkie`, `architecture:xverse` | immutable representative GGUF; full-logit prefill and cached-decode parity; deterministic generation/state evidence |
| `evidence-only` | `mtp-runtime-evidence` | `mtp:hy_v3`, `mtp:qwen35` | target acceptance loop; cache-threaded draft/target parity |
| `evidence-only` | `projector-runtime-evidence` | `projector:adapter`, `projector:cogvlm`, `projector:deepseekocr`, `projector:deepseekocr2`, `projector:dots3note_a`, `projector:dots3note_v`, `projector:dots_ocr`, `projector:exaone4_5`, `projector:gemma3`, `projector:gemma3na`, `projector:gemma3nv`, `projector:gemma4a`, `projector:gemma4ua`, `projector:gemma4uv`, `projector:gemma4v`, `projector:glm4v`, `projector:glma`, `projector:granite4_vision`, `projector:granite_speech`, `projector:hunyuanvl`, `projector:idefics3`, `projector:internvl`, `projector:janus_pro`, `projector:kimik25`, `projector:kimivl`, `projector:ldp`, `projector:ldpv2`, `projector:lfm2`, `projector:lfm2a`, `projector:lightonocr`, `projector:llama4`, `projector:meralion`, `projector:mimo_audio`, `projector:mimovl`, `projector:minicpmv4_6`, `projector:minimax_m3`, `projector:mlp`, `projector:muse-glimmer`, `projector:musicflamingo`, `projector:nemotron_v2_vl`, `projector:paddleocr`, `projector:parakeet`, `projector:pixtral`, `projector:pockettts_spkenc`, `projector:qwen2.5o`, `projector:qwen2.5vl_merger`, `projector:qwen2a`, `projector:qwen2vl_merger`, `projector:qwen3a`, `projector:qwen3tts_spkenc`, `projector:qwen3vl_merger`, `projector:step3vl`, `projector:ultravox`, `projector:voxtral`, `projector:yasa2`, `projector:youtuvl` | paired text target; processor boundary; deterministic multimodal package execution |
| `intentionally-rejected` | `policy-rejections` | `architecture:bailingmoe2`, `architecture:clip`, `architecture:dots3note`, `architecture:exaone-moe`, `architecture:exaone4`, `architecture:glm4`, `architecture:glm4moe`, `architecture:gptj` | policy change plus independent correctness proof |
@@ -182,7 +183,7 @@ Reason codes are concise user-facing categories; detailed architecture audits re
| Canonical architecture | Aliases | Import route | Tensor exactness | Config/tensor/graph/runtime/quantized import | Restriction or evidence gap |
|---|---|---|---|---|---|
| `afmoe` | — | none (fails before config extraction) | audited-direct-loader-conditional-union | config=deferred; tensor_map=deferred; graph=deferred; runtime=deferred; quantized_import=supported | CONFIG_DEFERRED — AFMoE combines sandwich norms, Q/K norms, sigmoid-gated attention, MuP embedding scaling, a dense prefix, correction-biased routed/shared experts, and optional interleaved sliding-window attention. |
-| `apertus` | — | model=`apertus`; tensor=`llama`+`apertus_extras` | exact-direct-loader-conditional-union | config=supported; tensor_map=supported; graph=supported; runtime=deferred; quantized_import=supported | RUNTIME_EVIDENCE_PENDING — Config extraction, exact tensor-name closure, and a full synthetic GGUF graph build are covered, but no representative real-weight GGUF has yet passed ORT parity or generation validation. |
+| `apertus` | — | model=`apertus`; tensor=`llama`+`apertus_extras` | exact-direct-loader-conditional-union | config=supported; tensor_map=supported; graph=supported; runtime=supported; quantized_import=supported | EVIDENCED_SCOPE — Runtime support is restricted to the pinned Apertus-v1.1-1.5B-Instruct BF16 artifact's exact-float CPU route, official tokenizer revision, full-logit stateful evidence, and ORT GenAI 0.15.2. |
| `arcee` | — | model=`arcee`; tensor=`arcee` | not claimed | config=supported; tensor_map=supported; graph=supported; runtime=deferred; quantized_import=supported | RUNTIME_EVIDENCE_PENDING — Config extraction, exact tensor-name closure, and a full synthetic GGUF graph build are covered, but no representative real-weight GGUF has yet passed ORT parity or generation validation. |
| `arctic` | — | model=`arctic`; module=`arctic_gguf`; tensor=`llama`+`arctic_extras` | not claimed | config=supported; tensor_map=supported; graph=supported; runtime=deferred; quantized_import=rejected | RUNTIME_EVIDENCE_PENDING / FLOAT_IMPORT_ONLY — The mobius graph uses floating Linear modules for this architecture, so no MatMulNBits or BlockQuantizedMatMul target can consume preserved GGUF projection weights. |
| `arwkv7` | — | none (fails before config extraction) | audited-direct-loader-conditional-union | config=deferred; tensor_map=deferred; graph=deferred; runtime=deferred; quantized_import=supported | CONFIG_DEFERRED — ARWKV7 wraps RWKV7's delta-rule matrix recurrence in a distinct one-shift RMSNorm/Qwen residual topology with optional five-versus-six-way interpolation, optional gate/group norm, and Qwen SwiGLU. |
diff --git a/src/mobius/integrations/gguf/_apertus_test.py b/src/mobius/integrations/gguf/_apertus_test.py
index 7aee0fa3c..7ac9e84b2 100644
--- a/src/mobius/integrations/gguf/_apertus_test.py
+++ b/src/mobius/integrations/gguf/_apertus_test.py
@@ -110,7 +110,8 @@ def test_apertus_registry_promotes_exact_float_and_quantized_graph_import() -> N
assert spec.tensor_map_recipe == ("llama", "apertus_extras")
assert spec.config_postprocessor == "apertus"
assert spec.quantized_import.value == "supported"
- assert spec.runtime.value == "deferred"
+ assert spec.runtime.value == "supported"
+ assert spec.runtime_evidence_ids == ("apertus-v1.1-1.5b-instruct-bf16-ort-genai-0.15.2",)
def test_apertus_consumes_serialized_rope_and_xielu_values_exactly() -> None:
@@ -132,6 +133,42 @@ def test_apertus_consumes_serialized_rope_and_xielu_values_exactly() -> None:
assert config.attn_o_bias
+def test_apertus_accepts_generic_xielu_metadata_and_default_rope() -> None:
+ metadata = _metadata()
+ for suffix in ("alpha_p", "alpha_n", "beta", "eps"):
+ metadata[f"xielu.{suffix}"] = metadata.pop(f"apertus.xielu.{suffix}")
+ tensors = _tensors()
+ tensors.pop("rope_freqs.weight")
+
+ config = gguf_to_config(_FakeGGUF(metadata, tensors))
+
+ assert config.rope_type == "default"
+ assert config.rope_scaling is None
+ assert config.xielu_alpha_p == (-0.2, -0.1)
+ assert config.xielu_alpha_n == (-1.0, -0.9)
+
+
+def test_apertus_rejects_conflicting_xielu_metadata_or_unserialized_rope_scaling() -> None:
+ metadata = _metadata()
+ metadata["xielu.beta"] = 0.25
+ with pytest.raises(ValueError, match=r"conflicting apertus\.xielu\.beta and xielu\.beta"):
+ gguf_to_config(_FakeGGUF(metadata, _tensors()))
+
+ metadata = _metadata()
+ metadata["apertus.rope.scaling.type"] = "longrope"
+ tensors = _tensors()
+ tensors.pop("rope_freqs.weight")
+ with pytest.raises(ValueError, match="factorless RoPE"):
+ gguf_to_config(_FakeGGUF(metadata, tensors))
+
+ metadata = _metadata()
+ metadata["apertus.rope.scaling.original_context_length"] = 16
+ tensors = _tensors()
+ tensors.pop("rope_freqs.weight")
+ with pytest.raises(ValueError, match="original_context_length requires"):
+ gguf_to_config(_FakeGGUF(metadata, tensors))
+
+
def test_apertus_maps_qk_norm_and_output_bias_without_value_transform() -> None:
assert (
map_gguf_to_hf_names("blk.1.attn_q_norm.bias", "apertus")
@@ -164,6 +201,11 @@ def test_apertus_rejects_quantized_or_malformed_serialized_rope_factors() -> Non
with pytest.raises(ValueError, match=r"shape \(2,\)"):
gguf_to_config(_FakeGGUF(_metadata(), tensors))
+ tensors = _tensors()
+ tensors["rope_factors_short.weight"] = np.ones((2,), np.float32)
+ with pytest.raises(ValueError, match="must contain exactly"):
+ gguf_to_config(_FakeGGUF(_metadata(), tensors))
+
def test_apertus_consumes_complete_longrope_factor_pair_exactly() -> None:
tensors = _tensors()
diff --git a/src/mobius/integrations/gguf/_arch_registry.py b/src/mobius/integrations/gguf/_arch_registry.py
index 3a62f7048..183bb140d 100644
--- a/src/mobius/integrations/gguf/_arch_registry.py
+++ b/src/mobius/integrations/gguf/_arch_registry.py
@@ -2258,15 +2258,14 @@
model_type="apertus",
tensor_map_recipe=("llama", "apertus_extras"),
config_postprocessor="apertus",
- required_metadata=(
- "attention.layer_norm_rms_epsilon",
- "xielu.alpha_n",
- "xielu.alpha_p",
- "xielu.beta",
- "xielu.eps",
+ required_metadata=("attention.layer_norm_rms_epsilon",),
+ runtime=Support.SUPPORTED,
+ runtime_evidence_ids=("apertus-v1.1-1.5b-instruct-bf16-ort-genai-0.15.2",),
+ reason=(
+ "Runtime support is restricted to the pinned Apertus-v1.1-1.5B-Instruct "
+ "BF16 artifact's exact-float CPU route, official tokenizer revision, full-logit "
+ "stateful evidence, and ORT GenAI 0.15.2."
),
- runtime=Support.DEFERRED,
- reason=_RUNTIME_VALIDATION_PENDING,
),
GGUFArchitectureSpec(
gguf_arch="minicpm",
diff --git a/src/mobius/integrations/gguf/_component_export.py b/src/mobius/integrations/gguf/_component_export.py
index 932906883..306815d65 100644
--- a/src/mobius/integrations/gguf/_component_export.py
+++ b/src/mobius/integrations/gguf/_component_export.py
@@ -248,7 +248,10 @@ def attach_runtime_unvalidated_report(
else None
)
tokenizer_verdict = getattr(pkg, "gguf_tokenizer_verdict", None)
- if tokenizer is None:
+ if tokenizer is None or (
+ tokenizer_exported
+ and (tokenizer.support != "supported" or tokenizer.output != "exported")
+ ):
if tokenizer_verdict is None:
raise ValueError(
"Runtime-unvalidated packaging requires the tokenizer disposition captured "
diff --git a/src/mobius/integrations/gguf/_config_mapping.py b/src/mobius/integrations/gguf/_config_mapping.py
index 4177fb2d0..beaac253f 100644
--- a/src/mobius/integrations/gguf/_config_mapping.py
+++ b/src/mobius/integrations/gguf/_config_mapping.py
@@ -473,6 +473,7 @@ def _validate_closed_rope_scaling_metadata(
arch: str,
*,
allowed_suffixes: set[str] | None = None,
+ allowed_types: set[str] | None = None,
) -> None:
"""Reject RoPE-scaling metadata outside an architecture's exact subset."""
allowed = {f"{arch}.rope.scaling.{suffix}" for suffix in (allowed_suffixes or set())}
@@ -495,7 +496,8 @@ def _validate_closed_rope_scaling_metadata(
f"{arch} has unsupported RoPE scaling metadata: {', '.join(sorted(unsupported))}"
)
scaling_type = metadata.get(f"{arch}.rope.scaling.type")
- if "type" in (allowed_suffixes or set()) and scaling_type not in (None, "", "none"):
+ accepted_types = {None, "", "none", *(allowed_types or set())}
+ if "type" in (allowed_suffixes or set()) and scaling_type not in accepted_types:
raise ValueError(
f"{arch} rope.scaling.type={scaling_type!r} is not in the exact supported subset"
)
@@ -1415,7 +1417,24 @@ def _apertus_postprocess(
layers = config.num_hidden_layers
def per_layer(suffix: str) -> tuple[float, ...]:
- raw = metadata[f"{arch}.xielu.{suffix}"]
+ qualified = f"{arch}.xielu.{suffix}"
+ generic = f"xielu.{suffix}"
+ if qualified in metadata and generic in metadata:
+ if not np.array_equal(
+ np.asarray(metadata[qualified]), np.asarray(metadata[generic])
+ ):
+ raise ValueError(
+ f"GGUF architecture {arch!r} has conflicting {qualified} and {generic}"
+ )
+ raw = metadata[qualified]
+ elif qualified in metadata:
+ raw = metadata[qualified]
+ elif generic in metadata:
+ raw = metadata[generic]
+ else:
+ raise ValueError(
+ f"GGUF architecture {arch!r} is missing required metadata: xielu.{suffix}"
+ )
values = list(raw) if isinstance(raw, (list, tuple, np.ndarray)) else [raw] * layers
if len(values) != layers:
raise ValueError(
@@ -1440,15 +1459,36 @@ def per_layer(suffix: str) -> tuple[float, ...]:
}
if any(type_id not in {0, 1, 30} for type_id in raw_types.values()):
raise ValueError("Apertus serialized RoPE factors must use F32/F16/BF16 storage")
+ _validate_closed_rope_scaling_metadata(
+ metadata,
+ arch,
+ allowed_suffixes={"original_context_length", "type"},
+ allowed_types={"longrope"},
+ )
+ scaling_type = metadata.get(f"{arch}.rope.scaling.type")
+ original_context_key = f"{arch}.rope.scaling.original_context_length"
+ factor_pair = {"rope_factors_long.weight", "rope_factors_short.weight"}
+ if original_context_key in metadata and rope_names != factor_pair:
+ raise ValueError(
+ "Apertus rope.scaling.original_context_length requires the complete "
+ "rope_factors_long.weight/rope_factors_short.weight pair"
+ )
- if rope_names == {"rope_freqs.weight"}:
+ if not rope_names:
+ if scaling_type not in {None, "", "none"}:
+ raise ValueError(
+ "Apertus factorless RoPE requires rope.scaling.type to be absent or 'none'"
+ )
+ short_factors = long_factors = None
+ original_context = config.max_position_embeddings
+ elif rope_names == {"rope_freqs.weight"}:
factors = np.asarray(model.get_tensor("rope_freqs.weight"), dtype=np.float32).reshape(
-1
)
short_factors = long_factors = factors
original_context = config.max_position_embeddings
- elif rope_names == {"rope_factors_long.weight", "rope_factors_short.weight"}:
- original_context_raw = metadata.get(f"{arch}.rope.scaling.original_context_length")
+ elif rope_names == factor_pair:
+ original_context_raw = metadata.get(original_context_key)
if original_context_raw is None:
raise ValueError(
"Apertus LongRoPE factors require apertus.rope.scaling.original_context_length"
@@ -1470,18 +1510,24 @@ def per_layer(suffix: str) -> tuple[float, ...]:
"Apertus GGUF must contain exactly rope_freqs.weight or the complete "
"rope_factors_long.weight/rope_factors_short.weight pair"
)
+ if rope_names and scaling_type not in {None, "", "none", "longrope"}:
+ raise ValueError(
+ "Apertus serialized RoPE factors require rope.scaling.type to be absent, "
+ "'none', or 'longrope'"
+ )
- expected = config.head_dim // 2
- for name, factors in (("short", short_factors), ("long", long_factors)):
- if (
- factors.shape != (expected,)
- or not np.all(np.isfinite(factors))
- or np.any(factors <= 0)
- ):
- raise ValueError(
- f"Apertus {name} RoPE factors must have shape ({expected},) and be "
- "finite positive values"
- )
+ if short_factors is not None and long_factors is not None:
+ expected = config.head_dim // 2
+ for name, factors in (("short", short_factors), ("long", long_factors)):
+ if (
+ factors.shape != (expected,)
+ or not np.all(np.isfinite(factors))
+ or np.any(factors <= 0)
+ ):
+ raise ValueError(
+ f"Apertus {name} RoPE factors must have shape ({expected},) and be "
+ "finite positive values"
+ )
q_biases = tuple(f"blk.{layer}.attn_q_norm.bias" in names for layer in range(layers))
k_biases = tuple(f"blk.{layer}.attn_k_norm.bias" in names for layer in range(layers))
@@ -1491,11 +1537,15 @@ def per_layer(suffix: str) -> tuple[float, ...]:
attn_qk_norm=True,
attn_q_norm_biases=q_biases,
attn_k_norm_biases=k_biases,
- rope_type="longrope",
- rope_scaling={
- "short_factor": short_factors.tolist(),
- "long_factor": long_factors.tolist(),
- },
+ rope_type="default" if short_factors is None else "longrope",
+ rope_scaling=(
+ None
+ if short_factors is None or long_factors is None
+ else {
+ "short_factor": short_factors.tolist(),
+ "long_factor": long_factors.tolist(),
+ }
+ ),
original_max_position_embeddings=original_context,
xielu_alpha_p=per_layer("alpha_p"),
xielu_alpha_n=per_layer("alpha_n"),
diff --git a/src/mobius/integrations/gguf/_docs_test.py b/src/mobius/integrations/gguf/_docs_test.py
index 9b4cc85d2..c48e4a51f 100644
--- a/src/mobius/integrations/gguf/_docs_test.py
+++ b/src/mobius/integrations/gguf/_docs_test.py
@@ -147,6 +147,10 @@ def test_runtime_support_requires_structured_evidence() -> None:
("gpt2", ("gpt2-q2-k-ort-genai-0.15.2",)),
("starcoder2", ("tiny-starcoder2-q2-k-ort-genai-0.15.2",)),
("olmo", ("tiny-olmo-q2-k-ort-genai-0.15.2",)),
+ (
+ "apertus",
+ ("apertus-v1.1-1.5b-instruct-bf16-ort-genai-0.15.2",),
+ ),
("mpt", ("tiny-mpt-q2-k-ort-genai-0.15.2",)),
("gptneox", ("pythia-70m-q2-k-ort-genai-0.15.2",)),
("starcoder", ("tiny-starcoder-q2-k-ort-genai-0.15.2",)),
diff --git a/src/mobius/integrations/gguf/_quant_capabilities_test.py b/src/mobius/integrations/gguf/_quant_capabilities_test.py
index 0a8c7e288..a2d6e24c5 100644
--- a/src/mobius/integrations/gguf/_quant_capabilities_test.py
+++ b/src/mobius/integrations/gguf/_quant_capabilities_test.py
@@ -203,10 +203,10 @@ def test_selected_real_artifacts_stay_within_global_budget() -> None:
assert isinstance(policy, dict)
assert isinstance(artifacts, list)
assert isinstance(lossy, list)
- assert len(artifacts) == 10
+ assert len(artifacts) == 11
assert len(lossy) == 1
selected = sum(int(record["size"]) for record in [*artifacts, *lossy])
- assert selected == 2_724_371_296
+ assert selected == 5_752_423_904
assert selected == policy["selected_artifact_bytes"]
assert selected <= policy["max_selected_artifact_bytes"]
assert lossy[0]["lfs_sha256"] == (
diff --git a/src/mobius/integrations/gguf/_route_census_test.py b/src/mobius/integrations/gguf/_route_census_test.py
index cdf66ad69..4699cc3b3 100644
--- a/src/mobius/integrations/gguf/_route_census_test.py
+++ b/src/mobius/integrations/gguf/_route_census_test.py
@@ -54,14 +54,14 @@ def test_every_route_has_one_actionable_classification() -> None:
assert {item.category for item in items} == allowed
assert all(item.batch and item.dependencies and item.reason.strip() for item in items)
assert Counter(item.kind for item in items) == {
- "architecture": 136,
+ "architecture": 135,
"projector": 60,
"tokenizer": 56,
"mtp": 22,
}
assert Counter(item.category for item in items) == {
"dependency-or-mobius-abi-blocked": 99,
- "evidence-only": 151,
+ "evidence-only": 150,
"intentionally-rejected": 19,
"artifact-unavailable": 5,
}
diff --git a/src/mobius/integrations/gguf/_runtime_evidence.py b/src/mobius/integrations/gguf/_runtime_evidence.py
index 5fa9bae9e..bc9106fe1 100644
--- a/src/mobius/integrations/gguf/_runtime_evidence.py
+++ b/src/mobius/integrations/gguf/_runtime_evidence.py
@@ -534,6 +534,71 @@ def _is_hex(value: str) -> bool:
"dynamic KV cache prefill, full-sequence replay, rollback, reorder, and 20 decode steps"
)
+_APERTUS_15B_BF16_ORT_GENAI = GGUFRuntimeEvidence(
+ evidence_id="apertus-v1.1-1.5b-instruct-bf16-ort-genai-0.15.2",
+ architecture="apertus",
+ repository="MrMeOrYou/Apertus-v1.1-1.5B-Instruct-GGUF",
+ revision="88c75ad49566d3c2157d03709bf772262c3241ed",
+ filename="Apertus-v1.1-1.5B-Instruct-BF16.gguf",
+ size=3_028_052_608,
+ lfs_sha256="f9ec154d0ec29dad1f6465b458b7f27bd25ad7b9a3899233ae98ca6d358501c2",
+ config_repository="swiss-ai/Apertus-v1.1-1.5B-Instruct",
+ config_revision="9e9d01154446a645d30f04174cf1515a38058be7",
+ tokenizer_repository="swiss-ai/Apertus-v1.1-1.5B-Instruct",
+ tokenizer_revision="9e9d01154446a645d30f04174cf1515a38058be7",
+ tokenizer_metadata_sha256=(
+ "3097bd9f22efd32db9045c4705d978539dc8adeeae763324062a5e1a73fc24a5"
+ ),
+ tokenizer_assets=(
+ (
+ "chat_template.jinja",
+ 5_250,
+ "4afab8361a4bd0c2994404e0b0851dbaad461a06e56bdfdb635aae3977473c19",
+ ),
+ (
+ "special_tokens_map.json",
+ 560,
+ "9f69883bd70fc5d8b55822799837a216d3ac4fb565e05256a0d4f9850404bbc5",
+ ),
+ (
+ "tokenizer.json",
+ 17_078_368,
+ "be12f4375d655cc740864e3a9041bcddd8477942f209d9e7f27f6c8767162638",
+ ),
+ (
+ "tokenizer_config.json",
+ 177_274,
+ "77b14a0664585c26065f07d7a4c852a4615c83348d9378e23def01957bbd3f57",
+ ),
+ ),
+ tensor_count=163,
+ tensor_qtypes=(("BF16", 98), ("F32", 65)),
+ import_route='{"architecture":"apertus","config_sha256":"2a9660c48656c0980e76f1b374262bf8383256603e8b4221aa99bcfa11c510b0","execution_provider":"cpu","model_type":"apertus","module_type":"apertus","preserve_quantization":false,"registry_import":{"config_key_map":null,"config_postprocessor":"apertus","llama_qk_permute":false,"offset_norm":false,"required_metadata":["attention.layer_norm_rms_epsilon"],"rope_interleave":false,"tensor_processor":null,"v_head_reorder":false,"vlm_builder":null},"route_schema":1,"static_cache":false,"task":{"class":"builtins.str","state":"text-generation"},"tensor_map_recipe":["llama","apertus_extras"]}',
+ source_fidelity=True,
+ storage_quantized=False,
+ target_storage_format="float",
+ compute_mode="float operators",
+ graph_files=_LOW_COST_GRAPH_FILES,
+ graph_sha256="4bb91bade19d41559cb524e28692453851fe67274cc58109787ee968df3e0fe5",
+ runtime_package_files=(
+ "chat_template.jinja",
+ "export_report.json",
+ *_LOW_COST_RUNTIME_PACKAGE_FILES,
+ ),
+ runtime_package_sha256="97582549e9c5b4114f3bcfa81c92f16aba9d14dd0ea0d2b20acaefd9060e6486",
+ parity_test=("test_promoted_gguf_full_runtime_evidence[apertus-v1.1-1.5b-instruct-bf16]"),
+ parity_kind="full-logit",
+ deterministic_test=(
+ "test_promoted_gguf_full_runtime_evidence[apertus-v1.1-1.5b-instruct-bf16]"
+ ),
+ stateful_semantics=_LOW_COST_STATEFUL_SEMANTICS,
+ execution_provider="CPUExecutionProvider",
+ onnxruntime_version="1.29.0",
+ runtime="ort-genai",
+ runtime_version="0.15.2",
+ runtime_package_schema=FINAL_RUNTIME_PACKAGE_SCHEMA,
+)
+
_GPT2_Q2_K_ORT_GENAI = GGUFRuntimeEvidence(
evidence_id="gpt2-q2-k-ort-genai-0.15.2",
architecture="gpt2",
@@ -863,6 +928,7 @@ def _is_hex(value: str) -> bool:
{
record.evidence_id: record
for record in (
+ _APERTUS_15B_BF16_ORT_GENAI,
_LFM2_350M_F16_ORT_GENAI,
_QWEN25_Q8_ORT_GENAI,
_QWEN35MOE_087B_Q2_K_ORT_GENAI,
@@ -993,12 +1059,7 @@ def find_matching_runtime_evidence(
"The GGUF source no longer matches the exact artifact identity captured during "
f"graph construction: built={built_identity!r}, current={current_identity!r}."
)
- if (
- not evidence_ids
- or runtime_version is None
- or tokenizer_repository is None
- or tokenizer_revision is None
- ):
+ if not evidence_ids or runtime_version is None:
return None
identity = built_identity
candidates = [
@@ -1014,8 +1075,14 @@ def find_matching_runtime_evidence(
and _RUNTIME_EVIDENCE[evidence_id].tensor_count == identity.tensor_count
and _RUNTIME_EVIDENCE[evidence_id].tensor_qtypes == identity.tensor_qtypes
and _RUNTIME_EVIDENCE[evidence_id].import_route == import_route
- and _RUNTIME_EVIDENCE[evidence_id].tokenizer_repository == tokenizer_repository
- and _RUNTIME_EVIDENCE[evidence_id].tokenizer_revision == tokenizer_revision
+ and (
+ tokenizer_repository is None
+ or _RUNTIME_EVIDENCE[evidence_id].tokenizer_repository == tokenizer_repository
+ )
+ and (
+ tokenizer_revision is None
+ or _RUNTIME_EVIDENCE[evidence_id].tokenizer_revision == tokenizer_revision
+ )
]
if len(candidates) > 1:
raise RuntimeError(
diff --git a/src/mobius/integrations/gguf/_runtime_evidence_test.py b/src/mobius/integrations/gguf/_runtime_evidence_test.py
index fed5e26b9..4f0c96ec7 100644
--- a/src/mobius/integrations/gguf/_runtime_evidence_test.py
+++ b/src/mobius/integrations/gguf/_runtime_evidence_test.py
@@ -144,9 +144,11 @@ def test_production_runtime_evidence_requires_final_package_regeneration() -> No
records = iter_runtime_evidence()
assert records
- assert all(
- record.runtime_package_schema != FINAL_RUNTIME_PACKAGE_SCHEMA for record in records
- )
+ assert [
+ record.evidence_id
+ for record in records
+ if record.runtime_package_schema == FINAL_RUNTIME_PACKAGE_SCHEMA
+ ] == ["apertus-v1.1-1.5b-instruct-bf16-ort-genai-0.15.2"]
def test_low_cost_runtime_batch_manifest_is_closed_and_within_budget() -> None:
@@ -284,6 +286,21 @@ def test_matching_evidence_binds_arch_runtime_source_qtypes_and_route(
)
is record
)
+ assert (
+ matching_runtime_evidence(
+ (record.evidence_id,),
+ architecture="llama",
+ runtime="onnx-genai",
+ source_path=source,
+ gguf_model=_model(),
+ built_identity=gguf_artifact_identity(source, _model(), architecture="llama"),
+ import_route=record.import_route,
+ runtime_version="1.0.0",
+ tokenizer_repository=None,
+ tokenizer_revision=None,
+ )
+ is record
+ )
with pytest.raises(ValueError, match="No unique GGUF runtime evidence"):
matching_runtime_evidence(
diff --git a/src/mobius/integrations/gguf/_runtime_package.py b/src/mobius/integrations/gguf/_runtime_package.py
index 2ce04f0ed..55752fd4b 100644
--- a/src/mobius/integrations/gguf/_runtime_package.py
+++ b/src/mobius/integrations/gguf/_runtime_package.py
@@ -32,6 +32,7 @@
from mobius.integrations.gguf._arch_registry import get_arch_spec
from mobius.integrations.gguf._runtime_evidence import (
+ FINAL_RUNTIME_PACKAGE_SCHEMA,
RuntimeEvidenceUnavailableError,
gguf_artifact_identity,
gguf_graph_package_identity,
@@ -382,11 +383,7 @@ def write_gguf_runtime_package(
)
evidence = None
evidence_warning: str | None = None
- if (
- tokenizer_repository is not None
- and tokenizer_revision is not None
- and architecture_spec.runtime_evidence_ids
- ):
+ if architecture_spec.runtime_evidence_ids:
try:
evidence = matching_runtime_evidence(
architecture_spec.runtime_evidence_ids,
@@ -397,11 +394,24 @@ def write_gguf_runtime_package(
built_identity=built_identity,
import_route=import_route,
runtime_version=runtime_version,
- tokenizer_repository=tokenizer_repository,
- tokenizer_revision=tokenizer_revision,
+ tokenizer_repository=None,
+ tokenizer_revision=None,
)
except RuntimeEvidenceUnavailableError as error:
evidence_warning = f"{error} Export continues without claiming runtime validation."
+ if (
+ evidence is not None
+ and tokenizer_repository is not None
+ and (
+ tokenizer_repository != evidence.tokenizer_repository
+ or tokenizer_revision != evidence.tokenizer_revision
+ )
+ ):
+ raise ValueError(
+ "The explicit tokenizer source conflicts with exact runtime evidence: "
+ f"requested={tokenizer_repository}@{tokenizer_revision}, "
+ f"evidence={evidence.tokenizer_repository}@{evidence.tokenizer_revision}."
+ )
if not source_model.source_matches_path():
raise ValueError(
"The GGUF source changed while runtime evidence was being matched; "
@@ -433,9 +443,12 @@ def write_gguf_runtime_package(
"The GGUF tokenizer metadata no longer matches the identity captured during "
"graph construction; refusing to pair the graph with a replaced tokenizer source."
)
- if getattr(built_verdict, "blocker_category", None) is not None:
- # Exact runtime evidence cannot override a tokenizer route that the
- # authoritative tokenizer census has explicitly withheld.
+ if (
+ getattr(built_verdict, "blocker_category", None) is not None
+ and getattr(evidence, "runtime_package_schema", None) != FINAL_RUNTIME_PACKAGE_SCHEMA
+ ):
+ # Identifier-level blockers remain authoritative unless a final-package
+ # record proves this exact artifact through the strict checks below.
evidence = None
output_dir.parent.mkdir(parents=True, exist_ok=True)
@@ -452,7 +465,14 @@ def write_gguf_runtime_package(
"The GGUF source changed while target/MTP graphs were being serialized; "
"refusing package publication."
)
- graph_identity = gguf_graph_package_identity(stage)
+ graph_files = tuple(
+ sorted(
+ path.relative_to(stage).as_posix()
+ for path in stage.rglob("*")
+ if path.is_file() and path.name != "export_report.json"
+ )
+ )
+ graph_identity = gguf_graph_package_identity(stage, files=graph_files)
validation_warnings: list[str] = []
if evidence_warning is not None:
validation_warnings.append(evidence_warning)
diff --git a/src/mobius/integrations/gguf/_runtime_package_test.py b/src/mobius/integrations/gguf/_runtime_package_test.py
index ea00feb97..55db6b1fb 100644
--- a/src/mobius/integrations/gguf/_runtime_package_test.py
+++ b/src/mobius/integrations/gguf/_runtime_package_test.py
@@ -132,12 +132,14 @@ def _source_model():
def _write_runtime(pkg, source, output, **kwargs):
+ tokenizer_repository = kwargs.pop("tokenizer_repository", _TOKENIZER_REPOSITORY)
+ tokenizer_revision = kwargs.pop("tokenizer_revision", _TOKENIZER_REVISION)
return write_gguf_runtime_package(
pkg,
source,
output,
- tokenizer_repository=_TOKENIZER_REPOSITORY,
- tokenizer_revision=_TOKENIZER_REVISION,
+ tokenizer_repository=tokenizer_repository,
+ tokenizer_revision=tokenizer_revision,
**kwargs,
)
@@ -476,8 +478,29 @@ def test_atomically_emits_graph_tokenizer_and_runtime_config(self, tmp_path):
assert (out / "export_report.json").is_file()
assert not list(tmp_path.glob(".out.*.tmp"))
- def test_exact_runtime_evidence_marks_package_validated(self, tmp_path):
+ @pytest.mark.parametrize(
+ "tokenizer_kwargs",
+ (
+ {},
+ {"tokenizer_repository": None, "tokenizer_revision": None},
+ ),
+ ids=("matching-explicit-tokenizer", "evidence-pinned-tokenizer"),
+ )
+ def test_exact_runtime_evidence_marks_package_validated(self, tmp_path, tokenizer_kwargs):
pkg = _FakePackage()
+ blocker = GGUFTokenizerVerdict(
+ route="deferred",
+ model="gpt2",
+ pre="blocked-pre",
+ canonical_pre="blocked-pre",
+ token_count=2,
+ metadata_sha256="f" * 64,
+ blocker_category="compiled-llama.cpp-semantic-dependency",
+ audit_status="deferred-compiled-semantics",
+ reason="compiled tokenizer semantics require exact evidence",
+ )
+ pkg.gguf_tokenizer_verdict = blocker
+ attach_tokenizer_export_report(pkg, blocker, model_route="llama")
out = tmp_path / "out"
evidence = SimpleNamespace(
evidence_id="test-evidence",
@@ -489,6 +512,7 @@ def test_exact_runtime_evidence_marks_package_validated(self, tmp_path):
"tokenizer.json",
),
runtime_package_sha256="c" * 64,
+ runtime_package_schema=_runtime_package.FINAL_RUNTIME_PACKAGE_SCHEMA,
tokenizer_repository=_TOKENIZER_REPOSITORY,
tokenizer_revision=_TOKENIZER_REVISION,
tokenizer_metadata_sha256="f" * 64,
@@ -500,6 +524,14 @@ def test_exact_runtime_evidence_marks_package_validated(self, tmp_path):
"mobius.integrations.gguf._runtime_package.matching_runtime_evidence",
return_value=evidence,
),
+ mock.patch(
+ "mobius.integrations.gguf._runtime_package.inspect_gguf_tokenizer",
+ return_value=blocker,
+ ),
+ mock.patch(
+ "mobius.integrations.gguf._component_export.resolve_tokenizer_export_verdict",
+ return_value=blocker,
+ ),
mock.patch(
"mobius.integrations.gguf._runtime_package.gguf_graph_package_identity",
side_effect=(
@@ -509,12 +541,20 @@ def test_exact_runtime_evidence_marks_package_validated(self, tmp_path):
sha256=evidence.runtime_package_sha256,
),
),
- ),
+ ) as package_identity,
):
- _write_runtime(pkg, tmp_path / "m.gguf", out, runtime_version="0.15.2")
+ _write_runtime(
+ pkg,
+ tmp_path / "m.gguf",
+ out,
+ runtime_version="0.15.2",
+ **tokenizer_kwargs,
+ )
+ assert package_identity.call_args_list[0].kwargs["files"] == ("model.onnx",)
assert pkg.export_report is not None
assert pkg.export_report.export_status == "complete"
+ assert pkg.export_report.component("tokenizer").output == "exported"
assert pkg.export_report.runtime_validation_status == "validated"
assert pkg.export_report.end_to_end_runnable is True
runtime_component = pkg.export_report.component("runtime")
@@ -525,6 +565,24 @@ def test_exact_runtime_evidence_marks_package_validated(self, tmp_path):
assert compatibility["runtime_validation_status"] == "validated"
assert compatibility["runtime_evidence_id"] == "test-evidence"
+ def test_explicit_tokenizer_conflict_rejects_before_package_save(self, tmp_path):
+ pkg = _FakePackage()
+ with (
+ _successful_runtime_dependencies(),
+ pytest.raises(ValueError, match="conflicts with exact runtime evidence"),
+ ):
+ _write_runtime(
+ pkg,
+ tmp_path / "m.gguf",
+ tmp_path / "out",
+ runtime_version="0.15.2",
+ tokenizer_repository="attacker/replacement",
+ tokenizer_revision="d" * 40,
+ )
+
+ assert pkg.saved_to is None
+ assert not (tmp_path / "out").exists()
+
def test_missing_evidence_runtime_version_does_not_block_export(self, tmp_path):
pkg = _FakePackage()
out = tmp_path / "out"
@@ -662,6 +720,10 @@ def test_tokenizer_blocker_omits_only_tokenizer_assets(self, tmp_path, caplog):
with (
caplog.at_level("WARNING"),
_successful_runtime_dependencies(),
+ mock.patch(
+ "mobius.integrations.gguf._runtime_package.matching_runtime_evidence",
+ return_value=None,
+ ),
mock.patch(
"mobius.integrations.gguf._runtime_package.inspect_gguf_tokenizer",
return_value=blocker,
@@ -669,7 +731,13 @@ def test_tokenizer_blocker_omits_only_tokenizer_assets(self, tmp_path, caplog):
):
pkg.gguf_tokenizer_verdict = blocker
attach_tokenizer_export_report(pkg, blocker, model_route="llama")
- artifacts = _write_runtime(pkg, tmp_path / "m.gguf", out)
+ artifacts = _write_runtime(
+ pkg,
+ tmp_path / "m.gguf",
+ out,
+ tokenizer_repository=None,
+ tokenizer_revision=None,
+ )
assert (out / "model.onnx").is_file()
assert Path(artifacts["export_report"]).is_file()
diff --git a/testdata/cases/causal-lm/apertus-v1.1-1.5b-instruct-bf16.yaml b/testdata/cases/causal-lm/apertus-v1.1-1.5b-instruct-bf16.yaml
new file mode 100644
index 000000000..d12df910f
--- /dev/null
+++ b/testdata/cases/causal-lm/apertus-v1.1-1.5b-instruct-bf16.yaml
@@ -0,0 +1,61 @@
+model_id: "swiss-ai/Apertus-v1.1-1.5B-Instruct"
+model_type: "apertus"
+revision: "9e9d01154446a645d30f04174cf1515a38058be7"
+task_type: "text-generation"
+dtype: "float32"
+
+gguf:
+ repository: "MrMeOrYou/Apertus-v1.1-1.5B-Instruct-GGUF"
+ revision: "88c75ad49566d3c2157d03709bf772262c3241ed"
+ filename: "Apertus-v1.1-1.5B-Instruct-BF16.gguf"
+ size: 3028052608
+ lfs_sha256: "f9ec154d0ec29dad1f6465b458b7f27bd25ad7b9a3899233ae98ca6d358501c2"
+ tensor_count: 163
+ tensor_qtypes:
+ BF16: 98
+ F32: 65
+ execution_provider: "cpu"
+ config_sha256: "2a9660c48656c0980e76f1b374262bf8383256603e8b4221aa99bcfa11c510b0"
+ preserve_quantization: false
+ keep_quantized: true
+ tokenizer:
+ repository: "swiss-ai/Apertus-v1.1-1.5B-Instruct"
+ revision: "9e9d01154446a645d30f04174cf1515a38058be7"
+ metadata_sha256: "3097bd9f22efd32db9045c4705d978539dc8adeeae763324062a5e1a73fc24a5"
+ identity_status: "exact"
+ assets:
+ - filename: "chat_template.jinja"
+ size: 5250
+ sha256: "4afab8361a4bd0c2994404e0b0851dbaad461a06e56bdfdb635aae3977473c19"
+ - filename: "special_tokens_map.json"
+ size: 560
+ sha256: "9f69883bd70fc5d8b55822799837a216d3ac4fb565e05256a0d4f9850404bbc5"
+ - filename: "tokenizer.json"
+ size: 17078368
+ sha256: "be12f4375d655cc740864e3a9041bcddd8477942f209d9e7f27f6c8767162638"
+ - filename: "tokenizer_config.json"
+ size: 177274
+ sha256: "77b14a0664585c26065f07d7a4c852a4615c83348d9378e23def01957bbd3f57"
+
+ort_genai:
+ tier: "real"
+ runtime_versions: ["0.15.2"]
+ model_type: "decoder"
+ runtime_evidence_id: "apertus-v1.1-1.5b-instruct-bf16-ort-genai-0.15.2"
+ execution_provider: "cpu"
+ max_download_bytes: 3050000000
+ released_capabilities:
+ "0.15.2": {generic_decoder: true, state_groups: false}
+
+inputs:
+ prompts:
+ - "Hello"
+
+level: "L4+L5"
+
+generation:
+ max_new_tokens: 20
+ do_sample: false
+ exact_match: true
+
+notes: "Pinned BF16 GGUF; independent raw-reader same-value oracle, full logits, 20-step cached decode, state semantics, and ORT GenAI generation are verified."
diff --git a/testdata/cases/schema.json b/testdata/cases/schema.json
index 42882605d..06f4149d6 100644
--- a/testdata/cases/schema.json
+++ b/testdata/cases/schema.json
@@ -308,7 +308,7 @@
"max_download_bytes": {
"type": "integer",
"minimum": 1,
- "maximum": 1073741824,
+ "maximum": 17179869184,
"description": "Hard per-case aggregate artifact download budget."
},
"released_capabilities": {
diff --git a/testdata/evidence/gguf_quantization_capabilities.json b/testdata/evidence/gguf_quantization_capabilities.json
index 2f4806485..3bb328889 100644
--- a/testdata/evidence/gguf_quantization_capabilities.json
+++ b/testdata/evidence/gguf_quantization_capabilities.json
@@ -384,7 +384,7 @@
"max_selected_artifact_bytes": 17179869184,
"preserved_definition": "Only byte-identical native blocks or exact affine repacks preserve source represented values. Dequantize/requantize is never preserved.",
"runtime_definition": "Runtime support requires immutable same-artifact full-logit parity plus deterministic prefill/decode/replay/rollback/reorder evidence.",
- "selected_artifact_bytes": 2724371296,
+ "selected_artifact_bytes": 5752423904,
"target_storage_definition": "Lossy affine normalization may still produce supported packed target storage while source_fidelity is false."
},
"qtypes": [
@@ -3010,6 +3010,42 @@
"F32": 55
}
},
+ {
+ "evidence_ids": [
+ "apertus-v1.1-1.5b-instruct-bf16-ort-genai-0.15.2"
+ ],
+ "filename": "Apertus-v1.1-1.5B-Instruct-BF16.gguf",
+ "lfs_sha256": "f9ec154d0ec29dad1f6465b458b7f27bd25ad7b9a3899233ae98ca6d358501c2",
+ "repository": "MrMeOrYou/Apertus-v1.1-1.5B-Instruct-GGUF",
+ "revision": "88c75ad49566d3c2157d03709bf772262c3241ed",
+ "runtime_results": [
+ {
+ "compute_mode": "float operators",
+ "deterministic_test": "test_promoted_gguf_full_runtime_evidence[apertus-v1.1-1.5b-instruct-bf16]",
+ "downstream_runtime": "ort-genai",
+ "downstream_runtime_version": "0.15.2",
+ "evidence_id": "apertus-v1.1-1.5b-instruct-bf16-ort-genai-0.15.2",
+ "execution_provider": "CPUExecutionProvider",
+ "graph_sha256": "4bb91bade19d41559cb524e28692453851fe67274cc58109787ee968df3e0fe5",
+ "import_route": "{\"architecture\":\"apertus\",\"config_sha256\":\"2a9660c48656c0980e76f1b374262bf8383256603e8b4221aa99bcfa11c510b0\",\"execution_provider\":\"cpu\",\"model_type\":\"apertus\",\"module_type\":\"apertus\",\"preserve_quantization\":false,\"registry_import\":{\"config_key_map\":null,\"config_postprocessor\":\"apertus\",\"llama_qk_permute\":false,\"offset_norm\":false,\"required_metadata\":[\"attention.layer_norm_rms_epsilon\"],\"rope_interleave\":false,\"tensor_processor\":null,\"v_head_reorder\":false,\"vlm_builder\":null},\"route_schema\":1,\"static_cache\":false,\"task\":{\"class\":\"builtins.str\",\"state\":\"text-generation\"},\"tensor_map_recipe\":[\"llama\",\"apertus_extras\"]}",
+ "limitations": null,
+ "onnxruntime_version": "1.29.0",
+ "parity_kind": "full-logit",
+ "parity_test": "test_promoted_gguf_full_runtime_evidence[apertus-v1.1-1.5b-instruct-bf16]",
+ "result": "passed",
+ "runtime_package_sha256": "97582549e9c5b4114f3bcfa81c92f16aba9d14dd0ea0d2b20acaefd9060e6486",
+ "source_fidelity": true,
+ "stateful_semantics": "dynamic KV cache prefill, full-sequence replay, rollback, reorder, and 20 decode steps",
+ "storage_quantized": false,
+ "target_storage_format": "float"
+ }
+ ],
+ "size": 3028052608,
+ "tensor_qtypes": {
+ "BF16": 98,
+ "F32": 65
+ }
+ },
{
"evidence_ids": [
"qwen2.5-0.5b-instruct-q8-ort-genai-0.15.2"
diff --git a/tests/gguf_small_model_runtime_integration_test.py b/tests/gguf_small_model_runtime_integration_test.py
index 8a6ce53a1..ea3454c62 100644
--- a/tests/gguf_small_model_runtime_integration_test.py
+++ b/tests/gguf_small_model_runtime_integration_test.py
@@ -57,6 +57,7 @@
)
from mobius.integrations.gguf._reader import GGUFModel
from mobius.integrations.gguf._runtime_evidence import (
+ FINAL_RUNTIME_PACKAGE_SCHEMA,
GGUFRuntimeEvidence,
gguf_graph_package_identity,
runtime_evidence,
@@ -260,6 +261,37 @@ class _PromotedRuntimeCase:
_PROMOTED_RUNTIME_CASES = (
+ _PromotedRuntimeCase(
+ name="apertus-v1.1-1.5b-instruct-bf16",
+ evidence_id="apertus-v1.1-1.5b-instruct-bf16-ort-genai-0.15.2",
+ prompt="Hello",
+ generated_tokens=(
+ 1044,
+ 1362,
+ 4525,
+ 1875,
+ 1317,
+ 1278,
+ 4304,
+ 1307,
+ 19466,
+ 1321,
+ 1362,
+ 6483,
+ 2151,
+ 6370,
+ 1317,
+ 8178,
+ 17616,
+ 1046,
+ 1362,
+ 6483,
+ ),
+ atol=2e-4,
+ cache_atol=4e-5,
+ reference_kind="apertus",
+ required_free_bytes=13_000_000_000,
+ ),
_PromotedRuntimeCase(
name="gpt2-q2-k",
evidence_id="gpt2-q2-k-ort-genai-0.15.2",
@@ -740,7 +772,8 @@ def _load_low_cost_same_value_reference(
# The oracle reads and maps upstream GGUF tensors directly. It deliberately
# shares no Mobius tensor loader, name mapping, value processor, or normalizer.
source: dict[str, torch.Tensor] = {}
- for tensor in GGUFReader(gguf_path).tensors:
+ reader = GGUFReader(gguf_path)
+ for tensor in reader.tensors:
shape = tuple(int(dimension) for dimension in reversed(tensor.shape))
if tensor.tensor_type in (GGMLQuantizationType.F32, GGMLQuantizationType.F16):
value = tensor.data.reshape(shape)
@@ -776,6 +809,11 @@ def undo_llama_qk_permutation(value: torch.Tensor, heads: int) -> torch.Tensor:
renamed: dict[str, torch.Tensor] = {}
global_names = {
+ "apertus": {
+ "token_embd.weight": "model.embed_tokens.weight",
+ "output_norm.weight": "model.norm.weight",
+ "output.weight": "lm_head.weight",
+ },
"gptneox": {
"token_embd.weight": "gpt_neox.embed_in.weight",
"output_norm.weight": "gpt_neox.final_layer_norm.weight",
@@ -798,6 +836,18 @@ def undo_llama_qk_permutation(value: torch.Tensor, heads: int) -> torch.Tensor:
},
}[reference_kind]
layer_names = {
+ "apertus": {
+ "attn_norm": "attention_layernorm",
+ "attn_q": "self_attn.q_proj",
+ "attn_k": "self_attn.k_proj",
+ "attn_v": "self_attn.v_proj",
+ "attn_output": "self_attn.o_proj",
+ "attn_q_norm": "self_attn.q_norm",
+ "attn_k_norm": "self_attn.k_norm",
+ "ffn_norm": "feedforward_layernorm",
+ "ffn_up": "mlp.up_proj",
+ "ffn_down": "mlp.down_proj",
+ },
"gptneox": {
"attn_norm": "input_layernorm",
"attn_qkv": "attention.query_key_value",
@@ -833,6 +883,7 @@ def undo_llama_qk_permutation(value: torch.Tensor, heads: int) -> torch.Tensor:
},
}[reference_kind]
layer_prefix = {
+ "apertus": "model.layers",
"gptneox": "gpt_neox.layers",
"mpt": "transformer.blocks",
"olmo": "model.layers",
@@ -855,9 +906,27 @@ def undo_llama_qk_permutation(value: torch.Tensor, heads: int) -> torch.Tensor:
)
value = undo_llama_qk_permutation(value, heads)
renamed[target] = value
- assert len(renamed) == len(source)
- if reference_kind == "gptneox":
+ if reference_kind == "apertus":
+ for suffix in ("alpha_p", "alpha_n", "beta", "eps"):
+ field = reader.fields[f"xielu.{suffix}"]
+ values = [float(field.parts[index][0]) for index in field.data]
+ assert len(values) == int(hf_config.num_hidden_layers)
+ for layer, value in enumerate(values):
+ target = f"model.layers.{layer}.mlp.act_fn.{suffix}"
+ renamed[target] = torch.tensor(
+ [value] if suffix in {"alpha_p", "alpha_n"} else value,
+ dtype=torch.float32,
+ )
+ expected_metadata_parameters = (
+ 4 * int(hf_config.num_hidden_layers) if reference_kind == "apertus" else 0
+ )
+ assert len(renamed) == len(source) + expected_metadata_parameters
+
+ if reference_kind == "apertus":
+ reference = AutoModelForCausalLM.from_config(hf_config)
+ reference.load_state_dict(renamed, assign=True, strict=True)
+ elif reference_kind == "gptneox":
reference = GPTNeoXForCausalLM(hf_config)
reference.load_state_dict(renamed, strict=True)
elif reference_kind == "starcoder":
@@ -1615,10 +1684,6 @@ def capture_save(package: ModelPackage, *args: object, **kwargs: object) -> None
"ort-genai",
"--runtime-version",
installed_version,
- "--tokenizer-repository",
- evidence.tokenizer_repository,
- "--tokenizer-revision",
- evidence.tokenizer_revision,
"--local-files-only",
]
if case.dequantize:
@@ -1649,6 +1714,12 @@ def capture_save(package: ModelPackage, *args: object, **kwargs: object) -> None
package = ModelPackage.load(output_dir)
assert tuple(package) == ("model",)
assert package.gguf_quantization_report == source_report
+ if evidence.runtime_package_schema == FINAL_RUNTIME_PACKAGE_SCHEMA:
+ assert package.export_report is not None
+ assert package.export_report.export_status == "complete"
+ assert package.export_report.runtime_validation_status == "validated"
+ assert package.export_report.end_to_end_runnable is True
+ assert package.export_report.component("tokenizer").output == "exported"
assert (
GGUFQuantizationReport.read_json(output_dir / "quantization_report.json")
== source_report
@@ -1689,7 +1760,7 @@ def capture_save(package: ModelPackage, *args: object, **kwargs: object) -> None
)
else:
with ExitStack() as reference_guard:
- if case.reference_kind in {"gptneox", "mpt", "olmo", "starcoder"}:
+ if case.reference_kind in {"apertus", "gptneox", "mpt", "olmo", "starcoder"}:
for helper in (
"mobius.integrations.gguf._builder._load_dequantized_state_dict",
"mobius.integrations.gguf._builder._normalize_gguf_weights",