diff --git a/docs/api/build_from_gguf.md b/docs/api/build_from_gguf.md index e4a3d381e..7ac39b573 100644 --- a/docs/api/build_from_gguf.md +++ b/docs/api/build_from_gguf.md @@ -9,7 +9,7 @@ Support is capability-specific: graph import does not imply runtime packaging. | Census | Total | Closure | |---|---:|---| -| Architectures | 148 | graph verdicts: {'deferred': 41, 'rejected': 2, 'supported': 105}; importable: 105; quantized import: {'rejected': 47, 'supported': 101}; runtime: {'deferred': 136, 'rejected': 2, 'supported': 10} | +| Architectures | 148 | graph verdicts: {'deferred': 41, 'rejected': 2, 'supported': 105}; importable: 105; quantized import: {'rejected': 47, 'supported': 101}; runtime: {'deferred': 135, 'rejected': 2, 'supported': 11} | | Active stored qtypes | 25 | 24 have an import route; 1 are explicitly deferred with no route | | Serialized projector strings | 60 | {'graph-importable': 56, 'runtime-evidenced': 0} | | Tokenizer pre identifiers | 87 | 56 semantic groups; route dispositions: {'deferred-compiled-semantics': 45, 'deferred-pinned-artifact-evidence': 4, 'deferred-pinned-artifact-mismatch': 7, 'validated-pinned-source': 31} | @@ -88,6 +88,7 @@ network-free selection, budget, exclusions, and fail-closed candidate reasons ar | Evidence ID | GGUF identity | Config identity | Tokenizer identity | Runtime proof | |---|---|---|---|---| +| `apertus-v1.1-1.5b-instruct-bf16-ort-genai-0.15.2` | `MrMeOrYou/Apertus-v1.1-1.5B-Instruct-GGUF@88c75ad49566d3c2157d03709bf772262c3241ed`
`Apertus-v1.1-1.5B-Instruct-BF16.gguf`
3,028,052,608 B
`f9ec154d0ec29dad1f6465b458b7f27bd25ad7b9a3899233ae98ca6d358501c2` | `swiss-ai/Apertus-v1.1-1.5B-Instruct@9e9d01154446a645d30f04174cf1515a38058be7` | `swiss-ai/Apertus-v1.1-1.5B-Instruct@9e9d01154446a645d30f04174cf1515a38058be7`
`chat_template.jinja` 5,250 B `4afab8361a4bd0c2994404e0b0851dbaad461a06e56bdfdb635aae3977473c19`, `special_tokens_map.json` 560 B `9f69883bd70fc5d8b55822799837a216d3ac4fb565e05256a0d4f9850404bbc5`, `tokenizer.json` 17,078,368 B `be12f4375d655cc740864e3a9041bcddd8477942f209d9e7f27f6c8767162638`, `tokenizer_config.json` 177,274 B `77b14a0664585c26065f07d7a4c852a4615c83348d9378e23def01957bbd3f57`
metadata `3097bd9f22efd32db9045c4705d978539dc8adeeae763324062a5e1a73fc24a5` | ONNX Runtime 1.29.0 `CPUExecutionProvider`; ort-genai 0.15.2; result=passed; full-logit; dynamic KV cache prefill, full-sequence replay, rollback, reorder, and 20 decode steps | | `gpt2-q2-k-ort-genai-0.15.2` | `tensorblock/gpt2-GGUF@5b01870b15c4b2e43695d7f3f3bfb5b26106f23b`
`gpt2-Q2_K.gguf`
81,196,544 B
`4234545f917ec1df10dab4d926796a83422b68e9010d85a4c111b8b541f32892` | `openai-community/gpt2@607a30d783dfa663caf39e06633721c8d4cfcd7e` | `Xenova/gpt2@bf2c7f02e0b826c60d03af341171bde20893da66`
`special_tokens_map.json` 99 B `6f50ab5a5a509a1c309d6171f339b196a900dc9c99ad0408ff23bb615fdae7ad`, `tokenizer.json` 2,107,653 B `cda20b8ca044949aa07ac4078420c80d1a57139d5f9f33700e46fb2d891e7c66`, `tokenizer_config.json` 234 B `551e26ec611d8d0c8edc3ef72e518a38418cb71f40de1347dd486a595e1557d7`
metadata `b2417176025f8500d864004b0bf93b1403dc3c52238f6628f82fb0e3c498977e` | ONNX Runtime 1.29.0 `CPUExecutionProvider`; ort-genai 0.15.2; result=passed; full-logit; dynamic KV cache prefill, full-sequence replay, rollback, reorder, and 20 decode steps; Explicit-float correctness route; source GGUF blocks are dequantized. | | `lfm2-350m-f16-ort-genai-0.15.2` | `LiquidAI/LFM2-350M-GGUF@8fdc9d526b7ed346b19257551b05816c7912ecc2`
`LFM2-350M-F16.gguf`
711,482,304 B
`379ffdcbf08147c0313f6f1ce7ff558a2bc935eda633f4b46c52347032419c42` | `LiquidAI/LFM2-350M@f37d3f5c8c5484bc01dad379a595cf4c68c4e70e` | `LiquidAI/LFM2-350M@73e3c253078a3b97c2e14b4c4665679f4d9b6d56`
`chat_template.jinja` 209 B `a805e50fed68938a076b07e2e602639611b50b1ced0e50f11eb92f1ba25be4dc`, `special_tokens_map.json` 434 B `742aefe2b7dec496e8caffdba03a75d0c1a9925d53bd3f3e0d388c96b591b6f4`, `tokenizer.json` 4,732,426 B `98cff83b4f6d7e9d8929bebc62b07e92cf1b3f99c80d16bafe8b84a75448f40b`, `tokenizer_config.json` 91,509 B `36f511115e9d8952cbc9d15d9a20dfa7ce7d1444940e5c1dc42a762020c99bf5`
metadata `e5626d605bb50bc53fdb0fbfcf374fb33dfbaa0cc698d9746ba1e9b0b7e6d07d` | ONNX Runtime 1.29.0 `CPUExecutionProvider`; ort-genai 0.15.2; result=passed; full-logit; hybrid convolution and KV state prefill, replay, rollback, reorder, and 20 decode steps | | `pythia-70m-q2-k-ort-genai-0.15.2` | `mradermacher/pythia-70m-GGUF@52d6f045404c9f93418df2a0144d20c9de34316b`
`pythia-70m.Q2_K.gguf`
38,508,192 B
`8e331c8c8016bed8ff1863b78fafe51e86b2364f32c2d7f3e201687e081cf7f7` | `EleutherAI/pythia-70m@a39f36b100fe8a5377810d56c3f4789b9c53ac42` | `EleutherAI/pythia-70m@a39f36b100fe8a5377810d56c3f4789b9c53ac42`
`special_tokens_map.json` 99 B `6f50ab5a5a509a1c309d6171f339b196a900dc9c99ad0408ff23bb615fdae7ad`, `tokenizer.json` 2,113,710 B `c24618a1b3e6a38167beff1c72cffd126c3a66254347304b50547d12c5f25624`, `tokenizer_config.json` 396 B `70e38394e494931c6f773ba41e19460dd4436526b852207367f04341b4066d3f`
metadata `d5c722f646ff6462ac217da5e3514d1fa8b4a7b33aedced3daa8e7f00cc74f78` | ONNX Runtime 1.29.0 `CPUExecutionProvider`; ort-genai 0.15.2; result=passed; full-logit; dynamic KV cache prefill, full-sequence replay, rollback, reorder, and 20 decode steps; Explicit-float correctness route on the portable graph; source GGUF blocks are dequantized and the pinned tokenizer vocabulary is extended only with deterministic padding IDs present in the GGUF. | @@ -129,7 +130,7 @@ remain machine-readable in `_route_census.py`; this table groups only shared nex | `dependency-or-mobius-abi-blocked` | `projector-package-abi` | `projector:resampler` | Mobius dynamic processor-to-graph media shape contract | | `dependency-or-mobius-abi-blocked` | `tokenizer-compiled-semantics` | `tokenizer:afmoe`, `tokenizer:bloom`, `tokenizer:chameleon`, `tokenizer:codeshell`, `tokenizer:command-r`, `tokenizer:dbrx`, `tokenizer:deepseek-coder`, `tokenizer:deepseek-llm`, `tokenizer:deepseek-v3`, `tokenizer:default`, `tokenizer:exaone`, `tokenizer:exaone-moe`, `tokenizer:falcon`, `tokenizer:gpt3-finnish`, `tokenizer:granite-docling`, `tokenizer:granite-embed-multi-97m`, `tokenizer:grok-2`, `tokenizer:hunyuan`, `tokenizer:hunyuan-dense`, `tokenizer:jais`, `tokenizer:jais-2`, `tokenizer:joyai-llm`, `tokenizer:kimi-k2`, `tokenizer:laguna`, `tokenizer:megrez`, `tokenizer:mellum2`, `tokenizer:minerva-7b`, `tokenizer:minicpm5`, `tokenizer:minimax-m2`, `tokenizer:mpt`, `tokenizer:olmo`, `tokenizer:poro-chat`, `tokenizer:refact`, `tokenizer:sarvam-moe`, `tokenizer:seed-coder`, `tokenizer:smaug-bpe`, `tokenizer:solar-open`, `tokenizer:stablelm2`, `tokenizer:starcoder`, `tokenizer:superbpe`, `tokenizer:tekken`, `tokenizer:trillion`, `tokenizer:viking`, `tokenizer:whitespace`, `tokenizer:youtu` | compiled pinned llama.cpp oracle; dispatch-equivalence fixture | | `dependency-or-mobius-abi-blocked` | `tokenizer-compiled-semantics` | `tokenizer:bailingmoe`, `tokenizer:bailingmoe2`, `tokenizer:chatglm-bpe`, `tokenizer:cohere2moe`, `tokenizer:glm4`, `tokenizer:llada-moe`, `tokenizer:tiny_aya` | upstream tokenizer semantic parity; independently proven replacement reconstruction | -| `evidence-only` | `architecture-runtime-evidence` | `architecture:apertus`, `architecture:arcee`, `architecture:arctic`, `architecture:baichuan`, `architecture:bailingmoe`, `architecture:bert`, `architecture:bitnet`, `architecture:bloom`, `architecture:chatglm`, `architecture:codeshell`, `architecture:cohere2`, `architecture:command-r`, `architecture:dbrx`, `architecture:deci`, `architecture:deepseek`, `architecture:dots1`, `architecture:dream`, `architecture:ernie4_5`, `architecture:ernie4_5-moe`, `architecture:eurobert`, `architecture:exaone`, `architecture:falcon`, `architecture:falcon-h1`, `architecture:gemma`, `architecture:gemma-embedding`, `architecture:gemma2`, `architecture:gemma3`, `architecture:gemma4`, `architecture:glm-dsa`, `architecture:granite`, `architecture:granitehybrid`, `architecture:granitemoe`, `architecture:grok`, `architecture:grovemoe`, `architecture:hunyuan-dense`, `architecture:hunyuan-moe`, `architecture:hy_v3`, `architecture:internlm2`, `architecture:jais`, `architecture:jais2`, `architecture:jamba`, `architecture:jina-bert-v2`, `architecture:jina-bert-v3`, `architecture:kimi-k3`, `architecture:kimi-linear`, `architecture:lfm2moe`, `architecture:llada`, `architecture:llada-moe`, `architecture:llama-embed`, `architecture:maincoder`, `architecture:mamba`, `architecture:mamba2`, `architecture:minicpm`, `architecture:minicpm3`, `architecture:minimax-01`, `architecture:minimax-m2`, `architecture:mistral4`, `architecture:modern-bert`, `architecture:muse-glimmer`, `architecture:nemotron`, `architecture:nemotron_h`, `architecture:nemotron_h_moe`, `architecture:neo-bert`, `architecture:nomic-bert`, `architecture:nomic-bert-moe`, `architecture:olmo2`, `architecture:olmoe`, `architecture:openelm`, `architecture:orion`, `architecture:pangu-embedded`, `architecture:phi2`, `architecture:phi3`, `architecture:phimoe`, `architecture:plamo`, `architecture:plamo2`, `architecture:plm`, `architecture:qwen`, `architecture:qwen2moe`, `architecture:qwen2vl`, `architecture:qwen3`, `architecture:qwen35`, `architecture:qwen3moe`, `architecture:qwen3next`, `architecture:refact`, `architecture:rnd1`, `architecture:seed_oss`, `architecture:smallthinker`, `architecture:smollm3`, `architecture:stablelm`, `architecture:t5`, `architecture:t5encoder`, `architecture:talkie`, `architecture:xverse` | immutable representative GGUF; full-logit prefill and cached-decode parity; deterministic generation/state evidence | +| `evidence-only` | `architecture-runtime-evidence` | `architecture:arcee`, `architecture:arctic`, `architecture:baichuan`, `architecture:bailingmoe`, `architecture:bert`, `architecture:bitnet`, `architecture:bloom`, `architecture:chatglm`, `architecture:codeshell`, `architecture:cohere2`, `architecture:command-r`, `architecture:dbrx`, `architecture:deci`, `architecture:deepseek`, `architecture:dots1`, `architecture:dream`, `architecture:ernie4_5`, `architecture:ernie4_5-moe`, `architecture:eurobert`, `architecture:exaone`, `architecture:falcon`, `architecture:falcon-h1`, `architecture:gemma`, `architecture:gemma-embedding`, `architecture:gemma2`, `architecture:gemma3`, `architecture:gemma4`, `architecture:glm-dsa`, `architecture:granite`, `architecture:granitehybrid`, `architecture:granitemoe`, `architecture:grok`, `architecture:grovemoe`, `architecture:hunyuan-dense`, `architecture:hunyuan-moe`, `architecture:hy_v3`, `architecture:internlm2`, `architecture:jais`, `architecture:jais2`, `architecture:jamba`, `architecture:jina-bert-v2`, `architecture:jina-bert-v3`, `architecture:kimi-k3`, `architecture:kimi-linear`, `architecture:lfm2moe`, `architecture:llada`, `architecture:llada-moe`, `architecture:llama-embed`, `architecture:maincoder`, `architecture:mamba`, `architecture:mamba2`, `architecture:minicpm`, `architecture:minicpm3`, `architecture:minimax-01`, `architecture:minimax-m2`, `architecture:mistral4`, `architecture:modern-bert`, `architecture:muse-glimmer`, `architecture:nemotron`, `architecture:nemotron_h`, `architecture:nemotron_h_moe`, `architecture:neo-bert`, `architecture:nomic-bert`, `architecture:nomic-bert-moe`, `architecture:olmo2`, `architecture:olmoe`, `architecture:openelm`, `architecture:orion`, `architecture:pangu-embedded`, `architecture:phi2`, `architecture:phi3`, `architecture:phimoe`, `architecture:plamo`, `architecture:plamo2`, `architecture:plm`, `architecture:qwen`, `architecture:qwen2moe`, `architecture:qwen2vl`, `architecture:qwen3`, `architecture:qwen35`, `architecture:qwen3moe`, `architecture:qwen3next`, `architecture:refact`, `architecture:rnd1`, `architecture:seed_oss`, `architecture:smallthinker`, `architecture:smollm3`, `architecture:stablelm`, `architecture:t5`, `architecture:t5encoder`, `architecture:talkie`, `architecture:xverse` | immutable representative GGUF; full-logit prefill and cached-decode parity; deterministic generation/state evidence | | `evidence-only` | `mtp-runtime-evidence` | `mtp:hy_v3`, `mtp:qwen35` | target acceptance loop; cache-threaded draft/target parity | | `evidence-only` | `projector-runtime-evidence` | `projector:adapter`, `projector:cogvlm`, `projector:deepseekocr`, `projector:deepseekocr2`, `projector:dots3note_a`, `projector:dots3note_v`, `projector:dots_ocr`, `projector:exaone4_5`, `projector:gemma3`, `projector:gemma3na`, `projector:gemma3nv`, `projector:gemma4a`, `projector:gemma4ua`, `projector:gemma4uv`, `projector:gemma4v`, `projector:glm4v`, `projector:glma`, `projector:granite4_vision`, `projector:granite_speech`, `projector:hunyuanvl`, `projector:idefics3`, `projector:internvl`, `projector:janus_pro`, `projector:kimik25`, `projector:kimivl`, `projector:ldp`, `projector:ldpv2`, `projector:lfm2`, `projector:lfm2a`, `projector:lightonocr`, `projector:llama4`, `projector:meralion`, `projector:mimo_audio`, `projector:mimovl`, `projector:minicpmv4_6`, `projector:minimax_m3`, `projector:mlp`, `projector:muse-glimmer`, `projector:musicflamingo`, `projector:nemotron_v2_vl`, `projector:paddleocr`, `projector:parakeet`, `projector:pixtral`, `projector:pockettts_spkenc`, `projector:qwen2.5o`, `projector:qwen2.5vl_merger`, `projector:qwen2a`, `projector:qwen2vl_merger`, `projector:qwen3a`, `projector:qwen3tts_spkenc`, `projector:qwen3vl_merger`, `projector:step3vl`, `projector:ultravox`, `projector:voxtral`, `projector:yasa2`, `projector:youtuvl` | paired text target; processor boundary; deterministic multimodal package execution | | `intentionally-rejected` | `policy-rejections` | `architecture:bailingmoe2`, `architecture:clip`, `architecture:dots3note`, `architecture:exaone-moe`, `architecture:exaone4`, `architecture:glm4`, `architecture:glm4moe`, `architecture:gptj` | policy change plus independent correctness proof | @@ -182,7 +183,7 @@ Reason codes are concise user-facing categories; detailed architecture audits re | Canonical architecture | Aliases | Import route | Tensor exactness | Config/tensor/graph/runtime/quantized import | Restriction or evidence gap | |---|---|---|---|---|---| | `afmoe` | — | none (fails before config extraction) | audited-direct-loader-conditional-union | config=deferred; tensor_map=deferred; graph=deferred; runtime=deferred; quantized_import=supported | CONFIG_DEFERRED — AFMoE combines sandwich norms, Q/K norms, sigmoid-gated attention, MuP embedding scaling, a dense prefix, correction-biased routed/shared experts, and optional interleaved sliding-window attention. | -| `apertus` | — | model=`apertus`; tensor=`llama`+`apertus_extras` | exact-direct-loader-conditional-union | config=supported; tensor_map=supported; graph=supported; runtime=deferred; quantized_import=supported | RUNTIME_EVIDENCE_PENDING — Config extraction, exact tensor-name closure, and a full synthetic GGUF graph build are covered, but no representative real-weight GGUF has yet passed ORT parity or generation validation. | +| `apertus` | — | model=`apertus`; tensor=`llama`+`apertus_extras` | exact-direct-loader-conditional-union | config=supported; tensor_map=supported; graph=supported; runtime=supported; quantized_import=supported | EVIDENCED_SCOPE — Runtime support is restricted to the pinned Apertus-v1.1-1.5B-Instruct BF16 artifact's exact-float CPU route, official tokenizer revision, full-logit stateful evidence, and ORT GenAI 0.15.2. | | `arcee` | — | model=`arcee`; tensor=`arcee` | not claimed | config=supported; tensor_map=supported; graph=supported; runtime=deferred; quantized_import=supported | RUNTIME_EVIDENCE_PENDING — Config extraction, exact tensor-name closure, and a full synthetic GGUF graph build are covered, but no representative real-weight GGUF has yet passed ORT parity or generation validation. | | `arctic` | — | model=`arctic`; module=`arctic_gguf`; tensor=`llama`+`arctic_extras` | not claimed | config=supported; tensor_map=supported; graph=supported; runtime=deferred; quantized_import=rejected | RUNTIME_EVIDENCE_PENDING / FLOAT_IMPORT_ONLY — The mobius graph uses floating Linear modules for this architecture, so no MatMulNBits or BlockQuantizedMatMul target can consume preserved GGUF projection weights. | | `arwkv7` | — | none (fails before config extraction) | audited-direct-loader-conditional-union | config=deferred; tensor_map=deferred; graph=deferred; runtime=deferred; quantized_import=supported | CONFIG_DEFERRED — ARWKV7 wraps RWKV7's delta-rule matrix recurrence in a distinct one-shift RMSNorm/Qwen residual topology with optional five-versus-six-way interpolation, optional gate/group norm, and Qwen SwiGLU. | diff --git a/src/mobius/integrations/gguf/_apertus_test.py b/src/mobius/integrations/gguf/_apertus_test.py index 7aee0fa3c..7ac9e84b2 100644 --- a/src/mobius/integrations/gguf/_apertus_test.py +++ b/src/mobius/integrations/gguf/_apertus_test.py @@ -110,7 +110,8 @@ def test_apertus_registry_promotes_exact_float_and_quantized_graph_import() -> N assert spec.tensor_map_recipe == ("llama", "apertus_extras") assert spec.config_postprocessor == "apertus" assert spec.quantized_import.value == "supported" - assert spec.runtime.value == "deferred" + assert spec.runtime.value == "supported" + assert spec.runtime_evidence_ids == ("apertus-v1.1-1.5b-instruct-bf16-ort-genai-0.15.2",) def test_apertus_consumes_serialized_rope_and_xielu_values_exactly() -> None: @@ -132,6 +133,42 @@ def test_apertus_consumes_serialized_rope_and_xielu_values_exactly() -> None: assert config.attn_o_bias +def test_apertus_accepts_generic_xielu_metadata_and_default_rope() -> None: + metadata = _metadata() + for suffix in ("alpha_p", "alpha_n", "beta", "eps"): + metadata[f"xielu.{suffix}"] = metadata.pop(f"apertus.xielu.{suffix}") + tensors = _tensors() + tensors.pop("rope_freqs.weight") + + config = gguf_to_config(_FakeGGUF(metadata, tensors)) + + assert config.rope_type == "default" + assert config.rope_scaling is None + assert config.xielu_alpha_p == (-0.2, -0.1) + assert config.xielu_alpha_n == (-1.0, -0.9) + + +def test_apertus_rejects_conflicting_xielu_metadata_or_unserialized_rope_scaling() -> None: + metadata = _metadata() + metadata["xielu.beta"] = 0.25 + with pytest.raises(ValueError, match=r"conflicting apertus\.xielu\.beta and xielu\.beta"): + gguf_to_config(_FakeGGUF(metadata, _tensors())) + + metadata = _metadata() + metadata["apertus.rope.scaling.type"] = "longrope" + tensors = _tensors() + tensors.pop("rope_freqs.weight") + with pytest.raises(ValueError, match="factorless RoPE"): + gguf_to_config(_FakeGGUF(metadata, tensors)) + + metadata = _metadata() + metadata["apertus.rope.scaling.original_context_length"] = 16 + tensors = _tensors() + tensors.pop("rope_freqs.weight") + with pytest.raises(ValueError, match="original_context_length requires"): + gguf_to_config(_FakeGGUF(metadata, tensors)) + + def test_apertus_maps_qk_norm_and_output_bias_without_value_transform() -> None: assert ( map_gguf_to_hf_names("blk.1.attn_q_norm.bias", "apertus") @@ -164,6 +201,11 @@ def test_apertus_rejects_quantized_or_malformed_serialized_rope_factors() -> Non with pytest.raises(ValueError, match=r"shape \(2,\)"): gguf_to_config(_FakeGGUF(_metadata(), tensors)) + tensors = _tensors() + tensors["rope_factors_short.weight"] = np.ones((2,), np.float32) + with pytest.raises(ValueError, match="must contain exactly"): + gguf_to_config(_FakeGGUF(_metadata(), tensors)) + def test_apertus_consumes_complete_longrope_factor_pair_exactly() -> None: tensors = _tensors() diff --git a/src/mobius/integrations/gguf/_arch_registry.py b/src/mobius/integrations/gguf/_arch_registry.py index 3a62f7048..183bb140d 100644 --- a/src/mobius/integrations/gguf/_arch_registry.py +++ b/src/mobius/integrations/gguf/_arch_registry.py @@ -2258,15 +2258,14 @@ model_type="apertus", tensor_map_recipe=("llama", "apertus_extras"), config_postprocessor="apertus", - required_metadata=( - "attention.layer_norm_rms_epsilon", - "xielu.alpha_n", - "xielu.alpha_p", - "xielu.beta", - "xielu.eps", + required_metadata=("attention.layer_norm_rms_epsilon",), + runtime=Support.SUPPORTED, + runtime_evidence_ids=("apertus-v1.1-1.5b-instruct-bf16-ort-genai-0.15.2",), + reason=( + "Runtime support is restricted to the pinned Apertus-v1.1-1.5B-Instruct " + "BF16 artifact's exact-float CPU route, official tokenizer revision, full-logit " + "stateful evidence, and ORT GenAI 0.15.2." ), - runtime=Support.DEFERRED, - reason=_RUNTIME_VALIDATION_PENDING, ), GGUFArchitectureSpec( gguf_arch="minicpm", diff --git a/src/mobius/integrations/gguf/_component_export.py b/src/mobius/integrations/gguf/_component_export.py index 932906883..306815d65 100644 --- a/src/mobius/integrations/gguf/_component_export.py +++ b/src/mobius/integrations/gguf/_component_export.py @@ -248,7 +248,10 @@ def attach_runtime_unvalidated_report( else None ) tokenizer_verdict = getattr(pkg, "gguf_tokenizer_verdict", None) - if tokenizer is None: + if tokenizer is None or ( + tokenizer_exported + and (tokenizer.support != "supported" or tokenizer.output != "exported") + ): if tokenizer_verdict is None: raise ValueError( "Runtime-unvalidated packaging requires the tokenizer disposition captured " diff --git a/src/mobius/integrations/gguf/_config_mapping.py b/src/mobius/integrations/gguf/_config_mapping.py index 4177fb2d0..beaac253f 100644 --- a/src/mobius/integrations/gguf/_config_mapping.py +++ b/src/mobius/integrations/gguf/_config_mapping.py @@ -473,6 +473,7 @@ def _validate_closed_rope_scaling_metadata( arch: str, *, allowed_suffixes: set[str] | None = None, + allowed_types: set[str] | None = None, ) -> None: """Reject RoPE-scaling metadata outside an architecture's exact subset.""" allowed = {f"{arch}.rope.scaling.{suffix}" for suffix in (allowed_suffixes or set())} @@ -495,7 +496,8 @@ def _validate_closed_rope_scaling_metadata( f"{arch} has unsupported RoPE scaling metadata: {', '.join(sorted(unsupported))}" ) scaling_type = metadata.get(f"{arch}.rope.scaling.type") - if "type" in (allowed_suffixes or set()) and scaling_type not in (None, "", "none"): + accepted_types = {None, "", "none", *(allowed_types or set())} + if "type" in (allowed_suffixes or set()) and scaling_type not in accepted_types: raise ValueError( f"{arch} rope.scaling.type={scaling_type!r} is not in the exact supported subset" ) @@ -1415,7 +1417,24 @@ def _apertus_postprocess( layers = config.num_hidden_layers def per_layer(suffix: str) -> tuple[float, ...]: - raw = metadata[f"{arch}.xielu.{suffix}"] + qualified = f"{arch}.xielu.{suffix}" + generic = f"xielu.{suffix}" + if qualified in metadata and generic in metadata: + if not np.array_equal( + np.asarray(metadata[qualified]), np.asarray(metadata[generic]) + ): + raise ValueError( + f"GGUF architecture {arch!r} has conflicting {qualified} and {generic}" + ) + raw = metadata[qualified] + elif qualified in metadata: + raw = metadata[qualified] + elif generic in metadata: + raw = metadata[generic] + else: + raise ValueError( + f"GGUF architecture {arch!r} is missing required metadata: xielu.{suffix}" + ) values = list(raw) if isinstance(raw, (list, tuple, np.ndarray)) else [raw] * layers if len(values) != layers: raise ValueError( @@ -1440,15 +1459,36 @@ def per_layer(suffix: str) -> tuple[float, ...]: } if any(type_id not in {0, 1, 30} for type_id in raw_types.values()): raise ValueError("Apertus serialized RoPE factors must use F32/F16/BF16 storage") + _validate_closed_rope_scaling_metadata( + metadata, + arch, + allowed_suffixes={"original_context_length", "type"}, + allowed_types={"longrope"}, + ) + scaling_type = metadata.get(f"{arch}.rope.scaling.type") + original_context_key = f"{arch}.rope.scaling.original_context_length" + factor_pair = {"rope_factors_long.weight", "rope_factors_short.weight"} + if original_context_key in metadata and rope_names != factor_pair: + raise ValueError( + "Apertus rope.scaling.original_context_length requires the complete " + "rope_factors_long.weight/rope_factors_short.weight pair" + ) - if rope_names == {"rope_freqs.weight"}: + if not rope_names: + if scaling_type not in {None, "", "none"}: + raise ValueError( + "Apertus factorless RoPE requires rope.scaling.type to be absent or 'none'" + ) + short_factors = long_factors = None + original_context = config.max_position_embeddings + elif rope_names == {"rope_freqs.weight"}: factors = np.asarray(model.get_tensor("rope_freqs.weight"), dtype=np.float32).reshape( -1 ) short_factors = long_factors = factors original_context = config.max_position_embeddings - elif rope_names == {"rope_factors_long.weight", "rope_factors_short.weight"}: - original_context_raw = metadata.get(f"{arch}.rope.scaling.original_context_length") + elif rope_names == factor_pair: + original_context_raw = metadata.get(original_context_key) if original_context_raw is None: raise ValueError( "Apertus LongRoPE factors require apertus.rope.scaling.original_context_length" @@ -1470,18 +1510,24 @@ def per_layer(suffix: str) -> tuple[float, ...]: "Apertus GGUF must contain exactly rope_freqs.weight or the complete " "rope_factors_long.weight/rope_factors_short.weight pair" ) + if rope_names and scaling_type not in {None, "", "none", "longrope"}: + raise ValueError( + "Apertus serialized RoPE factors require rope.scaling.type to be absent, " + "'none', or 'longrope'" + ) - expected = config.head_dim // 2 - for name, factors in (("short", short_factors), ("long", long_factors)): - if ( - factors.shape != (expected,) - or not np.all(np.isfinite(factors)) - or np.any(factors <= 0) - ): - raise ValueError( - f"Apertus {name} RoPE factors must have shape ({expected},) and be " - "finite positive values" - ) + if short_factors is not None and long_factors is not None: + expected = config.head_dim // 2 + for name, factors in (("short", short_factors), ("long", long_factors)): + if ( + factors.shape != (expected,) + or not np.all(np.isfinite(factors)) + or np.any(factors <= 0) + ): + raise ValueError( + f"Apertus {name} RoPE factors must have shape ({expected},) and be " + "finite positive values" + ) q_biases = tuple(f"blk.{layer}.attn_q_norm.bias" in names for layer in range(layers)) k_biases = tuple(f"blk.{layer}.attn_k_norm.bias" in names for layer in range(layers)) @@ -1491,11 +1537,15 @@ def per_layer(suffix: str) -> tuple[float, ...]: attn_qk_norm=True, attn_q_norm_biases=q_biases, attn_k_norm_biases=k_biases, - rope_type="longrope", - rope_scaling={ - "short_factor": short_factors.tolist(), - "long_factor": long_factors.tolist(), - }, + rope_type="default" if short_factors is None else "longrope", + rope_scaling=( + None + if short_factors is None or long_factors is None + else { + "short_factor": short_factors.tolist(), + "long_factor": long_factors.tolist(), + } + ), original_max_position_embeddings=original_context, xielu_alpha_p=per_layer("alpha_p"), xielu_alpha_n=per_layer("alpha_n"), diff --git a/src/mobius/integrations/gguf/_docs_test.py b/src/mobius/integrations/gguf/_docs_test.py index 9b4cc85d2..c48e4a51f 100644 --- a/src/mobius/integrations/gguf/_docs_test.py +++ b/src/mobius/integrations/gguf/_docs_test.py @@ -147,6 +147,10 @@ def test_runtime_support_requires_structured_evidence() -> None: ("gpt2", ("gpt2-q2-k-ort-genai-0.15.2",)), ("starcoder2", ("tiny-starcoder2-q2-k-ort-genai-0.15.2",)), ("olmo", ("tiny-olmo-q2-k-ort-genai-0.15.2",)), + ( + "apertus", + ("apertus-v1.1-1.5b-instruct-bf16-ort-genai-0.15.2",), + ), ("mpt", ("tiny-mpt-q2-k-ort-genai-0.15.2",)), ("gptneox", ("pythia-70m-q2-k-ort-genai-0.15.2",)), ("starcoder", ("tiny-starcoder-q2-k-ort-genai-0.15.2",)), diff --git a/src/mobius/integrations/gguf/_quant_capabilities_test.py b/src/mobius/integrations/gguf/_quant_capabilities_test.py index 0a8c7e288..a2d6e24c5 100644 --- a/src/mobius/integrations/gguf/_quant_capabilities_test.py +++ b/src/mobius/integrations/gguf/_quant_capabilities_test.py @@ -203,10 +203,10 @@ def test_selected_real_artifacts_stay_within_global_budget() -> None: assert isinstance(policy, dict) assert isinstance(artifacts, list) assert isinstance(lossy, list) - assert len(artifacts) == 10 + assert len(artifacts) == 11 assert len(lossy) == 1 selected = sum(int(record["size"]) for record in [*artifacts, *lossy]) - assert selected == 2_724_371_296 + assert selected == 5_752_423_904 assert selected == policy["selected_artifact_bytes"] assert selected <= policy["max_selected_artifact_bytes"] assert lossy[0]["lfs_sha256"] == ( diff --git a/src/mobius/integrations/gguf/_route_census_test.py b/src/mobius/integrations/gguf/_route_census_test.py index cdf66ad69..4699cc3b3 100644 --- a/src/mobius/integrations/gguf/_route_census_test.py +++ b/src/mobius/integrations/gguf/_route_census_test.py @@ -54,14 +54,14 @@ def test_every_route_has_one_actionable_classification() -> None: assert {item.category for item in items} == allowed assert all(item.batch and item.dependencies and item.reason.strip() for item in items) assert Counter(item.kind for item in items) == { - "architecture": 136, + "architecture": 135, "projector": 60, "tokenizer": 56, "mtp": 22, } assert Counter(item.category for item in items) == { "dependency-or-mobius-abi-blocked": 99, - "evidence-only": 151, + "evidence-only": 150, "intentionally-rejected": 19, "artifact-unavailable": 5, } diff --git a/src/mobius/integrations/gguf/_runtime_evidence.py b/src/mobius/integrations/gguf/_runtime_evidence.py index 5fa9bae9e..bc9106fe1 100644 --- a/src/mobius/integrations/gguf/_runtime_evidence.py +++ b/src/mobius/integrations/gguf/_runtime_evidence.py @@ -534,6 +534,71 @@ def _is_hex(value: str) -> bool: "dynamic KV cache prefill, full-sequence replay, rollback, reorder, and 20 decode steps" ) +_APERTUS_15B_BF16_ORT_GENAI = GGUFRuntimeEvidence( + evidence_id="apertus-v1.1-1.5b-instruct-bf16-ort-genai-0.15.2", + architecture="apertus", + repository="MrMeOrYou/Apertus-v1.1-1.5B-Instruct-GGUF", + revision="88c75ad49566d3c2157d03709bf772262c3241ed", + filename="Apertus-v1.1-1.5B-Instruct-BF16.gguf", + size=3_028_052_608, + lfs_sha256="f9ec154d0ec29dad1f6465b458b7f27bd25ad7b9a3899233ae98ca6d358501c2", + config_repository="swiss-ai/Apertus-v1.1-1.5B-Instruct", + config_revision="9e9d01154446a645d30f04174cf1515a38058be7", + tokenizer_repository="swiss-ai/Apertus-v1.1-1.5B-Instruct", + tokenizer_revision="9e9d01154446a645d30f04174cf1515a38058be7", + tokenizer_metadata_sha256=( + "3097bd9f22efd32db9045c4705d978539dc8adeeae763324062a5e1a73fc24a5" + ), + tokenizer_assets=( + ( + "chat_template.jinja", + 5_250, + "4afab8361a4bd0c2994404e0b0851dbaad461a06e56bdfdb635aae3977473c19", + ), + ( + "special_tokens_map.json", + 560, + "9f69883bd70fc5d8b55822799837a216d3ac4fb565e05256a0d4f9850404bbc5", + ), + ( + "tokenizer.json", + 17_078_368, + "be12f4375d655cc740864e3a9041bcddd8477942f209d9e7f27f6c8767162638", + ), + ( + "tokenizer_config.json", + 177_274, + "77b14a0664585c26065f07d7a4c852a4615c83348d9378e23def01957bbd3f57", + ), + ), + tensor_count=163, + tensor_qtypes=(("BF16", 98), ("F32", 65)), + import_route='{"architecture":"apertus","config_sha256":"2a9660c48656c0980e76f1b374262bf8383256603e8b4221aa99bcfa11c510b0","execution_provider":"cpu","model_type":"apertus","module_type":"apertus","preserve_quantization":false,"registry_import":{"config_key_map":null,"config_postprocessor":"apertus","llama_qk_permute":false,"offset_norm":false,"required_metadata":["attention.layer_norm_rms_epsilon"],"rope_interleave":false,"tensor_processor":null,"v_head_reorder":false,"vlm_builder":null},"route_schema":1,"static_cache":false,"task":{"class":"builtins.str","state":"text-generation"},"tensor_map_recipe":["llama","apertus_extras"]}', + source_fidelity=True, + storage_quantized=False, + target_storage_format="float", + compute_mode="float operators", + graph_files=_LOW_COST_GRAPH_FILES, + graph_sha256="4bb91bade19d41559cb524e28692453851fe67274cc58109787ee968df3e0fe5", + runtime_package_files=( + "chat_template.jinja", + "export_report.json", + *_LOW_COST_RUNTIME_PACKAGE_FILES, + ), + runtime_package_sha256="97582549e9c5b4114f3bcfa81c92f16aba9d14dd0ea0d2b20acaefd9060e6486", + parity_test=("test_promoted_gguf_full_runtime_evidence[apertus-v1.1-1.5b-instruct-bf16]"), + parity_kind="full-logit", + deterministic_test=( + "test_promoted_gguf_full_runtime_evidence[apertus-v1.1-1.5b-instruct-bf16]" + ), + stateful_semantics=_LOW_COST_STATEFUL_SEMANTICS, + execution_provider="CPUExecutionProvider", + onnxruntime_version="1.29.0", + runtime="ort-genai", + runtime_version="0.15.2", + runtime_package_schema=FINAL_RUNTIME_PACKAGE_SCHEMA, +) + _GPT2_Q2_K_ORT_GENAI = GGUFRuntimeEvidence( evidence_id="gpt2-q2-k-ort-genai-0.15.2", architecture="gpt2", @@ -863,6 +928,7 @@ def _is_hex(value: str) -> bool: { record.evidence_id: record for record in ( + _APERTUS_15B_BF16_ORT_GENAI, _LFM2_350M_F16_ORT_GENAI, _QWEN25_Q8_ORT_GENAI, _QWEN35MOE_087B_Q2_K_ORT_GENAI, @@ -993,12 +1059,7 @@ def find_matching_runtime_evidence( "The GGUF source no longer matches the exact artifact identity captured during " f"graph construction: built={built_identity!r}, current={current_identity!r}." ) - if ( - not evidence_ids - or runtime_version is None - or tokenizer_repository is None - or tokenizer_revision is None - ): + if not evidence_ids or runtime_version is None: return None identity = built_identity candidates = [ @@ -1014,8 +1075,14 @@ def find_matching_runtime_evidence( and _RUNTIME_EVIDENCE[evidence_id].tensor_count == identity.tensor_count and _RUNTIME_EVIDENCE[evidence_id].tensor_qtypes == identity.tensor_qtypes and _RUNTIME_EVIDENCE[evidence_id].import_route == import_route - and _RUNTIME_EVIDENCE[evidence_id].tokenizer_repository == tokenizer_repository - and _RUNTIME_EVIDENCE[evidence_id].tokenizer_revision == tokenizer_revision + and ( + tokenizer_repository is None + or _RUNTIME_EVIDENCE[evidence_id].tokenizer_repository == tokenizer_repository + ) + and ( + tokenizer_revision is None + or _RUNTIME_EVIDENCE[evidence_id].tokenizer_revision == tokenizer_revision + ) ] if len(candidates) > 1: raise RuntimeError( diff --git a/src/mobius/integrations/gguf/_runtime_evidence_test.py b/src/mobius/integrations/gguf/_runtime_evidence_test.py index fed5e26b9..4f0c96ec7 100644 --- a/src/mobius/integrations/gguf/_runtime_evidence_test.py +++ b/src/mobius/integrations/gguf/_runtime_evidence_test.py @@ -144,9 +144,11 @@ def test_production_runtime_evidence_requires_final_package_regeneration() -> No records = iter_runtime_evidence() assert records - assert all( - record.runtime_package_schema != FINAL_RUNTIME_PACKAGE_SCHEMA for record in records - ) + assert [ + record.evidence_id + for record in records + if record.runtime_package_schema == FINAL_RUNTIME_PACKAGE_SCHEMA + ] == ["apertus-v1.1-1.5b-instruct-bf16-ort-genai-0.15.2"] def test_low_cost_runtime_batch_manifest_is_closed_and_within_budget() -> None: @@ -284,6 +286,21 @@ def test_matching_evidence_binds_arch_runtime_source_qtypes_and_route( ) is record ) + assert ( + matching_runtime_evidence( + (record.evidence_id,), + architecture="llama", + runtime="onnx-genai", + source_path=source, + gguf_model=_model(), + built_identity=gguf_artifact_identity(source, _model(), architecture="llama"), + import_route=record.import_route, + runtime_version="1.0.0", + tokenizer_repository=None, + tokenizer_revision=None, + ) + is record + ) with pytest.raises(ValueError, match="No unique GGUF runtime evidence"): matching_runtime_evidence( diff --git a/src/mobius/integrations/gguf/_runtime_package.py b/src/mobius/integrations/gguf/_runtime_package.py index 2ce04f0ed..55752fd4b 100644 --- a/src/mobius/integrations/gguf/_runtime_package.py +++ b/src/mobius/integrations/gguf/_runtime_package.py @@ -32,6 +32,7 @@ from mobius.integrations.gguf._arch_registry import get_arch_spec from mobius.integrations.gguf._runtime_evidence import ( + FINAL_RUNTIME_PACKAGE_SCHEMA, RuntimeEvidenceUnavailableError, gguf_artifact_identity, gguf_graph_package_identity, @@ -382,11 +383,7 @@ def write_gguf_runtime_package( ) evidence = None evidence_warning: str | None = None - if ( - tokenizer_repository is not None - and tokenizer_revision is not None - and architecture_spec.runtime_evidence_ids - ): + if architecture_spec.runtime_evidence_ids: try: evidence = matching_runtime_evidence( architecture_spec.runtime_evidence_ids, @@ -397,11 +394,24 @@ def write_gguf_runtime_package( built_identity=built_identity, import_route=import_route, runtime_version=runtime_version, - tokenizer_repository=tokenizer_repository, - tokenizer_revision=tokenizer_revision, + tokenizer_repository=None, + tokenizer_revision=None, ) except RuntimeEvidenceUnavailableError as error: evidence_warning = f"{error} Export continues without claiming runtime validation." + if ( + evidence is not None + and tokenizer_repository is not None + and ( + tokenizer_repository != evidence.tokenizer_repository + or tokenizer_revision != evidence.tokenizer_revision + ) + ): + raise ValueError( + "The explicit tokenizer source conflicts with exact runtime evidence: " + f"requested={tokenizer_repository}@{tokenizer_revision}, " + f"evidence={evidence.tokenizer_repository}@{evidence.tokenizer_revision}." + ) if not source_model.source_matches_path(): raise ValueError( "The GGUF source changed while runtime evidence was being matched; " @@ -433,9 +443,12 @@ def write_gguf_runtime_package( "The GGUF tokenizer metadata no longer matches the identity captured during " "graph construction; refusing to pair the graph with a replaced tokenizer source." ) - if getattr(built_verdict, "blocker_category", None) is not None: - # Exact runtime evidence cannot override a tokenizer route that the - # authoritative tokenizer census has explicitly withheld. + if ( + getattr(built_verdict, "blocker_category", None) is not None + and getattr(evidence, "runtime_package_schema", None) != FINAL_RUNTIME_PACKAGE_SCHEMA + ): + # Identifier-level blockers remain authoritative unless a final-package + # record proves this exact artifact through the strict checks below. evidence = None output_dir.parent.mkdir(parents=True, exist_ok=True) @@ -452,7 +465,14 @@ def write_gguf_runtime_package( "The GGUF source changed while target/MTP graphs were being serialized; " "refusing package publication." ) - graph_identity = gguf_graph_package_identity(stage) + graph_files = tuple( + sorted( + path.relative_to(stage).as_posix() + for path in stage.rglob("*") + if path.is_file() and path.name != "export_report.json" + ) + ) + graph_identity = gguf_graph_package_identity(stage, files=graph_files) validation_warnings: list[str] = [] if evidence_warning is not None: validation_warnings.append(evidence_warning) diff --git a/src/mobius/integrations/gguf/_runtime_package_test.py b/src/mobius/integrations/gguf/_runtime_package_test.py index ea00feb97..55db6b1fb 100644 --- a/src/mobius/integrations/gguf/_runtime_package_test.py +++ b/src/mobius/integrations/gguf/_runtime_package_test.py @@ -132,12 +132,14 @@ def _source_model(): def _write_runtime(pkg, source, output, **kwargs): + tokenizer_repository = kwargs.pop("tokenizer_repository", _TOKENIZER_REPOSITORY) + tokenizer_revision = kwargs.pop("tokenizer_revision", _TOKENIZER_REVISION) return write_gguf_runtime_package( pkg, source, output, - tokenizer_repository=_TOKENIZER_REPOSITORY, - tokenizer_revision=_TOKENIZER_REVISION, + tokenizer_repository=tokenizer_repository, + tokenizer_revision=tokenizer_revision, **kwargs, ) @@ -476,8 +478,29 @@ def test_atomically_emits_graph_tokenizer_and_runtime_config(self, tmp_path): assert (out / "export_report.json").is_file() assert not list(tmp_path.glob(".out.*.tmp")) - def test_exact_runtime_evidence_marks_package_validated(self, tmp_path): + @pytest.mark.parametrize( + "tokenizer_kwargs", + ( + {}, + {"tokenizer_repository": None, "tokenizer_revision": None}, + ), + ids=("matching-explicit-tokenizer", "evidence-pinned-tokenizer"), + ) + def test_exact_runtime_evidence_marks_package_validated(self, tmp_path, tokenizer_kwargs): pkg = _FakePackage() + blocker = GGUFTokenizerVerdict( + route="deferred", + model="gpt2", + pre="blocked-pre", + canonical_pre="blocked-pre", + token_count=2, + metadata_sha256="f" * 64, + blocker_category="compiled-llama.cpp-semantic-dependency", + audit_status="deferred-compiled-semantics", + reason="compiled tokenizer semantics require exact evidence", + ) + pkg.gguf_tokenizer_verdict = blocker + attach_tokenizer_export_report(pkg, blocker, model_route="llama") out = tmp_path / "out" evidence = SimpleNamespace( evidence_id="test-evidence", @@ -489,6 +512,7 @@ def test_exact_runtime_evidence_marks_package_validated(self, tmp_path): "tokenizer.json", ), runtime_package_sha256="c" * 64, + runtime_package_schema=_runtime_package.FINAL_RUNTIME_PACKAGE_SCHEMA, tokenizer_repository=_TOKENIZER_REPOSITORY, tokenizer_revision=_TOKENIZER_REVISION, tokenizer_metadata_sha256="f" * 64, @@ -500,6 +524,14 @@ def test_exact_runtime_evidence_marks_package_validated(self, tmp_path): "mobius.integrations.gguf._runtime_package.matching_runtime_evidence", return_value=evidence, ), + mock.patch( + "mobius.integrations.gguf._runtime_package.inspect_gguf_tokenizer", + return_value=blocker, + ), + mock.patch( + "mobius.integrations.gguf._component_export.resolve_tokenizer_export_verdict", + return_value=blocker, + ), mock.patch( "mobius.integrations.gguf._runtime_package.gguf_graph_package_identity", side_effect=( @@ -509,12 +541,20 @@ def test_exact_runtime_evidence_marks_package_validated(self, tmp_path): sha256=evidence.runtime_package_sha256, ), ), - ), + ) as package_identity, ): - _write_runtime(pkg, tmp_path / "m.gguf", out, runtime_version="0.15.2") + _write_runtime( + pkg, + tmp_path / "m.gguf", + out, + runtime_version="0.15.2", + **tokenizer_kwargs, + ) + assert package_identity.call_args_list[0].kwargs["files"] == ("model.onnx",) assert pkg.export_report is not None assert pkg.export_report.export_status == "complete" + assert pkg.export_report.component("tokenizer").output == "exported" assert pkg.export_report.runtime_validation_status == "validated" assert pkg.export_report.end_to_end_runnable is True runtime_component = pkg.export_report.component("runtime") @@ -525,6 +565,24 @@ def test_exact_runtime_evidence_marks_package_validated(self, tmp_path): assert compatibility["runtime_validation_status"] == "validated" assert compatibility["runtime_evidence_id"] == "test-evidence" + def test_explicit_tokenizer_conflict_rejects_before_package_save(self, tmp_path): + pkg = _FakePackage() + with ( + _successful_runtime_dependencies(), + pytest.raises(ValueError, match="conflicts with exact runtime evidence"), + ): + _write_runtime( + pkg, + tmp_path / "m.gguf", + tmp_path / "out", + runtime_version="0.15.2", + tokenizer_repository="attacker/replacement", + tokenizer_revision="d" * 40, + ) + + assert pkg.saved_to is None + assert not (tmp_path / "out").exists() + def test_missing_evidence_runtime_version_does_not_block_export(self, tmp_path): pkg = _FakePackage() out = tmp_path / "out" @@ -662,6 +720,10 @@ def test_tokenizer_blocker_omits_only_tokenizer_assets(self, tmp_path, caplog): with ( caplog.at_level("WARNING"), _successful_runtime_dependencies(), + mock.patch( + "mobius.integrations.gguf._runtime_package.matching_runtime_evidence", + return_value=None, + ), mock.patch( "mobius.integrations.gguf._runtime_package.inspect_gguf_tokenizer", return_value=blocker, @@ -669,7 +731,13 @@ def test_tokenizer_blocker_omits_only_tokenizer_assets(self, tmp_path, caplog): ): pkg.gguf_tokenizer_verdict = blocker attach_tokenizer_export_report(pkg, blocker, model_route="llama") - artifacts = _write_runtime(pkg, tmp_path / "m.gguf", out) + artifacts = _write_runtime( + pkg, + tmp_path / "m.gguf", + out, + tokenizer_repository=None, + tokenizer_revision=None, + ) assert (out / "model.onnx").is_file() assert Path(artifacts["export_report"]).is_file() diff --git a/testdata/cases/causal-lm/apertus-v1.1-1.5b-instruct-bf16.yaml b/testdata/cases/causal-lm/apertus-v1.1-1.5b-instruct-bf16.yaml new file mode 100644 index 000000000..d12df910f --- /dev/null +++ b/testdata/cases/causal-lm/apertus-v1.1-1.5b-instruct-bf16.yaml @@ -0,0 +1,61 @@ +model_id: "swiss-ai/Apertus-v1.1-1.5B-Instruct" +model_type: "apertus" +revision: "9e9d01154446a645d30f04174cf1515a38058be7" +task_type: "text-generation" +dtype: "float32" + +gguf: + repository: "MrMeOrYou/Apertus-v1.1-1.5B-Instruct-GGUF" + revision: "88c75ad49566d3c2157d03709bf772262c3241ed" + filename: "Apertus-v1.1-1.5B-Instruct-BF16.gguf" + size: 3028052608 + lfs_sha256: "f9ec154d0ec29dad1f6465b458b7f27bd25ad7b9a3899233ae98ca6d358501c2" + tensor_count: 163 + tensor_qtypes: + BF16: 98 + F32: 65 + execution_provider: "cpu" + config_sha256: "2a9660c48656c0980e76f1b374262bf8383256603e8b4221aa99bcfa11c510b0" + preserve_quantization: false + keep_quantized: true + tokenizer: + repository: "swiss-ai/Apertus-v1.1-1.5B-Instruct" + revision: "9e9d01154446a645d30f04174cf1515a38058be7" + metadata_sha256: "3097bd9f22efd32db9045c4705d978539dc8adeeae763324062a5e1a73fc24a5" + identity_status: "exact" + assets: + - filename: "chat_template.jinja" + size: 5250 + sha256: "4afab8361a4bd0c2994404e0b0851dbaad461a06e56bdfdb635aae3977473c19" + - filename: "special_tokens_map.json" + size: 560 + sha256: "9f69883bd70fc5d8b55822799837a216d3ac4fb565e05256a0d4f9850404bbc5" + - filename: "tokenizer.json" + size: 17078368 + sha256: "be12f4375d655cc740864e3a9041bcddd8477942f209d9e7f27f6c8767162638" + - filename: "tokenizer_config.json" + size: 177274 + sha256: "77b14a0664585c26065f07d7a4c852a4615c83348d9378e23def01957bbd3f57" + +ort_genai: + tier: "real" + runtime_versions: ["0.15.2"] + model_type: "decoder" + runtime_evidence_id: "apertus-v1.1-1.5b-instruct-bf16-ort-genai-0.15.2" + execution_provider: "cpu" + max_download_bytes: 3050000000 + released_capabilities: + "0.15.2": {generic_decoder: true, state_groups: false} + +inputs: + prompts: + - "Hello" + +level: "L4+L5" + +generation: + max_new_tokens: 20 + do_sample: false + exact_match: true + +notes: "Pinned BF16 GGUF; independent raw-reader same-value oracle, full logits, 20-step cached decode, state semantics, and ORT GenAI generation are verified." diff --git a/testdata/cases/schema.json b/testdata/cases/schema.json index 42882605d..06f4149d6 100644 --- a/testdata/cases/schema.json +++ b/testdata/cases/schema.json @@ -308,7 +308,7 @@ "max_download_bytes": { "type": "integer", "minimum": 1, - "maximum": 1073741824, + "maximum": 17179869184, "description": "Hard per-case aggregate artifact download budget." }, "released_capabilities": { diff --git a/testdata/evidence/gguf_quantization_capabilities.json b/testdata/evidence/gguf_quantization_capabilities.json index 2f4806485..3bb328889 100644 --- a/testdata/evidence/gguf_quantization_capabilities.json +++ b/testdata/evidence/gguf_quantization_capabilities.json @@ -384,7 +384,7 @@ "max_selected_artifact_bytes": 17179869184, "preserved_definition": "Only byte-identical native blocks or exact affine repacks preserve source represented values. Dequantize/requantize is never preserved.", "runtime_definition": "Runtime support requires immutable same-artifact full-logit parity plus deterministic prefill/decode/replay/rollback/reorder evidence.", - "selected_artifact_bytes": 2724371296, + "selected_artifact_bytes": 5752423904, "target_storage_definition": "Lossy affine normalization may still produce supported packed target storage while source_fidelity is false." }, "qtypes": [ @@ -3010,6 +3010,42 @@ "F32": 55 } }, + { + "evidence_ids": [ + "apertus-v1.1-1.5b-instruct-bf16-ort-genai-0.15.2" + ], + "filename": "Apertus-v1.1-1.5B-Instruct-BF16.gguf", + "lfs_sha256": "f9ec154d0ec29dad1f6465b458b7f27bd25ad7b9a3899233ae98ca6d358501c2", + "repository": "MrMeOrYou/Apertus-v1.1-1.5B-Instruct-GGUF", + "revision": "88c75ad49566d3c2157d03709bf772262c3241ed", + "runtime_results": [ + { + "compute_mode": "float operators", + "deterministic_test": "test_promoted_gguf_full_runtime_evidence[apertus-v1.1-1.5b-instruct-bf16]", + "downstream_runtime": "ort-genai", + "downstream_runtime_version": "0.15.2", + "evidence_id": "apertus-v1.1-1.5b-instruct-bf16-ort-genai-0.15.2", + "execution_provider": "CPUExecutionProvider", + "graph_sha256": "4bb91bade19d41559cb524e28692453851fe67274cc58109787ee968df3e0fe5", + "import_route": "{\"architecture\":\"apertus\",\"config_sha256\":\"2a9660c48656c0980e76f1b374262bf8383256603e8b4221aa99bcfa11c510b0\",\"execution_provider\":\"cpu\",\"model_type\":\"apertus\",\"module_type\":\"apertus\",\"preserve_quantization\":false,\"registry_import\":{\"config_key_map\":null,\"config_postprocessor\":\"apertus\",\"llama_qk_permute\":false,\"offset_norm\":false,\"required_metadata\":[\"attention.layer_norm_rms_epsilon\"],\"rope_interleave\":false,\"tensor_processor\":null,\"v_head_reorder\":false,\"vlm_builder\":null},\"route_schema\":1,\"static_cache\":false,\"task\":{\"class\":\"builtins.str\",\"state\":\"text-generation\"},\"tensor_map_recipe\":[\"llama\",\"apertus_extras\"]}", + "limitations": null, + "onnxruntime_version": "1.29.0", + "parity_kind": "full-logit", + "parity_test": "test_promoted_gguf_full_runtime_evidence[apertus-v1.1-1.5b-instruct-bf16]", + "result": "passed", + "runtime_package_sha256": "97582549e9c5b4114f3bcfa81c92f16aba9d14dd0ea0d2b20acaefd9060e6486", + "source_fidelity": true, + "stateful_semantics": "dynamic KV cache prefill, full-sequence replay, rollback, reorder, and 20 decode steps", + "storage_quantized": false, + "target_storage_format": "float" + } + ], + "size": 3028052608, + "tensor_qtypes": { + "BF16": 98, + "F32": 65 + } + }, { "evidence_ids": [ "qwen2.5-0.5b-instruct-q8-ort-genai-0.15.2" diff --git a/tests/gguf_small_model_runtime_integration_test.py b/tests/gguf_small_model_runtime_integration_test.py index 8a6ce53a1..ea3454c62 100644 --- a/tests/gguf_small_model_runtime_integration_test.py +++ b/tests/gguf_small_model_runtime_integration_test.py @@ -57,6 +57,7 @@ ) from mobius.integrations.gguf._reader import GGUFModel from mobius.integrations.gguf._runtime_evidence import ( + FINAL_RUNTIME_PACKAGE_SCHEMA, GGUFRuntimeEvidence, gguf_graph_package_identity, runtime_evidence, @@ -260,6 +261,37 @@ class _PromotedRuntimeCase: _PROMOTED_RUNTIME_CASES = ( + _PromotedRuntimeCase( + name="apertus-v1.1-1.5b-instruct-bf16", + evidence_id="apertus-v1.1-1.5b-instruct-bf16-ort-genai-0.15.2", + prompt="Hello", + generated_tokens=( + 1044, + 1362, + 4525, + 1875, + 1317, + 1278, + 4304, + 1307, + 19466, + 1321, + 1362, + 6483, + 2151, + 6370, + 1317, + 8178, + 17616, + 1046, + 1362, + 6483, + ), + atol=2e-4, + cache_atol=4e-5, + reference_kind="apertus", + required_free_bytes=13_000_000_000, + ), _PromotedRuntimeCase( name="gpt2-q2-k", evidence_id="gpt2-q2-k-ort-genai-0.15.2", @@ -740,7 +772,8 @@ def _load_low_cost_same_value_reference( # The oracle reads and maps upstream GGUF tensors directly. It deliberately # shares no Mobius tensor loader, name mapping, value processor, or normalizer. source: dict[str, torch.Tensor] = {} - for tensor in GGUFReader(gguf_path).tensors: + reader = GGUFReader(gguf_path) + for tensor in reader.tensors: shape = tuple(int(dimension) for dimension in reversed(tensor.shape)) if tensor.tensor_type in (GGMLQuantizationType.F32, GGMLQuantizationType.F16): value = tensor.data.reshape(shape) @@ -776,6 +809,11 @@ def undo_llama_qk_permutation(value: torch.Tensor, heads: int) -> torch.Tensor: renamed: dict[str, torch.Tensor] = {} global_names = { + "apertus": { + "token_embd.weight": "model.embed_tokens.weight", + "output_norm.weight": "model.norm.weight", + "output.weight": "lm_head.weight", + }, "gptneox": { "token_embd.weight": "gpt_neox.embed_in.weight", "output_norm.weight": "gpt_neox.final_layer_norm.weight", @@ -798,6 +836,18 @@ def undo_llama_qk_permutation(value: torch.Tensor, heads: int) -> torch.Tensor: }, }[reference_kind] layer_names = { + "apertus": { + "attn_norm": "attention_layernorm", + "attn_q": "self_attn.q_proj", + "attn_k": "self_attn.k_proj", + "attn_v": "self_attn.v_proj", + "attn_output": "self_attn.o_proj", + "attn_q_norm": "self_attn.q_norm", + "attn_k_norm": "self_attn.k_norm", + "ffn_norm": "feedforward_layernorm", + "ffn_up": "mlp.up_proj", + "ffn_down": "mlp.down_proj", + }, "gptneox": { "attn_norm": "input_layernorm", "attn_qkv": "attention.query_key_value", @@ -833,6 +883,7 @@ def undo_llama_qk_permutation(value: torch.Tensor, heads: int) -> torch.Tensor: }, }[reference_kind] layer_prefix = { + "apertus": "model.layers", "gptneox": "gpt_neox.layers", "mpt": "transformer.blocks", "olmo": "model.layers", @@ -855,9 +906,27 @@ def undo_llama_qk_permutation(value: torch.Tensor, heads: int) -> torch.Tensor: ) value = undo_llama_qk_permutation(value, heads) renamed[target] = value - assert len(renamed) == len(source) - if reference_kind == "gptneox": + if reference_kind == "apertus": + for suffix in ("alpha_p", "alpha_n", "beta", "eps"): + field = reader.fields[f"xielu.{suffix}"] + values = [float(field.parts[index][0]) for index in field.data] + assert len(values) == int(hf_config.num_hidden_layers) + for layer, value in enumerate(values): + target = f"model.layers.{layer}.mlp.act_fn.{suffix}" + renamed[target] = torch.tensor( + [value] if suffix in {"alpha_p", "alpha_n"} else value, + dtype=torch.float32, + ) + expected_metadata_parameters = ( + 4 * int(hf_config.num_hidden_layers) if reference_kind == "apertus" else 0 + ) + assert len(renamed) == len(source) + expected_metadata_parameters + + if reference_kind == "apertus": + reference = AutoModelForCausalLM.from_config(hf_config) + reference.load_state_dict(renamed, assign=True, strict=True) + elif reference_kind == "gptneox": reference = GPTNeoXForCausalLM(hf_config) reference.load_state_dict(renamed, strict=True) elif reference_kind == "starcoder": @@ -1615,10 +1684,6 @@ def capture_save(package: ModelPackage, *args: object, **kwargs: object) -> None "ort-genai", "--runtime-version", installed_version, - "--tokenizer-repository", - evidence.tokenizer_repository, - "--tokenizer-revision", - evidence.tokenizer_revision, "--local-files-only", ] if case.dequantize: @@ -1649,6 +1714,12 @@ def capture_save(package: ModelPackage, *args: object, **kwargs: object) -> None package = ModelPackage.load(output_dir) assert tuple(package) == ("model",) assert package.gguf_quantization_report == source_report + if evidence.runtime_package_schema == FINAL_RUNTIME_PACKAGE_SCHEMA: + assert package.export_report is not None + assert package.export_report.export_status == "complete" + assert package.export_report.runtime_validation_status == "validated" + assert package.export_report.end_to_end_runnable is True + assert package.export_report.component("tokenizer").output == "exported" assert ( GGUFQuantizationReport.read_json(output_dir / "quantization_report.json") == source_report @@ -1689,7 +1760,7 @@ def capture_save(package: ModelPackage, *args: object, **kwargs: object) -> None ) else: with ExitStack() as reference_guard: - if case.reference_kind in {"gptneox", "mpt", "olmo", "starcoder"}: + if case.reference_kind in {"apertus", "gptneox", "mpt", "olmo", "starcoder"}: for helper in ( "mobius.integrations.gguf._builder._load_dequantized_state_dict", "mobius.integrations.gguf._builder._normalize_gguf_weights",