From b541f19f98eab677326592d3ea829e29b70f0c36 Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Tue, 22 Sep 2026 17:58:17 +0800 Subject: [PATCH 01/78] feat(v41): add model metadata and text tokenizer entry --- docs/developer-guide/deepseek-v41-entry.md | 61 ++++ mkdocs.yml | 1 + pyproject.toml | 3 + pypto_serving/cli/main.py | 7 + pypto_serving/model/deepseek_v41/__init__.py | 10 + pypto_serving/model/deepseek_v41/config.py | 75 ++++ .../model/deepseek_v41/encoding.LICENSE | 21 ++ pypto_serving/model/deepseek_v41/encoding.py | 128 +++++++ pypto_serving/model/deepseek_v41/tokenizer.py | 58 +++ pypto_serving/model/model_family.py | 13 +- pypto_serving/model/model_loader.py | 10 +- pypto_serving/model/tokenizer.py | 14 +- .../deepseek_v41/chat_encoding_golden.json | 337 ++++++++++++++++++ tests/fixtures/deepseek_v41/config.json | 170 +++++++++ .../unit/model/deepseek_v41/test_encoding.py | 199 +++++++++++ tests/unit/model/deepseek_v41/test_entry.py | 126 +++++++ 16 files changed, 1226 insertions(+), 7 deletions(-) create mode 100644 docs/developer-guide/deepseek-v41-entry.md create mode 100644 pypto_serving/model/deepseek_v41/__init__.py create mode 100644 pypto_serving/model/deepseek_v41/config.py create mode 100644 pypto_serving/model/deepseek_v41/encoding.LICENSE create mode 100644 pypto_serving/model/deepseek_v41/encoding.py create mode 100644 pypto_serving/model/deepseek_v41/tokenizer.py create mode 100644 tests/fixtures/deepseek_v41/chat_encoding_golden.json create mode 100644 tests/fixtures/deepseek_v41/config.json create mode 100644 tests/unit/model/deepseek_v41/test_encoding.py create mode 100644 tests/unit/model/deepseek_v41/test_entry.py diff --git a/docs/developer-guide/deepseek-v41-entry.md b/docs/developer-guide/deepseek-v41-entry.md new file mode 100644 index 00000000..9b181520 --- /dev/null +++ b/docs/developer-guide/deepseek-v41-entry.md @@ -0,0 +1,61 @@ +# DeepSeek V4.1 text entry + +This is the first stage of V4.1 integration into the upstream serving architecture: +model identification, metadata validation and local text tokenization. It does +not load checkpoint tensors, allocate devices or enable generation. Both the +model loader and the serving CLI reject V4.1 execution explicitly until its +weight loader and executor are integrated, instead of treating it as Qwen. + +```python +from pypto_serving.model.deepseek_v41.config import load_text_config +from pypto_serving.model.tokenizer import load_tokenizer + +model_dir = "/path/to/DeepSeek-V4.1-Flash" +config = load_text_config(model_dir) +tokenizer = load_tokenizer(model_dir) +ids = tokenizer.apply_chat_template( + [{"role": "user", "content": "Hello"}], tokenize=True, +) +``` + +The text config reads dimensions from nested `text_config`, special IDs from +the checkpoint metadata, and compression modes for the target backbone only. +Trailing draft-layer compression entries are not backbone layers. This is +metadata validation, not a guarantee that every shape is supported by kernels. + +The tokenizer loads local files through the existing shared fast-tokenizer +loader. Raw `encode` adds no special tokens. Chat encoding inserts the model's +own BOS and assistant prefix once. Supported messages have string content; +tools, images and structured content are rejected. The text prompt format is +derived from the independent DeepSeek encoder at revision +`dba1be0a40aa45a94ad051997016db3960a90277`; the checked-in golden fixture records +its source and digest. A synthetic local tokenizer exercises actual token IDs +and decoding but is not a full-checkpoint tokenizer validation. + +## Staged integration + +1. Model identification, text config and tokenizer (this stage). +2. Selective checkpoint loading, weight formats and shard contracts. +3. Embedding and initial mHC residual/pre-mix state. +4. Attention mHC and RMSNorm. +5. SWA, then C2A and C1A attention with cache publication and prefill/decode state. +6. Attention mHC post, FFN mHC/norm, MoE and FFN mHC post. +7. Cross-layer state and TP/EP composition, then the full backbone. +8. Final HC/norm, LM head and greedy decode loop. +9. Scheduler/HTTP lifecycle, recovery, memory observations and M0 acceptance. + +Engram, vision and speculative decoding are deferred. Optional metadata for +these modules may be present in config.json; this stage does not initialize or +load them. Later numerical acceptance must use a reference with the same +Engram-disabled scope, not claim equality with the complete official model. + +Development starts from upstream main. Existing experimental V4.1 code can be +reused selectively with tests; its backend and runtime are not prerequisites. +Kernel stages should track current pypto-lib interfaces and record the revision +used for validation. This metadata/tokenizer stage does not change the lib pin. + +## Validation + +```bash +python -m pytest tests/unit/model/deepseek_v41 tests/unit/model/test_tokenizer.py tests/unit/cli/test_parallel_options.py -q +``` diff --git a/mkdocs.yml b/mkdocs.yml index 68f375a4..d325dc93 100644 --- a/mkdocs.yml +++ b/mkdocs.yml @@ -111,3 +111,4 @@ nav: - Weight Staging: developer-guide/weight-staging.md - DeepSeek V4 Runtime: developer-guide/deepseek-v4-runtime.md - DeepSeek V4 DSpark: developer-guide/deepseek-v4-dspark.md + - DeepSeek V4.1 Entry: developer-guide/deepseek-v41-entry.md diff --git a/pyproject.toml b/pyproject.toml index 0471d368..adaa685a 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -26,3 +26,6 @@ pypto-prepack-deepseek-v4 = "pypto_serving.tools.prepack_deepseek_v4:main" where = ["."] include = ["pypto_serving*"] namespaces = false + +[tool.setuptools.package-data] +"pypto_serving.model.deepseek_v41" = ["encoding.LICENSE"] diff --git a/pypto_serving/cli/main.py b/pypto_serving/cli/main.py index 06d07e3c..23817261 100644 --- a/pypto_serving/cli/main.py +++ b/pypto_serving/cli/main.py @@ -277,6 +277,13 @@ def build_serving_engine_config(args: argparse.Namespace) -> EngineConfig: devices = parse_device_ids(args.devices, default_device=args.device) model_config_data = read_model_config(model_dir) model_family = detect_model_family(model_config_data) + if model_family == "deepseek_v41": + from pypto_serving.model.deepseek_v41.config import load_text_config + + load_text_config(model_dir) + raise NotImplementedError( + "V4.1 configuration and tokenizer are supported; serving execution is not integrated yet." + ) model_variant = _resolve_model_variant(args) _validate_prefill_chunk_size( model_family, diff --git a/pypto_serving/model/deepseek_v41/__init__.py b/pypto_serving/model/deepseek_v41/__init__.py new file mode 100644 index 00000000..4b13ba7c --- /dev/null +++ b/pypto_serving/model/deepseek_v41/__init__.py @@ -0,0 +1,10 @@ +# Copyright (c) PyPTO Contributors. +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of +# CANN Open Software License Agreement Version 2.0 (the "License"). +# Please refer to the License for details. You may not use this file except in compliance with the License. +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. +# See LICENSE in the root of the software repository for the full text of the License. +# ----------------------------------------------------------------------------------------------------------- + +"""DeepSeek V4.1 text model metadata and tokenization.""" diff --git a/pypto_serving/model/deepseek_v41/config.py b/pypto_serving/model/deepseek_v41/config.py new file mode 100644 index 00000000..0dd917e3 --- /dev/null +++ b/pypto_serving/model/deepseek_v41/config.py @@ -0,0 +1,75 @@ +# Copyright (c) PyPTO Contributors. +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of +# CANN Open Software License Agreement Version 2.0 (the "License"). +# Please refer to the License for details. You may not use this file except in compliance with the License. +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. +# See LICENSE in the root of the software repository for the full text of the License. +# ----------------------------------------------------------------------------------------------------------- +"""Metadata-only V4.1 text configuration; no weights or device runtime are loaded.""" + +from dataclasses import dataclass +import json +import math +from pathlib import Path + +from ..model_family import is_deepseek_v41_config + + +@dataclass(frozen=True) +class V41TextConfig: + """Backbone dimensions, not a claim that a checkpoint can execute yet. + + Vision, Engram and draft metadata may exist in the checkpoint. They are + outside this stage; compression modes here cover only the target backbone. + """ + + vocab_size: int + hidden_size: int + num_hidden_layers: int + num_attention_heads: int + head_dim: int + max_position_embeddings: int + hc_mult: int + rms_norm_eps: float + compress_ratios: tuple[int, ...] + bos_token_id: int | None + eos_token_id: int | None + pad_token_id: int | None + + @classmethod + def from_dict(cls, raw: dict) -> "V41TextConfig": + if not isinstance(raw, dict) or not is_deepseek_v41_config(raw): + raise ValueError("expected a DeepSeek V4.1 model config") + text = raw.get("text_config") + if not isinstance(text, dict) or text.get("model_type") != "deepseek_v41_text": + raise ValueError("text_config.model_type must be deepseek_v41_text") + names = ("vocab_size", "hidden_size", "num_hidden_layers", "num_attention_heads", + "head_dim", "max_position_embeddings", "hc_mult") + values = {} + for name in names: + value = text.get(name) + if type(value) is not int or value <= 0: + raise ValueError(f"text_config.{name} must be a positive integer") + values[name] = value + eps = text.get("rms_norm_eps") + if type(eps) not in (float, int) or not math.isfinite(eps) or eps <= 0: + raise ValueError("text_config.rms_norm_eps must be finite and positive") + ratios = text.get("compress_ratios") + layers = values["num_hidden_layers"] + if not isinstance(ratios, list) or len(ratios) < layers: + raise ValueError("compress_ratios must cover every backbone layer") + if any(type(value) is not int or value not in (0, 1, 2) for value in ratios[:layers]): + raise ValueError("backbone compress_ratios must use 0 (SWA), 1 (C1A), or 2 (C2A)") + for name in ("bos_token_id", "eos_token_id", "pad_token_id"): + value = raw.get(name, text.get(name)) + if value is not None and (type(value) is not int or not 0 <= value < values["vocab_size"]): + raise ValueError(f"{name} must be a vocabulary index or null") + values[name] = value + return cls(**values, rms_norm_eps=float(eps), compress_ratios=tuple(ratios[:layers])) + + +def load_text_config(model_dir: str | Path) -> V41TextConfig: + """Read only config.json; unknown optional modules are not initialized.""" + raw = json.loads((Path(model_dir) / "config.json").read_text(encoding="utf-8")) + return V41TextConfig.from_dict(raw) diff --git a/pypto_serving/model/deepseek_v41/encoding.LICENSE b/pypto_serving/model/deepseek_v41/encoding.LICENSE new file mode 100644 index 00000000..d84f527e --- /dev/null +++ b/pypto_serving/model/deepseek_v41/encoding.LICENSE @@ -0,0 +1,21 @@ +MIT License + +Copyright (c) 2023 DeepSeek + +Permission is hereby granted, free of charge, to any person obtaining a copy +of this software and associated documentation files (the "Software"), to deal +in the Software without restriction, including without limitation the rights +to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +copies of the Software, and to permit persons to whom the Software is +furnished to do so, subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +SOFTWARE. diff --git a/pypto_serving/model/deepseek_v41/encoding.py b/pypto_serving/model/deepseek_v41/encoding.py new file mode 100644 index 00000000..53b6272e --- /dev/null +++ b/pypto_serving/model/deepseek_v41/encoding.py @@ -0,0 +1,128 @@ +# Copyright (c) PyPTO Contributors. +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of +# CANN Open Software License Agreement Version 2.0 (the "License"). +# Please refer to the License for details. You may not use this file except in compliance with the License. +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. +# See LICENSE in the root of the software repository for the full text of the License. +# ----------------------------------------------------------------------------------------------------------- +"""Text-only V4.1 generation prompts, independent of the V4 encoding rules. + +Format source: DeepSeek-V4.1-Flash encoding/encoding.py at revision +dba1be0a40aa45a94ad051997016db3960a90277, SHA256 +502bdaec8a3fd88ebc24c4721a7038fbe42f2063c664638127056107920035c1. +This implementation covers string messages and historical reasoning, not tools, +image inputs, response schemas, context deltas, or auxiliary classification tasks. +The source format is MIT licensed; see encoding.LICENSE for its notice. +""" + +from __future__ import annotations + +from collections.abc import Mapping, Sequence + + +BOS_TOKEN = "<\uff5cbegin\u2581of\u2581sentence\uff5c>" +EOS_TOKEN = "<\uff5cend\u2581of\u2581sentence\uff5c>" +SYSTEM_TOKEN = "<\uff5cSystem\uff5c>" +USER_TOKEN = "<\uff5cUser\uff5c>" +ASSISTANT_TOKEN = "<\uff5cAssistant\uff5c>" +LATEST_REMINDER_TOKEN = "<\uff5clatest_reminder\uff5c>" +IMAGE_TOKEN = "<\uff5cdeepseek_image\uff5c>" +REFERENCE_REVISION = "dba1be0a40aa45a94ad051997016db3960a90277" +REFERENCE_SHA256 = "502bdaec8a3fd88ebc24c4721a7038fbe42f2063c664638127056107920035c1" + + +def _reasoning_budget(effort: str | int | None) -> int: + if effort is None: + return 75 + if type(effort) is int and 1 <= effort <= 100: + return effort + if isinstance(effort, str) and effort in {"low", "high", "max"}: + return {"low": 50, "high": 75, "max": 100}[effort] + raise ValueError("V4.1 reasoning_effort must be low/high/max or an integer in [1, 100]") + + +def encode_messages( + messages: Sequence[Mapping[str, object]], + *, + thinking_mode: str = "chat", + reasoning_effort: str | int | None = None, + drop_thinking: bool = True, +) -> str: + """Render the supported pure-text subset byte-for-byte like the pinned reference.""" + if thinking_mode not in ("chat", "thinking") or type(drop_thinking) is not bool: + raise ValueError("thinking_mode must be chat/thinking and drop_thinking must be bool") + budget = _reasoning_budget(reasoning_effort) + if not messages: + raise ValueError("V4.1 chat requires at least one message") + records = tuple(messages) + for message in records: + if not isinstance(message, Mapping): + raise ValueError("V4.1 chat messages must be objects") + unknown = set(message) - {"role", "content", "reasoning_content", "wo_eos"} + if unknown: + raise ValueError(f"V4.1 text encoding does not support message fields: {sorted(unknown)}") + if message.get("role") not in ("system", "user", "assistant", "latest_reminder"): + raise ValueError(f"V4.1 text encoding does not support role {message.get('role')!r}") + if not isinstance(message.get("content"), str): + raise ValueError( + "V4.1 text message content must be a string; image/content blocks are unsupported" + ) + reasoning = message.get("reasoning_content") + if reasoning is not None and (not isinstance(reasoning, str) or message["role"] != "assistant"): + raise ValueError("reasoning_content is supported only as an assistant string") + if type(message.get("wo_eos", False)) is not bool: + raise ValueError("wo_eos must be bool") + forbidden = (IMAGE_TOKEN, "", "") + if any(marker in message["content"] for marker in forbidden) or ( + isinstance(reasoning, str) and IMAGE_TOKEN in reasoning + ): + raise ValueError("raw image special tokens require an implemented vision input path") + # The official tool preprocessor also merges adjacent text-only user messages. + merged = [] + for message in records: + if merged and message["role"] == "user" and merged[-1]["role"] == "user": + merged[-1]["content"] += "\n\n" + message["content"] + else: + merged.append(dict(message)) + records = tuple(merged) + last_user = max( + ( + i + for i, message in enumerate(records) + if message["role"] == "user" or (message["role"] == "system" and i > 0) + ), + default=-1, + ) + parts = [BOS_TOKEN] + for index, message in enumerate(records): + role, content = message["role"], message["content"] + if index == 0: + if thinking_mode == "thinking" or role == "system": + parts.append(SYSTEM_TOKEN) + if thinking_mode == "thinking": + parts.append( + f"Reasoning Effort: {budget} " + "(range 1-100, the higher the value, the more thorough the reasoning)\n\n" + ) + if role == "system": + if index > 0: + parts.append(SYSTEM_TOKEN) + parts.append(content) + elif role == "user": + parts.extend((USER_TOKEN, content)) + elif role == "latest_reminder": + parts.extend((LATEST_REMINDER_TOKEN, content)) + else: + if thinking_mode == "thinking" and (not drop_thinking or index > last_user): + parts.extend((message.get("reasoning_content") or "", "")) + parts.append(content) + if not message.get("wo_eos", False): + parts.append(EOS_TOKEN) + next_role = records[index + 1]["role"] if index + 1 < len(records) else None + if next_role not in (None, "assistant", "latest_reminder"): + continue + if role == "user" or (role == "system" and index > 0): + thinking = thinking_mode == "thinking" and (not drop_thinking or index >= last_user) + parts.extend((ASSISTANT_TOKEN, "" if thinking else "")) + return "".join(parts) diff --git a/pypto_serving/model/deepseek_v41/tokenizer.py b/pypto_serving/model/deepseek_v41/tokenizer.py new file mode 100644 index 00000000..39d5a867 --- /dev/null +++ b/pypto_serving/model/deepseek_v41/tokenizer.py @@ -0,0 +1,58 @@ +# Copyright (c) PyPTO Contributors. +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of +# CANN Open Software License Agreement Version 2.0 (the "License"). +# Please refer to the License for details. You may not use this file except in compliance with the License. +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. +# See LICENSE in the root of the software repository for the full text of the License. +# ----------------------------------------------------------------------------------------------------------- +"""V4.1 chat prompts over the existing local fast-tokenizer loading and decoding.""" + +from __future__ import annotations + +from collections.abc import Mapping, Sequence + +from pypto_serving.model.tokenizer import TransformersTokenizerAdapter + +from .encoding import encode_messages + + +class DeepSeekV41TokenizerAdapter(TransformersTokenizerAdapter): + """Pure-text generation adapter; the checkpoint has no Jinja chat_template.""" + + def apply_chat_template(self, messages: Sequence[Mapping[str, object]], **kwargs): + allowed = { + "thinking", + "enable_thinking", + "thinking_mode", + "reasoning_effort", + "drop_thinking", + "tokenize", + "add_generation_prompt", + } + unknown = set(kwargs) - allowed + if unknown: + raise ValueError(f"V4.1 text generation does not support template options: {sorted(unknown)}") + for name in ("thinking", "enable_thinking", "tokenize", "add_generation_prompt", "drop_thinking"): + if name in kwargs and type(kwargs[name]) is not bool: + raise ValueError(f"V4.1 {name} must be bool") + if not kwargs.get("add_generation_prompt", True): + raise ValueError("V4.1 currently supports add_generation_prompt=True only") + flags = [kwargs[name] for name in ("thinking", "enable_thinking") if name in kwargs] + if len(set(flags)) > 1: + raise ValueError("thinking and enable_thinking must agree") + mode = kwargs.get("thinking_mode", "thinking" if flags and flags[0] else "chat") + if mode not in ("chat", "thinking"): + raise ValueError("thinking_mode must be chat/thinking") + if "thinking_mode" in kwargs and flags and (mode == "thinking") != flags[0]: + raise ValueError("thinking_mode conflicts with the supplied thinking flag") + effort = kwargs.get("reasoning_effort") + if effort == "none": + mode, effort = "chat", None + prompt = encode_messages( + messages, + thinking_mode=mode, + reasoning_effort=effort, + drop_thinking=kwargs.get("drop_thinking", True), + ) + return self.encode(prompt) if kwargs.get("tokenize", False) else prompt diff --git a/pypto_serving/model/model_family.py b/pypto_serving/model/model_family.py index a4d0d002..f535d945 100644 --- a/pypto_serving/model/model_family.py +++ b/pypto_serving/model/model_family.py @@ -14,7 +14,7 @@ from typing import Literal -ModelFamily = Literal["deepseek_v4", "qwen"] +ModelFamily = Literal["deepseek_v41", "deepseek_v4", "qwen"] def read_model_config(model_dir: str | Path) -> dict[str, object]: @@ -41,6 +41,17 @@ def is_deepseek_v4_config(config_data: dict[str, object]) -> bool: return model_type == "deepseek_v4" or "deepseekv4forcausallm" in architectures +def is_deepseek_v41_config(config_data: dict[str, object]) -> bool: + """Identify V4.1 before the V4 and generic Hugging Face paths.""" + architectures = config_data.get("architectures", ()) + return str(config_data.get("model_type", "")).lower() == "deepseek_v41" or ( + isinstance(architectures, (list, tuple)) + and any(str(name).lower() == "deepseekv41forcausallm" for name in architectures) + ) + + def detect_model_family(config_data: dict[str, object]) -> ModelFamily: """Return the serving model family inferred from config metadata.""" + if is_deepseek_v41_config(config_data): + return "deepseek_v41" return "deepseek_v4" if is_deepseek_v4_config(config_data) else "qwen" diff --git a/pypto_serving/model/model_loader.py b/pypto_serving/model/model_loader.py index d7ae5f90..f344f8d5 100644 --- a/pypto_serving/model/model_loader.py +++ b/pypto_serving/model/model_loader.py @@ -24,7 +24,7 @@ RuntimeModel, ) -from .model_family import is_deepseek_v4_config, read_model_config +from .model_family import is_deepseek_v4_config, is_deepseek_v41_config, read_model_config from .tokenizer import TokenizerAdapter, load_tokenizer @@ -457,6 +457,14 @@ def load( **loader_options: object, ) -> LoadedModel: """Load a model directory using an explicit or inferred format.""" + if is_deepseek_v41_config(read_model_config(model_dir)): + from .deepseek_v41.config import load_text_config + + load_text_config(model_dir) + raise NotImplementedError( + "V4.1 configuration and tokenizer are supported; weight loading and execution " + "are not integrated yet. Use load_text_config() and load_tokenizer() for inspection." + ) request = ModelLoadRequest( model_id=model_id, model_dir=model_dir, diff --git a/pypto_serving/model/tokenizer.py b/pypto_serving/model/tokenizer.py index 351fa827..8e8d1a60 100644 --- a/pypto_serving/model/tokenizer.py +++ b/pypto_serving/model/tokenizer.py @@ -198,11 +198,15 @@ def output_parser_id(self) -> str | None: def load_tokenizer(model_dir: str | Path, *, trust_remote_code: bool = False) -> TokenizerAdapter: """Load a local tokenizer and select any model-specific chat encoding.""" model_path = Path(model_dir) - adapter_cls = ( - DeepSeekV4TokenizerAdapter - if detect_model_family(read_model_config(model_path)) == "deepseek_v4" - else TransformersTokenizerAdapter - ) + family = detect_model_family(read_model_config(model_path)) + if family == "deepseek_v41": + from .deepseek_v41.tokenizer import DeepSeekV41TokenizerAdapter + + adapter_cls = DeepSeekV41TokenizerAdapter + elif family == "deepseek_v4": + adapter_cls = DeepSeekV4TokenizerAdapter + else: + adapter_cls = TransformersTokenizerAdapter if (model_path / "tokenizer.json").exists(): return adapter_cls.from_tokenizer_file(str(model_path)) return adapter_cls.from_pretrained( diff --git a/tests/fixtures/deepseek_v41/chat_encoding_golden.json b/tests/fixtures/deepseek_v41/chat_encoding_golden.json new file mode 100644 index 00000000..710af2bd --- /dev/null +++ b/tests/fixtures/deepseek_v41/chat_encoding_golden.json @@ -0,0 +1,337 @@ +{ + "source": "https://huggingface.co/deepseek-ai/DeepSeek-V4.1-Flash/blob/dba1be0a40aa45a94ad051997016db3960a90277/encoding/encoding.py", + "revision": "dba1be0a40aa45a94ad051997016db3960a90277", + "source_sha256": "502bdaec8a3fd88ebc24c4721a7038fbe42f2063c664638127056107920035c1", + "cases": [ + { + "name": "basic", + "messages": [ + { + "role": "user", + "content": "What is 2+2?" + } + ], + "options": { + "thinking_mode": "chat" + }, + "expected": "<\uff5cbegin\u2581of\u2581sentence\uff5c><\uff5cUser\uff5c>What is 2+2?<\uff5cAssistant\uff5c>" + }, + { + "name": "first_system", + "messages": [ + { + "role": "system", + "content": "Be concise." + }, + { + "role": "user", + "content": "What is 2+2?" + } + ], + "options": { + "thinking_mode": "chat" + }, + "expected": "<\uff5cbegin\u2581of\u2581sentence\uff5c><\uff5cSystem\uff5c>Be concise.<\uff5cUser\uff5c>What is 2+2?<\uff5cAssistant\uff5c>" + }, + { + "name": "multi_turn", + "messages": [ + { + "role": "user", + "content": "What is 2+2?" + }, + { + "role": "assistant", + "content": "4", + "reasoning_content": "Add two and two." + }, + { + "role": "user", + "content": "And +1?" + } + ], + "options": { + "thinking_mode": "chat" + }, + "expected": "<\uff5cbegin\u2581of\u2581sentence\uff5c><\uff5cUser\uff5c>What is 2+2?<\uff5cAssistant\uff5c>4<\uff5cend\u2581of\u2581sentence\uff5c><\uff5cUser\uff5c>And +1?<\uff5cAssistant\uff5c>" + }, + { + "name": "mid_system", + "messages": [ + { + "role": "user", + "content": "What is 2+2?" + }, + { + "role": "assistant", + "content": "4", + "reasoning_content": "Add two and two." + }, + { + "role": "system", + "content": "Now use French." + } + ], + "options": { + "thinking_mode": "chat" + }, + "expected": "<\uff5cbegin\u2581of\u2581sentence\uff5c><\uff5cUser\uff5c>What is 2+2?<\uff5cAssistant\uff5c>4<\uff5cend\u2581of\u2581sentence\uff5c><\uff5cSystem\uff5c>Now use French.<\uff5cAssistant\uff5c>" + }, + { + "name": "mid_system_answer", + "messages": [ + { + "role": "user", + "content": "What is 2+2?" + }, + { + "role": "assistant", + "content": "4", + "reasoning_content": "Add two and two." + }, + { + "role": "system", + "content": "Now use French." + }, + { + "role": "assistant", + "content": "D accord." + } + ], + "options": { + "thinking_mode": "chat" + }, + "expected": "<\uff5cbegin\u2581of\u2581sentence\uff5c><\uff5cUser\uff5c>What is 2+2?<\uff5cAssistant\uff5c>4<\uff5cend\u2581of\u2581sentence\uff5c><\uff5cSystem\uff5c>Now use French.<\uff5cAssistant\uff5c>D accord.<\uff5cend\u2581of\u2581sentence\uff5c>" + }, + { + "name": "thinking_default", + "messages": [ + { + "role": "user", + "content": "What is 2+2?" + } + ], + "options": { + "thinking_mode": "thinking" + }, + "expected": "<\uff5cbegin\u2581of\u2581sentence\uff5c><\uff5cSystem\uff5c>Reasoning Effort: 75 (range 1-100, the higher the value, the more thorough the reasoning)\n\n<\uff5cUser\uff5c>What is 2+2?<\uff5cAssistant\uff5c>" + }, + { + "name": "thinking_system_low", + "messages": [ + { + "role": "system", + "content": "Be concise." + }, + { + "role": "user", + "content": "What is 2+2?" + } + ], + "options": { + "thinking_mode": "thinking", + "reasoning_effort": "low" + }, + "expected": "<\uff5cbegin\u2581of\u2581sentence\uff5c><\uff5cSystem\uff5c>Reasoning Effort: 50 (range 1-100, the higher the value, the more thorough the reasoning)\n\nBe concise.<\uff5cUser\uff5c>What is 2+2?<\uff5cAssistant\uff5c>" + }, + { + "name": "thinking_numeric", + "messages": [ + { + "role": "user", + "content": "What is 2+2?" + } + ], + "options": { + "thinking_mode": "thinking", + "reasoning_effort": 64 + }, + "expected": "<\uff5cbegin\u2581of\u2581sentence\uff5c><\uff5cSystem\uff5c>Reasoning Effort: 64 (range 1-100, the higher the value, the more thorough the reasoning)\n\n<\uff5cUser\uff5c>What is 2+2?<\uff5cAssistant\uff5c>" + }, + { + "name": "thinking_max", + "messages": [ + { + "role": "user", + "content": "What is 2+2?" + } + ], + "options": { + "thinking_mode": "thinking", + "reasoning_effort": "max" + }, + "expected": "<\uff5cbegin\u2581of\u2581sentence\uff5c><\uff5cSystem\uff5c>Reasoning Effort: 100 (range 1-100, the higher the value, the more thorough the reasoning)\n\n<\uff5cUser\uff5c>What is 2+2?<\uff5cAssistant\uff5c>" + }, + { + "name": "thinking_drop_old", + "messages": [ + { + "role": "user", + "content": "What is 2+2?" + }, + { + "role": "assistant", + "content": "4", + "reasoning_content": "Add two and two." + }, + { + "role": "user", + "content": "Why?" + } + ], + "options": { + "thinking_mode": "thinking" + }, + "expected": "<\uff5cbegin\u2581of\u2581sentence\uff5c><\uff5cSystem\uff5c>Reasoning Effort: 75 (range 1-100, the higher the value, the more thorough the reasoning)\n\n<\uff5cUser\uff5c>What is 2+2?<\uff5cAssistant\uff5c>4<\uff5cend\u2581of\u2581sentence\uff5c><\uff5cUser\uff5c>Why?<\uff5cAssistant\uff5c>" + }, + { + "name": "thinking_retain_old", + "messages": [ + { + "role": "user", + "content": "What is 2+2?" + }, + { + "role": "assistant", + "content": "4", + "reasoning_content": "Add two and two." + }, + { + "role": "user", + "content": "Why?" + } + ], + "options": { + "thinking_mode": "thinking", + "drop_thinking": false + }, + "expected": "<\uff5cbegin\u2581of\u2581sentence\uff5c><\uff5cSystem\uff5c>Reasoning Effort: 75 (range 1-100, the higher the value, the more thorough the reasoning)\n\n<\uff5cUser\uff5c>What is 2+2?<\uff5cAssistant\uff5c>Add two and two.4<\uff5cend\u2581of\u2581sentence\uff5c><\uff5cUser\uff5c>Why?<\uff5cAssistant\uff5c>" + }, + { + "name": "thinking_final_answer", + "messages": [ + { + "role": "user", + "content": "What is 2+2?" + }, + { + "role": "assistant", + "content": "4", + "reasoning_content": "Add two and two." + } + ], + "options": { + "thinking_mode": "thinking" + }, + "expected": "<\uff5cbegin\u2581of\u2581sentence\uff5c><\uff5cSystem\uff5c>Reasoning Effort: 75 (range 1-100, the higher the value, the more thorough the reasoning)\n\n<\uff5cUser\uff5c>What is 2+2?<\uff5cAssistant\uff5c>Add two and two.4<\uff5cend\u2581of\u2581sentence\uff5c>" + }, + { + "name": "reminder", + "messages": [ + { + "role": "user", + "content": "What is 2+2?" + }, + { + "role": "latest_reminder", + "content": "Today is Friday." + } + ], + "options": { + "thinking_mode": "chat" + }, + "expected": "<\uff5cbegin\u2581of\u2581sentence\uff5c><\uff5cUser\uff5c>What is 2+2?<\uff5cAssistant\uff5c><\uff5clatest_reminder\uff5c>Today is Friday." + }, + { + "name": "without_eos", + "messages": [ + { + "role": "user", + "content": "What is 2+2?" + }, + { + "role": "assistant", + "content": "4", + "reasoning_content": "Add two and two.", + "wo_eos": true + } + ], + "options": { + "thinking_mode": "chat" + }, + "expected": "<\uff5cbegin\u2581of\u2581sentence\uff5c><\uff5cUser\uff5c>What is 2+2?<\uff5cAssistant\uff5c>4" + }, + { + "name": "system_only", + "messages": [ + { + "role": "system", + "content": "Be concise." + } + ], + "options": { + "thinking_mode": "chat" + }, + "expected": "<\uff5cbegin\u2581of\u2581sentence\uff5c><\uff5cSystem\uff5c>Be concise." + }, + { + "name": "multiple_systems", + "messages": [ + { + "role": "system", + "content": "Be concise." + }, + { + "role": "system", + "content": "Also be precise." + }, + { + "role": "user", + "content": "What is 2+2?" + } + ], + "options": { + "thinking_mode": "chat" + }, + "expected": "<\uff5cbegin\u2581of\u2581sentence\uff5c><\uff5cSystem\uff5c>Be concise.<\uff5cSystem\uff5c>Also be precise.<\uff5cUser\uff5c>What is 2+2?<\uff5cAssistant\uff5c>" + }, + { + "name": "adjacent_users", + "messages": [ + { + "role": "user", + "content": "A" + }, + { + "role": "user", + "content": "B" + } + ], + "options": { + "thinking_mode": "chat" + }, + "expected": "<\uff5cbegin\u2581of\u2581sentence\uff5c><\uff5cUser\uff5c>A\n\nB<\uff5cAssistant\uff5c>" + }, + { + "name": "adjacent_users_thinking", + "messages": [ + { + "role": "user", + "content": "A" + }, + { + "role": "user", + "content": "" + }, + { + "role": "user", + "content": "B" + } + ], + "options": { + "thinking_mode": "thinking" + }, + "expected": "<\uff5cbegin\u2581of\u2581sentence\uff5c><\uff5cSystem\uff5c>Reasoning Effort: 75 (range 1-100, the higher the value, the more thorough the reasoning)\n\n<\uff5cUser\uff5c>A\n\n\n\nB<\uff5cAssistant\uff5c>" + } + ] +} diff --git a/tests/fixtures/deepseek_v41/config.json b/tests/fixtures/deepseek_v41/config.json new file mode 100644 index 00000000..09917a91 --- /dev/null +++ b/tests/fixtures/deepseek_v41/config.json @@ -0,0 +1,170 @@ +{ + "architectures": [ + "DeepseekV41ForCausalLM" + ], + "model_type": "deepseek_v41", + "dtype": "bfloat16", + "transformers_version": "5.6.0", + "bos_token_id": 0, + "eos_token_id": 1, + "pad_token_id": 2, + "image_token_id": 129264, + "quantization_config": { + "quant_method": "fp8", + "activation_scheme": "dynamic", + "weight_block_size": [ + 32, + 32 + ], + "scale_fmt": "ue8m0", + "expert_dtype": "fp4" + }, + "text_config": { + "model_type": "deepseek_v41_text", + "vocab_size": 129280, + "hidden_size": 5120, + "moe_intermediate_size": 2304, + "num_hidden_layers": 40, + "num_attention_heads": 64, + "num_key_value_heads": 1, + "head_dim": 512, + "qk_rope_head_dim": 64, + "q_lora_rank": 1280, + "o_lora_rank": 1024, + "o_groups": 8, + "hidden_act": "silu", + "swiglu_limit": 10.0, + "rms_norm_eps": 1e-20, + "attention_bias": false, + "attention_dropout": 0.0, + "initializer_range": 0.02, + "use_cache": true, + "tie_word_embeddings": false, + "max_position_embeddings": 1048576, + "rope_theta": 10000, + "rope_scaling": { + "rope_type": "yarn", + "factor": 16, + "beta_fast": 32, + "beta_slow": 1, + "original_max_position_embeddings": 65536 + }, + "n_routed_experts": 384, + "n_shared_experts": 1, + "num_experts_per_tok": 6, + "scoring_func": "sqrtsoftplus", + "topk_method": "noaux_tc", + "norm_topk_prob": true, + "routed_scaling_factor": 1.5, + "sliding_window": 128, + "compress_ratios": [ + 0, + 0, + 2, + 2, + 2, + 2, + 2, + 2, + 2, + 2, + 2, + 2, + 2, + 2, + 2, + 2, + 2, + 2, + 2, + 2, + 1, + 1, + 1, + 1, + 1, + 1, + 1, + 1, + 1, + 1, + 1, + 1, + 1, + 1, + 1, + 1, + 1, + 1, + 1, + 1, + 0, + 0, + 0 + ], + "compress_rope_theta": 160000, + "kv_source_layer_ids": [ + 2, + 8, + 14, + 20 + ], + "index_source_layer_ids": [ + 2, + 8, + 14, + 20, + 24, + 28, + 32, + 36 + ], + "index_n_heads": 32, + "index_head_dim": 128, + "index_topk": 512, + "candidate_source_layer_id": 20, + "candidate_topk_blocks": 2048, + "candidate_block_size": 8, + "hc_mult": 4, + "hc_sinkhorn_iters": 20, + "hc_eps": 1e-06, + "engram_layer_ids": [ + 1, + 14 + ], + "engram_num_embeddings": [ + 384006168, + 384016682 + ], + "engram_max_ngram_size": 4, + "engram_vocab_size": 16000000, + "engram_n_heads": 8, + "engram_head_dim": 256, + "engram_pad_token_id": 2, + "engram_compressed_vocab_size": 99092, + "num_nextn_predict_layers": 3, + "dspark_block_size": 5, + "dspark_noise_token_id": 128799, + "dspark_target_layer_ids": [ + 37, + 38, + 39 + ], + "dspark_markov_rank": 256, + "dspark_n_routed_experts": 128, + "dspark_num_experts_per_tok": 3 + }, + "vision_config": { + "model_type": "deepseek_v41_vision", + "num_hidden_layers": 32, + "hidden_size": 1024, + "num_attention_heads": 16, + "intermediate_size": 2816, + "patch_size": 14, + "rope_theta": 10000, + "downsample_ratio": 3, + "max_image_tokens": 1024, + "min_pixels": 295936, + "max_wh_ratio": null + } +} diff --git a/tests/unit/model/deepseek_v41/test_encoding.py b/tests/unit/model/deepseek_v41/test_encoding.py new file mode 100644 index 00000000..fa72a4af --- /dev/null +++ b/tests/unit/model/deepseek_v41/test_encoding.py @@ -0,0 +1,199 @@ +# Copyright (c) PyPTO Contributors. +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of +# CANN Open Software License Agreement Version 2.0 (the "License"). +# Please refer to the License for details. You may not use this file except in compliance with the License. +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. +# See LICENSE in the root of the software repository for the full text of the License. +# ----------------------------------------------------------------------------------------------------------- +"""Pinned official V4.1 text-prompt goldens and serving adapter input contracts.""" + +from __future__ import annotations + +import copy +import importlib.util +import json +from pathlib import Path +import sys +from types import ModuleType + +import pytest + + +ROOT = Path(__file__).resolve().parents[4] +MODEL_ROOT = ROOT / "pypto_serving/model" +FIXTURE = json.loads((ROOT / "tests/fixtures/deepseek_v41/chat_encoding_golden.json").read_text()) + + +def load(name, path, monkeypatch=None): + spec = importlib.util.spec_from_file_location(name, path) + module = importlib.util.module_from_spec(spec) + if monkeypatch is not None: + monkeypatch.setitem(sys.modules, name, module) + spec.loader.exec_module(module) + return module + + +encoding = load("v41_text_encoding_test", MODEL_ROOT / "deepseek_v41/encoding.py") + + +@pytest.mark.parametrize("case", FIXTURE["cases"], ids=lambda case: case["name"]) +def test_pinned_official_text_prompt(case): + before = copy.deepcopy(case["messages"]) + actual = encoding.encode_messages(case["messages"], **case["options"]) + assert actual == case["expected"] + assert actual.encode("utf-8") == case["expected"].encode("utf-8") + assert case["messages"] == before + + +def test_golden_provenance_is_the_pinned_encoder(): + assert FIXTURE["revision"] == encoding.REFERENCE_REVISION + assert FIXTURE["source_sha256"] == encoding.REFERENCE_SHA256 + assert f"/blob/{encoding.REFERENCE_REVISION}/encoding/encoding.py" in FIXTURE["source"] + + +@pytest.mark.parametrize( + "messages,match", + [ + ([], "at least one"), + (["user"], "objects"), + ([{"role": "developer", "content": "rules"}], "role"), + ([{"role": "tool", "content": "result"}], "role"), + ([{"role": "user"}], "string"), + ([{"role": "user", "content": [{"type": "text", "text": "hi"}]}], "string"), + ([{"role": "user", "content": "hi", "tools": []}], "fields"), + ([{"role": "assistant", "content": "hi", "tool_calls": []}], "fields"), + ([{"role": "system", "content": "hi", "response_format": {}}], "fields"), + ([{"role": "user", "content": "hi", "content_blocks": []}], "fields"), + ([{"role": "user", "content": "hi", "task": "action"}], "fields"), + ([{"role": "user", "content": "hi", "reasoning_content": "why"}], "assistant"), + ([{"role": "assistant", "content": "hi", "reasoning_content": []}], "assistant"), + ([{"role": "assistant", "content": "hi", "wo_eos": 1}], "bool"), + ([{"role": "user", "content": encoding.IMAGE_TOKEN}], "image"), + ([{"role": "user", "content": "describe cat.jpg"}], "image"), + ([{"role": "user", "content": "malformed "}], "image"), + ([{"role": "assistant", "content": "hi", "reasoning_content": encoding.IMAGE_TOKEN}], "image"), + ], +) +def test_unsupported_or_invalid_message_is_explicit(messages, match): + with pytest.raises(ValueError, match=match): + encoding.encode_messages(messages) + + +@pytest.mark.parametrize("effort", ["medium", "xhigh", "none", "50", 0, 101, True, 75.0, []]) +def test_invalid_reference_reasoning_effort(effort): + with pytest.raises(ValueError, match="reasoning_effort"): + encoding.encode_messages([{"role": "user", "content": "hi"}], reasoning_effort=effort) + + +@pytest.fixture +def adapter(monkeypatch): + # Load the actual shared adapter without executing the torch-dependent package initializer. + for name, path in ( + ("pypto_serving", ROOT / "pypto_serving"), + ("pypto_serving.model", MODEL_ROOT), + ("pypto_serving.model.deepseek_v41", MODEL_ROOT / "deepseek_v41"), + ): + package = ModuleType(name) + package.__path__ = [str(path)] + monkeypatch.setitem(sys.modules, name, package) + for suffix, path in ( + ("model_family", "model_family.py"), + ("tokenizer", "tokenizer.py"), + ("deepseek_v41.encoding", "deepseek_v41/encoding.py"), + ("deepseek_v41.tokenizer", "deepseek_v41/tokenizer.py"), + ): + module = load(f"pypto_serving.model.{suffix}", MODEL_ROOT / path, monkeypatch) + + class RecordingTokenizer: + def __init__(self): + self.calls = [] + + def encode(self, text, *, add_special_tokens): + self.calls.append((text, add_special_tokens)) + return [100, 200, 300] + + return module.DeepSeekV41TokenizerAdapter(RecordingTokenizer()) + + +@pytest.mark.parametrize( + "options,budget", + [ + ({"thinking": True}, 75), + ({"enable_thinking": True, "reasoning_effort": "low"}, 50), + ({"enable_thinking": True, "reasoning_effort": "high"}, 75), + ({"thinking_mode": "thinking", "reasoning_effort": "max"}, 100), + ({"thinking_mode": "thinking", "reasoning_effort": 1}, 1), + ({"thinking_mode": "thinking", "reasoning_effort": 100}, 100), + ], +) +def test_server_thinking_options(adapter, options, budget): + actual = adapter.apply_chat_template( + [{"role": "user", "content": "hi"}], tokenize=False, add_generation_prompt=True, **options + ) + assert actual == ( + encoding.BOS_TOKEN + encoding.SYSTEM_TOKEN + f"Reasoning Effort: {budget} " + "(range 1-100, the higher the value, the more thorough the reasoning)\n\n" + + encoding.USER_TOKEN + + "hi" + + encoding.ASSISTANT_TOKEN + + "" + ) + + +@pytest.mark.parametrize( + "options", [{}, {"enable_thinking": False}, {"thinking": True, "reasoning_effort": "none"}] +) +def test_chat_and_none_budget(adapter, options): + case = FIXTURE["cases"][0] + assert adapter.apply_chat_template(case["messages"], **options) == case["expected"] + + +def test_tokenize_preserves_single_official_bos(adapter): + case = FIXTURE["cases"][0] + assert adapter.apply_chat_template(case["messages"], tokenize=True) == [100, 200, 300] + assert adapter.tokenizer.calls == [(case["expected"], False)] + + +@pytest.mark.parametrize( + "options,match", + [ + ({"enable_thinking": 1}, "bool"), + ({"tokenize": 1}, "bool"), + ({"drop_thinking": 0}, "bool"), + ({"add_generation_prompt": False}, "add_generation_prompt=True"), + ({"thinking": True, "enable_thinking": False}, "agree"), + ({"thinking_mode": "chat", "thinking": True}, "conflicts"), + ({"thinking_mode": "invalid", "reasoning_effort": "none"}, "thinking_mode"), + ({"reasoning_effort": "medium"}, "reasoning_effort"), + ({"reasoning_effort": "xhigh"}, "reasoning_effort"), + ({"tools": []}, "options"), + ({"return_tensors": "pt"}, "options"), + ], +) +def test_adapter_rejects_unsupported_options(adapter, options, match): + with pytest.raises(ValueError, match=match): + adapter.apply_chat_template([{"role": "user", "content": "hi"}], **options) + + +@pytest.mark.parametrize( + "family,expected", + [ + ("deepseek_v41", "DeepSeekV41TokenizerAdapter"), + ("deepseek_v4", "DeepSeekV4TokenizerAdapter"), + ("qwen3", "TransformersTokenizerAdapter"), + ], +) +def test_shared_tokenizer_loader_dispatches_independent_adapter( + adapter, tmp_path, monkeypatch, family, expected +): + shared = sys.modules["pypto_serving.model.tokenizer"] + monkeypatch.setattr( + shared.TransformersTokenizerAdapter, + "from_tokenizer_file", + classmethod(lambda cls, model_dir: cls(object())), + ) + (tmp_path / "config.json").write_text(json.dumps({"model_type": family})) + (tmp_path / "tokenizer.json").write_text("{}") + result = shared.load_tokenizer(tmp_path) + assert type(result).__name__ == expected diff --git a/tests/unit/model/deepseek_v41/test_entry.py b/tests/unit/model/deepseek_v41/test_entry.py new file mode 100644 index 00000000..ebc506b6 --- /dev/null +++ b/tests/unit/model/deepseek_v41/test_entry.py @@ -0,0 +1,126 @@ +# Copyright (c) PyPTO Contributors. +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of +# CANN Open Software License Agreement Version 2.0 (the "License"). +# Please refer to the License for details. You may not use this file except in compliance with the License. +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. +# See LICENSE in the root of the software repository for the full text of the License. +# ----------------------------------------------------------------------------------------------------------- + +"""Metadata and tokenizer entry tests through the public serving interfaces.""" + +import copy +import json +from pathlib import Path + +import pytest + +from pypto_serving.model.deepseek_v41.config import V41TextConfig, load_text_config +from pypto_serving.model.model_family import detect_model_family +from pypto_serving.model.model_loader import ModelLoader +from pypto_serving.model.tokenizer import load_tokenizer + + +FIXTURE = Path(__file__).resolve().parents[4] / "tests/fixtures/deepseek_v41/config.json" + + +@pytest.fixture +def raw(): + return json.loads(FIXTURE.read_text(encoding="utf-8")) + + +@pytest.fixture +def model_dir(tmp_path, raw): + (tmp_path / "config.json").write_text(json.dumps(raw), encoding="utf-8") + return tmp_path + + +@pytest.mark.parametrize("metadata", [ + {"model_type": "deepseek_v41"}, + {"architectures": ["DeepseekV41ForCausalLM"]}, + {"model_type": "deepseek_v4", "architectures": ["DeepseekV41ForCausalLM"]}, +]) +def test_v41_detection_precedes_other_families(metadata): + assert detect_model_family(metadata) == "deepseek_v41" + + +@pytest.mark.parametrize("metadata,expected", [ + ({"model_type": "deepseek_v4"}, "deepseek_v4"), + ({"model_type": "qwen3"}, "qwen"), + ({"architectures": None}, "qwen"), +]) +def test_existing_family_detection(metadata, expected): + assert detect_model_family(metadata) == expected + + +def test_metadata_only_read_ignores_optional_modules(model_dir, raw): + before = copy.deepcopy(raw) + config = load_text_config(model_dir) + assert (config.hidden_size, config.num_hidden_layers, config.hc_mult) == (5120, 40, 4) + assert len(config.compress_ratios) == 40 + assert set(config.compress_ratios) == {0, 1, 2} + assert (config.bos_token_id, config.eos_token_id, config.pad_token_id) == (0, 1, 2) + raw["vision_config"] = {"unimplemented": True} + raw["text_config"]["engram_config"] = {"unimplemented": True} + assert V41TextConfig.from_dict(raw) == config + assert V41TextConfig.from_dict(before) == config + assert list(model_dir.iterdir()) == [model_dir / "config.json"] + + +@pytest.mark.parametrize("field,value", [ + ("hidden_size", True), ("num_hidden_layers", 0), ("hc_mult", -1), + ("rms_norm_eps", float("nan")), ("rms_norm_eps", 0), + ("compress_ratios", [0]), ("compress_ratios", [3] * 40), + ("compress_ratios", [True] * 40), ("model_type", "deepseek_v4"), +]) +def test_invalid_text_config(raw, field, value): + raw["text_config"][field] = value + with pytest.raises(ValueError): + V41TextConfig.from_dict(raw) + + +@pytest.mark.parametrize("value", [-1, True, 129280]) +def test_invalid_special_id(raw, value): + raw["eos_token_id"] = value + with pytest.raises(ValueError, match="eos_token_id"): + V41TextConfig.from_dict(raw) + + +@pytest.mark.parametrize("model_format", [None, "hf", "deepseek_v4"]) +def test_loading_never_falls_through_to_qwen(model_dir, model_format): + with pytest.raises(NotImplementedError, match="weight loading and execution"): + ModelLoader().load("v41", str(model_dir), model_format=model_format) + + +def test_cli_rejects_execution_before_device_setup(model_dir): + from pypto_serving.cli.main import build_parser, build_serving_engine_config + + args = build_parser().parse_args(["--model", str(model_dir)]) + with pytest.raises(NotImplementedError, match="serving execution"): + build_serving_engine_config(args) + + +def test_local_tokenizer_round_trip_and_chat(model_dir): + from tokenizers import Tokenizer, models, pre_tokenizers + from pypto_serving.model.deepseek_v41.encoding import ( + BOS_TOKEN, EOS_TOKEN, USER_TOKEN, ASSISTANT_TOKEN, + ) + + special = [BOS_TOKEN, EOS_TOKEN, "", USER_TOKEN, ASSISTANT_TOKEN, ""] + vocab = {token: i for i, token in enumerate(special + ["[UNK]", "hello", "world"])} + tokenizer = Tokenizer(models.WordLevel(vocab, unk_token="[UNK]")) + tokenizer.pre_tokenizer = pre_tokenizers.Whitespace() + tokenizer.add_special_tokens(special) + tokenizer.save(str(model_dir / "tokenizer.json")) + (model_dir / "tokenizer_config.json").write_text(json.dumps({ + "bos_token": BOS_TOKEN, "eos_token": EOS_TOKEN, "pad_token": "", + }), encoding="utf-8") + adapter = load_tokenizer(model_dir) + assert adapter.encode("hello world") == [vocab["hello"], vocab["world"]] + assert adapter.decode(adapter.encode("hello world")) == "hello world" + ids = adapter.apply_chat_template([{"role": "user", "content": "hello"}], tokenize=True) + assert ids == [vocab[BOS_TOKEN], vocab[USER_TOKEN], vocab["hello"], + vocab[ASSISTANT_TOKEN], vocab[""]] + assert adapter.bos_token_id == 0 and adapter.eos_token_id == 1 + assert adapter.pad_token_id == 2 + assert adapter.decode(ids) == "hello" From 65b74728391ae3bbcf148247d427fb656c2faad5 Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Tue, 22 Sep 2026 18:41:44 +0800 Subject: [PATCH 02/78] feat(v41): add selective checkpoint loading and native weight packing --- docs/developer-guide/deepseek-v41-entry.md | 75 ++++- pypto_serving/model/common/weights/store.py | 31 +++ .../model/deepseek_v41/weight_loader.py | 223 +++++++++++++++ .../model/deepseek_v41/weight_packing.py | 74 +++++ .../model/deepseek_v41/weight_spec.py | 170 ++++++++++++ pypto_serving/model/model_loader.py | 4 +- tests/unit/model/deepseek_v41/test_entry.py | 2 +- .../model/deepseek_v41/test_weight_loader.py | 261 ++++++++++++++++++ 8 files changed, 827 insertions(+), 13 deletions(-) create mode 100644 pypto_serving/model/deepseek_v41/weight_loader.py create mode 100644 pypto_serving/model/deepseek_v41/weight_packing.py create mode 100644 pypto_serving/model/deepseek_v41/weight_spec.py create mode 100644 tests/unit/model/deepseek_v41/test_weight_loader.py diff --git a/docs/developer-guide/deepseek-v41-entry.md b/docs/developer-guide/deepseek-v41-entry.md index 9b181520..b3ec48cb 100644 --- a/docs/developer-guide/deepseek-v41-entry.md +++ b/docs/developer-guide/deepseek-v41-entry.md @@ -1,10 +1,10 @@ # DeepSeek V4.1 text entry -This is the first stage of V4.1 integration into the upstream serving architecture: -model identification, metadata validation and local text tokenization. It does -not load checkpoint tensors, allocate devices or enable generation. Both the -model loader and the serving CLI reject V4.1 execution explicitly until its -weight loader and executor are integrated, instead of treating it as Qwen. +V4.1 integration currently provides model identification, metadata validation, +local text tokenization and selective CPU weight loading. Configuration and +tokenization do not read checkpoint tensors. None of these APIs allocate NPU +resources or enable generation. Both the model loader and serving CLI reject +V4.1 execution until its executor is integrated, instead of treating it as Qwen. ```python from pypto_serving.model.deepseek_v41.config import load_text_config @@ -34,12 +34,12 @@ and decoding but is not a full-checkpoint tokenizer validation. ## Staged integration -1. Model identification, text config and tokenizer (this stage). -2. Selective checkpoint loading, weight formats and shard contracts. +1. Model identification, text config and tokenizer (implemented). +2. Selective checkpoint loading, weight formats and shard contracts (implemented on CPU). 3. Embedding and initial mHC residual/pre-mix state. -4. Attention mHC and RMSNorm. -5. SWA, then C2A and C1A attention with cache publication and prefill/decode state. -6. Attention mHC post, FFN mHC/norm, MoE and FFN mHC post. +4. Input preparation at the selected lib composite boundary. +5. SWA, C2A and C1A prefill composite entries and cache state. +6. Decode composite entries and prefill-to-decode transitions. 7. Cross-layer state and TP/EP composition, then the full backbone. 8. Final HC/norm, LM head and greedy decode loop. 9. Scheduler/HTTP lifecycle, recovery, memory observations and M0 acceptance. @@ -54,6 +54,61 @@ reused selectively with tests; its backend and runtime are not prerequisites. Kernel stages should track current pypto-lib interfaces and record the revision used for validation. This metadata/tokenizer stage does not change the lib pin. +## Selective weight loading + +The weight path follows V4's declarative source specs and shared +`LazySafetensorsStore`. A constructor reads only config/index JSON and checks +required text names. Requested tensor headers are validated before payload +slicing. It never loads an entire shard or deferred Engram/vision/draft weights. + +```python +from pypto_serving.model.deepseek_v41.weight_loader import V41WeightLoader + +loader = V41WeightLoader(model_dir, tp_size=4, tp_rank=0, ep_size=8, ep_rank=0) +projection = loader.load("layers.0.attn.wq_b.weight") +expert = loader.load("layers.0.ffn.experts.0.w1.weight") +embedding_rows = loader.load_rows("embed.weight", 0, 32) +``` + +`load` returns a `WeightBundle` with owned CPU weight/scale tensors, layout, +source names, rank identities and the per-call buffer estimate. Callers retain +global checkpoint names and map them to the chosen composite's parameters in +the Executor stage. This API does not invent an executable whole-layer ABI. + +| Source | CPU result | +| --- | --- | +| FP8 block32 `[N,K]` + UE8M0 grid | Contiguous FP8 `[K,N]`, expanded scale codes in MX_B_NN order | +| Routed FP4 `[N,K/2]` bytes | UINT8 tiles `[K*N/256,128]`, with MX_B_NN E8M0 scales | +| `wo_a` FP8 | Dequantized grouped BF16 after selecting this rank's output groups | +| Dense HC/norm weights | Preserved dense layout/dtype | +| Gate and ratio-2 compressor | Required FP32 promotion; compressor/index matrices transposed where required | +| Embedding/head | BF16 vocabulary shard or a bounded local-row range | + +Packing matches lib `4c3eab2` host layout helpers; these pure CPU operations do +not import PyPTO or execute small model operators. Runtime integration will +call composite lib entries, following V4 Executor/Runner structure. Verify the +layouts again when lib changes. Current FP4 tiles require K/N multiples of 256; +native FP8 matrices require K divisible by 64 and N by 32. Unsupported geometry +is rejected. Torch/safetensors must support the checkpoint E8M0 dtype. + +TP slices grouped/head projections and aligns their scales. Shared experts and +router weights are replicated; EP selects whole routed experts, retaining +global expert IDs. TP and EP ranks are explicit; mapping physical ranks to +these coordinates belongs to the Executor. This loader does not certify a +multi-device execution path. + +The default 256 MiB budget covers a conservative estimate for one operation's +tensor buffers, including conversion scratch. It excludes mmap address space, +Python overhead, device storage and previously returned bundles. Large loads +are rejected before payload reads. Use `load_rows` for vocabulary tables and +load expert matrices individually; do not accumulate a full model without a +separate residency budget. `load_rows` offsets are relative to the local TP +vocabulary shard. + +Validation uses synthetic checkpoint tensors stored in actual safetensors files, +independent packing-address checks, and shared-store regression tests. These +checks are not real-checkpoint numerical inference or NPU acceptance. + ## Validation ```bash diff --git a/pypto_serving/model/common/weights/store.py b/pypto_serving/model/common/weights/store.py index 4c74358e..f5fa775b 100644 --- a/pypto_serving/model/common/weights/store.py +++ b/pypto_serving/model/common/weights/store.py @@ -31,6 +31,10 @@ def get_tensor(self, name: str) -> torch.Tensor: """Return one tensor by name.""" raise NotImplementedError + def get_slice(self, name: str): + """Return a lazy handle exposing shape/dtype and bounded slicing.""" + raise NotImplementedError + class SafeOpenFn(Protocol): """Callable shape for injectable safetensors openers.""" @@ -110,6 +114,33 @@ def load_tensor(self, name: str) -> torch.Tensor: """Load one tensor by name, leaving all unrelated shard tensors untouched.""" return self.load_many([name])[name] + def load_slice( + self, name: str, ranges: tuple[slice, ...], *, shape: tuple[int, ...], dtype: str, + ) -> torch.Tensor: + """Validate metadata before reading an owned contiguous slice. + + The caller supplies the expected physical storage shape and safetensors + dtype, and is responsible for its allocation budget. Existing whole-tensor + staging paths continue to use load_many. + """ + if len(ranges) != len(shape) or any( + not isinstance(part, slice) or part.step not in (None, 1) + or type(part.start) is not int or type(part.stop) is not int + or not 0 <= part.start < part.stop <= size + for part, size in zip(ranges, shape) + ): + raise ValueError(f"invalid tensor slice: {name}") + path = self.path_for(name) + if not path.exists(): + raise FileNotFoundError(self.missing_shard_error.format(path=path)) + with self._safe_open_fn(path, self.device) as reader: + source = reader.get_slice(name) + if tuple(source.get_shape()) != shape or source.get_dtype() != dtype: + raise ValueError(f"checkpoint shape/dtype mismatch: {name}; expected {shape}/{dtype}") + # Clone even contiguous slices: do not retain mmap storage or a + # full source-row stride after the reader is closed. + return source[ranges].clone(memory_format=torch.contiguous_format) + def load_many(self, names: Sequence[str]) -> dict[str, torch.Tensor]: """Load a set of named tensors grouped by shard file. diff --git a/pypto_serving/model/deepseek_v41/weight_loader.py b/pypto_serving/model/deepseek_v41/weight_loader.py new file mode 100644 index 00000000..6b76d0ca --- /dev/null +++ b/pypto_serving/model/deepseek_v41/weight_loader.py @@ -0,0 +1,223 @@ +# Copyright (c) PyPTO Contributors. +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of +# CANN Open Software License Agreement Version 2.0 (the "License"). +# Please refer to the License for details. You may not use this file except in compliance with the License. +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. +# See LICENSE in the root of the software repository for the full text of the License. +# ----------------------------------------------------------------------------------------------------------- + +"""Selective checkpoint reads for V4.1 composite weight ABIs. + +Follows V4's name/spec/store separation. A load returns one matrix and its +scales or one dense tensor, not a resident model or an executable backend. +TP shards attention projections; EP selects whole routed experts. The caller +owns device placement and the lifetime/budget of previously returned bundles. +""" + +from dataclasses import dataclass +import json +import math +from pathlib import Path +import re + +import torch + +from pypto_serving.model.common.weights.store import LazySafetensorsStore +from .config import V41TextConfig +from .weight_spec import backbone_weight_specs +from .weight_packing import dequantize_output_groups, fp8_input_major, pack_fp4_tiles, pack_mx_scale + + +_BYTES = {"BF16": 2, "F32": 4, "F8_E4M3": 1, "F8_E8M0": 1, "I8": 1} +_EXPERT = re.compile(r"^layers\.\d+\.ffn\.experts\.(\d+)\.") + + +def _unique(pairs): + result = {} + for key, value in pairs: + if key in result: + raise ValueError(f"duplicate checkpoint metadata key: {key}") + result[key] = value + return result + + +def _positive(value, name): + if type(value) is not int or value <= 0: + raise ValueError(f"{name} must be a positive integer") + return value + + +@dataclass(frozen=True) +class WeightBundle: + """Owned CPU tensors, with explicit layout and checkpoint provenance.""" + + weight: torch.Tensor + scale: torch.Tensor | None + layout: str + source_names: tuple[str, ...] + tp_rank: int + ep_rank: int + estimated_peak_bytes: int + + +class V41WeightLoader: + """Read only selected text weights; reject budget overflow before payload I/O. + + max_load_bytes bounds conservative tensor buffers for one operation, not mmap + address space, the complete process or accumulated caller-owned results. + """ + + def __init__(self, model_dir, *, tp_size=1, tp_rank=0, ep_size=1, ep_rank=0, + max_load_bytes=256 << 20, safe_open_fn=None): + self.model_dir = Path(model_dir).resolve() + raw = json.loads((self.model_dir / "config.json").read_text(encoding="utf-8"), object_pairs_hook=_unique) + self.config = V41TextConfig.from_dict(raw) + self.specs = backbone_weight_specs(raw) + self.text = raw["text_config"] + self.tp_size = _positive(tp_size, "tp_size") + self.ep_size = _positive(ep_size, "ep_size") + for rank, size, name in ((tp_rank, tp_size, "tp_rank"), (ep_rank, ep_size, "ep_rank")): + if type(rank) is not int or not 0 <= rank < size: + raise ValueError(f"invalid {name}") + self.tp_rank, self.ep_rank = tp_rank, ep_rank + self.max_load_bytes = _positive(max_load_bytes, "max_load_bytes") + if self.text["n_routed_experts"] % ep_size: + raise ValueError("routed expert count must divide EP size") + for name in ("num_attention_heads", "o_groups", "index_n_heads", "vocab_size"): + if self.text[name] % tp_size: + raise ValueError(f"{name} must divide TP size") + index = json.loads((self.model_dir / "model.safetensors.index.json").read_text(encoding="utf-8"), + object_pairs_hook=_unique) + weight_map = index.get("weight_map") if isinstance(index, dict) else None + if not isinstance(weight_map, dict) or not weight_map: + raise ValueError("checkpoint index requires a nonempty weight_map") + for name, filename in weight_map.items(): + if not isinstance(name, str) or not isinstance(filename, str): + raise ValueError("weight_map must map names to shard filenames") + path = (self.model_dir / filename).resolve() + if Path(filename).is_absolute() or not path.is_relative_to(self.model_dir): + raise ValueError("checkpoint shard path must stay inside model directory") + # Deferred-module entries are not required, validated or opened. Only + # declared text weights are reachable through this loader. + self.store = LazySafetensorsStore(model_dir=self.model_dir, weight_map=weight_map, + safe_open_fn=safe_open_fn) + self.store.require(self.specs) + + def names(self, layer_id=None): + """Selected rank's names; None lists global weights, excluding deferred modules.""" + if layer_id is not None and (type(layer_id) is not int or not 0 <= layer_id < self.config.num_hidden_layers): + raise ValueError("invalid backbone layer_id") + prefix = None if layer_id is None else f"layers.{layer_id}." + return tuple(name for name in self.specs + if (not name.startswith("layers.") if prefix is None else name.startswith(prefix)) + and self._owns(name)) + + def _owns(self, name): + match = _EXPERT.match(name) + if not match: + return True + count = self.text["n_routed_experts"] // self.ep_size + return self.ep_rank * count <= int(match[1]) < (self.ep_rank + 1) * count + + def _ranges(self, name): + if name not in self.specs: + raise KeyError(f"not a supported text-backbone weight: {name}") + if not self._owns(name): + raise ValueError(f"expert is not owned by this EP rank: {name}") + spec = self.specs[name] + ranges = [slice(0, size) for size in spec.shape] + for axis in (0, 1): + if f"tp_shard_axis{axis}" in spec.conversion: + if spec.shape[axis] % self.tp_size: + raise ValueError(f"unaligned TP shard: {name}") + size = spec.shape[axis] // self.tp_size + ranges[axis] = slice(self.tp_rank * size, (self.tp_rank + 1) * size) + return tuple(ranges) + + def _estimate(self, requests): + # Include selected source copies, output/reordering, per-block conversion + # and byte scratch. No full-model or full FP4 float expansion is performed. + total = sum(math.prod(p.stop-p.start for p in ranges) * _BYTES[self.specs[name].dtype] + for name, ranges in requests) + estimate = total * 12 + (1 << 20) + if estimate > self.max_load_bytes: + raise ValueError(f"weight load requires estimated {estimate} bytes; budget={self.max_load_bytes}") + return estimate + + def _read(self, name, ranges): + spec = self.specs[name] + value = self.store.load_slice(name, ranges, shape=spec.shape, dtype=spec.dtype) + if spec.dtype == "F8_E8M0" and bool((value.view(torch.uint8) == 255).any()): + raise ValueError(f"non-finite E8M0 scale: {name}") + return value + + def load_rows(self, name, start, stop): + """Read local TP vocabulary rows without loading the complete table. + + start/stop are relative to this rank's embedding/head shard. Selection + of arbitrary token IDs and device lookup belong to the input stage. + """ + if name not in ("embed.weight", "head.weight"): + raise ValueError("load_rows supports embedding and LM-head weights only") + ranges = self._ranges(name) + shard = ranges[0] + if type(start) is not int or type(stop) is not int or not 0 <= start < stop <= shard.stop-shard.start: + raise ValueError("invalid local vocabulary row range") + ranges = (slice(shard.start+start, shard.start+stop), ranges[1]) + estimate = self._estimate([(name, ranges)]) + return WeightBundle(self._read(name, ranges), None, "vocabulary_rows_bf16", (name,), + self.tp_rank, self.ep_rank, estimate) + + def load(self, name): + """Load one dense weight or one projection with its required scales. + + Pass the checkpoint .weight name for projections; scales cannot be + loaded independently because TP slicing and packing must stay aligned. + """ + ranges = self._ranges(name) + spec = self.specs[name] + if spec.dtype == "F8_E8M0": + raise ValueError("load the projection weight to obtain aligned scales") + quantized = spec.dtype in ("F8_E4M3", "I8") + requests = [(name, ranges)] + scale_name = name.removesuffix(".weight") + ".scale" + if quantized: + n_part, k_part = ranges + n, k = n_part.stop-n_part.start, k_part.stop-k_part.start + if spec.dtype == "I8" and (n % 256 or (2*k) % 256): + raise ValueError("routed FP4 weights require complete 256x256 tiles") + if spec.dtype == "F8_E4M3" and "dequantize_fp8" not in spec.conversion and k % 64: + raise ValueError("FP8 native projection requires K divisible by 64") + divisor = 16 if spec.dtype == "I8" else 32 + if any(v % 32 for v in (n_part.start, n_part.stop)) and spec.dtype == "F8_E4M3": + raise ValueError("FP8 TP shard must align to output block32") + if k_part.start % divisor or k_part.stop % divisor: + raise ValueError("quantized TP shard must align to input block32") + scale_ranges = (n_part if spec.dtype == "I8" else slice(n_part.start//32, n_part.stop//32), + slice(k_part.start//divisor, k_part.stop//divisor)) + requests.append((scale_name, scale_ranges)) + estimate = self._estimate(requests) + weight = self._read(name, ranges) + scale = self._read(*requests[1]) if quantized else None + if spec.dtype == "I8": + weight = pack_fp4_tiles(weight) + scale = pack_mx_scale(scale.view(torch.uint8).T.contiguous()).view(torch.float8_e8m0fnu) + layout = "routed_fp4_tiles_256;scale_mx_b_nn" + elif spec.dtype == "F8_E4M3" and "dequantize_fp8" in spec.conversion: + weight = dequantize_output_groups(weight, scale, self.text["o_groups"] // self.tp_size) + scale, layout = None, "grouped_bf16" + elif spec.dtype == "F8_E4M3": + weight, scale = fp8_input_major(weight, scale) + layout = "input_major_fp8;scale_mx_b_nn" + else: + if "cast_" in spec.conversion: + weight = weight.float() + if name.endswith((".compressor.wkv.weight", ".compressor.wgate.weight", + ".indexer.wk.weight", ".indexer.weights_proj.weight")): + weight = weight.T.contiguous() + layout = "input_major_dense" + else: + layout = "checkpoint_dense" + return WeightBundle(weight, scale, layout, tuple(item[0] for item in requests), + self.tp_rank, self.ep_rank, estimate) diff --git a/pypto_serving/model/deepseek_v41/weight_packing.py b/pypto_serving/model/deepseek_v41/weight_packing.py new file mode 100644 index 00000000..ab87f3b7 --- /dev/null +++ b/pypto_serving/model/deepseek_v41/weight_packing.py @@ -0,0 +1,74 @@ +# Copyright (c) PyPTO Contributors. +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of +# CANN Open Software License Agreement Version 2.0 (the "License"). +# Please refer to the License for details. You may not use this file except in compliance with the License. +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. +# See LICENSE in the root of the software repository for the full text of the License. +# ----------------------------------------------------------------------------------------------------------- + +"""CPU-only layouts consumed by V4.1 lib composite weights. + +Matches pypto-lib 4c3eab2 quantization.pack_mx_b_scale and +pack_mxfp4_weight_tiles, with bounded FP4 tile scratch. No compiler imports. +""" + +import torch + + +def pack_mx_scale(codes): + """Logical [K/32,N] E8M0 bytes to MX_B_NN physical order.""" + groups, width = codes.shape + if groups % 2 or width % 16: + raise ValueError("MX_B_NN requires K divisible by 64 and N divisible by 16") + return codes.reshape(groups // 2, 2, width // 16, 16).permute(2, 0, 3, 1).contiguous().reshape( + groups, width, + ) + + +def pack_fp4_tiles(payload, tile=256): + """Reorder packed [N,K/2] to lib [K*N/256,128], without float expansion.""" + source = payload.contiguous().view(torch.uint8) + n, half_k = source.shape + k = half_k * 2 + if k % tile or n % tile: + raise ValueError("routed FP4 weights require complete 256x256 tiles") + output = torch.empty((n // tile, k // tile, tile, tile // 2), dtype=torch.uint8) + for nb in range(n // tile): + for kb in range(k // tile): + block = source[nb*tile:(nb+1)*tile, kb*(tile//2):(kb+1)*(tile//2)] + codes = torch.empty((tile, tile), dtype=torch.uint8) + codes[:, 0::2] = block & 15 + codes[:, 1::2] = block >> 4 + transposed = codes.T + output[nb, kb] = transposed[:, 0::2] | (transposed[:, 1::2] << 4) + return output.reshape(k * n // tile, tile // 2) + + +def fp8_input_major(payload, scale): + """Checkpoint FP8 block32 [N,K] to [K,N] plus E8M0 MX_B_NN backing.""" + n, k = payload.shape + if k % 64 or n % 32: + raise ValueError("FP8 native projection requires K divisible by 64 and N by 32") + codes = scale.contiguous().view(torch.uint8) + if tuple(codes.shape) != (n // 32, k // 32): + raise ValueError("FP8 block scales do not match the projection") + return payload.T.contiguous(), pack_mx_scale(codes.T.repeat_interleave(32, dim=1)).view( + torch.float8_e8m0fnu, + ) + + +def dequantize_output_groups(payload, scale, groups): + """The wo_a ABI uses grouped BF16, with checkpoint block32 scales applied.""" + n, k = payload.shape + if n % groups or n % 32 or k % 32: + raise ValueError("wo_a requires complete output groups and 32x32 quantization blocks") + codes = scale.contiguous().view(torch.uint8) + if tuple(codes.shape) != (n // 32, k // 32): + raise ValueError("wo_a scales do not match the projection") + # Convert one output block at a time instead of materializing a full FP32 matrix. + output = torch.empty((n, k), dtype=torch.bfloat16) + for row in range(0, n, 32): + factors = torch.exp2(codes[row // 32].float() - 127).repeat_interleave(32) + output[row:row+32] = (payload[row:row+32].float() * factors).to(torch.bfloat16) + return output.reshape(groups, n // groups, k) diff --git a/pypto_serving/model/deepseek_v41/weight_spec.py b/pypto_serving/model/deepseek_v41/weight_spec.py new file mode 100644 index 00000000..fc5a3fa2 --- /dev/null +++ b/pypto_serving/model/deepseek_v41/weight_spec.py @@ -0,0 +1,170 @@ +# Copyright (c) PyPTO Contributors. +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of +# CANN Open Software License Agreement Version 2.0 (the "License"). +# Please refer to the License for details. You may not use this file except in compliance with the License. +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. +# See LICENSE in the root of the software repository for the full text of the License. +# ----------------------------------------------------------------------------------------------------------- +"""Checkpoint storage contracts for the V4.1 text backbone, excluding deferred modules. + +Names and dtypes describe the published checkpoint, not PyTorch module defaults. +Conversion strings document the pinned reference's transformations and sharding; +they do not execute conversion, define an Ascend pack ABI, or upload weights. +Engram, vision, DSpark and the unused VL router bias are outside this scope. +""" + +from __future__ import annotations + +from collections.abc import Mapping +from dataclasses import dataclass +from typing import Any + + +@dataclass(frozen=True) +class TensorSpec: + """One source tensor and its required reference-layout transformation. + + ``shape`` is the physical checkpoint shape, so FP4 weights have K/2 bytes. + Multiple sources can target one runtime tensor: wo_a's scale is consumed by + dequantization into wo_a.weight, rather than retained as a runtime parameter. + """ + + shape: tuple[int, ...] + dtype: str + runtime_name: str + conversion: str + + +def backbone_weight_specs(config: Mapping[str, Any]) -> dict[str, TensorSpec]: + """Build required source specs from a Hugging Face V4.1 config object. + + This accepts the outer config containing ``text_config`` and + ``quantization_config``. It fails on unsupported quantization instead of + treating an arbitrary I8 tensor as E2M1. Layer ownership comes from source IDs. + The independent config parser should validate full model semantics first. + """ + text = config["text_config"] + quant = config["quantization_config"] + if not isinstance(text, Mapping) or not isinstance(quant, Mapping): + raise ValueError("text_config and quantization_config must be objects") + if ( + quant.get("quant_method") != "fp8" + or tuple(quant.get("weight_block_size", ())) != (32, 32) + or quant.get("scale_fmt") != "ue8m0" + or quant.get("expert_dtype") != "fp4" + ): + raise ValueError("V4.1 weight specs require FP8 32x32, UE8M0 and FP4 block-32 experts") + + def positive(name: str) -> int: + value = text[name] + if type(value) is not int or value <= 0: + raise ValueError(f"text_config.{name} must be a positive integer") + return value + + dim, inter = positive("hidden_size"), positive("moe_intermediate_size") + layers, heads = positive("num_hidden_layers"), positive("num_attention_heads") + experts, shared = positive("n_routed_experts"), positive("n_shared_experts") + head_dim, q_rank = positive("head_dim"), positive("q_lora_rank") + o_rank, groups = positive("o_lora_rank"), positive("o_groups") + hc, vocab = positive("hc_mult"), positive("vocab_size") + index_heads, index_dim = positive("index_n_heads"), positive("index_head_dim") + ratios = text["compress_ratios"] + kv_sources, index_sources = set(text["kv_source_layer_ids"]), set(text["index_source_layer_ids"]) + if len(ratios) < layers: + raise ValueError("compression metadata must cover backbone layers") + if any(type(i) is not int or not 0 <= i < layers for i in (*kv_sources, *index_sources)): + raise ValueError("weight source IDs must identify backbone layers") + if not kv_sources <= index_sources or any(ratios[i] < 1 for i in kv_sources | index_sources): + raise ValueError("KV sources must also own an indexer and use a positive compression ratio") + if heads % groups or shared != 1: + raise ValueError("V4.1 requires divisible attention groups and one shared expert") + + specs: dict[str, TensorSpec] = {} + + def add(name: str, shape: tuple[int, ...], dtype: str, conversion: str = "identity", target=None): + if any(type(size) is not int or size <= 0 for size in shape): + raise ValueError(f"{name}: invalid tensor shape {shape}") + if name in specs: + raise ValueError(f"duplicate source spec: {name}") + specs[name] = TensorSpec(shape, dtype, target or name, conversion) + + def dense(name: str, out_dim: int, in_dim: int, shard: str = "replicate", dequant: bool = False): + target = name + ".weight" + operation = "dequantize_fp8_32x32_ue8m0_to_bf16" if dequant else "preserve_fp8_32x32_ue8m0" + add(target, (out_dim, in_dim), "F8_E4M3", f"{operation};{shard}") + add( + name + ".scale", + ((out_dim + 31) // 32, (in_dim + 31) // 32), + "F8_E8M0", + f"consume_scale_for_dequantization;{shard}" if dequant else f"preserve_ue8m0;{shard}", + target=target if dequant else None, + ) + + def expert(name: str, routed: bool): + for proj, out_dim, in_dim in (("w1", inter, dim), ("w2", dim, inter), ("w3", inter, dim)): + prefix = f"{name}.{proj}" + if routed: + if in_dim % 32: + raise ValueError(f"{prefix}: FP4 input dimension must be divisible by 32") + add( + prefix + ".weight", + (out_dim, in_dim // 2), + "I8", + "reinterpret_packed_e2m1_low_nibble_first;ep_select_expert", + ) + add( + prefix + ".scale", + (out_dim, in_dim // 32), + "F8_E8M0", + "preserve_ue8m0_per_row_block32;ep_select_expert", + ) + else: + dense(prefix, out_dim, in_dim) + + add("embed.weight", (vocab, dim), "BF16", "tp_shard_axis0") + add("head.weight", (vocab, dim), "BF16", "tp_shard_axis0") + add("norm.weight", (dim,), "BF16") + for layer in range(layers): + base = f"layers.{layer}" + attn = base + ".attn" + add(attn + ".attn_sink", (heads,), "F32", "tp_shard_axis0") + dense(attn + ".wq_a", q_rank, dim) + dense(attn + ".wq_b", heads * head_dim, q_rank, "tp_shard_axis0") + dense(attn + ".wkv", head_dim, dim) + dense( + attn + ".wo_a", + groups * o_rank, + heads * head_dim // groups, + "tp_shard_axis0;view_grouped_output_projection", + dequant=True, + ) + dense(attn + ".wo_b", dim, groups * o_rank, "tp_shard_axis1") + for name, size in (("q_norm", q_rank), ("kv_norm", head_dim)): + add(f"{attn}.{name}.weight", (size,), "BF16") + for sublayer in ("attn", "ffn"): + add(f"{base}.{sublayer}_norm.weight", (dim,), "BF16") + mix = (2 + hc) * hc + add(f"{base}.hc_{sublayer}_fn", (mix, hc * dim), "F32") + add(f"{base}.hc_{sublayer}_base", (mix,), "F32") + add(f"{base}.hc_{sublayer}_scale", (3,), "F32") + add(base + ".ffn.gate.weight", (experts, dim), "BF16", "cast_to_fp32_for_routing") + add(base + ".ffn.gate.bias", (experts,), "F32") + for expert_id in range(experts): + expert(f"{base}.ffn.experts.{expert_id}", routed=True) + expert(base + ".ffn.shared_experts", routed=False) + if layer in kv_sources: + prefix = attn + ".compressor" + promotion = "cast_bf16_to_fp32" if ratios[layer] > 1 else "identity" + add(prefix + ".wkv.weight", (head_dim, dim), "BF16", promotion) + add(prefix + ".norm.weight", (head_dim,), "BF16") + if ratios[layer] > 1: + add(prefix + ".wgate.weight", (head_dim, dim), "BF16", promotion) + if layer in index_sources: + prefix = attn + ".indexer" + dense(prefix + ".wq_b", index_heads * index_dim, q_rank, "tp_shard_axis0") + add(prefix + ".weights_proj.weight", (index_heads, dim), "BF16", "tp_shard_axis0") + if layer in kv_sources: + add(prefix + ".wk.weight", (index_dim, head_dim), "BF16") + add(prefix + ".k_norm.weight", (index_dim,), "BF16") + return specs diff --git a/pypto_serving/model/model_loader.py b/pypto_serving/model/model_loader.py index f344f8d5..8e3be25d 100644 --- a/pypto_serving/model/model_loader.py +++ b/pypto_serving/model/model_loader.py @@ -462,8 +462,8 @@ def load( load_text_config(model_dir) raise NotImplementedError( - "V4.1 configuration and tokenizer are supported; weight loading and execution " - "are not integrated yet. Use load_text_config() and load_tokenizer() for inspection." + "V4.1 serving execution is not integrated yet. Use load_text_config() and " + "load_tokenizer() for inspection, or V41WeightLoader for selective CPU weight loading." ) request = ModelLoadRequest( model_id=model_id, diff --git a/tests/unit/model/deepseek_v41/test_entry.py b/tests/unit/model/deepseek_v41/test_entry.py index ebc506b6..bee082ac 100644 --- a/tests/unit/model/deepseek_v41/test_entry.py +++ b/tests/unit/model/deepseek_v41/test_entry.py @@ -88,7 +88,7 @@ def test_invalid_special_id(raw, value): @pytest.mark.parametrize("model_format", [None, "hf", "deepseek_v4"]) def test_loading_never_falls_through_to_qwen(model_dir, model_format): - with pytest.raises(NotImplementedError, match="weight loading and execution"): + with pytest.raises(NotImplementedError, match="serving execution"): ModelLoader().load("v41", str(model_dir), model_format=model_format) diff --git a/tests/unit/model/deepseek_v41/test_weight_loader.py b/tests/unit/model/deepseek_v41/test_weight_loader.py new file mode 100644 index 00000000..1f555e3b --- /dev/null +++ b/tests/unit/model/deepseek_v41/test_weight_loader.py @@ -0,0 +1,261 @@ +# Copyright (c) PyPTO Contributors. +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of +# CANN Open Software License Agreement Version 2.0 (the "License"). +# Please refer to the License for details. You may not use this file except in compliance with the License. +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. +# See LICENSE in the root of the software repository for the full text of the License. +# ----------------------------------------------------------------------------------------------------------- + +"""Real safetensors reads, ownership and independent byte-layout checks.""" + +import json +from pathlib import Path +from contextlib import contextmanager + +import pytest +import torch +from safetensors.torch import save_file +from safetensors import safe_open + +from pypto_serving.model.deepseek_v41.weight_spec import backbone_weight_specs +from pypto_serving.model.deepseek_v41.weight_loader import V41WeightLoader +from pypto_serving.model.deepseek_v41.weight_packing import pack_fp4_tiles, pack_mx_scale + + +@pytest.fixture +def checkpoint(tmp_path): + raw = json.loads((Path(__file__).resolve().parents[4] / + "tests/fixtures/deepseek_v41/config.json").read_text(encoding="utf-8")) + raw["text_config"].update(hidden_size=256, vocab_size=256, num_hidden_layers=1, + num_attention_heads=8, head_dim=64, q_lora_rank=256, o_lora_rank=128, o_groups=4, + n_routed_experts=2, moe_intermediate_size=256, index_n_heads=8, index_head_dim=32, + kv_source_layer_ids=[0], index_source_layer_ids=[0], compress_ratios=[2]) + raw.update(bos_token_id=0, eos_token_id=1, pad_token_id=2) + specs = backbone_weight_specs(raw) + tensors = {} + types = {"BF16": torch.bfloat16, "F32": torch.float32, "I8": torch.int8, + "F8_E4M3": torch.float8_e4m3fn, "F8_E8M0": torch.float8_e8m0fnu} + for name, spec in specs.items(): + size = 1 + for dim in spec.shape: + size *= dim + if spec.dtype == "F8_E8M0": + value = (torch.arange(size) % 5 + 125).to(torch.uint8).view(types[spec.dtype]) + elif spec.dtype == "I8": + value = (torch.arange(size) % 256).to(torch.uint8).view(torch.int8) + else: + value = ((torch.arange(size) % 13 - 6).float() / 4).to(types[spec.dtype]) + tensors[name] = value.reshape(spec.shape) + save_file(tensors, str(tmp_path / "text.safetensors")) + index = {name: "text.safetensors" for name in tensors} + index.update({"layers.0.engram.embed.weight": "not-opened.safetensors", + "vision.weight": "not-opened.safetensors", "mtp.weight": "not-opened.safetensors"}) + (tmp_path / "config.json").write_text(json.dumps(raw), encoding="utf-8") + (tmp_path / "model.safetensors.index.json").write_text(json.dumps({"weight_map": index}), encoding="utf-8") + return tmp_path, raw, tensors + + +def logical_scales(packed): + codes = packed.view(torch.uint8) + groups, n = codes.shape + # Independent physical address decoding; do not use an inverse of the packer. + result = torch.empty_like(codes) + flat = codes.flatten() + for kg in range(groups): + for col in range(n): + index = (col//16)*(groups//2)*32 + (kg//2)*32 + (col%16)*2 + kg%2 + result[kg, col] = flat[index] + return result + + +def test_index_and_ownership_do_not_read_payload(checkpoint): + path, _, _ = checkpoint + @contextmanager + def forbidden(*args): + raise AssertionError("constructor must not open any shard") + yield + loader = V41WeightLoader(path, ep_size=2, ep_rank=1, safe_open_fn=forbidden) + assert all("engram" not in n and "experts.0." not in n for n in loader.names(0)) + assert any("experts.1." in n for n in loader.names(0)) + with pytest.raises(KeyError, match="supported text"): + loader.load("layers.0.engram.embed.weight") + with pytest.raises(ValueError, match="not owned"): + loader.load("layers.0.ffn.experts.0.w1.weight") + + +def test_dense_and_fp8_tp_slices(checkpoint): + path, _, tensors = checkpoint + chunks = [] + for rank in range(2): + loader = V41WeightLoader(path, tp_size=2, tp_rank=rank) + embedding = loader.load("embed.weight") + assert torch.equal(embedding.weight, tensors["embed.weight"][rank*128:(rank+1)*128]) + bundle = loader.load("layers.0.attn.wq_b.weight") + chunks.append(bundle.weight) + expected = tensors["layers.0.attn.wq_b.weight"][rank*256:(rank+1)*256].T + assert torch.equal(bundle.weight.view(torch.uint8), expected.contiguous().view(torch.uint8)) + raw_scale = tensors["layers.0.attn.wq_b.scale"].view(torch.uint8)[rank*8:(rank+1)*8] + assert torch.equal(logical_scales(bundle.scale), raw_scale.T.repeat_interleave(32, 1)) + assert torch.equal(torch.cat(chunks, dim=1).T.contiguous().view(torch.uint8), + tensors["layers.0.attn.wq_b.weight"].view(torch.uint8)) + + +def test_input_axis_shard_keeps_scales_aligned(checkpoint): + path, _, tensors = checkpoint + result = V41WeightLoader(path, tp_size=2, tp_rank=1).load("layers.0.attn.wo_b.weight") + expected = tensors["layers.0.attn.wo_b.weight"][:, 256:] + assert torch.equal(result.weight.view(torch.uint8), expected.T.contiguous().view(torch.uint8)) + scales = tensors["layers.0.attn.wo_b.scale"].view(torch.uint8)[:, 8:] + assert torch.equal(logical_scales(result.scale), scales.T.repeat_interleave(32, 1)) + + +def test_wo_a_group_dequantization(checkpoint): + path, _, tensors = checkpoint + result = V41WeightLoader(path, tp_size=2, tp_rank=1).load("layers.0.attn.wo_a.weight") + values = tensors["layers.0.attn.wo_a.weight"][256:].float() + scales = torch.exp2(tensors["layers.0.attn.wo_a.scale"].view(torch.uint8)[8:].float() - 127) + expected = (values * scales.repeat_interleave(32, 0).repeat_interleave(32, 1)).bfloat16() + assert result.weight.shape == (2, 128, 128) + assert torch.equal(result.weight.flatten(0, 1), expected) + assert result.scale is None + + +def test_dense_promotions_and_transpose(checkpoint): + path, _, tensors = checkpoint + loader = V41WeightLoader(path) + for name in ("layers.0.hc_attn_fn", "layers.0.attn_norm.weight"): + assert torch.equal(loader.load(name).weight, tensors[name]) + name = "layers.0.attn.compressor.wkv.weight" + assert torch.equal(loader.load(name).weight, tensors[name].float().T) + assert loader.load("layers.0.ffn.gate.weight").weight.dtype == torch.float32 + + +def test_fp4_payload_is_reordered_without_loss(checkpoint): + path, _, tensors = checkpoint + name = "layers.0.ffn.experts.1.w1.weight" + result = V41WeightLoader(path, ep_size=2, ep_rank=1).load(name) + packed = result.weight.reshape(256, 128) + raw = tensors[name].view(torch.uint8) + # Address definition: adjacent output channels share one byte, K rows remain ordered. + for k in (0, 1, 127, 128, 255): + for n in (0, 1, 126, 127, 254, 255): + expected = (int(raw[n, k//2]) >> (4*(k%2))) & 15 + actual = (int(packed[k, n//2]) >> (4*(n%2))) & 15 + assert actual == expected + assert result.weight.dtype == torch.uint8 + expected_scale = tensors[name.replace(".weight", ".scale")].view(torch.uint8).T + assert torch.equal(logical_scales(result.scale), expected_scale) + + +def test_fp4_multiple_tiles(): + n, k = 512, 768 + raw = (torch.arange(n*(k//2)) % 251).to(torch.uint8).reshape(n, k//2) + result = pack_fp4_tiles(raw).reshape(n//256, k//256, 256, 128) + for row in (0, 255, 256, 511): + for col in (0, 255, 256, 511, 512, 767): + expected = (int(raw[row, col//2]) >> (4*(col%2))) & 15 + byte = int(result[row//256, col//256, col%256, (row%256)//2]) + assert (byte >> (4*(row%2))) & 15 == expected + + +def test_budget_rejects_before_open(checkpoint): + path, _, _ = checkpoint + @contextmanager + def forbidden(*args): + raise AssertionError("budget must be checked before payload I/O") + yield + loader = V41WeightLoader(path, max_load_bytes=1, safe_open_fn=forbidden) + with pytest.raises(ValueError, match="budget"): + loader.load("layers.0.attn.wq_b.weight") + + +def test_slice_reader_never_calls_get_tensor(checkpoint): + path, _, tensors = checkpoint + reads = [] + @contextmanager + def opener(path, device): + with safe_open(str(path), framework="pt", device=device) as reader: + class Reader: + def get_tensor(self, name): + raise AssertionError("must use bounded slicing") + def get_slice(self, name): + source = reader.get_slice(name) + class Slice: + def get_shape(self): return source.get_shape() + def get_dtype(self): return source.get_dtype() + def __getitem__(self, ranges): + reads.append((name, ranges)) + return source[ranges] + return Slice() + yield Reader() + result = V41WeightLoader(path, tp_size=2, tp_rank=1, safe_open_fn=opener).load("embed.weight") + assert reads == [("embed.weight", (slice(128, 256), slice(0, 256)))] + result.weight.zero_() + assert tensors["embed.weight"].count_nonzero() > 0 + + +@pytest.mark.parametrize("change", ["shape", "dtype"]) +def test_header_mismatch(checkpoint, change): + path, _, _ = checkpoint + value = torch.zeros(12) if change == "shape" else torch.zeros(256, 256, dtype=torch.float32) + save_file({"embed.weight": value}, str(path / "text.safetensors")) + with pytest.raises(ValueError, match="shape/dtype"): + V41WeightLoader(path).load("embed.weight") + + +@pytest.mark.parametrize("options", [{"tp_size": 3}, {"ep_size": 3}, {"tp_rank": -1}, + {"ep_rank": True}, {"max_load_bytes": 0}]) +def test_bad_topology(checkpoint, options): + with pytest.raises(ValueError): + V41WeightLoader(checkpoint[0], **options) + + +def test_missing_index_weight(checkpoint): + path, _, _ = checkpoint + index_path = path / "model.safetensors.index.json" + index = json.loads(index_path.read_text()) + del index["weight_map"]["norm.weight"] + index_path.write_text(json.dumps(index)) + with pytest.raises(KeyError, match="norm.weight"): + V41WeightLoader(path) + + +def test_scale_nan(checkpoint): + path, _, tensors = checkpoint + tensors["layers.0.attn.wq_a.scale"].view(torch.uint8)[0, 0] = 255 + save_file(tensors, str(path / "text.safetensors")) + with pytest.raises(ValueError, match="non-finite E8M0"): + V41WeightLoader(path).load("layers.0.attn.wq_a.weight") + + +def test_pack_rejects_invalid_geometry(): + with pytest.raises(ValueError): + pack_fp4_tiles(torch.zeros(32, 32, dtype=torch.uint8)) + with pytest.raises(ValueError): + pack_mx_scale(torch.zeros(3, 16, dtype=torch.uint8)) + + +@pytest.mark.parametrize("name", ["embed.weight", "head.weight"]) +def test_vocabulary_row_chunks(checkpoint, name): + path, _, tensors = checkpoint + loader = V41WeightLoader(path, tp_size=2, tp_rank=1, max_load_bytes=(1 << 20) + 40000) + with pytest.raises(ValueError, match="budget"): + loader.load(name) + chunk = loader.load_rows(name, 3, 7) + assert torch.equal(chunk.weight, tensors[name][131:135]) + assert chunk.weight.dtype == torch.bfloat16 + for start, stop in ((-1, 2), (0, 129), (3, 3), (True, 3)): + with pytest.raises(ValueError): + loader.load_rows(name, start, stop) + + +@pytest.mark.parametrize("filename", ["../outside.safetensors", "/absolute.safetensors"]) +def test_index_rejects_path_escape(checkpoint, filename): + path, _, _ = checkpoint + index_path = path / "model.safetensors.index.json" + index = json.loads(index_path.read_text()) + index["weight_map"]["norm.weight"] = filename + index_path.write_text(json.dumps(index)) + with pytest.raises(ValueError, match="inside model"): + V41WeightLoader(path) From a0df435df5429fece9a64dcea60789792ada1b47 Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Tue, 22 Sep 2026 19:22:40 +0800 Subject: [PATCH 03/78] feat(v41): prepare layer ownership and logical rank execution plans --- docs/developer-guide/deepseek-v41-entry.md | 53 ++++++- .../model/deepseek_v41/execution_plan.py | 150 ++++++++++++++++++ .../model/deepseek_v41/test_execution_plan.py | 95 +++++++++++ 3 files changed, 295 insertions(+), 3 deletions(-) create mode 100644 pypto_serving/model/deepseek_v41/execution_plan.py create mode 100644 tests/unit/model/deepseek_v41/test_execution_plan.py diff --git a/docs/developer-guide/deepseek-v41-entry.md b/docs/developer-guide/deepseek-v41-entry.md index b3ec48cb..57bb135a 100644 --- a/docs/developer-guide/deepseek-v41-entry.md +++ b/docs/developer-guide/deepseek-v41-entry.md @@ -36,8 +36,9 @@ and decoding but is not a full-checkpoint tokenizer validation. 1. Model identification, text config and tokenizer (implemented). 2. Selective checkpoint loading, weight formats and shard contracts (implemented on CPU). -3. Embedding and initial mHC residual/pre-mix state. -4. Input preparation at the selected lib composite boundary. +3. Executor/Runner preparation and composite contract verification (metadata planning implemented; + device registration blocked on composite integration). +4. Embedding and initial residual/pre-mix state at the selected composite boundary. 5. SWA, C2A and C1A prefill composite entries and cache state. 6. Decode composite entries and prefill-to-decode transitions. 7. Cross-layer state and TP/EP composition, then the full backbone. @@ -109,7 +110,53 @@ Validation uses synthetic checkpoint tensors stored in actual safetensors files, independent packing-address checks, and shared-store regression tests. These checks are not real-checkpoint numerical inference or NPU acceptance. -## Validation +## Executor/Runner preparation + +Stage numbers follow serving issue #240. `V41ExecutionPlan` implements the +CPU preparation portion of stage 3; it is not a registered executor, device +runner or substitute inference backend. + +```python +from pypto_serving.model.deepseek_v41.execution_plan import RankPlacement, V41ExecutionPlan + +plan = V41ExecutionPlan(model_dir, RankPlacement(rank=7)) +layer = plan.layer(24) # C1A Reindex: KV producer 20, index producer 24 +names = plan.weight_names(24) # local experts; no duplicate scale loads +bundle = plan.load_weight(24, "layers.24.hc_attn_scale") +``` + +Logical ranks form contiguous TP groups. With TP4/DP2/EP8, rank 7 is +TP rank 3, DP rank 1 and EP rank 7. These are logical coordinates, not +physical NPU IDs. The checkpoint loader still enforces dimension divisibility. +The plan preserves checkpoint names and the loader's per-operation budget; +it does not invent parameter bindings for an unsupported composite signature. + +Layer planning follows lib's `config.layer_config` ownership rules. SWA +windows are layer-local. Compressed layers select the latest preceding +producer of the same compression ratio. Full layers publish KV/index state; +Reindex layers use the KV producer's index-key cache and publish a new Top-K +selection; Reuse layers consume that selection without loading producer +weights. C1A candidates must address the same KV producer as their consumers. +These layer references are not scheduler page IDs or request-global state. + +The execution integration will follow upstream V4's `PyptoExecutor` and +`ModelRunner` lifecycle: the executor validates lib contracts and compiles +composites; the runner owns uploaded weights, buffers and dispatch completion; +the scheduler owns request/page reservations. Do not inherit the generic +dense K/V allocator for V4.1's window/compressed/index/pending-state pools. +Request and DP identity must scope every pool. Active token counts, padding, +physical page layouts and buffer reuse must come from the selected composite +contract, not from the V4 constants or a tensor capacity alone. + +At inspected lib revision `4c3eab2`, complete-layer coverage and the routed +weight contract are not ready for this registration. The current prefill +layer takes FP8 routed weights, whereas this loader preserves packed FP4; +the decode block factory raises `NotImplementedError`. Independent MoE FP4 +support does not establish full-layer compatibility. No device runner is +registered, generic fallback enabled, or FP4 model expanded to work around +these constraints. Stage 3 remains partially complete pending these contracts. + +## Test command ```bash python -m pytest tests/unit/model/deepseek_v41 tests/unit/model/test_tokenizer.py tests/unit/cli/test_parallel_options.py -q diff --git a/pypto_serving/model/deepseek_v41/execution_plan.py b/pypto_serving/model/deepseek_v41/execution_plan.py new file mode 100644 index 00000000..b29836a7 --- /dev/null +++ b/pypto_serving/model/deepseek_v41/execution_plan.py @@ -0,0 +1,150 @@ +# Copyright (c) PyPTO Contributors. +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of +# CANN Open Software License Agreement Version 2.0 (the "License"). +# Please refer to the License for details. You may not use this file except in compliance with the License. +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. +# See LICENSE in the root of the software repository for the full text of the License. +# ----------------------------------------------------------------------------------------------------------- +"""CPU preparation for the V4-style Executor/Runner boundary, not a device backend. + +Resolve logical ranks and cache producers before allocating weights or device +state. Kernel selection and cache allocation stay blocked until the composite +ABI is integrated; mode names here describe model semantics, not availability. +""" + +from dataclasses import dataclass +import json +from pathlib import Path + +from .config import V41TextConfig +from .weight_loader import V41WeightLoader + + +@dataclass(frozen=True) +class RankPlacement: + """Contiguous TP groups inside an EP world; rank is not a physical device ID.""" + + rank: int + tp_size: int = 4 + dp_size: int = 2 + ep_size: int = 8 + + def __post_init__(self): + if any(type(v) is not int or v <= 0 for v in (self.tp_size, self.dp_size, self.ep_size)): + raise ValueError("parallel sizes must be positive integers") + if self.tp_size * self.dp_size != self.ep_size: + raise ValueError("Attention TP * DP must equal the MoE EP world size") + if type(self.rank) is not int or not 0 <= self.rank < self.ep_size: + raise ValueError("logical rank must belong to the EP world") + + @property + def tp_rank(self): + return self.rank % self.tp_size + + @property + def dp_rank(self): + return self.rank // self.tp_size + + @property + def tp_group_start(self): + return self.dp_rank * self.tp_size + + +@dataclass(frozen=True) +class LayerPlan: + """Each layer owns its SWA window; compressed pools use explicit producers. + + REINDEX consumes its KV producer's index-key cache and publishes a new + Top-K selection. REUSE consumes the index producer's Top-K selection. + Producers are layer IDs within one request/DP partition, never page IDs. + """ + + layer_id: int + mode: str + kv_source: int | None + index_source: int | None + candidate_source: int | None + + +def plan_layers(raw): + """Follow lib config.layer_config source resolution without importing JIT modules.""" + config = V41TextConfig.from_dict(raw) + text = raw["text_config"] + ratios = config.compress_ratios + + def sources(name): + values = text.get(name) + if not isinstance(values, list) or any( + type(i) is not int or not 0 <= i < len(ratios) or ratios[i] == 0 for i in values + ): + raise ValueError(f"invalid {name}") + if len(set(values)) != len(values): + raise ValueError(f"duplicate {name}") + return set(values) + + kv, index = sources("kv_source_layer_ids"), sources("index_source_layer_ids") + if not kv <= index: + raise ValueError("KV producers must also produce index keys/selections") + candidate = text.get("candidate_source_layer_id") + if 1 in ratios and (type(candidate) is not int or candidate not in kv or ratios[candidate] != 1): + raise ValueError("C1A candidate source must be a C1A Full layer") + + def active(layer, ratio, owners): + matches = [i for i in owners if i <= layer and ratios[i] == ratio] + if not matches: + raise ValueError(f"layer {layer} has no preceding source with compression ratio {ratio}") + return max(matches) + + result = [] + for layer, ratio in enumerate(ratios): + if ratio == 0: + result.append(LayerPlan(layer, "swa", None, None, None)) + continue + kv_owner, index_owner = active(layer, ratio, kv), active(layer, ratio, index) + mode = "full" if layer in kv else "reindex" if layer in index else "reuse" + if ratio == 2 and mode == "reindex": + raise ValueError("C2A Reindex has no supported model contract") + if ratio == 1 and (candidate > layer or kv_owner != candidate): + raise ValueError("C1A candidates must refer to the same preceding compressed KV producer") + result.append(LayerPlan(layer, f"c{ratio}a_{mode}", kv_owner, index_owner, + candidate if ratio == 1 else None)) + return tuple(result) + + +class V41ExecutionPlan: + """Metadata and lazy weight preparation to be consumed by the future Executor. + + This deliberately does not register a ModelRunner or allocate generic K/V + pools: V4.1 needs request/DP-owned window, compressed, index and pending + state pools whose runtime contracts have not yet been integrated. + """ + + def __init__(self, model_dir, placement: RankPlacement, *, max_load_bytes=256 << 20): + self.placement = placement + raw = json.loads((Path(model_dir) / "config.json").read_text(encoding="utf-8")) + self.layers = plan_layers(raw) + self.weights = V41WeightLoader( + model_dir, tp_size=placement.tp_size, tp_rank=placement.tp_rank, + ep_size=placement.ep_size, ep_rank=placement.rank, max_load_bytes=max_load_bytes, + ) + + def layer(self, layer_id): + if type(layer_id) is not int or not 0 <= layer_id < len(self.layers): + raise ValueError("invalid backbone layer_id") + return self.layers[layer_id] + + def weight_names(self, layer_id): + """One projection name per bundle, owned experts only; scales load with payloads. + + Reuse layers consume shared cache state, not their producer's weights. + Keep checkpoint names until a specific composite binding is validated. + """ + self.layer(layer_id) + return tuple(name for name in self.weights.names(layer_id) if not name.endswith(".scale")) + + def load_weight(self, layer_id, name): + """Load one owned CPU bundle; the eventual runner owns upload/residency limits.""" + if name not in self.weight_names(layer_id): + raise ValueError("weight is not owned by the selected layer and rank") + return self.weights.load(name) diff --git a/tests/unit/model/deepseek_v41/test_execution_plan.py b/tests/unit/model/deepseek_v41/test_execution_plan.py new file mode 100644 index 00000000..8946961b --- /dev/null +++ b/tests/unit/model/deepseek_v41/test_execution_plan.py @@ -0,0 +1,95 @@ +# Copyright (c) PyPTO Contributors. +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of +# CANN Open Software License Agreement Version 2.0 (the "License"). +# Please refer to the License for details. You may not use this file except in compliance with the License. +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. +# See LICENSE in the root of the software repository for the full text of the License. +# ----------------------------------------------------------------------------------------------------------- +"""Preparation must preserve layer ownership without allocating a device backend.""" + +import json +from pathlib import Path + +import pytest +import torch +from safetensors.torch import save_file + +from pypto_serving.model.deepseek_v41.execution_plan import RankPlacement, V41ExecutionPlan, plan_layers +from pypto_serving.model.deepseek_v41.weight_spec import backbone_weight_specs + + +@pytest.fixture +def raw(): + return json.loads((Path(__file__).resolve().parents[4] / + "tests/fixtures/deepseek_v41/config.json").read_text(encoding="utf-8")) + + +def test_all_backbone_modes_and_producers(raw): + layers = plan_layers(raw) + assert len(layers) == 40 + assert [p.mode for p in layers] == ( + ["swa"] * 2 + (["c2a_full"] + ["c2a_reuse"] * 5) * 3 + + ["c1a_full"] + ["c1a_reuse"] * 3 + (["c1a_reindex"] + ["c1a_reuse"] * 3) * 4 + ) + assert [(p.kv_source, p.index_source) for p in layers[18:26]] == [ + (14, 14), (14, 14), (20, 20), (20, 20), (20, 20), (20, 20), (20, 24), (20, 24)] + assert all(p.candidate_source == 20 for p in layers[20:]) + assert all(p.candidate_source is None for p in layers[:20]) + + +@pytest.mark.parametrize("rank", range(8)) +def test_rank_coordinates(rank): + p = RankPlacement(rank) + assert (p.tp_rank, p.dp_rank, p.tp_group_start) == (rank % 4, rank // 4, rank // 4 * 4) + + +@pytest.mark.parametrize("kwargs", [{"rank": -1}, {"rank": 8}, {"rank": True}, + {"rank": 0, "tp_size": 0}, {"rank": 0, "ep_size": 4}]) +def test_invalid_topology(kwargs): + with pytest.raises(ValueError): + RankPlacement(**kwargs) + + +@pytest.mark.parametrize("field,value", [ + ("kv_source_layer_ids", [8, 14, 20]), + ("index_source_layer_ids", [2, 8, 14, 24, 28, 32, 36]), + ("kv_source_layer_ids", [2, 2, 8, 14, 20]), + ("kv_source_layer_ids", [0, 2, 8, 14, 20]), + ("index_source_layer_ids", [2, 3, 8, 14, 20, 24, 28, 32, 36]), + ("candidate_source_layer_id", 24), + ("candidate_source_layer_id", True), +]) +def test_bad_source_metadata_rejected(raw, field, value): + raw["text_config"][field] = value + with pytest.raises(ValueError): + plan_layers(raw) + + +def test_metadata_only_then_bounded_real_read(raw, tmp_path): + (tmp_path / "config.json").write_text(json.dumps(raw)) + names = backbone_weight_specs(raw) + (tmp_path / "model.safetensors.index.json").write_text(json.dumps({ + "weight_map": {name: "part.safetensors" for name in names}})) + # No shard exists. Construction and name inspection must not read payloads. + plan = V41ExecutionPlan(tmp_path, RankPlacement(7)) + assert plan.layer(24).kv_source == 20 + assert plan.layer(24).index_source == 24 + reuse = plan.weight_names(21) + assert not any("compressor" in n or "indexer" in n or n.endswith(".scale") for n in reuse) + assert "layers.21.ffn.experts.336.w1.weight" in reuse + assert "layers.21.ffn.experts.383.w3.weight" in reuse + assert "layers.21.ffn.experts.335.w1.weight" not in reuse + reindex = plan.weight_names(24) + assert "layers.24.attn.indexer.wq_b.weight" in reindex + assert not any("compressor" in n or "indexer.wk." in n for n in reindex) + with pytest.raises(ValueError, match="not owned"): + plan.load_weight(21, "layers.20.attn.compressor.wkv.weight") + with pytest.raises(ValueError): + plan.layer(-1) + name = "layers.21.hc_attn_scale" + expected = torch.tensor([1., 2., 3.]) + save_file({name: expected}, tmp_path / "part.safetensors") + bundle = plan.load_weight(21, name) + assert torch.equal(bundle.weight, expected) + assert (bundle.tp_rank, bundle.ep_rank) == (3, 7) From 39ddff01b22bcccba7def73536ba3eb0fc673c94 Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Wed, 23 Sep 2026 09:59:07 +0800 Subject: [PATCH 04/78] feat(v41): define composite executor and runner lifecycle contracts --- pypto_serving/model/deepseek_v41/composite.py | 88 +++++++++++++++++++ .../model/deepseek_v41/npu_executor.py | 65 ++++++++++++++ .../model/deepseek_v41/npu_runner.py | 72 +++++++++++++++ tests/unit/model/deepseek_v41/test_runner.py | 71 +++++++++++++++ 4 files changed, 296 insertions(+) create mode 100644 pypto_serving/model/deepseek_v41/composite.py create mode 100644 pypto_serving/model/deepseek_v41/npu_executor.py create mode 100644 pypto_serving/model/deepseek_v41/npu_runner.py create mode 100644 tests/unit/model/deepseek_v41/test_runner.py diff --git a/pypto_serving/model/deepseek_v41/composite.py b/pypto_serving/model/deepseek_v41/composite.py new file mode 100644 index 00000000..eec9db0d --- /dev/null +++ b/pypto_serving/model/deepseek_v41/composite.py @@ -0,0 +1,88 @@ +# Copyright (c) PyPTO Contributors. +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of +# CANN Open Software License Agreement Version 2.0 (the "License"). +# Please refer to the License for details. You may not use this file except in compliance with the License. +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. +# See LICENSE in the root of the software repository for the full text of the License. +# ----------------------------------------------------------------------------------------------------------- +"""Serving-owned composite boundary; no guessed lib signatures or CPU fallback. + +An integration supplies complete layer entries (Attention plus FFN), resource +allocation and the model output boundary. Callbacks may enqueue work: wait() +returns only after all ranks have stopped accessing the supplied buffers. +""" +from dataclasses import dataclass +from typing import Callable, Mapping + +from .execution_plan import LayerPlan, RankPlacement + + +class MissingCompositeInterface(NotImplementedError): + """The selected lib revision cannot execute the requested serving segment.""" + + +@dataclass(frozen=True) +class LayerState: + """Opaque device state carried between composites, never converted on host.""" + residual: object + pre_mix: object + layout: str = "tp_replicated" + + +@dataclass(frozen=True) +class CompositeBindings: + """Explicit adapter seam for a verified lib revision. + + Entries consume (LayerPlan, LayerState, step, resources, weights) and return + LayerState. initialize consumes (embeddings, step, resources); output consumes + (state, step, resources) and returns host logits in original request order. + The adapter owns device uploads and lib ABI binding. Serving never expands + packed FP4 or calls the sub-operators of a complete layer. + + No default implementation fabricates results. A supplied adapter must handle + all DP partitions collectively, including empty partitions, and retain input + objects until wait() completes. reset_request must clear every persistent + cache and compressor slot for that request before returning. + """ + revision: str + entries: Mapping[tuple[str, str], Callable] + initialize: Callable + output: Callable + allocate: Callable + prepare_weights: Callable + reset_request: Callable + wait: Callable + close: Callable + cache_groups: tuple = () + input_layout: str = "tp_replicated" + output_layout: str = "tp_replicated" + + def require(self, layers: tuple[LayerPlan, ...], placement: RankPlacement) -> None: + if not self.revision: + raise ValueError("composite bindings must identify the validated lib revision") + missing = sorted({f"{phase}/{layer.mode}" for phase in ("prefill", "decode") + for layer in layers if not callable(self.entries.get((phase, layer.mode)))}) + if missing: + raise MissingCompositeInterface("missing complete layer composites: " + ", ".join(missing)) + for name in ("initialize", "output", "allocate", "prepare_weights", "reset_request", "wait", "close"): + if not callable(getattr(self, name)): + raise MissingCompositeInterface(f"missing composite resource/output operation: {name}") + if self.input_layout not in ("tp_replicated", "tp_local_token") or self.output_layout != self.input_layout: + raise MissingCompositeInterface("layer entries must preserve their declared residual/pre_mix layout") + if (placement.tp_size, placement.dp_size, placement.ep_size) != (4, 2, 8): + raise ValueError("V4.1 serving currently targets TP4/DP2/EP8") + + +def load_composite_bindings() -> CompositeBindings: + """Fail before resource allocation until a complete adapter is implemented. + + This is an intentional integration placeholder, not a discovery heuristic: + importing a lib module or finding a function does not establish its ABI. + """ + raise MissingCompositeInterface( + "V4.1 serving execution requires verified lib composite bindings: " + "packed-FP4 full-layer prefill/decode for all modes, initial residual/pre_mix, " + "cache allocation/reset/completion and final HC/Norm/LM head. " + "Track pypto-lib #1205, #1275 and #1287; no Torch fallback is enabled." + ) diff --git a/pypto_serving/model/deepseek_v41/npu_executor.py b/pypto_serving/model/deepseek_v41/npu_executor.py new file mode 100644 index 00000000..6f69ba39 --- /dev/null +++ b/pypto_serving/model/deepseek_v41/npu_executor.py @@ -0,0 +1,65 @@ +# Copyright (c) PyPTO Contributors. +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of +# CANN Open Software License Agreement Version 2.0 (the "License"). +# Please refer to the License for details. You may not use this file except in compliance with the License. +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. +# See LICENSE in the root of the software repository for the full text of the License. +# ----------------------------------------------------------------------------------------------------------- +"""Executor routing for V4.1; devices are opened only after contract validation.""" +from pypto_serving.model.common.executor.executor import ModelExecutor + +from .composite import MissingCompositeInterface, load_composite_bindings +from .execution_plan import RankPlacement, V41ExecutionPlan +from .npu_runner import V41ModelRunner + + +class DeepSeekV41PyptoExecutor(ModelExecutor): + """Shared worker interface with a synchronous, collective V4.1 runner.""" + def __init__(self, kv_cache_manager=None, *, platform="a5", device_ids=tuple(range(8)), + pypto_build_dir="build_output", use_compile_cache=False, bindings=None): + super().__init__(kv_cache_manager) + if platform != "a5": + raise ValueError("V4.1 M0 currently requires the A5 platform") + self.device_ids = tuple(device_ids) + self.bindings = bindings + self.runners = {} + self._failed_runners = [] + self.platform = platform + self.pypto_build_dir = pypto_build_dir + self.use_compile_cache = use_compile_cache + + def register_model(self, model_id, record): + if model_id in self.runners: + raise ValueError("model already registered") + bindings = self.bindings or load_composite_bindings() + if not bindings.cache_groups: + raise MissingCompositeInterface("V4.1 requires explicit grouped cache layouts and capacities") + if record.runtime.kv_cache_groups != bindings.cache_groups: + raise ValueError("scheduler and composite cache group contracts differ") + plan = V41ExecutionPlan(record.runtime_model.extra["model_dir"], RankPlacement(0)) + runner = V41ModelRunner(plan, bindings, device_ids=self.device_ids, runtime=record.runtime) + try: + pages = runner.preflight() + except Exception: + if not runner.closed: + self._failed_runners.append(runner) + raise + self.runners[model_id] = runner + return pages + + def run_prefill(self, model, batch): + return self.runners[model.config.model_id].run_prefill(model, batch) + + def run_decode(self, model, batch): + return self.runners[model.config.model_id].run_decode(model, batch) + + def release_finished_requests(self, request_ids): + for runner in self.runners.values(): + runner.release_finished_requests(request_ids) + + def close(self): + for runner in (*self.runners.values(), *self._failed_runners): + runner.close() + self.runners.clear() + self._failed_runners.clear() diff --git a/pypto_serving/model/deepseek_v41/npu_runner.py b/pypto_serving/model/deepseek_v41/npu_runner.py new file mode 100644 index 00000000..183d3576 --- /dev/null +++ b/pypto_serving/model/deepseek_v41/npu_runner.py @@ -0,0 +1,72 @@ +# Copyright (c) PyPTO Contributors. +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of +# CANN Open Software License Agreement Version 2.0 (the "License"). +# Please refer to the License for details. You may not use this file except in compliance with the License. +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. +# See LICENSE in the root of the software repository for the full text of the License. +# ----------------------------------------------------------------------------------------------------------- +"""V4-style runner lifecycle for explicit V4.1 composite bindings.""" +from .composite import CompositeBindings, MissingCompositeInterface +from .execution_plan import V41ExecutionPlan + + +class V41ModelRunner: + """Own one collective session; missing entries fail before resource creation. + + This follows the shared runner lifecycle but deliberately does not inherit + its generic K/V allocator: V4.1 pools include index and compressor state. + """ + def __init__(self, plan: V41ExecutionPlan, bindings: CompositeBindings, *, device_ids, runtime): + self.plan, self.bindings = plan, bindings + self.device_ids = tuple(device_ids) + if len(self.device_ids) != plan.placement.ep_size or len(set(self.device_ids)) != len(self.device_ids): + raise ValueError("one distinct physical device is required for each logical EP rank") + if any(type(i) is not int or i < 0 for i in self.device_ids): + raise ValueError("device IDs must be nonnegative integers") + bindings.require(plan.layers, plan.placement) + self.runtime = runtime + self.resources = None + self.num_pages = None + self.closed = False + self.failed = False + + def preflight(self): + if self.failed: + raise RuntimeError("runner initialization failed; close the session before retrying") + if self.closed: + raise RuntimeError("runner is closed") + if self.resources is not None: + return self.num_pages + try: + resources, pages = self.bindings.allocate(self.plan, self.device_ids, self.runtime) + self.resources = resources + if resources is None or type(pages) is not int or pages <= 0: + raise ValueError("composite allocator must return resources and a positive page capacity") + self.bindings.wait(resources) + self.num_pages = pages + except Exception: + self.failed = True + self.close() + raise + return self.num_pages + + def close(self): + if self.closed: + return + if self.resources is not None: + # Do not free or reuse storage if completion itself fails. + self.bindings.wait(self.resources) + self.bindings.close(self.resources) + self.resources = None + self.closed = True + + def run_prefill(self, model, batch): + raise MissingCompositeInterface("prefill request/state binding is not connected yet") + + def run_decode(self, model, batch): + raise MissingCompositeInterface("decode request/state binding is not connected yet") + + def release_finished_requests(self, request_ids): + if request_ids: + raise MissingCompositeInterface("request cache lifecycle is not connected yet") diff --git a/tests/unit/model/deepseek_v41/test_runner.py b/tests/unit/model/deepseek_v41/test_runner.py new file mode 100644 index 00000000..883aa877 --- /dev/null +++ b/tests/unit/model/deepseek_v41/test_runner.py @@ -0,0 +1,71 @@ +# Copyright (c) PyPTO Contributors. +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of +# CANN Open Software License Agreement Version 2.0 (the "License"). +# Please refer to the License for details. You may not use this file except in compliance with the License. +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. +# See LICENSE in the root of the software repository for the full text of the License. +# ----------------------------------------------------------------------------------------------------------- +"""CPU contract tests; these do not emulate lib kernels or claim device coverage.""" +from types import SimpleNamespace +import pytest + +from pypto_serving.model.deepseek_v41.composite import CompositeBindings, MissingCompositeInterface +from pypto_serving.model.deepseek_v41.execution_plan import LayerPlan, RankPlacement +from pypto_serving.model.deepseek_v41.npu_runner import V41ModelRunner + + +def bindings(events, **overrides): + def allocate(*args): + events.append("allocate") + return object(), 12 + values = dict(revision="test-double-only", entries={(p, "swa"): lambda *a: None + for p in ("prefill", "decode")}, initialize=lambda *a: None, output=lambda *a: None, + allocate=allocate, prepare_weights=lambda *a: None, reset_request=lambda *a: None, + wait=lambda *a: events.append("wait"), close=lambda *a: events.append("close")) + return CompositeBindings(**(values | overrides)) + + +def plan(): + return SimpleNamespace(placement=RankPlacement(0), layers=(LayerPlan(0, "swa", None, None, None),)) + + +def test_missing_entry_before_allocation(): + events = [] + with pytest.raises(MissingCompositeInterface, match="decode/swa"): + V41ModelRunner(plan(), bindings(events, entries={("prefill", "swa"): lambda *a: None}), + device_ids=range(8), runtime=None) + assert events == [] + + +def test_lifecycle_waits_before_free_and_is_idempotent(): + events = [] + runner = V41ModelRunner(plan(), bindings(events), device_ids=range(8), runtime=None) + assert runner.preflight() == runner.preflight() == 12 + runner.close() + runner.close() + assert events == ["allocate", "wait", "wait", "close"] + with pytest.raises(RuntimeError, match="closed"): + runner.preflight() + + +def test_bad_allocator_result_is_closed(): + events = [] + runner = V41ModelRunner(plan(), bindings(events, allocate=lambda *a: (object(), 0)), + device_ids=range(8), runtime=None) + with pytest.raises(ValueError, match="capacity"): + runner.preflight() + assert events == ["wait", "close"] + + +def test_wait_failure_never_frees_live_buffers(): + events = [] + def wait(*args): + raise RuntimeError("device completion failed") + runner = V41ModelRunner(plan(), bindings(events, wait=wait), device_ids=range(8), runtime=None) + with pytest.raises(RuntimeError, match="completion"): + runner.preflight() + assert events == ["allocate"] + assert runner.resources is not None + with pytest.raises(RuntimeError, match="initialization failed"): + runner.preflight() From 453461b68591d9d4a833290692fbeab652975157 Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Wed, 23 Sep 2026 09:59:33 +0800 Subject: [PATCH 05/78] feat(v41): prepare bounded embeddings from TP vocabulary shards --- .../model/deepseek_v41/input_preparation.py | 98 +++++++++++++ .../deepseek_v41/test_input_preparation.py | 132 ++++++++++++++++++ 2 files changed, 230 insertions(+) create mode 100644 pypto_serving/model/deepseek_v41/input_preparation.py create mode 100644 tests/unit/model/deepseek_v41/test_input_preparation.py diff --git a/pypto_serving/model/deepseek_v41/input_preparation.py b/pypto_serving/model/deepseek_v41/input_preparation.py new file mode 100644 index 00000000..4af602c1 --- /dev/null +++ b/pypto_serving/model/deepseek_v41/input_preparation.py @@ -0,0 +1,98 @@ +# Copyright (c) PyPTO Contributors. +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of +# CANN Open Software License Agreement Version 2.0 (the "License"). +# Please refer to the License for details. You may not use this file except in compliance with the License. +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. +# See LICENSE in the root of the software repository for the full text of the License. +# ----------------------------------------------------------------------------------------------------------- +"""Bounded CPU embedding preparation using the checkpoint's vocabulary shards. + +Like V4's executor lookup, this is explicit host input preparation. It is not a +device embedding implementation or a fallback for a missing lib composite. +Residual expansion and the initial delayed pre_mix belong to the selected lib +initialization contract; an Attention golden fixture does not establish them. +""" + +from collections.abc import Sequence + +import torch + +from .weight_loader import V41WeightLoader + + +def lookup_token_embeddings( + loaders: Sequence[V41WeightLoader], + token_ids: torch.Tensor, + *, + max_prepare_bytes: int = 256 << 20, + max_rows_per_read: int = 64, +) -> torch.Tensor: + """Return owned CPU BF16 rows with shape ``(*token_ids.shape, hidden_size)``. + + Supply one complete set of TP vocabulary shards from the same checkpoint, + in any order. DP groups share this checkpoint table; token order and all + leading dimensions are preserved, without duplicating rows for TP ranks. + IDs must be CPU int32/int64 with at least one dimension. Empty inputs are + allowed, but padding sentinels such as -1 are not vocabulary indices. + + Repeated IDs are read once. Adjacent unique rows in the same TP shard are + read together, with at most max_rows_per_read rows per load_rows call. + max_prepare_bytes bounds conservative preparation tensor storage, including + the result, unique rows, indexing scratch and a read buffer; the loader's + separate max_load_bytes still bounds its own read/conversion operation. + Neither budget represents total process memory or previously returned data. + """ + for value, name in ((max_prepare_bytes, "max_prepare_bytes"), + (max_rows_per_read, "max_rows_per_read")): + if type(value) is not int or value <= 0: + raise ValueError(f"{name} must be a positive integer") + shards = tuple(loaders) + if not shards or any(not isinstance(loader, V41WeightLoader) for loader in shards): + raise ValueError("embedding lookup requires V41WeightLoader TP shards") + first = shards[0] + if len(shards) != first.tp_size or {loader.tp_rank for loader in shards} != set(range(first.tp_size)): + raise ValueError("embedding lookup requires exactly one loader per TP shard") + if any(loader.model_dir != first.model_dir or loader.config != first.config + or loader.tp_size != first.tp_size or loader.ep_size != first.ep_size for loader in shards): + raise ValueError("embedding TP shards must use the same checkpoint and parallel sizes") + if not isinstance(token_ids, torch.Tensor) or token_ids.layout != torch.strided: + raise ValueError("token_ids must be a strided CPU tensor") + if token_ids.device.type != "cpu" or token_ids.dtype not in (torch.int32, torch.int64): + raise ValueError("token_ids must be a CPU int32/int64 tensor") + if token_ids.ndim == 0: + raise ValueError("token_ids must have at least one dimension") + + hidden = first.config.hidden_size + count = token_ids.numel() + if count == 0: + return torch.empty((*token_ids.shape, hidden), dtype=torch.bfloat16) + # Bound allocations before flattening or sorting. Worst case every ID is + # unique; count*128 allows conservative int64 indexing/sorting scratch. + row_bytes = hidden * 2 + estimate = (2 * count + min(count, max_rows_per_read)) * row_bytes + count * 128 + if estimate > max_prepare_bytes: + raise ValueError( + f"embedding preparation requires estimated {estimate} bytes; budget={max_prepare_bytes}" + ) + if bool(((token_ids < 0) | (token_ids >= first.config.vocab_size)).any()): + raise ValueError("token_ids must be inside the checkpoint vocabulary") + + flat = token_ids.reshape(-1).to(torch.int64) + unique, inverse = torch.unique(flat, sorted=True, return_inverse=True) + unique_rows = torch.empty((unique.numel(), hidden), dtype=torch.bfloat16) + by_rank = {loader.tp_rank: loader for loader in shards} + shard_size = first.config.vocab_size // first.tp_size + start = 0 + while start < unique.numel(): + token = int(unique[start]) + rank, local_row = divmod(token, shard_size) + stop = start + 1 + limit = min(unique.numel(), start + max_rows_per_read, start + shard_size - local_row) + while stop < limit and int(unique[stop]) == token + stop - start: + stop += 1 + bundle = by_rank[rank].load_rows("embed.weight", local_row, local_row + stop - start) + unique_rows[start:stop].copy_(bundle.weight) + del bundle + start = stop + return unique_rows.index_select(0, inverse).reshape(*token_ids.shape, hidden) diff --git a/tests/unit/model/deepseek_v41/test_input_preparation.py b/tests/unit/model/deepseek_v41/test_input_preparation.py new file mode 100644 index 00000000..059eb059 --- /dev/null +++ b/tests/unit/model/deepseek_v41/test_input_preparation.py @@ -0,0 +1,132 @@ +# Copyright (c) PyPTO Contributors. +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of +# CANN Open Software License Agreement Version 2.0 (the "License"). +# Please refer to the License for details. You may not use this file except in compliance with the License. +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. +# See LICENSE in the root of the software repository for the full text of the License. +# ----------------------------------------------------------------------------------------------------------- +"""Real checkpoint row reads preserve token order and honor preparation limits.""" + +import json +from pathlib import Path + +import pytest +import torch +from safetensors.torch import save_file + +from pypto_serving.model.deepseek_v41.input_preparation import lookup_token_embeddings +from pypto_serving.model.deepseek_v41.weight_loader import V41WeightLoader +from pypto_serving.model.deepseek_v41.weight_spec import backbone_weight_specs + + +@pytest.fixture +def embedding_checkpoint(tmp_path): + raw = json.loads((Path(__file__).resolve().parents[4] / + "tests/fixtures/deepseek_v41/config.json").read_text(encoding="utf-8")) + raw["text_config"].update(hidden_size=32, vocab_size=256, num_hidden_layers=1, + num_attention_heads=8, head_dim=64, q_lora_rank=256, o_lora_rank=128, o_groups=4, + n_routed_experts=2, moe_intermediate_size=256, index_n_heads=8, index_head_dim=32, + kv_source_layer_ids=[0], index_source_layer_ids=[0], compress_ratios=[2]) + raw.update(bos_token_id=0, eos_token_id=1, pad_token_id=2) + table = (torch.arange(256).reshape(-1, 1) + torch.arange(32).reshape(1, -1) / 32).bfloat16() + save_file({"embed.weight": table}, str(tmp_path / "embedding.safetensors")) + specs = backbone_weight_specs(raw) + index = {name: "embedding.safetensors" if name == "embed.weight" else "not-opened.safetensors" + for name in specs} + (tmp_path / "config.json").write_text(json.dumps(raw), encoding="utf-8") + (tmp_path / "model.safetensors.index.json").write_text( + json.dumps({"weight_map": index}), encoding="utf-8" + ) + loaders = [V41WeightLoader(tmp_path, tp_size=2, tp_rank=rank) for rank in range(2)] + return loaders, table + + +def test_lookup_restores_dp_order_and_deduplicates_reads(embedding_checkpoint, monkeypatch): + loaders, table = embedding_checkpoint + calls = [] + for loader in loaders: + original = loader.load_rows + def read(name, start, stop, original=original, rank=loader.tp_rank): + calls.append((rank, name, start, stop)) + return original(name, start, stop) + monkeypatch.setattr(loader, "load_rows", read) + ids = torch.tensor([[130, 0, 129, 127, 128], [0, 131, 2, 1, 128]], dtype=torch.int32) + result = lookup_token_embeddings(loaders[::-1], ids, max_rows_per_read=2) + assert torch.equal(result, table[ids.long()]) + assert result.dtype == torch.bfloat16 and result.device.type == "cpu" + assert calls == [(0, "embed.weight", 0, 2), (0, "embed.weight", 2, 3), + (0, "embed.weight", 127, 128), (1, "embed.weight", 0, 2), + (1, "embed.weight", 2, 4)] + + +def test_noncontiguous_token_order_and_owned_rows(embedding_checkpoint): + loaders, table = embedding_checkpoint + ids = torch.tensor([[3, 255], [128, 3]], dtype=torch.int64).T + result = lookup_token_embeddings(loaders, ids) + assert torch.equal(result, table[ids]) + result[0, 0].zero_() + assert torch.equal(result[1, 1], table[3]) + assert torch.equal(lookup_token_embeddings(loaders, torch.tensor([3]))[0], table[3]) + + +@pytest.mark.parametrize("shape", [(0,), (2, 0)]) +def test_empty_tokens_require_no_payload(embedding_checkpoint, monkeypatch, shape): + loaders, _ = embedding_checkpoint + def forbidden(*args): + raise AssertionError("empty lookup must not open a checkpoint") + for loader in loaders: + monkeypatch.setattr(loader, "load_rows", forbidden) + result = lookup_token_embeddings(loaders, torch.empty(shape, dtype=torch.int64), max_prepare_bytes=1) + assert result.shape == (*shape, 32) + assert result.dtype == torch.bfloat16 + + +@pytest.mark.parametrize("ids", [torch.tensor([-1]), torch.tensor([256]), torch.tensor([1.0]), + torch.tensor([True]), torch.tensor(1)]) +def test_invalid_token_ids_fail_before_read(embedding_checkpoint, monkeypatch, ids): + loaders, _ = embedding_checkpoint + def forbidden(*args): + raise AssertionError("invalid IDs must fail before checkpoint I/O") + for loader in loaders: + monkeypatch.setattr(loader, "load_rows", forbidden) + with pytest.raises(ValueError, match="token_ids"): + lookup_token_embeddings(loaders, ids) + + +def test_incomplete_or_mixed_shards_rejected(embedding_checkpoint, tmp_path): + loaders, _ = embedding_checkpoint + for values in ([], loaders[:1], [loaders[0], loaders[0]], [*loaders, *loaders]): + with pytest.raises(ValueError, match="TP shard"): + lookup_token_embeddings(values, torch.tensor([0])) + other = tmp_path / "different_checkpoint" + other.mkdir() + for filename in ("config.json", "model.safetensors.index.json"): + (other / filename).write_bytes((tmp_path / filename).read_bytes()) + mixed = V41WeightLoader(other, tp_size=2, tp_rank=1) + with pytest.raises(ValueError, match="same checkpoint"): + lookup_token_embeddings([loaders[0], mixed], torch.tensor([0])) + + +def test_preparation_budget_rejected_before_payload(embedding_checkpoint, monkeypatch): + loaders, _ = embedding_checkpoint + def forbidden(*args): + raise AssertionError("preparation budget must be checked before I/O") + for loader in loaders: + monkeypatch.setattr(loader, "load_rows", forbidden) + with pytest.raises(ValueError, match="preparation.*budget"): + lookup_token_embeddings(loaders, torch.tensor([0, 1, 128, 129]), max_prepare_bytes=1) + + +def test_loader_budget_is_preserved(embedding_checkpoint): + loaders, _ = embedding_checkpoint + loaders[0].max_load_bytes = 1 + with pytest.raises(ValueError, match="weight load.*budget"): + lookup_token_embeddings(loaders, torch.tensor([0])) + + +@pytest.mark.parametrize("options", [{"max_prepare_bytes": 0}, {"max_prepare_bytes": True}, + {"max_rows_per_read": 0}, {"max_rows_per_read": 1.5}]) +def test_invalid_limits(embedding_checkpoint, options): + with pytest.raises(ValueError, match="positive integer"): + lookup_token_embeddings(embedding_checkpoint[0], torch.tensor([0]), **options) From 399c729be7c70795323a959951661ee848a9189d Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Wed, 23 Sep 2026 10:02:37 +0800 Subject: [PATCH 06/78] feat(v41): track chunked prefill and request-owned cache state --- pypto_serving/config/types.py | 3 + pypto_serving/model/deepseek_v41/metadata.py | 130 ++++++++++++++ .../model/deepseek_v41/npu_runner.py | 48 +++++- .../model/deepseek_v41/request_state.py | 159 ++++++++++++++++++ pypto_serving/serving/utils/prefill.py | 2 + .../unit/model/deepseek_v41/test_metadata.py | 146 ++++++++++++++++ .../model/deepseek_v41/test_request_state.py | 73 ++++++++ 7 files changed, 558 insertions(+), 3 deletions(-) create mode 100644 pypto_serving/model/deepseek_v41/metadata.py create mode 100644 pypto_serving/model/deepseek_v41/request_state.py create mode 100644 tests/unit/model/deepseek_v41/test_metadata.py create mode 100644 tests/unit/model/deepseek_v41/test_request_state.py diff --git a/pypto_serving/config/types.py b/pypto_serving/config/types.py index 75184fec..c82d6e39 100644 --- a/pypto_serving/config/types.py +++ b/pypto_serving/config/types.py @@ -304,6 +304,9 @@ class PrefillBatch: block_ids: list[list[int]] = field(default_factory=list) block_ids_by_group: list[dict[str, list[int]]] = field(default_factory=list) cache_partitions: list[int | None] = field(default_factory=list) + # Total original prompt lengths, distinct from seq_lens (this chunk's end). + # Required by integrations that own prefill-to-decode persistent state. + prompt_lens: list[int] = field(default_factory=list) @dataclass diff --git a/pypto_serving/model/deepseek_v41/metadata.py b/pypto_serving/model/deepseek_v41/metadata.py new file mode 100644 index 00000000..c1393e7f --- /dev/null +++ b/pypto_serving/model/deepseek_v41/metadata.py @@ -0,0 +1,130 @@ +# Copyright (c) PyPTO Contributors. +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of +# CANN Open Software License Agreement Version 2.0 (the "License"). +# Please refer to the License for details. You may not use this file except in compliance with the License. +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. +# See LICENSE in the root of the software repository for the full text of the License. +# ----------------------------------------------------------------------------------------------------------- +"""Validate scheduler metadata before taking request/cache state ownership. + +Positions and page tables stay logical here; the lib composite adapter lowers +them to its own physical buffers. There is no implicit assignment from batch +row to DP partition or compressor slot. +""" + +from collections.abc import Mapping, Sequence +import math + +import torch + +from pypto_serving.config.types import KVCacheGroupSpec +from .composite import MissingCompositeInterface + + +def validate_page_tables(rows, partitions, ends, groups): + """Copy private full-history page tables and reject active request aliases. + + Physical IDs are local to each DP partition. Without an explicit num_blocks + the device allocator must check its actual capacity before launching work. + Rolling groups require a position-to-ring-slot ABI and are not guessed here. + """ + if not groups or any(not isinstance(group, KVCacheGroupSpec) for group in groups): + raise MissingCompositeInterface("V4.1 requires explicit grouped cache specifications") + if len({group.name for group in groups}) != len(groups): + raise ValueError("cache group names must be unique") + if any(group.num_partitions != 2 for group in groups): + raise ValueError("V4.1 cache groups require two DP partitions") + if any(group.sliding_window is not None for group in groups): + raise MissingCompositeInterface("rolling cache page lowering requires a verified composite contract") + if len(rows) != len(partitions) or len(rows) != len(ends): + raise ValueError("page tables, partitions and request extents must have matching rows") + required = {group.name for group in groups} + owners, result = set(), [] + for pages, partition, end in zip(rows, partitions, ends): + if type(partition) is not int or not 0 <= partition < 2: + raise ValueError("requests require an explicit DP cache partition in [0, 2)") + if type(end) is not int or end <= 0: + raise ValueError("cache request extent must be a positive integer") + if not isinstance(pages, Mapping) or set(pages) != required: + raise ValueError("request page tables must match the declared cache groups") + copied = {} + for group in groups: + ids = pages[group.name] + if not isinstance(ids, Sequence) or isinstance(ids, (str, bytes)): + raise ValueError("a cache page table must be a sequence of physical IDs") + ids = tuple(ids) + if any(type(page) is not int or page < 0 for page in ids): + raise ValueError("physical page IDs must be nonnegative integers") + if group.num_blocks is not None and any(page >= group.num_blocks for page in ids): + raise ValueError("physical page ID exceeds the configured cache pool") + required_pages = math.ceil(end / group.spec.token_capacity) + if not required_pages <= len(ids) <= group.max_blocks_per_seq: + raise ValueError("cache page table does not cover the request extent within its capacity") + if len(set(ids)) != len(ids): + raise ValueError("a full-history request must not alias its own cache pages") + addresses = {(partition, group.name, page) for page in ids} + if owners.intersection(addresses): + raise ValueError("active requests must not share writable cache pages") + owners.update(addresses) + copied[group.name] = ids + result.append(copied) + return tuple(result) + + +def prefill_requests(batch, config, runtime, groups): + """Build RequestLedger.begin_prefill items from the shared packed batch. + + seq_lens is the end of the current chunk; prompt_lens is the original total + prompt length. Confusing them would mark the first chunk as terminal and + prevent subsequent chunks from continuing their persistent state. + """ + count = len(batch.request_ids) + if not count or count > runtime.max_batch_size: + raise ValueError("prefill request count exceeds the configured batch capacity") + if any(not isinstance(key, str) or not key for key in batch.request_ids): + raise ValueError("request IDs must be nonempty strings") + if len(set(batch.request_ids)) != count: + raise ValueError("prefill requests must have distinct IDs") + for name in ("chunk_lens", "chunk_offsets", "chunk_starts", "seq_lens", "prompt_lens", + "block_ids_by_group", "cache_partitions"): + if len(getattr(batch, name)) != count: + raise ValueError(f"prefill {name} must contain one entry per request") + tokens = batch.token_ids + if not isinstance(tokens, torch.Tensor) or tokens.device.type != "cpu" or tokens.ndim != 1: + raise ValueError("prefill token IDs must be a flat CPU tensor") + if tokens.dtype not in (torch.int32, torch.int64): + raise ValueError("prefill token IDs must use int32 or int64") + if tokens.numel() > runtime.max_num_batched_tokens: + raise ValueError("prefill tokens exceed the configured dispatch capacity") + token_ids = tokens.tolist() + if any(not 0 <= token < config.vocab_size for token in token_ids): + raise ValueError("prefill token ID is outside the vocabulary") + + cursor, chunks = 0, [] + for size, offset, start, end, prompt in zip( + batch.chunk_lens, batch.chunk_offsets, batch.chunk_starts, batch.seq_lens, batch.prompt_lens, + ): + if any(type(value) is not int for value in (size, offset, start, end, prompt)): + raise ValueError("prefill offsets and lengths must be integers") + if size <= 0 or offset != cursor or offset + size > len(token_ids): + raise ValueError("prefill chunks must cover consecutive packed token spans") + if not 0 <= start < end == start + size <= prompt <= min( + runtime.max_seq_len, config.max_position_embeddings, + ): + raise ValueError("prefill chunk extent or total prompt length is invalid") + if runtime.max_prefill_tokens_per_request is not None and size > runtime.max_prefill_tokens_per_request: + raise ValueError("prefill chunk exceeds the per-request capacity") + chunks.append(tuple(token_ids[offset:offset + size])) + cursor += size + if cursor != len(token_ids): + raise ValueError("prefill chunk spans leave unclaimed packed tokens") + pages = validate_page_tables( + batch.block_ids_by_group, batch.cache_partitions, batch.seq_lens, groups, + ) + return [ + (key, partition, start, chunk, prompt, table) + for key, partition, start, chunk, prompt, table in zip( + batch.request_ids, batch.cache_partitions, batch.chunk_starts, chunks, batch.prompt_lens, pages, + ) + ] diff --git a/pypto_serving/model/deepseek_v41/npu_runner.py b/pypto_serving/model/deepseek_v41/npu_runner.py index 183d3576..1c17117e 100644 --- a/pypto_serving/model/deepseek_v41/npu_runner.py +++ b/pypto_serving/model/deepseek_v41/npu_runner.py @@ -9,6 +9,9 @@ """V4-style runner lifecycle for explicit V4.1 composite bindings.""" from .composite import CompositeBindings, MissingCompositeInterface from .execution_plan import V41ExecutionPlan +from .metadata import prefill_requests +from .request_state import RequestLedger +from threading import RLock class V41ModelRunner: @@ -30,6 +33,8 @@ def __init__(self, plan: V41ExecutionPlan, bindings: CompositeBindings, *, devic self.num_pages = None self.closed = False self.failed = False + self.ledger = None + self._lock = RLock() def preflight(self): if self.failed: @@ -61,12 +66,49 @@ def close(self): self.resources = None self.closed = True + def _request_ledger(self): + self.preflight() + if self.ledger is None: + self.ledger = RequestLedger(max_requests=self.runtime.max_batch_size, + max_seq_len=self.runtime.max_seq_len) + return self.ledger + + def _reset_request(self, key, owner): + self.bindings.reset_request(self.resources, key, owner) + self.bindings.wait(self.resources) + def run_prefill(self, model, batch): - raise MissingCompositeInterface("prefill request/state binding is not connected yet") + with self._lock: + ledger = self._request_ledger() + requests = prefill_requests(batch, model.config, self.runtime, self.bindings.cache_groups) + step = ledger.begin_prefill(requests) + return self._run_transaction(step, batch.input_embeddings) + + def _run_transaction(self, step, embeddings): + try: + result = self._execute_step(step, embeddings) + self.bindings.wait(self.resources) + self.ledger.commit(step) + return result + except Exception: + try: + self.bindings.wait(self.resources) + except Exception: + self.failed = True + self.ledger.poisoned = True + # Keep pending state and buffers owned: completion is unknown. + raise + self.ledger.abort(step, self._reset_request) + raise + + def _execute_step(self, step, embeddings): + raise MissingCompositeInterface("backbone/output composite dispatch is not connected yet") def run_decode(self, model, batch): raise MissingCompositeInterface("decode request/state binding is not connected yet") def release_finished_requests(self, request_ids): - if request_ids: - raise MissingCompositeInterface("request cache lifecycle is not connected yet") + with self._lock: + if self.ledger is not None: + self.bindings.wait(self.resources) + self.ledger.release(request_ids, self._reset_request) diff --git a/pypto_serving/model/deepseek_v41/request_state.py b/pypto_serving/model/deepseek_v41/request_state.py new file mode 100644 index 00000000..2b5ca7e5 --- /dev/null +++ b/pypto_serving/model/deepseek_v41/request_state.py @@ -0,0 +1,159 @@ +# Copyright (c) PyPTO Contributors. +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of +# CANN Open Software License Agreement Version 2.0 (the "License"). +# Please refer to the License for details. You may not use this file except in compliance with the License. +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. +# See LICENSE in the root of the software repository for the full text of the License. +# ----------------------------------------------------------------------------------------------------------- +"""Request-owned positions and stable compressor slots, independent of batch rows. + +Physical pages remain owned by the shared scheduler. A failed in-place kernel +cannot be rolled back by restoring a length: its request must be reset and +recomputed. Prefix-cache attachment is intentionally unsupported here. +""" +from dataclasses import dataclass, field +from types import MappingProxyType +from typing import Mapping + + +@dataclass(frozen=True) +class RequestSlice: + request_id: str + partition: int + state_slot: int + start: int + token_ids: tuple[int, ...] + prompt_length: int + pages: Mapping[str, tuple[int, ...]] + + @property + def end(self): + return self.start + len(self.token_ids) + + +@dataclass(frozen=True) +class ForwardStep: + phase: str + requests: tuple[RequestSlice, ...] + epoch: int + + @property + def token_ids(self): + return tuple(token for request in self.requests for token in request.token_ids) + + @property + def positions(self): + return tuple(p for request in self.requests for p in range(request.start, request.end)) + + @property + def request_rows(self): + return tuple(i for i, request in enumerate(self.requests) for _ in request.token_ids) + + @property + def terminal_prefill(self): + return tuple(request.end == request.prompt_length for request in self.requests) + + +@dataclass +class RequestProgress: + partition: int + slot: int + prompt_length: int + length: int = 0 + pages: Mapping[str, tuple[int, ...]] = field(default_factory=dict) + + +class RequestLedger: + """One synchronous in-flight batch; slots survive omission and reordering.""" + def __init__(self, *, max_requests, max_seq_len, partitions=2): + if any(type(v) is not int or v <= 0 for v in (max_requests, max_seq_len, partitions)): + raise ValueError("request capacities must be positive integers") + self.max_seq_len = max_seq_len + self.owners = {} + self.free = [list(range(max_requests - 1, -1, -1)) for _ in range(partitions)] + self.pending = None + self.epoch = 1 + self.poisoned = False + + def begin_prefill(self, requests): + """Validate the complete batch before taking ownership of any new slot. + + Each item is (request_id, partition, start, token_ids, prompt_length, pages). + Input order is retained for output mapping; it never determines state slots. + """ + if self.poisoned: + raise RuntimeError("session requires recovery after device reset failure") + if self.pending is not None: + raise RuntimeError("a request batch is already in flight") + if not requests or len({r[0] for r in requests}) != len(requests): + raise ValueError("batch must contain distinct request IDs") + available = [list(slots) for slots in self.free] + additions, slices = {}, [] + for key, partition, start, tokens, prompt_length, pages in requests: + if not isinstance(key, str) or not key: + raise ValueError("request ID must be nonempty") + if type(partition) is not int or not 0 <= partition < len(available): + raise ValueError("invalid DP cache partition") + if any(type(v) is not int for v in (start, prompt_length)): + raise ValueError("positions and prompt length must be integers") + if not tokens or not 0 <= start < start + len(tokens) <= prompt_length <= self.max_seq_len: + raise ValueError("invalid prefill extent") + owner = self.owners.get(key) + if owner is None: + if start != 0: + raise ValueError("cold requests require prefill from position zero; prefix restore is unavailable") + if not available[partition]: + raise ValueError("compressor state slot capacity exhausted") + owner = RequestProgress(partition, available[partition].pop(), prompt_length) + additions[key] = owner + if owner.partition != partition or owner.length != start or owner.prompt_length != prompt_length: + raise ValueError("request partition, committed position or prompt length changed") + if owner.length == owner.prompt_length: + raise ValueError("prefill already completed") + immutable_pages = MappingProxyType({name: tuple(ids) for name, ids in pages.items()}) + slices.append(RequestSlice(key, partition, owner.slot, start, tuple(tokens), + prompt_length, immutable_pages)) + self.free = available + self.owners.update(additions) + self.pending = ForwardStep("prefill", tuple(slices), self.epoch) + return self.pending + + def commit(self, step): + if step is not self.pending: + raise ValueError("stale or foreign forward completion") + for request in step.requests: + owner = self.owners[request.request_id] + owner.length, owner.pages = request.end, request.pages + self.pending = None + self.epoch += 1 + + def release(self, request_ids, reset): + """reset is synchronous; failed reset retains ownership and poisons reuse.""" + if self.pending is not None: + raise RuntimeError("cannot release an in-flight batch") + for key in dict.fromkeys(request_ids): + owner = self.owners.get(key) + if owner is None: + continue + try: + reset(key, owner) + except Exception: + self.poisoned = True + raise + del self.owners[key] + self.free[owner.partition].append(owner.slot) + + def abort(self, step, reset): + if step is not self.pending: + raise ValueError("stale or foreign failed forward") + for request in step.requests: + owner = self.owners[request.request_id] + names = owner.pages.keys() | request.pages.keys() + owner.pages = MappingProxyType({name: tuple(dict.fromkeys( + (*owner.pages.get(name, ()), *request.pages.get(name, ())))) for name in names}) + self.pending = None + # All layer/cache writes are potentially partial. Invalidate the whole + # affected request, including its previously committed prefix. + self.release([r.request_id for r in step.requests], reset) + self.epoch += 1 diff --git a/pypto_serving/serving/utils/prefill.py b/pypto_serving/serving/utils/prefill.py index c680003a..bb25e7fb 100644 --- a/pypto_serving/serving/utils/prefill.py +++ b/pypto_serving/serving/utils/prefill.py @@ -30,6 +30,7 @@ def pack_prefill_batch( block_ids: Sequence[Sequence[int]] = (), block_ids_by_group: Sequence[dict[str, list[int]]] = (), cache_partitions: Sequence[int | None] = (), + prompt_lens: Sequence[int] = (), ) -> PrefillBatch: """Pack request chunks and optionally look up all host embeddings once.""" chunk_lens = [len(chunk) for chunk in token_chunks] @@ -61,4 +62,5 @@ def pack_prefill_batch( block_ids=[list(row) for row in block_ids], block_ids_by_group=list(block_ids_by_group), cache_partitions=list(cache_partitions), + prompt_lens=list(prompt_lens), ) diff --git a/tests/unit/model/deepseek_v41/test_metadata.py b/tests/unit/model/deepseek_v41/test_metadata.py new file mode 100644 index 00000000..4c48bc43 --- /dev/null +++ b/tests/unit/model/deepseek_v41/test_metadata.py @@ -0,0 +1,146 @@ +# Copyright (c) PyPTO Contributors. +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of +# CANN Open Software License Agreement Version 2.0 (the "License"). +# Please refer to the License for details. You may not use this file except in compliance with the License. +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. +# See LICENSE in the root of the software repository for the full text of the License. +# ----------------------------------------------------------------------------------------------------------- +"""Bounded scheduler-to-request metadata tests, with no model/device execution.""" + +from dataclasses import replace +from types import SimpleNamespace + +import pytest +import torch + +from pypto_serving.config.types import KVCacheGroupSpec, KVCacheSpec, RuntimeConfig +from pypto_serving.model.deepseek_v41.composite import MissingCompositeInterface +from pypto_serving.model.deepseek_v41.metadata import prefill_requests +from pypto_serving.model.deepseek_v41.request_state import RequestLedger +from pypto_serving.serving.utils.prefill import pack_prefill_batch + + +@pytest.fixture +def inputs(): + config = SimpleNamespace(vocab_size=32, max_position_embeddings=16) + runtime = RuntimeConfig(max_batch_size=2, max_seq_len=16, max_num_batched_tokens=8) + groups = ( + KVCacheGroupSpec("window", (0,), KVCacheSpec(2, 16), 8, num_blocks=16, num_partitions=2), + KVCacheGroupSpec("compressed", (2,), KVCacheSpec(4, 16, 2), 4, num_blocks=16, num_partitions=2), + ) + batch = pack_prefill_batch( + request_ids=["A", "B"], token_chunks=[[4, 5, 6], [7, 8]], seq_lens=[3, 2], + chunk_starts=[0, 0], prompt_lens=[5, 2], device="cpu", cache_partitions=[1, 0], + block_ids_by_group=[{"window": [0, 1], "compressed": [3]}, {"window": [0], "compressed": [3]}], + ) + return batch, config, runtime, groups + + +def test_chunk_end_is_not_total_prompt_length(inputs): + batch, config, runtime, groups = inputs + ledger = RequestLedger(max_requests=2, max_seq_len=16) + step = ledger.begin_prefill(prefill_requests(batch, config, runtime, groups)) + assert step.positions == (0, 1, 2, 0, 1) + assert step.request_rows == (0, 0, 0, 1, 1) + assert step.terminal_prefill == (False, True) + assert step.requests[0].partition == 1 + ledger.commit(step) + next_batch = pack_prefill_batch( + request_ids=["A"], token_chunks=[[9, 10]], seq_lens=[5], chunk_starts=[3], prompt_lens=[5], + device="cpu", cache_partitions=[1], + block_ids_by_group=[{"window": [0, 1, 2], "compressed": [3, 4]}], + ) + tail = ledger.begin_prefill(prefill_requests(next_batch, config, runtime, groups)) + assert tail.positions == (3, 4) + assert tail.terminal_prefill == (True,) + assert tail.requests[0].state_slot == step.requests[0].state_slot + + +def test_tables_are_copied_and_page_ids_are_partition_local(inputs): + batch, config, runtime, groups = inputs + requests = prefill_requests(batch, config, runtime, groups) + batch.block_ids_by_group[0]["window"][0] = 15 + assert requests[0][-1]["window"] == (0, 1) + assert requests[1][-1]["window"] == (0,) + + +@pytest.mark.parametrize("field,value,message", [ + ("request_ids", ["A", "A"], "distinct"), + ("chunk_offsets", [0, 2], "consecutive"), + ("chunk_offsets", [1, 4], "consecutive"), + ("chunk_lens", [0, 2], "consecutive"), + ("chunk_starts", [True, 0], "must be integers"), + ("seq_lens", [4, 2], "extent"), + ("prompt_lens", [], "prompt_lens"), + ("prompt_lens", [2, 2], "extent"), + ("prompt_lens", [17, 2], "extent"), + ("cache_partitions", [None, 0], "explicit DP"), + ("cache_partitions", [2, 0], "explicit DP"), + ("cache_partitions", [0, 0], "share writable"), +]) +def test_invalid_batch_metadata_is_rejected(inputs, field, value, message): + batch, config, runtime, groups = inputs + setattr(batch, field, value) + with pytest.raises(ValueError, match=message): + prefill_requests(batch, config, runtime, groups) + + +@pytest.mark.parametrize("tokens,message", [ + (torch.tensor([4., 5., 6., 7., 8.]), "int32 or int64"), + (torch.tensor([[4, 5, 6, 7, 8]]), "flat CPU"), + (torch.tensor([-1, 5, 6, 7, 8]), "vocabulary"), + (torch.tensor([32, 5, 6, 7, 8]), "vocabulary"), + (torch.tensor([4, 5, 6, 7, 8, 9]), "unclaimed"), +]) +def test_invalid_token_storage_or_range_is_rejected(inputs, tokens, message): + batch, config, runtime, groups = inputs + batch.token_ids = tokens + with pytest.raises(ValueError, match=message): + prefill_requests(batch, config, runtime, groups) + + +@pytest.mark.parametrize("pages,message", [ + ({"window": [0, 1]}, "declared cache groups"), + ({"window": [0, 1], "compressed": [3], "unknown": [0]}, "declared cache groups"), + ({"window": [0], "compressed": [3]}, "cover the request extent"), + ({"window": [0, 0], "compressed": [3]}, "alias its own"), + ({"window": [-1, 1], "compressed": [3]}, "nonnegative integers"), + ({"window": [True, 1], "compressed": [3]}, "nonnegative integers"), + ({"window": [16, 1], "compressed": [3]}, "exceeds the configured cache pool"), +]) +def test_invalid_page_tables_are_rejected(inputs, pages, message): + batch, config, runtime, groups = inputs + batch.block_ids_by_group[0] = pages + with pytest.raises(ValueError, match=message): + prefill_requests(batch, config, runtime, groups) + + +def test_rolling_lowering_fails_without_guessing_the_ring_contract(inputs): + batch, config, runtime, groups = inputs + groups = (replace(groups[0], sliding_window=2), groups[1]) + with pytest.raises(MissingCompositeInterface, match="rolling cache"): + prefill_requests(batch, config, runtime, groups) + + +def test_missing_group_contract_cannot_use_generic_pages(inputs): + batch, config, runtime, _ = inputs + batch.block_ids = [[0, 1], [0]] + with pytest.raises(MissingCompositeInterface, match="explicit grouped"): + prefill_requests(batch, config, runtime, ()) + + +def test_dynamic_physical_capacity_is_left_for_allocator_validation(inputs): + batch, config, runtime, groups = inputs + groups = (replace(groups[0], num_blocks=None), groups[1]) + batch.block_ids_by_group[0]["window"] = [300, 301] + result = prefill_requests(batch, config, runtime, groups) + assert result[0][-1]["window"] == (300, 301) + + +def test_dispatch_and_per_request_token_limits(inputs): + batch, config, runtime, groups = inputs + with pytest.raises(ValueError, match="dispatch capacity"): + prefill_requests(batch, config, replace(runtime, max_num_batched_tokens=4), groups) + with pytest.raises(ValueError, match="per-request capacity"): + prefill_requests(batch, config, replace(runtime, max_prefill_tokens_per_request=2), groups) diff --git a/tests/unit/model/deepseek_v41/test_request_state.py b/tests/unit/model/deepseek_v41/test_request_state.py new file mode 100644 index 00000000..c6b7ffad --- /dev/null +++ b/tests/unit/model/deepseek_v41/test_request_state.py @@ -0,0 +1,73 @@ +# Copyright (c) PyPTO Contributors. +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of +# CANN Open Software License Agreement Version 2.0 (the "License"). +# Please refer to the License for details. You may not use this file except in compliance with the License. +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. +# See LICENSE in the root of the software repository for the full text of the License. +# ----------------------------------------------------------------------------------------------------------- +"""Lifecycle evidence uses no model math or device emulation.""" +import pytest +from pypto_serving.model.deepseek_v41.request_state import RequestLedger + + +def item(key="a", partition=0, start=0, tokens=(4, 5), length=4, pages=(3,)): + return key, partition, start, tokens, length, {"window": pages} + + +def test_slots_survive_pause_and_batch_reorder(): + ledger = RequestLedger(max_requests=2, max_seq_len=128) + first = ledger.begin_prefill([item("a"), item("b")]) + slots = {r.request_id: r.state_slot for r in first.requests} + ledger.commit(first) + second = ledger.begin_prefill([item("b", start=2)]) + assert second.requests[0].state_slot == slots["b"] + assert ledger.owners["a"].slot == slots["a"] + assert second.positions == (2, 3) + ledger.commit(second) + with pytest.raises(ValueError, match="stale"): + ledger.commit(first) + + +def test_bad_batch_does_not_reserve_partial_ownership(): + ledger = RequestLedger(max_requests=1, max_seq_len=128) + with pytest.raises(ValueError, match="capacity"): + ledger.begin_prefill([item("a"), item("b")]) + assert ledger.owners == {} and ledger.free[0] == [0] + + +def test_failed_chunk_invalidates_prefix_and_clears_new_pages(): + ledger = RequestLedger(max_requests=2, max_seq_len=128) + step = ledger.begin_prefill([item()]) + ledger.commit(step) + failed = ledger.begin_prefill([item(start=2, pages=(3, 9))]) + calls = [] + ledger.abort(failed, lambda key, owner: calls.append((key, owner.slot, owner.pages["window"]))) + assert calls == [("a", 0, (3, 9))] + assert ledger.owners == {} and ledger.pending is None + with pytest.raises(ValueError, match="position zero"): + ledger.begin_prefill([item(start=2)]) + retry = ledger.begin_prefill([item()]) + assert retry.requests[0].state_slot == 0 + + +def test_reset_failure_keeps_slot_and_blocks_future_requests(): + ledger = RequestLedger(max_requests=1, max_seq_len=128) + step = ledger.begin_prefill([item()]) + def reset(*args): + raise RuntimeError("device lost") + with pytest.raises(RuntimeError, match="device lost"): + ledger.abort(step, reset) + assert "a" in ledger.owners and not ledger.free[0] + with pytest.raises(RuntimeError, match="recovery"): + ledger.begin_prefill([item("b")]) + + +def test_cannot_release_until_forward_completes(): + ledger = RequestLedger(max_requests=1, max_seq_len=128) + step = ledger.begin_prefill([item()]) + with pytest.raises(RuntimeError, match="in-flight"): + ledger.release(["a"], lambda *args: None) + ledger.commit(step) + ledger.release(["a", "a", "unknown"], lambda *args: None) + assert ledger.free[0] == [0] From deab8dbb01438905239b6fd6b382b7fac4a656b2 Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Wed, 23 Sep 2026 10:03:53 +0800 Subject: [PATCH 07/78] feat(v41): continue decode from committed prefill state --- pypto_serving/model/deepseek_v41/metadata.py | 28 +++++++++++++++++++ .../model/deepseek_v41/npu_runner.py | 8 ++++-- .../model/deepseek_v41/request_state.py | 21 ++++++++++++++ .../unit/model/deepseek_v41/test_metadata.py | 15 ++++++++++ .../model/deepseek_v41/test_request_state.py | 22 +++++++++++++++ 5 files changed, 92 insertions(+), 2 deletions(-) diff --git a/pypto_serving/model/deepseek_v41/metadata.py b/pypto_serving/model/deepseek_v41/metadata.py index c1393e7f..74415657 100644 --- a/pypto_serving/model/deepseek_v41/metadata.py +++ b/pypto_serving/model/deepseek_v41/metadata.py @@ -128,3 +128,31 @@ def prefill_requests(batch, config, runtime, groups): batch.request_ids, batch.cache_partitions, batch.chunk_starts, chunks, batch.prompt_lens, pages, ) ] + + +def decode_requests(batch, config, runtime, groups): + """Decode consumes exactly one supplied token per request; no padding rows.""" + count = len(batch.request_ids) + if not count or count > min(runtime.max_batch_size, runtime.max_num_batched_tokens): + raise ValueError("decode request count exceeds the configured batch capacity") + if len(set(batch.request_ids)) != count or any(not isinstance(k, str) or not k for k in batch.request_ids): + raise ValueError("decode requires distinct nonempty request IDs") + tokens, lengths = batch.token_ids, batch.seq_lens + for value in (tokens, lengths): + if not isinstance(value, torch.Tensor) or value.device.type != "cpu" or value.dtype not in ( + torch.int32, torch.int64, + ): + raise ValueError("decode tokens and lengths must be CPU integer tensors") + if tuple(tokens.shape) not in ((count,), (count, 1)) or tuple(lengths.shape) != (count,): + raise ValueError("decode requires one token and one sequence length per request") + ids, ends = tokens.reshape(-1).tolist(), lengths.tolist() + if any(not 0 <= token < config.vocab_size for token in ids): + raise ValueError("decode token ID is outside the vocabulary") + if any(not 1 <= end <= min(runtime.max_seq_len, config.max_position_embeddings) for end in ends): + raise ValueError("decode sequence length exceeds model capacity") + if len(batch.cache_partitions) != count or len(batch.block_ids_by_group) != count: + raise ValueError("decode requires one partition and grouped page table per request") + pages = validate_page_tables(batch.block_ids_by_group, batch.cache_partitions, ends, groups) + return [(key, partition, end - 1, token, table) + for key, partition, end, token, table in zip( + batch.request_ids, batch.cache_partitions, ends, ids, pages)] diff --git a/pypto_serving/model/deepseek_v41/npu_runner.py b/pypto_serving/model/deepseek_v41/npu_runner.py index 1c17117e..640bffd0 100644 --- a/pypto_serving/model/deepseek_v41/npu_runner.py +++ b/pypto_serving/model/deepseek_v41/npu_runner.py @@ -9,7 +9,7 @@ """V4-style runner lifecycle for explicit V4.1 composite bindings.""" from .composite import CompositeBindings, MissingCompositeInterface from .execution_plan import V41ExecutionPlan -from .metadata import prefill_requests +from .metadata import prefill_requests, decode_requests from .request_state import RequestLedger from threading import RLock @@ -105,7 +105,11 @@ def _execute_step(self, step, embeddings): raise MissingCompositeInterface("backbone/output composite dispatch is not connected yet") def run_decode(self, model, batch): - raise MissingCompositeInterface("decode request/state binding is not connected yet") + with self._lock: + ledger = self._request_ledger() + requests = decode_requests(batch, model.config, self.runtime, self.bindings.cache_groups) + step = ledger.begin_decode(requests) + return self._run_transaction(step, batch.hidden_states) def release_finished_requests(self, request_ids): with self._lock: diff --git a/pypto_serving/model/deepseek_v41/request_state.py b/pypto_serving/model/deepseek_v41/request_state.py index 2b5ca7e5..0b9cfe36 100644 --- a/pypto_serving/model/deepseek_v41/request_state.py +++ b/pypto_serving/model/deepseek_v41/request_state.py @@ -119,6 +119,27 @@ def begin_prefill(self, requests): self.pending = ForwardStep("prefill", tuple(slices), self.epoch) return self.pending + def begin_decode(self, requests): + if self.poisoned: + raise RuntimeError("session requires recovery after device reset failure") + if self.pending is not None: + raise RuntimeError("a request batch is already in flight") + if not requests or len({r[0] for r in requests}) != len(requests): + raise ValueError("batch must contain distinct request IDs") + slices = [] + for key, partition, start, token, pages in requests: + owner = self.owners.get(key) + if owner is None or owner.length < owner.prompt_length: + raise ValueError("decode requires a completed prefill for the same request") + if type(start) is not int or owner.length != start or owner.partition != partition: + raise ValueError("decode must continue at the committed position and DP partition") + if start + 1 > self.max_seq_len: + raise ValueError("decode exceeds sequence capacity") + slices.append(RequestSlice(key, partition, owner.slot, start, (token,), owner.prompt_length, + MappingProxyType({name: tuple(ids) for name, ids in pages.items()}))) + self.pending = ForwardStep("decode", tuple(slices), self.epoch) + return self.pending + def commit(self, step): if step is not self.pending: raise ValueError("stale or foreign forward completion") diff --git a/tests/unit/model/deepseek_v41/test_metadata.py b/tests/unit/model/deepseek_v41/test_metadata.py index 4c48bc43..95f0f98b 100644 --- a/tests/unit/model/deepseek_v41/test_metadata.py +++ b/tests/unit/model/deepseek_v41/test_metadata.py @@ -144,3 +144,18 @@ def test_dispatch_and_per_request_token_limits(inputs): prefill_requests(batch, config, replace(runtime, max_num_batched_tokens=4), groups) with pytest.raises(ValueError, match="per-request capacity"): prefill_requests(batch, config, replace(runtime, max_prefill_tokens_per_request=2), groups) + + +def test_decode_worker_column_tokens_preserve_request_order(inputs): + from pypto_serving.config.types import DecodeBatch + from pypto_serving.model.deepseek_v41.metadata import decode_requests + _, config, runtime, groups = inputs + batch = DecodeBatch(request_ids=["B", "A"], token_ids=torch.tensor([[7], [9]]), + hidden_states=None, seq_lens=torch.tensor([3, 6]), cache_partitions=[0, 1], + block_ids_by_group=[{"window": [0, 1], "compressed": [3]}, + {"window": [0, 1, 2], "compressed": [3, 4]}]) + rows = decode_requests(batch, config, runtime, groups) + assert [r[:4] for r in rows] == [("B", 0, 2, 7), ("A", 1, 5, 9)] + batch.token_ids = torch.tensor([[7, 8], [9, 10]]) + with pytest.raises(ValueError, match="one token"): + decode_requests(batch, config, runtime, groups) diff --git a/tests/unit/model/deepseek_v41/test_request_state.py b/tests/unit/model/deepseek_v41/test_request_state.py index c6b7ffad..c258dbb1 100644 --- a/tests/unit/model/deepseek_v41/test_request_state.py +++ b/tests/unit/model/deepseek_v41/test_request_state.py @@ -71,3 +71,25 @@ def test_cannot_release_until_forward_completes(): ledger.commit(step) ledger.release(["a", "a", "unknown"], lambda *args: None) assert ledger.free[0] == [0] + + +def test_decode_continues_prefill_and_cannot_skip_or_repeat_positions(): + ledger = RequestLedger(max_requests=2, max_seq_len=128) + first = ledger.begin_prefill([item(length=2)]) + slot = first.requests[0].state_slot + ledger.commit(first) + for position in range(2, 6): + step = ledger.begin_decode([("a", 0, position, 7, {"window": (3,)})]) + assert step.positions == (position,) and step.requests[0].state_slot == slot + ledger.commit(step) + for position in (4, 7): + with pytest.raises(ValueError, match="committed position"): + ledger.begin_decode([("a", 0, position, 7, {"window": (3,)})]) + + +def test_decode_rejects_partial_prefill_and_unknown_requests(): + ledger = RequestLedger(max_requests=2, max_seq_len=128) + ledger.commit(ledger.begin_prefill([item(length=4)])) + for key in ("a", "unknown"): + with pytest.raises(ValueError, match="completed prefill"): + ledger.begin_decode([(key, 0, 2, 7, {"window": (3,)})]) From 820ff9c6e1a6ba6a577fdffd73a46c139dad7158 Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Wed, 23 Sep 2026 10:04:52 +0800 Subject: [PATCH 08/78] feat(v41): orchestrate rank-owned weights and complete layer composites --- .../model/deepseek_v41/execution_plan.py | 7 ++ .../model/deepseek_v41/npu_executor.py | 3 + .../model/deepseek_v41/npu_runner.py | 50 +++++++++++++- .../deepseek_v41/test_composite_dispatch.py | 67 +++++++++++++++++++ 4 files changed, 125 insertions(+), 2 deletions(-) create mode 100644 tests/unit/model/deepseek_v41/test_composite_dispatch.py diff --git a/pypto_serving/model/deepseek_v41/execution_plan.py b/pypto_serving/model/deepseek_v41/execution_plan.py index b29836a7..1e9b5297 100644 --- a/pypto_serving/model/deepseek_v41/execution_plan.py +++ b/pypto_serving/model/deepseek_v41/execution_plan.py @@ -148,3 +148,10 @@ def load_weight(self, layer_id, name): if name not in self.weight_names(layer_id): raise ValueError("weight is not owned by the selected layer and rank") return self.weights.load(name) + + def for_rank(self, rank): + """Create metadata-only owned weight access for one collective participant.""" + placement = RankPlacement(rank, self.placement.tp_size, self.placement.dp_size, self.placement.ep_size) + if placement == self.placement: + return self + return V41ExecutionPlan(self.weights.model_dir, placement, max_load_bytes=self.weights.max_load_bytes) diff --git a/pypto_serving/model/deepseek_v41/npu_executor.py b/pypto_serving/model/deepseek_v41/npu_executor.py index 6f69ba39..14aa9380 100644 --- a/pypto_serving/model/deepseek_v41/npu_executor.py +++ b/pypto_serving/model/deepseek_v41/npu_executor.py @@ -48,6 +48,9 @@ def register_model(self, model_id, record): self.runners[model_id] = runner return pages + def lookup_embeddings(self, model, token_ids): + return self.runners[model.config.model_id].lookup_embeddings(token_ids) + def run_prefill(self, model, batch): return self.runners[model.config.model_id].run_prefill(model, batch) diff --git a/pypto_serving/model/deepseek_v41/npu_runner.py b/pypto_serving/model/deepseek_v41/npu_runner.py index 640bffd0..e1a707f4 100644 --- a/pypto_serving/model/deepseek_v41/npu_runner.py +++ b/pypto_serving/model/deepseek_v41/npu_runner.py @@ -7,7 +7,9 @@ # See LICENSE in the root of the software repository for the full text of the License. # ----------------------------------------------------------------------------------------------------------- """V4-style runner lifecycle for explicit V4.1 composite bindings.""" -from .composite import CompositeBindings, MissingCompositeInterface +from .composite import CompositeBindings, MissingCompositeInterface, LayerState +from .input_preparation import lookup_token_embeddings +import torch from .execution_plan import V41ExecutionPlan from .metadata import prefill_requests, decode_requests from .request_state import RequestLedger @@ -35,6 +37,7 @@ def __init__(self, plan: V41ExecutionPlan, bindings: CompositeBindings, *, devic self.failed = False self.ledger = None self._lock = RLock() + self._plans = None def preflight(self): if self.failed: @@ -101,8 +104,51 @@ def _run_transaction(self, step, embeddings): self.ledger.abort(step, self._reset_request) raise + def _rank_plans(self): + if self._plans is None: + self._plans = tuple(self.plan.for_rank(rank) for rank in range(self.plan.placement.ep_size)) + return self._plans + + def lookup_embeddings(self, token_ids): + # A complete TP vocabulary exists in each DP group; read it once on host. + plans = self._rank_plans()[:self.plan.placement.tp_size] + return lookup_token_embeddings([plan.weights for plan in plans], token_ids) + + def _check_state(self, state): + if not isinstance(state, LayerState) or state.residual is None or state.pre_mix is None: + raise ValueError("composite must return both residual and delayed pre_mix") + if state.layout != self.bindings.output_layout: + raise ValueError("composite output token layout differs from the next layer input") + return state + + def _run_layers(self, step, embeddings): + config = self.plan.weights.config + if embeddings is None: + embeddings = self.lookup_embeddings(torch.tensor(step.token_ids, dtype=torch.int64)) + if (not isinstance(embeddings, torch.Tensor) or embeddings.device.type != "cpu" + or embeddings.dtype != torch.bfloat16 + or tuple(embeddings.shape) != (len(step.token_ids), config.hidden_size)): + raise ValueError("input embeddings must be packed CPU BF16 [active_tokens, hidden_size]") + plans = self._rank_plans() + state = self.bindings.initialize(embeddings, step, self.resources) + self.bindings.wait(self.resources) + self._check_state(state) + for layer in self.plan.layers: + # The adapter chooses bounded staging or residency; payloads stay packed. + weights = self.bindings.prepare_weights(plans, layer, self.resources) + self.bindings.wait(self.resources) + state = self.bindings.entries[(step.phase, layer.mode)]( + layer, state, step, self.resources, weights) + self.bindings.wait(self.resources) + self._check_state(state) + return state + def _execute_step(self, step, embeddings): - raise MissingCompositeInterface("backbone/output composite dispatch is not connected yet") + state = self._run_layers(step, embeddings) + return self._finish_step(state, step) + + def _finish_step(self, state, step): + raise MissingCompositeInterface("final HC/Norm/LM-head result binding is not connected yet") def run_decode(self, model, batch): with self._lock: diff --git a/tests/unit/model/deepseek_v41/test_composite_dispatch.py b/tests/unit/model/deepseek_v41/test_composite_dispatch.py new file mode 100644 index 00000000..fabd01c9 --- /dev/null +++ b/tests/unit/model/deepseek_v41/test_composite_dispatch.py @@ -0,0 +1,67 @@ +# Copyright (c) PyPTO Contributors. +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of +# CANN Open Software License Agreement Version 2.0 (the "License"). +# Please refer to the License for details. You may not use this file except in compliance with the License. +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. +# See LICENSE in the root of the software repository for the full text of the License. +# ----------------------------------------------------------------------------------------------------------- +"""Recording adapters test orchestration only, not model numerical correctness.""" +from types import SimpleNamespace +import pytest +import torch + +from pypto_serving.model.deepseek_v41.composite import CompositeBindings, LayerState +from pypto_serving.model.deepseek_v41.execution_plan import LayerPlan, RankPlacement +from pypto_serving.model.deepseek_v41.npu_runner import V41ModelRunner +from pypto_serving.model.deepseek_v41.request_state import ForwardStep, RequestSlice + + +def make_runner(*, layout="tp_local_token", entry=None, output=None): + events = [] + layers = tuple(LayerPlan(i, mode, None if i == 0 else 1, None if i == 0 else 1, None) + for i, mode in enumerate(("swa", "c2a_full", "c2a_reuse"))) + def initialize(embeddings, step, resources): + events.append(("initialize", step.positions)) + return LayerState(object(), object(), layout) + def call(layer, state, step, resources, weights): + events.append(("layer", layer.layer_id, step.phase, layer.kv_source, weights)) + return LayerState(object(), object(), layout) + def prepare(plans, layer, resources): + assert [p.placement.rank for p in plans] == list(range(8)) + events.append(("weights", layer.layer_id)) + return layer.layer_id + bindings = CompositeBindings(revision="recording-test-adapter", input_layout=layout, output_layout=layout, + entries={(phase, layer.mode): entry or call for phase in ("prefill", "decode") for layer in layers}, + initialize=initialize, output=output or (lambda *a: None), allocate=lambda *a: (object(), 8), + prepare_weights=prepare, reset_request=lambda *a: events.append(("reset", a[1])), + wait=lambda *a: events.append(("wait",)), close=lambda *a: events.append(("close",))) + config = SimpleNamespace(hidden_size=4, vocab_size=16, max_position_embeddings=128) + plan = SimpleNamespace(placement=RankPlacement(0), layers=layers, weights=SimpleNamespace(config=config)) + plan.for_rank = lambda rank: SimpleNamespace(placement=RankPlacement(rank)) + runtime = SimpleNamespace(max_batch_size=2, max_seq_len=128) + runner = V41ModelRunner(plan, bindings, device_ids=(9, 3, 8, 4, 1, 7, 2, 0), runtime=runtime) + runner.preflight() + return runner, events + + +@pytest.mark.parametrize("phase", ["prefill", "decode"]) +def test_all_layers_keep_state_and_logical_rank_mapping(phase): + runner, events = make_runner() + step = ForwardStep(phase, (RequestSlice("a", 1, 0, 7, (2,), 8, {}),), 4) + state = runner._run_layers(step, torch.ones(1, 4, dtype=torch.bfloat16)) + assert state.layout == "tp_local_token" + assert [e for e in events if e[0] == "layer"] == [ + ("layer", 0, phase, None, 0), ("layer", 1, phase, 1, 1), ("layer", 2, phase, 1, 2)] + # A collective completion separates every layer from weight/window reuse. + for i, event in enumerate(events): + if event[0] == "layer": + assert events[i + 1] == ("wait",) + + +def test_wrong_layout_fails_before_next_layer(): + runner, events = make_runner(entry=lambda *a: LayerState(object(), object(), "tp_replicated")) + step = ForwardStep("prefill", (RequestSlice("a", 0, 0, 0, (2,), 1, {}),), 1) + with pytest.raises(ValueError, match="layout"): + runner._run_layers(step, torch.ones(1, 4, dtype=torch.bfloat16)) + assert [e for e in events if e[0] == "weights"] == [("weights", 0)] From 927e6aa71432d07e229e94bf10b9d14b1d14d4c4 Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Wed, 23 Sep 2026 10:06:08 +0800 Subject: [PATCH 09/78] feat(v41): return validated logits to the shared generation loop --- .../model/deepseek_v41/npu_runner.py | 19 ++++++++- .../deepseek_v41/test_composite_dispatch.py | 42 +++++++++++++++++++ 2 files changed, 59 insertions(+), 2 deletions(-) diff --git a/pypto_serving/model/deepseek_v41/npu_runner.py b/pypto_serving/model/deepseek_v41/npu_runner.py index e1a707f4..4353db10 100644 --- a/pypto_serving/model/deepseek_v41/npu_runner.py +++ b/pypto_serving/model/deepseek_v41/npu_runner.py @@ -7,9 +7,10 @@ # See LICENSE in the root of the software repository for the full text of the License. # ----------------------------------------------------------------------------------------------------------- """V4-style runner lifecycle for explicit V4.1 composite bindings.""" -from .composite import CompositeBindings, MissingCompositeInterface, LayerState +from .composite import CompositeBindings, LayerState from .input_preparation import lookup_token_embeddings import torch +from pypto_serving.config.types import PrefillResult, DecodeResult from .execution_plan import V41ExecutionPlan from .metadata import prefill_requests, decode_requests from .request_state import RequestLedger @@ -148,7 +149,21 @@ def _execute_step(self, step, embeddings): return self._finish_step(state, step) def _finish_step(self, state, step): - raise MissingCompositeInterface("final HC/Norm/LM-head result binding is not connected yet") + logits = self.bindings.output(state, step, self.resources) + self.bindings.wait(self.resources) + shape = (len(step.requests), self.plan.weights.config.vocab_size) + if (not isinstance(logits, torch.Tensor) or logits.device.type != "cpu" + or tuple(logits.shape) != shape or logits.dtype not in ( + torch.float32, torch.bfloat16, torch.float16)): + raise ValueError("output composite must return CPU logits [requests, vocabulary] in request order") + if not bool(torch.isfinite(logits).all()): + raise ValueError("output composite returned non-finite logits") + # Shared workers may sample after the next dispatch reuses device output + # scratch. Keep returned host logits independent of adapter-owned buffers. + owned_logits = logits.detach().clone() + if step.phase == "prefill": + return PrefillResult(last_hidden=None, logits=owned_logits) + return DecodeResult(hidden_states=None, logits=owned_logits) def run_decode(self, model, batch): with self._lock: diff --git a/tests/unit/model/deepseek_v41/test_composite_dispatch.py b/tests/unit/model/deepseek_v41/test_composite_dispatch.py index fabd01c9..020158ce 100644 --- a/tests/unit/model/deepseek_v41/test_composite_dispatch.py +++ b/tests/unit/model/deepseek_v41/test_composite_dispatch.py @@ -65,3 +65,45 @@ def test_wrong_layout_fails_before_next_layer(): with pytest.raises(ValueError, match="layout"): runner._run_layers(step, torch.ones(1, 4, dtype=torch.bfloat16)) assert [e for e in events if e[0] == "weights"] == [("weights", 0)] + + +def test_output_uses_shared_greedy_sampler_and_feeds_decode(): + from pypto_serving.config.types import SamplingParams + from pypto_serving.model.common.executor.sampler import Sampler + def head(state, step, resources): + logits = torch.full((len(step.requests), 16), -10.0) + for row, request in enumerate(step.requests): + logits[row, (request.token_ids[-1] + 1) % 16] = 10.0 + return logits + runner, events = make_runner(output=head) + ledger = runner._request_ledger() + step = ledger.begin_prefill([("a", 0, 0, (2, 3), 2, {})]) + result = runner._run_transaction(step, torch.ones(2, 4, dtype=torch.bfloat16)) + sampler, generated = Sampler(), [] + for position in range(2, 6): + token = sampler.sample(result.logits[0], SamplingParams(temperature=0.0), "a") + generated.append(token) + step = ledger.begin_decode([("a", 0, position, token, {})]) + result = runner._run_transaction(step, torch.ones(1, 4, dtype=torch.bfloat16)) + assert generated == [4, 5, 6, 7] + assert ledger.owners["a"].length == 6 + runner.release_finished_requests(["a"]) + assert ledger.owners == {} and ("reset", "a") in events + + +def test_bad_output_invalidates_state_before_sampling(): + runner, events = make_runner(output=lambda *a: torch.full((1, 16), float("nan"))) + ledger = runner._request_ledger() + step = ledger.begin_prefill([("a", 0, 0, (2,), 1, {"window": (4,)})]) + with pytest.raises(ValueError, match="non-finite"): + runner._run_transaction(step, torch.ones(1, 4, dtype=torch.bfloat16)) + assert ledger.owners == {} and ("reset", "a") in events + + +def test_result_does_not_alias_reused_adapter_output(): + scratch = torch.arange(16, dtype=torch.float32).view(1, 16) + runner, _ = make_runner(output=lambda *a: scratch) + step = ForwardStep("decode", (RequestSlice("a", 0, 0, 4, (3,), 4, {}),), 1) + result = runner._finish_step(LayerState(object(), object(), "tp_local_token"), step) + scratch.zero_() + assert result.logits[0, -1].item() == 15 From e09ce2030868ac6e04760b85ff3138e38a6936e1 Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Wed, 23 Sep 2026 10:13:34 +0800 Subject: [PATCH 10/78] feat(v41): connect model loading and shared serving lifecycle --- docs/developer-guide/deepseek-v41-entry.md | 133 ++++++++--- pypto_serving/cli/main.py | 81 ++++++- pypto_serving/model/deepseek_v41/composite.py | 11 + .../model/deepseek_v41/model_loader.py | 87 ++++++++ .../model/deepseek_v41/npu_executor.py | 7 +- .../model/deepseek_v41/npu_runner.py | 9 +- pypto_serving/model/model_loader.py | 15 +- .../serving/server/serving_worker.py | 5 + tests/unit/model/deepseek_v41/test_entry.py | 11 +- .../deepseek_v41/test_framework_lifecycle.py | 206 ++++++++++++++++++ tests/unit/model/deepseek_v41/test_runner.py | 14 +- .../model/deepseek_v41/test_serving_entry.py | 187 ++++++++++++++++ 12 files changed, 710 insertions(+), 56 deletions(-) create mode 100644 pypto_serving/model/deepseek_v41/model_loader.py create mode 100644 tests/unit/model/deepseek_v41/test_framework_lifecycle.py create mode 100644 tests/unit/model/deepseek_v41/test_serving_entry.py diff --git a/docs/developer-guide/deepseek-v41-entry.md b/docs/developer-guide/deepseek-v41-entry.md index 57bb135a..2454e2b5 100644 --- a/docs/developer-guide/deepseek-v41-entry.md +++ b/docs/developer-guide/deepseek-v41-entry.md @@ -1,10 +1,11 @@ # DeepSeek V4.1 text entry -V4.1 integration currently provides model identification, metadata validation, -local text tokenization and selective CPU weight loading. Configuration and -tokenization do not read checkpoint tensors. None of these APIs allocate NPU -resources or enable generation. Both the model loader and serving CLI reject -V4.1 execution until its executor is integrated, instead of treating it as Qwen. +V4.1 integration provides metadata loading, local text tokenization, selective +CPU weight loading and a serving framework for explicit lib composite bindings. +The model loader registers metadata without opening checkpoint payloads. The +CLI routes V4.1 to its own executor, but rejects execution before allocating +devices until `load_composite_bindings()` supplies a verified adapter. The +default adapter is an explicit placeholder; this is not an executable M0 model. ```python from pypto_serving.model.deepseek_v41.config import load_text_config @@ -36,14 +37,13 @@ and decoding but is not a full-checkpoint tokenizer validation. 1. Model identification, text config and tokenizer (implemented). 2. Selective checkpoint loading, weight formats and shard contracts (implemented on CPU). -3. Executor/Runner preparation and composite contract verification (metadata planning implemented; - device registration blocked on composite integration). -4. Embedding and initial residual/pre-mix state at the selected composite boundary. -5. SWA, C2A and C1A prefill composite entries and cache state. -6. Decode composite entries and prefill-to-decode transitions. -7. Cross-layer state and TP/EP composition, then the full backbone. -8. Final HC/norm, LM head and greedy decode loop. -9. Scheduler/HTTP lifecycle, recovery, memory observations and M0 acceptance. +3. Executor/Runner lifecycle and explicit composite contract (framework implemented). +4. Bounded CPU embedding rows from TP shards (implemented); device residual/pre-mix initialization is an adapter callback. +5. Chunked prefill and request/cache ownership (framework implemented); numerical composites remain adapter callbacks. +6. Decode continuity and prefill-to-decode transitions (framework implemented). +7. Rank-owned weights and all backbone layers (dispatch implemented); device TP/EP execution remains in the adapter. +8. Final HC/norm/head boundary and shared greedy sampling (framework implemented); numerical head remains an adapter callback. +9. Model loader, CLI, scheduler metadata and request release (framework implemented); HTTP/device M0 acceptance remains pending. Engram, vision and speculative decoding are deferred. Optional metadata for these modules may be present in config.json; this stage does not initialize or @@ -53,7 +53,7 @@ Engram-disabled scope, not claim equality with the complete official model. Development starts from upstream main. Existing experimental V4.1 code can be reused selectively with tests; its backend and runtime are not prerequisites. Kernel stages should track current pypto-lib interfaces and record the revision -used for validation. This metadata/tokenizer stage does not change the lib pin. +used for validation. This integration does not change the lib pin. ## Selective weight loading @@ -112,9 +112,9 @@ checks are not real-checkpoint numerical inference or NPU acceptance. ## Executor/Runner preparation -Stage numbers follow serving issue #240. `V41ExecutionPlan` implements the -CPU preparation portion of stage 3; it is not a registered executor, device -runner or substitute inference backend. +Stage numbers follow serving issue #240. `V41ExecutionPlan` describes the +checkpoint and rank ownership without opening devices. The executor uses one +plan per logical rank and dispatches complete layer entries through its runner. ```python from pypto_serving.model.deepseek_v41.execution_plan import RankPlacement, V41ExecutionPlan @@ -139,22 +139,89 @@ selection; Reuse layers consume that selection without loading producer weights. C1A candidates must address the same KV producer as their consumers. These layer references are not scheduler page IDs or request-global state. -The execution integration will follow upstream V4's `PyptoExecutor` and -`ModelRunner` lifecycle: the executor validates lib contracts and compiles -composites; the runner owns uploaded weights, buffers and dispatch completion; -the scheduler owns request/page reservations. Do not inherit the generic -dense K/V allocator for V4.1's window/compressed/index/pending-state pools. -Request and DP identity must scope every pool. Active token counts, padding, -physical page layouts and buffer reuse must come from the selected composite -contract, not from the V4 constants or a tensor capacity alone. - -At inspected lib revision `4c3eab2`, complete-layer coverage and the routed -weight contract are not ready for this registration. The current prefill -layer takes FP8 routed weights, whereas this loader preserves packed FP4; -the decode block factory raises `NotImplementedError`. Independent MoE FP4 -support does not establish full-layer compatibility. No device runner is -registered, generic fallback enabled, or FP4 model expanded to work around -these constraints. Stage 3 remains partially complete pending these contracts. +`DeepSeekV41PyptoExecutor` and `V41ModelRunner` follow the shared executor +lifecycle, with one synchronous collective session for TP4/DP2/EP8 on A5. +The runner does not inherit the generic dense K/V allocator. The scheduler +owns page reservations; the adapter owns device pools, compilation, uploads +and completion. Physical device IDs are independent of logical ranks. + +## Composite adapter contract + +`CompositeBindings` is a serving-owned integration boundary, not a declaration +that current lib functions already have these signatures. A concrete adapter +must identify its tested lib revision and implement every operation below. +All callbacks may enqueue device work; `wait(resources)` must establish +completion across every participating rank before host code reuses storage. + +| Operation | Responsibility | +| --- | --- | +| `allocate(plan, device_ids, runtime, build_options)` | Return `(resources, num_pages)` with positive scheduler page capacity. Honor the worker's platform, build directory and compile-cache choice. Allocate the declared cache groups and compressor state, enforce memory budgets, and prepare global embedding/HC/norm/head resources as needed. Clean up partially created resources if allocation raises before returning. | +| `initialize(embeddings, step, resources)` | Upload packed CPU BF16 `[active_tokens, hidden_size]` and initialize the production residual and delayed pre-mix state. No fixture pre-mix value is assumed. | +| `prepare_weights(rank_plans, layer, resources)` | Bind each rank's TP projections and EP experts, retaining packed FP4. Own bounded staging or persistent residency and reuse producer weights correctly. | +| `entries[(phase, mode)](layer, state, step, resources, weights)` | Execute the complete Attention + FFN layer for prefill/decode, preserving the declared residual/pre-mix layout and returning `LayerState`. | +| `output(state, step, resources)` | Execute final HC/norm/head and return CPU float logits `[requests, vocabulary]` in original request order, including a row for each prefill chunk. Only terminal prompt chunks are sampled by the shared worker. | +| `reset_request(resources, key, owner)` | Clear every cache/state location owned by this request. `owner` includes its DP partition, stable compressor block ID, committed length and page tables, including pages touched by a failed step. | +| `wait(resources)` / `close(resources)` | Establish collective completion / release all session resources. A failed wait must never be treated as permission to reuse buffers. | + +The runner carries opaque `LayerState` between layers. Both replicated and +TP-local-token residual layouts are supported when all callbacks agree; it +does not insert an unconditional residual AllGather. Every collective must +include the two DP partitions even when one has no active requests. The +adapter lowers logical positions, active tokens and page IDs into the exact +lib ABI, including padding and communication-window lifetimes. + +`cache_groups` must describe the same concrete layouts and capacities to the +scheduler and allocator, with two DP partitions. Full-history page tables are +supported by the framework. Rolling page reuse requires a verified lowering +contract and is explicitly rejected for now. Cache payloads must not use a +generic dense K/V substitute. The adapter must bound physical page IDs against +actual allocated pools when a group leaves `num_blocks` unspecified. + +At inspected upstream lib revision `1b8caa4`, packed-FP4 MoE and token-local +decode Attention progress do not yet provide a compatible complete-layer +adapter. Decode `stage="block"` remains disabled; the prefill fixture still +uses the older routed-weight/call contract. Initial residual/pre-mix, complete +prefill/decode, cache allocation/reset and final HC/norm/head remain explicit +integration work. These facts do not block testing the serving state machine, +but they do block real-model generation and M0 numerical acceptance. + +## Request state and serving lifecycle + +`RequestLedger` allocates a stable logical compressor state block per request +and DP partition. This is an ownership ID, not a physical layout: the adapter +must translate it to the selected lib's ring-state block table. Paused or +omitted requests keep their blocks. Batch reordering does not change ownership. +The adapter must provision enough state blocks for the configured request +capacity; no batch row is used as a persistent state address. + +Packed prefill metadata carries both the end of the current chunk (`seq_lens`) +and the original full prompt length (`prompt_lens`). A subsequent chunk must +start at the committed position. Decode requires completed prefill and consumes +exactly one supplied token per request. Lengths are committed only after all +layers and output finish. Failed execution waits, resets affected requests and +releases their ownership; failed completion/reset poisons the session instead +of reusing uncertain state. This does not attempt to roll back in-place device +updates to an earlier token position. + +The existing worker owns sampling, EOS handling, HTTP request lifecycle and +finished-request notifications. Returned logits own their host storage so an +adapter cannot overwrite them while the worker samples. Prefix caching and +asynchronous scheduling are disabled: restoring a prefix also needs matching +compressor state, and concurrent dispatch needs separate mutable state tickets. + +The CLI requires A5, eight distinct device IDs, TP4/DP2/EP8 and 128-row pages. +The shared engine sees one overlapped EP worker; internal TP/DP placement lives +in the rank plan and the two cache partitions. A3 execution, Engram, MTP and +vision are outside this framework's current execution contract. + +## Validation boundary + +Serving tests use small safetensors checkpoints and explicitly named recording +adapters. They verify ownership, metadata, call order, failures and sampling +integration without inventing numerical results for missing production kernels. +Running these CPU tests on an A5 host is not NPU acceptance. Real M0 still needs +the concrete adapter, original weights, matched-reference 8K + 128-token greedy +results, repeated-request cleanup and device memory observations. ## Test command diff --git a/pypto_serving/cli/main.py b/pypto_serving/cli/main.py index 23817261..31ec6c2c 100644 --- a/pypto_serving/cli/main.py +++ b/pypto_serving/cli/main.py @@ -278,12 +278,7 @@ def build_serving_engine_config(args: argparse.Namespace) -> EngineConfig: model_config_data = read_model_config(model_dir) model_family = detect_model_family(model_config_data) if model_family == "deepseek_v41": - from pypto_serving.model.deepseek_v41.config import load_text_config - - load_text_config(model_dir) - raise NotImplementedError( - "V4.1 configuration and tokenizer are supported; serving execution is not integrated yet." - ) + return _build_v41_engine_config(args, model_dir, model_config_data, devices) model_variant = _resolve_model_variant(args) _validate_prefill_chunk_size( model_family, @@ -357,6 +352,78 @@ def build_serving_engine_config(args: argparse.Namespace) -> EngineConfig: ) +def _build_v41_engine_config(args, model_dir, raw, devices): + """Resolve the V4.1 worker boundary before starting any device resources. + + Like DSpark placement, TP/DP are internal to one overlapped EP worker. + The scheduler uses two cache partitions instead of creating DP replicas. + """ + from pypto_serving.config.types import KVCacheGroupSpec + from pypto_serving.model.deepseek_v41.composite import ( + MissingCompositeInterface, load_composite_bindings, + ) + from pypto_serving.model.deepseek_v41.config import V41TextConfig + from pypto_serving.model.deepseek_v41.execution_plan import RankPlacement, plan_layers + from pypto_serving.serving.engine.async_engine import EngineConfig + + text = V41TextConfig.from_dict(raw) + if getattr(args, "speculative_config", None) is not None or getattr(args, "num_speculative_tokens", 0): + raise ValueError("V4.1 text serving does not support DSpark or MTP") + if args.platform != "a5": + raise ValueError("V4.1 M0 serving requires --platform a5") + topology = (args.tensor_parallel_size, args.data_parallel_size, args.expert_parallel_size) + if topology != (4, 2, 8) or len(devices) != 8: + raise ValueError("V4.1 serving requires --tp 4 --dp 2 --ep 8 and exactly eight device IDs") + if args.block_size != 128: + raise ValueError("V4.1 serving requires --block-size 128") + if not 0 < args.max_model_len <= text.max_position_embeddings: + raise ValueError("V4.1 --max-model-len must fit the checkpoint position capacity") + if args.max_num_seqs <= 0 or args.max_num_batched_tokens <= 0: + raise ValueError("V4.1 batch and token capacities must be positive") + + parallel = ParallelConfig( + data_parallel_size=1, tensor_parallel_size=1, expert_parallel_size=8, + enable_expert_parallel=True, placement_mode="overlapped", devices=devices, + data_parallel_routing=args.data_parallel_routing, + ) + # This resolver intentionally raises while the lib adapter is missing. + # No default generic KV layout may be substituted for the V4.1 pools. + bindings = load_composite_bindings() + bindings.require(plan_layers(raw), RankPlacement(0)) + groups = tuple(bindings.cache_groups) + if not groups or any(not isinstance(group, KVCacheGroupSpec) for group in groups): + raise MissingCompositeInterface("V4.1 bindings must provide concrete grouped cache specifications") + if any(group.num_partitions != 2 for group in groups): + raise ValueError("V4.1 cache groups must use two logical DP partitions") + if len({group.name for group in groups}) != len(groups): + raise ValueError("V4.1 cache group names must be unique") + runtime = dataclasses.replace( + _build_runtime_config(args), + kv_cache_groups=groups, + requires_homogeneous_prefill_decode=True, + ) + return EngineConfig( + model_id=args.served_model_name or Path(model_dir).name, + model_dir=model_dir, + platform=args.platform, + device_id=devices[0], + device_ids=devices, + parallel_config=parallel, + executor_cls=_executor_cls_for_model_family("deepseek_v41"), + executor_kwargs={"use_compile_cache": args.use_compile_cache}, + runtime_config=runtime, + profile_config=_build_profile_config(args), + max_num_running_reqs=args.max_num_seqs, + max_num_scheduled_tokens=args.max_num_batched_tokens, + long_prefill_token_threshold=args.long_prefill_token_threshold, + # Prefix restore needs both KV pages and matching compressor state. + # Pipelining needs independently owned mutable snapshots and tickets. + enable_prefix_cache=False, + async_scheduling=False, + enable_chunk_prefill=args.enable_chunked_prefill, + ) + + def _build_runtime_config( args: argparse.Namespace, *, @@ -644,6 +711,8 @@ def _warn_deprecated_serving_profile_env(args: argparse.Namespace) -> None: def _executor_cls_for_model_family(model_family: str, *, variant: str = "") -> str: """Map model family metadata to the worker executor class id.""" + if model_family == "deepseek_v41": + return "PyptoDeepSeekV41Executor" if model_family == "deepseek_v4": if variant == "dspark": return "PyptoDeepSeekV4DSparkExecutor" diff --git a/pypto_serving/model/deepseek_v41/composite.py b/pypto_serving/model/deepseek_v41/composite.py index eec9db0d..8eed98b1 100644 --- a/pypto_serving/model/deepseek_v41/composite.py +++ b/pypto_serving/model/deepseek_v41/composite.py @@ -22,6 +22,14 @@ class MissingCompositeInterface(NotImplementedError): """The selected lib revision cannot execute the requested serving segment.""" +@dataclass(frozen=True) +class BuildOptions: + """Worker-selected compilation settings passed unchanged to the adapter.""" + platform: str = "a5" + pypto_build_dir: str = "build_output" + use_compile_cache: bool = False + + @dataclass(frozen=True) class LayerState: """Opaque device state carried between composites, never converted on host.""" @@ -39,6 +47,9 @@ class CompositeBindings: (state, step, resources) and returns host logits in original request order. The adapter owns device uploads and lib ABI binding. Serving never expands packed FP4 or calls the sub-operators of a complete layer. + allocate consumes (plan, device_ids, runtime, BuildOptions), including the + worker's build directory and compile-cache choice, and returns (resources, + num_pages). It also prepares global weights needed by initialize/output. No default implementation fabricates results. A supplied adapter must handle all DP partitions collectively, including empty partitions, and retain input diff --git a/pypto_serving/model/deepseek_v41/model_loader.py b/pypto_serving/model/deepseek_v41/model_loader.py new file mode 100644 index 00000000..4e6f8503 --- /dev/null +++ b/pypto_serving/model/deepseek_v41/model_loader.py @@ -0,0 +1,87 @@ +# Copyright (c) PyPTO Contributors. +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of +# CANN Open Software License Agreement Version 2.0 (the "License"). +# Please refer to the License for details. You may not use this file except in compliance with the License. +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. +# See LICENSE in the root of the software repository for the full text of the License. +# ----------------------------------------------------------------------------------------------------------- +"""Metadata-only registration of the original V4.1 text checkpoint. + +The executor owns device readiness. Loading metadata does not assert that the +selected lib revision has the composite entries needed to execute the model. +""" + +import json +from pathlib import Path + +import torch + +from pypto_serving.config.types import LoadedModel, ModelConfig, RuntimeConfig, RuntimeModel +from pypto_serving.model.model_family import is_deepseek_v41_config, read_model_config +from pypto_serving.model.model_loader import SafetensorsDirectoryLoader, _build_layer_specs +from pypto_serving.model.tokenizer import load_tokenizer +from .weight_loader import V41WeightLoader + + +class DeepSeekV41DirectoryLoader(SafetensorsDirectoryLoader): + """Validate text metadata and the index without opening weight payloads.""" + + format_names = ("deepseek_v41", "deepseek-v41", "dsv41") + + def _recognises(self, model_path: Path) -> bool: + return is_deepseek_v41_config(read_model_config(model_path)) + + def load(self, request) -> LoadedModel: + model_path = Path(request.model_dir).resolve() + weights = V41WeightLoader(model_path) + text = weights.config + raw = json.loads((model_path / "config.json").read_text(encoding="utf-8")) + values = raw["text_config"] + tokenizer = load_tokenizer( + model_path, trust_remote_code=bool(request.loader_options.get("trust_remote_code", False)), + ) + config = ModelConfig( + model_id=request.model_id, + architecture="DeepseekV41ForCausalLM", + vocab_size=text.vocab_size, + hidden_size=text.hidden_size, + intermediate_size=int(values["moe_intermediate_size"]), + num_hidden_layers=text.num_hidden_layers, + num_attention_heads=text.num_attention_heads, + num_key_value_heads=int(values.get("num_key_value_heads", 1)), + head_dim=text.head_dim, + max_position_embeddings=text.max_position_embeddings, + rms_norm_eps=text.rms_norm_eps, + rope_theta=float(values["rope_theta"]), + bos_token_id=text.bos_token_id, + eos_token_id=text.eos_token_id, + pad_token_id=text.pad_token_id, + torch_dtype="bfloat16", + ) + runtime = request.runtime_config or RuntimeConfig( + page_size=128, max_seq_len=min(text.max_position_embeddings, 8192 + 128), + ) + if runtime.device != "cpu": + raise ValueError("V4.1 metadata and checkpoint preparation require runtime.device='cpu'") + if not 0 < runtime.max_seq_len <= text.max_position_embeddings: + raise ValueError("V4.1 max_seq_len must fit the checkpoint position capacity") + placeholder = torch.empty((0, text.hidden_size), dtype=torch.bfloat16) + model = RuntimeModel( + config=config, + runtime=runtime, + embed_tokens=placeholder, + final_norm_weight=torch.empty(0, dtype=torch.bfloat16), + lm_head=placeholder, + extra={ + "family": "deepseek_v41", + "checkpoint_format": "fp8-ue8m0-packed-fp4", + "config_data": raw, + "model_dir": str(model_path), + "compress_ratios": text.compress_ratios, + }, + ) + return LoadedModel( + model_id=request.model_id, model_dir=str(model_path), config=config, + tokenizer=tokenizer, layer_specs=_build_layer_specs(config), runtime_model=model, + ) diff --git a/pypto_serving/model/deepseek_v41/npu_executor.py b/pypto_serving/model/deepseek_v41/npu_executor.py index 14aa9380..38fb5198 100644 --- a/pypto_serving/model/deepseek_v41/npu_executor.py +++ b/pypto_serving/model/deepseek_v41/npu_executor.py @@ -9,7 +9,7 @@ """Executor routing for V4.1; devices are opened only after contract validation.""" from pypto_serving.model.common.executor.executor import ModelExecutor -from .composite import MissingCompositeInterface, load_composite_bindings +from .composite import BuildOptions, MissingCompositeInterface, load_composite_bindings from .execution_plan import RankPlacement, V41ExecutionPlan from .npu_runner import V41ModelRunner @@ -38,7 +38,10 @@ def register_model(self, model_id, record): if record.runtime.kv_cache_groups != bindings.cache_groups: raise ValueError("scheduler and composite cache group contracts differ") plan = V41ExecutionPlan(record.runtime_model.extra["model_dir"], RankPlacement(0)) - runner = V41ModelRunner(plan, bindings, device_ids=self.device_ids, runtime=record.runtime) + runner = V41ModelRunner( + plan, bindings, device_ids=self.device_ids, runtime=record.runtime, + build_options=BuildOptions(self.platform, self.pypto_build_dir, self.use_compile_cache), + ) try: pages = runner.preflight() except Exception: diff --git a/pypto_serving/model/deepseek_v41/npu_runner.py b/pypto_serving/model/deepseek_v41/npu_runner.py index 4353db10..2be780ae 100644 --- a/pypto_serving/model/deepseek_v41/npu_runner.py +++ b/pypto_serving/model/deepseek_v41/npu_runner.py @@ -7,7 +7,7 @@ # See LICENSE in the root of the software repository for the full text of the License. # ----------------------------------------------------------------------------------------------------------- """V4-style runner lifecycle for explicit V4.1 composite bindings.""" -from .composite import CompositeBindings, LayerState +from .composite import BuildOptions, CompositeBindings, LayerState from .input_preparation import lookup_token_embeddings import torch from pypto_serving.config.types import PrefillResult, DecodeResult @@ -23,7 +23,8 @@ class V41ModelRunner: This follows the shared runner lifecycle but deliberately does not inherit its generic K/V allocator: V4.1 pools include index and compressor state. """ - def __init__(self, plan: V41ExecutionPlan, bindings: CompositeBindings, *, device_ids, runtime): + def __init__(self, plan: V41ExecutionPlan, bindings: CompositeBindings, *, device_ids, runtime, + build_options: BuildOptions = BuildOptions()): self.plan, self.bindings = plan, bindings self.device_ids = tuple(device_ids) if len(self.device_ids) != plan.placement.ep_size or len(set(self.device_ids)) != len(self.device_ids): @@ -32,6 +33,7 @@ def __init__(self, plan: V41ExecutionPlan, bindings: CompositeBindings, *, devic raise ValueError("device IDs must be nonnegative integers") bindings.require(plan.layers, plan.placement) self.runtime = runtime + self.build_options = build_options self.resources = None self.num_pages = None self.closed = False @@ -48,7 +50,8 @@ def preflight(self): if self.resources is not None: return self.num_pages try: - resources, pages = self.bindings.allocate(self.plan, self.device_ids, self.runtime) + resources, pages = self.bindings.allocate( + self.plan, self.device_ids, self.runtime, self.build_options) self.resources = resources if resources is None or type(pages) is not int or pages <= 0: raise ValueError("composite allocator must return resources and a positive page capacity") diff --git a/pypto_serving/model/model_loader.py b/pypto_serving/model/model_loader.py index 8e3be25d..e481e2be 100644 --- a/pypto_serving/model/model_loader.py +++ b/pypto_serving/model/model_loader.py @@ -442,7 +442,11 @@ class ModelLoader: def __init__(self, format_loaders: list[ModelFormatLoader] | None = None) -> None: """Create a loader registry with optional custom format loaders.""" - self._format_loaders = format_loaders or [DeepSeekV4W8A8DirectoryLoader(), HuggingFaceDirectoryLoader()] + from .deepseek_v41.model_loader import DeepSeekV41DirectoryLoader + + self._format_loaders = format_loaders or [ + DeepSeekV41DirectoryLoader(), DeepSeekV4W8A8DirectoryLoader(), HuggingFaceDirectoryLoader(), + ] def register(self, format_loader: ModelFormatLoader) -> None: """Register an additional model format loader.""" @@ -458,13 +462,10 @@ def load( ) -> LoadedModel: """Load a model directory using an explicit or inferred format.""" if is_deepseek_v41_config(read_model_config(model_dir)): - from .deepseek_v41.config import load_text_config + from .deepseek_v41.model_loader import DeepSeekV41DirectoryLoader - load_text_config(model_dir) - raise NotImplementedError( - "V4.1 serving execution is not integrated yet. Use load_text_config() and " - "load_tokenizer() for inspection, or V41WeightLoader for selective CPU weight loading." - ) + if model_format is not None and not DeepSeekV41DirectoryLoader().supports_format(model_format): + raise ValueError("a V4.1 checkpoint requires model_format='deepseek_v41'") request = ModelLoadRequest( model_id=model_id, model_dir=model_dir, diff --git a/pypto_serving/serving/server/serving_worker.py b/pypto_serving/serving/server/serving_worker.py index aa54338b..5383d8a7 100644 --- a/pypto_serving/serving/server/serving_worker.py +++ b/pypto_serving/serving/server/serving_worker.py @@ -194,6 +194,10 @@ def init_device_and_model(self) -> int: return num_pages def _resolve_executor_cls(self): + if self.config.executor_cls == "PyptoDeepSeekV41Executor": + from pypto_serving.model.deepseek_v41.npu_executor import DeepSeekV41PyptoExecutor + + return DeepSeekV41PyptoExecutor if self.config.executor_cls == "PyptoQwen14BExecutor": from pypto_serving.model.qwen.npu_executor import Qwen314BPyptoExecutor @@ -776,6 +780,7 @@ def _batch_prefill( block_ids=block_ids_list, block_ids_by_group=[pr.block_ids_by_group for pr in scheduled], cache_partitions=[pr.cache_partition for pr in scheduled], + prompt_lens=[len(self._req_cache[pr.request_id].prompt_token_ids) for pr in scheduled], ), ) diff --git a/tests/unit/model/deepseek_v41/test_entry.py b/tests/unit/model/deepseek_v41/test_entry.py index bee082ac..184df570 100644 --- a/tests/unit/model/deepseek_v41/test_entry.py +++ b/tests/unit/model/deepseek_v41/test_entry.py @@ -86,17 +86,20 @@ def test_invalid_special_id(raw, value): V41TextConfig.from_dict(raw) -@pytest.mark.parametrize("model_format", [None, "hf", "deepseek_v4"]) +@pytest.mark.parametrize("model_format", ["hf", "deepseek_v4"]) def test_loading_never_falls_through_to_qwen(model_dir, model_format): - with pytest.raises(NotImplementedError, match="serving execution"): + with pytest.raises(ValueError, match="requires model_format='deepseek_v41'"): ModelLoader().load("v41", str(model_dir), model_format=model_format) def test_cli_rejects_execution_before_device_setup(model_dir): from pypto_serving.cli.main import build_parser, build_serving_engine_config - args = build_parser().parse_args(["--model", str(model_dir)]) - with pytest.raises(NotImplementedError, match="serving execution"): + args = build_parser().parse_args([ + "--model", str(model_dir), "--platform", "a5", "--tp", "4", "--dp", "2", "--ep", "8", + "--devices", "0,1,2,3,4,5,6,7", + ]) + with pytest.raises(NotImplementedError, match="verified lib composite bindings"): build_serving_engine_config(args) diff --git a/tests/unit/model/deepseek_v41/test_framework_lifecycle.py b/tests/unit/model/deepseek_v41/test_framework_lifecycle.py new file mode 100644 index 00000000..535980f1 --- /dev/null +++ b/tests/unit/model/deepseek_v41/test_framework_lifecycle.py @@ -0,0 +1,206 @@ +# Copyright (c) PyPTO Contributors. +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of +# CANN Open Software License Agreement Version 2.0 (the "License"). +# Please refer to the License for details. You may not use this file except in compliance with the License. +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. +# See LICENSE in the root of the software repository for the full text of the License. +# ----------------------------------------------------------------------------------------------------------- +"""8K/128 framework regression with a recording adapter, not M0 numerical evidence. + +No model kernels, physical cache allocator, HTTP server or NPU are executed. +The adapter records ownership/reset callbacks and emits synthetic next-token +logits; full-history test pages do not claim the model's rolling SWA cache ABI. +""" + +import json +from pathlib import Path +from types import SimpleNamespace + +import torch + +from pypto_serving.config.types import ( + DecodeBatch, KVCacheGroupSpec, KVCacheSpec, RuntimeConfig, SamplingParams, +) +from pypto_serving.model.common.executor.sampler import Sampler +from pypto_serving.model.deepseek_v41.composite import CompositeBindings, LayerState +from pypto_serving.model.deepseek_v41.config import V41TextConfig +from pypto_serving.model.deepseek_v41.execution_plan import RankPlacement, plan_layers +from pypto_serving.model.deepseek_v41.npu_runner import V41ModelRunner +from pypto_serving.serving.utils.prefill import pack_prefill_batch + + +def test_public_runner_chunked_8k_128_decode_release_and_reuse(): + raw = json.loads((Path(__file__).resolve().parents[4] / + "tests/fixtures/deepseek_v41/config.json").read_text(encoding="utf-8")) + raw["text_config"].update(hidden_size=4, vocab_size=16, max_position_embeddings=8320) + config = V41TextConfig.from_dict(raw) + layers = plan_layers(raw) + assert len(layers) == 40 + assert {layer.mode for layer in layers} == { + "swa", "c2a_full", "c2a_reuse", "c1a_full", "c1a_reindex", "c1a_reuse", + } + groups = ( + KVCacheGroupSpec("window", tuple(range(40)), KVCacheSpec(128, 16), + 65, num_blocks=65, num_partitions=2), + KVCacheGroupSpec("compressed", (2, 8, 14, 20), KVCacheSpec(256, 16, 2), + 33, num_blocks=33, num_partitions=2), + ) + runtime = RuntimeConfig(max_batch_size=2, max_seq_len=8320, max_num_batched_tokens=2048, + max_prefill_tokens_per_request=1024, kv_cache_groups=groups) + table = torch.arange(64, dtype=torch.float32).reshape(16, 4).bfloat16() + resources = {"slots": {}, "pages": {}} + observed_positions, observed_slots, steps, layer_calls, resets = {}, {}, [], [], [] + completion = {"pending": False, "closed": False} + + def embeddings(ids): + return table.index_select(0, ids.reshape(-1)) + + def initialize(values, step, allocated): + assert allocated is resources + assert not completion["pending"] + assert torch.equal(values, embeddings(torch.tensor(step.token_ids))) + steps.append(step) + for request in step.requests: + key = request.request_id + history = observed_positions.setdefault(key, []) + assert request.start == len(history) + history.extend(range(request.start, request.end)) + slot = (request.partition, request.state_slot) + assert observed_slots.setdefault(key, slot) == slot + assert resources["slots"].setdefault(slot, key) == key + for name, pages in request.pages.items(): + for page in pages: + assert resources["pages"].setdefault((request.partition, name, page), key) == key + completion["pending"] = True + return LayerState(object(), object()) + + def prepare(plans, layer, allocated): + assert not completion["pending"] + assert [(plan.placement.tp_rank, plan.placement.dp_rank) for plan in plans] == [ + (rank % 4, rank // 4) for rank in range(8) + ] + return layer.layer_id + + def call(layer, state, step, allocated, weights): + assert not completion["pending"] + assert weights == layer.layer_id + layer_calls.append((step.epoch, step.phase, layer)) + completion["pending"] = True + return state + + def output(state, step, allocated): + assert not completion["pending"] + logits = torch.full((len(step.requests), 16), -1.0) + for row, request in enumerate(step.requests): + logits[row, (request.token_ids[-1] + 1) % 16] = 1.0 + completion["pending"] = True + return logits + + def reset(allocated, key, owner): + assert not completion["pending"] + slot = (owner.partition, owner.slot) + assert resources["slots"].pop(slot) == key + pages = {(owner.partition, name, page) for name, ids in owner.pages.items() for page in ids} + assert pages == {address for address, value in resources["pages"].items() if value == key} + for address in pages: + assert resources["pages"].pop(address) == key + resets.append((key, slot, pages)) + completion["pending"] = True + + def wait(allocated): + completion["pending"] = False + + def close(allocated): + assert not completion["pending"] and not any(resources.values()) + completion["closed"] = True + + bindings = CompositeBindings( + revision="recording-lifecycle-test-only", cache_groups=groups, + entries={(phase, layer.mode): call for phase in ("prefill", "decode") for layer in layers}, + initialize=initialize, output=output, allocate=lambda *args: (resources, 65), + prepare_weights=prepare, reset_request=reset, wait=wait, close=close, + ) + plan = SimpleNamespace(placement=RankPlacement(0), layers=layers, weights=SimpleNamespace(config=config)) + plan.for_rank = lambda rank: SimpleNamespace(placement=RankPlacement(rank)) + runner = V41ModelRunner(plan, bindings, device_ids=(7, 2, 5, 0, 6, 1, 4, 3), runtime=runtime) + model = SimpleNamespace(config=config) + partitions, lengths = {"A": 1, "B": 0, "C": 1}, {"A": 0, "B": 0} + prompts = {"A": [(i + 3) % 16 for i in range(8192)], + "B": [(i + 7) % 16 for i in range(8192)]} + + def pages(end): + return {group.name: list(range((end + group.spec.token_capacity - 1) // group.spec.token_capacity)) + for group in groups} + + def prefill(order, size): + ends = [lengths[key] + size for key in order] + batch = pack_prefill_batch( + request_ids=order, token_chunks=[prompts[key][lengths[key]:end] for key, end in zip(order, ends)], + seq_lens=ends, chunk_starts=[lengths[key] for key in order], + prompt_lens=[len(prompts[key]) for key in order], device="cpu", embedding_lookup=embeddings, + cache_partitions=[partitions[key] for key in order], + block_ids_by_group=[pages(end) for end in ends], + ) + result = runner.run_prefill(model, batch) + lengths.update(zip(order, ends)) + return result + + # A pauses while B advances, then B pauses while A catches up. Both retain + # their state and partition-local page IDs despite changing packed row order. + orders = [("A", "B"), ("B",), ("B", "A"), ("A",)] + orders += [("B", "A") if i % 2 else ("A", "B") for i in range(5)] + for order in orders: + omitted = {key: owner.length for key, owner in (runner.ledger.owners.items() if runner.ledger else []) + if key not in order} + result = prefill(order, 1024) + assert all(runner.ledger.owners[key].length == length for key, length in omitted.items()) + assert lengths == {"A": 8192, "B": 8192} + assert steps[-1].terminal_prefill == (True, True) + assert not any(terminal for step in steps[:-1] for terminal in step.terminal_prefill) + + sampler, params = Sampler(), SamplingParams(temperature=0.0) + next_token = {key: sampler.sample(result.logits[row], params, key) for row, key in enumerate(order)} + generated = {key: [] for key in prompts} + for turn in range(128): + order = ("B", "A") if turn % 2 else ("A", "B") + ids = torch.tensor([[next_token[key]] for key in order]) + ends = [lengths[key] + 1 for key in order] + batch = DecodeBatch( + request_ids=list(order), token_ids=ids, hidden_states=embeddings(ids), + seq_lens=torch.tensor(ends), cache_partitions=[partitions[key] for key in order], + block_ids_by_group=[pages(end) for end in ends], + ) + result = runner.run_decode(model, batch) + for row, key in enumerate(order): + generated[key].append(next_token[key]) + next_token[key] = sampler.sample(result.logits[row], params, key) + lengths.update(zip(order, ends)) + for key in prompts: + assert generated[key] == [(prompts[key][-1] + i + 1) % 16 for i in range(128)] + assert observed_positions[key] == list(range(8320)) + assert runner.ledger.owners[key].length == 8320 + assert observed_slots == {"A": (1, 0), "B": (0, 0)} + assert len(layer_calls) == len(steps) * 40 + for index, step in enumerate(steps): + assert layer_calls[index * 40:(index + 1) * 40] == [ + (step.epoch, step.phase, layer) for layer in layers + ] + assert [step.epoch for step in steps] == list(range(1, len(steps) + 1)) + + runner.release_finished_requests(["A", "A"]) + assert [event[0] for event in resets] == ["A"] + assert "A" not in runner.ledger.owners and runner.ledger.owners["B"].length == 8320 + assert resets[0][2] == {(1, name, page) for name, ids in pages(8320).items() for page in ids} + assert set(resources["pages"].values()) == {"B"} + prompts["C"], lengths["C"] = [4, 5, 6], 0 + prefill(("C",), 3) + assert observed_slots["C"] == observed_slots["A"] + assert observed_positions["C"] == [0, 1, 2] + assert resources["pages"][(1, "window", 0)] == "C" + assert resources["pages"][(0, "window", 0)] == "B" + runner.release_finished_requests(["C", "B"]) + assert [event[0] for event in resets] == ["A", "C", "B"] + assert not runner.ledger.owners and not any(resources.values()) + runner.close() + assert completion["closed"] diff --git a/tests/unit/model/deepseek_v41/test_runner.py b/tests/unit/model/deepseek_v41/test_runner.py index 883aa877..a2603cd6 100644 --- a/tests/unit/model/deepseek_v41/test_runner.py +++ b/tests/unit/model/deepseek_v41/test_runner.py @@ -10,7 +10,7 @@ from types import SimpleNamespace import pytest -from pypto_serving.model.deepseek_v41.composite import CompositeBindings, MissingCompositeInterface +from pypto_serving.model.deepseek_v41.composite import BuildOptions, CompositeBindings, MissingCompositeInterface from pypto_serving.model.deepseek_v41.execution_plan import LayerPlan, RankPlacement from pypto_serving.model.deepseek_v41.npu_runner import V41ModelRunner @@ -49,6 +49,18 @@ def test_lifecycle_waits_before_free_and_is_idempotent(): runner.preflight() +def test_allocator_receives_build_options(): + options = BuildOptions(pypto_build_dir="worker-7-build", use_compile_cache=True) + received = [] + runner = V41ModelRunner( + plan(), bindings([], allocate=lambda *args: (received.append(args[-1]) or object(), 12)), + device_ids=range(8), runtime=None, build_options=options, + ) + runner.preflight() + assert received == [options] + runner.close() + + def test_bad_allocator_result_is_closed(): events = [] runner = V41ModelRunner(plan(), bindings(events, allocate=lambda *a: (object(), 0)), diff --git a/tests/unit/model/deepseek_v41/test_serving_entry.py b/tests/unit/model/deepseek_v41/test_serving_entry.py new file mode 100644 index 00000000..aa081916 --- /dev/null +++ b/tests/unit/model/deepseek_v41/test_serving_entry.py @@ -0,0 +1,187 @@ +# Copyright (c) PyPTO Contributors. +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of +# CANN Open Software License Agreement Version 2.0 (the "License"). +# Please refer to the License for details. You may not use this file except in compliance with the License. +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. +# See LICENSE in the root of the software repository for the full text of the License. +# ----------------------------------------------------------------------------------------------------------- +"""Serving registration and lifecycle contracts without device execution.""" + +from dataclasses import replace +import json +from pathlib import Path +from types import SimpleNamespace + +import pytest + +from pypto_serving.cli.main import build_parser, build_serving_engine_config +from pypto_serving.config.types import KVCacheGroupSpec, KVCacheSpec, RuntimeConfig +from pypto_serving.model.deepseek_v41.composite import CompositeBindings, MissingCompositeInterface +from pypto_serving.model.deepseek_v41.execution_plan import plan_layers +from pypto_serving.model.deepseek_v41.weight_spec import backbone_weight_specs +from pypto_serving.model.model_loader import ModelLoader + + +FIXTURE = Path(__file__).resolve().parents[4] / "tests/fixtures/deepseek_v41/config.json" + + +@pytest.fixture +def metadata_checkpoint(tmp_path, monkeypatch): + raw = json.loads(FIXTURE.read_text(encoding="utf-8")) + raw["text_config"].update( + hidden_size=256, vocab_size=256, num_hidden_layers=1, num_attention_heads=8, + head_dim=64, q_lora_rank=256, o_lora_rank=128, o_groups=4, n_routed_experts=8, + moe_intermediate_size=256, index_n_heads=8, index_head_dim=32, + kv_source_layer_ids=[0], index_source_layer_ids=[0], compress_ratios=[2], + ) + (tmp_path / "config.json").write_text(json.dumps(raw), encoding="utf-8") + # There is deliberately no shard file: metadata loading must never open one. + index = {name: "missing-payload.safetensors" for name in backbone_weight_specs(raw)} + index["vision.weight"] = "also-missing.safetensors" + (tmp_path / "model.safetensors.index.json").write_text( + json.dumps({"weight_map": index}), encoding="utf-8", + ) + monkeypatch.setattr( + "pypto_serving.model.deepseek_v41.model_loader.load_tokenizer", + lambda *args, **kwargs: SimpleNamespace(bos_token_id=0, eos_token_id=1, pad_token_id=2), + ) + return tmp_path, raw + + +@pytest.mark.parametrize("model_format", [None, "deepseek_v41", "deepseek-v41", "dsv41"]) +def test_metadata_loader_registers_without_weight_payload(metadata_checkpoint, model_format): + path, raw = metadata_checkpoint + runtime = RuntimeConfig(page_size=128, max_seq_len=8320) + loaded = ModelLoader().load("text", str(path), runtime, model_format=model_format) + assert loaded.config.architecture == "DeepseekV41ForCausalLM" + assert loaded.config.head_dim == 64 # Not hidden_size / num_attention_heads. + assert loaded.runtime_model.extra["family"] == "deepseek_v41" + assert loaded.runtime_model.extra["config_data"] == raw + assert loaded.runtime_model.runtime is runtime + assert loaded.runtime_model.embed_tokens.numel() == 0 + assert loaded.runtime_model.lm_head.numel() == 0 + assert len(loaded.layer_specs) == 1 + + +def test_loader_keeps_unsupported_quantization_closed(metadata_checkpoint): + path, raw = metadata_checkpoint + raw["quantization_config"]["quant_method"] = "compressed-tensors" + (path / "config.json").write_text(json.dumps(raw), encoding="utf-8") + with pytest.raises(ValueError, match="FP8 32x32"): + ModelLoader().load("text", str(path)) + + +def _args(path, *extra): + return build_parser().parse_args([ + "--model", str(path), "--platform", "a5", "--tp", "4", "--dp", "2", "--ep", "8", + "--devices", "0,1,2,3,4,5,6,7", "--max-model-len", "8320", *extra, + ]) + + +@pytest.fixture +def binding_metadata(metadata_checkpoint, monkeypatch): + _, raw = metadata_checkpoint + + def unreachable(*args, **kwargs): + raise AssertionError("configuration must not dispatch or allocate") + + groups = (KVCacheGroupSpec( + "test_cache", (0,), KVCacheSpec(block_size=128, page_size_bytes=128), + max_blocks_per_seq=65, num_partitions=2, + ),) + bindings = CompositeBindings( + revision="test-contract-only", + entries={(phase, layer.mode): unreachable for phase in ("prefill", "decode") + for layer in plan_layers(raw)}, + initialize=unreachable, output=unreachable, allocate=unreachable, + prepare_weights=unreachable, reset_request=unreachable, wait=unreachable, + close=unreachable, cache_groups=groups, + ) + monkeypatch.setattr( + "pypto_serving.model.deepseek_v41.composite.load_composite_bindings", lambda: bindings, + ) + return bindings + + +def test_cli_routes_one_ep_worker_with_two_cache_partitions(metadata_checkpoint, binding_metadata): + path, _ = metadata_checkpoint + config = build_serving_engine_config(_args(path)) + assert config.executor_cls == "PyptoDeepSeekV41Executor" + assert config.parallel_config.num_replicas == 1 + assert config.worker_device_ids() == tuple(range(8)) + assert config.parallel_config.tensor_parallel_size == 1 + assert config.parallel_config.data_parallel_size == 1 + assert config.parallel_config.expert_parallel_size == 8 + assert config.runtime_config.kv_cache_groups == binding_metadata.cache_groups + assert config.runtime_config.requires_homogeneous_prefill_decode + assert not config.enable_prefix_cache + assert not config.resolve_async_scheduling() + + +@pytest.mark.parametrize("extra,message", [ + (("--platform", "a2a3"), "platform a5"), + (("--tp", "2"), "--tp 4 --dp 2 --ep 8"), + (("--dp", "1"), "--tp 4 --dp 2 --ep 8"), + (("--devices", "0,1,2,3"), "exactly eight"), + (("--block-size", "64"), "block-size 128"), + (("--num-speculative-tokens", "1"), "DSpark or MTP"), + (("--speculative-config", '{"method":"dspark"}'), "DSpark or MTP"), +]) +def test_cli_rejects_wrong_topology_before_binding(metadata_checkpoint, monkeypatch, extra, message): + path, _ = metadata_checkpoint + + def forbidden(): + raise AssertionError("invalid settings must fail before binding resolution") + + monkeypatch.setattr("pypto_serving.model.deepseek_v41.composite.load_composite_bindings", forbidden) + with pytest.raises(ValueError, match=message): + build_serving_engine_config(_args(path, *extra)) + + +def test_cli_cannot_fall_back_to_generic_kv(metadata_checkpoint, binding_metadata, monkeypatch): + path, _ = metadata_checkpoint + monkeypatch.setattr( + "pypto_serving.model.deepseek_v41.composite.load_composite_bindings", + lambda: replace(binding_metadata, cache_groups=()), + ) + with pytest.raises(MissingCompositeInterface, match="grouped cache"): + build_serving_engine_config(_args(path)) + + +def test_worker_resolves_v41_executor_and_releases_request_state(): + from pypto_serving.model.deepseek_v41.npu_executor import DeepSeekV41PyptoExecutor + from pypto_serving.serving.server.serving_worker import WorkerProcess + + worker = WorkerProcess(SimpleNamespace(executor_cls="PyptoDeepSeekV41Executor"), None, None) + assert worker._resolve_executor_cls() is DeepSeekV41PyptoExecutor + released = [] + worker.executor = SimpleNamespace(release_finished_requests=lambda ids: released.extend(ids)) + worker._req_cache = {"finished": object(), "live": object()} + worker._last_tokens = {"finished": [4], "live": [8]} + worker._release_finished_request_state(["finished"]) + assert released == ["finished"] + assert set(worker._req_cache) == {"live"} + assert worker._last_tokens == {"live": [8]} + + +def test_executor_passes_worker_build_options(metadata_checkpoint, binding_metadata, monkeypatch): + from pypto_serving.model.deepseek_v41.npu_executor import DeepSeekV41PyptoExecutor + + path, _ = metadata_checkpoint + received = [] + monkeypatch.setattr( + "pypto_serving.model.deepseek_v41.npu_runner.V41ModelRunner.preflight", + lambda runner: received.append(runner.build_options) or 4, + ) + record = SimpleNamespace( + runtime=RuntimeConfig(kv_cache_groups=binding_metadata.cache_groups), + runtime_model=SimpleNamespace(extra={"model_dir": str(path)}), + ) + executor = DeepSeekV41PyptoExecutor( + bindings=binding_metadata, pypto_build_dir="worker-7-build", use_compile_cache=True, + ) + assert executor.register_model("text", record) == 4 + assert received[0].platform == "a5" + assert received[0].pypto_build_dir == "worker-7-build" + assert received[0].use_compile_cache From 62c00b22a6560866156e14a7dafb981f5e9f6657 Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Wed, 23 Sep 2026 10:15:11 +0800 Subject: [PATCH 11/78] test(v41): supply explicit greedy sampling parameters --- tests/unit/model/deepseek_v41/test_composite_dispatch.py | 2 +- tests/unit/model/deepseek_v41/test_framework_lifecycle.py | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/tests/unit/model/deepseek_v41/test_composite_dispatch.py b/tests/unit/model/deepseek_v41/test_composite_dispatch.py index 020158ce..34acc7be 100644 --- a/tests/unit/model/deepseek_v41/test_composite_dispatch.py +++ b/tests/unit/model/deepseek_v41/test_composite_dispatch.py @@ -81,7 +81,7 @@ def head(state, step, resources): result = runner._run_transaction(step, torch.ones(2, 4, dtype=torch.bfloat16)) sampler, generated = Sampler(), [] for position in range(2, 6): - token = sampler.sample(result.logits[0], SamplingParams(temperature=0.0), "a") + token = sampler.sample(result.logits[0], SamplingParams(temperature=0.0, top_p=1.0), "a") generated.append(token) step = ledger.begin_decode([("a", 0, position, token, {})]) result = runner._run_transaction(step, torch.ones(1, 4, dtype=torch.bfloat16)) diff --git a/tests/unit/model/deepseek_v41/test_framework_lifecycle.py b/tests/unit/model/deepseek_v41/test_framework_lifecycle.py index 535980f1..1bd3291b 100644 --- a/tests/unit/model/deepseek_v41/test_framework_lifecycle.py +++ b/tests/unit/model/deepseek_v41/test_framework_lifecycle.py @@ -159,7 +159,7 @@ def prefill(order, size): assert steps[-1].terminal_prefill == (True, True) assert not any(terminal for step in steps[:-1] for terminal in step.terminal_prefill) - sampler, params = Sampler(), SamplingParams(temperature=0.0) + sampler, params = Sampler(), SamplingParams(temperature=0.0, top_p=1.0) next_token = {key: sampler.sample(result.logits[row], params, key) for row, key in enumerate(order)} generated = {key: [] for key in prompts} for turn in range(128): From 2a2e1274b46e5ca46f94398c440e57a6243832d1 Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Mon, 28 Sep 2026 01:04:09 +0800 Subject: [PATCH 12/78] Add bounded V4.1 SWA and packed-FP4 MoE device segment --- docs/developer-guide/v41-swa-segment.md | 38 ++++ .../model/deepseek_v41/swa_segment.py | 212 ++++++++++++++++++ tests/unit/test_v41_swa_segment.py | 100 +++++++++ tools/validate_v41_swa_segment.py | 105 +++++++++ 4 files changed, 455 insertions(+) create mode 100644 docs/developer-guide/v41-swa-segment.md create mode 100644 pypto_serving/model/deepseek_v41/swa_segment.py create mode 100644 tests/unit/test_v41_swa_segment.py create mode 100644 tools/validate_v41_swa_segment.py diff --git a/docs/developer-guide/v41-swa-segment.md b/docs/developer-guide/v41-swa-segment.md new file mode 100644 index 00000000..f568b67c --- /dev/null +++ b/docs/developer-guide/v41-swa-segment.md @@ -0,0 +1,38 @@ +# V4.1 bounded SWA segment + +`pypto_serving/model/deepseek_v41/swa_segment.py` adds the concrete half-layer +execution boundary inspected against lib `216456332c2a74d89cca23b7824dab264ce34bff`: + +1. `prefill_swa.make_hc_program(..., epochs=1)` includes mHC, input Norm, + attention, TP communication and mHC post. +2. `moe.l3_moe` includes mHC, Norm, packed-FP4 routed/shared experts and mHC post. +3. The returned FP32 `LayerState` is passed directly to the next SWA layer. + +The shared V4 `KernelCompiler` compiles both entries. `SwaSegment` uses one +persistent `DistributedWorker`, matching V4's runtime ownership. It does not +call small operators or copy intermediate state to CPU. The caller owns +uploaded weights, per-layer caches, positions, shared count metadata and scratch. +The example uses independent caches and tied synthetic weights for two layers. + +Current bound: 16 physical token rows per rank, as fixed by lib `MOE_TOKENS`. +TP4 permits 64 rows per DP group; TP2 permits 32. Counts use contiguous slabs, +including padded ranks. Exceeding this capacity is rejected. Capacity checks and +count conversion do not establish numerical correctness of empty/ragged cases. +Device failures poison the segment; its worker must be closed before recovery. +Do not reset only the request cache and resume an uncertain collective. + +Run the explicit synthetic diagnostic on A5 through the device queue: + +```bash +PYTHONPATH=. python tools/validate_v41_swa_segment.py \ + --lib-root /path/to/current/pypto-lib --tp 2 --devices 0,1,2,3 +``` + +`--compile-only` checks code generation without device execution. The device +smoke checks completion and finite/nonzero outputs, not numerical acceptance. +CPU dispatch tests use a mocked worker and do not establish NPU correctness. + +This segment does not enable `load_composite_bindings()` for complete serving: +checkpoint-to-resident bundle assembly, input initialization, all attention +modes, decode, cache lifecycle and the final output boundary still need adapters +and validation. Engram is excluded. Full TP4/DP2/EP8 8K-to-128 M0 is not claimed. diff --git a/pypto_serving/model/deepseek_v41/swa_segment.py b/pypto_serving/model/deepseek_v41/swa_segment.py new file mode 100644 index 00000000..c73ca502 --- /dev/null +++ b/pypto_serving/model/deepseek_v41/swa_segment.py @@ -0,0 +1,212 @@ +# Copyright (c) PyPTO Contributors. +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of +# CANN Open Software License Agreement Version 2.0 (the "License"). +# Please refer to the License for details. You may not use this file except in compliance with the License. +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. +# See LICENSE in the root of the software repository for the full text of the License. +# ----------------------------------------------------------------------------------------------------------- +"""Bounded device-resident SWA Attention -> FP4 MoE layer composition. + +This is an execution segment, not the complete V41 CLI backend. The owner +allocates/uploads weights, caches and scratch before worker creation, retains +all buffers until completion, and closes the worker on a failed dispatch. +""" +from dataclasses import dataclass +import ctypes +import importlib +from pathlib import Path +import sys + +from .composite import LayerState + +ATTENTION_ARGS = ( + "x_hc incoming_pre_mix hc_attn_fn hc_attn_scale hc_attn_base attn_norm_weight " + "wq_a wq_a_scale q_norm_weight wq_b wq_b_scale wkv wkv_scale kv_norm_weight " + "attn_sink wo_a wo_b wo_b_scale rope_cos rope_sin window_slots window_indices " + "window_cache window_cache_scale output next_pre_mix hidden attn_out num_tokens attention_epoch" +).split() +MOE_ARGS = ( + "x_hc pre_mix hc_ffn_fn hc_ffn_scale hc_ffn_base norm_weight gate_weight correction_bias " + "routed_w1 routed_w1_scale routed_w2 routed_w2_scale routed_w3 routed_w3_scale mxfp4_pair_lut " + "shared_w1 shared_w1_scale shared_w2 shared_w2_scale shared_w3 shared_w3_scale " + "next_pre_mix x_mixed x_next num_tokens moe_epoch" +).split() + + +@dataclass(frozen=True) +class SegmentTopology: + tp: int = 4 + dp: int = 2 + local_capacity: int = 16 + + def __post_init__(self): + if self.tp not in (1, 2, 4, 8) or type(self.tp) is not int: + raise ValueError("unsupported TP size") + if type(self.dp) is not int or self.dp <= 0 or self.world not in (2, 4, 8): + raise ValueError("TP * DP must be a supported EP size") + if type(self.local_capacity) is not int or self.local_capacity <= 0: + raise ValueError("local capacity must be positive") + + @property + def world(self): + return self.tp * self.dp + + @property + def capacity(self): + return self.tp * self.local_capacity + + def counts(self, group_counts): + """Contiguous SP slabs; padded/empty ranks still participate in collectives.""" + if len(group_counts) != self.dp or any( + type(n) is not int or not 0 <= n <= self.capacity for n in group_counts + ): + raise ValueError("one bounded active-token count is required per DP group") + attention = tuple(n for n in group_counts for _ in range(self.tp)) + local = tuple(max(0, min(self.local_capacity, n - r * self.local_capacity)) + for n in group_counts for r in range(self.tp)) + return attention, local + + +def load_segment_modules(lib_root, topology): + """Use V4's explicit lib path and import-time topology, in a model worker. + + Never reload an already imported model with another topology/revision. + Lib currently captures dimensions from argv when modules are imported. + """ + root = Path(lib_root).resolve() + package = "models.deepseek_v4_1_flash" + existing = sys.modules.get(package + ".config") + expected = root / "models/deepseek_v4_1_flash/config.py" + if existing is not None and Path(existing.__file__).resolve() != expected: + raise RuntimeError("V41 lib from another checkout is already imported; use a fresh worker") + old_argv, old_path = sys.argv[:], sys.path[:] + try: + sys.path.insert(0, str(root)) + sys.argv = [old_argv[0], "--tp", str(topology.tp), "--ep", str(topology.world)] + config = importlib.import_module(package + ".config") + if Path(config.__file__).resolve() != expected: + raise RuntimeError("import resolved to a different lib checkout") + if (config.TP_SIZE, config.EP_SIZE, config.MOE_TOKENS) != ( + topology.tp, topology.world, topology.local_capacity + ): + raise ValueError("lib topology/MOE_TOKENS disagrees with the segment capacity") + attention = importlib.import_module(package + ".prefill_swa") + moe = importlib.import_module(package + ".moe") + if moe.SKIP_SHARED_TEST or moe.SKIP_TRANSPORT_TEST: + raise ValueError("cannot run serving with lib test-only MoE bypasses") + return attention, moe + finally: + sys.argv, sys.path = old_argv, old_path + + +def compile_segment(compiler, lib_root, topology): + """Compile the two existing lib composites with the shared V4 compiler.""" + import pypto.language as pl + + attention, moe = load_segment_modules(lib_root, topology) + swa = compiler.compile( + "v41_swa_segment", attention.make_hc_program(topology.capacity, topology.world, epochs=1), + attention_epoch=pl.RUNTIME, + ) + ffn = compiler.compile("v41_moe_segment", moe.l3_moe, moe_epoch=pl.RUNTIME) + return swa, ffn + + +class SwaSegment: + """Synchronous half-layer dispatcher on one persistent DistributedWorker. + + Metadata must be shared CPU tensors created before worker startup: + attention_counts [world, 1] and moe_counts [world], both int32. + Per-layer argument maps contain all ABI tensors except input state, counts + and epoch. They own separate output/scratch/cache buffers. No host copies + occur between half-layers or between consecutive calls to run_layer(). + """ + + def __init__(self, worker, programs, topology, attention_counts, moe_counts, run_config): + import torch + + for value, shape in ((attention_counts, (topology.world, 1)), + (moe_counts, (topology.world,))): + if (not isinstance(value, torch.Tensor) or value.device.type != "cpu" + or tuple(value.shape) != shape or value.dtype != torch.int32 or not value.is_shared()): + raise ValueError("counts must be shared CPU int32 metadata allocated before worker startup") + self.worker, self.programs, self.topology = worker, programs, topology + self.attention_counts, self.moe_counts = attention_counts, moe_counts + self.run_config = run_config + self._epoch = 0 + self._failed = False + + def _check_state(self, state): + import torch + from pypto.runtime import StackedDeviceTensor + + if state.layout != "tp_local_token": + raise ValueError("SWA segment requires contiguous TP-local token state") + prefix = (self.topology.world, self.topology.local_capacity) + for value, shape in ((state.residual, (*prefix, 4, 5120)), (state.pre_mix, (*prefix, 4))): + if (not isinstance(value, StackedDeviceTensor) or tuple(value.shape) != shape + or value.dtype != torch.float32 + or tuple(value.worker_ids) != tuple(range(self.topology.world))): + raise ValueError("state must be FP32 device shards with canonical rank placement and capacity") + + def run_layer(self, state, attention, moe, *, group_counts): + """Return MoE outputs directly for the following layer's Attention input. + + Request packing, RoPE, page mapping and cache initialization are owned + by the caller and must use the same contiguous group/slab ordering. + """ + if self._failed: + raise RuntimeError("segment dispatch failed; close this worker before attempting recovery") + self._check_state(state) + group, local = self.topology.counts(group_counts) + a = dict(attention, x_hc=state.residual, incoming_pre_mix=state.pre_mix, + num_tokens=self.attention_counts) + attention_state = LayerState(a["output"], a["next_pre_mix"], "tp_local_token") + m = dict(moe, x_hc=attention_state.residual, pre_mix=attention_state.pre_mix, + num_tokens=self.moe_counts) + result = LayerState(m["x_next"], m["next_pre_mix"], "tp_local_token") + self._check_state(attention_state) + self._check_state(result) + # Reject aliases before launching either kernel, including distinct + # handles pointing at the same allocation. Reuse across layers is fine. + buffers = (state.residual, state.pre_mix, attention_state.residual, + attention_state.pre_mix, result.residual, result.pre_mix) + seen = set() + for buffer in buffers: + for rank, shard in zip(buffer.worker_ids, buffer.shards): + key = (rank, shard.data_ptr) + if key in seen: + raise ValueError("input and half-layer outputs must not alias") + seen.add(key) + epoch = self._epoch + 1 + if epoch >= 2**31: + raise OverflowError("communication epoch exhausted; recreate worker") + a["attention_epoch"] = m["moe_epoch"] = ctypes.c_int32(epoch) + # Resolve all arguments before the first collective: a missing MoE + # weight must not leave an already-mutated Attention cache behind. + a_args = [a[name] for name in ATTENTION_ARGS] + m_args = [m[name] for name in MOE_ARGS] + self._check_weights(m) + for rank, (global_n, local_n) in enumerate(zip(group, local)): + self.attention_counts[rank, 0] = global_n + self.moe_counts[rank] = local_n + self._epoch = epoch + try: + self.worker.run(self.programs[0].compiled, *a_args, config=self.run_config) + self.worker.run(self.programs[1].compiled, *m_args, config=self.run_config) + except BaseException: + self._failed = True + raise + return result + + @staticmethod + def _check_weights(moe): + import torch + from pypto.runtime import StackedDeviceTensor + + for name in ("routed_w1", "routed_w2", "routed_w3", "mxfp4_pair_lut"): + value = moe[name] + expected = torch.int16 if name == "mxfp4_pair_lut" else torch.uint8 + if not isinstance(value, StackedDeviceTensor) or value.dtype != expected: + raise ValueError(f"{name} must be a resident packed-FP4 ABI tensor ({expected})") diff --git a/tests/unit/test_v41_swa_segment.py b/tests/unit/test_v41_swa_segment.py new file mode 100644 index 00000000..97c2e576 --- /dev/null +++ b/tests/unit/test_v41_swa_segment.py @@ -0,0 +1,100 @@ +# Copyright (c) PyPTO Contributors. +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of +# CANN Open Software License Agreement Version 2.0 (the "License"). +# Please refer to the License for details. You may not use this file except in compliance with the License. +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. +# See LICENSE in the root of the software repository for the full text of the License. +# ----------------------------------------------------------------------------------------------------------- +"""CPU dispatch-contract tests; these do not execute NPU kernels.""" +from types import SimpleNamespace +from unittest.mock import Mock + +import pytest +import torch + +from pypto_serving.model.deepseek_v41.composite import LayerState +from pypto_serving.model.deepseek_v41.swa_segment import ( + ATTENTION_ARGS, MOE_ARGS, SegmentTopology, SwaSegment, +) + + +@pytest.mark.parametrize("counts,expected", [ + ([64, 64], (16,) * 8), ([17, 0], (16, 1, 0, 0, 0, 0, 0, 0)), + ([0, 63], (0, 0, 0, 0, 16, 16, 16, 15)), +]) +def test_local_counts(counts, expected): + group, local = SegmentTopology().counts(counts) + assert local == expected + assert group == tuple(n for n in counts for _ in range(4)) + assert sum(local) == sum(counts) + + +@pytest.mark.parametrize("counts", [[65, 0], [-1, 16], [True, 0], [16]]) +def test_reject_counts(counts): + with pytest.raises(ValueError): + SegmentTopology().counts(counts) + + +def fixture(monkeypatch): + # PyPTO is optional for CPU serving tests. The dispatcher shape/type guard + # is independently exercised in the device smoke, not mocked as evidence. + monkeypatch.setattr(SwaSegment, "_check_state", lambda self, state: None) + monkeypatch.setattr(SwaSegment, "_check_weights", lambda self, weights: None) + def buffer(address): + return SimpleNamespace(worker_ids=(0, 1), shards=( + SimpleNamespace(data_ptr=address), SimpleNamespace(data_ptr=address))) + topology = SegmentTopology(tp=1, dp=2) + worker = Mock() + programs = (SimpleNamespace(compiled="attention"), SimpleNamespace(compiled="moe")) + segment = SwaSegment(worker, programs, topology, + torch.zeros(2, 1, dtype=torch.int32).share_memory_(), + torch.zeros(2, dtype=torch.int32).share_memory_(), None) + state = LayerState(buffer(1), buffer(2), "tp_local_token") + a, m = dict.fromkeys(ATTENTION_ARGS), dict.fromkeys(MOE_ARGS) + a.update(output=buffer(3), next_pre_mix=buffer(4)) + m.update(x_next=buffer(5), next_pre_mix=buffer(6)) + return segment, state, a, m + + +def test_device_handle_handoff_and_next_layer(monkeypatch): + segment, state, a, m = fixture(monkeypatch) + first = segment.run_layer(state, a, m, group_counts=[16, 3]) + calls = segment.worker.run.call_args_list + assert calls[0].args[1] is state.residual + assert calls[1].args[1] is a["output"] + assert calls[1].args[2] is a["next_pre_mix"] + assert first.residual is m["x_next"] + assert first.pre_mix is m["next_pre_mix"] + # Ping-pong outputs, retaining the returned handles as next-layer inputs. + m.update(x_next=state.residual, next_pre_mix=state.pre_mix) + segment.run_layer(first, a, m, group_counts=[16, 3]) + assert segment.worker.run.call_args_list[2].args[1] is first.residual + assert segment.worker.run.call_args_list[2].args[-1].value == 2 + assert segment.moe_counts.tolist() == [16, 3] + + +def test_failed_dispatch_poisoned(monkeypatch): + segment, state, a, m = fixture(monkeypatch) + segment.worker.run.side_effect = RuntimeError("device failure") + with pytest.raises(RuntimeError, match="device failure"): + segment.run_layer(state, a, m, group_counts=[16, 16]) + with pytest.raises(RuntimeError, match="close this worker"): + segment.run_layer(state, a, m, group_counts=[16, 16]) + assert segment.worker.run.call_count == 1 + + +def test_validate_both_stages_before_dispatch(monkeypatch): + segment, state, a, m = fixture(monkeypatch) + del m["routed_w3"] + with pytest.raises(KeyError): + segment.run_layer(state, a, m, group_counts=[16, 16]) + segment.worker.run.assert_not_called() + + +def test_alias_rejected_before_dispatch(monkeypatch): + segment, state, a, m = fixture(monkeypatch) + m["x_next"] = state.residual + with pytest.raises(ValueError, match="alias"): + segment.run_layer(state, a, m, group_counts=[16, 16]) + segment.worker.run.assert_not_called() diff --git a/tools/validate_v41_swa_segment.py b/tools/validate_v41_swa_segment.py new file mode 100644 index 00000000..627e640f --- /dev/null +++ b/tools/validate_v41_swa_segment.py @@ -0,0 +1,105 @@ +# Copyright (c) PyPTO Contributors. +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of +# CANN Open Software License Agreement Version 2.0 (the "License"). +# Please refer to the License for details. You may not use this file except in compliance with the License. +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. +# See LICENSE in the root of the software repository for the full text of the License. +# ----------------------------------------------------------------------------------------------------------- +"""Synthetic A5 smoke of serving SWA -> MoE -> next-layer SWA/MoE. + +This is not real-checkpoint validation or numerical acceptance. Lib fixture +builders are used only here, never by the serving segment implementation. +""" +import argparse +from pathlib import Path +from types import SimpleNamespace +import sys + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--lib-root", required=True) + parser.add_argument("--devices", default="0,1,2,3") + parser.add_argument("--tp", type=int, default=2) + parser.add_argument("--build-dir", default="build_output/v41-swa-segment") + parser.add_argument("--compile-only", action="store_true") + args = parser.parse_args() + sys.path.insert(0, str(Path(args.lib_root).resolve())) + import torch + from pypto.ir import DistributedConfig + from pypto.runtime import DistributedWorker, RunConfig + from golden.spec import TensorSpec + from pypto_serving.model.common.compiler.compiler import KernelCompiler + from pypto_serving.model.deepseek_v41.composite import LayerState + from pypto_serving.model.deepseek_v41.swa_segment import ( + SegmentTopology, SwaSegment, compile_segment, load_segment_modules, + ) + + torch.set_num_threads(4) + devices = [int(v) for v in args.devices.split(",")] + if len(devices) % args.tp or len(set(devices)) != len(devices): + raise ValueError("unique devices must form complete TP groups") + topology = SegmentTopology(tp=args.tp, dp=len(devices) // args.tp) + config = RunConfig(platform="a5", distributed_config=DistributedConfig(device_ids=devices), + ring_heap=536870912, ring_task_window=131072, ring_dep_pool=131072) + compiler = KernelCompiler(run_config=config, cache_dir=args.build_dir) + programs = compile_segment(compiler, args.lib_root, topology) + print("COMPILE PASS", flush=True) + if args.compile_only: + return + swa, moe = load_segment_modules(args.lib_root, topology) + fixture = SimpleNamespace(tokens=topology.capacity, requests=1, dp=topology.dp, + seed=11, case="normal", fixture="checkpoint", dp_tokens=None, + epochs=1, bench=False) + + def materialize(specs): + return {s.name: s.create_tensor().contiguous() for s in specs if isinstance(s, TensorSpec)} + + a = materialize(swa.build_hc_specs(fixture)) + m = materialize(moe.build_tensor_specs([16] * topology.world)) + ac = torch.zeros(topology.world, 1, dtype=torch.int32).share_memory_() + mc = torch.zeros(topology.world, dtype=torch.int32).share_memory_() + readback = torch.empty_like(m["x_next"]).share_memory_() + mix_readback = torch.empty_like(m["next_pre_mix"]).share_memory_() + sources = [*a.values(), *m.values()] + with DistributedWorker([p.compiled for p in programs], persistent=True, + inherited_host_tensors=sources, config=config) as worker: + allocations = [] + + def upload(values): + result = {} + for name, value in values.items(): + if name == "num_tokens": + continue + device = worker.alloc_stacked_tensor(value) + allocations.append(device) + result[name] = device + return result + + da, dm = upload(a), upload(m) + state = LayerState(da.pop("x_hc"), da.pop("incoming_pre_mix"), "tp_local_token") + dm.pop("x_hc") + dm.pop("pre_mix") + # Diagnostic uses tied synthetic weights but independent layer caches. + second_a = dict(da) + for name in ("window_cache", "window_cache_scale"): + second_a[name] = worker.alloc_stacked_tensor(a[name]) + allocations.append(second_a[name]) + runner = SwaSegment(worker, programs, topology, ac, mc, config) + first = runner.run_layer(state, da, dm, group_counts=[topology.capacity] * topology.dp) + second_m = dict(dm, x_next=state.residual, next_pre_mix=state.pre_mix) + final = runner.run_layer(first, second_a, second_m, + group_counts=[topology.capacity] * topology.dp) + # Read back only after the full chain; no intermediate host round trip. + worker.copy_stacked_from(final.residual, readback) + worker.copy_stacked_from(final.pre_mix, mix_readback) + assert torch.isfinite(readback).all() and torch.isfinite(mix_readback).all() + assert readback.abs().max() > 0 and mix_readback.abs().max() > 0 + print("DEVICE TWO-LAYER SMOKE PASS (finite/nonzero only; not numerical acceptance)", flush=True) + for value in reversed(allocations): + worker.free_stacked_tensor(value) + + +if __name__ == "__main__": + main() From 323bac6f42f10a48250cd427e24d844f7a758d7b Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Mon, 28 Sep 2026 01:15:59 +0800 Subject: [PATCH 13/78] Fix V4.1 cross-layer communication window lifecycle --- pypto_serving/model/deepseek_v41/composite.py | 3 ++- .../model/deepseek_v41/swa_segment.py | 21 ++++++++++++++++-- tests/unit/test_v41_swa_segment.py | 22 ++++++++++++++++++- tools/validate_v41_swa_segment.py | 11 ++++++---- 4 files changed, 49 insertions(+), 8 deletions(-) diff --git a/pypto_serving/model/deepseek_v41/composite.py b/pypto_serving/model/deepseek_v41/composite.py index 8eed98b1..05eda6f1 100644 --- a/pypto_serving/model/deepseek_v41/composite.py +++ b/pypto_serving/model/deepseek_v41/composite.py @@ -93,7 +93,8 @@ def load_composite_bindings() -> CompositeBindings: """ raise MissingCompositeInterface( "V4.1 serving execution requires verified lib composite bindings: " - "packed-FP4 full-layer prefill/decode for all modes, initial residual/pre_mix, " + "all-mode prefill/decode adapters, initial residual/pre_mix, " "cache allocation/reset/completion and final HC/Norm/LM head. " + "The bounded SWA Attention/MoE segment is available separately; it is not a complete model backend. " "Track pypto-lib #1205, #1275 and #1287; no Torch fallback is enabled." ) diff --git a/pypto_serving/model/deepseek_v41/swa_segment.py b/pypto_serving/model/deepseek_v41/swa_segment.py index c73ca502..4120f864 100644 --- a/pypto_serving/model/deepseek_v41/swa_segment.py +++ b/pypto_serving/model/deepseek_v41/swa_segment.py @@ -113,10 +113,27 @@ def compile_segment(compiler, lib_root, topology): return swa, ffn +def make_segment_worker(programs, run_config, inherited_host_tensors=()): + """Retain both programs' communication windows as in the V4 runner. + + Attention waits for (epoch - 1) * 2 before publishing the next phase. + The runtime's default reset-on-reuse policy would erase that completion + signal and stall the second layer. Each compiled program owns its windows. + """ + from pypto.runtime import DistributedWorker + + return DistributedWorker( + [program.compiled for program in programs], config=run_config, + persistent=True, reset_persistent_windows=False, + inherited_host_tensors=inherited_host_tensors, + ) + + class SwaSegment: """Synchronous half-layer dispatcher on one persistent DistributedWorker. - Metadata must be shared CPU tensors created before worker startup: + Construct the worker with make_segment_worker(). Metadata must be shared + CPU tensors created before worker startup: attention_counts [world, 1] and moe_counts [world], both int32. Per-layer argument maps contain all ABI tensors except input state, counts and epoch. They own separate output/scratch/cache buffers. No host copies @@ -180,7 +197,7 @@ def run_layer(self, state, attention, moe, *, group_counts): raise ValueError("input and half-layer outputs must not alias") seen.add(key) epoch = self._epoch + 1 - if epoch >= 2**31: + if epoch > (2**31 - 1) // 2: raise OverflowError("communication epoch exhausted; recreate worker") a["attention_epoch"] = m["moe_epoch"] = ctypes.c_int32(epoch) # Resolve all arguments before the first collective: a missing MoE diff --git a/tests/unit/test_v41_swa_segment.py b/tests/unit/test_v41_swa_segment.py index 97c2e576..013d5632 100644 --- a/tests/unit/test_v41_swa_segment.py +++ b/tests/unit/test_v41_swa_segment.py @@ -9,13 +9,14 @@ """CPU dispatch-contract tests; these do not execute NPU kernels.""" from types import SimpleNamespace from unittest.mock import Mock +import sys import pytest import torch from pypto_serving.model.deepseek_v41.composite import LayerState from pypto_serving.model.deepseek_v41.swa_segment import ( - ATTENTION_ARGS, MOE_ARGS, SegmentTopology, SwaSegment, + ATTENTION_ARGS, MOE_ARGS, SegmentTopology, SwaSegment, make_segment_worker, ) @@ -98,3 +99,22 @@ def test_alias_rejected_before_dispatch(monkeypatch): with pytest.raises(ValueError, match="alias"): segment.run_layer(state, a, m, group_counts=[16, 16]) segment.worker.run.assert_not_called() + + +def test_retained_window_policy(monkeypatch): + constructor = Mock() + monkeypatch.setitem(sys.modules, "pypto.runtime", SimpleNamespace(DistributedWorker=constructor)) + programs = [SimpleNamespace(compiled="a"), SimpleNamespace(compiled="m")] + make_segment_worker(programs, "config", ["source"]) + constructor.assert_called_once_with( + ["a", "m"], config="config", persistent=True, + reset_persistent_windows=False, inherited_host_tensors=["source"], + ) + + +def test_attention_epoch_cannot_overflow_int32(monkeypatch): + segment, state, a, m = fixture(monkeypatch) + segment._epoch = (2**31 - 1) // 2 + with pytest.raises(OverflowError): + segment.run_layer(state, a, m, group_counts=[16, 16]) + segment.worker.run.assert_not_called() diff --git a/tools/validate_v41_swa_segment.py b/tools/validate_v41_swa_segment.py index 627e640f..8160cdd4 100644 --- a/tools/validate_v41_swa_segment.py +++ b/tools/validate_v41_swa_segment.py @@ -28,12 +28,12 @@ def main(): sys.path.insert(0, str(Path(args.lib_root).resolve())) import torch from pypto.ir import DistributedConfig - from pypto.runtime import DistributedWorker, RunConfig + from pypto.runtime import RunConfig from golden.spec import TensorSpec from pypto_serving.model.common.compiler.compiler import KernelCompiler from pypto_serving.model.deepseek_v41.composite import LayerState from pypto_serving.model.deepseek_v41.swa_segment import ( - SegmentTopology, SwaSegment, compile_segment, load_segment_modules, + SegmentTopology, SwaSegment, compile_segment, load_segment_modules, make_segment_worker, ) torch.set_num_threads(4) @@ -57,14 +57,15 @@ def materialize(specs): return {s.name: s.create_tensor().contiguous() for s in specs if isinstance(s, TensorSpec)} a = materialize(swa.build_hc_specs(fixture)) + print("Attention fixture ready; preparing packed-FP4 expert weights", flush=True) m = materialize(moe.build_tensor_specs([16] * topology.world)) + print("MoE fixture ready", flush=True) ac = torch.zeros(topology.world, 1, dtype=torch.int32).share_memory_() mc = torch.zeros(topology.world, dtype=torch.int32).share_memory_() readback = torch.empty_like(m["x_next"]).share_memory_() mix_readback = torch.empty_like(m["next_pre_mix"]).share_memory_() sources = [*a.values(), *m.values()] - with DistributedWorker([p.compiled for p in programs], persistent=True, - inherited_host_tensors=sources, config=config) as worker: + with make_segment_worker(programs, config, sources) as worker: allocations = [] def upload(values): @@ -96,6 +97,8 @@ def upload(values): worker.copy_stacked_from(final.pre_mix, mix_readback) assert torch.isfinite(readback).all() and torch.isfinite(mix_readback).all() assert readback.abs().max() > 0 and mix_readback.abs().max() > 0 + assert not torch.equal(readback, a["x_hc"]), "residual was not updated" + assert not torch.equal(mix_readback, a["incoming_pre_mix"]), "pre_mix was not updated" print("DEVICE TWO-LAYER SMOKE PASS (finite/nonzero only; not numerical acceptance)", flush=True) for value in reversed(allocations): worker.free_stacked_tensor(value) From a3b2d5f4cc8c14d59bdc82cec12d0a2b7fb02e9a Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Mon, 28 Sep 2026 09:36:03 +0800 Subject: [PATCH 14/78] Add real-weight SWA/MoE bundles and two-layer numerical validation --- .../model/deepseek_v41/swa_weights.py | 115 ++++++++++++++++++ tests/unit/test_v41_swa_weights.py | 37 ++++++ tools/validate_v41_swa_segment.py | 104 +++++++++++----- 3 files changed, 226 insertions(+), 30 deletions(-) create mode 100644 pypto_serving/model/deepseek_v41/swa_weights.py create mode 100644 tests/unit/test_v41_swa_weights.py diff --git a/pypto_serving/model/deepseek_v41/swa_weights.py b/pypto_serving/model/deepseek_v41/swa_weights.py new file mode 100644 index 00000000..9e31049c --- /dev/null +++ b/pypto_serving/model/deepseek_v41/swa_weights.py @@ -0,0 +1,115 @@ +# Copyright (c) PyPTO Contributors. +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of +# CANN Open Software License Agreement Version 2.0 (the "License"). +# Please refer to the License for details. You may not use this file except in compliance with the License. +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. +# See LICENSE in the root of the software repository for the full text of the License. +# ----------------------------------------------------------------------------------------------------------- + +"""Checkpoint-to-CPU ABI bundles for the bounded SWA/MoE segment. + +Uses the existing selective loader, as V4 separates loading from execution. +No lib fixtures, device execution or FP4 float expansion belongs here. +""" +import torch + +from .weight_loader import V41WeightLoader +from .weight_packing import pack_mx_scale + + +def stack_bytes(values): + """Stack even float8 tensors on CPU, preserving every payload bit.""" + dtype = values[0].dtype + if any(v.dtype != dtype or v.shape != values[0].shape for v in values): + raise ValueError("rank weights must have matching shape and dtype") + return torch.stack([v.contiguous().view(torch.uint8) for v in values]).view(dtype) + + +def merge_expert_scales(scales): + """Repack per-expert MX_B_NN into the lib's flattened expert/K-group ABI. + + Concatenating the already packed buffers would put expert before output + block, whereas the kernel expects output block before expert/K-group. + Only scale bytes are reordered; routed weight payloads stay packed FP4. + """ + groups, width = scales[0].shape + if groups % 2 or width % 16 or any(s.shape != scales[0].shape for s in scales): + raise ValueError("expert scales must share a valid MX_B_NN shape") + logical = [s.contiguous().view(torch.uint8).reshape(width // 16, groups // 2, 16, 2) + .permute(1, 3, 0, 2).reshape(groups, width) for s in scales] + return pack_mx_scale(torch.cat(logical)).view(scales[0].dtype) + + +def mxfp4_pair_lut(): + """Decode two E2M1 nibbles into two E4M3FN bytes in the lib LUT ABI.""" + codes = torch.tensor([0x00, 0x30, 0x38, 0x3C, 0x40, 0x44, 0x48, 0x4C, + 0x80, 0xB0, 0xB8, 0xBC, 0xC0, 0xC4, 0xC8, 0xCC], dtype=torch.int64) + packed = torch.arange(256) + pair = codes[packed & 15] | (codes[packed >> 4] << 8) + pair = torch.where(pair < 32768, pair, pair - 65536) + return pair.to(torch.int16).reshape(1, 256).repeat(2, 1) + + +def load_swa_layer_weights(model_dir, layer_id, topology, *, max_bundle_bytes=32 << 30): + """Return stacked Attention and MoE weight maps for one SWA layer. + + The caller owns the results and uploads them once. max_bundle_bytes is a + conservative bound on stacking buffers for this call, not process memory + or the accumulated weights of other layers. Individual payload reads retain + V41WeightLoader's separate pre-read budget. + """ + if type(max_bundle_bytes) is not int or max_bundle_bytes <= 0: + raise ValueError("max_bundle_bytes must be positive") + attention, moe = [], [] + held_bytes = 0 + for rank in range(topology.world): + loader = V41WeightLoader(model_dir, tp_size=topology.tp, tp_rank=rank % topology.tp, + ep_size=topology.world, ep_rank=rank, max_load_bytes=512 << 20) + if (type(layer_id) is not int or not 0 <= layer_id < loader.config.num_hidden_layers + or loader.text["compress_ratios"][layer_id] != 0): + raise ValueError("selected layer must be a SWA backbone layer") + prefix = f"layers.{layer_id}." + a, m = {}, {} + + def load(name): + nonlocal held_bytes + bundle = loader.load(prefix + name) + held_bytes += sum(t.numel() * t.element_size() for t in (bundle.weight, bundle.scale) + if t is not None) + if 3 * held_bytes > max_bundle_bytes: + raise ValueError("stacked layer weight budget exceeded") + return bundle + + for target, source in { + "hc_attn_fn": "hc_attn_fn", "hc_attn_scale": "hc_attn_scale", + "hc_attn_base": "hc_attn_base", "attn_norm_weight": "attn_norm.weight", + "q_norm_weight": "attn.q_norm.weight", "kv_norm_weight": "attn.kv_norm.weight", + "attn_sink": "attn.attn_sink", + }.items(): + a[target] = load(source).weight + for name in ("wq_a", "wq_b", "wkv", "wo_a", "wo_b"): + bundle = load(f"attn.{name}.weight") + a[name] = bundle.weight + if bundle.scale is not None: + a[name + "_scale"] = bundle.scale + for target, source in { + "hc_ffn_fn": "hc_ffn_fn", "hc_ffn_scale": "hc_ffn_scale", "hc_ffn_base": "hc_ffn_base", + "norm_weight": "ffn_norm.weight", "gate_weight": "ffn.gate.weight", + "correction_bias": "ffn.gate.bias", + }.items(): + m[target] = load(source).weight + for name in ("w1", "w2", "w3"): + bundle = load(f"ffn.shared_experts.{name}.weight") + m["shared_" + name], m["shared_" + name + "_scale"] = bundle.weight, bundle.scale + count = loader.text["n_routed_experts"] // topology.world + bundles = [load(f"ffn.experts.{expert}.{name}.weight") + for expert in range(rank * count, (rank + 1) * count)] + m["routed_" + name] = stack_bytes([b.weight for b in bundles]) + m["routed_" + name + "_scale"] = merge_expert_scales([b.scale for b in bundles]) + del bundles + m["mxfp4_pair_lut"] = mxfp4_pair_lut() + attention.append(a) + moe.append(m) + return ({name: stack_bytes([r[name] for r in attention]) for name in attention[0]}, + {name: stack_bytes([r[name] for r in moe]) for name in moe[0]}) diff --git a/tests/unit/test_v41_swa_weights.py b/tests/unit/test_v41_swa_weights.py new file mode 100644 index 00000000..de00c9d0 --- /dev/null +++ b/tests/unit/test_v41_swa_weights.py @@ -0,0 +1,37 @@ +# Copyright (c) PyPTO Contributors. +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of +# CANN Open Software License Agreement Version 2.0 (the "License"). +# Please refer to the License for details. You may not use this file except in compliance with the License. +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. +# See LICENSE in the root of the software repository for the full text of the License. +# ----------------------------------------------------------------------------------------------------------- + +"""Byte-level checks for multi-expert scale and LUT assembly.""" +import torch + +from pypto_serving.model.deepseek_v41.swa_weights import merge_expert_scales, mxfp4_pair_lut +from pypto_serving.model.deepseek_v41.weight_packing import pack_mx_scale + + +def test_multi_expert_scale_layout(): + logical = [((torch.arange(4 * 32).reshape(4, 32) + e * 17) % 254).to(torch.uint8) + for e in range(3)] + individual = [pack_mx_scale(s) for s in logical] + actual = merge_expert_scales(individual) + # Independent physical-address formula: output block, expert/group-pair, + # output lane, adjacent group parity. This catches a simple concatenation. + for expert in range(3): + for group in range(4): + for col in range(32): + offset = (((col // 16) * 6 + expert * 2 + group // 2) * 16 + col % 16) * 2 + group % 2 + assert actual.flatten()[offset] == logical[expert][group, col] + assert not torch.equal(actual, torch.cat(individual)) + + +def test_pair_lut_matches_fp4_values(): + table = mxfp4_pair_lut().view(torch.uint8).view(torch.float8_e4m3fn).float() + values = torch.tensor([0, .5, 1, 1.5, 2, 3, 4, 6, -0., -.5, -1, -1.5, -2, -3, -4, -6]) + expected = torch.stack([values[torch.arange(256) % 16], values[torch.arange(256) // 16]], dim=1) + for lane in range(2): + torch.testing.assert_close(table[lane].reshape(256, 2), expected, rtol=0, atol=0) diff --git a/tools/validate_v41_swa_segment.py b/tools/validate_v41_swa_segment.py index 8160cdd4..a3835b8f 100644 --- a/tools/validate_v41_swa_segment.py +++ b/tools/validate_v41_swa_segment.py @@ -6,10 +6,10 @@ # INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. # See LICENSE in the root of the software repository for the full text of the License. # ----------------------------------------------------------------------------------------------------------- -"""Synthetic A5 smoke of serving SWA -> MoE -> next-layer SWA/MoE. +"""A5 two-layer SWA/MoE diagnostic with optional checkpoint weights/reference. -This is not real-checkpoint validation or numerical acceptance. Lib fixture -builders are used only here, never by the serving segment implementation. +Activations and request metadata are controlled fixtures, including when real +checkpoint weights are selected. This is not text-generation acceptance. """ import argparse from pathlib import Path @@ -24,6 +24,9 @@ def main(): parser.add_argument("--tp", type=int, default=2) parser.add_argument("--build-dir", default="build_output/v41-swa-segment") parser.add_argument("--compile-only", action="store_true") + parser.add_argument("--model-dir", help="Use actual checkpoint weights for layers 0 and 1") + parser.add_argument("--reference", action="store_true", help="Compare against composed Torch references") + parser.add_argument("--artifact-dir", default=".validation-artifacts/swa-segment") args = parser.parse_args() sys.path.insert(0, str(Path(args.lib_root).resolve())) import torch @@ -57,51 +60,92 @@ def materialize(specs): return {s.name: s.create_tensor().contiguous() for s in specs if isinstance(s, TensorSpec)} a = materialize(swa.build_hc_specs(fixture)) - print("Attention fixture ready; preparing packed-FP4 expert weights", flush=True) - m = materialize(moe.build_tensor_specs([16] * topology.world)) - print("MoE fixture ready", flush=True) + print("Attention fixture ready", flush=True) + if args.model_dir: + from pypto_serving.model.deepseek_v41.swa_weights import load_swa_layer_weights + + layers = [] + for layer_id in (0, 1): + print(f"Loading real checkpoint layer {layer_id}", flush=True) + aw, mw = load_swa_layer_weights(args.model_dir, layer_id, topology) + layer_a = dict(a, **aw) + layer_m = dict(mw, next_pre_mix=torch.zeros_like(a["next_pre_mix"]), + x_mixed=torch.zeros_like(a["attn_out"]), x_next=torch.zeros_like(a["output"]), + num_tokens=torch.full((topology.world,), 16, dtype=torch.int32)) + layers.append((layer_a, layer_m)) + else: + print("Preparing synthetic packed-FP4 experts", flush=True) + m = materialize(moe.build_tensor_specs([16] * topology.world)) + layers = [(a, m), (dict(a), dict(m))] + # Independent output/scratch/cache per layer, even with tied fixture weights. + for index, (la, lm) in enumerate(layers): + if index: + for name in ("window_cache", "window_cache_scale", "output", "next_pre_mix", "hidden", "attn_out"): + la[name] = la[name].clone() + for name in ("next_pre_mix", "x_mixed", "x_next"): + lm[name] = lm[name].clone() ac = torch.zeros(topology.world, 1, dtype=torch.int32).share_memory_() mc = torch.zeros(topology.world, dtype=torch.int32).share_memory_() - readback = torch.empty_like(m["x_next"]).share_memory_() - mix_readback = torch.empty_like(m["next_pre_mix"]).share_memory_() - sources = [*a.values(), *m.values()] + readback = torch.empty_like(layers[-1][1]["x_next"]).share_memory_() + mix_readback = torch.empty_like(layers[-1][1]["next_pre_mix"]).share_memory_() + sources = [v for pair in layers for mapping in pair for v in mapping.values()] + print("Weights ready; executing two-layer device segment", flush=True) with make_segment_worker(programs, config, sources) as worker: - allocations = [] + allocations, uploaded = [], {} def upload(values): result = {} for name, value in values.items(): if name == "num_tokens": continue - device = worker.alloc_stacked_tensor(value) - allocations.append(device) - result[name] = device + if id(value) not in uploaded: + device = worker.alloc_stacked_tensor(value) + allocations.append(device) + uploaded[id(value)] = device + result[name] = uploaded[id(value)] return result - da, dm = upload(a), upload(m) - state = LayerState(da.pop("x_hc"), da.pop("incoming_pre_mix"), "tp_local_token") - dm.pop("x_hc") - dm.pop("pre_mix") - # Diagnostic uses tied synthetic weights but independent layer caches. - second_a = dict(da) - for name in ("window_cache", "window_cache_scale"): - second_a[name] = worker.alloc_stacked_tensor(a[name]) - allocations.append(second_a[name]) + device_layers = [(upload(la), upload(lm)) for la, lm in layers] + da = device_layers[0][0] + state = LayerState(da["x_hc"], da["incoming_pre_mix"], "tp_local_token") runner = SwaSegment(worker, programs, topology, ac, mc, config) - first = runner.run_layer(state, da, dm, group_counts=[topology.capacity] * topology.dp) - second_m = dict(dm, x_next=state.residual, next_pre_mix=state.pre_mix) - final = runner.run_layer(first, second_a, second_m, - group_counts=[topology.capacity] * topology.dp) - # Read back only after the full chain; no intermediate host round trip. - worker.copy_stacked_from(final.residual, readback) - worker.copy_stacked_from(final.pre_mix, mix_readback) + for layer_id, (da, dm) in enumerate(device_layers): + state = runner.run_layer(state, da, dm, group_counts=[topology.capacity] * topology.dp) + print(f"Device layer {layer_id} complete", flush=True) + worker.copy_stacked_from(state.residual, readback) + worker.copy_stacked_from(state.pre_mix, mix_readback) assert torch.isfinite(readback).all() and torch.isfinite(mix_readback).all() assert readback.abs().max() > 0 and mix_readback.abs().max() > 0 assert not torch.equal(readback, a["x_hc"]), "residual was not updated" assert not torch.equal(mix_readback, a["incoming_pre_mix"]), "pre_mix was not updated" - print("DEVICE TWO-LAYER SMOKE PASS (finite/nonzero only; not numerical acceptance)", flush=True) for value in reversed(allocations): worker.free_stacked_tensor(value) + print("DEVICE TWO-LAYER SMOKE PASS", flush=True) + if args.reference: + # References run after worker shutdown and never feed device execution. + # All weight buffers are read-only; clone only mutable state/scratch. + residual, mix = a["x_hc"], a["incoming_pre_mix"] + for layer_id, (la, lm) in enumerate(layers): + ra = dict(la, x_hc=residual, incoming_pre_mix=mix) + for name in ("window_cache", "window_cache_scale", "output", "next_pre_mix", "hidden", "attn_out"): + ra[name] = la[name].clone() + swa.golden_prefill_swa_case(ra) + rm = dict(lm, x_hc=ra["output"], pre_mix=ra["next_pre_mix"]) + for name in ("next_pre_mix", "x_mixed", "x_next"): + rm[name] = lm[name].clone() + moe.golden_moe(rm) + residual, mix = rm["x_next"], rm["next_pre_mix"] + print(f"Torch reference layer {layer_id} complete", flush=True) + artifact = Path(args.artifact_dir) + artifact.mkdir(parents=True, exist_ok=True) + torch.save({"actual_residual": readback, "expected_residual": residual, + "actual_pre_mix": mix_readback, "expected_pre_mix": mix}, artifact / "comparison.pt") + compare = moe._local_mhc_compare([16] * topology.world) + ok, message = compare(readback, residual) + print("Final residual reference check:", ok, message, flush=True) + assert ok, message + torch.testing.assert_close(mix_readback, mix, rtol=1e-2, atol=1e-4) + print("TWO-LAYER TORCH REFERENCE PASS", flush=True) if __name__ == "__main__": From 1633399e9684ef0583c6c52398d42442e76ae8ed Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Mon, 28 Sep 2026 09:46:34 +0800 Subject: [PATCH 15/78] Use lib MoE heap budget in the real-weight diagnostic --- docs/developer-guide/v41-swa-segment.md | 29 ++++++++++++++++++++++++- tools/validate_v41_swa_segment.py | 4 +++- 2 files changed, 31 insertions(+), 2 deletions(-) diff --git a/docs/developer-guide/v41-swa-segment.md b/docs/developer-guide/v41-swa-segment.md index f568b67c..212919b2 100644 --- a/docs/developer-guide/v41-swa-segment.md +++ b/docs/developer-guide/v41-swa-segment.md @@ -1,4 +1,4 @@ -# V4.1 bounded SWA segment +# V4.1 bounded SWA segment `pypto_serving/model/deepseek_v41/swa_segment.py` adds the concrete half-layer execution boundary inspected against lib `216456332c2a74d89cca23b7824dab264ce34bff`: @@ -36,3 +36,30 @@ This segment does not enable `load_composite_bindings()` for complete serving: checkpoint-to-resident bundle assembly, input initialization, all attention modes, decode, cache lifecycle and the final output boundary still need adapters and validation. Engram is excluded. Full TP4/DP2/EP8 8K-to-128 M0 is not claimed. + +## Real checkpoint bundles and numerical diagnostic + +`load_swa_layer_weights(model_dir, layer_id, topology)` returns CPU Attention +and MoE weight maps for a SWA layer. It uses the selective checkpoint loader; +TP projections and EP expert ownership retain their existing rules. Routed +payloads stay packed FP4. Per-expert scales must be unpacked and repacked into +the combined expert/K-group MX layout, not concatenated in their already packed +order. The caller still owns upload, request metadata and cache allocation. + +To exercise distinct real weights for layers 0 and 1 and compare the final +state against composed Torch references: + +```bash +PYTHONPATH=. python tools/validate_v41_swa_segment.py \ + --lib-root /path/to/current/pypto-lib --tp 2 --devices 0,1,2,3 \ + --model-dir /path/to/DeepSeek-V4.1-Flash --reference +``` + +The diagnostic uses controlled activations and request metadata, including in +checkpoint mode. It does not yet validate embedding initialization, real prompt +semantics, Engram or generation. The reference runs only after the device worker +closes, never supplies intermediate device inputs, and shares read-only weights +with the device path. Final actual/expected residual and pre_mix are saved to +`--artifact-dir/comparison.pt` even when the numerical comparison fails. The +residual gate uses lib's local MoE relative-error comparator; pre_mix uses +rtol=0.01 and atol=0.0001. A completed smoke is not a numerical pass. diff --git a/tools/validate_v41_swa_segment.py b/tools/validate_v41_swa_segment.py index a3835b8f..0ed69028 100644 --- a/tools/validate_v41_swa_segment.py +++ b/tools/validate_v41_swa_segment.py @@ -27,6 +27,8 @@ def main(): parser.add_argument("--model-dir", help="Use actual checkpoint weights for layers 0 and 1") parser.add_argument("--reference", action="store_true", help="Compare against composed Torch references") parser.add_argument("--artifact-dir", default=".validation-artifacts/swa-segment") + parser.add_argument("--ring-heap-mib", type=int, default=1024, + help="Per-ring temporary heap; lib MoE validation uses 1024 MiB") args = parser.parse_args() sys.path.insert(0, str(Path(args.lib_root).resolve())) import torch @@ -45,7 +47,7 @@ def main(): raise ValueError("unique devices must form complete TP groups") topology = SegmentTopology(tp=args.tp, dp=len(devices) // args.tp) config = RunConfig(platform="a5", distributed_config=DistributedConfig(device_ids=devices), - ring_heap=536870912, ring_task_window=131072, ring_dep_pool=131072) + ring_heap=args.ring_heap_mib << 20, ring_task_window=131072, ring_dep_pool=131072) compiler = KernelCompiler(run_config=config, cache_dir=args.build_dir) programs = compile_segment(compiler, args.lib_root, topology) print("COMPILE PASS", flush=True) From b52934f9f9bcf3f194e53a3f29f7aa685244b91c Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Mon, 28 Sep 2026 09:53:39 +0800 Subject: [PATCH 16/78] docs: register the V4.1 segment guide in navigation --- mkdocs.yml | 1 + 1 file changed, 1 insertion(+) diff --git a/mkdocs.yml b/mkdocs.yml index d325dc93..91028a8e 100644 --- a/mkdocs.yml +++ b/mkdocs.yml @@ -112,3 +112,4 @@ nav: - DeepSeek V4 Runtime: developer-guide/deepseek-v4-runtime.md - DeepSeek V4 DSpark: developer-guide/deepseek-v4-dspark.md - DeepSeek V4.1 Entry: developer-guide/deepseek-v41-entry.md + - DeepSeek V4.1 SWA Segment: developer-guide/v41-swa-segment.md From 2bfcdcf2242f84fc09a9875e6d70487eeb601eb3 Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Mon, 28 Sep 2026 10:32:18 +0800 Subject: [PATCH 17/78] Fix lib reference comparator invocation and allow saved-output checks --- tools/validate_v41_swa_segment.py | 45 +++++++++++++++++++++++++------ 1 file changed, 37 insertions(+), 8 deletions(-) diff --git a/tools/validate_v41_swa_segment.py b/tools/validate_v41_swa_segment.py index 0ed69028..2f26cd71 100644 --- a/tools/validate_v41_swa_segment.py +++ b/tools/validate_v41_swa_segment.py @@ -17,6 +17,34 @@ import sys +def compare_saved(data, moe, topology): + """Apply unchanged lib budgets to live or saved post-run outputs on CPU.""" + import torch + + actual = data["actual_residual"] + expected = data["expected_residual"] + for name in ("residual", "pre_mix"): + a, e = data["actual_" + name].double(), data["expected_" + name].double() + error = a - e + print(f"{name}: rel_l2={(error.norm() / e.norm().clamp_min(1e-12)).item():.8g} " + f"max_abs={error.abs().max().item():.8g}", flush=True) + counts = [topology.local_capacity] * topology.world + compare = moe._local_mhc_compare(counts) + ok, message = compare( + actual, expected, actual_outputs={"x_next": actual}, expected_outputs={"x_next": expected}, + inputs={"num_tokens": torch.tensor(counts, dtype=torch.int32)}, rtol=1e-5, atol=1e-5, + ) + print("Final residual reference check:", ok, message, flush=True) + mix_error = None + try: + torch.testing.assert_close(data["actual_pre_mix"], data["expected_pre_mix"], rtol=1e-2, atol=1e-4) + except AssertionError as exc: + mix_error = str(exc) + print("Final pre_mix reference check:", mix_error is None, mix_error or "", flush=True) + assert ok and mix_error is None, message + (mix_error or "") + print("TWO-LAYER TORCH REFERENCE PASS", flush=True) + + def main(): parser = argparse.ArgumentParser(description=__doc__) parser.add_argument("--lib-root", required=True) @@ -26,6 +54,7 @@ def main(): parser.add_argument("--compile-only", action="store_true") parser.add_argument("--model-dir", help="Use actual checkpoint weights for layers 0 and 1") parser.add_argument("--reference", action="store_true", help="Compare against composed Torch references") + parser.add_argument("--compare-only", help="Recheck a saved comparison.pt on CPU without compilation/device use") parser.add_argument("--artifact-dir", default=".validation-artifacts/swa-segment") parser.add_argument("--ring-heap-mib", type=int, default=1024, help="Per-ring temporary heap; lib MoE validation uses 1024 MiB") @@ -46,6 +75,10 @@ def main(): if len(devices) % args.tp or len(set(devices)) != len(devices): raise ValueError("unique devices must form complete TP groups") topology = SegmentTopology(tp=args.tp, dp=len(devices) // args.tp) + if args.compare_only: + _, moe = load_segment_modules(args.lib_root, topology) + compare_saved(torch.load(args.compare_only, map_location="cpu", weights_only=True), moe, topology) + return config = RunConfig(platform="a5", distributed_config=DistributedConfig(device_ids=devices), ring_heap=args.ring_heap_mib << 20, ring_task_window=131072, ring_dep_pool=131072) compiler = KernelCompiler(run_config=config, cache_dir=args.build_dir) @@ -140,14 +173,10 @@ def upload(values): print(f"Torch reference layer {layer_id} complete", flush=True) artifact = Path(args.artifact_dir) artifact.mkdir(parents=True, exist_ok=True) - torch.save({"actual_residual": readback, "expected_residual": residual, - "actual_pre_mix": mix_readback, "expected_pre_mix": mix}, artifact / "comparison.pt") - compare = moe._local_mhc_compare([16] * topology.world) - ok, message = compare(readback, residual) - print("Final residual reference check:", ok, message, flush=True) - assert ok, message - torch.testing.assert_close(mix_readback, mix, rtol=1e-2, atol=1e-4) - print("TWO-LAYER TORCH REFERENCE PASS", flush=True) + data = {"actual_residual": readback, "expected_residual": residual, + "actual_pre_mix": mix_readback, "expected_pre_mix": mix} + torch.save(data, artifact / "comparison.pt") + compare_saved(data, moe, topology) if __name__ == "__main__": From 7ef97ca8658dfcf500da9d235a2b323560df0aaa Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Mon, 28 Sep 2026 10:34:39 +0800 Subject: [PATCH 18/78] Add post-run half-layer reference diagnostics without device feedback --- tools/validate_v41_swa_segment.py | 55 +++++++++++++++++++++++++++++++ 1 file changed, 55 insertions(+) diff --git a/tools/validate_v41_swa_segment.py b/tools/validate_v41_swa_segment.py index 2f26cd71..92c663f4 100644 --- a/tools/validate_v41_swa_segment.py +++ b/tools/validate_v41_swa_segment.py @@ -16,6 +16,45 @@ from types import SimpleNamespace import sys +ATTENTION_OUTPUTS = ("output", "next_pre_mix", "hidden", "attn_out", "window_cache", "window_cache_scale") +MOE_OUTPUTS = ("x_next", "next_pre_mix", "x_mixed") + + +def compare_stages(layers, captured, initial, swa, moe, topology): + """Localize errors against each half-layer's actual input, after execution.""" + from golden.validation import ratio_allclose + + residual, mix = initial + records = [] + for layer_id, ((la, lm), (actual_a, actual_m)) in enumerate(zip(layers, captured)): + ra = dict(la, x_hc=residual, incoming_pre_mix=mix) + for name in ATTENTION_OUTPUTS: + ra[name] = la[name].clone() + swa.golden_prefill_swa_case(ra) + rm = dict(lm, x_hc=actual_a["output"], pre_mix=actual_a["next_pre_mix"]) + for name in MOE_OUTPUTS: + rm[name] = lm[name].clone() + moe.golden_moe(rm) + checks = ( + ("attention", actual_a, ra, swa.make_staged_compare()), + ("moe", actual_m, rm, { + "next_pre_mix": ratio_allclose(atol=2.5e-5, rtol=5e-3), + "x_mixed": ratio_allclose(atol=1e-4, rtol=1.0 / 128), + "x_next": moe._local_mhc_compare([topology.local_capacity] * topology.world), + }), + ) + for label, actual, reference, comparators in checks: + results = {} + for name, compare in comparators.items(): + ok, message = compare(actual[name], reference[name], actual_outputs=actual, + expected_outputs=reference, inputs=reference, rtol=1e-3, atol=1e-3) + print(f"STAGE layer={layer_id} {label}.{name}: {ok} {message}", flush=True) + results[name] = (bool(ok), message) + records.append({"layer": layer_id, "stage": label, "results": results, + "actual": actual, "expected": {name: reference[name] for name in actual}}) + residual, mix = actual_m["x_next"], actual_m["next_pre_mix"] + return records + def compare_saved(data, moe, topology): """Apply unchanged lib budgets to live or saved post-run outputs on CPU.""" @@ -54,11 +93,15 @@ def main(): parser.add_argument("--compile-only", action="store_true") parser.add_argument("--model-dir", help="Use actual checkpoint weights for layers 0 and 1") parser.add_argument("--reference", action="store_true", help="Compare against composed Torch references") + parser.add_argument("--stage-reference", action="store_true", + help="Also localize errors on each half-layer's device input, after the full run") parser.add_argument("--compare-only", help="Recheck a saved comparison.pt on CPU without compilation/device use") parser.add_argument("--artifact-dir", default=".validation-artifacts/swa-segment") parser.add_argument("--ring-heap-mib", type=int, default=1024, help="Per-ring temporary heap; lib MoE validation uses 1024 MiB") args = parser.parse_args() + if args.stage_reference: + args.reference = True sys.path.insert(0, str(Path(args.lib_root).resolve())) import torch from pypto.ir import DistributedConfig @@ -123,6 +166,9 @@ def materialize(specs): mc = torch.zeros(topology.world, dtype=torch.int32).share_memory_() readback = torch.empty_like(layers[-1][1]["x_next"]).share_memory_() mix_readback = torch.empty_like(layers[-1][1]["next_pre_mix"]).share_memory_() + captured = [({name: torch.empty_like(la[name]).share_memory_() for name in ATTENTION_OUTPUTS}, + {name: torch.empty_like(lm[name]).share_memory_() for name in MOE_OUTPUTS}) + for la, lm in layers] if args.stage_reference else [] sources = [v for pair in layers for mapping in pair for v in mapping.values()] print("Weights ready; executing two-layer device segment", flush=True) with make_segment_worker(programs, config, sources) as worker: @@ -149,6 +195,11 @@ def upload(values): print(f"Device layer {layer_id} complete", flush=True) worker.copy_stacked_from(state.residual, readback) worker.copy_stacked_from(state.pre_mix, mix_readback) + # Read only after both layers finish; never feed diagnostic state back. + for (da, dm), (ca, cm) in zip(device_layers, captured): + for device, cpu in ((da, ca), (dm, cm)): + for name, destination in cpu.items(): + worker.copy_stacked_from(device[name], destination) assert torch.isfinite(readback).all() and torch.isfinite(mix_readback).all() assert readback.abs().max() > 0 and mix_readback.abs().max() > 0 assert not torch.equal(readback, a["x_hc"]), "residual was not updated" @@ -176,6 +227,10 @@ def upload(values): data = {"actual_residual": readback, "expected_residual": residual, "actual_pre_mix": mix_readback, "expected_pre_mix": mix} torch.save(data, artifact / "comparison.pt") + if captured: + data["stages"] = compare_stages(layers, captured, (a["x_hc"], a["incoming_pre_mix"]), + swa, moe, topology) + torch.save(data, artifact / "comparison.pt") compare_saved(data, moe, topology) From 0f59700e0c1110569fa0caadd5acd222e96a107a Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Mon, 28 Sep 2026 10:35:53 +0800 Subject: [PATCH 19/78] docs: record real-weight memory and numerical validation boundaries --- docs/developer-guide/v41-swa-segment.md | 26 +++++++++++++++++++++++-- 1 file changed, 24 insertions(+), 2 deletions(-) diff --git a/docs/developer-guide/v41-swa-segment.md b/docs/developer-guide/v41-swa-segment.md index 212919b2..f01f31eb 100644 --- a/docs/developer-guide/v41-swa-segment.md +++ b/docs/developer-guide/v41-swa-segment.md @@ -33,7 +33,7 @@ smoke checks completion and finite/nonzero outputs, not numerical acceptance. CPU dispatch tests use a mocked worker and do not establish NPU correctness. This segment does not enable `load_composite_bindings()` for complete serving: -checkpoint-to-resident bundle assembly, input initialization, all attention +production resident bundle management, input initialization, all attention modes, decode, cache lifecycle and the final output boundary still need adapters and validation. Engram is excluded. Full TP4/DP2/EP8 8K-to-128 M0 is not claimed. @@ -52,7 +52,7 @@ state against composed Torch references: ```bash PYTHONPATH=. python tools/validate_v41_swa_segment.py \ --lib-root /path/to/current/pypto-lib --tp 2 --devices 0,1,2,3 \ - --model-dir /path/to/DeepSeek-V4.1-Flash --reference + --model-dir /path/to/DeepSeek-V4.1-Flash --reference --ring-heap-mib 4096 ``` The diagnostic uses controlled activations and request metadata, including in @@ -63,3 +63,25 @@ with the device path. Final actual/expected residual and pre_mix are saved to `--artifact-dir/comparison.pt` even when the numerical comparison fails. The residual gate uses lib's local MoE relative-error comparator; pre_mix uses rtol=0.01 and atol=0.0001. A completed smoke is not a numerical pass. + +The real-weight TP2/DP2/EP4 diagnostic exhausted the temporary heap at both +512 MiB and 1024 MiB per ring. With 4096 MiB per ring, both device layers and +their Torch references completed. This runtime has four rings, so a per-ring +setting is not the total allocation; check device headroom before running. +The saved final outputs did not pass the numerical gate (residual relative L2 +about 0.00760, pre_mix about 0.00105). These are diagnostic observations, not +an accepted end-to-end tolerance or a production memory recommendation. + +`--stage-reference` also captures each half-layer's outputs after the entire +device chain completes, and applies lib's stage comparators to references +computed on that half-layer's actual input. This localizes accumulated errors; +it does not replace the independent end-to-end check or feed CPU values back +to the device. The final check still determines success. + +To recheck a saved final comparison without compiling or allocating devices: + +```bash +PYTHONPATH=. python tools/validate_v41_swa_segment.py \ + --lib-root /path/to/current/pypto-lib --tp 2 --devices 0,1,2,3 \ + --compare-only /path/to/comparison.pt +``` From c257c9316aa81f39a8b3943476ad4468593d6752 Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Mon, 28 Sep 2026 10:40:39 +0800 Subject: [PATCH 20/78] Keep stage and end-to-end numerical gates distinct and mandatory --- docs/developer-guide/v41-swa-segment.md | 6 ++++ tests/unit/test_v41_swa_validation.py | 42 +++++++++++++++++++++++++ tools/validate_v41_swa_segment.py | 5 ++- 3 files changed, 52 insertions(+), 1 deletion(-) create mode 100644 tests/unit/test_v41_swa_validation.py diff --git a/docs/developer-guide/v41-swa-segment.md b/docs/developer-guide/v41-swa-segment.md index f01f31eb..c07c9d04 100644 --- a/docs/developer-guide/v41-swa-segment.md +++ b/docs/developer-guide/v41-swa-segment.md @@ -78,6 +78,12 @@ computed on that half-layer's actual input. This localizes accumulated errors; it does not replace the independent end-to-end check or feed CPU values back to the device. The final check still determines success. +At serving `7ef97ca`, all stage checks passed for both real-weight SWA/MoE +layers, including Attention caches and MoE residuals. The independent two-layer +check still failed with the errors above. Agreement on each stage's actual input +does not establish the accumulated numerical budget across layers; that boundary +still needs validation before full-model acceptance. + To recheck a saved final comparison without compiling or allocating devices: ```bash diff --git a/tests/unit/test_v41_swa_validation.py b/tests/unit/test_v41_swa_validation.py new file mode 100644 index 00000000..512f4ed9 --- /dev/null +++ b/tests/unit/test_v41_swa_validation.py @@ -0,0 +1,42 @@ +# Copyright (c) PyPTO Contributors. +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of +# CANN Open Software License Agreement Version 2.0 (the "License"). +# Please refer to the License for details. You may not use this file except in compliance with the License. +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. +# See LICENSE in the root of the software repository for the full text of the License. +# ----------------------------------------------------------------------------------------------------------- + +"""Reference diagnostics must reject local or accumulated numerical failures.""" +from types import SimpleNamespace + +import pytest +import torch + +from tools.validate_v41_swa_segment import compare_saved + + +@pytest.mark.parametrize("failure", [None, "residual", "pre_mix", "stage"]) +def test_saved_comparison_keeps_all_acceptance_gates(failure): + residual = torch.ones(1, 2, 4, 3) + mix = torch.ones(1, 2, 4) + data = {"actual_residual": residual.clone(), "expected_residual": residual, + "actual_pre_mix": mix.clone(), "expected_pre_mix": mix} + if failure in ("residual", "pre_mix"): + data["actual_" + failure].add_(1) + if failure == "stage": + data["stages"] = [{"results": {"output": (False, "stage mismatch")}}] + + def comparator(actual, expected, *, actual_outputs, expected_outputs, inputs, rtol, atol): + assert inputs["num_tokens"].tolist() == [2] + assert actual_outputs["x_next"] is actual + assert expected_outputs["x_next"] is expected + return torch.equal(actual, expected), "residual mismatch" + + moe = SimpleNamespace(_local_mhc_compare=lambda counts: comparator) + topology = SimpleNamespace(local_capacity=2, world=1) + if failure is None: + compare_saved(data, moe, topology) + else: + with pytest.raises(AssertionError): + compare_saved(data, moe, topology) diff --git a/tools/validate_v41_swa_segment.py b/tools/validate_v41_swa_segment.py index 92c663f4..6939e240 100644 --- a/tools/validate_v41_swa_segment.py +++ b/tools/validate_v41_swa_segment.py @@ -80,7 +80,10 @@ def compare_saved(data, moe, topology): except AssertionError as exc: mix_error = str(exc) print("Final pre_mix reference check:", mix_error is None, mix_error or "", flush=True) - assert ok and mix_error is None, message + (mix_error or "") + stages_ok = all(result[0] for stage in data.get("stages", []) for result in stage["results"].values()) + assert ok and mix_error is None and stages_ok, message + (mix_error or "") + ( + "Half-layer stage comparison failed" if not stages_ok else "" + ) print("TWO-LAYER TORCH REFERENCE PASS", flush=True) From d0a7e176538acc254239c07ff602a67d2e32f69d Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Mon, 28 Sep 2026 10:53:21 +0800 Subject: [PATCH 21/78] Add CPU trace replay to locate the first accumulated precision divergence --- tools/diagnose_v41_swa_precision.py | 85 +++++++++++++++++++++++++++++ 1 file changed, 85 insertions(+) create mode 100644 tools/diagnose_v41_swa_precision.py diff --git a/tools/diagnose_v41_swa_precision.py b/tools/diagnose_v41_swa_precision.py new file mode 100644 index 00000000..f8aea9fc --- /dev/null +++ b/tools/diagnose_v41_swa_precision.py @@ -0,0 +1,85 @@ +# Copyright (c) PyPTO Contributors. +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of +# CANN Open Software License Agreement Version 2.0 (the "License"). +# Please refer to the License for details. You may not use this file except in compliance with the License. +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. +# See LICENSE in the root of the software repository for the full text of the License. +# ----------------------------------------------------------------------------------------------------------- +"""CPU replay of a saved two-layer device trace to bisect accumulated error. + +Uses the original seed-11 full-prefix fixture and checkpoint layers 0/1. +No device execution, production operator composition or tolerance changes. +""" +import argparse +import json +from pathlib import Path +import sys +from types import SimpleNamespace + + +def metrics(actual, expected): + import torch + + a, e = actual.double(), expected.double() + diff = (a - e).abs() + floor = (1.0 / (1 << 14)) / 0.003 + denom = torch.maximum(a.abs(), e.abs()).clamp_min(floor) + 1e-9 + relative = torch.where(diff < 0.003, diff, diff / denom) + return {"rel_l2": float(diff.norm() / e.norm().clamp_min(1e-12)), + "max_abs": float(diff.max()), "bad_fraction": float((relative > .003).double().mean()), + "equal_fraction": float((a == e).double().mean())} + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--lib-root", required=True) + parser.add_argument("--model-dir", required=True) + parser.add_argument("--saved", required=True) + parser.add_argument("--output", required=True) + args = parser.parse_args() + sys.path.insert(0, str(Path(args.lib_root).resolve())) + import torch + from golden.spec import TensorSpec + from pypto_serving.model.deepseek_v41.swa_segment import SegmentTopology, load_segment_modules + from pypto_serving.model.deepseek_v41.swa_weights import load_swa_layer_weights + + torch.set_num_threads(4) + topology = SegmentTopology(tp=2, dp=2) + swa, moe = load_segment_modules(args.lib_root, topology) + saved = torch.load(args.saved, map_location="cpu", weights_only=True) + fixture = SimpleNamespace(tokens=32, requests=1, dp=2, seed=11, case="normal", + fixture="checkpoint", dp_tokens=None, epochs=1, bench=False) + a = {s.name: s.create_tensor().contiguous() for s in swa.build_hc_specs(fixture) + if isinstance(s, TensorSpec)} + residual, mix = a["x_hc"], a["incoming_pre_mix"] + records = [] + for layer in (0, 1): + print(f"Loading layer {layer}", flush=True) + aw, mw = load_swa_layer_weights(args.model_dir, layer, topology) + la = dict(a, **aw, x_hc=residual, incoming_pre_mix=mix) + for name in ("output", "next_pre_mix", "hidden", "attn_out", "window_cache", "window_cache_scale"): + la[name] = a[name].clone() + swa.golden_prefill_swa_case(la) + lm = dict(mw, x_hc=la["output"], pre_mix=la["next_pre_mix"], + next_pre_mix=torch.zeros_like(mix), x_mixed=torch.zeros_like(a["attn_out"]), + x_next=torch.zeros_like(residual), num_tokens=torch.full((4,), 16, dtype=torch.int32)) + moe.golden_moe(lm) + for kind, expected in (("attention", la), ("moe", lm)): + actual = saved["stages"][2 * layer + (kind == "moe")]["actual"] + result = {name: metrics(actual[name], expected[name]) for name in actual + if name not in ("window_cache", "window_cache_scale")} + print(json.dumps({"layer": layer, "stage": kind, "metrics": result}), flush=True) + records.append({"layer": layer, "stage": kind, "metrics": result, + "expected": {name: expected[name].clone() for name in actual}}) + residual, mix = lm["x_next"], lm["next_pre_mix"] + del aw, mw, la, lm + print("Replay matches saved reference:", torch.equal(residual, saved["expected_residual"]), + torch.equal(mix, saved["expected_pre_mix"]), flush=True) + assert torch.equal(residual, saved["expected_residual"]) + assert torch.equal(mix, saved["expected_pre_mix"]) + torch.save(records, args.output) + + +if __name__ == "__main__": + main() From 3b06edce4b653bbf5998472344829050a96e74a2 Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Mon, 28 Sep 2026 10:57:34 +0800 Subject: [PATCH 22/78] Trace attention projection and quantization divergence on saved inputs --- tools/diagnose_v41_swa_precision.py | 54 +++++++++++++++++++++++++++++ 1 file changed, 54 insertions(+) diff --git a/tools/diagnose_v41_swa_precision.py b/tools/diagnose_v41_swa_precision.py index f8aea9fc..5d5adc14 100644 --- a/tools/diagnose_v41_swa_precision.py +++ b/tools/diagnose_v41_swa_precision.py @@ -31,12 +31,57 @@ def metrics(actual, expected): "equal_fraction": float((a == e).double().mean())} +def trace_attention(swa, tensors, actual_hidden, reference_hidden): + """Bisect the second attention on CPU, recording its public reference calls.""" + from models.deepseek_v4_1_flash import decode_attn_swa as ref + + original_linear, original_rope = ref.official_linear, ref.official_rope + traces = [] + try: + for hidden in (actual_hidden, reference_hidden): + trace = {} + calls = iter(("q_a", "q_b", "kv", "o_b")) + ropes = iter(("q_rope", "kv_rope", "out_rope")) + + def linear(x, weight, scale, fp32=False): + name = next(calls) + trace[name + ".input"] = x.clone() + payload, codes = ref.official_quantize(x) + trace[name + ".quant"] = payload.float() + trace[name + ".scale"] = codes.float() + result = original_linear(x, weight, scale, fp32=fp32) + trace[name + ".output"] = result.clone() + return result + + def rope(x, cos, sin, inverse=False): + name = next(ropes) + trace[name + ".input"] = x.clone() + result = original_rope(x, cos, sin, inverse=inverse) + trace[name + ".output"] = result.clone() + return result + + ref.official_linear, ref.official_rope = linear, rope + inputs = {name: tensors[name][0] for name in swa.HC_INPUT_NAMES if name not in ( + "x_hc", "incoming_pre_mix", "hc_attn_fn", "hc_attn_scale", "hc_attn_base", "attn_norm_weight", + )} + inputs["x"] = hidden + ref.official_reference(inputs) + traces.append(trace) + finally: + ref.official_linear, ref.official_rope = original_linear, original_rope + for name in traces[0]: + print("TRACE", name, json.dumps(metrics(traces[0][name], traces[1][name])), flush=True) + return traces + + def main(): parser = argparse.ArgumentParser(description=__doc__) parser.add_argument("--lib-root", required=True) parser.add_argument("--model-dir", required=True) parser.add_argument("--saved", required=True) parser.add_argument("--output", required=True) + parser.add_argument("--trace-attention", action="store_true", + help="Use the saved full-reference replay to bisect layer-1 rank-0 attention") args = parser.parse_args() sys.path.insert(0, str(Path(args.lib_root).resolve())) import torch @@ -52,6 +97,15 @@ def main(): fixture="checkpoint", dp_tokens=None, epochs=1, bench=False) a = {s.name: s.create_tensor().contiguous() for s in swa.build_hc_specs(fixture) if isinstance(s, TensorSpec)} + if args.trace_attention: + full = torch.load(args.output, map_location="cpu", weights_only=True) + print("Loading layer 1 for attention trace", flush=True) + aw, unused_moe = load_swa_layer_weights(args.model_dir, 1, topology) + del unused_moe + traces = trace_attention(swa, dict(a, **aw), saved["stages"][2]["actual"]["hidden"][0], + full[2]["expected"]["hidden"][0]) + torch.save(traces, str(args.output) + ".attention-trace.pt") + return residual, mix = a["x_hc"], a["incoming_pre_mix"] records = [] for layer in (0, 1): From e33cc07da32f12cd5ec1919e3d4add50e74b768d Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Mon, 28 Sep 2026 11:01:07 +0800 Subject: [PATCH 23/78] Add saved-boundary interventions for precision bisection --- tools/diagnose_v41_swa_precision.py | 18 ++++++++++++++++++ 1 file changed, 18 insertions(+) diff --git a/tools/diagnose_v41_swa_precision.py b/tools/diagnose_v41_swa_precision.py index 5d5adc14..797e4b0b 100644 --- a/tools/diagnose_v41_swa_precision.py +++ b/tools/diagnose_v41_swa_precision.py @@ -82,6 +82,8 @@ def main(): parser.add_argument("--output", required=True) parser.add_argument("--trace-attention", action="store_true", help="Use the saved full-reference replay to bisect layer-1 rank-0 attention") + parser.add_argument("--cut-after", type=int, choices=(0, 1, 2), + help="Restart the CPU reference from a saved device boundary; diagnostic only") args = parser.parse_args() sys.path.insert(0, str(Path(args.lib_root).resolve())) import torch @@ -109,12 +111,19 @@ def main(): residual, mix = a["x_hc"], a["incoming_pre_mix"] records = [] for layer in (0, 1): + if args.cut_after is not None and args.cut_after >= 2 * layer + 1: + cut = saved["stages"][2 * layer + 1]["actual"] + residual, mix = cut["x_next"], cut["next_pre_mix"] + continue print(f"Loading layer {layer}", flush=True) aw, mw = load_swa_layer_weights(args.model_dir, layer, topology) la = dict(a, **aw, x_hc=residual, incoming_pre_mix=mix) for name in ("output", "next_pre_mix", "hidden", "attn_out", "window_cache", "window_cache_scale"): la[name] = a[name].clone() swa.golden_prefill_swa_case(la) + if args.cut_after == 2 * layer: + cut = saved["stages"][2 * layer]["actual"] + la["output"], la["next_pre_mix"] = cut["output"], cut["next_pre_mix"] lm = dict(mw, x_hc=la["output"], pre_mix=la["next_pre_mix"], next_pre_mix=torch.zeros_like(mix), x_mixed=torch.zeros_like(a["attn_out"]), x_next=torch.zeros_like(residual), num_tokens=torch.full((4,), 16, dtype=torch.int32)) @@ -128,6 +137,15 @@ def main(): "expected": {name: expected[name].clone() for name in actual}}) residual, mix = lm["x_next"], lm["next_pre_mix"] del aw, mw, la, lm + if args.cut_after is not None: + print("CUT", args.cut_after, "final residual", json.dumps(metrics(saved["actual_residual"], residual)), + "final pre_mix", json.dumps(metrics(saved["actual_pre_mix"], mix)), flush=True) + from validate_v41_swa_segment import compare_saved + result = {"actual_residual": saved["actual_residual"], "expected_residual": residual, + "actual_pre_mix": saved["actual_pre_mix"], "expected_pre_mix": mix} + torch.save(result, str(args.output) + f".cut-{args.cut_after}.pt") + compare_saved(result, moe, topology) + return print("Replay matches saved reference:", torch.equal(residual, saved["expected_residual"]), torch.equal(mix, saved["expected_pre_mix"]), flush=True) assert torch.equal(residual, saved["expected_residual"]) From 9d9ce9af4de487e19882f12660b1732f3c3e8718 Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Mon, 28 Sep 2026 11:06:29 +0800 Subject: [PATCH 24/78] Allow selective internal tensor capture for precision diagnosis --- tools/validate_v41_swa_segment.py | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/tools/validate_v41_swa_segment.py b/tools/validate_v41_swa_segment.py index 6939e240..ea18e838 100644 --- a/tools/validate_v41_swa_segment.py +++ b/tools/validate_v41_swa_segment.py @@ -99,6 +99,8 @@ def main(): parser.add_argument("--stage-reference", action="store_true", help="Also localize errors on each half-layer's device input, after the full run") parser.add_argument("--compare-only", help="Recheck a saved comparison.pt on CPU without compilation/device use") + parser.add_argument("--dump-tagged", action="store_true", + help="Enable selective runtime dumps from a separately instrumented lib checkout") parser.add_argument("--artifact-dir", default=".validation-artifacts/swa-segment") parser.add_argument("--ring-heap-mib", type=int, default=1024, help="Per-ring temporary heap; lib MoE validation uses 1024 MiB") @@ -126,7 +128,8 @@ def main(): compare_saved(torch.load(args.compare_only, map_location="cpu", weights_only=True), moe, topology) return config = RunConfig(platform="a5", distributed_config=DistributedConfig(device_ids=devices), - ring_heap=args.ring_heap_mib << 20, ring_task_window=131072, ring_dep_pool=131072) + ring_heap=args.ring_heap_mib << 20, ring_task_window=131072, ring_dep_pool=131072, + enable_dump_args=1 if args.dump_tagged else 0) compiler = KernelCompiler(run_config=config, cache_dir=args.build_dir) programs = compile_segment(compiler, args.lib_root, topology) print("COMPILE PASS", flush=True) From 5db84429b64aa26e98e65a50e10ab3dc4495e1fa Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Mon, 28 Sep 2026 11:09:41 +0800 Subject: [PATCH 25/78] Support first-layer reference capture for matching device dumps --- tools/diagnose_v41_swa_precision.py | 12 +++++++----- 1 file changed, 7 insertions(+), 5 deletions(-) diff --git a/tools/diagnose_v41_swa_precision.py b/tools/diagnose_v41_swa_precision.py index 797e4b0b..ea056ebf 100644 --- a/tools/diagnose_v41_swa_precision.py +++ b/tools/diagnose_v41_swa_precision.py @@ -82,6 +82,7 @@ def main(): parser.add_argument("--output", required=True) parser.add_argument("--trace-attention", action="store_true", help="Use the saved full-reference replay to bisect layer-1 rank-0 attention") + parser.add_argument("--trace-layer", type=int, choices=(0, 1), default=1) parser.add_argument("--cut-after", type=int, choices=(0, 1, 2), help="Restart the CPU reference from a saved device boundary; diagnostic only") args = parser.parse_args() @@ -101,12 +102,13 @@ def main(): if isinstance(s, TensorSpec)} if args.trace_attention: full = torch.load(args.output, map_location="cpu", weights_only=True) - print("Loading layer 1 for attention trace", flush=True) - aw, unused_moe = load_swa_layer_weights(args.model_dir, 1, topology) + layer = args.trace_layer + print(f"Loading layer {layer} for attention trace", flush=True) + aw, unused_moe = load_swa_layer_weights(args.model_dir, layer, topology) del unused_moe - traces = trace_attention(swa, dict(a, **aw), saved["stages"][2]["actual"]["hidden"][0], - full[2]["expected"]["hidden"][0]) - torch.save(traces, str(args.output) + ".attention-trace.pt") + traces = trace_attention(swa, dict(a, **aw), saved["stages"][2 * layer]["actual"]["hidden"][0], + full[2 * layer]["expected"]["hidden"][0]) + torch.save(traces, str(args.output) + f".attention-trace-layer{layer}.pt") return residual, mix = a["x_hc"], a["incoming_pre_mix"] records = [] From bc0023ac6bb73e0a58700611ce71df06331627f3 Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Mon, 28 Sep 2026 11:17:25 +0800 Subject: [PATCH 26/78] Preserve per-layer runtime dumps for dataflow bisection --- tools/validate_v41_swa_segment.py | 12 ++++++++++++ 1 file changed, 12 insertions(+) diff --git a/tools/validate_v41_swa_segment.py b/tools/validate_v41_swa_segment.py index ea18e838..8ce8edc5 100644 --- a/tools/validate_v41_swa_segment.py +++ b/tools/validate_v41_swa_segment.py @@ -199,6 +199,18 @@ def upload(values): for layer_id, (da, dm) in enumerate(device_layers): state = runner.run_layer(state, da, dm, group_counts=[topology.capacity] * topology.dp) print(f"Device layer {layer_id} complete", flush=True) + if args.dump_tagged: + # Preserve completed runtime dumps before the next dispatch reuses its path. + import shutil + + destination = Path(args.artifact_dir) / f"layer-{layer_id}-dumps" + for manifest in Path(args.build_dir).rglob("args_dump.json"): + relative = manifest.parent.relative_to(Path(args.build_dir)) + target = destination / relative + target.parent.mkdir(parents=True, exist_ok=True) + if target.exists(): + raise FileExistsError(f"Refusing to overwrite diagnostic dumps: {target}") + shutil.move(str(manifest.parent), str(target)) worker.copy_stacked_from(state.residual, readback) worker.copy_stacked_from(state.pre_mix, mix_readback) # Read only after both layers finish; never feed diagnostic state back. From 466b5c045dbd6428e053c2e4901c96a9e2af7fe5 Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Mon, 28 Sep 2026 14:36:38 +0800 Subject: [PATCH 27/78] Fix dump traversal before moving completed layer artifacts --- tools/validate_v41_swa_segment.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tools/validate_v41_swa_segment.py b/tools/validate_v41_swa_segment.py index 8ce8edc5..f3f3680e 100644 --- a/tools/validate_v41_swa_segment.py +++ b/tools/validate_v41_swa_segment.py @@ -204,7 +204,7 @@ def upload(values): import shutil destination = Path(args.artifact_dir) / f"layer-{layer_id}-dumps" - for manifest in Path(args.build_dir).rglob("args_dump.json"): + for manifest in list(Path(args.build_dir).rglob("args_dump.json")): relative = manifest.parent.relative_to(Path(args.build_dir)) target = destination / relative target.parent.mkdir(parents=True, exist_ok=True) From 83f7a1751a5c9536746e578ae3f9ab2a3f03f753 Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Mon, 28 Sep 2026 14:46:05 +0800 Subject: [PATCH 28/78] Add independent attention reference precision sensitivity replay --- tools/diagnose_v41_swa_precision.py | 16 ++++++++++++++++ 1 file changed, 16 insertions(+) diff --git a/tools/diagnose_v41_swa_precision.py b/tools/diagnose_v41_swa_precision.py index ea056ebf..a8439a98 100644 --- a/tools/diagnose_v41_swa_precision.py +++ b/tools/diagnose_v41_swa_precision.py @@ -82,6 +82,8 @@ def main(): parser.add_argument("--output", required=True) parser.add_argument("--trace-attention", action="store_true", help="Use the saved full-reference replay to bisect layer-1 rank-0 attention") + parser.add_argument("--reference-fp64-attention", action="store_true", + help="Diagnostic reference sensitivity only; keep the original gate result") parser.add_argument("--trace-layer", type=int, choices=(0, 1), default=1) parser.add_argument("--cut-after", type=int, choices=(0, 1, 2), help="Restart the CPU reference from a saved device boundary; diagnostic only") @@ -95,6 +97,12 @@ def main(): torch.set_num_threads(4) topology = SegmentTopology(tp=2, dp=2) swa, moe = load_segment_modules(args.lib_root, topology) + if args.reference_fp64_attention: + from models.deepseek_v4_1_flash import decode_attn_swa + + original_reference = decode_attn_swa.official_reference + decode_attn_swa.official_reference = lambda tensors: original_reference( + tensors, attention_dtype=torch.float64) saved = torch.load(args.saved, map_location="cpu", weights_only=True) fixture = SimpleNamespace(tokens=32, requests=1, dp=2, seed=11, case="normal", fixture="checkpoint", dp_tokens=None, epochs=1, bench=False) @@ -148,6 +156,14 @@ def main(): torch.save(result, str(args.output) + f".cut-{args.cut_after}.pt") compare_saved(result, moe, topology) return + if args.reference_fp64_attention: + from validate_v41_swa_segment import compare_saved + + result = {"actual_residual": saved["actual_residual"], "expected_residual": residual, + "actual_pre_mix": saved["actual_pre_mix"], "expected_pre_mix": mix} + torch.save(result, str(args.output) + ".fp64-attention.pt") + compare_saved(result, moe, topology) + return print("Replay matches saved reference:", torch.equal(residual, saved["expected_residual"]), torch.equal(mix, saved["expected_pre_mix"]), flush=True) assert torch.equal(residual, saved["expected_residual"]) From 80eb504d3fa12a28f1259819bd3e25522e5d8578 Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Mon, 28 Sep 2026 15:28:53 +0800 Subject: [PATCH 29/78] Validate SWA chain with checkpoint embedding input and replayable initial state --- .../deepseek_v41/test_input_preparation.py | 17 ++++++++++ tools/diagnose_v41_swa_precision.py | 3 ++ tools/validate_v41_swa_segment.py | 32 +++++++++++++++++-- 3 files changed, 50 insertions(+), 2 deletions(-) diff --git a/tests/unit/model/deepseek_v41/test_input_preparation.py b/tests/unit/model/deepseek_v41/test_input_preparation.py index 059eb059..48c886a1 100644 --- a/tests/unit/model/deepseek_v41/test_input_preparation.py +++ b/tests/unit/model/deepseek_v41/test_input_preparation.py @@ -130,3 +130,20 @@ def test_loader_budget_is_preserved(embedding_checkpoint): def test_invalid_limits(embedding_checkpoint, options): with pytest.raises(ValueError, match="positive integer"): lookup_token_embeddings(embedding_checkpoint[0], torch.tensor([0]), **options) + +def test_segment_embedding_control_has_identity_hc_state(embedding_checkpoint): + from types import SimpleNamespace + from tools.validate_v41_swa_segment import prepare_checkpoint_inputs + + loaders, table = embedding_checkpoint + topology = SimpleNamespace(tp=2, dp=1, world=2, capacity=4, local_capacity=2) + tensors = {"x_hc": torch.empty(2, 2, 4, 32), + "incoming_pre_mix": torch.full((2, 2, 4), 17.0)} + ids = prepare_checkpoint_inputs(tensors, loaders[0].model_dir, topology) + expected = table[ids].reshape(2, 2, 32).float() + collapsed = (tensors["x_hc"] * tensors["incoming_pre_mix"].unsqueeze(-1)).sum(2) + assert torch.equal(collapsed, expected) + assert torch.equal(tensors["x_hc"][:, :, 3], expected) + assert tensors["x_hc"].dtype == torch.float32 and tensors["x_hc"].is_contiguous() + tensors["x_hc"][0, 0, 0].zero_() + assert torch.equal(tensors["x_hc"][0, 0, 1], expected[0, 0]) diff --git a/tools/diagnose_v41_swa_precision.py b/tools/diagnose_v41_swa_precision.py index a8439a98..4e108e85 100644 --- a/tools/diagnose_v41_swa_precision.py +++ b/tools/diagnose_v41_swa_precision.py @@ -108,6 +108,9 @@ def main(): fixture="checkpoint", dp_tokens=None, epochs=1, bench=False) a = {s.name: s.create_tensor().contiguous() for s in swa.build_hc_specs(fixture) if isinstance(s, TensorSpec)} + if "initial_state" in saved: + a["x_hc"] = saved["initial_state"]["residual"].clone() + a["incoming_pre_mix"] = saved["initial_state"]["pre_mix"].clone() if args.trace_attention: full = torch.load(args.output, map_location="cpu", weights_only=True) layer = args.trace_layer diff --git a/tools/validate_v41_swa_segment.py b/tools/validate_v41_swa_segment.py index f3f3680e..055db32e 100644 --- a/tools/validate_v41_swa_segment.py +++ b/tools/validate_v41_swa_segment.py @@ -20,6 +20,25 @@ MOE_OUTPUTS = ("x_next", "next_pre_mix", "x_mixed") +def prepare_checkpoint_inputs(tensors, model_dir, topology): + """Prepare an embedding-broadcast control; retain the random stress fixture separately.""" + import torch + from pypto_serving.model.deepseek_v41.input_preparation import lookup_token_embeddings + from pypto_serving.model.deepseek_v41.weight_loader import V41WeightLoader + + loaders = [V41WeightLoader(model_dir, tp_size=topology.tp, tp_rank=rank, + ep_size=topology.world, ep_rank=rank, max_load_bytes=512 << 20) + for rank in range(topology.tp)] + ids = torch.arange(1, topology.dp * topology.capacity + 1, dtype=torch.int64).reshape( + topology.dp, topology.capacity) + embeddings = lookup_token_embeddings(loaders, ids) + rows = embeddings.reshape(topology.world, topology.local_capacity, -1) + tensors["x_hc"] = rows.unsqueeze(2).expand_as(tensors["x_hc"]).float().contiguous() + tensors["incoming_pre_mix"].zero_() + tensors["incoming_pre_mix"][..., 0] = 1 + return ids + + def compare_stages(layers, captured, initial, swa, moe, topology): """Localize errors against each half-layer's actual input, after execution.""" from golden.validation import ratio_allclose @@ -94,6 +113,8 @@ def main(): parser.add_argument("--tp", type=int, default=2) parser.add_argument("--build-dir", default="build_output/v41-swa-segment") parser.add_argument("--compile-only", action="store_true") + parser.add_argument("--input-source", choices=("stress", "embeddings"), default="stress", + help="Keep the random stress input, or load real embedding rows and broadcast HC streams") parser.add_argument("--model-dir", help="Use actual checkpoint weights for layers 0 and 1") parser.add_argument("--reference", action="store_true", help="Compare against composed Torch references") parser.add_argument("--stage-reference", action="store_true", @@ -105,6 +126,8 @@ def main(): parser.add_argument("--ring-heap-mib", type=int, default=1024, help="Per-ring temporary heap; lib MoE validation uses 1024 MiB") args = parser.parse_args() + if args.input_source == "embeddings" and not args.model_dir: + parser.error("--input-source embeddings requires --model-dir") if args.stage_reference: args.reference = True sys.path.insert(0, str(Path(args.lib_root).resolve())) @@ -144,7 +167,11 @@ def materialize(specs): return {s.name: s.create_tensor().contiguous() for s in specs if isinstance(s, TensorSpec)} a = materialize(swa.build_hc_specs(fixture)) - print("Attention fixture ready", flush=True) + token_ids = None + if args.input_source == "embeddings": + token_ids = prepare_checkpoint_inputs(a, args.model_dir, topology) + initial_state = {"residual": a["x_hc"].clone(), "pre_mix": a["incoming_pre_mix"].clone()} + print(f"Attention fixture ready: input_source={args.input_source}", flush=True) if args.model_dir: from pypto_serving.model.deepseek_v41.swa_weights import load_swa_layer_weights @@ -242,7 +269,8 @@ def upload(values): print(f"Torch reference layer {layer_id} complete", flush=True) artifact = Path(args.artifact_dir) artifact.mkdir(parents=True, exist_ok=True) - data = {"actual_residual": readback, "expected_residual": residual, + data = {"input_source": args.input_source, "token_ids": token_ids, "initial_state": initial_state, + "actual_residual": readback, "expected_residual": residual, "actual_pre_mix": mix_readback, "expected_pre_mix": mix} torch.save(data, artifact / "comparison.pt") if captured: From 8d029500bd5f9577ea49f3c975ae9174f210b0de Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Mon, 28 Sep 2026 15:37:44 +0800 Subject: [PATCH 30/78] test(v41): allow explicit token IDs in embedding precision control --- .../deepseek_v41/test_input_preparation.py | 15 +++++++++++++++ tools/validate_v41_swa_segment.py | 19 +++++++++++++++---- 2 files changed, 30 insertions(+), 4 deletions(-) diff --git a/tests/unit/model/deepseek_v41/test_input_preparation.py b/tests/unit/model/deepseek_v41/test_input_preparation.py index 48c886a1..0fdd78db 100644 --- a/tests/unit/model/deepseek_v41/test_input_preparation.py +++ b/tests/unit/model/deepseek_v41/test_input_preparation.py @@ -147,3 +147,18 @@ def test_segment_embedding_control_has_identity_hc_state(embedding_checkpoint): assert tensors["x_hc"].dtype == torch.float32 and tensors["x_hc"].is_contiguous() tensors["x_hc"][0, 0, 0].zero_() assert torch.equal(tensors["x_hc"][0, 0, 1], expected[0, 0]) + + +def test_segment_embedding_control_preserves_supplied_dp_tokens(embedding_checkpoint): + from types import SimpleNamespace + from tools.validate_v41_swa_segment import prepare_checkpoint_inputs + + loaders, table = embedding_checkpoint + topology = SimpleNamespace(tp=2, dp=1, world=2, capacity=4, local_capacity=2) + tensors = {"x_hc": torch.empty(2, 2, 4, 32), "incoming_pre_mix": torch.empty(2, 2, 4)} + ids = prepare_checkpoint_inputs(tensors, loaders[0].model_dir, topology, [[255, 1, 128, 255]]) + assert ids.tolist() == [[255, 1, 128, 255]] + assert torch.equal(tensors["x_hc"][:, :, 0], table[ids].reshape(2, 2, 32).float()) + for invalid in ([[1, 2]], [[1.0, 2.0, 3.0, 4.0]], [[True] * 4]): + with pytest.raises(ValueError, match="full integer token row"): + prepare_checkpoint_inputs(tensors, loaders[0].model_dir, topology, invalid) diff --git a/tools/validate_v41_swa_segment.py b/tools/validate_v41_swa_segment.py index 055db32e..31bb6876 100644 --- a/tools/validate_v41_swa_segment.py +++ b/tools/validate_v41_swa_segment.py @@ -20,17 +20,22 @@ MOE_OUTPUTS = ("x_next", "next_pre_mix", "x_mixed") -def prepare_checkpoint_inputs(tensors, model_dir, topology): +def prepare_checkpoint_inputs(tensors, model_dir, topology, token_ids=None): """Prepare an embedding-broadcast control; retain the random stress fixture separately.""" import torch from pypto_serving.model.deepseek_v41.input_preparation import lookup_token_embeddings from pypto_serving.model.deepseek_v41.weight_loader import V41WeightLoader + if token_ids is None: + ids = torch.arange(1, topology.dp * topology.capacity + 1, dtype=torch.int64).reshape( + topology.dp, topology.capacity) + else: + ids = torch.as_tensor(token_ids) + if ids.dtype not in (torch.int32, torch.int64) or ids.shape != (topology.dp, topology.capacity): + raise ValueError("token_ids must contain exactly one full integer token row per DP group") loaders = [V41WeightLoader(model_dir, tp_size=topology.tp, tp_rank=rank, ep_size=topology.world, ep_rank=rank, max_load_bytes=512 << 20) for rank in range(topology.tp)] - ids = torch.arange(1, topology.dp * topology.capacity + 1, dtype=torch.int64).reshape( - topology.dp, topology.capacity) embeddings = lookup_token_embeddings(loaders, ids) rows = embeddings.reshape(topology.world, topology.local_capacity, -1) tensors["x_hc"] = rows.unsqueeze(2).expand_as(tensors["x_hc"]).float().contiguous() @@ -115,6 +120,7 @@ def main(): parser.add_argument("--compile-only", action="store_true") parser.add_argument("--input-source", choices=("stress", "embeddings"), default="stress", help="Keep the random stress input, or load real embedding rows and broadcast HC streams") + parser.add_argument("--token-ids", help="JSON array [DP, capacity] of embedding control IDs; no padding") parser.add_argument("--model-dir", help="Use actual checkpoint weights for layers 0 and 1") parser.add_argument("--reference", action="store_true", help="Compare against composed Torch references") parser.add_argument("--stage-reference", action="store_true", @@ -128,6 +134,8 @@ def main(): args = parser.parse_args() if args.input_source == "embeddings" and not args.model_dir: parser.error("--input-source embeddings requires --model-dir") + if args.token_ids and args.input_source != "embeddings": + parser.error("--token-ids requires --input-source embeddings") if args.stage_reference: args.reference = True sys.path.insert(0, str(Path(args.lib_root).resolve())) @@ -169,7 +177,10 @@ def materialize(specs): a = materialize(swa.build_hc_specs(fixture)) token_ids = None if args.input_source == "embeddings": - token_ids = prepare_checkpoint_inputs(a, args.model_dir, topology) + import json + + supplied_ids = json.loads(Path(args.token_ids).read_text(encoding="utf-8")) if args.token_ids else None + token_ids = prepare_checkpoint_inputs(a, args.model_dir, topology, supplied_ids) initial_state = {"residual": a["x_hc"].clone(), "pre_mix": a["incoming_pre_mix"].clone()} print(f"Attention fixture ready: input_source={args.input_source}", flush=True) if args.model_dir: From 6b4b8737f0f8e7cb54da778bbe01f7722c9d3eb4 Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Mon, 28 Sep 2026 15:46:16 +0800 Subject: [PATCH 31/78] fix(v41): compare decoded FP8 values in precision diagnostics --- docs/developer-guide/v41-swa-segment.md | 33 ++++++++++++++++++++++--- tests/unit/test_v41_swa_validation.py | 14 +++++++++++ tools/diagnose_v41_swa_precision.py | 22 ++++++++++++++++- 3 files changed, 65 insertions(+), 4 deletions(-) diff --git a/docs/developer-guide/v41-swa-segment.md b/docs/developer-guide/v41-swa-segment.md index c07c9d04..90540cc2 100644 --- a/docs/developer-guide/v41-swa-segment.md +++ b/docs/developer-guide/v41-swa-segment.md @@ -55,9 +55,18 @@ PYTHONPATH=. python tools/validate_v41_swa_segment.py \ --model-dir /path/to/DeepSeek-V4.1-Flash --reference --ring-heap-mib 4096 ``` -The diagnostic uses controlled activations and request metadata, including in -checkpoint mode. It does not yet validate embedding initialization, real prompt -semantics, Engram or generation. The reference runs only after the device worker +The default `--input-source stress` uses controlled random HC activations and +request metadata, including in checkpoint mode. `--input-source embeddings` +instead loads checkpoint embedding rows, broadcasts each row to four HC lanes, +and uses an identity pre-mix selecting lane zero. By default it selects sequential +token IDs. Supply `--token-ids /path/to/ids.json` for an integer JSON array shaped +`[DP, capacity]`, such as tokenizer-produced full token slabs. IDs are not padded, +repeated or truncated by the diagnostic. Both controls retain fixture RoPE/pages; +neither establishes complete prompt semantics, Engram or generation. The selected +IDs and exact initial state are saved for CPU replay. This diagnostic input setup +does not enable the production composite adapter. + +The reference runs only after the device worker closes, never supplies intermediate device inputs, and shares read-only weights with the device path. Final actual/expected residual and pre_mix are saved to `--artifact-dir/comparison.pt` even when the numerical comparison fails. The @@ -84,6 +93,24 @@ check still failed with the errors above. Agreement on each stage's actual input does not establish the accumulated numerical budget across layers; that boundary still needs validation before full-model acceptance. +The baseline embedding control also fails the unchanged accumulated gate: +residual relative L2 is 0.01316 and pre-mix relative L2 is 0.00281, while the native +half-layer checks pass. Its rank-zero outlier fraction is 3.896%, versus 32.572% +on the original random stress input. The lower outlier fraction does not mean +the relative L2 improved: these are distinct measurements on distinct workloads. +The original stress failure remains an unresolved regression case. + +For precision bisection, follow lib's `docs/debug-and-tune/precision-tuning.md` +and PyPTO's `docs/en/user/precision/00-workflow.md`. Check dtype, rounding and +reference operation order before changing kernels. The full-chain gate currently +reuses a single-MoE comparator; it is not an agreed model-wide error budget. +Report relative L2, maximum absolute error and outlier fraction separately, and +retain the existing failing gate while the accumulated contract is unresolved. +FP8 trace comparisons decode payloads with their own scales; different encoding +pairs can represent identical values. Local checks on actual device inputs and +CPU boundary substitutions only localize errors, never replace full-chain +acceptance or supply intermediate values to device execution. + To recheck a saved final comparison without compiling or allocating devices: ```bash diff --git a/tests/unit/test_v41_swa_validation.py b/tests/unit/test_v41_swa_validation.py index 512f4ed9..efc11994 100644 --- a/tests/unit/test_v41_swa_validation.py +++ b/tests/unit/test_v41_swa_validation.py @@ -14,6 +14,7 @@ import torch from tools.validate_v41_swa_segment import compare_saved +from tools.diagnose_v41_swa_precision import quantization_metrics @pytest.mark.parametrize("failure", [None, "residual", "pre_mix", "stage"]) @@ -40,3 +41,16 @@ def comparator(actual, expected, *, actual_outputs, expected_outputs, inputs, rt else: with pytest.raises(AssertionError): compare_saved(data, moe, topology) + + +def test_quantization_probe_compares_values_instead_of_payload_codes(): + payload = torch.ones(2, 64) + codes = torch.full((2, 2), 127, dtype=torch.uint8) + # Different payload/exponent pairs can represent exactly the same values. + same = quantization_metrics(payload, codes, payload / 2, codes + 1) + assert same["dequantized"]["rel_l2"] == 0 + assert same["payload_changed"] == 128 and same["scale_changed"] == 4 + # Identical payloads are not equal physical values when exponents differ. + different = quantization_metrics(payload, codes + 1, payload, codes) + assert different["payload_changed"] == 0 + assert different["dequantized"]["rel_l2"] == 1 diff --git a/tools/diagnose_v41_swa_precision.py b/tools/diagnose_v41_swa_precision.py index 4e108e85..80a22fc3 100644 --- a/tools/diagnose_v41_swa_precision.py +++ b/tools/diagnose_v41_swa_precision.py @@ -8,7 +8,7 @@ # ----------------------------------------------------------------------------------------------------------- """CPU replay of a saved two-layer device trace to bisect accumulated error. -Uses the original seed-11 full-prefix fixture and checkpoint layers 0/1. +Uses the saved initial state (or legacy seed-11 fixture) and checkpoint layers 0/1. No device execution, production operator composition or tolerance changes. """ import argparse @@ -31,6 +31,19 @@ def metrics(actual, expected): "equal_fraction": float((a == e).double().mean())} +def quantization_metrics(actual_payload, actual_codes, expected_payload, expected_codes): + """Compare physical FP8 values; payload distance alone ignores the shared exponent.""" + import torch + + def decode(payload, codes): + return payload.float().unflatten(-1, (-1, 32)) * torch.exp2(codes.float() - 127).unsqueeze(-1) + + return {"dequantized": metrics(decode(actual_payload, actual_codes), + decode(expected_payload, expected_codes)), + "payload_changed": int((actual_payload != expected_payload).sum()), + "scale_changed": int((actual_codes != expected_codes).sum())} + + def trace_attention(swa, tensors, actual_hidden, reference_hidden): """Bisect the second attention on CPU, recording its public reference calls.""" from models.deepseek_v4_1_flash import decode_attn_swa as ref @@ -70,6 +83,13 @@ def rope(x, cos, sin, inverse=False): finally: ref.official_linear, ref.official_rope = original_linear, original_rope for name in traces[0]: + if name.endswith(".scale"): + continue + if name.endswith(".quant"): + scale = name.removesuffix(".quant") + ".scale" + print("QUANT_TRACE", name, json.dumps(quantization_metrics( + traces[0][name], traces[0][scale], traces[1][name], traces[1][scale])), flush=True) + continue print("TRACE", name, json.dumps(metrics(traces[0][name], traces[1][name])), flush=True) return traces From 70191bed21c5f5c3b4f3217fcf9cd070c90b6c5d Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Mon, 28 Sep 2026 15:50:21 +0800 Subject: [PATCH 32/78] docs(v41): record text embedding precision control and remaining failure --- docs/developer-guide/v41-swa-segment.md | 24 ++++++++++++++++++------ 1 file changed, 18 insertions(+), 6 deletions(-) diff --git a/docs/developer-guide/v41-swa-segment.md b/docs/developer-guide/v41-swa-segment.md index 90540cc2..73e2a286 100644 --- a/docs/developer-guide/v41-swa-segment.md +++ b/docs/developer-guide/v41-swa-segment.md @@ -93,12 +93,24 @@ check still failed with the errors above. Agreement on each stage's actual input does not establish the accumulated numerical budget across layers; that boundary still needs validation before full-model acceptance. -The baseline embedding control also fails the unchanged accumulated gate: -residual relative L2 is 0.01316 and pre-mix relative L2 is 0.00281, while the native -half-layer checks pass. Its rank-zero outlier fraction is 3.896%, versus 32.572% -on the original random stress input. The lower outlier fraction does not mean -the relative L2 improved: these are distinct measurements on distinct workloads. -The original stress failure remains an unresolved regression case. +The baseline embedding controls also fail the unchanged accumulated gate, while +their 18 native half-layer checks pass: + +| Initial state | Rank-zero outliers | Residual relative L2 | Pre-mix relative L2 | +| --- | ---: | ---: | ---: | +| Independent random HC streams | 32.572% | 0.00760 | 0.00105 | +| Sequential checkpoint embedding rows | 3.896% | 0.01316 | 0.00281 | +| Checkpoint embeddings for text token prefixes | 4.094% | 0.01348 | 0.00284 | + +The text control uses 32 tokens per DP group with fixture metadata. The lower +outlier fraction does not mean relative L2 improved: these are distinct +measurements on distinct workloads. The original stress failure remains an +unresolved regression case. On the sequential embedding control, CPU replay +matches the saved full reference bitwise; accumulated residual relative L2 at +Attention 0, MoE 0, Attention 1, and MoE 1 is 0.00101, 0.00353, 0.01029, and +0.01316 respectively. CPU router replay on inherited device inputs changes four +second-layer expert sets in the random control and none in this embedding +control. Routing changes alone therefore do not explain the chain failure. For precision bisection, follow lib's `docs/debug-and-tune/precision-tuning.md` and PyPTO's `docs/en/user/precision/00-workflow.md`. Check dtype, rounding and From 85f3362993d086f8492eba83870a7b76821deecc Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Mon, 28 Sep 2026 17:05:23 +0800 Subject: [PATCH 33/78] test(v41): adopt DSV4 layer residual budget and retain legacy profile --- docs/developer-guide/v41-swa-segment.md | 24 ++++++++++++---- tests/unit/test_v41_swa_validation.py | 34 +++++++++++++++++++++-- tools/diagnose_v41_swa_precision.py | 5 ++-- tools/validate_v41_swa_segment.py | 37 ++++++++++++++++++------- 4 files changed, 80 insertions(+), 20 deletions(-) diff --git a/docs/developer-guide/v41-swa-segment.md b/docs/developer-guide/v41-swa-segment.md index 73e2a286..3121a393 100644 --- a/docs/developer-guide/v41-swa-segment.md +++ b/docs/developer-guide/v41-swa-segment.md @@ -70,8 +70,11 @@ The reference runs only after the device worker closes, never supplies intermediate device inputs, and shares read-only weights with the device path. Final actual/expected residual and pre_mix are saved to `--artifact-dir/comparison.pt` even when the numerical comparison fails. The -residual gate uses lib's local MoE relative-error comparator; pre_mix uses -rtol=0.01 and atol=0.0001. A completed smoke is not a numerical pass. +residual gate defaults to `--residual-profile dsv4-layer`: the DSV4 complete-layer +comparator `ratio_reldiff(diff_thd=0.01, pct_thd=0.05)`, applied separately to each +rank. `--residual-profile v41-local` retains the historical V4.1 single-MoE +comparator (0.003/2%, with its single-point cap). Pre_mix retains rtol=0.01 and +atol=0.0001; local stage gates are unchanged. A completed smoke is not a numerical pass. The real-weight TP2/DP2/EP4 diagnostic exhausted the temporary heap at both 512 MiB and 1024 MiB per ring. With 4096 MiB per ring, both device layers and @@ -93,7 +96,7 @@ check still failed with the errors above. Agreement on each stage's actual input does not establish the accumulated numerical budget across layers; that boundary still needs validation before full-model acceptance. -The baseline embedding controls also fail the unchanged accumulated gate, while +Under the historical `v41-local` profile, the baseline embedding controls fail the accumulated gate, while their 18 native half-layer checks pass: | Initial state | Rank-zero outliers | Residual relative L2 | Pre-mix relative L2 | @@ -114,15 +117,24 @@ control. Routing changes alone therefore do not explain the chain failure. For precision bisection, follow lib's `docs/debug-and-tune/precision-tuning.md` and PyPTO's `docs/en/user/precision/00-workflow.md`. Check dtype, rounding and -reference operation order before changing kernels. The full-chain gate currently -reuses a single-MoE comparator; it is not an agreed model-wide error budget. +reference operation order before changing kernels. The historical full-chain gate +reused a single-MoE comparator; it was not an agreed model-wide error budget. Report relative L2, maximum absolute error and outlier fraction separately, and -retain the existing failing gate while the accumulated contract is unresolved. +retain the historical results when changing acceptance profiles. FP8 trace comparisons decode payloads with their own scales; different encoding pairs can represent identical values. Local checks on actual device inputs and CPU boundary substitutions only localize errors, never replace full-chain acceptance or supply intermediate values to device execution. +Rechecking those same saved tensors with `dsv4-layer` passes residual on every +rank. Worst-rank outlier fractions are 3.3542% (random streams), 0.02167% +(sequential embeddings), and 0.12879% (text embeddings), below the 5% budget. +This is a requested acceptance-profile change, not reduced numerical error. +Pre_mix still fails in 14/256, 30/256 and 24/256 entries respectively, so the +overall diagnostic still fails. DSV4's layer comparison is not an independent +43-layer accuracy guarantee, and it does not specify V4.1's delayed pre_mix +contract. These results do not establish complete model or M0 acceptance. + To recheck a saved final comparison without compiling or allocating devices: ```bash diff --git a/tests/unit/test_v41_swa_validation.py b/tests/unit/test_v41_swa_validation.py index efc11994..0853450b 100644 --- a/tests/unit/test_v41_swa_validation.py +++ b/tests/unit/test_v41_swa_validation.py @@ -37,10 +37,40 @@ def comparator(actual, expected, *, actual_outputs, expected_outputs, inputs, rt moe = SimpleNamespace(_local_mhc_compare=lambda counts: comparator) topology = SimpleNamespace(local_capacity=2, world=1) if failure is None: - compare_saved(data, moe, topology) + compare_saved(data, moe, topology, "v41-local") else: with pytest.raises(AssertionError): - compare_saved(data, moe, topology) + compare_saved(data, moe, topology, "v41-local") + + +@pytest.mark.parametrize("failure", [None, "rank", "pre_mix", "stage"]) +def test_dsv4_profile_keeps_per_rank_and_auxiliary_gates(monkeypatch, failure): + import sys + + calls = [] + def factory(**settings): + assert settings == {"diff_thd": 0.01, "pct_thd": 0.05} + def comparator(actual, expected, **kwargs): + calls.append(actual.shape) + return torch.equal(actual, expected), "rank mismatch" + return comparator + monkeypatch.setitem(sys.modules, "golden.validation", SimpleNamespace(ratio_reldiff=factory)) + actual = torch.ones(2, 2, 4, 3) + data = {"actual_residual": actual, "expected_residual": actual.clone(), + "actual_pre_mix": torch.ones(2, 2, 4), "expected_pre_mix": torch.ones(2, 2, 4)} + if failure == "rank": + actual[1].add_(1) + elif failure == "pre_mix": + data["actual_pre_mix"].add_(1) + elif failure == "stage": + data["stages"] = [{"results": {"output": (False, "stage mismatch")}}] + topology = SimpleNamespace(local_capacity=2, world=2) + if failure is None: + compare_saved(data, None, topology) + else: + with pytest.raises(AssertionError): + compare_saved(data, None, topology) + assert calls == [(2, 4, 3), (2, 4, 3)] def test_quantization_probe_compares_values_instead_of_payload_codes(): diff --git a/tools/diagnose_v41_swa_precision.py b/tools/diagnose_v41_swa_precision.py index 80a22fc3..9f946d4a 100644 --- a/tools/diagnose_v41_swa_precision.py +++ b/tools/diagnose_v41_swa_precision.py @@ -105,6 +105,7 @@ def main(): parser.add_argument("--reference-fp64-attention", action="store_true", help="Diagnostic reference sensitivity only; keep the original gate result") parser.add_argument("--trace-layer", type=int, choices=(0, 1), default=1) + parser.add_argument("--residual-profile", choices=("dsv4-layer", "v41-local"), default="dsv4-layer") parser.add_argument("--cut-after", type=int, choices=(0, 1, 2), help="Restart the CPU reference from a saved device boundary; diagnostic only") args = parser.parse_args() @@ -177,7 +178,7 @@ def main(): result = {"actual_residual": saved["actual_residual"], "expected_residual": residual, "actual_pre_mix": saved["actual_pre_mix"], "expected_pre_mix": mix} torch.save(result, str(args.output) + f".cut-{args.cut_after}.pt") - compare_saved(result, moe, topology) + compare_saved(result, moe, topology, args.residual_profile) return if args.reference_fp64_attention: from validate_v41_swa_segment import compare_saved @@ -185,7 +186,7 @@ def main(): result = {"actual_residual": saved["actual_residual"], "expected_residual": residual, "actual_pre_mix": saved["actual_pre_mix"], "expected_pre_mix": mix} torch.save(result, str(args.output) + ".fp64-attention.pt") - compare_saved(result, moe, topology) + compare_saved(result, moe, topology, args.residual_profile) return print("Replay matches saved reference:", torch.equal(residual, saved["expected_residual"]), torch.equal(mix, saved["expected_pre_mix"]), flush=True) diff --git a/tools/validate_v41_swa_segment.py b/tools/validate_v41_swa_segment.py index 31bb6876..30891fa2 100644 --- a/tools/validate_v41_swa_segment.py +++ b/tools/validate_v41_swa_segment.py @@ -80,24 +80,38 @@ def compare_stages(layers, captured, initial, swa, moe, topology): return records -def compare_saved(data, moe, topology): - """Apply unchanged lib budgets to live or saved post-run outputs on CPU.""" +def compare_saved(data, moe, topology, residual_profile="dsv4-layer"): + """Apply the selected residual budget, retaining pre-mix and local-stage gates.""" import torch actual = data["actual_residual"] expected = data["expected_residual"] + if actual.shape != expected.shape or actual.shape[0] != topology.world: + raise ValueError("residual shapes must match and contain every logical rank") for name in ("residual", "pre_mix"): a, e = data["actual_" + name].double(), data["expected_" + name].double() error = a - e print(f"{name}: rel_l2={(error.norm() / e.norm().clamp_min(1e-12)).item():.8g} " f"max_abs={error.abs().max().item():.8g}", flush=True) counts = [topology.local_capacity] * topology.world - compare = moe._local_mhc_compare(counts) - ok, message = compare( - actual, expected, actual_outputs={"x_next": actual}, expected_outputs={"x_next": expected}, - inputs={"num_tokens": torch.tensor(counts, dtype=torch.int32)}, rtol=1e-5, atol=1e-5, - ) - print("Final residual reference check:", ok, message, flush=True) + kwargs = dict(actual_outputs={"x_next": actual}, expected_outputs={"x_next": expected}, + inputs={"num_tokens": torch.tensor(counts, dtype=torch.int32)}, rtol=1e-5, atol=1e-5) + if residual_profile == "v41-local": + ok, message = moe._local_mhc_compare(counts)(actual, expected, **kwargs) + elif residual_profile == "dsv4-layer": + from golden.validation import ratio_reldiff + + compare = ratio_reldiff(diff_thd=0.01, pct_thd=0.05) + results = [] + for rank, count in enumerate(counts): + passed, detail = compare(actual[rank, :count], expected[rank, :count], **kwargs) + print(f"Final residual rank={rank} profile={residual_profile}: {passed} {detail}", flush=True) + results.append((passed, detail)) + ok = all(passed for passed, _ in results) + message = "\n".join(detail for passed, detail in results if not passed) + else: + raise ValueError(f"Unknown residual profile: {residual_profile}") + print(f"Final residual reference check ({residual_profile}):", ok, message, flush=True) mix_error = None try: torch.testing.assert_close(data["actual_pre_mix"], data["expected_pre_mix"], rtol=1e-2, atol=1e-4) @@ -126,6 +140,8 @@ def main(): parser.add_argument("--stage-reference", action="store_true", help="Also localize errors on each half-layer's device input, after the full run") parser.add_argument("--compare-only", help="Recheck a saved comparison.pt on CPU without compilation/device use") + parser.add_argument("--residual-profile", choices=("dsv4-layer", "v41-local"), default="dsv4-layer", + help="DSV4 layer residual budget (0.01/5%%), or the historical V4.1 local MoE budget") parser.add_argument("--dump-tagged", action="store_true", help="Enable selective runtime dumps from a separately instrumented lib checkout") parser.add_argument("--artifact-dir", default=".validation-artifacts/swa-segment") @@ -156,7 +172,8 @@ def main(): topology = SegmentTopology(tp=args.tp, dp=len(devices) // args.tp) if args.compare_only: _, moe = load_segment_modules(args.lib_root, topology) - compare_saved(torch.load(args.compare_only, map_location="cpu", weights_only=True), moe, topology) + compare_saved(torch.load(args.compare_only, map_location="cpu", weights_only=True), moe, topology, + args.residual_profile) return config = RunConfig(platform="a5", distributed_config=DistributedConfig(device_ids=devices), ring_heap=args.ring_heap_mib << 20, ring_task_window=131072, ring_dep_pool=131072, @@ -288,7 +305,7 @@ def upload(values): data["stages"] = compare_stages(layers, captured, (a["x_hc"], a["incoming_pre_mix"]), swa, moe, topology) torch.save(data, artifact / "comparison.pt") - compare_saved(data, moe, topology) + compare_saved(data, moe, topology, args.residual_profile) if __name__ == "__main__": From f74847b2851b7e080d3f06b0499605005e9bc064 Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Tue, 29 Sep 2026 01:11:46 +0800 Subject: [PATCH 34/78] test(v41): record approved accumulated pre-mix budget --- docs/developer-guide/v41-swa-segment.md | 19 +++++++++++++++---- tests/unit/test_v41_swa_validation.py | 17 +++++++++++++++++ tools/validate_v41_swa_segment.py | 3 ++- 3 files changed, 34 insertions(+), 5 deletions(-) diff --git a/docs/developer-guide/v41-swa-segment.md b/docs/developer-guide/v41-swa-segment.md index 3121a393..02dd7e7c 100644 --- a/docs/developer-guide/v41-swa-segment.md +++ b/docs/developer-guide/v41-swa-segment.md @@ -73,8 +73,9 @@ with the device path. Final actual/expected residual and pre_mix are saved to residual gate defaults to `--residual-profile dsv4-layer`: the DSV4 complete-layer comparator `ratio_reldiff(diff_thd=0.01, pct_thd=0.05)`, applied separately to each rank. `--residual-profile v41-local` retains the historical V4.1 single-MoE -comparator (0.003/2%, with its single-point cap). Pre_mix retains rtol=0.01 and -atol=0.0001; local stage gates are unchanged. A completed smoke is not a numerical pass. +comparator (0.003/2%, with its single-point cap). Accumulated pre_mix uses the user-approved provisional rtol=0.01 and +atol=0.001, requiring every element to pass; local stage gates are unchanged. +This is limited to this two-layer diagnostic, not a V4/vLLM multi-layer standard. A completed smoke is not a numerical pass. The real-weight TP2/DP2/EP4 diagnostic exhausted the temporary heap at both 512 MiB and 1024 MiB per ring. With 4096 MiB per ring, both device layers and @@ -130,8 +131,8 @@ Rechecking those same saved tensors with `dsv4-layer` passes residual on every rank. Worst-rank outlier fractions are 3.3542% (random streams), 0.02167% (sequential embeddings), and 0.12879% (text embeddings), below the 5% budget. This is a requested acceptance-profile change, not reduced numerical error. -Pre_mix still fails in 14/256, 30/256 and 24/256 entries respectively, so the -overall diagnostic still fails. DSV4's layer comparison is not an independent +With the historical pre_mix atol=0.0001, 14/256, 30/256 and 24/256 entries +respectively failed, so that historical overall diagnostic failed. DSV4's layer comparison is not an independent 43-layer accuracy guarantee, and it does not specify V4.1's delayed pre_mix contract. These results do not establish complete model or M0 acceptance. @@ -142,3 +143,13 @@ PYTHONPATH=. python tools/validate_v41_swa_segment.py \ --lib-root /path/to/current/pypto-lib --tp 2 --devices 0,1,2,3 \ --compare-only /path/to/comparison.pt ``` + + +On 2026-09-28 the user approved provisional accumulated pre_mix tolerances +rtol=0.01, atol=0.001, with no allowed failing elements. CPU rechecks of the +three saved device runs still fail in 2/256 (random), 7/256 (sequential +embeddings), and 7/256 (text embeddings) entries. Residual and all native stage +checks pass, but overall two-layer acceptance remains unresolved. This changes +the acceptance budget only; numerical errors are unchanged. No new NPU run was +performed. Evidence: `.validation-artifacts/approved-premix-budget-recheck.json` +on the A5 validation checkout, using baseline lib `21645633`. diff --git a/tests/unit/test_v41_swa_validation.py b/tests/unit/test_v41_swa_validation.py index 0853450b..53d80241 100644 --- a/tests/unit/test_v41_swa_validation.py +++ b/tests/unit/test_v41_swa_validation.py @@ -84,3 +84,20 @@ def test_quantization_probe_compares_values_instead_of_payload_codes(): different = quantization_metrics(payload, codes + 1, payload, codes) assert different["payload_changed"] == 0 assert different["dequantized"]["rel_l2"] == 1 + + +@pytest.mark.parametrize("error,passes", [(0.0005, True), (0.002, False)]) +def test_provisional_premix_budget_requires_every_element(error, passes): + residual = torch.ones(1, 2, 4, 3) + expected = torch.zeros(1, 2, 4) + actual = expected.clone() + actual[0, 0, 0] = error + data = {"actual_residual": residual, "expected_residual": residual.clone(), + "actual_pre_mix": actual, "expected_pre_mix": expected} + moe = SimpleNamespace(_local_mhc_compare=lambda counts: lambda *args, **kwargs: (True, "")) + topology = SimpleNamespace(local_capacity=2, world=1) + if passes: + compare_saved(data, moe, topology, "v41-local") + else: + with pytest.raises(AssertionError): + compare_saved(data, moe, topology, "v41-local") diff --git a/tools/validate_v41_swa_segment.py b/tools/validate_v41_swa_segment.py index 30891fa2..8bc47cea 100644 --- a/tools/validate_v41_swa_segment.py +++ b/tools/validate_v41_swa_segment.py @@ -114,7 +114,8 @@ def compare_saved(data, moe, topology, residual_profile="dsv4-layer"): print(f"Final residual reference check ({residual_profile}):", ok, message, flush=True) mix_error = None try: - torch.testing.assert_close(data["actual_pre_mix"], data["expected_pre_mix"], rtol=1e-2, atol=1e-4) + # User-approved provisional budget for this two-layer accumulated output only. + torch.testing.assert_close(data["actual_pre_mix"], data["expected_pre_mix"], rtol=1e-2, atol=1e-3) except AssertionError as exc: mix_error = str(exc) print("Final pre_mix reference check:", mix_error is None, mix_error or "", flush=True) From f46d1f8c845313eb83d1d40236b749ad6d3e35c6 Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Tue, 29 Sep 2026 01:17:52 +0800 Subject: [PATCH 35/78] test(v41): probe FP8 projection reference accumulation sensitivity --- tools/diagnose_v41_swa_precision.py | 20 ++++++++++++++++++-- 1 file changed, 18 insertions(+), 2 deletions(-) diff --git a/tools/diagnose_v41_swa_precision.py b/tools/diagnose_v41_swa_precision.py index 9f946d4a..dce77ecb 100644 --- a/tools/diagnose_v41_swa_precision.py +++ b/tools/diagnose_v41_swa_precision.py @@ -104,6 +104,8 @@ def main(): help="Use the saved full-reference replay to bisect layer-1 rank-0 attention") parser.add_argument("--reference-fp64-attention", action="store_true", help="Diagnostic reference sensitivity only; keep the original gate result") + parser.add_argument("--reference-fp64-linear", action="store_true", + help="Accumulate quantized attention projections in FP64 for diagnosis only") parser.add_argument("--trace-layer", type=int, choices=(0, 1), default=1) parser.add_argument("--residual-profile", choices=("dsv4-layer", "v41-local"), default="dsv4-layer") parser.add_argument("--cut-after", type=int, choices=(0, 1, 2), @@ -124,6 +126,18 @@ def main(): original_reference = decode_attn_swa.official_reference decode_attn_swa.official_reference = lambda tensors: original_reference( tensors, attention_dtype=torch.float64) + if args.reference_fp64_linear: + from models.deepseek_v4_1_flash import decode_attn_swa + + def wide_linear(x, weight, packed_scale, fp32=False): + payload, codes = decode_attn_swa.official_quantize(x) + scale_a = decode_attn_swa.decode_e8m0(codes).double().repeat_interleave(32, -1) + scale_b = decode_attn_swa.decode_e8m0( + decode_attn_swa.unpack_mx_b_scale(packed_scale)).double().repeat_interleave(32, 0) + result = ((payload.double() * scale_a) @ (weight.double() * scale_b)).float() + return result if fp32 else result.bfloat16() + + decode_attn_swa.official_linear = wide_linear saved = torch.load(args.saved, map_location="cpu", weights_only=True) fixture = SimpleNamespace(tokens=32, requests=1, dp=2, seed=11, case="normal", fixture="checkpoint", dp_tokens=None, epochs=1, bench=False) @@ -180,12 +194,14 @@ def main(): torch.save(result, str(args.output) + f".cut-{args.cut_after}.pt") compare_saved(result, moe, topology, args.residual_profile) return - if args.reference_fp64_attention: + if args.reference_fp64_attention or args.reference_fp64_linear: from validate_v41_swa_segment import compare_saved result = {"actual_residual": saved["actual_residual"], "expected_residual": residual, "actual_pre_mix": saved["actual_pre_mix"], "expected_pre_mix": mix} - torch.save(result, str(args.output) + ".fp64-attention.pt") + suffix = ".fp64-linear" if args.reference_fp64_linear else "" + suffix += ".fp64-attention" if args.reference_fp64_attention else "" + torch.save(result, str(args.output) + suffix + ".pt") compare_saved(result, moe, topology, args.residual_profile) return print("Replay matches saved reference:", torch.equal(residual, saved["expected_residual"]), From 7aa79183b7c440a7cd2f9a55e8c14bc7d0d5f8dd Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Tue, 29 Sep 2026 01:30:51 +0800 Subject: [PATCH 36/78] test(v41): isolate attention RMSNorm reference reduction order --- tools/diagnose_v41_swa_precision.py | 7 ++++++- 1 file changed, 6 insertions(+), 1 deletion(-) diff --git a/tools/diagnose_v41_swa_precision.py b/tools/diagnose_v41_swa_precision.py index dce77ecb..150aa552 100644 --- a/tools/diagnose_v41_swa_precision.py +++ b/tools/diagnose_v41_swa_precision.py @@ -106,6 +106,8 @@ def main(): help="Diagnostic reference sensitivity only; keep the original gate result") parser.add_argument("--reference-fp64-linear", action="store_true", help="Accumulate quantized attention projections in FP64 for diagnosis only") + parser.add_argument("--reference-kernel-norm", action="store_true", + help="Use the standalone RMSNorm reference's chunk order in attention") parser.add_argument("--trace-layer", type=int, choices=(0, 1), default=1) parser.add_argument("--residual-profile", choices=("dsv4-layer", "v41-local"), default="dsv4-layer") parser.add_argument("--cut-after", type=int, choices=(0, 1, 2), @@ -120,6 +122,8 @@ def main(): torch.set_num_threads(4) topology = SegmentTopology(tp=2, dp=2) swa, moe = load_segment_modules(args.lib_root, topology) + if args.reference_kernel_norm: + swa.golden_rms_norm = moe.golden_rms_norm if args.reference_fp64_attention: from models.deepseek_v4_1_flash import decode_attn_swa @@ -194,13 +198,14 @@ def wide_linear(x, weight, packed_scale, fp32=False): torch.save(result, str(args.output) + f".cut-{args.cut_after}.pt") compare_saved(result, moe, topology, args.residual_profile) return - if args.reference_fp64_attention or args.reference_fp64_linear: + if args.reference_fp64_attention or args.reference_fp64_linear or args.reference_kernel_norm: from validate_v41_swa_segment import compare_saved result = {"actual_residual": saved["actual_residual"], "expected_residual": residual, "actual_pre_mix": saved["actual_pre_mix"], "expected_pre_mix": mix} suffix = ".fp64-linear" if args.reference_fp64_linear else "" suffix += ".fp64-attention" if args.reference_fp64_attention else "" + suffix += ".kernel-norm" if args.reference_kernel_norm else "" torch.save(result, str(args.output) + suffix + ".pt") compare_saved(result, moe, topology, args.residual_profile) return From 71e9929325c1d454f3f8d6e80db769358e0bc057 Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Tue, 29 Sep 2026 01:32:08 +0800 Subject: [PATCH 37/78] feat(v41): pack fresh HC inputs into request-aware TP token slabs --- docs/developer-guide/v41-swa-segment.md | 17 +++ .../model/deepseek_v41/segment_inputs.py | 100 ++++++++++++++++++ .../model/deepseek_v41/test_segment_inputs.py | 72 +++++++++++++ tools/validate_v41_swa_segment.py | 17 ++- 4 files changed, 202 insertions(+), 4 deletions(-) create mode 100644 pypto_serving/model/deepseek_v41/segment_inputs.py create mode 100644 tests/unit/model/deepseek_v41/test_segment_inputs.py diff --git a/docs/developer-guide/v41-swa-segment.md b/docs/developer-guide/v41-swa-segment.md index 02dd7e7c..8cd4835a 100644 --- a/docs/developer-guide/v41-swa-segment.md +++ b/docs/developer-guide/v41-swa-segment.md @@ -153,3 +153,20 @@ checks pass, but overall two-layer acceptance remains unresolved. This changes the acceptance budget only; numerical errors are unchanged. No new NPU run was performed. Evidence: `.validation-artifacts/approved-premix-budget-recheck.json` on the A5 validation checkout, using baseline lib `21645633`. + + +`segment_inputs.prepare_segment_inputs` prepares fresh host HC input buffers +from packed BF16 embeddings and a `ForwardStep`. Like V4 host input preparation, +it maps request rows before device upload. The initial four FP32 lanes and +lane-zero pre-mix follow lib `input_pack.pack_x_hc` and `golden.identity_pre_mix` +at `fbe92bfc`. This is data packing only; it does not run mHC or normalization. + +Requests may interleave DP partitions. Each partition keeps its own stable +packed order, split into contiguous TP token slabs. Returned source-row indices +also define the mapping for positions and other token metadata; returned final +row locations preserve the original request order for later output selection. +Inactive rows have zero residual/pre-mix, including fully empty DP partitions. +The helper rejects over-capacity steps before allocation instead of truncating. +It is used by the embedding diagnostic; cache lowering, device upload and the +complete production adapter remain separate work. Never invoke it between layers +to overwrite the residual/pre-mix produced by the previous composite. diff --git a/pypto_serving/model/deepseek_v41/segment_inputs.py b/pypto_serving/model/deepseek_v41/segment_inputs.py new file mode 100644 index 00000000..8b408b53 --- /dev/null +++ b/pypto_serving/model/deepseek_v41/segment_inputs.py @@ -0,0 +1,100 @@ +# Copyright (c) PyPTO Contributors. +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of +# CANN Open Software License Agreement Version 2.0 (the "License"). +# Please refer to the License for details. You may not use this file except in compliance with the License. +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. +# See LICENSE in the root of the software repository for the full text of the License. +# ----------------------------------------------------------------------------------------------------------- +"""Host input packing for the lib's TP-local-token SWA/MoE boundary. + +V4 prepares host embedding rows before device dispatch. V4.1 input_pack.pack_x_hc +(lib fbe92bfc) establishes four FP32 copies and golden.identity_pre_mix selects +lane zero. These are fresh per-token inputs at the model entrance, never a +replacement for residual/pre_mix between layers. Device upload is caller-owned. +""" +from dataclasses import dataclass + +import torch + +from .request_state import ForwardStep +from .swa_segment import SegmentTopology + + +@dataclass(frozen=True) +class SegmentInputs: + residual: torch.Tensor + pre_mix: torch.Tensor + # Source indices into the original packed request order; -1 means padding. + row_indices: torch.Tensor + # (world rank, local row) in the original request order, for output selection. + request_last_rows: tuple[tuple[int, int], ...] + group_counts: tuple[int, ...] + + +def prepare_segment_inputs( + embeddings: torch.Tensor, + step: ForwardStep, + topology: SegmentTopology, + *, + max_prepare_bytes: int = 256 << 20, +) -> SegmentInputs: + """Pack fresh embedding inputs without depending on request batch order. + + Each DP group contains contiguous local-capacity TP slabs, including empty + slabs. Padding is zero in both residual and pre_mix. row_indices supplies the + same mapping for positions and other token metadata. This does not lower + cache pages or initialize a production all-mode model adapter. + """ + if type(max_prepare_bytes) is not int or max_prepare_bytes <= 0: + raise ValueError("max_prepare_bytes must be a positive integer") + if not isinstance(step, ForwardStep) or not isinstance(topology, SegmentTopology): + raise ValueError("segment input preparation requires a ForwardStep and SegmentTopology") + if (not isinstance(embeddings, torch.Tensor) or embeddings.layout != torch.strided + or embeddings.device.type != "cpu" or embeddings.dtype != torch.bfloat16 + or embeddings.ndim != 2 or embeddings.shape[1] <= 0): + raise ValueError("embeddings must be packed CPU BF16 [active_tokens, hidden_size]") + counts = [0] * topology.dp + source_ranges, last_rows = [], [] + offset = 0 + request_ids = set() + for request in step.requests: + partition, count = request.partition, len(request.token_ids) + if type(partition) is not int or not 0 <= partition < topology.dp or count <= 0: + raise ValueError("requests require valid DP partitions and nonempty token slices") + if request.request_id in request_ids: + raise ValueError("duplicate request in forward step") + request_ids.add(request.request_id) + start = counts[partition] + if start + count > topology.capacity: + raise ValueError("DP token count exceeds segment capacity; split the step first") + if step.phase == "decode" and count != 1: + raise ValueError("decode requires one fresh token per request") + source_ranges.append((partition, start, offset, count)) + owner, row = divmod(start + count - 1, topology.local_capacity) + last_rows.append((partition * topology.tp + owner, row)) + counts[partition] += count + offset += count + if step.phase not in ("prefill", "decode") or embeddings.shape[0] != offset: + raise ValueError("embedding rows must match a prefill/decode forward step") + hidden = embeddings.shape[1] + # Include output storage and the largest temporary expanded FP32 row block. + allocated_rows = topology.world * topology.local_capacity + estimate = allocated_rows * (4 * hidden * 4 + 4 * 4 + 8) + offset * hidden * 4 + if estimate > max_prepare_bytes: + raise ValueError(f"segment preparation requires estimated {estimate} bytes; budget={max_prepare_bytes}") + if not bool(torch.isfinite(embeddings).all()): + raise ValueError("embeddings must be finite") + residual = torch.zeros(topology.dp, topology.capacity, 4, hidden, dtype=torch.float32) + pre_mix = torch.zeros(topology.dp, topology.capacity, 4, dtype=torch.float32) + row_indices = torch.full((topology.dp, topology.capacity), -1, dtype=torch.int64) + for partition, start, source, count in source_ranges: + residual[partition, start:start + count].copy_(embeddings[source:source + count, None]) + pre_mix[partition, start:start + count, 0] = 1 + row_indices[partition, start:start + count] = torch.arange(source, source + count) + return SegmentInputs( + residual.reshape(topology.world, topology.local_capacity, 4, hidden), + pre_mix.reshape(topology.world, topology.local_capacity, 4), + row_indices.reshape(topology.world, topology.local_capacity), + tuple(last_rows), tuple(counts), + ) diff --git a/tests/unit/model/deepseek_v41/test_segment_inputs.py b/tests/unit/model/deepseek_v41/test_segment_inputs.py new file mode 100644 index 00000000..4b37b798 --- /dev/null +++ b/tests/unit/model/deepseek_v41/test_segment_inputs.py @@ -0,0 +1,72 @@ +# Copyright (c) PyPTO Contributors. +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of +# CANN Open Software License Agreement Version 2.0 (the "License"). +# Please refer to the License for details. You may not use this file except in compliance with the License. +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. +# See LICENSE in the root of the software repository for the full text of the License. +# ----------------------------------------------------------------------------------------------------------- +"""Real tensor checks for fresh HC state and request-to-rank mapping.""" +import pytest +import torch + +from pypto_serving.model.deepseek_v41.request_state import ForwardStep, RequestSlice +from pypto_serving.model.deepseek_v41.segment_inputs import prepare_segment_inputs +from pypto_serving.model.deepseek_v41.swa_segment import SegmentTopology + + +def request(name, partition, count): + return RequestSlice(name, partition, 0, 3, tuple(range(count)), count + 3, {}) + + +def test_interleaved_requests_keep_tp_slabs_and_output_order(): + topology = SegmentTopology(tp=2, dp=2, local_capacity=2) + step = ForwardStep("prefill", (request("b", 1, 3), request("a", 0, 1), request("c", 1, 1)), 1) + embeddings = torch.arange(5 * 8).reshape(5, 8).bfloat16() + result = prepare_segment_inputs(embeddings, step, topology) + assert result.group_counts == (1, 4) + assert result.row_indices.tolist() == [[3, -1], [-1, -1], [0, 1], [2, 4]] + assert result.request_last_rows == ((3, 0), (0, 0), (3, 1)) + for rank, row in (result.row_indices >= 0).nonzero().tolist(): + source = result.row_indices[rank, row] + expected = embeddings[source].float() + assert torch.equal(result.residual[rank, row], expected.expand(4, -1)) + assert result.pre_mix[rank, row].tolist() == [1, 0, 0, 0] + padding = result.row_indices < 0 + assert torch.count_nonzero(result.residual[padding]) == 0 + assert torch.count_nonzero(result.pre_mix[padding]) == 0 + embeddings.zero_() + assert result.residual.abs().sum() > 0 # Own storage, safe until upload completes. + + +def test_decode_starts_fresh_token_hc_and_empty_partition_stays_zero(): + topology = SegmentTopology(tp=2, dp=2, local_capacity=2) + step = ForwardStep("decode", (request("a", 1, 1), request("b", 1, 1)), 8) + inputs = prepare_segment_inputs(torch.ones(2, 8).bfloat16(), step, topology) + assert inputs.group_counts == (0, 2) + assert inputs.request_last_rows == ((2, 0), (2, 1)) + assert torch.count_nonzero(inputs.residual[:2]) == 0 + assert torch.equal(inputs.residual[2], torch.ones(2, 4, 8)) + + +@pytest.mark.parametrize("requests,phase,rows,match", [ + ((request("a", 0, 5),), "prefill", 5, "capacity"), + ((request("a", 2, 1),), "prefill", 1, "partitions"), + ((request("a", 0, 2),), "decode", 2, "one fresh token"), + ((request("a", 0, 1), request("a", 1, 1)), "prefill", 2, "duplicate"), + ((request("a", 0, 1),), "prefill", 2, "embedding rows"), +]) +def test_invalid_mapping_is_rejected(requests, phase, rows, match): + with pytest.raises(ValueError, match=match): + prepare_segment_inputs(torch.zeros(rows, 8).bfloat16(), ForwardStep(phase, requests, 1), + SegmentTopology(tp=2, dp=2, local_capacity=2)) + + +def test_allocation_budget_rejects_before_output_allocation(monkeypatch): + embeddings = torch.ones(1, 8).bfloat16() + def forbidden(*args, **kwargs): + raise AssertionError("must not allocate output before budget check") + monkeypatch.setattr(torch, "zeros", forbidden) + with pytest.raises(ValueError, match="budget"): + prepare_segment_inputs(embeddings, ForwardStep("prefill", (request("a", 0, 1),), 1), + SegmentTopology(tp=2, dp=2, local_capacity=2), max_prepare_bytes=1) diff --git a/tools/validate_v41_swa_segment.py b/tools/validate_v41_swa_segment.py index 8bc47cea..e290d2ba 100644 --- a/tools/validate_v41_swa_segment.py +++ b/tools/validate_v41_swa_segment.py @@ -24,6 +24,9 @@ def prepare_checkpoint_inputs(tensors, model_dir, topology, token_ids=None): """Prepare an embedding-broadcast control; retain the random stress fixture separately.""" import torch from pypto_serving.model.deepseek_v41.input_preparation import lookup_token_embeddings + from pypto_serving.model.deepseek_v41.request_state import ForwardStep, RequestSlice + from pypto_serving.model.deepseek_v41.segment_inputs import prepare_segment_inputs + from pypto_serving.model.deepseek_v41.swa_segment import SegmentTopology from pypto_serving.model.deepseek_v41.weight_loader import V41WeightLoader if token_ids is None: @@ -37,10 +40,16 @@ def prepare_checkpoint_inputs(tensors, model_dir, topology, token_ids=None): ep_size=topology.world, ep_rank=rank, max_load_bytes=512 << 20) for rank in range(topology.tp)] embeddings = lookup_token_embeddings(loaders, ids) - rows = embeddings.reshape(topology.world, topology.local_capacity, -1) - tensors["x_hc"] = rows.unsqueeze(2).expand_as(tensors["x_hc"]).float().contiguous() - tensors["incoming_pre_mix"].zero_() - tensors["incoming_pre_mix"][..., 0] = 1 + step = ForwardStep("prefill", tuple( + RequestSlice(str(group), group, 0, 0, tuple(row.tolist()), topology.capacity, {}) + for group, row in enumerate(ids) + ), 1) + prepared = prepare_segment_inputs( + embeddings.flatten(0, 1), step, + SegmentTopology(tp=topology.tp, dp=topology.dp, local_capacity=topology.local_capacity), + ) + tensors["x_hc"] = prepared.residual + tensors["incoming_pre_mix"] = prepared.pre_mix return ids From b5f07cd0396f66ddd705c04c8b4ec3141fbca85b Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Tue, 29 Sep 2026 01:44:22 +0800 Subject: [PATCH 38/78] feat(v41): lower private SWA pages into causal chunk metadata --- docs/developer-guide/v41-swa-segment.md | 15 +++ .../model/deepseek_v41/swa_metadata.py | 94 +++++++++++++++++++ .../model/deepseek_v41/test_swa_metadata.py | 75 +++++++++++++++ 3 files changed, 184 insertions(+) create mode 100644 pypto_serving/model/deepseek_v41/swa_metadata.py create mode 100644 tests/unit/model/deepseek_v41/test_swa_metadata.py diff --git a/docs/developer-guide/v41-swa-segment.md b/docs/developer-guide/v41-swa-segment.md index 8cd4835a..34000288 100644 --- a/docs/developer-guide/v41-swa-segment.md +++ b/docs/developer-guide/v41-swa-segment.md @@ -170,3 +170,18 @@ The helper rejects over-capacity steps before allocation instead of truncating. It is used by the embedding diagnostic; cache lowering, device upload and the complete production adapter remain separate work. Never invoke it between layers to overwrite the residual/pre-mix produced by the previous composite. + + +`swa_metadata.prepare_swa_window_metadata` lowers scheduler-owned full-history +128-row window pages into the lib's INT64 write slots and INT32 causal read +indices. TP peers receive identical metadata within each DP group; physical +page IDs may be reused across DP groups but never shared by active requests in +one group. Each query reads at most 128 positions ending at itself. Full-history +pages avoid overwriting an early query's history when an entire new chunk is +published before attention. Rolling/modulo page reuse is not implemented. + +The metadata helper covers page crossing, chunk continuation into decode, +interleaved requests, padding and empty partitions. It does not allocate or clear +the cache or choose a RoPE profile. Its positions must select the corresponding +checkpoint RoPE rows before dispatch. Integration into a complete model adapter +and device validation of that lifecycle remain pending. diff --git a/pypto_serving/model/deepseek_v41/swa_metadata.py b/pypto_serving/model/deepseek_v41/swa_metadata.py new file mode 100644 index 00000000..2c63e68b --- /dev/null +++ b/pypto_serving/model/deepseek_v41/swa_metadata.py @@ -0,0 +1,94 @@ +# Copyright (c) PyPTO Contributors. +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of +# CANN Open Software License Agreement Version 2.0 (the "License"). +# Please refer to the License for details. You may not use this file except in compliance with the License. +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. +# See LICENSE in the root of the software repository for the full text of the License. +# ----------------------------------------------------------------------------------------------------------- +"""Lower private full-history pages to the existing SWA composite metadata ABI.""" +from dataclasses import dataclass + +import torch + +from .request_state import ForwardStep +from .swa_segment import SegmentTopology + + +@dataclass(frozen=True) +class SwaWindowMetadata: + positions: torch.Tensor + window_slots: torch.Tensor + window_indices: torch.Tensor + group_counts: tuple[int, ...] + + +def prepare_swa_window_metadata( + step: ForwardStep, + topology: SegmentTopology, + *, + cache_pages: int, + window_group: str = "window", + max_prepare_bytes: int = 16 << 20, +) -> SwaWindowMetadata: + """Map each active row to a unique cache write and at most 128 causal reads. + + Page IDs address a partition-private cache of [pages, 128, 1, 512] payloads. + Full-history tables preserve old rows while the whole new chunk is published; + a 128-slot modulo ring would overwrite keys needed by early chunk queries. + TP peers receive the same group-global metadata. No cache contents, RoPE, + request ownership or completion state are changed by this host preparation. + """ + if not isinstance(step, ForwardStep) or not isinstance(topology, SegmentTopology): + raise ValueError("SWA metadata requires a ForwardStep and SegmentTopology") + if type(cache_pages) is not int or not 0 < cache_pages <= (2**31 - 1) // 128: + raise ValueError("cache_pages must fit the INT32 window-index ABI") + if type(max_prepare_bytes) is not int or max_prepare_bytes <= 0: + raise ValueError("max_prepare_bytes must be a positive integer") + if step.phase not in ("prefill", "decode"): + raise ValueError("unsupported forward phase") + counts, ranges, owners, requests = [0] * topology.dp, [], set(), set() + for request in step.requests: + group, count = request.partition, len(request.token_ids) + if (type(group) is not int or not 0 <= group < topology.dp or count <= 0 + or type(request.start) is not int or request.start < 0): + raise ValueError("request partition, start and token count must be valid") + if request.request_id in requests: + raise ValueError("duplicate request in forward step") + requests.add(request.request_id) + if step.phase == "decode" and count != 1: + raise ValueError("decode requires one token per request") + if counts[group] + count > topology.capacity: + raise ValueError("DP token count exceeds SWA capacity") + pages = request.pages.get(window_group) + if not isinstance(pages, (tuple, list)) or len(pages) < (request.end + 127) // 128: + raise ValueError("SWA requires a full-history physical page table through the chunk end") + for page in pages: + if type(page) is not int or not 0 <= page < cache_pages: + raise ValueError("SWA physical page is outside its partition cache pool") + key = (group, page) + if key in owners: + raise ValueError("SWA pages must be private; shared or repeated physical pages are unsupported") + owners.add(key) + ranges.append((group, counts[group], request.start, count, tuple(pages))) + counts[group] += count + rows = topology.dp * topology.capacity + estimate = rows * (128 * 4 + 2 * 8) * (topology.tp + 1) + if estimate > max_prepare_bytes: + raise ValueError(f"SWA metadata requires estimated {estimate} bytes; budget={max_prepare_bytes}") + positions = torch.zeros(topology.dp, topology.capacity, dtype=torch.int64) + slots = torch.full_like(positions, -1) + indices = torch.full((topology.dp, topology.capacity, 128), -1, dtype=torch.int32) + for group, row, start, count, pages in ranges: + for local in range(count): + position = start + local + positions[group, row + local] = position + slots[group, row + local] = pages[position // 128] * 128 + position % 128 + visible = range(max(0, position - 127), position + 1) + addresses = [pages[p // 128] * 128 + p % 128 for p in visible] + indices[group, row + local, :len(addresses)] = torch.tensor(addresses, dtype=torch.int32) + return SwaWindowMetadata( + positions.repeat_interleave(topology.tp, dim=0), + slots.repeat_interleave(topology.tp, dim=0), + indices.repeat_interleave(topology.tp, dim=0), tuple(counts), + ) diff --git a/tests/unit/model/deepseek_v41/test_swa_metadata.py b/tests/unit/model/deepseek_v41/test_swa_metadata.py new file mode 100644 index 00000000..e29be4a0 --- /dev/null +++ b/tests/unit/model/deepseek_v41/test_swa_metadata.py @@ -0,0 +1,75 @@ +# Copyright (c) PyPTO Contributors. +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of +# CANN Open Software License Agreement Version 2.0 (the "License"). +# Please refer to the License for details. You may not use this file except in compliance with the License. +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. +# See LICENSE in the root of the software repository for the full text of the License. +# ----------------------------------------------------------------------------------------------------------- +"""Causal cache addressing across chunks, physical pages and DP partitions.""" +import pytest +import torch + +from pypto_serving.model.deepseek_v41.request_state import ForwardStep, RequestSlice +from pypto_serving.model.deepseek_v41.segment_inputs import prepare_segment_inputs +from pypto_serving.model.deepseek_v41.swa_metadata import prepare_swa_window_metadata +from pypto_serving.model.deepseek_v41.swa_segment import SegmentTopology + + +TOPOLOGY = SegmentTopology(tp=2, dp=2, local_capacity=2) + + +def request(name="a", partition=0, start=126, count=4, pages=(5, 2)): + return RequestSlice(name, partition, 0, start, tuple(range(count)), start + count, {"window": pages}) + + +def test_page_crossing_preserves_early_chunk_history_and_causality(): + step = ForwardStep("prefill", (request(),), 1) + metadata = prepare_swa_window_metadata(step, TOPOLOGY, cache_pages=8) + assert metadata.group_counts == (4, 0) + assert metadata.positions[0].tolist() == [126, 127, 128, 129] + assert metadata.window_slots[0].tolist() == [766, 767, 256, 257] + assert metadata.window_indices[0, 0, :127].tolist() == list(range(640, 767)) + assert metadata.window_indices[0, 0, 127] == -1 + assert metadata.window_indices[0, 2].tolist() == list(range(641, 768)) + [256] + assert metadata.window_indices[0, 3].tolist() == list(range(642, 768)) + [256, 257] + assert torch.equal(metadata.window_indices[0], metadata.window_indices[1]) + assert (metadata.window_slots[2:] == -1).all() + assert (metadata.window_indices[2:] == -1).all() + continuation = ForwardStep("decode", (request(start=130, count=1),), 2) + decoded = prepare_swa_window_metadata(continuation, TOPOLOGY, cache_pages=8) + assert decoded.window_indices[0, 0].tolist() == list(range(643, 768)) + [256, 257, 258] + assert decoded.window_slots[0, 0] == 258 + + +def test_interleaved_dp_rows_match_hc_input_packing(): + requests = (request("b", 1, 0, 2, (3,)), request("a", 0, 2, 1, (3,)), + request("c", 1, 5, 1, (4,))) + step = ForwardStep("prefill", requests, 1) + inputs = prepare_segment_inputs(torch.ones(4, 8).bfloat16(), step, TOPOLOGY) + metadata = prepare_swa_window_metadata(step, TOPOLOGY, cache_pages=8) + assert metadata.group_counts == inputs.group_counts == (1, 3) + source_positions = torch.tensor(step.positions) + for rank, local in (inputs.row_indices >= 0).nonzero().tolist(): + group_row = rank % TOPOLOGY.tp * TOPOLOGY.local_capacity + local + assert metadata.positions[rank, group_row] == source_positions[inputs.row_indices[rank, local]] + assert metadata.window_slots[2].tolist() == [384, 385, 517, -1] + + +@pytest.mark.parametrize("requests,match", [ + ((request(pages=(5,)),), "full-history"), + ((request(pages=(5, 8)),), "outside"), + ((request(pages=(5, 5)),), "private"), + ((request("a", 0, 0, 1, (3,)), request("b", 0, 0, 1, (3,))), "private"), + ((request(start=-1),), "start"), + ((request(count=5),), "capacity"), +]) +def test_invalid_pages_and_extents_fail_before_dispatch(requests, match): + with pytest.raises(ValueError, match=match): + prepare_swa_window_metadata(ForwardStep("prefill", requests, 1), TOPOLOGY, cache_pages=8) + + +def test_small_budget_is_rejected(): + with pytest.raises(ValueError, match="budget"): + prepare_swa_window_metadata(ForwardStep("prefill", (request(),), 1), TOPOLOGY, + cache_pages=8, max_prepare_bytes=1) From 2ec6fd8bb3efe957faaa695ccfb3a8e2ab0963a2 Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Tue, 29 Sep 2026 02:00:34 +0800 Subject: [PATCH 39/78] feat(v41): wire checkpoint RoPE and causal request pages into SWA validation --- docs/developer-guide/v41-swa-segment.md | 10 +++++ .../model/deepseek_v41/swa_metadata.py | 40 +++++++++++++++++++ .../model/deepseek_v41/test_swa_metadata.py | 29 +++++++++++++- tools/diagnose_v41_swa_precision.py | 2 + tools/validate_v41_swa_segment.py | 36 +++++++++++++++++ 5 files changed, 116 insertions(+), 1 deletion(-) diff --git a/docs/developer-guide/v41-swa-segment.md b/docs/developer-guide/v41-swa-segment.md index 34000288..5f0cf1a5 100644 --- a/docs/developer-guide/v41-swa-segment.md +++ b/docs/developer-guide/v41-swa-segment.md @@ -185,3 +185,13 @@ interleaved requests, padding and empty partitions. It does not allocate or clea the cache or choose a RoPE profile. Its positions must select the corresponding checkpoint RoPE rows before dispatch. Integration into a complete model adapter and device validation of that lifecycle remain pending. + +`gather_swa_rope_rows` selects FP32 rows from the lib-built SWA tables using +those absolute positions, with identity rotation for padding. Like V4's +executor, model-specific table generation stays in the selected lib helper. +The diagnostic's optional `--request-metadata` requires embedding inputs and +a checkpoint. It uses fresh private request pages, empty caches and checkpoint +RoPE instead of fixture history/angles, saving all request inputs for exact +CPU replay. It currently exercises one full first chunk per DP group; it is +not a continuation, generation or complete backend acceptance test. Keep the +historical fixture failures separate from results on this changed workload. diff --git a/pypto_serving/model/deepseek_v41/swa_metadata.py b/pypto_serving/model/deepseek_v41/swa_metadata.py index 2c63e68b..36e77b20 100644 --- a/pypto_serving/model/deepseek_v41/swa_metadata.py +++ b/pypto_serving/model/deepseek_v41/swa_metadata.py @@ -23,6 +23,46 @@ class SwaWindowMetadata: group_counts: tuple[int, ...] +def gather_swa_rope_rows( + metadata: SwaWindowMetadata, + tables: tuple[torch.Tensor, torch.Tensor], + *, + max_prepare_bytes: int = 16 << 20, +) -> tuple[torch.Tensor, torch.Tensor]: + """Gather lib-built FP32 SWA tables in the same order as cache metadata. + + Like V4, the executor builds tables with the selected lib's host helper. + This boundary only selects rows; it does not implement model-side RoPE. + Padding receives identity rotation and does not index the table. + """ + if not isinstance(metadata, SwaWindowMetadata) or len(tables) != 2: + raise ValueError("SWA RoPE requires metadata and a cosine/sine table pair") + positions, slots = metadata.positions, metadata.window_slots + if (positions.device.type != "cpu" or positions.dtype != torch.int64 or positions.ndim != 2 + or slots.device.type != "cpu" or slots.dtype != torch.int64 or slots.shape != positions.shape): + raise ValueError("SWA positions and slots must be matching CPU INT64 matrices") + cos, sin = tables + if any(not isinstance(t, torch.Tensor) or t.device.type != "cpu" or t.dtype != torch.float32 + or t.layout != torch.strided or t.ndim != 2 for t in tables): + raise ValueError("SWA RoPE tables must be CPU FP32 matrices") + if cos.shape != sin.shape or min(cos.shape) <= 0: + raise ValueError("SWA cosine/sine tables must have matching nonempty shapes") + estimate = positions.numel() * (cos.shape[1] * 4 * 4 + 32) + if type(max_prepare_bytes) is not int or max_prepare_bytes <= 0 or estimate > max_prepare_bytes: + raise ValueError("SWA RoPE preparation exceeds its allocation budget") + active = slots >= 0 + selected = positions[active] + if bool(((selected < 0) | (selected >= cos.shape[0])).any()): + raise ValueError("active SWA position is outside the RoPE table") + selected_cos, selected_sin = cos[selected], sin[selected] + if not bool(torch.isfinite(selected_cos).all() and torch.isfinite(selected_sin).all()): + raise ValueError("selected SWA RoPE rows must be finite") + shape = (*positions.shape, cos.shape[1]) + rows_cos, rows_sin = torch.ones(shape, dtype=torch.float32), torch.zeros(shape, dtype=torch.float32) + rows_cos[active], rows_sin[active] = selected_cos, selected_sin + return rows_cos, rows_sin + + def prepare_swa_window_metadata( step: ForwardStep, topology: SegmentTopology, diff --git a/tests/unit/model/deepseek_v41/test_swa_metadata.py b/tests/unit/model/deepseek_v41/test_swa_metadata.py index e29be4a0..a5ab62db 100644 --- a/tests/unit/model/deepseek_v41/test_swa_metadata.py +++ b/tests/unit/model/deepseek_v41/test_swa_metadata.py @@ -12,7 +12,7 @@ from pypto_serving.model.deepseek_v41.request_state import ForwardStep, RequestSlice from pypto_serving.model.deepseek_v41.segment_inputs import prepare_segment_inputs -from pypto_serving.model.deepseek_v41.swa_metadata import prepare_swa_window_metadata +from pypto_serving.model.deepseek_v41.swa_metadata import gather_swa_rope_rows, prepare_swa_window_metadata from pypto_serving.model.deepseek_v41.swa_segment import SegmentTopology @@ -73,3 +73,30 @@ def test_small_budget_is_rejected(): with pytest.raises(ValueError, match="budget"): prepare_swa_window_metadata(ForwardStep("prefill", (request(),), 1), TOPOLOGY, cache_pages=8, max_prepare_bytes=1) + + +def test_rope_rows_use_absolute_positions_and_identity_padding(): + step = ForwardStep("prefill", (request("b", 1, 126, 2), request("a", 0, 3, 1, (2,))), 1) + metadata = prepare_swa_window_metadata(step, TOPOLOGY, cache_pages=8) + angles = torch.arange(130, dtype=torch.float32)[:, None] * torch.tensor([[1., .1]]) + cos, sin = gather_swa_rope_rows(metadata, (angles.cos(), angles.sin())) + for rank, positions in enumerate(([3], [3], [126, 127], [126, 127])): + torch.testing.assert_close(cos[rank, :len(positions)], angles[positions].cos(), rtol=0, atol=0) + torch.testing.assert_close(sin[rank, :len(positions)], angles[positions].sin(), rtol=0, atol=0) + assert (cos[rank, len(positions):] == 1).all() + assert (sin[rank, len(positions):] == 0).all() + + +@pytest.mark.parametrize("case,match", [("short", "outside"), ("nan", "finite"), + ("dtype", "FP32"), ("budget", "budget")]) +def test_rope_rejects_invalid_active_rows(case, match): + metadata = prepare_swa_window_metadata(ForwardStep("prefill", (request(),), 1), TOPOLOGY, cache_pages=8) + cos, sin = torch.ones(130, 2), torch.zeros(130, 2) + if case == "short": + cos, sin = cos[:129], sin[:129] + if case == "nan": + sin[129, 0] = float("nan") + if case == "dtype": + cos = cos.bfloat16() + with pytest.raises(ValueError, match=match): + gather_swa_rope_rows(metadata, (cos, sin), max_prepare_bytes=1 if case == "budget" else 16 << 20) diff --git a/tools/diagnose_v41_swa_precision.py b/tools/diagnose_v41_swa_precision.py index 150aa552..7876529c 100644 --- a/tools/diagnose_v41_swa_precision.py +++ b/tools/diagnose_v41_swa_precision.py @@ -150,6 +150,8 @@ def wide_linear(x, weight, packed_scale, fp32=False): if "initial_state" in saved: a["x_hc"] = saved["initial_state"]["residual"].clone() a["incoming_pre_mix"] = saved["initial_state"]["pre_mix"].clone() + if saved.get("request_inputs") is not None: + a.update({name: value.clone() for name, value in saved["request_inputs"].items()}) if args.trace_attention: full = torch.load(args.output, map_location="cpu", weights_only=True) layer = args.trace_layer diff --git a/tools/validate_v41_swa_segment.py b/tools/validate_v41_swa_segment.py index e290d2ba..d71c1d9b 100644 --- a/tools/validate_v41_swa_segment.py +++ b/tools/validate_v41_swa_segment.py @@ -18,6 +18,35 @@ ATTENTION_OUTPUTS = ("output", "next_pre_mix", "hidden", "attn_out", "window_cache", "window_cache_scale") MOE_OUTPUTS = ("x_next", "next_pre_mix", "x_mixed") +REQUEST_INPUTS = ("rope_cos", "rope_sin", "window_slots", "window_indices", "window_cache", "window_cache_scale") + + +def prepare_checkpoint_metadata(tensors, model_dir, topology, token_ids): + """First-chunk request metadata with checkpoint RoPE and private empty pages.""" + import json + import torch + from models.deepseek_v4_1_flash.rope_tables import precompute_rope_tables + from pypto_serving.model.deepseek_v41.request_state import ForwardStep, RequestSlice + from pypto_serving.model.deepseek_v41.swa_metadata import gather_swa_rope_rows, prepare_swa_window_metadata + + text = json.loads((Path(model_dir) / "config.json").read_text())["text_config"] + pages = (topology.capacity + 127) // 128 + step = ForwardStep("prefill", tuple( + RequestSlice(str(group), group, 0, 0, tuple(row.tolist()), topology.capacity, + {"window": tuple(range(pages))}) + for group, row in enumerate(token_ids) + ), 1) + # Keep fixture allocation shapes to isolate addressing/positions from compilation. + metadata = prepare_swa_window_metadata(step, topology, cache_pages=tensors["window_cache"].shape[1]) + rope_config = SimpleNamespace(qk_rope_head_dim=int(text["qk_rope_head_dim"]), + rope_theta=float(text["rope_theta"])) + tables = precompute_rope_tables(topology.capacity, False, config=rope_config) + tensors["rope_cos"], tensors["rope_sin"] = gather_swa_rope_rows(metadata, tables) + tensors["window_slots"], tensors["window_indices"] = metadata.window_slots, metadata.window_indices + tensors["window_cache"].view(torch.uint8).zero_() + # UE8M0 code 127 represents scale 1 for unpublished zero cache payloads. + tensors["window_cache_scale"].view(torch.uint8).fill_(127) + return {name: tensors[name].clone() for name in REQUEST_INPUTS} def prepare_checkpoint_inputs(tensors, model_dir, topology, token_ids=None): @@ -145,6 +174,8 @@ def main(): parser.add_argument("--input-source", choices=("stress", "embeddings"), default="stress", help="Keep the random stress input, or load real embedding rows and broadcast HC streams") parser.add_argument("--token-ids", help="JSON array [DP, capacity] of embedding control IDs; no padding") + parser.add_argument("--request-metadata", action="store_true", + help="Use fresh private request pages and checkpoint RoPE with embedding inputs") parser.add_argument("--model-dir", help="Use actual checkpoint weights for layers 0 and 1") parser.add_argument("--reference", action="store_true", help="Compare against composed Torch references") parser.add_argument("--stage-reference", action="store_true", @@ -158,6 +189,8 @@ def main(): parser.add_argument("--ring-heap-mib", type=int, default=1024, help="Per-ring temporary heap; lib MoE validation uses 1024 MiB") args = parser.parse_args() + if args.request_metadata and (args.input_source != "embeddings" or not args.model_dir): + parser.error("--request-metadata requires --input-source embeddings and --model-dir") if args.input_source == "embeddings" and not args.model_dir: parser.error("--input-source embeddings requires --model-dir") if args.token_ids and args.input_source != "embeddings": @@ -208,6 +241,8 @@ def materialize(specs): supplied_ids = json.loads(Path(args.token_ids).read_text(encoding="utf-8")) if args.token_ids else None token_ids = prepare_checkpoint_inputs(a, args.model_dir, topology, supplied_ids) + request_inputs = prepare_checkpoint_metadata(a, args.model_dir, topology, token_ids) if ( + args.request_metadata) else None initial_state = {"residual": a["x_hc"].clone(), "pre_mix": a["incoming_pre_mix"].clone()} print(f"Attention fixture ready: input_source={args.input_source}", flush=True) if args.model_dir: @@ -308,6 +343,7 @@ def upload(values): artifact = Path(args.artifact_dir) artifact.mkdir(parents=True, exist_ok=True) data = {"input_source": args.input_source, "token_ids": token_ids, "initial_state": initial_state, + "request_inputs": request_inputs, "actual_residual": readback, "expected_residual": residual, "actual_pre_mix": mix_readback, "expected_pre_mix": mix} torch.save(data, artifact / "comparison.pt") From 98a4311ab1225279fa86d10d8c17040a124ace63 Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Tue, 29 Sep 2026 02:05:58 +0800 Subject: [PATCH 40/78] fix(v41): replicate indexer heads required by current lib composites --- docs/developer-guide/deepseek-v41-entry.md | 7 +++++-- pypto_serving/model/deepseek_v41/weight_loader.py | 2 +- pypto_serving/model/deepseek_v41/weight_spec.py | 11 +++++++---- .../unit/model/deepseek_v41/test_weight_loader.py | 15 +++++++++++++++ 4 files changed, 28 insertions(+), 7 deletions(-) diff --git a/docs/developer-guide/deepseek-v41-entry.md b/docs/developer-guide/deepseek-v41-entry.md index 2454e2b5..9d75d347 100644 --- a/docs/developer-guide/deepseek-v41-entry.md +++ b/docs/developer-guide/deepseek-v41-entry.md @@ -92,8 +92,11 @@ layouts again when lib changes. Current FP4 tiles require K/N multiples of 256; native FP8 matrices require K divisible by 64 and N by 32. Unsupported geometry is rejected. Torch/safetensors must support the checkpoint E8M0 dtype. -TP slices grouped/head projections and aligns their scales. Shared experts and -router weights are replicated; EP selects whole routed experts, retaining +TP slices main-attention grouped/head projections and aligns their scales. +Indexer query/gating heads are replicated: the C1A/C2A composites at lib +`fbe92bfc` use global `INDEX_H` on every TP rank and require complete scores. +This overrides the earlier reference-derived index-head sharding assumption. +Shared experts and router weights are replicated; EP selects whole routed experts, retaining global expert IDs. TP and EP ranks are explicit; mapping physical ranks to these coordinates belongs to the Executor. This loader does not certify a multi-device execution path. diff --git a/pypto_serving/model/deepseek_v41/weight_loader.py b/pypto_serving/model/deepseek_v41/weight_loader.py index 6b76d0ca..2d2f5c48 100644 --- a/pypto_serving/model/deepseek_v41/weight_loader.py +++ b/pypto_serving/model/deepseek_v41/weight_loader.py @@ -84,7 +84,7 @@ def __init__(self, model_dir, *, tp_size=1, tp_rank=0, ep_size=1, ep_rank=0, self.max_load_bytes = _positive(max_load_bytes, "max_load_bytes") if self.text["n_routed_experts"] % ep_size: raise ValueError("routed expert count must divide EP size") - for name in ("num_attention_heads", "o_groups", "index_n_heads", "vocab_size"): + for name in ("num_attention_heads", "o_groups", "vocab_size"): if self.text[name] % tp_size: raise ValueError(f"{name} must divide TP size") index = json.loads((self.model_dir / "model.safetensors.index.json").read_text(encoding="utf-8"), diff --git a/pypto_serving/model/deepseek_v41/weight_spec.py b/pypto_serving/model/deepseek_v41/weight_spec.py index fc5a3fa2..a93e3ec6 100644 --- a/pypto_serving/model/deepseek_v41/weight_spec.py +++ b/pypto_serving/model/deepseek_v41/weight_spec.py @@ -9,8 +9,8 @@ """Checkpoint storage contracts for the V4.1 text backbone, excluding deferred modules. Names and dtypes describe the published checkpoint, not PyTorch module defaults. -Conversion strings document the pinned reference's transformations and sharding; -they do not execute conversion, define an Ascend pack ABI, or upload weights. +Conversion strings describe transformations and placement for the selected lib +boundary; they do not execute conversion, define a pack ABI, or upload weights. Engram, vision, DSpark and the unused VL router bias are outside this scope. """ @@ -162,8 +162,11 @@ def expert(name: str, routed: bool): add(prefix + ".wgate.weight", (head_dim, dim), "BF16", promotion) if layer in index_sources: prefix = attn + ".indexer" - dense(prefix + ".wq_b", index_heads * index_dim, q_rank, "tp_shard_axis0") - add(prefix + ".weights_proj.weight", (index_heads, dim), "BF16", "tp_shard_axis0") + # C1A/C2A composites at lib fbe92bfc compute complete index scores + # on every TP rank (INDEX_H is global). Main attention heads are + # sharded, but cutting index heads would omit part of each score. + dense(prefix + ".wq_b", index_heads * index_dim, q_rank) + add(prefix + ".weights_proj.weight", (index_heads, dim), "BF16", "replicate") if layer in kv_sources: add(prefix + ".wk.weight", (index_dim, head_dim), "BF16") add(prefix + ".k_norm.weight", (index_dim,), "BF16") diff --git a/tests/unit/model/deepseek_v41/test_weight_loader.py b/tests/unit/model/deepseek_v41/test_weight_loader.py index 1f555e3b..c933ede9 100644 --- a/tests/unit/model/deepseek_v41/test_weight_loader.py +++ b/tests/unit/model/deepseek_v41/test_weight_loader.py @@ -110,6 +110,21 @@ def test_input_axis_shard_keeps_scales_aligned(checkpoint): assert torch.equal(logical_scales(result.scale), scales.T.repeat_interleave(32, 1)) +def test_indexer_keeps_all_heads_on_every_tp_rank(checkpoint): + path, _, tensors = checkpoint + prefix = "layers.0.attn.indexer." + for rank in range(2): + loader = V41WeightLoader(path, tp_size=2, tp_rank=rank) + query = loader.load(prefix + "wq_b.weight") + expected = tensors[prefix + "wq_b.weight"].T.contiguous().view(torch.uint8) + assert torch.equal(query.weight.view(torch.uint8), expected) + scales = tensors[prefix + "wq_b.scale"].view(torch.uint8).T.repeat_interleave(32, 1) + assert torch.equal(logical_scales(query.scale), scales) + gate = loader.load(prefix + "weights_proj.weight") + assert torch.equal(gate.weight, tensors[prefix + "weights_proj.weight"].T) + assert query.tp_rank == gate.tp_rank == rank + + def test_wo_a_group_dequantization(checkpoint): path, _, tensors = checkpoint result = V41WeightLoader(path, tp_size=2, tp_rank=1).load("layers.0.attn.wo_a.weight") From 78c0c52af09f985c3bfbab023e183e3e9d4c31a2 Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Tue, 29 Sep 2026 02:11:41 +0800 Subject: [PATCH 41/78] feat(v41): dispatch prefill composites with independent communication epochs --- docs/developer-guide/v41-swa-segment.md | 14 +++ .../model/deepseek_v41/prefill_segment.py | 98 +++++++++++++++++++ .../model/deepseek_v41/swa_segment.py | 28 ++++-- tests/unit/test_v41_swa_segment.py | 42 ++++++++ tools/compile_v41_prefill_segments.py | 36 +++++++ 5 files changed, 211 insertions(+), 7 deletions(-) create mode 100644 pypto_serving/model/deepseek_v41/prefill_segment.py create mode 100644 tools/compile_v41_prefill_segments.py diff --git a/docs/developer-guide/v41-swa-segment.md b/docs/developer-guide/v41-swa-segment.md index 5f0cf1a5..b555ab2f 100644 --- a/docs/developer-guide/v41-swa-segment.md +++ b/docs/developer-guide/v41-swa-segment.md @@ -195,3 +195,17 @@ RoPE instead of fixture history/angles, saving all request inputs for exact CPU replay. It currently exercises one full first chunk per DP group; it is not a continuation, generation or complete backend acceptance test. Keep the historical fixture failures separate from results on this changed workload. + +`prefill_segment.PrefillSegment` extends the same device handoff to the audited +C2A Full/Reuse and C1A Full/Reindex/Reuse SP entry signatures. It calls existing +lib composites, with caller-supplied weights, metadata, scratch and caches. +C1A uses `prefill_c1a_sp.make_program`, preserving TP-local residuals. Each +compiled Attention program has its own retained-window epoch, while one shared +MoE program advances every layer. Missing arguments are rejected before either +half-layer runs. A failed dispatch poisons the entire worker. + +This dispatch support is not full-model readiness: real compressed-cache +allocation/lowering, producer bindings and device numerical validation remain +required. `tools/compile_v41_prefill_segments.py --lib-root ... --modes c2a_full` +provides an explicit compilation-only check without allocating devices. All +programs must be registered with the same persistent worker before execution. diff --git a/pypto_serving/model/deepseek_v41/prefill_segment.py b/pypto_serving/model/deepseek_v41/prefill_segment.py new file mode 100644 index 00000000..9349e104 --- /dev/null +++ b/pypto_serving/model/deepseek_v41/prefill_segment.py @@ -0,0 +1,98 @@ +# Copyright (c) PyPTO Contributors. +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of +# CANN Open Software License Agreement Version 2.0 (the "License"). +# Please refer to the License for details. You may not use this file except in compliance with the License. +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. +# See LICENSE in the root of the software repository for the full text of the License. +# ----------------------------------------------------------------------------------------------------------- +"""Explicit prefill half-layer dispatch for the lib's sequence-parallel entries. + +Signatures were audited at lib fbe92bfc. This module does not allocate caches, +invent metadata, or enable the full CLI backend. The caller must supply the +entire audited ABI and retain every program's communication windows. +""" +import importlib + +from .swa_segment import ATTENTION_ARGS, SwaSegment, load_segment_modules + + +_WEIGHTS = ( + "x_hc pre_mix hc_attn_fn hc_attn_scale hc_attn_base attn_norm_weight " + "wq_a wq_a_scale q_norm_weight wq_b wq_b_scale wkv wkv_scale kv_norm_weight " + "attn_sink wo_a wo_b wo_b_scale " +) +_WINDOW = "window_slots window_indices window_cache window_cache_scale " +_OUTPUTS = "attn_input attn_output next_pre_mix x_hc_out num_tokens attention_epoch" +C2A_FULL_ARGS = (_WEIGHTS + "freqs_cos freqs_sin " + _WINDOW + + "compressed_cache compressed_cache_scale token_to_req_indices compressed_lens " + "index_cache index_cache_scale index_block_table position_ids compressed_freqs_cos compressed_freqs_sin " + "compressed_rope_positions compressor_wkv compressor_wgate query_start_loc state_block_table state_cache " + "compressor_norm_weight compressed_slots index_wk index_norm_weight index_wq_b index_wq_b_scale " + "index_weights_proj topk_indices " + _OUTPUTS).split() +C2A_REUSE_ARGS = (_WEIGHTS + "rope_cos rope_sin " + _WINDOW + + "compressed_cache compressed_cache_scale compressed_indices " + _OUTPUTS).split() +C1A_ARGS = (_WEIGHTS + "rope_cos rope_sin " + _WINDOW + + "compressed_cache compressed_cache_scale request_ids compressed_lens index_cache index_cache_scale " + "index_block_table compressed_rope_cos compressed_rope_sin compressor_wkv compressor_norm_weight " + "compressed_slots index_wk index_norm_weight index_wq_b index_wq_b_scale index_weights_proj " + "topk_indices candidate_mask compressed_indices " + _OUTPUTS).split() +PREFILL_ARGUMENTS = { + "swa": ATTENTION_ARGS, "c2a_full": C2A_FULL_ARGS, "c2a_reuse": C2A_REUSE_ARGS, + "c1a_full": C1A_ARGS, "c1a_reindex": C1A_ARGS, "c1a_reuse": C1A_ARGS, +} + + +def compile_prefill_segments(compiler, lib_root, topology, modes): + """Compile existing composite entries, sharing one packed-FP4 MoE program. + + C1A must use prefill_c1a_sp, not the older replicated-residual wrappers. + No model operators are composed by this serving module. + """ + import pypto.language as pl + + modes = tuple(dict.fromkeys(modes)) + if not modes or any(mode not in PREFILL_ARGUMENTS for mode in modes): + raise ValueError("unsupported or empty prefill mode selection") + swa, moe = load_segment_modules(lib_root, topology) + attention = {} + for mode in modes: + if mode == "swa": + entry = swa.make_hc_program(topology.capacity, topology.world, epochs=1) + elif mode.startswith("c2a_"): + module = importlib.import_module("models.deepseek_v4_1_flash.prefill_" + mode) + entry = module.make_hc_program(topology.capacity, topology.world, epochs=1) + else: + module = importlib.import_module("models.deepseek_v4_1_flash.prefill_c1a_sp") + entry = module.make_program(mode.removeprefix("c1a_"), topology.world, epochs=1) + attention[mode] = compiler.compile("v41_prefill_" + mode, entry, attention_epoch=pl.RUNTIME) + ffn = compiler.compile("v41_moe_segment", moe.l3_moe, moe_epoch=pl.RUNTIME) + return attention, ffn + + +class PrefillSegment(SwaSegment): + """Carry device residual/pre_mix through a selected prefill mode and MoE. + + The worker must own all supplied programs via make_segment_worker(). Each + Attention program advances its own epoch; the shared MoE advances on every + layer. Cache ownership, cross-layer producer bindings and padded metadata + remain caller responsibilities. Any runtime failure poisons the whole worker. + """ + + def __init__(self, worker, attention_programs, moe_program, topology, + attention_counts, moe_counts, run_config): + if not attention_programs or any(mode not in PREFILL_ARGUMENTS for mode in attention_programs): + raise ValueError("unsupported or empty prefill program set") + self.attention_programs = dict(attention_programs) + super().__init__(worker, (next(iter(attention_programs.values())), moe_program), + topology, attention_counts, moe_counts, run_config) + + def run_layer(self, state, attention, moe, *, group_counts, mode): + if mode not in self.attention_programs: + raise ValueError(f"prefill mode was not compiled: {mode}") + return self._run_layer( + state, attention, moe, group_counts=group_counts, + attention_program=self.attention_programs[mode], argument_names=PREFILL_ARGUMENTS[mode], + input_mix="incoming_pre_mix" if mode == "swa" else "pre_mix", + output_residual="output" if mode == "swa" else "x_hc_out", + ) diff --git a/pypto_serving/model/deepseek_v41/swa_segment.py b/pypto_serving/model/deepseek_v41/swa_segment.py index 4120f864..74366fda 100644 --- a/pypto_serving/model/deepseek_v41/swa_segment.py +++ b/pypto_serving/model/deepseek_v41/swa_segment.py @@ -152,6 +152,7 @@ def __init__(self, worker, programs, topology, attention_counts, moe_counts, run self.attention_counts, self.moe_counts = attention_counts, moe_counts self.run_config = run_config self._epoch = 0 + self._attention_epochs = {} self._failed = False def _check_state(self, state): @@ -173,13 +174,20 @@ def run_layer(self, state, attention, moe, *, group_counts): Request packing, RoPE, page mapping and cache initialization are owned by the caller and must use the same contiguous group/slab ordering. """ + return self._run_layer(state, attention, moe, group_counts=group_counts, + attention_program=self.programs[0], argument_names=ATTENTION_ARGS, + input_mix="incoming_pre_mix", output_residual="output") + + def _run_layer(self, state, attention, moe, *, group_counts, attention_program, + argument_names, input_mix, output_residual): + """Dispatch an audited TP-local Attention ABI and the shared MoE entry.""" if self._failed: raise RuntimeError("segment dispatch failed; close this worker before attempting recovery") self._check_state(state) group, local = self.topology.counts(group_counts) - a = dict(attention, x_hc=state.residual, incoming_pre_mix=state.pre_mix, - num_tokens=self.attention_counts) - attention_state = LayerState(a["output"], a["next_pre_mix"], "tp_local_token") + a = dict(attention, x_hc=state.residual, num_tokens=self.attention_counts) + a[input_mix] = state.pre_mix + attention_state = LayerState(a[output_residual], a["next_pre_mix"], "tp_local_token") m = dict(moe, x_hc=attention_state.residual, pre_mix=attention_state.pre_mix, num_tokens=self.moe_counts) result = LayerState(m["x_next"], m["next_pre_mix"], "tp_local_token") @@ -197,20 +205,26 @@ def run_layer(self, state, attention, moe, *, group_counts): raise ValueError("input and half-layer outputs must not alias") seen.add(key) epoch = self._epoch + 1 - if epoch > (2**31 - 1) // 2: + program_key = id(attention_program.compiled) + attention_epoch = self._attention_epochs.get(program_key, 0) + 1 + if max(epoch, attention_epoch) > (2**31 - 1) // 2: raise OverflowError("communication epoch exhausted; recreate worker") - a["attention_epoch"] = m["moe_epoch"] = ctypes.c_int32(epoch) + # Each compiled Attention owns separate retained signal windows. A + # mode's first call must use epoch 1 even after another mode ran. + a["attention_epoch"] = ctypes.c_int32(attention_epoch) + m["moe_epoch"] = ctypes.c_int32(epoch) # Resolve all arguments before the first collective: a missing MoE # weight must not leave an already-mutated Attention cache behind. - a_args = [a[name] for name in ATTENTION_ARGS] + a_args = [a[name] for name in argument_names] m_args = [m[name] for name in MOE_ARGS] self._check_weights(m) for rank, (global_n, local_n) in enumerate(zip(group, local)): self.attention_counts[rank, 0] = global_n self.moe_counts[rank] = local_n self._epoch = epoch + self._attention_epochs[program_key] = attention_epoch try: - self.worker.run(self.programs[0].compiled, *a_args, config=self.run_config) + self.worker.run(attention_program.compiled, *a_args, config=self.run_config) self.worker.run(self.programs[1].compiled, *m_args, config=self.run_config) except BaseException: self._failed = True diff --git a/tests/unit/test_v41_swa_segment.py b/tests/unit/test_v41_swa_segment.py index 013d5632..1ac66c5f 100644 --- a/tests/unit/test_v41_swa_segment.py +++ b/tests/unit/test_v41_swa_segment.py @@ -18,6 +18,7 @@ from pypto_serving.model.deepseek_v41.swa_segment import ( ATTENTION_ARGS, MOE_ARGS, SegmentTopology, SwaSegment, make_segment_worker, ) +from pypto_serving.model.deepseek_v41.prefill_segment import PREFILL_ARGUMENTS, PrefillSegment @pytest.mark.parametrize("counts,expected", [ @@ -75,6 +76,47 @@ def test_device_handle_handoff_and_next_layer(monkeypatch): assert segment.moe_counts.tolist() == [16, 3] +def test_switching_attention_modes_keeps_per_program_epochs(monkeypatch): + base, state, a, m = fixture(monkeypatch) + programs = {mode: SimpleNamespace(compiled=object()) for mode in ("swa", "c2a_full", "c1a_reindex")} + segment = PrefillSegment(base.worker, programs, base.programs[1], base.topology, + base.attention_counts, base.moe_counts, None) + # The scratch/output allocations are owned and distinct for each half-layer. + original = state + sequence = ("swa", "swa", "c2a_full", "c1a_reindex", "c2a_full", "swa") + for mode in sequence: + args = dict.fromkeys(PREFILL_ARGUMENTS[mode]) + args.update(output=a["output"], x_hc_out=a["output"], next_pre_mix=a["next_pre_mix"]) + result = segment.run_layer(state, args, m, group_counts=[16, 3], mode=mode) + last = segment.worker.run.call_args_list[-2:] + assert last[0].args[0] is programs[mode].compiled + assert last[0].args[1] is state.residual + assert last[0].args[2] is state.pre_mix + assert last[1].args[1] is a["output"] + assert result.residual is m["x_next"] + m.update(x_next=state.residual, next_pre_mix=state.pre_mix) + state = result + calls = segment.worker.run.call_args_list + assert [c.args[-1].value for c in calls[::2]] == [1, 2, 1, 1, 2, 3] + assert [c.args[-1].value for c in calls[1::2]] == [1, 2, 3, 4, 5, 6] + assert original.layout == "tp_local_token" + + +def test_missing_compressed_metadata_fails_before_mutating_any_cache(monkeypatch): + base, state, a, m = fixture(monkeypatch) + segment = PrefillSegment(base.worker, {"c2a_full": base.programs[0]}, base.programs[1], + base.topology, base.attention_counts, base.moe_counts, None) + args = dict.fromkeys(PREFILL_ARGUMENTS["c2a_full"]) + args.update(x_hc_out=a["output"], next_pre_mix=a["next_pre_mix"]) + del args["state_block_table"] + with pytest.raises(KeyError, match="state_block_table"): + segment.run_layer(state, args, m, group_counts=[16, 0], mode="c2a_full") + with pytest.raises(ValueError, match="not compiled"): + segment.run_layer(state, args, m, group_counts=[16, 0], mode="c1a_full") + assert not segment.worker.run.called + assert segment._epoch == 0 and segment._attention_epochs == {} + + def test_failed_dispatch_poisoned(monkeypatch): segment, state, a, m = fixture(monkeypatch) segment.worker.run.side_effect = RuntimeError("device failure") diff --git a/tools/compile_v41_prefill_segments.py b/tools/compile_v41_prefill_segments.py new file mode 100644 index 00000000..8894e767 --- /dev/null +++ b/tools/compile_v41_prefill_segments.py @@ -0,0 +1,36 @@ +# Copyright (c) PyPTO Contributors. +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of +# CANN Open Software License Agreement Version 2.0 (the "License"). +# Please refer to the License for details. You may not use this file except in compliance with the License. +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. +# See LICENSE in the root of the software repository for the full text of the License. +# ----------------------------------------------------------------------------------------------------------- +"""Compile the selected V4.1 prefill composites without allocating NPU devices.""" +import argparse + + +def main(): + from pypto_serving.model.deepseek_v41.prefill_segment import PREFILL_ARGUMENTS, compile_prefill_segments + from pypto_serving.model.deepseek_v41.swa_segment import SegmentTopology + + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--lib-root", required=True) + parser.add_argument("--tp", type=int, default=2) + parser.add_argument("--dp", type=int, default=2) + parser.add_argument("--modes", nargs="+", choices=tuple(PREFILL_ARGUMENTS), required=True) + parser.add_argument("--build-dir", default="build_output/v41-prefill-composites") + args = parser.parse_args() + from pypto.ir import DistributedConfig + from pypto.runtime import RunConfig + from pypto_serving.model.common.compiler.compiler import KernelCompiler + + topology = SegmentTopology(tp=args.tp, dp=args.dp) + config = RunConfig(platform="a5", distributed_config=DistributedConfig(device_ids=list(range(topology.world)))) + compiler = KernelCompiler(run_config=config, cache_dir=args.build_dir) + attention, _ = compile_prefill_segments(compiler, args.lib_root, topology, args.modes) + print("COMPILE PASS:", ", ".join(attention), "+ packed-FP4 MoE; no device execution", flush=True) + + +if __name__ == "__main__": + main() From 32641e88c0c9a982d26434e83ad2538fc11426ed Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Tue, 29 Sep 2026 02:14:11 +0800 Subject: [PATCH 42/78] feat(v41): prepare producer-owned compressed attention weight bundles --- .../model/deepseek_v41/swa_weights.py | 35 +++++++++++++++++-- .../model/deepseek_v41/test_weight_loader.py | 23 ++++++++++++ 2 files changed, 56 insertions(+), 2 deletions(-) diff --git a/pypto_serving/model/deepseek_v41/swa_weights.py b/pypto_serving/model/deepseek_v41/swa_weights.py index 9e31049c..5c2adb8e 100644 --- a/pypto_serving/model/deepseek_v41/swa_weights.py +++ b/pypto_serving/model/deepseek_v41/swa_weights.py @@ -59,6 +59,21 @@ def load_swa_layer_weights(model_dir, layer_id, topology, *, max_bundle_bytes=32 or the accumulated weights of other layers. Individual payload reads retain V41WeightLoader's separate pre-read budget. """ + return _load_layer_weights(model_dir, layer_id, topology, max_bundle_bytes, swa_only=True) + + +def load_prefill_layer_weights(model_dir, layer_id, topology, *, max_bundle_bytes=32 << 30): + """Load this layer's real Attention/MoE weights for the prefill SP entries. + + Full modes own compressor and index-key weights; Reindex owns only index + query/gating weights; Reuse owns neither. Producer-owned caches and unused + C1A ABI buffers must be supplied by the resource adapter, not fabricated by + the loader. Index heads are replicated, matching lib's global INDEX_H. + """ + return _load_layer_weights(model_dir, layer_id, topology, max_bundle_bytes, swa_only=False) + + +def _load_layer_weights(model_dir, layer_id, topology, max_bundle_bytes, *, swa_only): if type(max_bundle_bytes) is not int or max_bundle_bytes <= 0: raise ValueError("max_bundle_bytes must be positive") attention, moe = [], [] @@ -66,8 +81,10 @@ def load_swa_layer_weights(model_dir, layer_id, topology, *, max_bundle_bytes=32 for rank in range(topology.world): loader = V41WeightLoader(model_dir, tp_size=topology.tp, tp_rank=rank % topology.tp, ep_size=topology.world, ep_rank=rank, max_load_bytes=512 << 20) - if (type(layer_id) is not int or not 0 <= layer_id < loader.config.num_hidden_layers - or loader.text["compress_ratios"][layer_id] != 0): + if type(layer_id) is not int or not 0 <= layer_id < loader.config.num_hidden_layers: + raise ValueError("selected layer must be a backbone layer") + ratio = loader.text["compress_ratios"][layer_id] + if swa_only and ratio != 0: raise ValueError("selected layer must be a SWA backbone layer") prefix = f"layers.{layer_id}." a, m = {}, {} @@ -93,6 +110,20 @@ def load(name): a[name] = bundle.weight if bundle.scale is not None: a[name + "_scale"] = bundle.scale + if layer_id in loader.text["kv_source_layer_ids"]: + for target, source in { + "compressor_wkv": "attn.compressor.wkv.weight", + "compressor_norm_weight": "attn.compressor.norm.weight", + "index_wk": "attn.indexer.wk.weight", + "index_norm_weight": "attn.indexer.k_norm.weight", + }.items(): + a[target] = load(source).weight + if ratio == 2: + a["compressor_wgate"] = load("attn.compressor.wgate.weight").weight + if layer_id in loader.text["index_source_layer_ids"]: + bundle = load("attn.indexer.wq_b.weight") + a["index_wq_b"], a["index_wq_b_scale"] = bundle.weight, bundle.scale + a["index_weights_proj"] = load("attn.indexer.weights_proj.weight").weight for target, source in { "hc_ffn_fn": "hc_ffn_fn", "hc_ffn_scale": "hc_ffn_scale", "hc_ffn_base": "hc_ffn_base", "norm_weight": "ffn_norm.weight", "gate_weight": "ffn.gate.weight", diff --git a/tests/unit/model/deepseek_v41/test_weight_loader.py b/tests/unit/model/deepseek_v41/test_weight_loader.py index c933ede9..b194b5a3 100644 --- a/tests/unit/model/deepseek_v41/test_weight_loader.py +++ b/tests/unit/model/deepseek_v41/test_weight_loader.py @@ -125,6 +125,29 @@ def test_indexer_keeps_all_heads_on_every_tp_rank(checkpoint): assert query.tp_rank == gate.tp_rank == rank +def test_c2a_full_bundle_preserves_composite_dtypes_and_index_heads(checkpoint): + from pypto_serving.model.deepseek_v41.swa_segment import SegmentTopology + from pypto_serving.model.deepseek_v41.swa_weights import load_prefill_layer_weights, load_swa_layer_weights + + path, _, tensors = checkpoint + topology = SegmentTopology(tp=2, dp=1) + with pytest.raises(ValueError, match="SWA"): + load_swa_layer_weights(path, 0, topology) + attention, moe = load_prefill_layer_weights(path, 0, topology) + assert attention["index_wq_b"].shape == (2, 256, 256) + assert attention["index_weights_proj"].shape == (2, 256, 8) + assert attention["wq_b"].shape == (2, 256, 256) # Half of the 512 main-head columns. + for rank in range(2): + assert torch.equal(attention["compressor_wkv"][rank], + tensors["layers.0.attn.compressor.wkv.weight"].float().T) + assert attention["compressor_wgate"].dtype == torch.float32 + assert attention["index_wk"].dtype == torch.bfloat16 + assert moe["routed_w1"].dtype == torch.uint8 + assert moe["routed_w1"].shape[1] == 1 # One whole expert per EP rank. + assert torch.equal(attention["index_wq_b"][0].view(torch.uint8), + attention["index_wq_b"][1].view(torch.uint8)) + + def test_wo_a_group_dequantization(checkpoint): path, _, tensors = checkpoint result = V41WeightLoader(path, tp_size=2, tp_rank=1).load("layers.0.attn.wo_a.weight") From b032b29cc34a6da962f969c70877028d49d0649a Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Tue, 29 Sep 2026 02:27:31 +0800 Subject: [PATCH 43/78] feat(v41): lower compressed pages and stable compressor state metadata --- docs/developer-guide/v41-swa-segment.md | 9 ++ .../model/deepseek_v41/compressed_metadata.py | 135 ++++++++++++++++++ .../deepseek_v41/test_compressed_metadata.py | 83 +++++++++++ tools/compile_v41_prefill_segments.py | 4 +- 4 files changed, 229 insertions(+), 2 deletions(-) create mode 100644 pypto_serving/model/deepseek_v41/compressed_metadata.py create mode 100644 tests/unit/model/deepseek_v41/test_compressed_metadata.py diff --git a/docs/developer-guide/v41-swa-segment.md b/docs/developer-guide/v41-swa-segment.md index b555ab2f..f558a93d 100644 --- a/docs/developer-guide/v41-swa-segment.md +++ b/docs/developer-guide/v41-swa-segment.md @@ -209,3 +209,12 @@ allocation/lowering, producer bindings and device numerical validation remain required. `tools/compile_v41_prefill_segments.py --lib-root ... --modes c2a_full` provides an explicit compilation-only check without allocating devices. All programs must be registered with the same persistent worker before execution. + +`compressed_metadata.prepare_compressed_metadata` prepares C1A/C2A positions, +causal compressed lengths, publication slots, index page tables and stable +compressor state IDs from a `ForwardStep`. It requires jointly numbered KV/index +pools, as current composites publish their index key at the compressed KV slot. +Ratio-2 publishes at odd positions and rotates at the pair's first position; +request state IDs survive batch reordering. Top-K/candidate selection remains +inside lib, and producer buffers stay caller-owned. Host tests cover pair/page +boundaries and continuation; this does not establish device lifecycle readiness. diff --git a/pypto_serving/model/deepseek_v41/compressed_metadata.py b/pypto_serving/model/deepseek_v41/compressed_metadata.py new file mode 100644 index 00000000..f627e259 --- /dev/null +++ b/pypto_serving/model/deepseek_v41/compressed_metadata.py @@ -0,0 +1,135 @@ +# Copyright (c) PyPTO Contributors. +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of +# CANN Open Software License Agreement Version 2.0 (the "License"). +# Please refer to the License for details. You may not use this file except in compliance with the License. +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. +# See LICENSE in the root of the software repository for the full text of the License. +# ----------------------------------------------------------------------------------------------------------- +"""Request-owned C1A/C2A page and compressor metadata for lib SP composites.""" +from dataclasses import dataclass + +import torch + +from .request_state import ForwardStep +from .swa_segment import SegmentTopology + + +@dataclass(frozen=True) +class CompressedMetadata: + position_ids: torch.Tensor + request_ids: torch.Tensor + query_start_loc: torch.Tensor + compressed_lens: torch.Tensor + compressed_slots: torch.Tensor + compressed_rope_positions: torch.Tensor + index_block_table: torch.Tensor + state_block_table: torch.Tensor + group_counts: tuple[int, ...] + + +def prepare_compressed_metadata( + step: ForwardStep, + topology: SegmentTopology, + *, + ratio: int, + compressed_group: str, + cache_pages: int, + max_requests: int, + state_blocks: int, + max_prepare_bytes: int = 16 << 20, +) -> CompressedMetadata: + """Lower full-history pages with joint KV/index physical page numbering. + + The resource allocator must use the same page IDs for compressed KV and + index-key pools. Each pool stores 128 compressed rows per page. This does + not select candidates/Top-K or materialize compressed attention indices. + Those remain device outputs from the appropriate lib producer composite. + + Request IDs are local packed batch rows; state slots are stable ledger IDs. + Ratio-2 publishes only completed pairs, rotating them at the pair's first + position. An odd-length chunk leaves its tail in the request's state block. + The formulas match lib metadata.paged_slots/compressor_metadata at fbe92bfc. + """ + if not isinstance(step, ForwardStep) or not isinstance(topology, SegmentTopology): + raise ValueError("compressed metadata requires a ForwardStep and SegmentTopology") + if type(ratio) is not int or ratio not in (1, 2): + raise ValueError("compression ratio must be 1 or 2") + if (type(cache_pages) is not int or not 0 < cache_pages <= (2**31 - 1) // 128 + or type(max_requests) is not int or max_requests <= 0 + or type(state_blocks) is not int or not 0 < state_blocks <= 2**31 - 1 + or type(max_prepare_bytes) is not int or max_prepare_bytes <= 0): + raise ValueError("cache, request, state and allocation capacities must be positive and fit the ABI") + if not isinstance(compressed_group, str) or not compressed_group: + raise ValueError("compressed_group must identify a jointly numbered KV/index pool") + if step.phase not in ("prefill", "decode"): + raise ValueError("unsupported forward phase") + groups = [[] for _ in range(topology.dp)] + counts, owners, slots, names = [0] * topology.dp, set(), set(), set() + columns = 1 + for request in step.requests: + group, count = request.partition, len(request.token_ids) + if (type(group) is not int or not 0 <= group < topology.dp or count <= 0 + or type(request.start) is not int or not 0 <= request.start < request.end <= 2**31 - 1): + raise ValueError("invalid compressed request partition or extent") + if request.request_id in names: + raise ValueError("duplicate request in forward step") + names.add(request.request_id) + if step.phase == "decode" and count != 1: + raise ValueError("decode requires one token per request") + if counts[group] + count > topology.capacity or len(groups[group]) >= max_requests: + raise ValueError("compressed requests exceed token or request capacity") + if type(request.state_slot) is not int or not 0 <= request.state_slot < state_blocks: + raise ValueError("request state slot is outside the compressor pool") + key = (group, request.state_slot) + if ratio == 2 and key in slots: + raise ValueError("requests must own distinct compressor state slots") + slots.add(key) + pages = request.pages.get(compressed_group) + needed = (request.end // ratio + 127) // 128 + if not isinstance(pages, (tuple, list)) or len(pages) < needed: + raise ValueError("compressed pages must cover all completed history") + for page in pages: + if type(page) is not int or not 0 <= page < cache_pages: + raise ValueError("compressed page is outside the joint KV/index pool") + if (group, page) in owners: + raise ValueError("compressed pages must be private within each DP group") + owners.add((group, page)) + columns = max(columns, len(pages)) + groups[group].append((request, counts[group], tuple(pages))) + counts[group] += count + # Include DP scratch plus replicated TP outputs and conservative small work arrays. + estimate = topology.dp * (topology.capacity * 40 + max_requests * (columns + 4) * 4) * (topology.tp + 1) + if estimate > max_prepare_bytes: + raise ValueError("compressed metadata exceeds its allocation budget") + shape = (topology.dp, topology.capacity) + positions = torch.zeros(shape, dtype=torch.int32) + ids = torch.full(shape, -1, dtype=torch.int32) + lens = torch.zeros(shape, dtype=torch.int32) + writes = torch.full(shape, -1, dtype=torch.int64) + rope = torch.full(shape, -1, dtype=torch.int32) + starts = torch.zeros(topology.dp, max_requests + 1, dtype=torch.int32) + tables = torch.full((topology.dp, max_requests, columns), -1, dtype=torch.int32) + states = torch.full((topology.dp, max_requests, 1), -1, dtype=torch.int32) + for group, requests in enumerate(groups): + for batch_row, (request, offset, pages) in enumerate(requests): + count = len(request.token_ids) + section = slice(offset, offset + count) + p = torch.arange(request.start, request.end, dtype=torch.int32) + positions[group, section] = p + ids[group, section] = batch_row + lens[group, section] = (p + 1) // ratio + if ratio == 2: + states[group, batch_row, 0] = request.state_slot + if pages: + tables[group, batch_row, :len(pages)] = torch.tensor(pages, dtype=torch.int32) + starts[group, batch_row + 1:] = offset + count + for local, position in enumerate(range(request.start, request.end)): + if (position + 1) % ratio == 0: + logical = position // ratio + writes[group, offset + local] = pages[logical // 128] * 128 + logical % 128 + rope[group, offset + local] = position + 1 - ratio + def replicate(tensor): + return tensor.repeat_interleave(topology.tp, dim=0) + return CompressedMetadata(*(replicate(t) for t in (positions, ids, starts, lens, writes, rope, tables, states)), + tuple(counts)) diff --git a/tests/unit/model/deepseek_v41/test_compressed_metadata.py b/tests/unit/model/deepseek_v41/test_compressed_metadata.py new file mode 100644 index 00000000..81535637 --- /dev/null +++ b/tests/unit/model/deepseek_v41/test_compressed_metadata.py @@ -0,0 +1,83 @@ +# Copyright (c) PyPTO Contributors. +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of +# CANN Open Software License Agreement Version 2.0 (the "License"). +# Please refer to the License for details. You may not use this file except in compliance with the License. +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. +# See LICENSE in the root of the software repository for the full text of the License. +# ----------------------------------------------------------------------------------------------------------- +"""Compressed history and stable compressor slots across chunk boundaries.""" +import pytest +import torch + +from pypto_serving.model.deepseek_v41.compressed_metadata import prepare_compressed_metadata +from pypto_serving.model.deepseek_v41.request_state import ForwardStep, RequestSlice +from pypto_serving.model.deepseek_v41.swa_segment import SegmentTopology + +TOPOLOGY = SegmentTopology(tp=2, dp=2, local_capacity=2) + + +def req(name="a", group=0, slot=5, start=255, count=3, pages=(7, 1)): + return RequestSlice(name, group, slot, start, tuple(range(count)), start + count, {"cmp": pages}) + + +def prepare(requests, ratio=2, **kwargs): + return prepare_compressed_metadata(ForwardStep("prefill", tuple(requests), 1), TOPOLOGY, + ratio=ratio, compressed_group="cmp", cache_pages=10, max_requests=3, state_blocks=10, **kwargs) + + +def test_pair_completion_page_crossing_and_interleaved_request_order(): + metadata = prepare([req("b", 1, 8, 0, 1, ()), req(), req("c", 0, 8, 0, 1, ())]) + assert metadata.group_counts == (4, 1) + assert metadata.position_ids[0].tolist() == [255, 256, 257, 0] + assert metadata.request_ids[0].tolist() == [0, 0, 0, 1] + assert metadata.query_start_loc[0].tolist() == [0, 3, 4, 4] + assert metadata.compressed_lens[0].tolist() == [128, 128, 129, 0] + assert metadata.compressed_slots[0].tolist() == [1023, -1, 128, -1] + assert metadata.compressed_rope_positions[0].tolist() == [254, -1, 256, -1] + assert metadata.state_block_table[0, :, 0].tolist() == [5, 8, -1] + assert metadata.index_block_table[0].tolist() == [[7, 1], [-1, -1], [-1, -1]] + assert metadata.state_block_table[2, :, 0].tolist() == [8, -1, -1] + assert metadata.request_ids[2].tolist() == [0, -1, -1, -1] + for tensor in (metadata.position_ids, metadata.query_start_loc, metadata.compressed_lens, + metadata.compressed_slots, metadata.index_block_table, metadata.state_block_table): + assert torch.equal(tensor[0], tensor[1]) and torch.equal(tensor[2], tensor[3]) + + +def test_decode_continuation_keeps_state_owner_after_batch_reordering(): + requests = [req("c", 0, 8, 1, 1, (2,)), req("a", 0, 5, 258, 1)] + step = ForwardStep("decode", tuple(requests), 2) + metadata = prepare_compressed_metadata(step, TOPOLOGY, ratio=2, compressed_group="cmp", + cache_pages=10, max_requests=3, state_blocks=10) + assert metadata.state_block_table[0, :, 0].tolist() == [8, 5, -1] + assert metadata.compressed_slots[0].tolist() == [256, -1, -1, -1] + assert metadata.compressed_rope_positions[0].tolist() == [0, -1, -1, -1] + assert metadata.compressed_lens[0].tolist() == [1, 129, 0, 0] + assert (metadata.state_block_table[2:] == -1).all() + assert (metadata.compressed_slots[2:] == -1).all() + + +def test_ratio_one_publishes_each_token_and_has_no_pending_state(): + metadata = prepare([req(start=127, count=3)], ratio=1) + assert metadata.compressed_slots[0].tolist() == [1023, 128, 129, -1] + assert metadata.compressed_lens[0].tolist() == [128, 129, 130, 0] + assert metadata.compressed_rope_positions[0].tolist() == [127, 128, 129, -1] + assert (metadata.state_block_table == -1).all() + + +@pytest.mark.parametrize("requests,match", [ + ([req(pages=(7,))], "completed history"), + ([req(pages=(7, 7))], "private"), + ([req(pages=(7, 10))], "outside"), + ([req(), req("b", slot=5, start=0, count=1, pages=())], "state slots"), + ([req(slot=10)], "state slot"), + ([req(start=-1)], "extent"), +]) +def test_invalid_compressed_ownership_fails_before_dispatch(requests, match): + with pytest.raises(ValueError, match=match): + prepare(requests) + + +def test_compressed_allocation_budget(): + with pytest.raises(ValueError, match="budget"): + prepare([req()], max_prepare_bytes=1) diff --git a/tools/compile_v41_prefill_segments.py b/tools/compile_v41_prefill_segments.py index 8894e767..b8e4d4d1 100644 --- a/tools/compile_v41_prefill_segments.py +++ b/tools/compile_v41_prefill_segments.py @@ -6,7 +6,7 @@ # INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. # See LICENSE in the root of the software repository for the full text of the License. # ----------------------------------------------------------------------------------------------------------- -"""Compile the selected V4.1 prefill composites without allocating NPU devices.""" +"""Generate selected V4.1 prefill code without device binary assembly or execution.""" import argparse @@ -29,7 +29,7 @@ def main(): config = RunConfig(platform="a5", distributed_config=DistributedConfig(device_ids=list(range(topology.world)))) compiler = KernelCompiler(run_config=config, cache_dir=args.build_dir) attention, _ = compile_prefill_segments(compiler, args.lib_root, topology, args.modes) - print("COMPILE PASS:", ", ".join(attention), "+ packed-FP4 MoE; no device execution", flush=True) + print("CODEGEN PASS:", ", ".join(attention), "+ packed-FP4 MoE; no binary assembly/device execution", flush=True) if __name__ == "__main__": From 589c5484672ea9c09ee440f16c0f3bd532efb2f3 Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Tue, 29 Sep 2026 02:47:05 +0800 Subject: [PATCH 44/78] feat(v41): bind resident prefill producers across layer modes --- docs/developer-guide/v41-swa-segment.md | 15 ++- .../model/deepseek_v41/prefill_segment.py | 100 ++++++++++++++++ .../deepseek_v41/test_prefill_producers.py | 113 ++++++++++++++++++ 3 files changed, 227 insertions(+), 1 deletion(-) create mode 100644 tests/unit/model/deepseek_v41/test_prefill_producers.py diff --git a/docs/developer-guide/v41-swa-segment.md b/docs/developer-guide/v41-swa-segment.md index f558a93d..00ff2133 100644 --- a/docs/developer-guide/v41-swa-segment.md +++ b/docs/developer-guide/v41-swa-segment.md @@ -205,7 +205,7 @@ MoE program advances every layer. Missing arguments are rejected before either half-layer runs. A failed dispatch poisons the entire worker. This dispatch support is not full-model readiness: real compressed-cache -allocation/lowering, producer bindings and device numerical validation remain +allocation, complete request integration and device numerical validation remain required. `tools/compile_v41_prefill_segments.py --lib-root ... --modes c2a_full` provides an explicit compilation-only check without allocating devices. All programs must be registered with the same persistent worker before execution. @@ -218,3 +218,16 @@ Ratio-2 publishes at odd positions and rotates at the pair's first position; request state IDs survive batch reordering. Top-K/candidate selection remains inside lib, and producer buffers stay caller-owned. Host tests cover pair/page boundaries and continuation; this does not establish device lifecycle readiness. + +`PrefillSegment.run_chain` binds resident producer handles before dispatching +consecutive layers for one packed request step. Full layers own compressed KV +and index caches; Reindex consumes the earlier candidate mask and produces its +own Top-K; Reuse consumes that index producer's physical Top-K rows directly. +Every layer keeps its own SWA cache. C1A's unused common-ABI weight slots refer +to real producer weights, without loading nonexistent Reuse weights. Each +producer must appear earlier in the same chain, preventing stale transient +selections from a previous step from satisfying an omitted producer. All +argument names and compiled modes are checked before the first dispatch. +The caller still owns allocation, step metadata, reset and buffer lifetime. +Unit tests cover the checkpoint's 40-layer producer plan; this is not evidence +of 40-layer device execution or numerical acceptance. diff --git a/pypto_serving/model/deepseek_v41/prefill_segment.py b/pypto_serving/model/deepseek_v41/prefill_segment.py index 9349e104..faea40af 100644 --- a/pypto_serving/model/deepseek_v41/prefill_segment.py +++ b/pypto_serving/model/deepseek_v41/prefill_segment.py @@ -15,6 +15,7 @@ import importlib from .swa_segment import ATTENTION_ARGS, SwaSegment, load_segment_modules +from .execution_plan import LayerPlan _WEIGHTS = ( @@ -43,6 +44,75 @@ } +def bind_prefill_producers(plans, arguments): + """Bind one ordered forward chain to its actual producer allocations. + + Like V4's worker-owned cache arguments, handles stay resident for the whole + dispatch. V4.1 additionally shares compressed KV/index pools and physical + Top-K rows according to LayerPlan. This performs no tensor computation. + Every producer must execute earlier in this chain, so a previous request's + transient Top-K/candidate rows cannot satisfy a missing dependency. + + C1A's common ABI retains unused Full weights in Reindex/Reuse. Bind those + slots to their real producer weights; never allocate placeholder weights. + Full/Reindex do not read compressed_indices; their own Top-K allocation is + a valid shape-compatible binding for that unused slot (lib fbe92bfc). + """ + plans = tuple(plans) + if not plans or any(not isinstance(p, LayerPlan) for p in plans): + raise ValueError("prefill chain requires LayerPlan entries") + if any(b.layer_id != a.layer_id + 1 for a, b in zip(plans, plans[1:])): + raise ValueError("prefill chain must execute consecutive layers in order") + result, seen = {}, {} + for plan in plans: + if plan.mode not in PREFILL_ARGUMENTS: + raise ValueError("unsupported prefill mode") + current = dict(arguments[plan.layer_id]) + if plan.mode != "swa": + family, mode = plan.mode.split("_") + + def producer(source, kind): + if source == plan.layer_id: + if mode != "full" and not (kind == "index" and mode == "reindex"): + raise ValueError(f"{plan.mode} cannot produce its own {kind}") + return current + if source not in seen or not seen[source].mode.startswith(family + "_"): + raise ValueError(f"missing preceding {kind} producer for layer {plan.layer_id}") + source_mode = seen[source].mode.split("_")[1] + if source_mode != "full" and not (kind == "index" and source_mode == "reindex"): + raise ValueError(f"invalid {kind} producer mode") + return result[source] + + if mode == "full" and (plan.kv_source != plan.layer_id or plan.index_source != plan.layer_id): + raise ValueError("Full must own its KV and index outputs") + if mode == "reindex" and (family != "c1a" or plan.index_source != plan.layer_id): + raise ValueError("Reindex must own its C1A index output") + kv = producer(plan.kv_source, "KV") + index = producer(plan.index_source, "index") + if plan.index_source != plan.layer_id and seen[plan.index_source].kv_source != plan.kv_source: + raise ValueError("index selection must address the same KV producer") + for name in ("compressed_cache", "compressed_cache_scale"): + current[name] = kv[name] + if family == "c1a": + candidate = producer(plan.candidate_source, "candidate") + if plan.candidate_source != plan.kv_source: + raise ValueError("C1A candidates must use the same KV producer") + for name in ("index_cache", "index_cache_scale", "compressor_wkv", "compressor_norm_weight", + "index_wk", "index_norm_weight"): + current[name] = kv[name] + current["candidate_mask"] = candidate["candidate_mask"] + if mode == "reuse": + for name in ("index_wq_b", "index_wq_b_scale", "index_weights_proj", "topk_indices"): + current[name] = index[name] + current["compressed_indices"] = index["topk_indices"] + elif mode == "reuse": + current["compressed_indices"] = index["topk_indices"] + elif any(source is not None for source in (plan.kv_source, plan.index_source, plan.candidate_source)): + raise ValueError("SWA must not declare compressed producers") + result[plan.layer_id], seen[plan.layer_id] = current, plan + return result + + def compile_prefill_segments(compiler, lib_root, topology, modes): """Compile existing composite entries, sharing one packed-FP4 MoE program. @@ -96,3 +166,33 @@ def run_layer(self, state, attention, moe, *, group_counts, mode): input_mix="incoming_pre_mix" if mode == "swa" else "pre_mix", output_residual="output" if mode == "swa" else "x_hc_out", ) + + def run_chain(self, state, plans, attention, moe, *, group_counts): + """Execute one request step through its producers and consumers in order. + + The resource owner must prepare every layer for the SAME packed step + and retain all allocations until this synchronous call returns. This + is still a bounded prefill segment, not a complete serving backend. + """ + from .swa_segment import MOE_ARGS + + plans = tuple(plans) + bound = bind_prefill_producers(plans, attention) + self.topology.counts(group_counts) + # Resolve the whole chain before any cache mutation. Runtime state, + # active counts and per-program epochs are provided by run_layer. + dynamic = {"x_hc", "pre_mix", "incoming_pre_mix", "num_tokens", "attention_epoch", "moe_epoch"} + for plan in plans: + if plan.mode not in self.attention_programs: + raise ValueError(f"prefill mode was not compiled: {plan.mode}") + for name in PREFILL_ARGUMENTS[plan.mode]: + if name not in dynamic: + bound[plan.layer_id][name] + for name in MOE_ARGS: + if name not in dynamic: + moe[plan.layer_id][name] + self._check_weights(moe[plan.layer_id]) + for plan in plans: + state = self.run_layer(state, bound[plan.layer_id], moe[plan.layer_id], + group_counts=group_counts, mode=plan.mode) + return state diff --git a/tests/unit/model/deepseek_v41/test_prefill_producers.py b/tests/unit/model/deepseek_v41/test_prefill_producers.py new file mode 100644 index 00000000..38aba52d --- /dev/null +++ b/tests/unit/model/deepseek_v41/test_prefill_producers.py @@ -0,0 +1,113 @@ +# Copyright (c) PyPTO Contributors. +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of +# CANN Open Software License Agreement Version 2.0 (the "License"). +# Please refer to the License for details. You may not use this file except in compliance with the License. +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. +# See LICENSE in the root of the software repository for the full text of the License. +# ----------------------------------------------------------------------------------------------------------- +"""Producer buffer handoff contracts, independent of device execution.""" +import json +from pathlib import Path +from unittest.mock import Mock + +import pytest + +from pypto_serving.model.deepseek_v41.execution_plan import LayerPlan, plan_layers +from pypto_serving.model.deepseek_v41.prefill_segment import ( + PREFILL_ARGUMENTS, PrefillSegment, bind_prefill_producers, +) +from pypto_serving.model.deepseek_v41.swa_segment import MOE_ARGS, SegmentTopology + + +def plans_and_buffers(): + path = Path(__file__).resolve().parents[4] / "tests/fixtures/deepseek_v41/config.json" + plans = plan_layers(json.loads(path.read_text())) + buffers = {p.layer_id: {name: object() for name in PREFILL_ARGUMENTS[p.mode]} for p in plans} + # Model loader deliberately omits weights owned by another layer. + for p in plans: + if p.mode in ("c1a_reindex", "c1a_reuse"): + for name in ("compressor_wkv", "compressor_norm_weight", "index_wk", "index_norm_weight"): + del buffers[p.layer_id][name] + if p.mode == "c1a_reuse": + for name in ("index_wq_b", "index_wq_b_scale", "index_weights_proj"): + del buffers[p.layer_id][name] + return plans, buffers + + +def test_actual_40_layer_plan_shares_producers_without_copying_or_mutating_inputs(): + plans, original = plans_and_buffers() + bound = bind_prefill_producers(plans, original) + for p in plans: + current = bound[p.layer_id] + assert current["window_cache"] is original[p.layer_id]["window_cache"] + if p.mode == "swa": + continue + assert current["compressed_cache"] is original[p.kv_source]["compressed_cache"] + if p.mode.endswith("reuse"): + assert current["compressed_indices"] is original[p.index_source]["topk_indices"] + if p.mode.startswith("c1a"): + assert current["candidate_mask"] is original[20]["candidate_mask"] + assert current["index_cache"] is original[20]["index_cache"] + assert current["compressor_wkv"] is original[20]["compressor_wkv"] + assert bound[24]["topk_indices"] is original[24]["topk_indices"] + assert bound[25]["compressed_indices"] is original[24]["topk_indices"] + assert "compressor_wkv" not in original[24] + assert original[21]["compressed_indices"] is not bound[21]["compressed_indices"] + + +@pytest.mark.parametrize("plans", [ + (LayerPlan(3, "c2a_reuse", 2, 2, None),), + (LayerPlan(20, "c1a_full", 20, 20, 20), LayerPlan(22, "c1a_reuse", 20, 20, 20)), + (LayerPlan(20, "c1a_full", 20, 20, 20), LayerPlan(21, "c1a_reuse", 20, 21, 20)), + (LayerPlan(2, "c2a_full", 2, 2, None), LayerPlan(3, "c1a_reuse", 2, 2, 2)), + (LayerPlan(0, "swa", 0, None, None),), +]) +def test_incomplete_or_invalid_chain_cannot_consume_stale_selection(plans): + _, buffers = plans_and_buffers() + with pytest.raises(ValueError): + bind_prefill_producers(plans, buffers) + + +def test_selection_from_different_compressed_pool_is_rejected(): + plans, buffers = plans_and_buffers() + changed = list(plans) + changed[9] = LayerPlan(9, "c2a_reuse", 8, 2, None) + with pytest.raises(ValueError, match="same KV"): + bind_prefill_producers(changed, buffers) + + +def test_run_chain_hands_state_forward_and_resolves_late_dependencies_before_dispatch(): + plans, buffers = plans_and_buffers() + plans = plans[2:4] + segment = PrefillSegment.__new__(PrefillSegment) + segment.topology = SegmentTopology(tp=2, dp=2) + segment.attention_programs = {p.mode: object() for p in plans} + segment._check_weights = Mock() + first, second, final = object(), object(), object() + segment.run_layer = Mock(side_effect=[second, final]) + moe = {p.layer_id: dict.fromkeys(MOE_ARGS, object()) for p in plans} + late_weight = moe[3].pop("routed_w3") + with pytest.raises(KeyError, match="routed_w3"): + segment.run_chain(first, plans, buffers, moe, group_counts=[17, 0]) + segment.run_layer.assert_not_called() + moe[3]["routed_w3"] = late_weight + assert segment.run_chain(first, plans, buffers, moe, group_counts=[17, 0]) is final + calls = segment.run_layer.call_args_list + assert calls[0].args[0] is first and calls[1].args[0] is second + assert calls[1].args[1]["compressed_indices"] is buffers[2]["topk_indices"] + assert all(c.kwargs["group_counts"] == [17, 0] for c in calls) + + +def test_runtime_failure_prevents_later_consumer_dispatch(): + plans, buffers = plans_and_buffers() + plans = plans[2:4] + segment = PrefillSegment.__new__(PrefillSegment) + segment.topology = SegmentTopology(tp=2, dp=2) + segment.attention_programs = {p.mode: object() for p in plans} + segment._check_weights = Mock() + segment.run_layer = Mock(side_effect=RuntimeError("producer failed")) + moe = {p.layer_id: dict.fromkeys(MOE_ARGS, object()) for p in plans} + with pytest.raises(RuntimeError, match="producer failed"): + segment.run_chain(object(), plans, buffers, moe, group_counts=[17, 0]) + assert segment.run_layer.call_count == 1 From 6f5251625fbb9dd002f71f22fdb8435561bd44aa Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Tue, 29 Sep 2026 02:51:05 +0800 Subject: [PATCH 45/78] test(v41): validate real-weight C2A producer and reuse chain --- docs/developer-guide/v41-swa-segment.md | 11 ++ tools/validate_v41_c2a_chain.py | 234 ++++++++++++++++++++++++ 2 files changed, 245 insertions(+) create mode 100644 tools/validate_v41_c2a_chain.py diff --git a/docs/developer-guide/v41-swa-segment.md b/docs/developer-guide/v41-swa-segment.md index 00ff2133..740aa1ca 100644 --- a/docs/developer-guide/v41-swa-segment.md +++ b/docs/developer-guide/v41-swa-segment.md @@ -231,3 +231,14 @@ argument names and compiled modes are checked before the first dispatch. The caller still owns allocation, step metadata, reset and buffer lifetime. Unit tests cover the checkpoint's 40-layer producer plan; this is not evidence of 40-layer device execution or numerical acceptance. + +`tools/validate_v41_c2a_chain.py` continues a saved fresh-request embedding +diagnostic through checkpoint layers 2/3 (C2A Full, packed-FP4 MoE, C2A Reuse, +packed-FP4 MoE). It uses the request metadata helpers, empty layer-owned caches, +checkpoint-compatible lib RoPE tables and resident producer bindings. +`--prepare-only` checks real-weight host preparation without allocating devices. +The device path retains each stage's outputs and applies the existing lib +same-input stage comparators after worker shutdown. A native-stage pass does +not resolve the incoming SWA accumulated error or establish a new accumulated +four-layer acceptance standard. Source state and numerical artifacts must be +reported with that limitation. diff --git a/tools/validate_v41_c2a_chain.py b/tools/validate_v41_c2a_chain.py new file mode 100644 index 00000000..9f94a887 --- /dev/null +++ b/tools/validate_v41_c2a_chain.py @@ -0,0 +1,234 @@ +# Copyright (c) PyPTO Contributors. +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of +# CANN Open Software License Agreement Version 2.0 (the "License"). +# Please refer to the License for details. You may not use this file except in compliance with the License. +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. +# See LICENSE in the root of the software repository for the full text of the License. +# ----------------------------------------------------------------------------------------------------------- +"""Bounded real-weight C2A Full -> MoE -> Reuse -> MoE device diagnostic. + +Start from the saved output of the two SWA layers. This does not claim that +that input passed its accumulated precision gate, or run a full model. Only +existing native same-input stage comparators gate this diagnostic. +""" +import argparse +import json +from pathlib import Path +from types import SimpleNamespace + + +def prepare(args, topology, module): + import torch + from golden.spec import TensorSpec + from models.deepseek_v4_1_flash.config import FLASH + from models.deepseek_v4_1_flash.rope_tables import precompute_rope_tables + from pypto_serving.model.deepseek_v41.compressed_metadata import prepare_compressed_metadata + from pypto_serving.model.deepseek_v41.execution_plan import plan_layers + from pypto_serving.model.deepseek_v41.request_state import ForwardStep, RequestSlice + from pypto_serving.model.deepseek_v41.swa_metadata import gather_swa_rope_rows, prepare_swa_window_metadata + from pypto_serving.model.deepseek_v41.swa_weights import load_prefill_layer_weights + + saved = torch.load(args.input_state, map_location="cpu", weights_only=True) + if saved.get("request_inputs") is None or saved.get("input_source") != "embeddings": + raise ValueError("input must come from the explicit fresh-request embedding diagnostic") + ids = saved["token_ids"] + if ids is None or tuple(ids.shape) != (topology.dp, topology.capacity): + raise ValueError("saved SWA diagnostic must contain exactly the same packed token capacity") + residual, mix = saved["actual_residual"], saved["actual_pre_mix"] + if residual.shape != (topology.world, topology.local_capacity, 4, 5120) or mix.shape != residual.shape[:-1]: + raise ValueError("saved SWA state has incompatible topology or shape") + if residual.dtype != torch.float32 or mix.dtype != torch.float32: + raise ValueError("saved SWA state must preserve its FP32 storage") + if not torch.isfinite(residual).all() or not torch.isfinite(mix).all(): + raise ValueError("saved SWA state must be finite") + raw = json.loads((Path(args.model_dir) / "config.json").read_text()) + plans = plan_layers(raw)[2:4] + if tuple(p.mode for p in plans) != ("c2a_full", "c2a_reuse"): + raise ValueError("checkpoint layers 2/3 are not the required Full/Reuse pair") + text = raw["text_config"] + for key in ("qk_rope_head_dim", "rope_theta", "compress_rope_theta"): + if text[key] != getattr(FLASH, key): + raise ValueError(f"lib/checkpoint RoPE mismatch: {key}") + for source, target in (("factor", "rope_factor"), ("beta_fast", "beta_fast"), + ("beta_slow", "beta_slow"), ("original_max_position_embeddings", + "original_max_position_embeddings")): + if text["rope_scaling"][source] != getattr(FLASH, target): + raise ValueError(f"lib/checkpoint compressed RoPE mismatch: {source}") + # Fresh first-chunk history at layer 2, private pages per DP group. + pages = (topology.capacity + 127) // 128 + step = ForwardStep("prefill", tuple(RequestSlice(str(g), g, 0, 0, tuple(row.tolist()), + topology.capacity, {"window": tuple(range(pages)), "cmp": tuple(range(pages))}) + for g, row in enumerate(ids)), 1) + tables = precompute_rope_tables(topology.capacity, False) + compressed_tables = precompute_rope_tables(topology.capacity, True) + fixture = SimpleNamespace(tokens=topology.capacity, requests=1, dp=topology.dp, + seed=11, case="mixed", dp_tokens=None, epochs=1, bench=False) + attention, moe = {}, {} + for plan in plans: + mode = plan.mode.removeprefix("c2a_") + values = {s.name: s.create_tensor().contiguous() for s in module.build_specs(fixture, mode, {}) + if isinstance(s, TensorSpec)} + print(f"Loading real checkpoint layer {plan.layer_id}", flush=True) + aw, mw = load_prefill_layer_weights(args.model_dir, plan.layer_id, topology) + values.update(aw, x_hc=residual, pre_mix=mix) + window = prepare_swa_window_metadata(step, topology, cache_pages=values["window_cache"].shape[1]) + values.update(window_slots=window.window_slots, window_indices=window.window_indices) + values["window_cache"].view(torch.uint8).zero_() + values["window_cache_scale"].view(torch.uint8).fill_(127) + if mode == "full": + cm = prepare_compressed_metadata(step, topology, ratio=2, compressed_group="cmp", + cache_pages=values["compressed_cache"].shape[1], max_requests=1, + state_blocks=values["state_cache"].shape[1]) + values.update(token_to_req_indices=cm.request_ids, compressed_lens=cm.compressed_lens, + compressed_slots=cm.compressed_slots, index_block_table=cm.index_block_table, + position_ids=cm.position_ids, query_start_loc=cm.query_start_loc, + state_block_table=cm.state_block_table, + compressed_rope_positions=cm.compressed_rope_positions) + for name, table in zip(("freqs_cos", "freqs_sin", "compressed_freqs_cos", "compressed_freqs_sin"), + (*tables, *compressed_tables)): + values[name] = table.unsqueeze(0).repeat(topology.world, 1, 1) + for name in ("compressed_cache", "index_cache"): + values[name].view(torch.uint8).zero_() + values["compressed_cache_scale"].view(torch.uint8).fill_(0x38) # E4M3 scale 1 + values["index_cache_scale"].view(torch.uint8).fill_(127) # E8M0 scale 1 + values["state_cache"].zero_() + values["topk_indices"].fill_(-1) + else: + values["rope_cos"], values["rope_sin"] = gather_swa_rope_rows(window, tables) + attention[plan.layer_id] = values + moe[plan.layer_id] = dict(mw, next_pre_mix=torch.zeros_like(mix), + x_mixed=torch.zeros(topology.world, topology.local_capacity, 5120, dtype=torch.bfloat16), + x_next=torch.zeros_like(residual), num_tokens=torch.full((topology.world,), + topology.local_capacity, dtype=torch.int32)) + return plans, attention, moe, residual, mix + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--lib-root", required=True) + parser.add_argument("--model-dir", required=True) + parser.add_argument("--input-state", required=True) + parser.add_argument("--devices", default="0,1,2,3") + parser.add_argument("--tp", type=int, default=2) + parser.add_argument("--prepare-only", action="store_true") + parser.add_argument("--build-dir", default="build_output/v41-c2a-chain") + parser.add_argument("--artifact-dir", default=".validation-artifacts/c2a-chain") + parser.add_argument("--ring-heap-mib", type=int, default=4096) + args = parser.parse_args() + import torch + from pypto_serving.model.deepseek_v41.swa_segment import SegmentTopology, load_segment_modules + from pypto_serving.model.deepseek_v41.prefill_segment import ( + PrefillSegment, bind_prefill_producers, compile_prefill_segments, + ) + devices = tuple(map(int, args.devices.split(","))) + if args.tp <= 0 or len(set(devices)) != len(devices) or len(devices) % args.tp: + raise ValueError("unique devices must form complete TP groups") + topology = SegmentTopology(tp=args.tp, dp=len(devices) // args.tp) + torch.set_num_threads(4) + _, moe_module = load_segment_modules(args.lib_root, topology) + from models.deepseek_v4_1_flash import prefill_c2a_full as c2a + plans, attention, moe, residual, mix = prepare(args, topology, c2a) + bound = bind_prefill_producers(plans, attention) + from pypto_serving.model.deepseek_v41.prefill_segment import PREFILL_ARGUMENTS + dynamic = {"attention_epoch", "num_tokens"} + for plan in plans: + for name in PREFILL_ARGUMENTS[plan.mode]: + if name not in dynamic: + assert isinstance(bound[plan.layer_id][name], torch.Tensor), name + print("REAL CHECKPOINT C2A PREPARATION PASS", flush=True) + if args.prepare_only: + return + from pypto.ir import DistributedConfig + from pypto.runtime import RunConfig + from pypto_serving.model.common.compiler.compiler import KernelCompiler + from pypto_serving.model.deepseek_v41.composite import LayerState + from pypto_serving.model.deepseek_v41.swa_segment import make_segment_worker + from golden.validation import ratio_allclose + artifact = Path(args.artifact_dir) + artifact.mkdir(parents=True, exist_ok=True) + if (artifact / "comparison.pt").exists(): + raise FileExistsError("refusing to overwrite a previous comparison") + config = RunConfig(platform="a5", distributed_config=DistributedConfig(device_ids=list(devices)), + ring_heap=args.ring_heap_mib << 20, ring_task_window=131072, ring_dep_pool=131072) + compiler = KernelCompiler(run_config=config, cache_dir=args.build_dir) + programs, ffn = compile_prefill_segments(compiler, args.lib_root, topology, [p.mode for p in plans]) + ac = torch.zeros(topology.world, 1, dtype=torch.int32).share_memory_() + mc = torch.zeros(topology.world, dtype=torch.int32).share_memory_() + captured = {} + for plan in plans: + layer = plan.layer_id + names = ("attn_input", "attn_output", "next_pre_mix", "x_hc_out") + c2a.ATTENTION_STATE[ + plan.mode.removeprefix("c2a_")] + captured[layer] = ({n: torch.empty_like(bound[layer][n]).share_memory_() for n in names}, + {n: torch.empty_like(moe[layer][n]).share_memory_() + for n in ("x_next", "next_pre_mix", "x_mixed")}) + sources = [v for maps in (bound, moe) for values in maps.values() for v in values.values()] + with make_segment_worker([*programs.values(), ffn], config, sources) as worker: + uploaded = {} + def upload(values): + result = {} + for name, value in values.items(): + if name == "num_tokens": + continue + if id(value) not in uploaded: + uploaded[id(value)] = worker.alloc_stacked_tensor(value) + result[name] = uploaded[id(value)] + return result + da = {layer: upload(values) for layer, values in bound.items()} + dm = {layer: upload(values) for layer, values in moe.items()} + runner = PrefillSegment(worker, programs, ffn, topology, ac, mc, config) + runner.run_chain(LayerState(da[2]["x_hc"], da[2]["pre_mix"], "tp_local_token"), + plans, da, dm, group_counts=[topology.capacity] * topology.dp) + for layer, (ca, cm) in captured.items(): + for device, host in ((da[layer], ca), (dm[layer], cm)): + for name, destination in host.items(): + worker.copy_stacked_from(device[name], destination) + for handle in reversed(list(uploaded.values())): + worker.free_stacked_tensor(handle) + # CPU same-input references run only after worker shutdown. Device state + # never depends on these diagnostic readbacks or reference results. + records, passed = [], True + for plan in plans: + layer, mode = plan.layer_id, plan.mode.removeprefix("c2a_") + actual_a, actual_m = captured[layer] + inputs = dict(bound[layer], x_hc=residual, pre_mix=mix) + if mode == "reuse": + for name in ("compressed_cache", "compressed_cache_scale"): + inputs[name] = captured[2][0][name] + inputs["compressed_indices"] = captured[2][0]["topk_indices"] + initial_cache = {n: inputs[n].clone() for n in c2a.MODES[mode][1]} + expected_a = dict(inputs) + for name in actual_a: + expected_a[name] = inputs[name].clone() + c2a.make_golden(mode, 1)(expected_a) + expected_m = dict(moe[layer], x_hc=actual_a["x_hc_out"], pre_mix=actual_a["next_pre_mix"]) + for name in actual_m: + expected_m[name] = moe[layer][name].clone() + moe_module.golden_moe(expected_m) + mc_check = { + "next_pre_mix": ratio_allclose(atol=2.5e-5, rtol=5e-3), + "x_mixed": ratio_allclose(atol=1e-4, rtol=1.0 / 128), + "x_next": moe_module._local_mhc_compare([topology.local_capacity] * topology.world), + } + for label, actual, expected, checks in (("attention", actual_a, expected_a, + c2a.make_compare(mode, 1, initial_cache)), ("moe", actual_m, expected_m, mc_check)): + results = {} + for name, check in checks.items(): + ok, detail = check(actual[name], expected[name], inputs=expected, + actual_outputs=actual, expected_outputs=expected, rtol=1e-3, atol=1e-3) + print(f"NATIVE STAGE layer={layer} {label}.{name}: {ok} {detail}", flush=True) + results[name] = (bool(ok), detail) + passed &= bool(ok) + records.append(dict(layer=layer, stage=label, results=results, actual=actual, + expected={n: expected[n] for n in actual})) + residual, mix = actual_m["x_next"], actual_m["next_pre_mix"] + torch.save({"input_state": str(args.input_state), "stages": records, + "actual_residual": residual, "actual_pre_mix": mix}, artifact / "comparison.pt") + if not passed: + raise AssertionError("C2A chain native stage check failed; see comparison.pt") + print("C2A CHAIN NATIVE STAGES PASS; accumulated full-model acceptance remains pending", flush=True) + + +if __name__ == "__main__": + main() From fd0b7620006558aceb182acdba0331994f7d81bb Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Tue, 29 Sep 2026 03:08:59 +0800 Subject: [PATCH 46/78] test(v41): trace precision boundaries in each DP partition --- tests/unit/test_v41_swa_validation.py | 24 +++++++++++++++++++++++- tools/diagnose_v41_swa_precision.py | 18 ++++++++++++------ 2 files changed, 35 insertions(+), 7 deletions(-) diff --git a/tests/unit/test_v41_swa_validation.py b/tests/unit/test_v41_swa_validation.py index 53d80241..ded90b21 100644 --- a/tests/unit/test_v41_swa_validation.py +++ b/tests/unit/test_v41_swa_validation.py @@ -14,7 +14,29 @@ import torch from tools.validate_v41_swa_segment import compare_saved -from tools.diagnose_v41_swa_precision import quantization_metrics +from tools.diagnose_v41_swa_precision import quantization_metrics, trace_attention + + +def test_attention_trace_uses_selected_rank_weights_and_cache(monkeypatch): + import sys + from types import ModuleType + + calls = [] + linear, rope = object(), object() + reference = SimpleNamespace(official_linear=linear, official_rope=rope, + official_reference=lambda inputs: calls.append(inputs)) + package = ModuleType("models.deepseek_v4_1_flash") + package.decode_attn_swa = reference + monkeypatch.setitem(sys.modules, "models", ModuleType("models")) + monkeypatch.setitem(sys.modules, "models.deepseek_v4_1_flash", package) + tensors = {name: torch.arange(4).reshape(4, 1) for name in ("weight", "cache")} + actual, expected = torch.ones(1), torch.zeros(1) + assert trace_attention(SimpleNamespace(HC_INPUT_NAMES=("weight", "cache", "x_hc")), + tensors, actual, expected, rank=2) == [{}, {}] + assert [int(c["weight"]) for c in calls] == [2, 2] + assert [int(c["cache"]) for c in calls] == [2, 2] + assert calls[0]["x"] is actual and calls[1]["x"] is expected + assert reference.official_linear is linear and reference.official_rope is rope @pytest.mark.parametrize("failure", [None, "residual", "pre_mix", "stage"]) diff --git a/tools/diagnose_v41_swa_precision.py b/tools/diagnose_v41_swa_precision.py index 7876529c..bfec766a 100644 --- a/tools/diagnose_v41_swa_precision.py +++ b/tools/diagnose_v41_swa_precision.py @@ -44,7 +44,7 @@ def decode(payload, codes): "scale_changed": int((actual_codes != expected_codes).sum())} -def trace_attention(swa, tensors, actual_hidden, reference_hidden): +def trace_attention(swa, tensors, actual_hidden, reference_hidden, *, rank=0): """Bisect the second attention on CPU, recording its public reference calls.""" from models.deepseek_v4_1_flash import decode_attn_swa as ref @@ -74,7 +74,7 @@ def rope(x, cos, sin, inverse=False): return result ref.official_linear, ref.official_rope = linear, rope - inputs = {name: tensors[name][0] for name in swa.HC_INPUT_NAMES if name not in ( + inputs = {name: tensors[name][rank] for name in swa.HC_INPUT_NAMES if name not in ( "x_hc", "incoming_pre_mix", "hc_attn_fn", "hc_attn_scale", "hc_attn_base", "attn_norm_weight", )} inputs["x"] = hidden @@ -109,6 +109,8 @@ def main(): parser.add_argument("--reference-kernel-norm", action="store_true", help="Use the standalone RMSNorm reference's chunk order in attention") parser.add_argument("--trace-layer", type=int, choices=(0, 1), default=1) + parser.add_argument("--trace-rank", type=int, default=0, + help="Logical rank for the per-operation Attention trace (including other DP groups)") parser.add_argument("--residual-profile", choices=("dsv4-layer", "v41-local"), default="dsv4-layer") parser.add_argument("--cut-after", type=int, choices=(0, 1, 2), help="Restart the CPU reference from a saved device boundary; diagnostic only") @@ -121,6 +123,8 @@ def main(): torch.set_num_threads(4) topology = SegmentTopology(tp=2, dp=2) + if not 0 <= args.trace_rank < topology.world: + parser.error("--trace-rank must be inside the diagnostic world") swa, moe = load_segment_modules(args.lib_root, topology) if args.reference_kernel_norm: swa.golden_rms_norm = moe.golden_rms_norm @@ -155,12 +159,14 @@ def wide_linear(x, weight, packed_scale, fp32=False): if args.trace_attention: full = torch.load(args.output, map_location="cpu", weights_only=True) layer = args.trace_layer - print(f"Loading layer {layer} for attention trace", flush=True) + rank = args.trace_rank + print(f"Loading layer {layer}, rank {rank} for attention trace", flush=True) aw, unused_moe = load_swa_layer_weights(args.model_dir, layer, topology) del unused_moe - traces = trace_attention(swa, dict(a, **aw), saved["stages"][2 * layer]["actual"]["hidden"][0], - full[2 * layer]["expected"]["hidden"][0]) - torch.save(traces, str(args.output) + f".attention-trace-layer{layer}.pt") + traces = trace_attention(swa, dict(a, **aw), saved["stages"][2 * layer]["actual"]["hidden"][rank], + full[2 * layer]["expected"]["hidden"][rank], rank=rank) + suffix = f"-rank{rank}" if rank else "" + torch.save(traces, str(args.output) + f".attention-trace-layer{layer}{suffix}.pt") return residual, mix = a["x_hc"], a["incoming_pre_mix"] records = [] From 9f9a5c3aa8f7ce819edc435328869f20e5b7d2ae Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Tue, 29 Sep 2026 03:16:43 +0800 Subject: [PATCH 47/78] refactor(v41): load attention probes without expert payloads --- .../model/deepseek_v41/swa_weights.py | 19 ++++++++++++++++--- .../model/deepseek_v41/test_weight_loader.py | 19 +++++++++++++++++++ tools/diagnose_v41_swa_precision.py | 5 ++--- 3 files changed, 37 insertions(+), 6 deletions(-) diff --git a/pypto_serving/model/deepseek_v41/swa_weights.py b/pypto_serving/model/deepseek_v41/swa_weights.py index 5c2adb8e..99c97f52 100644 --- a/pypto_serving/model/deepseek_v41/swa_weights.py +++ b/pypto_serving/model/deepseek_v41/swa_weights.py @@ -73,7 +73,18 @@ def load_prefill_layer_weights(model_dir, layer_id, topology, *, max_bundle_byte return _load_layer_weights(model_dir, layer_id, topology, max_bundle_bytes, swa_only=False) -def _load_layer_weights(model_dir, layer_id, topology, max_bundle_bytes, *, swa_only): +def load_prefill_attention_weights(model_dir, layer_id, topology, *, max_bundle_bytes=4 << 30): + """Read only Attention weights for isolated tracing or bounded preparation. + + Uses the identical packing/placement as the complete layer bundle. In + particular, an Attention probe must not read all routed expert payloads. + """ + attention, _ = _load_layer_weights(model_dir, layer_id, topology, max_bundle_bytes, + swa_only=False, include_moe=False) + return attention + + +def _load_layer_weights(model_dir, layer_id, topology, max_bundle_bytes, *, swa_only, include_moe=True): if type(max_bundle_bytes) is not int or max_bundle_bytes <= 0: raise ValueError("max_bundle_bytes must be positive") attention, moe = [], [] @@ -124,6 +135,9 @@ def load(name): bundle = load("attn.indexer.wq_b.weight") a["index_wq_b"], a["index_wq_b_scale"] = bundle.weight, bundle.scale a["index_weights_proj"] = load("attn.indexer.weights_proj.weight").weight + attention.append(a) + if not include_moe: + continue for target, source in { "hc_ffn_fn": "hc_ffn_fn", "hc_ffn_scale": "hc_ffn_scale", "hc_ffn_base": "hc_ffn_base", "norm_weight": "ffn_norm.weight", "gate_weight": "ffn.gate.weight", @@ -140,7 +154,6 @@ def load(name): m["routed_" + name + "_scale"] = merge_expert_scales([b.scale for b in bundles]) del bundles m["mxfp4_pair_lut"] = mxfp4_pair_lut() - attention.append(a) moe.append(m) return ({name: stack_bytes([r[name] for r in attention]) for name in attention[0]}, - {name: stack_bytes([r[name] for r in moe]) for name in moe[0]}) + {name: stack_bytes([r[name] for r in moe]) for name in moe[0]} if moe else {}) diff --git a/tests/unit/model/deepseek_v41/test_weight_loader.py b/tests/unit/model/deepseek_v41/test_weight_loader.py index b194b5a3..dc42ee91 100644 --- a/tests/unit/model/deepseek_v41/test_weight_loader.py +++ b/tests/unit/model/deepseek_v41/test_weight_loader.py @@ -159,6 +159,25 @@ def test_wo_a_group_dequantization(checkpoint): assert result.scale is None +def test_attention_only_bundle_does_not_open_expert_payloads(checkpoint): + from pypto_serving.model.deepseek_v41.swa_segment import SegmentTopology + from pypto_serving.model.deepseek_v41.swa_weights import load_prefill_attention_weights + + path, _, tensors = checkpoint + index_path = path / "model.safetensors.index.json" + index = json.loads(index_path.read_text()) + for name in index["weight_map"]: + if ".ffn." in name or "hc_ffn" in name or ".ffn_norm." in name: + index["weight_map"][name] = "unavailable-experts.safetensors" + index_path.write_text(json.dumps(index)) + result = load_prefill_attention_weights(path, 0, SegmentTopology(tp=2, dp=1)) + assert result["wq_b"].shape == (2, 256, 256) + assert result["index_wq_b"].shape == (2, 256, 256) + assert torch.equal(result["hc_attn_fn"][0], tensors["layers.0.hc_attn_fn"]) + with pytest.raises(ValueError, match="budget"): + load_prefill_attention_weights(path, 0, SegmentTopology(tp=2, dp=1), max_bundle_bytes=1) + + def test_dense_promotions_and_transpose(checkpoint): path, _, tensors = checkpoint loader = V41WeightLoader(path) diff --git a/tools/diagnose_v41_swa_precision.py b/tools/diagnose_v41_swa_precision.py index bfec766a..cc0305a2 100644 --- a/tools/diagnose_v41_swa_precision.py +++ b/tools/diagnose_v41_swa_precision.py @@ -119,7 +119,7 @@ def main(): import torch from golden.spec import TensorSpec from pypto_serving.model.deepseek_v41.swa_segment import SegmentTopology, load_segment_modules - from pypto_serving.model.deepseek_v41.swa_weights import load_swa_layer_weights + from pypto_serving.model.deepseek_v41.swa_weights import load_prefill_attention_weights, load_swa_layer_weights torch.set_num_threads(4) topology = SegmentTopology(tp=2, dp=2) @@ -161,8 +161,7 @@ def wide_linear(x, weight, packed_scale, fp32=False): layer = args.trace_layer rank = args.trace_rank print(f"Loading layer {layer}, rank {rank} for attention trace", flush=True) - aw, unused_moe = load_swa_layer_weights(args.model_dir, layer, topology) - del unused_moe + aw = load_prefill_attention_weights(args.model_dir, layer, topology) traces = trace_attention(swa, dict(a, **aw), saved["stages"][2 * layer]["actual"]["hidden"][rank], full[2 * layer]["expected"]["hidden"][rank], rank=rank) suffix = f"-rank{rank}" if rank else "" From 141c01acefde9ebc987dad4d152273ef73b723d3 Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Tue, 29 Sep 2026 03:25:35 +0800 Subject: [PATCH 48/78] Fix: retain V4.1 cache ownership across paused requests Validate full-history page ownership against every live request before opening a forward transaction. Reject relocation of committed history and preserve atomic slot allocation on invalid batches. --- .../model/deepseek_v41/request_state.py | 31 +++++++++++++ .../model/deepseek_v41/test_request_state.py | 46 ++++++++++++++++++- 2 files changed, 75 insertions(+), 2 deletions(-) diff --git a/pypto_serving/model/deepseek_v41/request_state.py b/pypto_serving/model/deepseek_v41/request_state.py index 0b9cfe36..39b1b420 100644 --- a/pypto_serving/model/deepseek_v41/request_state.py +++ b/pypto_serving/model/deepseek_v41/request_state.py @@ -76,6 +76,35 @@ def __init__(self, *, max_requests, max_seq_len, partitions=2): self.epoch = 1 self.poisoned = False + def _validate_page_ownership(self, slices): + """Retain full-history pages even when their owner is absent this step. + + There is no cache relocation/copy contract in this adapter. Continuations + may append pages, but cannot replace or drop their committed prefix. + Validate before reserving slots or exposing an in-flight transaction. + """ + occupied = {} + for key, owner in self.owners.items(): + for group, pages in owner.pages.items(): + for page in pages: + occupied[owner.partition, group, page] = key + for request in slices: + owner = self.owners.get(request.request_id) + if owner is not None: + for group, previous in owner.pages.items(): + current = request.pages.get(group, ()) + if tuple(current[:len(previous)]) != tuple(previous): + raise ValueError("continuation must preserve committed full-history pages") + for group, pages in request.pages.items(): + if len(set(pages)) != len(pages): + raise ValueError("request must not alias its own cache pages") + for page in pages: + address = (request.partition, group, page) + key = occupied.get(address) + if key is not None and key != request.request_id: + raise ValueError("cache page belongs to another live request") + occupied[address] = request.request_id + def begin_prefill(self, requests): """Validate the complete batch before taking ownership of any new slot. @@ -114,6 +143,7 @@ def begin_prefill(self, requests): immutable_pages = MappingProxyType({name: tuple(ids) for name, ids in pages.items()}) slices.append(RequestSlice(key, partition, owner.slot, start, tuple(tokens), prompt_length, immutable_pages)) + self._validate_page_ownership(slices) self.free = available self.owners.update(additions) self.pending = ForwardStep("prefill", tuple(slices), self.epoch) @@ -137,6 +167,7 @@ def begin_decode(self, requests): raise ValueError("decode exceeds sequence capacity") slices.append(RequestSlice(key, partition, owner.slot, start, (token,), owner.prompt_length, MappingProxyType({name: tuple(ids) for name, ids in pages.items()}))) + self._validate_page_ownership(slices) self.pending = ForwardStep("decode", tuple(slices), self.epoch) return self.pending diff --git a/tests/unit/model/deepseek_v41/test_request_state.py b/tests/unit/model/deepseek_v41/test_request_state.py index c258dbb1..7fb32620 100644 --- a/tests/unit/model/deepseek_v41/test_request_state.py +++ b/tests/unit/model/deepseek_v41/test_request_state.py @@ -17,10 +17,10 @@ def item(key="a", partition=0, start=0, tokens=(4, 5), length=4, pages=(3,)): def test_slots_survive_pause_and_batch_reorder(): ledger = RequestLedger(max_requests=2, max_seq_len=128) - first = ledger.begin_prefill([item("a"), item("b")]) + first = ledger.begin_prefill([item("a"), item("b", pages=(4,))]) slots = {r.request_id: r.state_slot for r in first.requests} ledger.commit(first) - second = ledger.begin_prefill([item("b", start=2)]) + second = ledger.begin_prefill([item("b", start=2, pages=(4,))]) assert second.requests[0].state_slot == slots["b"] assert ledger.owners["a"].slot == slots["a"] assert second.positions == (2, 3) @@ -93,3 +93,45 @@ def test_decode_rejects_partial_prefill_and_unknown_requests(): for key in ("a", "unknown"): with pytest.raises(ValueError, match="completed prefill"): ledger.begin_decode([(key, 0, 2, 7, {"window": (3,)})]) + + +def test_paused_request_keeps_pages_until_successful_release(): + ledger = RequestLedger(max_requests=2, max_seq_len=128) + ledger.commit(ledger.begin_prefill([item("a")])) + free = [list(slots) for slots in ledger.free] + with pytest.raises(ValueError, match="another live request"): + ledger.begin_prefill([item("b")]) + assert ledger.pending is None and ledger.free == free + assert set(ledger.owners) == {"a"} + ledger.release(["a"], lambda *_: None) + ledger.commit(ledger.begin_prefill([item("b")])) + assert ledger.owners["b"].pages["window"] == (3,) + + +@pytest.mark.parametrize("phase", ["prefill", "decode"]) +@pytest.mark.parametrize("pages", [(9,), (), (9, 3)]) +def test_continuation_cannot_relocate_or_drop_committed_pages(phase, pages): + ledger = RequestLedger(max_requests=2, max_seq_len=128) + ledger.commit(ledger.begin_prefill([item(length=4 if phase == "prefill" else 2)])) + with pytest.raises(ValueError, match="preserve committed"): + if phase == "prefill": + ledger.begin_prefill([item(start=2, pages=pages)]) + else: + ledger.begin_decode([("a", 0, 2, 7, {"window": pages})]) + assert ledger.pending is None and ledger.owners["a"].length == 2 + assert ledger.owners["a"].pages["window"] == (3,) + + +def test_decode_growth_cannot_steal_an_omitted_requests_page(): + ledger = RequestLedger(max_requests=2, max_seq_len=128) + ledger.commit(ledger.begin_prefill([item("a", length=2), item("b", length=2, pages=(9,))])) + with pytest.raises(ValueError, match="another live request"): + ledger.begin_decode([("a", 0, 2, 7, {"window": (3, 9)})]) + assert ledger.pending is None and ledger.owners["a"].length == 2 + ledger.commit(ledger.begin_decode([("a", 0, 2, 7, {"window": (3, 10)})])) + + +def test_same_page_numbers_are_independent_across_dp_partitions(): + ledger = RequestLedger(max_requests=1, max_seq_len=128) + ledger.commit(ledger.begin_prefill([item("a"), item("b", partition=1)])) + assert set(ledger.owners) == {"a", "b"} From 85ee35e54795f288cfec57c14588299782f41f11 Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Tue, 29 Sep 2026 03:28:44 +0800 Subject: [PATCH 49/78] Docs: record real-text precision candidate and page ownership limits --- docs/developer-guide/v41-swa-segment.md | 18 ++++++++++++++++++ 1 file changed, 18 insertions(+) diff --git a/docs/developer-guide/v41-swa-segment.md b/docs/developer-guide/v41-swa-segment.md index 740aa1ca..1307a77b 100644 --- a/docs/developer-guide/v41-swa-segment.md +++ b/docs/developer-guide/v41-swa-segment.md @@ -242,3 +242,21 @@ same-input stage comparators after worker shutdown. A native-stage pass does not resolve the incoming SWA accumulated error or establish a new accumulated four-layer acceptance standard. Source state and numerical artifacts must be reported with that limitation. + +The minimal Q-B group-32 diagnostic candidate (`cd759ce7`, based on official +lib `fbe92bfc`) was also tested with the same real-text inputs and explicit +request metadata, using serving `6f52516`. All native checks and all per-rank +V4 residual checks passed, but accumulated pre_mix still failed: 3/256 elements, +relative L2 0.0025390928 and maximum absolute error 0.0076903105. Residual +relative L2 was 0.012873608 (maximum absolute 0.02734375). The baseline has +7/256 pre_mix failures on this workload. This is an improvement, not an accepted +precision fix; the candidate remains on a diagnostic branch and is not a +production dependency. Neither reference arithmetic nor gates were changed. + +The request ledger retains committed full-history pages for omitted/paused +requests until their reset succeeds. A later batch cannot borrow those pages, +and a continuation may only append to its committed page table. Relocation or +dropping history is rejected because no device cache-copy contract is integrated. +Page IDs remain independent across DP partitions and cache groups. These host +ownership checks complement per-batch metadata validation; device reset/reuse +still requires its own validation. From fabf01120c11bc49ca82c9733e53e3897cdb466f Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Tue, 29 Sep 2026 03:33:52 +0800 Subject: [PATCH 50/78] Add: validate ragged C2A prefixes and empty DP groups Keep the saved SWA state immutable while selecting causal request prefixes. Pass matching global and local active counts through metadata, device dispatch and native references, including odd compressor tails and empty partitions. --- .../model/deepseek_v41/test_c2a_diagnostic.py | 33 +++++++++++++ tools/validate_v41_c2a_chain.py | 47 +++++++++++++------ 2 files changed, 66 insertions(+), 14 deletions(-) create mode 100644 tests/unit/model/deepseek_v41/test_c2a_diagnostic.py diff --git a/tests/unit/model/deepseek_v41/test_c2a_diagnostic.py b/tests/unit/model/deepseek_v41/test_c2a_diagnostic.py new file mode 100644 index 00000000..8673db4f --- /dev/null +++ b/tests/unit/model/deepseek_v41/test_c2a_diagnostic.py @@ -0,0 +1,33 @@ +# Copyright (c) PyPTO Contributors. +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of +# CANN Open Software License Agreement Version 2.0 (the "License"). +# Please refer to the License for details. You may not use this file except in compliance with the License. +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. +# See LICENSE in the root of the software repository for the full text of the License. +# ----------------------------------------------------------------------------------------------------------- +"""Ragged diagnostics retain real causal prefixes and explicit inactive padding.""" +import pytest +import torch + +from pypto_serving.model.deepseek_v41.swa_segment import SegmentTopology +from tools.validate_v41_c2a_chain import select_active_state + + +def test_select_ragged_prefix_without_mutating_saved_state(): + topology = SegmentTopology(tp=2, dp=2) + residual = torch.arange(4 * 16 * 4 * 5120, dtype=torch.float32).reshape(4, 16, 4, 5120) + mix = torch.ones(4, 16, 4) + saved = dict(actual_residual=residual, actual_pre_mix=mix) + selected, selected_mix = select_active_state(saved, topology, [31, 0]) + assert torch.equal(selected[0], residual[0]) + assert torch.equal(selected[1, :15], residual[1, :15]) + assert not selected[1, 15:].count_nonzero() and not selected[2:].count_nonzero() + assert not selected_mix[1, 15:].count_nonzero() and not selected_mix[2:].count_nonzero() + assert residual[2:].count_nonzero() and mix[2:].eq(1).all() + + +@pytest.mark.parametrize("counts", [[33, 0], [1], [-1, 32]]) +def test_select_prefix_rejects_invalid_counts_before_reading_state(counts): + with pytest.raises(ValueError, match="active-token count"): + select_active_state({}, SegmentTopology(tp=2, dp=2), counts) diff --git a/tools/validate_v41_c2a_chain.py b/tools/validate_v41_c2a_chain.py index 9f94a887..0a96d399 100644 --- a/tools/validate_v41_c2a_chain.py +++ b/tools/validate_v41_c2a_chain.py @@ -18,6 +18,25 @@ from types import SimpleNamespace +def select_active_state(saved, topology, group_counts): + """Select causal prefixes for a ragged diagnostic, preserving the source artifact.""" + import torch + + topology.counts(group_counts) + residual, mix = saved["actual_residual"], saved["actual_pre_mix"] + if residual.shape != (topology.world, topology.local_capacity, 4, 5120) or mix.shape != residual.shape[:-1]: + raise ValueError("saved SWA state has incompatible topology or shape") + if residual.dtype != torch.float32 or mix.dtype != torch.float32: + raise ValueError("saved SWA state must preserve its FP32 storage") + if not torch.isfinite(residual).all() or not torch.isfinite(mix).all(): + raise ValueError("saved SWA state must be finite") + residual, mix = residual.clone(), mix.clone() + for rank, count in enumerate(topology.counts(group_counts)[1]): + residual[rank, count:] = 0 + mix[rank, count:] = 0 + return residual, mix + + def prepare(args, topology, module): import torch from golden.spec import TensorSpec @@ -35,13 +54,9 @@ def prepare(args, topology, module): ids = saved["token_ids"] if ids is None or tuple(ids.shape) != (topology.dp, topology.capacity): raise ValueError("saved SWA diagnostic must contain exactly the same packed token capacity") - residual, mix = saved["actual_residual"], saved["actual_pre_mix"] - if residual.shape != (topology.world, topology.local_capacity, 4, 5120) or mix.shape != residual.shape[:-1]: - raise ValueError("saved SWA state has incompatible topology or shape") - if residual.dtype != torch.float32 or mix.dtype != torch.float32: - raise ValueError("saved SWA state must preserve its FP32 storage") - if not torch.isfinite(residual).all() or not torch.isfinite(mix).all(): - raise ValueError("saved SWA state must be finite") + group_counts = args.group_counts + global_counts, local_counts = topology.counts(group_counts) + residual, mix = select_active_state(saved, topology, group_counts) raw = json.loads((Path(args.model_dir) / "config.json").read_text()) plans = plan_layers(raw)[2:4] if tuple(p.mode for p in plans) != ("c2a_full", "c2a_reuse"): @@ -57,9 +72,9 @@ def prepare(args, topology, module): raise ValueError(f"lib/checkpoint compressed RoPE mismatch: {source}") # Fresh first-chunk history at layer 2, private pages per DP group. pages = (topology.capacity + 127) // 128 - step = ForwardStep("prefill", tuple(RequestSlice(str(g), g, 0, 0, tuple(row.tolist()), + step = ForwardStep("prefill", tuple(RequestSlice(str(g), g, 0, 0, tuple(row[:group_counts[g]].tolist()), topology.capacity, {"window": tuple(range(pages)), "cmp": tuple(range(pages))}) - for g, row in enumerate(ids)), 1) + for g, row in enumerate(ids) if group_counts[g]), 1) tables = precompute_rope_tables(topology.capacity, False) compressed_tables = precompute_rope_tables(topology.capacity, True) fixture = SimpleNamespace(tokens=topology.capacity, requests=1, dp=topology.dp, @@ -72,6 +87,7 @@ def prepare(args, topology, module): print(f"Loading real checkpoint layer {plan.layer_id}", flush=True) aw, mw = load_prefill_layer_weights(args.model_dir, plan.layer_id, topology) values.update(aw, x_hc=residual, pre_mix=mix) + values["num_tokens"] = torch.tensor(global_counts, dtype=torch.int32).reshape(-1, 1) window = prepare_swa_window_metadata(step, topology, cache_pages=values["window_cache"].shape[1]) values.update(window_slots=window.window_slots, window_indices=window.window_indices) values["window_cache"].view(torch.uint8).zero_() @@ -99,8 +115,7 @@ def prepare(args, topology, module): attention[plan.layer_id] = values moe[plan.layer_id] = dict(mw, next_pre_mix=torch.zeros_like(mix), x_mixed=torch.zeros(topology.world, topology.local_capacity, 5120, dtype=torch.bfloat16), - x_next=torch.zeros_like(residual), num_tokens=torch.full((topology.world,), - topology.local_capacity, dtype=torch.int32)) + x_next=torch.zeros_like(residual), num_tokens=torch.tensor(local_counts, dtype=torch.int32)) return plans, attention, moe, residual, mix @@ -111,6 +126,7 @@ def main(): parser.add_argument("--input-state", required=True) parser.add_argument("--devices", default="0,1,2,3") parser.add_argument("--tp", type=int, default=2) + parser.add_argument("--group-counts", help="Comma-separated causal prefix lengths per DP group") parser.add_argument("--prepare-only", action="store_true") parser.add_argument("--build-dir", default="build_output/v41-c2a-chain") parser.add_argument("--artifact-dir", default=".validation-artifacts/c2a-chain") @@ -125,6 +141,9 @@ def main(): if args.tp <= 0 or len(set(devices)) != len(devices) or len(devices) % args.tp: raise ValueError("unique devices must form complete TP groups") topology = SegmentTopology(tp=args.tp, dp=len(devices) // args.tp) + args.group_counts = ([topology.capacity] * topology.dp if args.group_counts is None + else [int(n) for n in args.group_counts.split(",")]) + topology.counts(args.group_counts) torch.set_num_threads(4) _, moe_module = load_segment_modules(args.lib_root, topology) from models.deepseek_v4_1_flash import prefill_c2a_full as c2a @@ -179,7 +198,7 @@ def upload(values): dm = {layer: upload(values) for layer, values in moe.items()} runner = PrefillSegment(worker, programs, ffn, topology, ac, mc, config) runner.run_chain(LayerState(da[2]["x_hc"], da[2]["pre_mix"], "tp_local_token"), - plans, da, dm, group_counts=[topology.capacity] * topology.dp) + plans, da, dm, group_counts=args.group_counts) for layer, (ca, cm) in captured.items(): for device, host in ((da[layer], ca), (dm[layer], cm)): for name, destination in host.items(): @@ -209,7 +228,7 @@ def upload(values): mc_check = { "next_pre_mix": ratio_allclose(atol=2.5e-5, rtol=5e-3), "x_mixed": ratio_allclose(atol=1e-4, rtol=1.0 / 128), - "x_next": moe_module._local_mhc_compare([topology.local_capacity] * topology.world), + "x_next": moe_module._local_mhc_compare(list(topology.counts(args.group_counts)[1])), } for label, actual, expected, checks in (("attention", actual_a, expected_a, c2a.make_compare(mode, 1, initial_cache)), ("moe", actual_m, expected_m, mc_check)): @@ -223,7 +242,7 @@ def upload(values): records.append(dict(layer=layer, stage=label, results=results, actual=actual, expected={n: expected[n] for n in actual})) residual, mix = actual_m["x_next"], actual_m["next_pre_mix"] - torch.save({"input_state": str(args.input_state), "stages": records, + torch.save({"input_state": str(args.input_state), "group_counts": args.group_counts, "stages": records, "actual_residual": residual, "actual_pre_mix": mix}, artifact / "comparison.pt") if not passed: raise AssertionError("C2A chain native stage check failed; see comparison.pt") From 0355d4089607162f90f6f76a5fcf738844b92c85 Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Tue, 29 Sep 2026 03:56:50 +0800 Subject: [PATCH 51/78] Add: validate resident C2A compressor continuation across chunks Prepare causal source slices and metadata for two consecutive request chunks while retaining device cache and compressor handles. Compare each native stage after device execution and keep captured outputs out of the forward input path. --- docs/developer-guide/v41-swa-segment.md | 15 ++ .../model/deepseek_v41/test_c2a_diagnostic.py | 18 ++ tools/validate_v41_c2a_chain.py | 177 +++++++++++------- 3 files changed, 141 insertions(+), 69 deletions(-) diff --git a/docs/developer-guide/v41-swa-segment.md b/docs/developer-guide/v41-swa-segment.md index 1307a77b..008b34af 100644 --- a/docs/developer-guide/v41-swa-segment.md +++ b/docs/developer-guide/v41-swa-segment.md @@ -243,6 +243,21 @@ not resolve the incoming SWA accumulated error or establish a new accumulated four-layer acceptance standard. Source state and numerical artifacts must be reported with that limitation. +The fresh real-weight C2A pair passed all 24 native checks on A5 TP2/DP2/EP4 +at serving `9f9a5c3` and official lib `fbe92bfc`. At serving `fabf011`, the +`--group-counts 31,0` case also passed, covering an odd compressor tail and an +empty DP group with untouched caches. Both device tasks exited zero and released +all four cards. These remain stage/state checks with an unresolved incoming +SWA accumulated error. + +`--continue-to-capacity` adds a second chunk for each initially nonempty request, +using the remaining rows of the saved causal source. For example, +`--group-counts 31,0 --continue-to-capacity` runs 31 then 1 token in the first DP +group, with the second group empty. Both chunks use the same resident per-layer +caches, compressor state, weights and communication windows. Step snapshots are +read-only diagnostics; references execute after worker shutdown. This checks +C2A continuation, not SWA chunk equivalence, full-model accuracy or request reset. + The minimal Q-B group-32 diagnostic candidate (`cd759ce7`, based on official lib `fbe92bfc`) was also tested with the same real-text inputs and explicit request metadata, using serving `6f52516`. All native checks and all per-rank diff --git a/tests/unit/model/deepseek_v41/test_c2a_diagnostic.py b/tests/unit/model/deepseek_v41/test_c2a_diagnostic.py index 8673db4f..544ee7db 100644 --- a/tests/unit/model/deepseek_v41/test_c2a_diagnostic.py +++ b/tests/unit/model/deepseek_v41/test_c2a_diagnostic.py @@ -31,3 +31,21 @@ def test_select_ragged_prefix_without_mutating_saved_state(): def test_select_prefix_rejects_invalid_counts_before_reading_state(counts): with pytest.raises(ValueError, match="active-token count"): select_active_state({}, SegmentTopology(tp=2, dp=2), counts) + + +def test_continuation_repacks_the_next_token_across_tp_slabs(): + topology = SegmentTopology(tp=2, dp=2) + residual = torch.arange(4 * 16 * 4 * 5120, dtype=torch.float32).reshape(4, 16, 4, 5120) + mix = torch.arange(4 * 16 * 4, dtype=torch.float32).reshape(4, 16, 4) + selected, selected_mix = select_active_state( + dict(actual_residual=residual, actual_pre_mix=mix), topology, [1, 0], [31, 0]) + assert torch.equal(selected[0, 0], residual[1, 15]) + assert torch.equal(selected_mix[0, 0], mix[1, 15]) + assert not selected[0, 1:].count_nonzero() and not selected[1:].count_nonzero() + assert not selected_mix[0, 1:].count_nonzero() and not selected_mix[1:].count_nonzero() + + +@pytest.mark.parametrize("starts", [[32, 0], [-1, 0], [0], [True, 0]]) +def test_continuation_cannot_exceed_saved_source(starts): + with pytest.raises(ValueError, match="saved causal source"): + select_active_state({}, SegmentTopology(tp=2, dp=2), [1, 0], starts) diff --git a/tools/validate_v41_c2a_chain.py b/tools/validate_v41_c2a_chain.py index 0a96d399..22b37943 100644 --- a/tools/validate_v41_c2a_chain.py +++ b/tools/validate_v41_c2a_chain.py @@ -18,11 +18,15 @@ from types import SimpleNamespace -def select_active_state(saved, topology, group_counts): +def select_active_state(saved, topology, group_counts, starts=None): """Select causal prefixes for a ragged diagnostic, preserving the source artifact.""" import torch topology.counts(group_counts) + starts = [0] * topology.dp if starts is None else starts + if len(starts) != topology.dp or any(type(start) is not int or start < 0 or + start + count > topology.capacity for start, count in zip(starts, group_counts)): + raise ValueError("diagnostic chunks must fit within the saved causal source") residual, mix = saved["actual_residual"], saved["actual_pre_mix"] if residual.shape != (topology.world, topology.local_capacity, 4, 5120) or mix.shape != residual.shape[:-1]: raise ValueError("saved SWA state has incompatible topology or shape") @@ -30,14 +34,15 @@ def select_active_state(saved, topology, group_counts): raise ValueError("saved SWA state must preserve its FP32 storage") if not torch.isfinite(residual).all() or not torch.isfinite(mix).all(): raise ValueError("saved SWA state must be finite") - residual, mix = residual.clone(), mix.clone() - for rank, count in enumerate(topology.counts(group_counts)[1]): - residual[rank, count:] = 0 - mix[rank, count:] = 0 - return residual, mix + selected, selected_mix = torch.zeros_like(residual), torch.zeros_like(mix) + for group, (start, count) in enumerate(zip(starts, group_counts)): + ranks = slice(group * topology.tp, (group + 1) * topology.tp) + selected[ranks].flatten(0, 1)[:count] = residual[ranks].flatten(0, 1)[start:start + count] + selected_mix[ranks].flatten(0, 1)[:count] = mix[ranks].flatten(0, 1)[start:start + count] + return selected, selected_mix -def prepare(args, topology, module): +def prepare(args, topology, module, weight_bundles=None): import torch from golden.spec import TensorSpec from models.deepseek_v4_1_flash.config import FLASH @@ -55,8 +60,10 @@ def prepare(args, topology, module): if ids is None or tuple(ids.shape) != (topology.dp, topology.capacity): raise ValueError("saved SWA diagnostic must contain exactly the same packed token capacity") group_counts = args.group_counts + starts = getattr(args, "starts", [0] * topology.dp) + weight_bundles = {} if weight_bundles is None else weight_bundles global_counts, local_counts = topology.counts(group_counts) - residual, mix = select_active_state(saved, topology, group_counts) + residual, mix = select_active_state(saved, topology, group_counts, starts) raw = json.loads((Path(args.model_dir) / "config.json").read_text()) plans = plan_layers(raw)[2:4] if tuple(p.mode for p in plans) != ("c2a_full", "c2a_reuse"): @@ -72,7 +79,8 @@ def prepare(args, topology, module): raise ValueError(f"lib/checkpoint compressed RoPE mismatch: {source}") # Fresh first-chunk history at layer 2, private pages per DP group. pages = (topology.capacity + 127) // 128 - step = ForwardStep("prefill", tuple(RequestSlice(str(g), g, 0, 0, tuple(row[:group_counts[g]].tolist()), + step = ForwardStep("prefill", tuple(RequestSlice(str(g), g, 0, starts[g], + tuple(row[starts[g]:starts[g] + group_counts[g]].tolist()), topology.capacity, {"window": tuple(range(pages)), "cmp": tuple(range(pages))}) for g, row in enumerate(ids) if group_counts[g]), 1) tables = precompute_rope_tables(topology.capacity, False) @@ -84,8 +92,10 @@ def prepare(args, topology, module): mode = plan.mode.removeprefix("c2a_") values = {s.name: s.create_tensor().contiguous() for s in module.build_specs(fixture, mode, {}) if isinstance(s, TensorSpec)} - print(f"Loading real checkpoint layer {plan.layer_id}", flush=True) - aw, mw = load_prefill_layer_weights(args.model_dir, plan.layer_id, topology) + if plan.layer_id not in weight_bundles: + print(f"Loading real checkpoint layer {plan.layer_id}", flush=True) + weight_bundles[plan.layer_id] = load_prefill_layer_weights(args.model_dir, plan.layer_id, topology) + aw, mw = weight_bundles[plan.layer_id] values.update(aw, x_hc=residual, pre_mix=mix) values["num_tokens"] = torch.tensor(global_counts, dtype=torch.int32).reshape(-1, 1) window = prepare_swa_window_metadata(step, topology, cache_pages=values["window_cache"].shape[1]) @@ -127,6 +137,8 @@ def main(): parser.add_argument("--devices", default="0,1,2,3") parser.add_argument("--tp", type=int, default=2) parser.add_argument("--group-counts", help="Comma-separated causal prefix lengths per DP group") + parser.add_argument("--continue-to-capacity", action="store_true", + help="Run a second chunk of each nonempty request using its resident caches") parser.add_argument("--prepare-only", action="store_true") parser.add_argument("--build-dir", default="build_output/v41-c2a-chain") parser.add_argument("--artifact-dir", default=".validation-artifacts/c2a-chain") @@ -147,14 +159,29 @@ def main(): torch.set_num_threads(4) _, moe_module = load_segment_modules(args.lib_root, topology) from models.deepseek_v4_1_flash import prefill_c2a_full as c2a - plans, attention, moe, residual, mix = prepare(args, topology, c2a) - bound = bind_prefill_producers(plans, attention) + chunks = [(args.group_counts, [0] * topology.dp)] + if args.continue_to_capacity: + remaining = [topology.capacity - n if n else 0 for n in args.group_counts] + if not any(remaining): + raise ValueError("continuation requires an unfinished nonempty request") + chunks.append((remaining, list(args.group_counts))) + prepared, weights = [], {} + for counts, starts in chunks: + options = SimpleNamespace(**{**vars(args), "group_counts": counts, "starts": starts}) + plans, attention, moe, residual, mix = prepare(options, topology, c2a, weights) + if prepared: + for plan in plans: + for name in c2a.ATTENTION_STATE[plan.mode.removeprefix("c2a_")]: + if "cache" in name: + attention[plan.layer_id][name] = prepared[0][0][plan.layer_id][name] + prepared.append((bind_prefill_producers(plans, attention), moe, residual, mix, counts)) from pypto_serving.model.deepseek_v41.prefill_segment import PREFILL_ARGUMENTS dynamic = {"attention_epoch", "num_tokens"} - for plan in plans: - for name in PREFILL_ARGUMENTS[plan.mode]: - if name not in dynamic: - assert isinstance(bound[plan.layer_id][name], torch.Tensor), name + for bound, *_ in prepared: + for plan in plans: + for name in PREFILL_ARGUMENTS[plan.mode]: + if name not in dynamic: + assert isinstance(bound[plan.layer_id][name], torch.Tensor), name print("REAL CHECKPOINT C2A PREPARATION PASS", flush=True) if args.prepare_only: return @@ -174,15 +201,19 @@ def main(): programs, ffn = compile_prefill_segments(compiler, args.lib_root, topology, [p.mode for p in plans]) ac = torch.zeros(topology.world, 1, dtype=torch.int32).share_memory_() mc = torch.zeros(topology.world, dtype=torch.int32).share_memory_() - captured = {} - for plan in plans: - layer = plan.layer_id - names = ("attn_input", "attn_output", "next_pre_mix", "x_hc_out") + c2a.ATTENTION_STATE[ - plan.mode.removeprefix("c2a_")] - captured[layer] = ({n: torch.empty_like(bound[layer][n]).share_memory_() for n in names}, - {n: torch.empty_like(moe[layer][n]).share_memory_() - for n in ("x_next", "next_pre_mix", "x_mixed")}) - sources = [v for maps in (bound, moe) for values in maps.values() for v in values.values()] + captures = [] + for bound, moe, *_ in prepared: + captured = {} + for plan in plans: + layer = plan.layer_id + names = ("attn_input", "attn_output", "next_pre_mix", "x_hc_out") + c2a.ATTENTION_STATE[ + plan.mode.removeprefix("c2a_")] + captured[layer] = ({n: torch.empty_like(bound[layer][n]).share_memory_() for n in names}, + {n: torch.empty_like(moe[layer][n]).share_memory_() + for n in ("x_next", "next_pre_mix", "x_mixed")}) + captures.append(captured) + sources = [v for bound, moe, *_ in prepared for maps in (bound, moe) + for values in maps.values() for v in values.values()] with make_segment_worker([*programs.values(), ffn], config, sources) as worker: uploaded = {} def upload(values): @@ -194,55 +225,63 @@ def upload(values): uploaded[id(value)] = worker.alloc_stacked_tensor(value) result[name] = uploaded[id(value)] return result - da = {layer: upload(values) for layer, values in bound.items()} - dm = {layer: upload(values) for layer, values in moe.items()} + device_steps = [({layer: upload(values) for layer, values in bound.items()}, + {layer: upload(values) for layer, values in moe.items()}) + for bound, moe, *_ in prepared] runner = PrefillSegment(worker, programs, ffn, topology, ac, mc, config) - runner.run_chain(LayerState(da[2]["x_hc"], da[2]["pre_mix"], "tp_local_token"), - plans, da, dm, group_counts=args.group_counts) - for layer, (ca, cm) in captured.items(): - for device, host in ((da[layer], ca), (dm[layer], cm)): - for name, destination in host.items(): - worker.copy_stacked_from(device[name], destination) + for step_id, ((da, dm), captured) in enumerate(zip(device_steps, captures)): + runner.run_chain(LayerState(da[2]["x_hc"], da[2]["pre_mix"], "tp_local_token"), + plans, da, dm, group_counts=prepared[step_id][4]) + for layer, (ca, cm) in captured.items(): + for device, host in ((da[layer], ca), (dm[layer], cm)): + for name, destination in host.items(): + worker.copy_stacked_from(device[name], destination) + print(f"C2A DEVICE CHUNK {step_id} COMPLETE", flush=True) for handle in reversed(list(uploaded.values())): worker.free_stacked_tensor(handle) # CPU same-input references run only after worker shutdown. Device state # never depends on these diagnostic readbacks or reference results. records, passed = [], True - for plan in plans: - layer, mode = plan.layer_id, plan.mode.removeprefix("c2a_") - actual_a, actual_m = captured[layer] - inputs = dict(bound[layer], x_hc=residual, pre_mix=mix) - if mode == "reuse": - for name in ("compressed_cache", "compressed_cache_scale"): - inputs[name] = captured[2][0][name] - inputs["compressed_indices"] = captured[2][0]["topk_indices"] - initial_cache = {n: inputs[n].clone() for n in c2a.MODES[mode][1]} - expected_a = dict(inputs) - for name in actual_a: - expected_a[name] = inputs[name].clone() - c2a.make_golden(mode, 1)(expected_a) - expected_m = dict(moe[layer], x_hc=actual_a["x_hc_out"], pre_mix=actual_a["next_pre_mix"]) - for name in actual_m: - expected_m[name] = moe[layer][name].clone() - moe_module.golden_moe(expected_m) - mc_check = { - "next_pre_mix": ratio_allclose(atol=2.5e-5, rtol=5e-3), - "x_mixed": ratio_allclose(atol=1e-4, rtol=1.0 / 128), - "x_next": moe_module._local_mhc_compare(list(topology.counts(args.group_counts)[1])), - } - for label, actual, expected, checks in (("attention", actual_a, expected_a, - c2a.make_compare(mode, 1, initial_cache)), ("moe", actual_m, expected_m, mc_check)): - results = {} - for name, check in checks.items(): - ok, detail = check(actual[name], expected[name], inputs=expected, - actual_outputs=actual, expected_outputs=expected, rtol=1e-3, atol=1e-3) - print(f"NATIVE STAGE layer={layer} {label}.{name}: {ok} {detail}", flush=True) - results[name] = (bool(ok), detail) - passed &= bool(ok) - records.append(dict(layer=layer, stage=label, results=results, actual=actual, - expected={n: expected[n] for n in actual})) - residual, mix = actual_m["x_next"], actual_m["next_pre_mix"] - torch.save({"input_state": str(args.input_state), "group_counts": args.group_counts, "stages": records, + for step_id, ((bound, moe, residual, mix, counts), captured) in enumerate(zip(prepared, captures)): + for plan in plans: + layer, mode = plan.layer_id, plan.mode.removeprefix("c2a_") + actual_a, actual_m = captured[layer] + inputs = dict(bound[layer], x_hc=residual, pre_mix=mix) + if step_id: + for name, value in captures[step_id - 1][layer][0].items(): + if "cache" in name: + inputs[name] = value + if mode == "reuse": + for name in ("compressed_cache", "compressed_cache_scale"): + inputs[name] = captured[2][0][name] + inputs["compressed_indices"] = captured[2][0]["topk_indices"] + initial_cache = {n: inputs[n].clone() for n in c2a.MODES[mode][1]} + expected_a = dict(inputs) + for name in actual_a: + expected_a[name] = inputs[name].clone() + c2a.make_golden(mode, 1)(expected_a) + expected_m = dict(moe[layer], x_hc=actual_a["x_hc_out"], pre_mix=actual_a["next_pre_mix"]) + for name in actual_m: + expected_m[name] = moe[layer][name].clone() + moe_module.golden_moe(expected_m) + mc_check = { + "next_pre_mix": ratio_allclose(atol=2.5e-5, rtol=5e-3), + "x_mixed": ratio_allclose(atol=1e-4, rtol=1.0 / 128), + "x_next": moe_module._local_mhc_compare(list(topology.counts(counts)[1])), + } + for label, actual, expected, checks in (("attention", actual_a, expected_a, + c2a.make_compare(mode, 1, initial_cache)), ("moe", actual_m, expected_m, mc_check)): + results = {} + for name, check in checks.items(): + ok, detail = check(actual[name], expected[name], inputs=expected, + actual_outputs=actual, expected_outputs=expected, rtol=1e-3, atol=1e-3) + print(f"NATIVE STAGE chunk={step_id} layer={layer} {label}.{name}: {ok} {detail}", flush=True) + results[name] = (bool(ok), detail) + passed &= bool(ok) + records.append(dict(chunk=step_id, layer=layer, stage=label, results=results, actual=actual, + expected={n: expected[n] for n in actual})) + residual, mix = actual_m["x_next"], actual_m["next_pre_mix"] + torch.save({"input_state": str(args.input_state), "group_counts": args.group_counts, "chunks": chunks, "stages": records, "actual_residual": residual, "actual_pre_mix": mix}, artifact / "comparison.pt") if not passed: raise AssertionError("C2A chain native stage check failed; see comparison.pt") From f28588dbd22a086badfe62e3dd852d4b467cff7e Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Tue, 29 Sep 2026 04:10:04 +0800 Subject: [PATCH 52/78] docs: record unresolved projection precision experiment --- docs/developer-guide/v41-swa-segment.md | 9 +++++++++ 1 file changed, 9 insertions(+) diff --git a/docs/developer-guide/v41-swa-segment.md b/docs/developer-guide/v41-swa-segment.md index 008b34af..4abccb34 100644 --- a/docs/developer-guide/v41-swa-segment.md +++ b/docs/developer-guide/v41-swa-segment.md @@ -268,6 +268,15 @@ relative L2 was 0.012873608 (maximum absolute 0.02734375). The baseline has precision fix; the candidate remains on a diagnostic branch and is not a production dependency. Neither reference arithmetic nor gates were changed. +The Q-B plus O-B group-32 diagnostic (`0e166cd9`) also completed the same +real-text/request workload. Residual passed on every rank (relative L2 +0.012825638, maximum absolute error 0.02734375), but pre_mix still failed in +2/256 elements (relative L2 0.0022762596, maximum absolute error 0.0088607967). +The smaller failure count is not acceptance: its maximum absolute error is +higher than the Q-B-only candidate. It remains diagnostic, with no lib pin or +reference change. Evidence: `swa-realweights-qb-ob32-text/comparison.pt` and +`swa-precision-qb-ob32-text.log` under the validation artifact directory. + The request ledger retains committed full-history pages for omitted/paused requests until their reset succeeds. A later batch cannot borrow those pages, and a continuation may only append to its committed page table. Relocation or From 861c36df449d9d91eb1a9d21bdb1004b5eac6044 Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Tue, 29 Sep 2026 04:13:17 +0800 Subject: [PATCH 53/78] docs: record successful real-weight C2A continuation --- docs/developer-guide/v41-swa-segment.md | 10 ++++++++++ 1 file changed, 10 insertions(+) diff --git a/docs/developer-guide/v41-swa-segment.md b/docs/developer-guide/v41-swa-segment.md index 4abccb34..b7ea5ab2 100644 --- a/docs/developer-guide/v41-swa-segment.md +++ b/docs/developer-guide/v41-swa-segment.md @@ -258,6 +258,16 @@ caches, compressor state, weights and communication windows. Step snapshots are read-only diagnostics; references execute after worker shutdown. This checks C2A continuation, not SWA chunk equivalence, full-model accuracy or request reset. +That `31 + 1` continuation passed all 48 native stage/state checks on A5 +TP2/DP2/EP4 at serving `0355d40` and official lib `fbe92bfc`. Task +`task_20260929_040845_198102410502` exited zero and released cards 0-3. +Evidence: `c2a-chain-continuation-retry/comparison.pt` and +`c2a-chain-continuation-retry.log` under the validation artifact directory. +An earlier attempt stopped before device execution with an empty PTOAS error; +the identical generated O-B source compiled in isolation. The successful retry +explicitly limited `PYPTO_CODEGEN_MAX_WORKERS=4` as well as build/OMP workers. +No source, reference or acceptance threshold changed between these attempts. + The minimal Q-B group-32 diagnostic candidate (`cd759ce7`, based on official lib `fbe92bfc`) was also tested with the same real-text inputs and explicit request metadata, using serving `6f52516`. All native checks and all per-rank From 4e83854c1cd3ebf7c4aa38a1f500396904bb2931 Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Tue, 29 Sep 2026 09:43:27 +0800 Subject: [PATCH 54/78] docs: record isolated V4.1 TP communication mismatch --- docs/developer-guide/v41-swa-segment.md | 26 +++++++++++++++++++++++++ 1 file changed, 26 insertions(+) diff --git a/docs/developer-guide/v41-swa-segment.md b/docs/developer-guide/v41-swa-segment.md index b7ea5ab2..e25210ef 100644 --- a/docs/developer-guide/v41-swa-segment.md +++ b/docs/developer-guide/v41-swa-segment.md @@ -294,3 +294,29 @@ dropping history is rejected because no device cache-copy contract is integrated Page IDs remain independent across DP partitions and cache groups. These host ownership checks complement per-batch metadata validation; device reset/reuse still requires its own validation. + +### TP communication precision isolation + +At serving `0355d40`, diagnostic lib `be5609d5` captures the O-B partial, +publication input, actual peer reads and pre-cast reduction. The producer and +publication inputs are bitwise equal. Four peer-read values across two layers +differ near the last row's final columns. Addition and BF16 conversion match +the captured reads, but do not always match the values published by the peers. +For layer 0, DP group 1, row 31, column 5115, the published FP32 partials sum to +1.0050979852676392 (BF16 1.0078125); the device returns 1.0. Instrumented final +outputs are bitwise identical to the uninstrumented Q-B group-32 diagnostic. + +The fixed-partial, reduction-only replay (`ea9e9a7c`) passes exactly. Adding +the input AllGather and its communication windows (`9061b7e0`) reproduces that +output mismatch while the AllGather output remains bitwise exact. This replay +loads captured partials, not checkpoint weights. Disabling the output +publication pipeline in the full workload (`0e524332`) does not change the +failure. Evidence includes `tp-gather-replay-layer0.log`, +`tp-gather-replay-layer0.pt` and `swa-realweights-tp-read-trace/` in the +validation artifact directory. All completed tasks released cards 0-3. + +Window layout versus communication execution is still being isolated; no +compiler/runtime attribution or production workaround is established. The +last-row discrepancy does not directly explain earlier causal rows that fail +the accumulated pre-mix gate. Resolving this replay alone will therefore not +establish model precision or M0 acceptance. From 043b7bf0013212c0db267ece3c42672db406f253 Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Tue, 29 Sep 2026 10:28:47 +0800 Subject: [PATCH 55/78] docs(v41): record verified communication alignment fix --- docs/developer-guide/v41-swa-segment.md | 29 ++++++++++++++++++++----- 1 file changed, 24 insertions(+), 5 deletions(-) diff --git a/docs/developer-guide/v41-swa-segment.md b/docs/developer-guide/v41-swa-segment.md index e25210ef..726aa695 100644 --- a/docs/developer-guide/v41-swa-segment.md +++ b/docs/developer-guide/v41-swa-segment.md @@ -315,8 +315,27 @@ failure. Evidence includes `tp-gather-replay-layer0.log`, `tp-gather-replay-layer0.pt` and `swa-realweights-tp-read-trace/` in the validation artifact directory. All completed tasks released cards 0-3. -Window layout versus communication execution is still being isolated; no -compiler/runtime attribution or production workaround is established. The -last-row discrepancy does not directly explain earlier causal rows that fail -the accumulated pre-mix gate. Resolving this replay alone will therefore not -establish model precision or M0 acceptance. +The four-window replay still fails when AllGather execution is omitted. +Reserving 64 bytes for each signal makes it pass without changing arithmetic. +The original 32-byte allocation padding places an output payload tail and its +signal on one 64-byte scalar cache line; signal cache maintenance can affect +the neighboring payload. Existing tracking: `hw-native-sys/pypto#2800` and +`hw-native-sys/simpler#2273`. + +PyPTO diagnostic fix `b792bde6`, based on `e8191e3c`, rounds both each physical +buffer size and their summed window capacity to 64 bytes. Logical views and the +runtime allocation ABI stay unchanged. The original 8-byte logical signal +replay now passes with zero mismatches (`task_20260929_101804_154960424725`). +The full real-weight trace (`task_20260929_102020_17364556272`, lib `d0478d78`, +serving `0355d40`) also has zero publication/peer-read, sum or cast mismatches +across both layers and all four ranks. Both tasks released cards 0-3. + +This fixes the captured communication corruption, but accumulated pre_mix +still fails in 3/256 elements: relative L2 0.0025418127, maximum absolute error +0.0076903105. Residual passes every rank's V4 gate (relative L2 0.012851512, +maximum absolute error 0.02734375). Reference outputs remain bitwise equal to +the earlier Q-B-only run. No arithmetic or acceptance threshold changed. +Artifacts: `tp-codegen64-replay-layer0.log`, `swa-precision-codegen64-trace.log` +and `swa-realweights-codegen64-trace/tp-boundary-audit.json`. +The compiler fix and Q-B group-32 arithmetic remain diagnostic dependencies; +these results do not establish full-model precision or M0 acceptance. From f3608ea654cf70079e6d690d0f8cd8414693a6ae Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Tue, 29 Sep 2026 11:20:02 +0800 Subject: [PATCH 56/78] test(v41): prepare real-weight C1A producer chains --- docs/developer-guide/v41-swa-segment.md | 18 +++ .../model/deepseek_v41/test_c2a_diagnostic.py | 45 ++++++ tools/validate_v41_c2a_chain.py | 132 ++++++++++++++---- 3 files changed, 166 insertions(+), 29 deletions(-) diff --git a/docs/developer-guide/v41-swa-segment.md b/docs/developer-guide/v41-swa-segment.md index 726aa695..a6b228f0 100644 --- a/docs/developer-guide/v41-swa-segment.md +++ b/docs/developer-guide/v41-swa-segment.md @@ -339,3 +339,21 @@ Artifacts: `tp-codegen64-replay-layer0.log`, `swa-precision-codegen64-trace.log` and `swa-realweights-codegen64-trace/tp-boundary-audit.json`. The compiler fix and Q-B group-32 arithmetic remain diagnostic dependencies; these results do not establish full-model precision or M0 acceptance. + +### C1A diagnostic preparation + +The compressed-chain diagnostic also accepts `--family c1a`. It resolves the +checkpoint's first C1A Full producer and executes consecutive layers through +`--last-layer` (inclusive). The default covers layers 20/21. Using +`--last-layer 25` retains producer 20 through Reuse layers 21-23, then executes +Reindex 24 and its Reuse consumer 25. KV/index caches, candidates and Top-K +buffers are bound to their declared producers, including read-only integrity +checks. Weight loading uses the same selective loader as the C2A path. + +For example, add `--family c1a --prepare-only` to the compressed-chain command +to check real weights and metadata before device execution. This mode uses +the saved SWA state as an explicitly injected diagnostic input; it does not +claim to have executed layers 2-19. Artifacts record the chosen family, layer +IDs and this limitation. The existing native C1A comparators are preserved. +Host plan/metadata tests pass; C1A real-weight preparation and device validation +remain pending. This is not accumulated full-model or M0 acceptance. diff --git a/tests/unit/model/deepseek_v41/test_c2a_diagnostic.py b/tests/unit/model/deepseek_v41/test_c2a_diagnostic.py index 544ee7db..fcc9f492 100644 --- a/tests/unit/model/deepseek_v41/test_c2a_diagnostic.py +++ b/tests/unit/model/deepseek_v41/test_c2a_diagnostic.py @@ -14,6 +14,51 @@ from tools.validate_v41_c2a_chain import select_active_state +def checkpoint_config(): + import json + from pathlib import Path + + return json.loads((Path(__file__).resolve().parents[4] / + "tests/fixtures/deepseek_v41/config.json").read_text()) + + +def test_c1a_chain_includes_full_before_reindex_and_reuse(): + from tools.validate_v41_c2a_chain import select_plans + + plans = select_plans(checkpoint_config(), "c1a", 25) + assert [p.layer_id for p in plans] == list(range(20, 26)) + assert [p.mode for p in plans] == ["c1a_full", *["c1a_reuse"] * 3, + "c1a_reindex", "c1a_reuse"] + assert all(p.kv_source == 20 and p.candidate_source == 20 for p in plans) + assert plans[-1].index_source == 24 + + +def test_default_c2a_diagnostic_preserves_original_two_layers(): + from tools.validate_v41_c2a_chain import select_plans + + plans = select_plans(checkpoint_config(), "c2a") + assert [(p.layer_id, p.mode) for p in plans] == [(2, "c2a_full"), (3, "c2a_reuse")] + + +@pytest.mark.parametrize("family,last", [("c1a", 19), ("c1a", 40), ("c2a", 20), + ("c2a", True), ("swa", None)]) +def test_diagnostic_cannot_skip_its_full_producer_or_cross_families(family, last): + from tools.validate_v41_c2a_chain import select_plans + + with pytest.raises(ValueError): + select_plans(checkpoint_config(), family, last) + + +def test_c1a_reuse_captures_readonly_producer_state_for_integrity_checks(): + from types import SimpleNamespace + from tools.validate_v41_c2a_chain import state_names + + module = SimpleNamespace(STATE_NAMES={"reuse": ("window_cache", "window_cache_scale")}) + assert set(state_names(module, "reuse")) == { + "window_cache", "window_cache_scale", "compressed_cache", "compressed_cache_scale", + "index_cache", "index_cache_scale", "topk_indices", "candidate_mask"} + + def test_select_ragged_prefix_without_mutating_saved_state(): topology = SegmentTopology(tp=2, dp=2) residual = torch.arange(4 * 16 * 4 * 5120, dtype=torch.float32).reshape(4, 16, 4, 5120) diff --git a/tools/validate_v41_c2a_chain.py b/tools/validate_v41_c2a_chain.py index 22b37943..cf089449 100644 --- a/tools/validate_v41_c2a_chain.py +++ b/tools/validate_v41_c2a_chain.py @@ -6,11 +6,13 @@ # INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. # See LICENSE in the root of the software repository for the full text of the License. # ----------------------------------------------------------------------------------------------------------- -"""Bounded real-weight C2A Full -> MoE -> Reuse -> MoE device diagnostic. +"""Bounded real-weight compressed Attention -> MoE device diagnostic. Start from the saved output of the two SWA layers. This does not claim that that input passed its accumulated precision gate, or run a full model. Only existing native same-input stage comparators gate this diagnostic. +With --family c1a the saved state is an explicitly injected diagnostic input, +not a claim that layers 2 through 19 have executed. """ import argparse import json @@ -18,6 +20,38 @@ from types import SimpleNamespace +def select_plans(raw, family, last_layer=None): + """Select a consecutive chain beginning with the family's real Full producer.""" + from pypto_serving.model.deepseek_v41.execution_plan import plan_layers + + if family not in ("c1a", "c2a"): + raise ValueError("diagnostic family must be c1a or c2a") + layers = plan_layers(raw) + first = next(p.layer_id for p in layers if p.mode == family + "_full") + last = first + 1 if last_layer is None else last_layer + if type(last) is not int or not first <= last < len(layers): + raise ValueError("last layer must belong to the selected compressed family") + selected = layers[first:last + 1] + if any(not p.mode.startswith(family + "_") for p in selected): + raise ValueError("diagnostic cannot cross compressed families") + seen = set() + for plan in selected: + seen.add(plan.layer_id) + if any(source is not None and source not in seen + for source in (plan.kv_source, plan.index_source, plan.candidate_source)): + raise ValueError("diagnostic is missing a preceding state producer") + return selected + + +def state_names(module, mode): + """Capture both mutable and read-only state checked by the native contract.""" + if hasattr(module, "STATE_NAMES"): + return tuple(dict.fromkeys((*module.STATE_NAMES[mode], + "compressed_cache", "compressed_cache_scale", "index_cache", "index_cache_scale", + "topk_indices", "candidate_mask"))) + return module.ATTENTION_STATE[mode] + + def select_active_state(saved, topology, group_counts, starts=None): """Select causal prefixes for a ragged diagnostic, preserving the source artifact.""" import torch @@ -48,7 +82,6 @@ def prepare(args, topology, module, weight_bundles=None): from models.deepseek_v4_1_flash.config import FLASH from models.deepseek_v4_1_flash.rope_tables import precompute_rope_tables from pypto_serving.model.deepseek_v41.compressed_metadata import prepare_compressed_metadata - from pypto_serving.model.deepseek_v41.execution_plan import plan_layers from pypto_serving.model.deepseek_v41.request_state import ForwardStep, RequestSlice from pypto_serving.model.deepseek_v41.swa_metadata import gather_swa_rope_rows, prepare_swa_window_metadata from pypto_serving.model.deepseek_v41.swa_weights import load_prefill_layer_weights @@ -65,9 +98,8 @@ def prepare(args, topology, module, weight_bundles=None): global_counts, local_counts = topology.counts(group_counts) residual, mix = select_active_state(saved, topology, group_counts, starts) raw = json.loads((Path(args.model_dir) / "config.json").read_text()) - plans = plan_layers(raw)[2:4] - if tuple(p.mode for p in plans) != ("c2a_full", "c2a_reuse"): - raise ValueError("checkpoint layers 2/3 are not the required Full/Reuse pair") + family = getattr(args, "family", "c2a") + plans = select_plans(raw, family, getattr(args, "last_layer", None)) text = raw["text_config"] for key in ("qk_rope_head_dim", "rope_theta", "compress_rope_theta"): if text[key] != getattr(FLASH, key): @@ -77,7 +109,7 @@ def prepare(args, topology, module, weight_bundles=None): "original_max_position_embeddings")): if text["rope_scaling"][source] != getattr(FLASH, target): raise ValueError(f"lib/checkpoint compressed RoPE mismatch: {source}") - # Fresh first-chunk history at layer 2, private pages per DP group. + # Private physical pages per DP group, shared only through declared producers. pages = (topology.capacity + 127) // 128 step = ForwardStep("prefill", tuple(RequestSlice(str(g), g, 0, starts[g], tuple(row[starts[g]:starts[g] + group_counts[g]].tolist()), @@ -89,8 +121,13 @@ def prepare(args, topology, module, weight_bundles=None): seed=11, case="mixed", dp_tokens=None, epochs=1, bench=False) attention, moe = {}, {} for plan in plans: - mode = plan.mode.removeprefix("c2a_") - values = {s.name: s.create_tensor().contiguous() for s in module.build_specs(fixture, mode, {}) + mode = plan.mode.split("_")[1] + if family == "c1a": + fixture.case, fixture.dp_tokens = "causal", list(group_counts) + specs = module.build_specs(fixture, {}) + else: + specs = module.build_specs(fixture, mode, {}) + values = {s.name: s.create_tensor().contiguous() for s in specs if isinstance(s, TensorSpec)} if plan.layer_id not in weight_bundles: print(f"Loading real checkpoint layer {plan.layer_id}", flush=True) @@ -102,7 +139,22 @@ def prepare(args, topology, module, weight_bundles=None): values.update(window_slots=window.window_slots, window_indices=window.window_indices) values["window_cache"].view(torch.uint8).zero_() values["window_cache_scale"].view(torch.uint8).fill_(127) - if mode == "full": + if family == "c1a": + cm = prepare_compressed_metadata(step, topology, ratio=1, compressed_group="cmp", + cache_pages=values["compressed_cache"].shape[1], max_requests=1, state_blocks=1) + values.update(request_ids=cm.request_ids, compressed_lens=cm.compressed_lens, + compressed_slots=cm.compressed_slots, index_block_table=cm.index_block_table) + values["rope_cos"], values["rope_sin"] = gather_swa_rope_rows(window, tables) + values["compressed_rope_cos"], values["compressed_rope_sin"] = gather_swa_rope_rows( + window, compressed_tables) + for name in ("compressed_cache", "index_cache"): + values[name].view(torch.uint8).zero_() + values["compressed_cache_scale"].view(torch.uint8).fill_(0x38) + values["index_cache_scale"].view(torch.uint8).fill_(127) + values["topk_indices"].fill_(-1) + values["compressed_indices"].fill_(-1) + values["candidate_mask"].zero_() + elif mode == "full": cm = prepare_compressed_metadata(step, topology, ratio=2, compressed_group="cmp", cache_pages=values["compressed_cache"].shape[1], max_requests=1, state_blocks=values["state_cache"].shape[1]) @@ -136,14 +188,18 @@ def main(): parser.add_argument("--input-state", required=True) parser.add_argument("--devices", default="0,1,2,3") parser.add_argument("--tp", type=int, default=2) + parser.add_argument("--family", choices=("c2a", "c1a"), default="c2a") + parser.add_argument("--last-layer", type=int, help="Inclusive last layer; starts at the real Full producer") parser.add_argument("--group-counts", help="Comma-separated causal prefix lengths per DP group") parser.add_argument("--continue-to-capacity", action="store_true", help="Run a second chunk of each nonempty request using its resident caches") parser.add_argument("--prepare-only", action="store_true") - parser.add_argument("--build-dir", default="build_output/v41-c2a-chain") - parser.add_argument("--artifact-dir", default=".validation-artifacts/c2a-chain") + parser.add_argument("--build-dir") + parser.add_argument("--artifact-dir") parser.add_argument("--ring-heap-mib", type=int, default=4096) args = parser.parse_args() + args.build_dir = args.build_dir or f"build_output/v41-{args.family}-chain" + args.artifact_dir = args.artifact_dir or f".validation-artifacts/{args.family}-chain" import torch from pypto_serving.model.deepseek_v41.swa_segment import SegmentTopology, load_segment_modules from pypto_serving.model.deepseek_v41.prefill_segment import ( @@ -158,7 +214,11 @@ def main(): topology.counts(args.group_counts) torch.set_num_threads(4) _, moe_module = load_segment_modules(args.lib_root, topology) - from models.deepseek_v4_1_flash import prefill_c2a_full as c2a + from models.deepseek_v4_1_flash import prefill_c2a_full as common + if args.family == "c1a": + from models.deepseek_v4_1_flash import prefill_c1a_sp as module + else: + module = common chunks = [(args.group_counts, [0] * topology.dp)] if args.continue_to_capacity: remaining = [topology.capacity - n if n else 0 for n in args.group_counts] @@ -168,10 +228,10 @@ def main(): prepared, weights = [], {} for counts, starts in chunks: options = SimpleNamespace(**{**vars(args), "group_counts": counts, "starts": starts}) - plans, attention, moe, residual, mix = prepare(options, topology, c2a, weights) + plans, attention, moe, residual, mix = prepare(options, topology, module, weights) if prepared: for plan in plans: - for name in c2a.ATTENTION_STATE[plan.mode.removeprefix("c2a_")]: + for name in state_names(module, plan.mode.split("_")[1]): if "cache" in name: attention[plan.layer_id][name] = prepared[0][0][plan.layer_id][name] prepared.append((bind_prefill_producers(plans, attention), moe, residual, mix, counts)) @@ -182,7 +242,8 @@ def main(): for name in PREFILL_ARGUMENTS[plan.mode]: if name not in dynamic: assert isinstance(bound[plan.layer_id][name], torch.Tensor), name - print("REAL CHECKPOINT C2A PREPARATION PASS", flush=True) + family_label = args.family.upper() + print(f"REAL CHECKPOINT {family_label} PREPARATION PASS; layers {[p.layer_id for p in plans]}", flush=True) if args.prepare_only: return from pypto.ir import DistributedConfig @@ -206,8 +267,8 @@ def main(): captured = {} for plan in plans: layer = plan.layer_id - names = ("attn_input", "attn_output", "next_pre_mix", "x_hc_out") + c2a.ATTENTION_STATE[ - plan.mode.removeprefix("c2a_")] + names = ("attn_input", "attn_output", "next_pre_mix", "x_hc_out") + state_names( + module, plan.mode.split("_")[1]) captured[layer] = ({n: torch.empty_like(bound[layer][n]).share_memory_() for n in names}, {n: torch.empty_like(moe[layer][n]).share_memory_() for n in ("x_next", "next_pre_mix", "x_mixed")}) @@ -230,13 +291,14 @@ def upload(values): for bound, moe, *_ in prepared] runner = PrefillSegment(worker, programs, ffn, topology, ac, mc, config) for step_id, ((da, dm), captured) in enumerate(zip(device_steps, captures)): - runner.run_chain(LayerState(da[2]["x_hc"], da[2]["pre_mix"], "tp_local_token"), + first = plans[0].layer_id + runner.run_chain(LayerState(da[first]["x_hc"], da[first]["pre_mix"], "tp_local_token"), plans, da, dm, group_counts=prepared[step_id][4]) for layer, (ca, cm) in captured.items(): for device, host in ((da[layer], ca), (dm[layer], cm)): for name, destination in host.items(): worker.copy_stacked_from(device[name], destination) - print(f"C2A DEVICE CHUNK {step_id} COMPLETE", flush=True) + print(f"{family_label} DEVICE CHUNK {step_id} COMPLETE", flush=True) for handle in reversed(list(uploaded.values())): worker.free_stacked_tensor(handle) # CPU same-input references run only after worker shutdown. Device state @@ -244,22 +306,32 @@ def upload(values): records, passed = [], True for step_id, ((bound, moe, residual, mix, counts), captured) in enumerate(zip(prepared, captures)): for plan in plans: - layer, mode = plan.layer_id, plan.mode.removeprefix("c2a_") + layer, mode = plan.layer_id, plan.mode.split("_")[1] actual_a, actual_m = captured[layer] inputs = dict(bound[layer], x_hc=residual, pre_mix=mix) if step_id: for name, value in captures[step_id - 1][layer][0].items(): if "cache" in name: inputs[name] = value - if mode == "reuse": + if mode != "full": for name in ("compressed_cache", "compressed_cache_scale"): - inputs[name] = captured[2][0][name] - inputs["compressed_indices"] = captured[2][0]["topk_indices"] - initial_cache = {n: inputs[n].clone() for n in c2a.MODES[mode][1]} + inputs[name] = captured[plan.kv_source][0][name] + if args.family == "c1a": + for name in ("index_cache", "index_cache_scale"): + inputs[name] = captured[plan.kv_source][0][name] + inputs["candidate_mask"] = captured[plan.candidate_source][0]["candidate_mask"] + if mode == "reuse": + inputs["compressed_indices"] = captured[plan.index_source][0]["topk_indices"] + if args.family == "c1a": + inputs["topk_indices"] = inputs["compressed_indices"] + initial_cache = {n: inputs[n].clone() for n in state_names(module, mode)} expected_a = dict(inputs) for name in actual_a: expected_a[name] = inputs[name].clone() - c2a.make_golden(mode, 1)(expected_a) + golden = (common.make_golden(mode, 1, attention_reference=module.reference_attention, + state_names=module.STATE_NAMES[mode]) if args.family == "c1a" + else common.make_golden(mode, 1)) + golden(expected_a) expected_m = dict(moe[layer], x_hc=actual_a["x_hc_out"], pre_mix=actual_a["next_pre_mix"]) for name in actual_m: expected_m[name] = moe[layer][name].clone() @@ -270,7 +342,7 @@ def upload(values): "x_next": moe_module._local_mhc_compare(list(topology.counts(counts)[1])), } for label, actual, expected, checks in (("attention", actual_a, expected_a, - c2a.make_compare(mode, 1, initial_cache)), ("moe", actual_m, expected_m, mc_check)): + module.make_compare(mode, 1, initial_cache)), ("moe", actual_m, expected_m, mc_check)): results = {} for name, check in checks.items(): ok, detail = check(actual[name], expected[name], inputs=expected, @@ -281,11 +353,13 @@ def upload(values): records.append(dict(chunk=step_id, layer=layer, stage=label, results=results, actual=actual, expected={n: expected[n] for n in actual})) residual, mix = actual_m["x_next"], actual_m["next_pre_mix"] - torch.save({"input_state": str(args.input_state), "group_counts": args.group_counts, "chunks": chunks, "stages": records, + torch.save({"input_state": str(args.input_state), "family": args.family, + "layer_ids": [p.layer_id for p in plans], "injected_boundary_input": True, + "group_counts": args.group_counts, "chunks": chunks, "stages": records, "actual_residual": residual, "actual_pre_mix": mix}, artifact / "comparison.pt") if not passed: - raise AssertionError("C2A chain native stage check failed; see comparison.pt") - print("C2A CHAIN NATIVE STAGES PASS; accumulated full-model acceptance remains pending", flush=True) + raise AssertionError(f"{family_label} chain native stage check failed; see comparison.pt") + print(f"{family_label} CHAIN NATIVE STAGES PASS; accumulated full-model acceptance remains pending", flush=True) if __name__ == "__main__": From a11532c9171104377057d85079b842370957c916 Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Tue, 29 Sep 2026 11:35:16 +0800 Subject: [PATCH 57/78] fix(v41): keep C1A index argument write ranges disjoint --- docs/developer-guide/deepseek-v41-entry.md | 21 ++++++++++++------- .../model/deepseek_v41/prefill_segment.py | 11 ++++++---- .../deepseek_v41/test_prefill_producers.py | 19 +++++++++++++++++ tools/validate_v41_c2a_chain.py | 2 -- 4 files changed, 40 insertions(+), 13 deletions(-) diff --git a/docs/developer-guide/deepseek-v41-entry.md b/docs/developer-guide/deepseek-v41-entry.md index 9d75d347..c18e1242 100644 --- a/docs/developer-guide/deepseek-v41-entry.md +++ b/docs/developer-guide/deepseek-v41-entry.md @@ -180,13 +180,20 @@ contract and is explicitly rejected for now. Cache payloads must not use a generic dense K/V substitute. The adapter must bound physical page IDs against actual allocated pools when a group leaves `num_blocks` unspecified. -At inspected upstream lib revision `1b8caa4`, packed-FP4 MoE and token-local -decode Attention progress do not yet provide a compatible complete-layer -adapter. Decode `stage="block"` remains disabled; the prefill fixture still -uses the older routed-weight/call contract. Initial residual/pre-mix, complete -prefill/decode, cache allocation/reset and final HC/norm/head remain explicit -integration work. These facts do not block testing the serving state machine, -but they do block real-model generation and M0 numerical acceptance. +At inspected upstream lib revision `fbe92bfc`, serving dispatches the existing +sequence-parallel Attention and packed-FP4 MoE composites through +`SwaSegment`/`PrefillSegment`. This does not depend on the older full-layer +prefill fixture. Initial residual/pre-mix, private request metadata and +producer-owned compressed caches have bounded integration evidence; see +[the segment validation record](v41-swa-segment.md) for tested modes and +precision limits. + +The default complete-model adapter is still unavailable. Decode `stage="block"` +remains disabled in this lib revision, and the final existing-state HC+Norm +composition is missing: `boundary_embed_to_norm` repacks embeddings and skips +the backbone, so it cannot consume the final layer's state. All-mode device +validation, reset/recovery and complete generation remain integration work. +Bounded half-layer dispatch does not establish full-model or M0 acceptance. ## Request state and serving lifecycle diff --git a/pypto_serving/model/deepseek_v41/prefill_segment.py b/pypto_serving/model/deepseek_v41/prefill_segment.py index faea40af..fa591772 100644 --- a/pypto_serving/model/deepseek_v41/prefill_segment.py +++ b/pypto_serving/model/deepseek_v41/prefill_segment.py @@ -55,8 +55,9 @@ def bind_prefill_producers(plans, arguments): C1A's common ABI retains unused Full weights in Reindex/Reuse. Bind those slots to their real producer weights; never allocate placeholder weights. - Full/Reindex do not read compressed_indices; their own Top-K allocation is - a valid shape-compatible binding for that unused slot (lib fbe92bfc). + Keep unused index slots in separate allocations. The common C1A ABI marks + them writable even when a selected mode does not use them; aliasing those + arguments makes the runtime reject overlapping write ranges (lib fbe92bfc). """ plans = tuple(plans) if not plans or any(not isinstance(p, LayerPlan) for p in plans): @@ -102,9 +103,11 @@ def producer(source, kind): current[name] = kv[name] current["candidate_mask"] = candidate["candidate_mask"] if mode == "reuse": - for name in ("index_wq_b", "index_wq_b_scale", "index_weights_proj", "topk_indices"): + for name in ("index_wq_b", "index_wq_b_scale", "index_weights_proj"): current[name] = index[name] - current["compressed_indices"] = index["topk_indices"] + current["compressed_indices"] = index["topk_indices"] + if current["compressed_indices"] is current["topk_indices"]: + raise ValueError("C1A index ABI arguments require separate allocations") elif mode == "reuse": current["compressed_indices"] = index["topk_indices"] elif any(source is not None for source in (plan.kv_source, plan.index_source, plan.candidate_source)): diff --git a/tests/unit/model/deepseek_v41/test_prefill_producers.py b/tests/unit/model/deepseek_v41/test_prefill_producers.py index 38aba52d..475c901a 100644 --- a/tests/unit/model/deepseek_v41/test_prefill_producers.py +++ b/tests/unit/model/deepseek_v41/test_prefill_producers.py @@ -47,6 +47,10 @@ def test_actual_40_layer_plan_shares_producers_without_copying_or_mutating_input if p.mode.endswith("reuse"): assert current["compressed_indices"] is original[p.index_source]["topk_indices"] if p.mode.startswith("c1a"): + assert current["topk_indices"] is original[p.layer_id]["topk_indices"] + assert current["topk_indices"] is not current["compressed_indices"] + if not p.mode.endswith("reuse"): + assert current["compressed_indices"] is original[p.layer_id]["compressed_indices"] assert current["candidate_mask"] is original[20]["candidate_mask"] assert current["index_cache"] is original[20]["index_cache"] assert current["compressor_wkv"] is original[20]["compressor_wkv"] @@ -56,6 +60,21 @@ def test_actual_40_layer_plan_shares_producers_without_copying_or_mutating_input assert original[21]["compressed_indices"] is not bound[21]["compressed_indices"] +@pytest.mark.parametrize("layer", [20, 24]) +def test_unused_c1a_input_cannot_alias_topk_write_argument(layer): + plans, buffers = plans_and_buffers() + buffers[layer]["compressed_indices"] = buffers[layer]["topk_indices"] + with pytest.raises(ValueError, match="separate allocations"): + bind_prefill_producers(plans, buffers) + + +def test_unused_reuse_output_cannot_alias_its_producer_selection(): + plans, buffers = plans_and_buffers() + buffers[25]["topk_indices"] = buffers[24]["topk_indices"] + with pytest.raises(ValueError, match="separate allocations"): + bind_prefill_producers(plans, buffers) + + @pytest.mark.parametrize("plans", [ (LayerPlan(3, "c2a_reuse", 2, 2, None),), (LayerPlan(20, "c1a_full", 20, 20, 20), LayerPlan(22, "c1a_reuse", 20, 20, 20)), diff --git a/tools/validate_v41_c2a_chain.py b/tools/validate_v41_c2a_chain.py index cf089449..92d544ae 100644 --- a/tools/validate_v41_c2a_chain.py +++ b/tools/validate_v41_c2a_chain.py @@ -322,8 +322,6 @@ def upload(values): inputs["candidate_mask"] = captured[plan.candidate_source][0]["candidate_mask"] if mode == "reuse": inputs["compressed_indices"] = captured[plan.index_source][0]["topk_indices"] - if args.family == "c1a": - inputs["topk_indices"] = inputs["compressed_indices"] initial_cache = {n: inputs[n].clone() for n in state_names(module, mode)} expected_a = dict(inputs) for name in actual_a: From 14e6a5aea91182b88eeee35fb25351c3b9d5aca2 Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Tue, 29 Sep 2026 11:44:29 +0800 Subject: [PATCH 58/78] test(v41): retain per-layer compressed attention dumps --- tools/validate_v41_c2a_chain.py | 25 ++++++++++++++++++++++--- 1 file changed, 22 insertions(+), 3 deletions(-) diff --git a/tools/validate_v41_c2a_chain.py b/tools/validate_v41_c2a_chain.py index 92d544ae..bc9d1b96 100644 --- a/tools/validate_v41_c2a_chain.py +++ b/tools/validate_v41_c2a_chain.py @@ -194,6 +194,8 @@ def main(): parser.add_argument("--continue-to-capacity", action="store_true", help="Run a second chunk of each nonempty request using its resident caches") parser.add_argument("--prepare-only", action="store_true") + parser.add_argument("--dump-tagged", action="store_true", + help="Preserve lib-tagged kernel arguments after each diagnostic layer") parser.add_argument("--build-dir") parser.add_argument("--artifact-dir") parser.add_argument("--ring-heap-mib", type=int, default=4096) @@ -257,7 +259,8 @@ def main(): if (artifact / "comparison.pt").exists(): raise FileExistsError("refusing to overwrite a previous comparison") config = RunConfig(platform="a5", distributed_config=DistributedConfig(device_ids=list(devices)), - ring_heap=args.ring_heap_mib << 20, ring_task_window=131072, ring_dep_pool=131072) + ring_heap=args.ring_heap_mib << 20, ring_task_window=131072, ring_dep_pool=131072, + enable_dump_args=1 if args.dump_tagged else 0) compiler = KernelCompiler(run_config=config, cache_dir=args.build_dir) programs, ffn = compile_prefill_segments(compiler, args.lib_root, topology, [p.mode for p in plans]) ac = torch.zeros(topology.world, 1, dtype=torch.int32).share_memory_() @@ -292,8 +295,24 @@ def upload(values): runner = PrefillSegment(worker, programs, ffn, topology, ac, mc, config) for step_id, ((da, dm), captured) in enumerate(zip(device_steps, captures)): first = plans[0].layer_id - runner.run_chain(LayerState(da[first]["x_hc"], da[first]["pre_mix"], "tp_local_token"), - plans, da, dm, group_counts=prepared[step_id][4]) + state = LayerState(da[first]["x_hc"], da[first]["pre_mix"], "tp_local_token") + if args.dump_tagged: + import shutil + + # Same composite calls and epochs; only completed diagnostic files + # are moved before a repeated program can reuse the dump path. + for plan in plans: + state = runner.run_layer(state, da[plan.layer_id], dm[plan.layer_id], + group_counts=prepared[step_id][4], mode=plan.mode) + destination = artifact / f"chunk-{step_id}-layer-{plan.layer_id}-dumps" + for manifest in list(Path(args.build_dir).rglob("args_dump.json")): + target = destination / manifest.parent.relative_to(Path(args.build_dir)) + if target.exists(): + raise FileExistsError(f"Refusing to overwrite diagnostic dumps: {target}") + target.parent.mkdir(parents=True, exist_ok=True) + shutil.move(str(manifest.parent), str(target)) + else: + runner.run_chain(state, plans, da, dm, group_counts=prepared[step_id][4]) for layer, (ca, cm) in captured.items(): for device, host in ((da[layer], ca), (dm[layer], cm)): for name, destination in host.items(): From 8da7adc40655fdfaaed3e5507f2351ef558cd19b Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Tue, 29 Sep 2026 11:47:58 +0800 Subject: [PATCH 59/78] docs(v41): record C1A device coverage and numerical failures --- docs/developer-guide/v41-swa-segment.md | 19 +++++++++++++++++-- 1 file changed, 17 insertions(+), 2 deletions(-) diff --git a/docs/developer-guide/v41-swa-segment.md b/docs/developer-guide/v41-swa-segment.md index a6b228f0..2b2025c6 100644 --- a/docs/developer-guide/v41-swa-segment.md +++ b/docs/developer-guide/v41-swa-segment.md @@ -355,5 +355,20 @@ to check real weights and metadata before device execution. This mode uses the saved SWA state as an explicitly injected diagnostic input; it does not claim to have executed layers 2-19. Artifacts record the chosen family, layer IDs and this limitation. The existing native C1A comparators are preserved. -Host plan/metadata tests pass; C1A real-weight preparation and device validation -remain pending. This is not accumulated full-model or M0 acceptance. +Host plan/metadata tests and real-checkpoint preparation pass. On A5 four-card +TP2/DP2/EP4, serving `a11532c`, official lib `fbe92bfc` and the previously +validated PyPTO communication fix `b792bde6` execute both layers and pass 28/30 +native checks. Full layer 20 fails `attn_output` and `x_hc_out`: DP group 1, +row 18 has attention error RMS 0.0090870445 against limit 0.0083356445. +Same-input HC post replay is within its local budget, but end-to-end residual +relative L2 exceeds the native 1% limit on ranks 2/3. No comparator changed. + +Full-layer cache, candidate and Top-K checks, all Reuse-layer checks (including +read-only producer state), and both MoE stages pass. The earlier task rejected +aliased index ABI arguments before executing Attention; serving now keeps +unused index slots disjoint and shares only the actual producer selection with +its consumer. The runtime rejection is resolved; the numerical failures remain. +Evidence: `c1a-chain-disjoint/comparison.pt`, `c1a-chain-disjoint.log`, task +`task_20260929_113608_405763621400` (exit 1, all four cards released). +Reindex and continuation validation remain pending. This is not accumulated +full-model or M0 acceptance. From c5945ebb3972adc169467b0653ef7ca9e0fc7964 Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Tue, 29 Sep 2026 12:02:17 +0800 Subject: [PATCH 60/78] docs(v41): record C1A probability narrowing isolation --- docs/developer-guide/v41-swa-segment.md | 19 +++++++++++++++++++ 1 file changed, 19 insertions(+) diff --git a/docs/developer-guide/v41-swa-segment.md b/docs/developer-guide/v41-swa-segment.md index 2b2025c6..a28ecf3d 100644 --- a/docs/developer-guide/v41-swa-segment.md +++ b/docs/developer-guide/v41-swa-segment.md @@ -372,3 +372,22 @@ Evidence: `c1a-chain-disjoint/comparison.pt`, `c1a-chain-disjoint.log`, task `task_20260929_113608_405763621400` (exit 1, all four cards released). Reindex and continuation validation remain pending. This is not accumulated full-model or M0 acceptance. + +The tag-only C1A capture (`b4fa1c91`, serving `14e6a5a`) leaves every saved +actual and expected stage tensor bitwise unchanged. Same-input QNorm, query +RoPE and inverse RoPE are exact; both DP groups' captured TP sums also match +exactly. Q-A, Q-B and O-A relative L2 errors are on the order of 1e-5, and O-B +is about 2e-7. Attention-core relative L2 is 0.00151-0.00165. A CPU replay +of the kernel's BF16 probability narrowing reproduces over 99.998% of the +captured core elements. Propagating the captured core through the independent +output projection reduces output relative L2 to 0.00012 / 0.000047, whereas +using FP32 probabilities on the same query/cache inputs differs from the +device output by about 1%. These cuts are diagnostic, not acceptance results. + +The C1A candidate `abfa7197` preserves probability rounding residuals with +two BF16 Cube products, without changing inputs, reference or native gates. +It is awaiting device validation and is not a production dependency. Do not +apply the same change blindly to SWA: its existing `official_reference` +explicitly includes BF16 probabilities, unlike the C1A sparse reference. +Evidence: `c1a-chain-trace/boundary-audit.json`, `boundary-audit.pt`, +`cpu-boundary-cuts.pt` and `c1a-cpu-boundary-cuts.log`. From 77e4844d6ae2641b2ebf0f6888edab8887801f27 Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Tue, 29 Sep 2026 12:34:28 +0800 Subject: [PATCH 61/78] docs(v41): correct C1A diagnostic reference and reject PV candidate --- docs/developer-guide/v41-swa-segment.md | 28 ++++++++++++++----------- 1 file changed, 16 insertions(+), 12 deletions(-) diff --git a/docs/developer-guide/v41-swa-segment.md b/docs/developer-guide/v41-swa-segment.md index a28ecf3d..ad27624c 100644 --- a/docs/developer-guide/v41-swa-segment.md +++ b/docs/developer-guide/v41-swa-segment.md @@ -377,17 +377,21 @@ The tag-only C1A capture (`b4fa1c91`, serving `14e6a5a`) leaves every saved actual and expected stage tensor bitwise unchanged. Same-input QNorm, query RoPE and inverse RoPE are exact; both DP groups' captured TP sums also match exactly. Q-A, Q-B and O-A relative L2 errors are on the order of 1e-5, and O-B -is about 2e-7. Attention-core relative L2 is 0.00151-0.00165. A CPU replay -of the kernel's BF16 probability narrowing reproduces over 99.998% of the -captured core elements. Propagating the captured core through the independent -output projection reduces output relative L2 to 0.00012 / 0.000047, whereas -using FP32 probabilities on the same query/cache inputs differs from the -device output by about 1%. These cuts are diagnostic, not acceptance results. - -The C1A candidate `abfa7197` preserves probability rounding residuals with -two BF16 Cube products, without changing inputs, reference or native gates. -It is awaiting device validation and is not a production dependency. Do not -apply the same change blindly to SWA: its existing `official_reference` -explicitly includes BF16 probabilities, unlike the C1A sparse reference. +is about 2e-7. A CPU replay of the kernel's BF16 probability narrowing +reproduces over 99.998% of the captured core elements. Propagating the captured +core through the independent output projection reduces output relative L2 to +0.00012 / 0.000047. These cuts are diagnostic, not acceptance results. + +The initial attention-core audit mistakenly used the generic FP32-probability +reference: its 0.00151-0.00165 relative L2 is not the native C1A contract. +All three modes inject `golden_prefill_c1a_attention`, which includes BF16 PV +and the first-vector FP32 patch. The probability-residual candidate `1d186943` +is rejected: it passes only 27/30 native checks and introduces a Reuse output +failure. First-layer expected tensors are bitwise unchanged from baseline; +the candidate never changed the acceptance reference or gates. The initial +`abfa7197` attempt stopped at accumulator dtype checking; the retry explicitly +declares FP32 and completed device execution before failing precision. +Further bisection must follow the specialized reference, not replace it with +generic sparse attention. The same caveat applies to SWA's BF16-P reference. Evidence: `c1a-chain-trace/boundary-audit.json`, `boundary-audit.pt`, `cpu-boundary-cuts.pt` and `c1a-cpu-boundary-cuts.log`. From de14b45f8514c69230bc2d7d407a831a0bdf9318 Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Tue, 29 Sep 2026 12:46:07 +0800 Subject: [PATCH 62/78] docs(v41): record Q-A improvement and cache error propagation --- docs/developer-guide/v41-swa-segment.md | 27 +++++++++++++++++++++++++ 1 file changed, 27 insertions(+) diff --git a/docs/developer-guide/v41-swa-segment.md b/docs/developer-guide/v41-swa-segment.md index ad27624c..4f9d937e 100644 --- a/docs/developer-guide/v41-swa-segment.md +++ b/docs/developer-guide/v41-swa-segment.md @@ -395,3 +395,30 @@ Further bisection must follow the specialized reference, not replace it with generic sparse attention. The same caveat applies to SWA's BF16-P reference. Evidence: `c1a-chain-trace/boundary-audit.json`, `boundary-audit.pt`, `cpu-boundary-cuts.pt` and `c1a-cpu-boundary-cuts.log`. + +With the specialized reference, the original DP1 token 18 attention error +is 1.0903%; continuing from captured Q-A reduces it to 0.1855%. Independent +FP64 evaluation confirms a Q-A BF16 rounding error at that token. Candidate +`b80bb8b3` routes all three C1A modes through the existing scale-corrected +group-32 prefill Q-A implementation. On the same Full20/Reuse21 case, it +passes **29/30 native checks**, including the previously failing Attention +output. Only Full20 end-to-end `x_hc_out` fails: ranks 2/3 relative L2 are +0.0161307 / 0.0119976. All first-layer expected tensors remain bitwise equal +to baseline. No probability arithmetic or acceptance check changed. + +Query/cache boundary cuts isolate this remaining error: using captured caches +with the canonical query reduces HC relative L2 to 0.003437 / 0.002168; +changing only the query does not improve it. The canonical CPU replay matches +the saved expected Attention output exactly. This is diagnostic evidence, +not a replacement reference or a passing full-layer result. + +The original normalized input differs in only six BF16 elements. At all six +coordinates, device values match the independent FP64 collapse/normalization +rounded to BF16. For example, DP1 token 0, feature 4237 has an exact FP64 +collapse sum of 0.252929660224396; the FP32 reference lands on the BF16 midpoint +0.2529296875 and rounds to the other neighbor. Subsequent quantization changes +cache values and amplifies the difference. This does not establish an incorrect +device HC implementation. The original gate remains failed; neither weakening +it nor changing the reference is part of this diagnostic. Evidence: +`c1a-chain-qa32/comparison.pt`, `query-cache-cuts.pt`, `c1a-chain-qa32.log`, +`c1a-query-cache-cuts.log` and `c1a-input-rounding-details.log`. From 9ad837ef31252fd9886fc00f175ba94b9ed1efe4 Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Tue, 29 Sep 2026 12:59:26 +0800 Subject: [PATCH 63/78] test(v41): isolate C1A reuse attention replay --- tools/replay_v41_c1a_attention.py | 127 ++++++++++++++++++++++++++++++ 1 file changed, 127 insertions(+) create mode 100644 tools/replay_v41_c1a_attention.py diff --git a/tools/replay_v41_c1a_attention.py b/tools/replay_v41_c1a_attention.py new file mode 100644 index 00000000..e84d2204 --- /dev/null +++ b/tools/replay_v41_c1a_attention.py @@ -0,0 +1,127 @@ +# Copyright (c) PyPTO Contributors. +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of +# CANN Open Software License Agreement Version 2.0 (the "License"). +# Please refer to the License for details. You may not use this file except in compliance with the License. +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. +# See LICENSE in the root of the software repository for the full text of the License. +# ----------------------------------------------------------------------------------------------------------- +"""Replay one C1A Reuse Attention composite from a saved same-input chain. + +This is a diagnostic readback/replay, not a production serving path or an +accumulated accuracy result. No routed expert weights are loaded. Establish +bitwise equivalence to the original capture before interpreting tagged cuts. +""" +import argparse +import ctypes +import json +from pathlib import Path +from types import SimpleNamespace + +from validate_v41_c2a_chain import prepare, select_plans, state_names + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--lib-root", required=True) + parser.add_argument("--model-dir", required=True) + parser.add_argument("--comparison", required=True) + parser.add_argument("--layer", type=int, default=22) + parser.add_argument("--tp", type=int, default=2) + parser.add_argument("--devices", default="0,1,2,3") + parser.add_argument("--build-dir", required=True) + parser.add_argument("--artifact-dir", required=True) + parser.add_argument("--dump-tagged", action="store_true") + args = parser.parse_args() + + import torch + import pypto.language as pl + from pypto.ir import DistributedConfig + from pypto.runtime import RunConfig + from pypto_serving.model.common.compiler.compiler import KernelCompiler + from pypto_serving.model.deepseek_v41.prefill_segment import C1A_ARGS, bind_prefill_producers + from pypto_serving.model.deepseek_v41.swa_segment import ( + SegmentTopology, load_segment_modules, make_segment_worker, + ) + from pypto_serving.model.deepseek_v41.swa_weights import load_prefill_attention_weights + + torch.set_num_threads(4) + devices = tuple(int(d) for d in args.devices.split(",")) + if args.tp <= 0 or len(set(devices)) != len(devices) or len(devices) % args.tp: + raise ValueError("devices must form complete TP groups") + topology = SegmentTopology(tp=args.tp, dp=len(devices) // args.tp) + load_segment_modules(args.lib_root, topology) + from models.deepseek_v4_1_flash import prefill_c1a_sp as module, prefill_c2a_full as common + + artifact = Path(args.artifact_dir) + artifact.mkdir(parents=True, exist_ok=True) + if (artifact / "comparison.pt").exists(): + raise FileExistsError("refusing to overwrite a previous replay") + saved = torch.load(args.comparison, map_location="cpu", weights_only=True) + if saved["family"] != "c1a" or len(saved["chunks"]) != 1: + raise ValueError("replay currently requires a single-chunk C1A capture") + records = {(r["layer"], r["stage"]): r for r in saved["stages"]} + raw = json.loads((Path(args.model_dir) / "config.json").read_text()) + plans = select_plans(raw, "c1a", args.layer) + plan = plans[-1] + if plan.mode != "c1a_reuse": + raise ValueError("isolated replay currently requires a Reuse consumer") + # Reuse the original metadata builder and identical Attention weight loader. + # Empty MoE weight maps are unused by this Attention-only diagnostic. + weights = {p.layer_id: (load_prefill_attention_weights(args.model_dir, p.layer_id, topology), {}) + for p in plans} + options = SimpleNamespace(model_dir=args.model_dir, input_state=saved["input_state"], + family="c1a", last_layer=args.layer, group_counts=saved["group_counts"]) + _, attention, _, _, _ = prepare(options, topology, module, weights) + values = bind_prefill_producers(plans, attention)[args.layer] + previous = records[args.layer - 1, "moe"]["actual"] + values.update(x_hc=previous["x_next"], pre_mix=previous["next_pre_mix"]) + for name in ("compressed_cache", "compressed_cache_scale", "index_cache", "index_cache_scale"): + values[name] = records[plan.kv_source, "attention"]["actual"][name] + values["candidate_mask"] = records[plan.candidate_source, "attention"]["actual"]["candidate_mask"] + values["compressed_indices"] = records[plan.index_source, "attention"]["actual"]["topk_indices"] + names = ("attn_input", "attn_output", "next_pre_mix", "x_hc_out") + state_names(module, "reuse") + initial = {n: values[n].clone() for n in state_names(module, "reuse")} + captured = {n: torch.empty_like(values[n]).share_memory_() for n in names} + counts = values["num_tokens"].clone().share_memory_() + config = RunConfig(platform="a5", distributed_config=DistributedConfig(device_ids=list(devices)), + ring_heap=4096 << 20, ring_task_window=131072, ring_dep_pool=131072, + enable_dump_args=int(args.dump_tagged)) + compiler = KernelCompiler(run_config=config, cache_dir=args.build_dir) + program = compiler.compile("v41_replay_c1a_reuse", + module.make_program("reuse", topology.world, epochs=1), attention_epoch=pl.RUNTIME) + with make_segment_worker([program], config, list(values.values())) as worker: + handles = {n: worker.alloc_stacked_tensor(values[n]) for n in C1A_ARGS + if n not in ("num_tokens", "attention_epoch")} + arguments = {**handles, "num_tokens": counts, "attention_epoch": ctypes.c_int32(1)} + worker.run(program.compiled, *(arguments[n] for n in C1A_ARGS), config=config) + for name, destination in captured.items(): + worker.copy_stacked_from(handles[name], destination) + for handle in reversed(list(handles.values())): + worker.free_stacked_tensor(handle) + original = records[args.layer, "attention"] + equivalent = {} + for name, value in captured.items(): + equivalent[name] = torch.equal(value.view(torch.uint8), original["actual"][name].view(torch.uint8)) + print("ORIGINAL REPLAY", name, equivalent[name], flush=True) + expected = {**values, **{n: values[n].clone() for n in captured}} + common.make_golden("reuse", 1, attention_reference=module.reference_attention, + state_names=module.STATE_NAMES["reuse"])(expected) + results = {} + for name, check in module.make_compare("reuse", 1, initial).items(): + results[name] = check(captured[name], expected[name], inputs=expected, + actual_outputs=captured, expected_outputs=expected, rtol=1e-3, atol=1e-3) + print("NATIVE REPLAY", name, results[name], flush=True) + expected_equal = {n: torch.equal(expected[n].view(torch.uint8), v.view(torch.uint8)) + for n, v in original["expected"].items()} + print("ORIGINAL EXPECTED", expected_equal, flush=True) + torch.save(dict(actual=captured, expected={n: expected[n] for n in captured}, + equivalent=equivalent, expected_equal=expected_equal, results=results, + comparison=args.comparison, layer=args.layer), artifact / "comparison.pt") + if not all(equivalent.values()) or not all(expected_equal.values()): + raise AssertionError("isolated replay differs from original; do not interpret cuts yet") + print("ISOLATED REPLAY EQUIVALENCE PASS; native accuracy results remain separate", flush=True) + + +if __name__ == "__main__": + main() From 78530b8bbdc7fdc3f0771d0e34ed2e1526521267 Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Tue, 29 Sep 2026 13:06:33 +0800 Subject: [PATCH 64/78] docs(v41): record C1A continuation and six-layer results --- docs/developer-guide/v41-swa-segment.md | 26 ++++++++++++++++++++++++- 1 file changed, 25 insertions(+), 1 deletion(-) diff --git a/docs/developer-guide/v41-swa-segment.md b/docs/developer-guide/v41-swa-segment.md index 4f9d937e..0c99f045 100644 --- a/docs/developer-guide/v41-swa-segment.md +++ b/docs/developer-guide/v41-swa-segment.md @@ -370,7 +370,7 @@ unused index slots disjoint and shares only the actual producer selection with its consumer. The runtime rejection is resolved; the numerical failures remain. Evidence: `c1a-chain-disjoint/comparison.pt`, `c1a-chain-disjoint.log`, task `task_20260929_113608_405763621400` (exit 1, all four cards released). -Reindex and continuation validation remain pending. This is not accumulated +Later Reindex and continuation results appear below. This is not accumulated full-model or M0 acceptance. The tag-only C1A capture (`b4fa1c91`, serving `14e6a5a`) leaves every saved @@ -422,3 +422,27 @@ device HC implementation. The original gate remains failed; neither weakening it nor changing the reference is part of this diagnostic. Evidence: `c1a-chain-qa32/comparison.pt`, `query-cache-cuts.pt`, `c1a-chain-qa32.log`, `c1a-query-cache-cuts.log` and `c1a-input-rounding-details.log`. + +The six-layer C1A chain (20-25, MoE after every Attention) completes with +**88/90 native checks passing** on the same Q-A candidate and compiler fix. +Full20 HC remains failed. Reuse22 additionally fails Attention at DP0 row 2: +error RMS 0.013177692 versus limit 0.010058112, and peak 0.056640625 versus +limit 0.050295558. Reindex24, Reuse25, all cache/selection integrity checks +and all MoE stages pass. Task `task_20260929_124358_401077729544` exits 1 +and releases all four cards; evidence is `c1a-through-reindex/comparison.pt`. + +C1A Full20/Reuse21 also completes **31+1 continuation with an empty DP +group**, using the same resident cache and communication allocations. +All **60/60 native stage/state checks pass**. Task +`task_20260929_125339_40626412299` exits 0 and releases cards 0-3; evidence +is `c1a-chain-continuation/comparison.pt` and its adjacent task log. +Both cases use serving `14e6a5a`, lib candidate `b80bb8b3` and compiler +`b792bde6`. This remains bounded four-card evidence, not accumulated or M0 +acceptance. + +`tools/replay_v41_c1a_attention.py` isolates a Reuse Attention composite from +a saved single-chunk chain. It loads Attention weights without routed experts, +restores the preceding MoE output and actual producer caches/selections, and +checks bitwise agreement with the original actual and expected stage tensors. +Only after equivalence passes should optional tagged captures be interpreted. +The replay is diagnostic and does not replace independent accumulated gates. From f570f02730920d3ff301e73a4537ad20e2e2d0a8 Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Tue, 29 Sep 2026 13:10:41 +0800 Subject: [PATCH 65/78] test(v41): preserve native gates for isolated kernel candidates --- tools/replay_v41_c1a_attention.py | 13 +++++++++++-- 1 file changed, 11 insertions(+), 2 deletions(-) diff --git a/tools/replay_v41_c1a_attention.py b/tools/replay_v41_c1a_attention.py index e84d2204..a0957d16 100644 --- a/tools/replay_v41_c1a_attention.py +++ b/tools/replay_v41_c1a_attention.py @@ -32,6 +32,8 @@ def main(): parser.add_argument("--build-dir", required=True) parser.add_argument("--artifact-dir", required=True) parser.add_argument("--dump-tagged", action="store_true") + parser.add_argument("--candidate", action="store_true", + help="Gate a changed kernel against the unchanged original reference") args = parser.parse_args() import torch @@ -118,9 +120,16 @@ def main(): torch.save(dict(actual=captured, expected={n: expected[n] for n in captured}, equivalent=equivalent, expected_equal=expected_equal, results=results, comparison=args.comparison, layer=args.layer), artifact / "comparison.pt") - if not all(equivalent.values()) or not all(expected_equal.values()): + if not all(expected_equal.values()): + raise AssertionError("canonical expected values changed; candidate comparison is invalid") + if args.candidate: + if not all(ok for ok, _ in results.values()): + raise AssertionError("candidate fails original native precision checks") + print("CANDIDATE NATIVE REPLAY PASS; not accumulated or M0 acceptance", flush=True) + elif not all(equivalent.values()): raise AssertionError("isolated replay differs from original; do not interpret cuts yet") - print("ISOLATED REPLAY EQUIVALENCE PASS; native accuracy results remain separate", flush=True) + else: + print("ISOLATED REPLAY EQUIVALENCE PASS; native accuracy results remain separate", flush=True) if __name__ == "__main__": From d19ba71f64919469c51fadd955bca5a2b4c70bf0 Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Tue, 29 Sep 2026 13:24:43 +0800 Subject: [PATCH 66/78] docs(v41): record isolated Q-A accumulation findings --- docs/developer-guide/v41-swa-segment.md | 21 +++++++++++++++++++++ 1 file changed, 21 insertions(+) diff --git a/docs/developer-guide/v41-swa-segment.md b/docs/developer-guide/v41-swa-segment.md index 0c99f045..c7761303 100644 --- a/docs/developer-guide/v41-swa-segment.md +++ b/docs/developer-guide/v41-swa-segment.md @@ -446,3 +446,24 @@ restores the preceding MoE output and actual producer caches/selections, and checks bitwise agreement with the original actual and expected stage tensors. Only after equivalence passes should optional tagged captures be interpreted. The replay is diagnostic and does not replace independent accumulated gates. + +Reuse22 isolated replay (`9ad837e` / tag-only lib `da929c70`) reproduces all +actual and expected stage tensors bitwise. The native row-2 failure remains. +Its native-reference Q-A boundary cut reduces that row's relative L2 from +0.01310287 to 4.24e-7. Independent FP64 confirms a Q-A accumulation rounding +error at DP0 row 2, column 308: device -0.0252685546875 versus +reference/FP64-rounded -0.025390625. + +A compensated Q-A accumulation candidate (`e537ef47`) corrects this row, +but still fails the original native Attention gate at DP0 row 31 (RMS +0.011790978 versus limit 0.0093322441). All other 11 checks pass and every +original expected tensor remains bitwise unchanged. This candidate is not +accepted. CPU FP64 analysis finds a separate reference-rounding difference +at row 31, column 722; a device capture is needed before attributing the +new failure. Evidence: `c1a-reuse22-kahan/comparison.pt` and the +`c1a-reuse22-qa-reference-fp64.log` diagnostic. + +Candidate replay uses explicit `--candidate`: it requires unchanged original +expected values and all native checks to pass. Default replay exit zero +means only equivalence to the original run, whose numerical failure may +remain. Neither mode replaces an accumulated acceptance test. From 9a60f59e2e3e0748f15d142435c8e0bad84da078 Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Tue, 29 Sep 2026 13:53:28 +0800 Subject: [PATCH 67/78] docs(v41): close Q-A device and FP64 comparison --- docs/developer-guide/v41-swa-segment.md | 17 +++++++++++++---- 1 file changed, 13 insertions(+), 4 deletions(-) diff --git a/docs/developer-guide/v41-swa-segment.md b/docs/developer-guide/v41-swa-segment.md index c7761303..cae9724d 100644 --- a/docs/developer-guide/v41-swa-segment.md +++ b/docs/developer-guide/v41-swa-segment.md @@ -458,10 +458,19 @@ A compensated Q-A accumulation candidate (`e537ef47`) corrects this row, but still fails the original native Attention gate at DP0 row 31 (RMS 0.011790978 versus limit 0.0093322441). All other 11 checks pass and every original expected tensor remains bitwise unchanged. This candidate is not -accepted. CPU FP64 analysis finds a separate reference-rounding difference -at row 31, column 722; a device capture is needed before attributing the -new failure. Evidence: `c1a-reuse22-kahan/comparison.pt` and the -`c1a-reuse22-qa-reference-fp64.log` diagnostic. +accepted. A subsequent Q-A-only device capture (`e7d72872`) leaves every +actual and expected tensor bitwise unchanged from the untagged candidate. +Captured Q-A relative L2 against independent FP64 is 2.18e-8 in DP0 and +zero in DP1, versus about 2.23e-5 / 2.27e-5 for the native FP32 reference. +At DP0 row 31, column 722, the FP64 sum is 0.09545897599309683; device and +FP64 round to BF16 0.09521484375, while the reference rounds to 0.095703125. +Continuing from captured Q-A reduces that row's output relative L2 from +0.0126360 to 0.00122239. This isolates amplification of a reference/device +accumulation-rounding difference; it does not make the original acceptance +pass. Neither the independent reference nor its gate was replaced. +Evidence: `c1a-reuse22-kahan/comparison.pt`, +`c1a-reuse22-kahan-trace/qa-captures.pt`, `qa-propagation-cuts.pt`, +`c1a-reuse22-kahan-device-fp64.log` and `c1a-kahan-qa-propagation.log`. Candidate replay uses explicit `--candidate`: it requires unchanged original expected values and all native checks to pass. Default replay exit zero From 2c3c6f77eb9d2046a67095d06d2d7d4852af853a Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Tue, 29 Sep 2026 14:20:49 +0800 Subject: [PATCH 68/78] test(v41): cover bounded cross-page prefill continuation --- docs/developer-guide/v41-swa-segment.md | 19 ++++ .../model/deepseek_v41/test_c2a_diagnostic.py | 60 +++++++++++++ tools/validate_v41_c2a_chain.py | 89 +++++++++++++++---- 3 files changed, 151 insertions(+), 17 deletions(-) diff --git a/docs/developer-guide/v41-swa-segment.md b/docs/developer-guide/v41-swa-segment.md index cae9724d..cc911fc7 100644 --- a/docs/developer-guide/v41-swa-segment.md +++ b/docs/developer-guide/v41-swa-segment.md @@ -476,3 +476,22 @@ Candidate replay uses explicit `--candidate`: it requires unchanged original expected values and all native checks to pass. Default replay exit zero means only equivalence to the original run, whose numerical failure may remain. Neither mode replaces an accumulated acceptance test. + +### Bounded cross-page continuation diagnostic + +`validate_v41_c2a_chain.py --repeat-input-chunks N` repeats the saved injected +boundary input at advancing absolute positions, retaining device caches and +communication windows. It accepts 2-16 full chunks, separately recording the +source offsets and request positions. It cannot combine repetition with ragged +counts or `--continue-to-capacity`. The source is not output from a complete +preceding model segment at those positions; this is a state/ABI diagnostic. + +At TP2, five 32-token chunks cross a C1A 128-row compressed-cache page; +nine chunks cross a ratio-2 C2A compressed page. The window cache, KV/index +pools, candidate mask and RoPE extent cover the full diagnostic context. +Native comparisons use the previous chunk's captured cache state and the +current chunk's producer outputs. There is no intermediate host feedback +into the device chain. Host tests cover disjoint successive write slots, +causal cross-page reads, compressed lengths and physical page extents. +Device cross-page results remain pending; this option does not establish +8K prefill, reset/reuse, full-model numerical accuracy or M0 acceptance. diff --git a/tests/unit/model/deepseek_v41/test_c2a_diagnostic.py b/tests/unit/model/deepseek_v41/test_c2a_diagnostic.py index fcc9f492..5c3aefad 100644 --- a/tests/unit/model/deepseek_v41/test_c2a_diagnostic.py +++ b/tests/unit/model/deepseek_v41/test_c2a_diagnostic.py @@ -94,3 +94,63 @@ def test_continuation_repacks_the_next_token_across_tp_slabs(): def test_continuation_cannot_exceed_saved_source(starts): with pytest.raises(ValueError, match="saved causal source"): select_active_state({}, SegmentTopology(tp=2, dp=2), [1, 0], starts) + + +@pytest.mark.parametrize("family,repeats,ratio", [("c1a", 5, 1), ("c2a", 9, 2)]) +def test_repeated_diagnostic_crosses_pages_without_slicing_past_source(family, repeats, ratio): + from tools.validate_v41_c2a_chain import diagnostic_chunks, diagnostic_step + from pypto_serving.model.deepseek_v41.compressed_metadata import prepare_compressed_metadata + from pypto_serving.model.deepseek_v41.swa_metadata import prepare_swa_window_metadata + + topology = SegmentTopology(tp=2, dp=2) + ids = torch.arange(64).reshape(2, 32) + chunks = diagnostic_chunks(topology, [32, 32], repeat_chunks=repeats) + context = repeats * 32 + prior_writes = set() + for counts, starts, source in chunks: + step = diagnostic_step(ids, topology, counts, starts, source, context, family) + assert step.requests[0].token_ids == tuple(range(32)) + assert step.requests[1].token_ids == tuple(range(32, 64)) + window = prepare_swa_window_metadata(step, topology, cache_pages=(context + 127) // 128) + writes = set(window.window_slots[0].tolist()) + assert not prior_writes.intersection(writes) + prior_writes.update(writes) + cm = prepare_compressed_metadata(step, topology, ratio=ratio, compressed_group="cmp", + cache_pages=(context // ratio + 127) // 128, max_requests=1, state_blocks=1) + assert prior_writes == set(range(context)) + assert int(cm.compressed_slots.max()) == context // ratio - 1 + assert int(cm.compressed_slots.max()) >= 128 + assert int(cm.compressed_lens[0, -1]) == context // ratio + assert window.window_indices[0, -1].tolist() == list(range(context - 128, context)) + assert torch.equal(cm.index_block_table[0], cm.index_block_table[1]) + + +@pytest.mark.parametrize("repeat,counts,continuation", [ + (0, [32, 32], False), (17, [32, 32], False), (True, [32, 32], False), + (5, [31, 32], False), (5, [32, 32], True), +]) +def test_repeat_diagnostic_rejects_unbounded_or_ambiguous_inputs(repeat, counts, continuation): + from tools.validate_v41_c2a_chain import diagnostic_chunks + + with pytest.raises(ValueError): + diagnostic_chunks(SegmentTopology(tp=2, dp=2), counts, continuation, repeat) + + +def test_cache_extension_preserves_payload_layout_and_independent_index_allocation(): + from tools.validate_v41_c2a_chain import size_diagnostic_caches + + values = { + "window_cache": torch.empty(4, 1, 128, 1, 512, dtype=torch.uint8), + "compressed_cache": torch.empty(4, 1, 128, 1, 256, dtype=torch.uint8), + "index_cache": torch.empty(4, 1, 128, 1, 64, dtype=torch.uint8), + "candidate_mask": torch.empty(4, 32, 128, dtype=torch.uint8), + "topk_indices": torch.full((4, 32, 512), -1, dtype=torch.int32), + } + topk = values["topk_indices"] + size_diagnostic_caches(values, 160, "c1a") + assert values["window_cache"].shape == (4, 2, 128, 1, 512) + assert values["compressed_cache"].shape == (4, 2, 128, 1, 256) + assert values["index_cache"].shape == (4, 2, 128, 1, 64) + assert values["candidate_mask"].shape == (4, 32, 256) + assert values["topk_indices"] is topk + assert all(t.dtype == torch.uint8 for n, t in values.items() if n != "topk_indices") diff --git a/tools/validate_v41_c2a_chain.py b/tools/validate_v41_c2a_chain.py index bc9d1b96..2ba451be 100644 --- a/tools/validate_v41_c2a_chain.py +++ b/tools/validate_v41_c2a_chain.py @@ -76,13 +76,66 @@ def select_active_state(saved, topology, group_counts, starts=None): return selected, selected_mix +def diagnostic_chunks(topology, counts, continue_to_capacity=False, repeat_chunks=1): + """Separate absolute positions from offsets into the bounded injected source.""" + topology.counts(counts) + if type(repeat_chunks) is not int or not 1 <= repeat_chunks <= 16: + raise ValueError("repeat chunks must be between 1 and 16") + if repeat_chunks > 1: + if continue_to_capacity or any(n != topology.capacity for n in counts): + raise ValueError("repeated input requires full chunks without partial continuation") + return [(list(counts), [i * topology.capacity] * topology.dp, [0] * topology.dp) + for i in range(repeat_chunks)] + chunks = [(list(counts), [0] * topology.dp, [0] * topology.dp)] + if continue_to_capacity: + remaining = [topology.capacity - n if n else 0 for n in counts] + if not any(remaining): + raise ValueError("continuation requires an unfinished nonempty request") + chunks.append((remaining, list(counts), list(counts))) + return chunks + + +def diagnostic_step(ids, topology, counts, starts, source_starts, context_tokens, family): + """Build private full-history pages; repeated IDs remain diagnostic input only.""" + from pypto_serving.model.deepseek_v41.request_state import ForwardStep, RequestSlice + + ratio = 1 if family == "c1a" else 2 + pages = {"window": tuple(range((context_tokens + 127) // 128)), + "cmp": tuple(range((context_tokens // ratio + 127) // 128))} + return ForwardStep("prefill", tuple( + RequestSlice(str(g), g, 0, starts[g], + tuple(row[source_starts[g]:source_starts[g] + counts[g]].tolist()), + context_tokens, pages) + for g, row in enumerate(ids) if counts[g]), 1) + + +def size_diagnostic_caches(values, context_tokens, family): + """Extend only dynamic context axes, preserving lib payload layouts and dtypes.""" + import torch + + window_pages = (context_tokens + 127) // 128 + compressed_pages = (context_tokens // (1 if family == "c1a" else 2) + 127) // 128 + for name in ("window_cache", "window_cache_scale", "compressed_cache", + "compressed_cache_scale", "index_cache", "index_cache_scale"): + if name not in values: + continue + old = values[name] + pages = window_pages if name.startswith("window") else compressed_pages + if pages > old.shape[1]: + values[name] = torch.empty((old.shape[0], pages, *old.shape[2:]), dtype=old.dtype) + if family == "c1a": + old = values["candidate_mask"] + columns = compressed_pages * 128 + if columns > old.shape[-1]: + values["candidate_mask"] = torch.empty((*old.shape[:-1], columns), dtype=old.dtype) + + def prepare(args, topology, module, weight_bundles=None): import torch from golden.spec import TensorSpec from models.deepseek_v4_1_flash.config import FLASH from models.deepseek_v4_1_flash.rope_tables import precompute_rope_tables from pypto_serving.model.deepseek_v41.compressed_metadata import prepare_compressed_metadata - from pypto_serving.model.deepseek_v41.request_state import ForwardStep, RequestSlice from pypto_serving.model.deepseek_v41.swa_metadata import gather_swa_rope_rows, prepare_swa_window_metadata from pypto_serving.model.deepseek_v41.swa_weights import load_prefill_layer_weights @@ -94,9 +147,11 @@ def prepare(args, topology, module, weight_bundles=None): raise ValueError("saved SWA diagnostic must contain exactly the same packed token capacity") group_counts = args.group_counts starts = getattr(args, "starts", [0] * topology.dp) + source_starts = getattr(args, "source_starts", starts) + context_tokens = getattr(args, "context_tokens", topology.capacity) weight_bundles = {} if weight_bundles is None else weight_bundles global_counts, local_counts = topology.counts(group_counts) - residual, mix = select_active_state(saved, topology, group_counts, starts) + residual, mix = select_active_state(saved, topology, group_counts, source_starts) raw = json.loads((Path(args.model_dir) / "config.json").read_text()) family = getattr(args, "family", "c2a") plans = select_plans(raw, family, getattr(args, "last_layer", None)) @@ -110,13 +165,9 @@ def prepare(args, topology, module, weight_bundles=None): if text["rope_scaling"][source] != getattr(FLASH, target): raise ValueError(f"lib/checkpoint compressed RoPE mismatch: {source}") # Private physical pages per DP group, shared only through declared producers. - pages = (topology.capacity + 127) // 128 - step = ForwardStep("prefill", tuple(RequestSlice(str(g), g, 0, starts[g], - tuple(row[starts[g]:starts[g] + group_counts[g]].tolist()), - topology.capacity, {"window": tuple(range(pages)), "cmp": tuple(range(pages))}) - for g, row in enumerate(ids) if group_counts[g]), 1) - tables = precompute_rope_tables(topology.capacity, False) - compressed_tables = precompute_rope_tables(topology.capacity, True) + step = diagnostic_step(ids, topology, group_counts, starts, source_starts, context_tokens, family) + tables = precompute_rope_tables(context_tokens, False) + compressed_tables = precompute_rope_tables(context_tokens, True) fixture = SimpleNamespace(tokens=topology.capacity, requests=1, dp=topology.dp, seed=11, case="mixed", dp_tokens=None, epochs=1, bench=False) attention, moe = {}, {} @@ -129,6 +180,7 @@ def prepare(args, topology, module, weight_bundles=None): specs = module.build_specs(fixture, mode, {}) values = {s.name: s.create_tensor().contiguous() for s in specs if isinstance(s, TensorSpec)} + size_diagnostic_caches(values, context_tokens, family) if plan.layer_id not in weight_bundles: print(f"Loading real checkpoint layer {plan.layer_id}", flush=True) weight_bundles[plan.layer_id] = load_prefill_layer_weights(args.model_dir, plan.layer_id, topology) @@ -193,6 +245,8 @@ def main(): parser.add_argument("--group-counts", help="Comma-separated causal prefix lengths per DP group") parser.add_argument("--continue-to-capacity", action="store_true", help="Run a second chunk of each nonempty request using its resident caches") + parser.add_argument("--repeat-input-chunks", type=int, default=1, + help="Diagnostic only: repeat the saved full input 2-16 times at advancing positions") parser.add_argument("--prepare-only", action="store_true") parser.add_argument("--dump-tagged", action="store_true", help="Preserve lib-tagged kernel arguments after each diagnostic layer") @@ -221,15 +275,14 @@ def main(): from models.deepseek_v4_1_flash import prefill_c1a_sp as module else: module = common - chunks = [(args.group_counts, [0] * topology.dp)] - if args.continue_to_capacity: - remaining = [topology.capacity - n if n else 0 for n in args.group_counts] - if not any(remaining): - raise ValueError("continuation requires an unfinished nonempty request") - chunks.append((remaining, list(args.group_counts))) + chunk_plan = diagnostic_chunks(topology, args.group_counts, args.continue_to_capacity, + args.repeat_input_chunks) + chunks = [(counts, starts) for counts, starts, _ in chunk_plan] + context_tokens = topology.capacity * args.repeat_input_chunks prepared, weights = [], {} - for counts, starts in chunks: - options = SimpleNamespace(**{**vars(args), "group_counts": counts, "starts": starts}) + for counts, starts, source_starts in chunk_plan: + options = SimpleNamespace(**{**vars(args), "group_counts": counts, "starts": starts, + "source_starts": source_starts, "context_tokens": context_tokens}) plans, attention, moe, residual, mix = prepare(options, topology, module, weights) if prepared: for plan in plans: @@ -373,6 +426,8 @@ def upload(values): torch.save({"input_state": str(args.input_state), "family": args.family, "layer_ids": [p.layer_id for p in plans], "injected_boundary_input": True, "group_counts": args.group_counts, "chunks": chunks, "stages": records, + "repeat_input_chunks": args.repeat_input_chunks, "context_tokens": context_tokens, + "source_starts": [source for _, _, source in chunk_plan], "actual_residual": residual, "actual_pre_mix": mix}, artifact / "comparison.pt") if not passed: raise AssertionError(f"{family_label} chain native stage check failed; see comparison.pt") From 4b39b384345e5dc90c0f13425107880d2b5090eb Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Tue, 29 Sep 2026 14:28:17 +0800 Subject: [PATCH 69/78] test(v41): retain empty partitions in cross-page diagnostics --- docs/developer-guide/v41-swa-segment.md | 3 ++- .../model/deepseek_v41/test_c2a_diagnostic.py | 19 ++++++++++++++++++- tools/validate_v41_c2a_chain.py | 4 ++-- 3 files changed, 22 insertions(+), 4 deletions(-) diff --git a/docs/developer-guide/v41-swa-segment.md b/docs/developer-guide/v41-swa-segment.md index cc911fc7..7c28c0b2 100644 --- a/docs/developer-guide/v41-swa-segment.md +++ b/docs/developer-guide/v41-swa-segment.md @@ -483,7 +483,8 @@ remain. Neither mode replaces an accumulated acceptance test. boundary input at advancing absolute positions, retaining device caches and communication windows. It accepts 2-16 full chunks, separately recording the source offsets and request positions. It cannot combine repetition with ragged -counts or `--continue-to-capacity`. The source is not output from a complete +nonempty counts or `--continue-to-capacity`; empty DP groups remain inactive. +The source is not output from a complete preceding model segment at those positions; this is a state/ABI diagnostic. At TP2, five 32-token chunks cross a C1A 128-row compressed-cache page; diff --git a/tests/unit/model/deepseek_v41/test_c2a_diagnostic.py b/tests/unit/model/deepseek_v41/test_c2a_diagnostic.py index 5c3aefad..67dc3d29 100644 --- a/tests/unit/model/deepseek_v41/test_c2a_diagnostic.py +++ b/tests/unit/model/deepseek_v41/test_c2a_diagnostic.py @@ -127,7 +127,7 @@ def test_repeated_diagnostic_crosses_pages_without_slicing_past_source(family, r @pytest.mark.parametrize("repeat,counts,continuation", [ (0, [32, 32], False), (17, [32, 32], False), (True, [32, 32], False), - (5, [31, 32], False), (5, [32, 32], True), + (5, [31, 32], False), (5, [32, 32], True), (5, [0, 0], False), ]) def test_repeat_diagnostic_rejects_unbounded_or_ambiguous_inputs(repeat, counts, continuation): from tools.validate_v41_c2a_chain import diagnostic_chunks @@ -154,3 +154,20 @@ def test_cache_extension_preserves_payload_layout_and_independent_index_allocati assert values["candidate_mask"].shape == (4, 32, 256) assert values["topk_indices"] is topk assert all(t.dtype == torch.uint8 for n, t in values.items() if n != "topk_indices") + + +def test_repeated_single_request_does_not_activate_empty_dp_group(): + from tools.validate_v41_c2a_chain import diagnostic_chunks, diagnostic_step + from pypto_serving.model.deepseek_v41.compressed_metadata import prepare_compressed_metadata + + topology = SegmentTopology(tp=2, dp=2) + counts, starts, source = diagnostic_chunks(topology, [32, 0], repeat_chunks=5)[-1] + assert starts == [128, 0] + step = diagnostic_step(torch.arange(64).reshape(2, 32), topology, counts, starts, source, 160, "c1a") + assert len(step.requests) == 1 and step.requests[0].start == 128 + cm = prepare_compressed_metadata(step, topology, ratio=1, compressed_group="cmp", + cache_pages=2, max_requests=1, state_blocks=1) + assert cm.group_counts == (32, 0) + assert cm.compressed_slots[2:].eq(-1).all() + assert cm.index_block_table[2:].eq(-1).all() + assert cm.compressed_lens[2:].eq(0).all() diff --git a/tools/validate_v41_c2a_chain.py b/tools/validate_v41_c2a_chain.py index 2ba451be..9f8c6c98 100644 --- a/tools/validate_v41_c2a_chain.py +++ b/tools/validate_v41_c2a_chain.py @@ -82,9 +82,9 @@ def diagnostic_chunks(topology, counts, continue_to_capacity=False, repeat_chunk if type(repeat_chunks) is not int or not 1 <= repeat_chunks <= 16: raise ValueError("repeat chunks must be between 1 and 16") if repeat_chunks > 1: - if continue_to_capacity or any(n != topology.capacity for n in counts): + if continue_to_capacity or not any(counts) or any(n not in (0, topology.capacity) for n in counts): raise ValueError("repeated input requires full chunks without partial continuation") - return [(list(counts), [i * topology.capacity] * topology.dp, [0] * topology.dp) + return [(list(counts), [i * count for count in counts], [0] * topology.dp) for i in range(repeat_chunks)] chunks = [(list(counts), [0] * topology.dp, [0] * topology.dp)] if continue_to_capacity: From 52cb5f92c3c2b9dd94aa0d52b357fe3bf593a8c4 Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Tue, 29 Sep 2026 14:33:05 +0800 Subject: [PATCH 70/78] docs(v41): record original Reuse Q-A control result --- docs/developer-guide/v41-swa-segment.md | 10 ++++++++++ 1 file changed, 10 insertions(+) diff --git a/docs/developer-guide/v41-swa-segment.md b/docs/developer-guide/v41-swa-segment.md index 7c28c0b2..21d15295 100644 --- a/docs/developer-guide/v41-swa-segment.md +++ b/docs/developer-guide/v41-swa-segment.md @@ -477,6 +477,16 @@ expected values and all native checks to pass. Default replay exit zero means only equivalence to the original run, whose numerical failure may remain. Neither mode replaces an accumulated acceptance test. +An additional control (`5b875e2a`, Full-only prefill Q-A change) restores +Reuse22's official original Q-A implementation on the same frozen input. +It still fails row 31 with exactly the compensated candidate's error RMS +0.011790978 and limit 0.0093322441. Every saved actual and expected tensor +is bitwise equal to the compensated candidate. Thus reverting Reuse Q-A +does not remove this native gate failure. Task +`task_20260929_140153_32103113893` exited 1 and released cards 0-3; +evidence is `c1a-reuse22-original-qa/comparison.pt` and +`c1a-reuse22-original-kahan-equivalence.log`. No candidate is promoted. + ### Bounded cross-page continuation diagnostic `validate_v41_c2a_chain.py --repeat-input-chunks N` repeats the saved injected From fac8eb1600dfdb9b11e99c9899e1a02bd8d96f3b Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Tue, 29 Sep 2026 14:41:47 +0800 Subject: [PATCH 71/78] docs(v41): record C2A cross-page device validation --- docs/developer-guide/v41-swa-segment.md | 12 ++++++++++++ 1 file changed, 12 insertions(+) diff --git a/docs/developer-guide/v41-swa-segment.md b/docs/developer-guide/v41-swa-segment.md index 21d15295..c4d9a874 100644 --- a/docs/developer-guide/v41-swa-segment.md +++ b/docs/developer-guide/v41-swa-segment.md @@ -506,3 +506,15 @@ into the device chain. Host tests cover disjoint successive write slots, causal cross-page reads, compressed lengths and physical page extents. Device cross-page results remain pending; this option does not establish 8K prefill, reset/reuse, full-model numerical accuracy or M0 acceptance. + +The A5 C2A single-request control at serving `4b39b38`, official lib +`fbe92bfc` and PyPTO `b792bde6` completed nine 32-token chunks. Its +36 Attention/MoE stages passed **216/216 native checks**; the last chunk +covered positions 256-287. Saved-state audit confirmed that the second +compressed page contains published payload, the other DP group's cache +is untouched, and all earlier chunk caches remained resident. Task +`task_20260929_143008_89019724616` exited zero and released devices 0-3. +Evidence is `c2a-crosspage-single-request/comparison.pt`, its adjacent +task log and `c2a-crosspage-audit.log`. The input was repeated from the +saved boundary state, so this is a physical-page continuation and native +state check, not an independent accumulated model run. From 0f3fb63d37d32844c30e91d2af730625abb1aa65 Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Tue, 29 Sep 2026 14:42:09 +0800 Subject: [PATCH 72/78] docs(v41): distinguish C1A crossing from validated C2A --- docs/developer-guide/v41-swa-segment.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/developer-guide/v41-swa-segment.md b/docs/developer-guide/v41-swa-segment.md index c4d9a874..7f8d3946 100644 --- a/docs/developer-guide/v41-swa-segment.md +++ b/docs/developer-guide/v41-swa-segment.md @@ -504,7 +504,7 @@ Native comparisons use the previous chunk's captured cache state and the current chunk's producer outputs. There is no intermediate host feedback into the device chain. Host tests cover disjoint successive write slots, causal cross-page reads, compressed lengths and physical page extents. -Device cross-page results remain pending; this option does not establish +C1A cross-page device results remain pending; this option does not establish 8K prefill, reset/reuse, full-model numerical accuracy or M0 acceptance. The A5 C2A single-request control at serving `4b39b38`, official lib From c56223197d63794c161fc528bac2597e44ca4f88 Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Tue, 29 Sep 2026 14:49:50 +0800 Subject: [PATCH 73/78] docs(v41): record C1A cross-page device validation --- docs/developer-guide/v41-swa-segment.md | 16 ++++++++++++++-- 1 file changed, 14 insertions(+), 2 deletions(-) diff --git a/docs/developer-guide/v41-swa-segment.md b/docs/developer-guide/v41-swa-segment.md index 7f8d3946..fa077caa 100644 --- a/docs/developer-guide/v41-swa-segment.md +++ b/docs/developer-guide/v41-swa-segment.md @@ -504,8 +504,8 @@ Native comparisons use the previous chunk's captured cache state and the current chunk's producer outputs. There is no intermediate host feedback into the device chain. Host tests cover disjoint successive write slots, causal cross-page reads, compressed lengths and physical page extents. -C1A cross-page device results remain pending; this option does not establish -8K prefill, reset/reuse, full-model numerical accuracy or M0 acceptance. +This option does not establish 8K prefill, reset/reuse, full-model numerical +accuracy or M0 acceptance. The A5 C2A single-request control at serving `4b39b38`, official lib `fbe92bfc` and PyPTO `b792bde6` completed nine 32-token chunks. Its @@ -518,3 +518,15 @@ Evidence is `c2a-crosspage-single-request/comparison.pt`, its adjacent task log and `c2a-crosspage-audit.log`. The input was repeated from the saved boundary state, so this is a physical-page continuation and native state check, not an independent accumulated model run. + +The A5 C1A Full20/Reuse21 control at serving `52cb5f9`, diagnostic +Full-only Q-A lib `5b875e2a` and the same PyPTO revision completed five +32-token chunks. All **150/150 native checks** passed across 20 stages; +the last chunk covered positions 128-159. Its second window, compressed +KV and index pages contain published payload, while the inactive DP group's +caches remain untouched. Task `task_20260929_144014_160907026376` exited +zero and released devices 0-3. Evidence is +`c1a-crosspage-single-request/comparison.pt`, its task log and +`c1a-crosspage-audit.log`. This one-active-DP diagnostic does not supersede +the prior two-active-DP Full20 HC and Reuse22 failures, nor prove that +layers 0-19 supplied the injected boundary input. From 5e7d5647377778dba588b98de0dc10ab5159fb33 Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Tue, 29 Sep 2026 15:53:14 +0800 Subject: [PATCH 74/78] Add V4.1 decode backbone and final output boundaries --- docs/developer-guide/deepseek-v41-entry.md | 19 ++- pypto_serving/model/deepseek_v41/composite.py | 4 +- .../model/deepseek_v41/decode_backbone.py | 68 +++++++++ .../model/deepseek_v41/final_output.py | 141 ++++++++++++++++++ .../model/deepseek_v41/npu_runner.py | 4 + .../model/deepseek_v41/swa_segment.py | 10 ++ .../deepseek_v41/test_composite_dispatch.py | 23 ++- .../deepseek_v41/test_decode_backbone.py | 63 ++++++++ .../model/deepseek_v41/test_final_output.py | 40 +++++ tools/compile_v41_decode_backbone.py | 27 ++++ tools/compile_v41_serving_output.py | 28 ++++ 11 files changed, 417 insertions(+), 10 deletions(-) create mode 100644 pypto_serving/model/deepseek_v41/decode_backbone.py create mode 100644 pypto_serving/model/deepseek_v41/final_output.py create mode 100644 tests/unit/model/deepseek_v41/test_decode_backbone.py create mode 100644 tests/unit/model/deepseek_v41/test_final_output.py create mode 100644 tools/compile_v41_decode_backbone.py create mode 100644 tools/compile_v41_serving_output.py diff --git a/docs/developer-guide/deepseek-v41-entry.md b/docs/developer-guide/deepseek-v41-entry.md index c18e1242..08144ccf 100644 --- a/docs/developer-guide/deepseek-v41-entry.md +++ b/docs/developer-guide/deepseek-v41-entry.md @@ -188,12 +188,19 @@ producer-owned compressed caches have bounded integration evidence; see [the segment validation record](v41-swa-segment.md) for tested modes and precision limits. -The default complete-model adapter is still unavailable. Decode `stage="block"` -remains disabled in this lib revision, and the final existing-state HC+Norm -composition is missing: `boundary_embed_to_norm` repacks embeddings and skips -the backbone, so it cannot consume the final layer's state. All-mode device -validation, reset/recovery and complete generation remain integration work. -Bounded half-layer dispatch does not establish full-model or M0 acceptance. +The default complete-model adapter is still unavailable. At lib `f1d35101`, +`decode_layer` composes all six attention modes with MoE, and `decode_fwd` +expands the 40-layer backbone. Serving's `decode_backbone.py` binds that +forward directly; `final_output.py` composes the existing HC-head, RMSNorm and +LM-head operators on the final residual/pre-mix instead of repacking embeddings. +Both serving entries passed A5 code generation with PTOAS 0.65, but neither has +a device run or real-checkpoint full-model validation. The lib decode-forward +fixture itself is compile-only, and #1385's A5 MoE precision job failed. +Production weight/cache allocation, request reset and the concrete +`CompositeBindings` registration remain unfinished. The bounded two-layer SWA +reference also still fails the provisional accumulated pre-mix gate in 3/256 +elements (`rtol=0.01`, `atol=0.001`); the reference and gate are unchanged. +This is a tracked precision issue, not evidence that output generation works. ## Request state and serving lifecycle diff --git a/pypto_serving/model/deepseek_v41/composite.py b/pypto_serving/model/deepseek_v41/composite.py index 05eda6f1..4d5ea6e6 100644 --- a/pypto_serving/model/deepseek_v41/composite.py +++ b/pypto_serving/model/deepseek_v41/composite.py @@ -68,11 +68,13 @@ class CompositeBindings: cache_groups: tuple = () input_layout: str = "tp_replicated" output_layout: str = "tp_replicated" + decode_backbone: Callable | None = None def require(self, layers: tuple[LayerPlan, ...], placement: RankPlacement) -> None: if not self.revision: raise ValueError("composite bindings must identify the validated lib revision") - missing = sorted({f"{phase}/{layer.mode}" for phase in ("prefill", "decode") + phases = ("prefill",) if callable(self.decode_backbone) else ("prefill", "decode") + missing = sorted({f"{phase}/{layer.mode}" for phase in phases for layer in layers if not callable(self.entries.get((phase, layer.mode)))}) if missing: raise MissingCompositeInterface("missing complete layer composites: " + ", ".join(missing)) diff --git a/pypto_serving/model/deepseek_v41/decode_backbone.py b/pypto_serving/model/deepseek_v41/decode_backbone.py new file mode 100644 index 00000000..5d91c4ba --- /dev/null +++ b/pypto_serving/model/deepseek_v41/decode_backbone.py @@ -0,0 +1,68 @@ +# Copyright (c) PyPTO Contributors. +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of +# CANN Open Software License Agreement Version 2.0 (the "License"). +# Please refer to the License for details. You may not use this file except in compliance with the License. +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. +# See LICENSE in the root of the software repository for the full text of the License. +# ----------------------------------------------------------------------------------------------------------- +"""Serving dispatch boundary for lib's complete V4.1 decode backbone.""" + +import ctypes +import importlib + +from .composite import LayerState +from .swa_segment import load_segment_modules + + +def compile_decode_backbone(compiler, lib_root, topology): + """Compile the existing 40-layer lib entry without rebuilding model operators.""" + import pypto.language as pl + + load_segment_modules(lib_root, topology) + module = importlib.import_module("models.deepseek_v4_1_flash.decode_fwd") + config = importlib.import_module("models.deepseek_v4_1_flash.config") + if module.N_LAYERS != config.FLASH.num_hidden_layers or module.N_LAYERS != 40: + raise ValueError("decode backbone and checkpoint layer schedules disagree") + program = compiler.compile( + "v41_decode_fwd", module.l3_decode_fwd, attention_num_tokens=pl.RUNTIME, + ) + return program, tuple(module.l3_decode_fwd.param_names) + + +class DecodeBackbone: + """Run one device-resident decode step with explicit lib-owned state/cache ABI.""" + + def __init__(self, worker, program, param_names, run_config): + self.worker = worker + self.program = program + self.param_names = tuple(param_names) + self.run_config = run_config + self.failed = False + if len(set(self.param_names)) != len(self.param_names): + raise ValueError("duplicate decode ABI parameter") + if not {"x_hc", "pre_mix", "attention_num_tokens"} <= set(self.param_names): + raise ValueError("decode ABI must expose residual, delayed mix and active count") + + def run(self, state: LayerState, arguments, *, active_tokens: int) -> LayerState: + if self.failed: + raise RuntimeError("decode dispatch failed; close the worker before retrying") + if state.layout != "tp_local_token": + raise ValueError("decode backbone requires TP-local residual and pre_mix") + if type(active_tokens) is not int or not 0 < active_tokens <= 192: + raise ValueError("decode requires a compact active batch of 1..192 tokens") + reserved = {"x_hc", "pre_mix", "attention_num_tokens"} + if reserved & arguments.keys(): + raise ValueError("decode state/count must come from this dispatch, not argument buffers") + missing = set(self.param_names) - reserved - arguments.keys() + if missing: + raise ValueError("missing decode ABI arguments: " + ", ".join(sorted(missing))) + bound = dict(arguments, x_hc=state.residual, pre_mix=state.pre_mix, + attention_num_tokens=ctypes.c_int32(active_tokens)) + try: + self.worker.run(self.program.compiled, *(bound[name] for name in self.param_names), + config=self.run_config) + except BaseException: + self.failed = True + raise + return state diff --git a/pypto_serving/model/deepseek_v41/final_output.py b/pypto_serving/model/deepseek_v41/final_output.py new file mode 100644 index 00000000..a07a47ee --- /dev/null +++ b/pypto_serving/model/deepseek_v41/final_output.py @@ -0,0 +1,141 @@ +# Copyright (c) PyPTO Contributors. +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of +# CANN Open Software License Agreement Version 2.0 (the "License"). +# Please refer to the License for details. You may not use this file except in compliance with the License. +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. +# See LICENSE in the root of the software repository for the full text of the License. +# ----------------------------------------------------------------------------------------------------------- +"""Final-state output composition using existing V4.1 lib operators.""" + +import ctypes +import importlib +import sys + +from .composite import LayerState +from .swa_segment import load_segment_modules + + +def make_final_output_program(lib_root, topology): + """Collapse the last delayed HC state, normalize, project and sample.""" + import pypto.language as pl + import pypto.language.distributed as pld + + load_segment_modules(lib_root, topology) + package = "models.deepseek_v4_1_flash" + config = importlib.import_module(package + ".config") + hc_head = importlib.import_module(package + ".hc_head").hc_head + rms_norm = importlib.import_module(package + ".rmsnorm").rms_norm + + # The LM-head module parses DP separately from the backbone's EP axis. + old_argv = sys.argv[:] + try: + sys.argv = [old_argv[0], "--tp", str(topology.tp), "--dp", str(topology.dp)] + lm = importlib.import_module(package + ".lm_head") + finally: + sys.argv = old_argv + if (lm.TP_SIZE, lm.DP_SIZE, lm.WORLD_SIZE) != (topology.tp, topology.dp, topology.world): + raise ValueError("LM head was imported for a different TP/DP topology") + + world = topology.world + local = config.LOCAL_T_DYN + hc = config.HC_MULT + d = config.D + rows = lm.MAX_LOGIT_ROWS + vocab = lm.VOCAB + vocab_per_tp = lm.VOCAB_PER_TP + sampled_pad = lm.SAMPLED_IDS_PAD + tp = topology.tp + group_logit_rows = lm.GROUP_LOGIT_ROWS + head_entry = lm.lm_head_with_sampling_test + + @pl.jit + def final_norm_rank( + x_hc: pl.Tensor[[local, hc, d], pl.FP32], + pre_mix: pl.Tensor[[local, hc], pl.FP32], + norm_weight: pl.Tensor[[d], pl.BF16], + normed: pl.Tensor[[local, d], pl.BF16], + ): + hidden = pl.create_tensor([pl.tensor.dim(x_hc, 0), d], dtype=pl.BF16) + hc_head(x_hc, pre_mix, hidden) + rms_norm(hidden, norm_weight, normed) + + @pl.jit.host + def l3_final_output( + x_hc: pl.Tensor[[world, local, hc, d], pl.FP32], + pre_mix: pl.Tensor[[world, local, hc], pl.FP32], + norm_weight: pl.Tensor[[world, d], pl.BF16], + head_weight: pl.Tensor[[world, vocab_per_tp, d], pl.BF16], + logit_row_indices: pl.Tensor[[world, rows], pl.INT32], + normed: pl.Out[pl.Tensor[[world, local, d], pl.BF16]], + logits: pl.Out[pl.Tensor[[world, rows, vocab], pl.FP32]], + sampled_ids: pl.Out[pl.Tensor[[world, rows, sampled_pad], pl.INT32]], + done_epoch: pl.Scalar[pl.INT32], + ): + hidden_window_buf = pld.alloc_window_buffer(group_logit_rows * d * 2) + logits_window_buf = pld.alloc_window_buffer(rows * vocab * 4) + hidden_done_buf = pld.alloc_window_buffer(tp * 4) + logits_done_buf = pld.alloc_window_buffer(tp * 4) + for rank in pl.range(pld.world_size()): + final_norm_rank(x_hc[rank], pre_mix[rank], norm_weight[rank], normed[rank], device=rank) + for rank in pl.range(pld.world_size()): + hidden_window = pld.window(hidden_window_buf, [group_logit_rows, d], dtype=pl.BF16) + hidden_done = pld.window(hidden_done_buf, [tp, 1], dtype=pl.INT32) + logits_window = pld.window(logits_window_buf, [rows, vocab], dtype=pl.FP32) + logits_done = pld.window(logits_done_buf, [tp, 1], dtype=pl.INT32) + head_entry( + normed[rank], head_weight[rank], logit_row_indices[rank], logits[rank], sampled_ids[rank], + hidden_window, hidden_done, logits_window, logits_done, + rank // tp * tp, rank % tp, done_epoch, + device=rank, + ) + + return l3_final_output + + +def compile_final_output(compiler, lib_root, topology): + import pypto.language as pl + + entry = make_final_output_program(lib_root, topology) + return compiler.compile("v41_final_output", entry, done_epoch=pl.RUNTIME), tuple(entry.param_names) + + +class FinalOutput: + """Dispatch final logits while retaining one completion epoch per worker.""" + + def __init__(self, worker, program, param_names, run_config): + self.worker = worker + self.program = program + self.param_names = tuple(param_names) + self.run_config = run_config + self.epoch = 0 + self.failed = False + required = {"x_hc", "pre_mix", "norm_weight", "head_weight", "logit_row_indices", + "normed", "logits", "sampled_ids", "done_epoch"} + if set(self.param_names) != required or len(self.param_names) != len(required): + raise ValueError("final output ABI differs from the serving composition") + + def run(self, state: LayerState, arguments): + if self.failed: + raise RuntimeError("output dispatch failed; close the worker before retrying") + if state.layout != "tp_local_token": + raise ValueError("final output requires TP-local residual and pre_mix") + reserved = {"x_hc", "pre_mix", "done_epoch"} + if reserved & arguments.keys(): + raise ValueError("final state and epoch must come from this dispatch") + missing = set(self.param_names) - reserved - arguments.keys() + if missing: + raise ValueError("missing output ABI arguments: " + ", ".join(sorted(missing))) + if self.epoch >= (2**31 - 1) // 16: + raise OverflowError("output completion epoch exhausted; recreate worker") + next_epoch = self.epoch + 1 + bound = dict(arguments, x_hc=state.residual, pre_mix=state.pre_mix, + done_epoch=ctypes.c_int32(next_epoch)) + try: + self.worker.run(self.program.compiled, *(bound[name] for name in self.param_names), + config=self.run_config) + except BaseException: + self.failed = True + raise + self.epoch = next_epoch + return arguments["logits"], arguments["sampled_ids"] diff --git a/pypto_serving/model/deepseek_v41/npu_runner.py b/pypto_serving/model/deepseek_v41/npu_runner.py index 2be780ae..cab231c8 100644 --- a/pypto_serving/model/deepseek_v41/npu_runner.py +++ b/pypto_serving/model/deepseek_v41/npu_runner.py @@ -137,6 +137,10 @@ def _run_layers(self, step, embeddings): state = self.bindings.initialize(embeddings, step, self.resources) self.bindings.wait(self.resources) self._check_state(state) + if step.phase == "decode" and self.bindings.decode_backbone is not None: + state = self.bindings.decode_backbone(state, step, self.resources, plans) + self.bindings.wait(self.resources) + return self._check_state(state) for layer in self.plan.layers: # The adapter chooses bounded staging or residency; payloads stay packed. weights = self.bindings.prepare_weights(plans, layer, self.resources) diff --git a/pypto_serving/model/deepseek_v41/swa_segment.py b/pypto_serving/model/deepseek_v41/swa_segment.py index 74366fda..07821863 100644 --- a/pypto_serving/model/deepseek_v41/swa_segment.py +++ b/pypto_serving/model/deepseek_v41/swa_segment.py @@ -40,6 +40,16 @@ class SegmentTopology: dp: int = 2 local_capacity: int = 16 + @classmethod + def for_decode(cls, *, tp: int = 4, dp: int = 2): + """Match lib's padded local MoE extent for a full decode batch.""" + if type(tp) is not int or tp not in (1, 2, 4, 8): + raise ValueError("unsupported decode TP size") + decode_tokens = 32 * (5 + 1) + row_tile = 16 + local_rows = (decode_tokens + tp - 1) // tp + return cls(tp=tp, dp=dp, local_capacity=(local_rows + row_tile - 1) // row_tile * row_tile) + def __post_init__(self): if self.tp not in (1, 2, 4, 8) or type(self.tp) is not int: raise ValueError("unsupported TP size") diff --git a/tests/unit/model/deepseek_v41/test_composite_dispatch.py b/tests/unit/model/deepseek_v41/test_composite_dispatch.py index 34acc7be..8529d23d 100644 --- a/tests/unit/model/deepseek_v41/test_composite_dispatch.py +++ b/tests/unit/model/deepseek_v41/test_composite_dispatch.py @@ -17,7 +17,7 @@ from pypto_serving.model.deepseek_v41.request_state import ForwardStep, RequestSlice -def make_runner(*, layout="tp_local_token", entry=None, output=None): +def make_runner(*, layout="tp_local_token", entry=None, output=None, decode_backbone=None): events = [] layers = tuple(LayerPlan(i, mode, None if i == 0 else 1, None if i == 0 else 1, None) for i, mode in enumerate(("swa", "c2a_full", "c2a_reuse"))) @@ -32,10 +32,12 @@ def prepare(plans, layer, resources): events.append(("weights", layer.layer_id)) return layer.layer_id bindings = CompositeBindings(revision="recording-test-adapter", input_layout=layout, output_layout=layout, - entries={(phase, layer.mode): entry or call for phase in ("prefill", "decode") for layer in layers}, + entries={(phase, layer.mode): entry or call + for phase in (("prefill",) if decode_backbone else ("prefill", "decode")) for layer in layers}, initialize=initialize, output=output or (lambda *a: None), allocate=lambda *a: (object(), 8), prepare_weights=prepare, reset_request=lambda *a: events.append(("reset", a[1])), - wait=lambda *a: events.append(("wait",)), close=lambda *a: events.append(("close",))) + wait=lambda *a: events.append(("wait",)), close=lambda *a: events.append(("close",)), + decode_backbone=decode_backbone) config = SimpleNamespace(hidden_size=4, vocab_size=16, max_position_embeddings=128) plan = SimpleNamespace(placement=RankPlacement(0), layers=layers, weights=SimpleNamespace(config=config)) plan.for_rank = lambda rank: SimpleNamespace(placement=RankPlacement(rank)) @@ -67,6 +69,21 @@ def test_wrong_layout_fails_before_next_layer(): assert [e for e in events if e[0] == "weights"] == [("weights", 0)] +def test_decode_backbone_runs_once_with_rank_plans_and_no_layer_dispatch(): + called = [] + + def backbone(state, step, resources, plans): + called.append((step.phase, tuple(plan.placement.rank for plan in plans))) + return state + + runner, events = make_runner(decode_backbone=backbone) + step = ForwardStep("decode", (RequestSlice("a", 1, 0, 7, (2,), 8, {}),), 4) + runner._run_layers(step, torch.ones(1, 4, dtype=torch.bfloat16)) + assert called == [("decode", tuple(range(8)))] + assert not [event for event in events if event[0] in ("layer", "weights")] + assert events[-1] == ("wait",) + + def test_output_uses_shared_greedy_sampler_and_feeds_decode(): from pypto_serving.config.types import SamplingParams from pypto_serving.model.common.executor.sampler import Sampler diff --git a/tests/unit/model/deepseek_v41/test_decode_backbone.py b/tests/unit/model/deepseek_v41/test_decode_backbone.py new file mode 100644 index 00000000..a1766ff7 --- /dev/null +++ b/tests/unit/model/deepseek_v41/test_decode_backbone.py @@ -0,0 +1,63 @@ +from types import SimpleNamespace + +import pytest + +from pypto_serving.model.deepseek_v41.composite import LayerState +from pypto_serving.model.deepseek_v41.decode_backbone import DecodeBackbone +from pypto_serving.model.deepseek_v41.swa_segment import SegmentTopology + + +class RecordingWorker: + def __init__(self, *, fail=False): + self.calls = [] + self.fail = fail + + def run(self, program, *args, config): + self.calls.append((program, args, config)) + if self.fail: + raise RuntimeError("device dispatch failed") + + +def test_decode_backbone_binds_resident_state_and_rejects_missing_cache(): + worker = RecordingWorker() + program = SimpleNamespace(compiled=object()) + decoder = DecodeBackbone( + worker, program, ("x_hc", "pre_mix", "window_cache", "attention_num_tokens"), "config", + ) + state = LayerState(object(), object(), "tp_local_token") + + with pytest.raises(ValueError, match="window_cache"): + decoder.run(state, {}, active_tokens=1) + assert worker.calls == [] + + assert decoder.run(state, {"window_cache": "resident-cache"}, active_tokens=7) is state + _, args, config = worker.calls[0] + assert args[:3] == (state.residual, state.pre_mix, "resident-cache") + assert args[3].value == 7 + assert config == "config" + + +def test_decode_backbone_rejects_invalid_batch_and_poisoned_worker(): + worker = RecordingWorker(fail=True) + decoder = DecodeBackbone(worker, SimpleNamespace(compiled="program"), + ("x_hc", "pre_mix", "attention_num_tokens"), "config") + state = LayerState(object(), object(), "tp_local_token") + + with pytest.raises(ValueError, match="compact active batch"): + decoder.run(state, {}, active_tokens=193) + with pytest.raises(ValueError, match="TP-local"): + decoder.run(LayerState(object(), object()), {}, active_tokens=1) + with pytest.raises(ValueError, match="not argument buffers"): + decoder.run(state, {"pre_mix": object()}, active_tokens=1) + with pytest.raises(RuntimeError, match="device dispatch failed"): + decoder.run(state, {}, active_tokens=1) + with pytest.raises(RuntimeError, match="close the worker"): + decoder.run(state, {}, active_tokens=1) + assert len(worker.calls) == 1 + + +@pytest.mark.parametrize("tp,expected", [(2, 96), (4, 48), (8, 32)]) +def test_decode_topology_uses_lib_padded_moe_rows(tp, expected): + topology = SegmentTopology.for_decode(tp=tp, dp=8 // tp) + assert topology.local_capacity == expected + assert topology.capacity >= 192 diff --git a/tests/unit/model/deepseek_v41/test_final_output.py b/tests/unit/model/deepseek_v41/test_final_output.py new file mode 100644 index 00000000..540b30d1 --- /dev/null +++ b/tests/unit/model/deepseek_v41/test_final_output.py @@ -0,0 +1,40 @@ +from types import SimpleNamespace + +import pytest + +from pypto_serving.model.deepseek_v41.composite import LayerState +from pypto_serving.model.deepseek_v41.final_output import FinalOutput + + +PARAMS = ("x_hc", "pre_mix", "norm_weight", "head_weight", "logit_row_indices", + "normed", "logits", "sampled_ids", "done_epoch") + + +class RecordingWorker: + def __init__(self): + self.calls = [] + + def run(self, program, *args, config): + self.calls.append((program, args, config)) + + +def test_output_uses_final_state_and_advances_epoch(): + worker = RecordingWorker() + output = FinalOutput(worker, SimpleNamespace(compiled="program"), PARAMS, "config") + state = LayerState("final-residual", "final-mix", "tp_local_token") + arguments = {name: object() for name in PARAMS if name not in ("x_hc", "pre_mix", "done_epoch")} + + for epoch in (1, 2): + assert output.run(state, arguments) == (arguments["logits"], arguments["sampled_ids"]) + _, values, _ = worker.calls[-1] + assert values[:2] == ("final-residual", "final-mix") + assert values[-1].value == epoch + + +def test_output_rejects_missing_weights_before_launch(): + worker = RecordingWorker() + output = FinalOutput(worker, SimpleNamespace(compiled="program"), PARAMS, "config") + state = LayerState(object(), object(), "tp_local_token") + with pytest.raises(ValueError, match="head_weight"): + output.run(state, {"norm_weight": object()}, ) + assert worker.calls == [] diff --git a/tools/compile_v41_decode_backbone.py b/tools/compile_v41_decode_backbone.py new file mode 100644 index 00000000..a2e60e77 --- /dev/null +++ b/tools/compile_v41_decode_backbone.py @@ -0,0 +1,27 @@ +"""Codegen-only check of the 40-layer V4.1 decode serving boundary.""" + +import argparse + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--lib-root", required=True) + parser.add_argument("--build-dir", default="build_output/v41-serving-decode") + args = parser.parse_args() + + from pypto.ir import DistributedConfig + from pypto.runtime import RunConfig + + from pypto_serving.model.common.compiler.compiler import KernelCompiler + from pypto_serving.model.deepseek_v41.decode_backbone import compile_decode_backbone + from pypto_serving.model.deepseek_v41.swa_segment import SegmentTopology + + topology = SegmentTopology.for_decode() + config = RunConfig(platform="a5", distributed_config=DistributedConfig(device_ids=list(range(topology.world)))) + compiler = KernelCompiler(run_config=config, cache_dir=args.build_dir) + _, arguments = compile_decode_backbone(compiler, args.lib_root, topology) + print(f"CODEGEN PASS: 40-layer decode, {len(arguments)} ABI arguments") + + +if __name__ == "__main__": + main() diff --git a/tools/compile_v41_serving_output.py b/tools/compile_v41_serving_output.py new file mode 100644 index 00000000..634e45a9 --- /dev/null +++ b/tools/compile_v41_serving_output.py @@ -0,0 +1,28 @@ +"""Codegen-only check of the V4.1 final-state output adapter.""" + +import argparse + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--lib-root", required=True) + parser.add_argument("--build-dir", default="build_output/v41-serving-output") + args = parser.parse_args() + + from pypto.ir import DistributedConfig + from pypto.runtime import RunConfig + + from pypto_serving.model.common.compiler.compiler import KernelCompiler + from pypto_serving.model.deepseek_v41.final_output import compile_final_output + from pypto_serving.model.deepseek_v41.swa_segment import SegmentTopology + + topology = SegmentTopology.for_decode() + config = RunConfig(platform="a5", distributed_config=DistributedConfig(device_ids=list(range(topology.world)))) + compiler = KernelCompiler(run_config=config, cache_dir=args.build_dir) + _, arguments = compile_final_output(compiler, args.lib_root, topology) + assert len(arguments) == 9 + print("CODEGEN PASS: final residual/pre_mix -> HC head -> RMSNorm -> LM head -> greedy") + + +if __name__ == "__main__": + main() From a28ce2cc50785c3ffcd91b02bebdf9096f600b4f Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Tue, 29 Sep 2026 16:15:30 +0800 Subject: [PATCH 75/78] Map selective V4.1 weights into decode source slots --- .../model/deepseek_v41/decode_weights.py | 127 ++++++++++++++++++ .../model/deepseek_v41/test_decode_weights.py | 82 +++++++++++ .../model/deepseek_v41/test_weight_loader.py | 21 +++ 3 files changed, 230 insertions(+) create mode 100644 pypto_serving/model/deepseek_v41/decode_weights.py create mode 100644 tests/unit/model/deepseek_v41/test_decode_weights.py diff --git a/pypto_serving/model/deepseek_v41/decode_weights.py b/pypto_serving/model/deepseek_v41/decode_weights.py new file mode 100644 index 00000000..47e85aef --- /dev/null +++ b/pypto_serving/model/deepseek_v41/decode_weights.py @@ -0,0 +1,127 @@ +# Copyright (c) PyPTO Contributors. +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of +# CANN Open Software License Agreement Version 2.0 (the "License"). +# Please refer to the License for details. You may not use this file except in compliance with the License. +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. +# See LICENSE in the root of the software repository for the full text of the License. +# ----------------------------------------------------------------------------------------------------------- +"""Bounded checkpoint slices for lib's 40-layer decode weight ABI. + +Each yielded tensor has an EP-leading axis and occupies one layer or producer +slot in ``decode_fwd.l3_decode_fwd``. The caller places slices into resident +buffers; this iterator never retains the whole checkpoint on the host. +""" + +from dataclasses import dataclass +import json +from pathlib import Path + +import torch + +from .execution_plan import plan_layers +from .swa_weights import load_prefill_layer_weights + + +_ATTENTION = { + "hc_attn_fn": "hc_attn_fn", "hc_attn_scale": "hc_attn_scale", + "hc_attn_base": "hc_attn_base", "attn_norm_weight": "attn_norm_weight", + "wq_a": "wq_a", "wq_a_scale": "wq_a_scale", "q_norm_weight": "q_norm_weight", + "wq_b": "wq_b", "wq_b_scale": "wq_b_scale", "wkv": "wkv", + "wkv_scale": "wkv_scale", "kv_norm_weight": "kv_norm_weight", + "attn_sink": "attn_sink", "wo_a": "wo_a", "wo_b": "wo_b", + "wo_b_scale": "wo_b_scale", +} +_MOE = { + "hc_ffn_fn": "hc_ffn_fn", "hc_ffn_scale": "hc_ffn_scale", + "hc_ffn_base": "hc_ffn_base", "ffn_norm_weight": "norm_weight", + "gate_weight": "gate_weight", "correction_bias": "correction_bias", + "routed_w1": "routed_w1", "routed_w1_scale": "routed_w1_scale", + "routed_w2": "routed_w2", "routed_w2_scale": "routed_w2_scale", + "routed_w3": "routed_w3", "routed_w3_scale": "routed_w3_scale", + "shared_w1": "shared_w1", "shared_w1_scale": "shared_w1_scale", + "shared_w2": "shared_w2", "shared_w2_scale": "shared_w2_scale", + "shared_w3": "shared_w3", "shared_w3_scale": "shared_w3_scale", +} + + +@dataclass(frozen=True) +class DecodeWeightSlice: + """One EP-stacked payload and its destination slot in the decode ABI.""" + + name: str + slot: int + value: torch.Tensor + + +def bind_decode_layer_weights(layer, attention, moe, *, kv_sources, index_sources, c2a_sources): + """Map one loaded layer to the lib's layer/source axes without copying data.""" + if not attention or not moe or "mxfp4_pair_lut" not in moe: + raise ValueError("decode requires complete Attention and packed-FP4 MoE weights") + result = [DecodeWeightSlice(target, layer.layer_id, attention[source]) + for target, source in _ATTENTION.items()] + result += [DecodeWeightSlice(target, layer.layer_id, moe[source]) + for target, source in _MOE.items()] + if layer.layer_id == 0: + result.append(DecodeWeightSlice("mxfp4_pair_lut", 0, moe["mxfp4_pair_lut"])) + if layer.layer_id in kv_sources: + kv_slot = kv_sources.index(layer.layer_id) + result.append(DecodeWeightSlice("compressor_norm_weight", kv_slot, + attention["compressor_norm_weight"])) + if layer.mode == "c2a_full": + slot = c2a_sources.index(layer.layer_id) + result.extend(( + DecodeWeightSlice("c2a_compressor_wkv", slot, attention["compressor_wkv"]), + DecodeWeightSlice("c2a_compressor_wgate", slot, attention["compressor_wgate"]), + )) + elif layer.mode == "c1a_full": + result.append(DecodeWeightSlice("c1a_compressor_wkv", 0, attention["compressor_wkv"])) + else: + raise ValueError("decode KV source must be a C1A/C2A Full layer") + # The decode ABI reserves eight slots, but only the four KV producers + # have index-key and key-norm checkpoint weights. + index_slot = index_sources.index(layer.layer_id) + result.extend(( + DecodeWeightSlice("index_wk", index_slot, attention["index_wk"]), + DecodeWeightSlice("index_norm_weight", index_slot, attention["index_norm_weight"]), + )) + if layer.layer_id in index_sources: + slot = index_sources.index(layer.layer_id) + result.extend(( + DecodeWeightSlice("index_wq_b", slot, attention["index_wq_b"]), + DecodeWeightSlice("index_wq_b_scale", slot, attention["index_wq_b_scale"]), + DecodeWeightSlice("index_weights_proj", slot, attention["index_weights_proj"]), + )) + return tuple(result) + + +def iter_decode_weight_slices(model_dir, topology, *, max_bundle_bytes=32 << 30): + """Yield one layer at a time in checkpoint order for bounded staging.""" + raw = json.loads((Path(model_dir) / "config.json").read_text(encoding="utf-8")) + layers = plan_layers(raw) + if len(layers) != 40 or (topology.tp, topology.dp, topology.world) != (4, 2, 8): + raise ValueError("lib decode_fwd requires the 40-layer TP4/DP2/EP8 schedule") + text = raw["text_config"] + kv_sources = tuple(text["kv_source_layer_ids"]) + index_sources = tuple(text["index_source_layer_ids"]) + c2a_sources = tuple(layer.layer_id for layer in layers if layer.mode == "c2a_full") + if len(kv_sources) != 4 or len(index_sources) != 8 or len(c2a_sources) != 3: + raise ValueError("lib decode_fwd source axes disagree with the checkpoint schedule") + for layer in layers: + attention, moe = load_prefill_layer_weights( + model_dir, layer.layer_id, topology, max_bundle_bytes=max_bundle_bytes, + ) + yield from bind_decode_layer_weights( + layer, attention, moe, kv_sources=kv_sources, index_sources=index_sources, + c2a_sources=c2a_sources, + ) + # decode_fwd reserves one key/norm slot per index source, although only KV + # producers have these checkpoint tensors. The trailing four slots are not + # read by the current graph; initialize them deterministically nonetheless. + head_dim = text["head_dim"] + index_dim = text["index_head_dim"] + for slot in range(len(kv_sources), len(index_sources)): + yield DecodeWeightSlice("index_wk", slot, + torch.zeros((topology.world, head_dim, index_dim), dtype=torch.bfloat16)) + yield DecodeWeightSlice("index_norm_weight", slot, + torch.zeros((topology.world, index_dim), dtype=torch.bfloat16)) diff --git a/tests/unit/model/deepseek_v41/test_decode_weights.py b/tests/unit/model/deepseek_v41/test_decode_weights.py new file mode 100644 index 00000000..3af53744 --- /dev/null +++ b/tests/unit/model/deepseek_v41/test_decode_weights.py @@ -0,0 +1,82 @@ +"""Decode weight-slot mapping against the published 40-layer schedule.""" + +from collections import Counter +import json +from pathlib import Path + +import pytest +import torch + +from pypto_serving.model.deepseek_v41 import decode_weights +from pypto_serving.model.deepseek_v41.execution_plan import plan_layers +from pypto_serving.model.deepseek_v41.swa_segment import SegmentTopology + + +CONFIG = Path(__file__).resolve().parents[4] / "tests/fixtures/deepseek_v41/config.json" + + +def _weights(layer_id): + value = torch.full((8, 1), layer_id, dtype=torch.int32) + attention = {name: value for name in decode_weights._ATTENTION.values()} + attention.update({name: value for name in ( + "compressor_norm_weight", "compressor_wkv", "compressor_wgate", + "index_wk", "index_norm_weight", "index_wq_b", "index_wq_b_scale", + "index_weights_proj", + )}) + moe = {name: value for name in decode_weights._MOE.values()} + moe["mxfp4_pair_lut"] = torch.zeros((8, 2, 256), dtype=torch.int16) + moe["routed_w1"] = torch.zeros((8, 1, 1, 128), dtype=torch.uint8) + return attention, moe + + +def test_decode_weight_slices_follow_layer_and_producer_order(monkeypatch, tmp_path): + raw = json.loads(CONFIG.read_text(encoding="utf-8")) + (tmp_path / "config.json").write_text(json.dumps(raw), encoding="utf-8") + calls = [] + + def load(_model_dir, layer_id, _topology, *, max_bundle_bytes): + calls.append((layer_id, max_bundle_bytes)) + return _weights(layer_id) + + monkeypatch.setattr(decode_weights, "load_prefill_layer_weights", load) + chunks = list(decode_weights.iter_decode_weight_slices( + tmp_path, SegmentTopology(tp=4, dp=2), max_bundle_bytes=1234, + )) + by_name = {} + for chunk in chunks: + by_name.setdefault(chunk.name, []).append(chunk) + assert calls == [(layer, 1234) for layer in range(40)] + assert set(by_name) == set(decode_weights._ATTENTION) | set(decode_weights._MOE) | { + "mxfp4_pair_lut", "compressor_norm_weight", "c2a_compressor_wkv", + "c2a_compressor_wgate", "c1a_compressor_wkv", "index_wk", + "index_norm_weight", "index_wq_b", "index_wq_b_scale", "index_weights_proj", + } + assert [c.slot for c in by_name["routed_w1"]] == list(range(40)) + assert all(c.value.dtype == torch.uint8 for c in by_name["routed_w1"]) + assert [c.slot for c in by_name["compressor_norm_weight"]] == list(range(4)) + assert [c.slot for c in by_name["index_wq_b"]] == list(range(8)) + assert [c.slot for c in by_name["c2a_compressor_wkv"]] == list(range(3)) + assert len(by_name["c1a_compressor_wkv"]) == 1 + assert [int(c.value[0, 0]) for c in by_name["index_wk"][:4]] == [2, 8, 14, 20] + assert [c.slot for c in by_name["index_wk"]] == list(range(8)) + assert all(not torch.count_nonzero(c.value) for c in by_name["index_wk"][4:]) + assert all(not torch.count_nonzero(c.value) for c in by_name["index_norm_weight"][4:]) + assert Counter(c.slot for c in by_name["mxfp4_pair_lut"]) == {0: 1} + + +def test_decode_weight_slices_reject_wrong_topology_before_payload_reads(monkeypatch, tmp_path): + raw = json.loads(CONFIG.read_text(encoding="utf-8")) + (tmp_path / "config.json").write_text(json.dumps(raw), encoding="utf-8") + monkeypatch.setattr(decode_weights, "load_prefill_layer_weights", + lambda *_args, **_kwargs: pytest.fail("must not read weights")) + with pytest.raises(ValueError, match="TP4/DP2/EP8"): + list(decode_weights.iter_decode_weight_slices(tmp_path, SegmentTopology(tp=2, dp=2))) + + +def test_decode_weight_slices_reject_incomplete_bundle(): + layer = plan_layers(json.loads(CONFIG.read_text(encoding="utf-8")))[0] + with pytest.raises(ValueError, match="complete Attention"): + decode_weights.bind_decode_layer_weights( + layer, {}, {}, kv_sources=(2, 8, 14, 20), + index_sources=(2, 8, 14, 20, 24, 28, 32, 36), c2a_sources=(2, 8, 14), + ) diff --git a/tests/unit/model/deepseek_v41/test_weight_loader.py b/tests/unit/model/deepseek_v41/test_weight_loader.py index dc42ee91..c0390b3d 100644 --- a/tests/unit/model/deepseek_v41/test_weight_loader.py +++ b/tests/unit/model/deepseek_v41/test_weight_loader.py @@ -148,6 +148,27 @@ def test_c2a_full_bundle_preserves_composite_dtypes_and_index_heads(checkpoint): attention["index_wq_b"][1].view(torch.uint8)) +def test_converted_c2a_bundle_maps_to_decode_weight_slots(checkpoint): + from pypto_serving.model.deepseek_v41.decode_weights import bind_decode_layer_weights + from pypto_serving.model.deepseek_v41.execution_plan import plan_layers + from pypto_serving.model.deepseek_v41.swa_segment import SegmentTopology + from pypto_serving.model.deepseek_v41.swa_weights import load_prefill_layer_weights + + path, raw, _ = checkpoint + attention, moe = load_prefill_layer_weights(path, 0, SegmentTopology(tp=2, dp=1)) + slices = bind_decode_layer_weights(plan_layers(raw)[0], attention, moe, + kv_sources=(0,), index_sources=(0,), c2a_sources=(0,)) + bound = {part.name: part for part in slices} + assert set(bound) >= {"wq_a", "wq_a_scale", "routed_w1", "routed_w1_scale", + "c2a_compressor_wkv", "c2a_compressor_wgate", "index_wq_b"} + assert all(part.slot == 0 for part in slices) + assert bound["wq_a"].value.dtype == torch.float8_e4m3fn + assert bound["wq_a_scale"].value.dtype == torch.float8_e8m0fnu + assert bound["routed_w1"].value.dtype == torch.uint8 + assert bound["routed_w1_scale"].value.dtype == torch.float8_e8m0fnu + assert bound["c2a_compressor_wkv"].value.dtype == torch.float32 + + def test_wo_a_group_dequantization(checkpoint): path, _, tensors = checkpoint result = V41WeightLoader(path, tp_size=2, tp_rank=1).load("layers.0.attn.wo_a.weight") From f59b6f3c63ec3d3865438d96f46f29b5474f49f2 Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Tue, 29 Sep 2026 16:25:19 +0800 Subject: [PATCH 76/78] Validate V4.1 decode weight geometry on checkpoint --- .../model/deepseek_v41/test_weight_loader.py | 6 ++ tools/check_v41_decode_weight_contract.py | 95 +++++++++++++++++++ 2 files changed, 101 insertions(+) create mode 100644 tools/check_v41_decode_weight_contract.py diff --git a/tests/unit/model/deepseek_v41/test_weight_loader.py b/tests/unit/model/deepseek_v41/test_weight_loader.py index c0390b3d..458c2802 100644 --- a/tests/unit/model/deepseek_v41/test_weight_loader.py +++ b/tests/unit/model/deepseek_v41/test_weight_loader.py @@ -167,6 +167,12 @@ def test_converted_c2a_bundle_maps_to_decode_weight_slots(checkpoint): assert bound["routed_w1"].value.dtype == torch.uint8 assert bound["routed_w1_scale"].value.dtype == torch.float8_e8m0fnu assert bound["c2a_compressor_wkv"].value.dtype == torch.float32 + from tools.check_v41_decode_weight_contract import expected_geometry + geometry = expected_geometry(raw["text_config"], SegmentTopology(tp=2, dp=1)) + for name, (shape, dtype) in geometry.items(): + if name in bound: + assert tuple(bound[name].value.shape) == shape, name + assert bound[name].value.dtype == dtype, name def test_wo_a_group_dequantization(checkpoint): diff --git a/tools/check_v41_decode_weight_contract.py b/tools/check_v41_decode_weight_contract.py new file mode 100644 index 00000000..2a67657c --- /dev/null +++ b/tools/check_v41_decode_weight_contract.py @@ -0,0 +1,95 @@ +"""Check real checkpoint weight slices against the current decode ABI geometry. + +CPU only. This checks selected layer types without retaining all 40 layers or +uploading any tensors; it is not a device or numerical model validation. +""" + +import argparse +import gc +import json +from pathlib import Path + +import torch + +from pypto_serving.model.deepseek_v41.decode_weights import bind_decode_layer_weights +from pypto_serving.model.deepseek_v41.execution_plan import plan_layers +from pypto_serving.model.deepseek_v41.swa_segment import SegmentTopology +from pypto_serving.model.deepseek_v41.swa_weights import load_prefill_layer_weights + + +def expected_geometry(text, topology): + world, tp = topology.world, topology.tp + d, inter = text["hidden_size"], text["moe_intermediate_size"] + q, head = text["q_lora_rank"], text["head_dim"] + heads, groups = text["num_attention_heads"], text["o_groups"] + index_h, index_d = text["index_n_heads"], text["index_head_dim"] + local_experts = text["n_routed_experts"] // world + local_heads, local_groups = heads // tp, groups // tp + local_o = local_groups * text["o_lora_rank"] + mix = (2 + text["hc_mult"]) * text["hc_mult"] + return { + "hc_attn_fn": ((world, mix, text["hc_mult"] * d), torch.float32), + "wq_a": ((world, d, q), torch.float8_e4m3fn), + "wq_a_scale": ((world, d // 32, q), torch.float8_e8m0fnu), + "wq_b": ((world, q, local_heads * head), torch.float8_e4m3fn), + "wq_b_scale": ((world, q // 32, local_heads * head), torch.float8_e8m0fnu), + "wkv": ((world, d, head), torch.float8_e4m3fn), + "wo_a": ((world, local_groups, text["o_lora_rank"], + heads * head // groups), torch.bfloat16), + "wo_b": ((world, local_o, d), torch.float8_e4m3fn), + "gate_weight": ((world, text["n_routed_experts"], d), torch.float32), + "routed_w1": ((world, local_experts, inter * d // 256, 128), torch.uint8), + "routed_w2": ((world, local_experts, inter * d // 256, 128), torch.uint8), + "routed_w3": ((world, local_experts, inter * d // 256, 128), torch.uint8), + "routed_w1_scale": ((world, local_experts * d // 32, inter), torch.float8_e8m0fnu), + "routed_w2_scale": ((world, local_experts * inter // 32, d), torch.float8_e8m0fnu), + "routed_w3_scale": ((world, local_experts * d // 32, inter), torch.float8_e8m0fnu), + "shared_w1": ((world, d, inter), torch.float8_e4m3fn), + "mxfp4_pair_lut": ((world, 2, 256), torch.int16), + "c2a_compressor_wkv": ((world, d, head), torch.float32), + "c2a_compressor_wgate": ((world, d, head), torch.float32), + "c1a_compressor_wkv": ((world, d, head), torch.bfloat16), + "compressor_norm_weight": ((world, head), torch.bfloat16), + "index_wk": ((world, head, index_d), torch.bfloat16), + "index_norm_weight": ((world, index_d), torch.bfloat16), + "index_wq_b": ((world, q, index_h * index_d), torch.float8_e4m3fn), + "index_wq_b_scale": ((world, q // 32, index_h * index_d), torch.float8_e8m0fnu), + "index_weights_proj": ((world, d, index_h), torch.bfloat16), + } + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("model_dir", type=Path) + parser.add_argument("--layers", type=int, nargs="+", default=(0, 2, 20, 24, 39)) + args = parser.parse_args() + raw = json.loads((args.model_dir / "config.json").read_text(encoding="utf-8")) + layers = plan_layers(raw) + topology = SegmentTopology(tp=4, dp=2) + geometry = expected_geometry(raw["text_config"], topology) + kv_sources = tuple(raw["text_config"]["kv_source_layer_ids"]) + index_sources = tuple(raw["text_config"]["index_source_layer_ids"]) + c2a_sources = tuple(layer.layer_id for layer in layers if layer.mode == "c2a_full") + if len(layers) != 40 or (len(kv_sources), len(index_sources), len(c2a_sources)) != (4, 8, 3): + raise ValueError("checkpoint schedule differs from lib decode_fwd") + for layer_id in args.layers: + layer = layers[layer_id] + attention, moe = load_prefill_layer_weights(args.model_dir, layer_id, topology) + bound = bind_decode_layer_weights(layer, attention, moe, kv_sources=kv_sources, + index_sources=index_sources, c2a_sources=c2a_sources) + for part in bound: + tensor = part.value + if not tensor.is_contiguous() or tensor.shape[0] != topology.world: + raise ValueError(f"layer {layer_id} {part.name}: invalid EP placement") + if part.name in geometry: + shape, dtype = geometry[part.name] + if tuple(tensor.shape) != shape or tensor.dtype != dtype: + raise ValueError(f"layer {layer_id} {part.name}: got {tensor.shape}/{tensor.dtype}; " + f"expected {shape}/{dtype}") + print(f"layer={layer_id} mode={layer.mode} checked_slices={len(bound)}") + del attention, moe, bound + gc.collect() + + +if __name__ == "__main__": + main() From 959f9593bc6f731bda5e5c35eb627c0ac890332c Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Tue, 29 Sep 2026 16:55:25 +0800 Subject: [PATCH 77/78] Check every selected V4.1 decode weight ABI field --- docs/developer-guide/deepseek-v41-entry.md | 10 +++++++ .../model/deepseek_v41/test_weight_loader.py | 9 ++++--- tools/check_v41_decode_weight_contract.py | 27 +++++++++++++++---- 3 files changed, 37 insertions(+), 9 deletions(-) diff --git a/docs/developer-guide/deepseek-v41-entry.md b/docs/developer-guide/deepseek-v41-entry.md index 08144ccf..1285644e 100644 --- a/docs/developer-guide/deepseek-v41-entry.md +++ b/docs/developer-guide/deepseek-v41-entry.md @@ -113,6 +113,16 @@ Validation uses synthetic checkpoint tensors stored in actual safetensors files, independent packing-address checks, and shared-store regression tests. These checks are not real-checkpoint numerical inference or NPU acceptance. +For lib `f1d3510`, `decode_weights.iter_decode_weight_slices` maps the 40-layer +checkpoint to decode-forward layer and producer slots one layer at a time. +Routed experts remain packed FP4; the four index-key/norm slots reserved by the +ABI but absent from the checkpoint are explicitly zeroed. The A5 original +checkpoint passed CPU shape/dtype/EP-placement checks for 189 slices across +SWA layer 0, C2A Full layer 2, C1A Full layer 20, C1A Reindex layer 24, and +C1A Reuse layer 39. Earlier validation checked all required text headers and +54 selected real payload conversions. This does not scan every payload in all +40 layers or establish device residency; that belongs to Executor/Runner work. + ## Executor/Runner preparation Stage numbers follow serving issue #240. `V41ExecutionPlan` describes the diff --git a/tests/unit/model/deepseek_v41/test_weight_loader.py b/tests/unit/model/deepseek_v41/test_weight_loader.py index 458c2802..0a60c68e 100644 --- a/tests/unit/model/deepseek_v41/test_weight_loader.py +++ b/tests/unit/model/deepseek_v41/test_weight_loader.py @@ -169,10 +169,11 @@ def test_converted_c2a_bundle_maps_to_decode_weight_slots(checkpoint): assert bound["c2a_compressor_wkv"].value.dtype == torch.float32 from tools.check_v41_decode_weight_contract import expected_geometry geometry = expected_geometry(raw["text_config"], SegmentTopology(tp=2, dp=1)) - for name, (shape, dtype) in geometry.items(): - if name in bound: - assert tuple(bound[name].value.shape) == shape, name - assert bound[name].value.dtype == dtype, name + assert set(bound) <= set(geometry) + for name, part in bound.items(): + shape, dtype = geometry[name] + assert tuple(part.value.shape) == shape, name + assert part.value.dtype == dtype, name def test_wo_a_group_dequantization(checkpoint): diff --git a/tools/check_v41_decode_weight_contract.py b/tools/check_v41_decode_weight_contract.py index 2a67657c..c6047a96 100644 --- a/tools/check_v41_decode_weight_contract.py +++ b/tools/check_v41_decode_weight_contract.py @@ -29,15 +29,28 @@ def expected_geometry(text, topology): mix = (2 + text["hc_mult"]) * text["hc_mult"] return { "hc_attn_fn": ((world, mix, text["hc_mult"] * d), torch.float32), + "hc_attn_scale": ((world, 3), torch.float32), + "hc_attn_base": ((world, mix), torch.float32), + "attn_norm_weight": ((world, d), torch.bfloat16), "wq_a": ((world, d, q), torch.float8_e4m3fn), "wq_a_scale": ((world, d // 32, q), torch.float8_e8m0fnu), + "q_norm_weight": ((world, q), torch.bfloat16), "wq_b": ((world, q, local_heads * head), torch.float8_e4m3fn), "wq_b_scale": ((world, q // 32, local_heads * head), torch.float8_e8m0fnu), "wkv": ((world, d, head), torch.float8_e4m3fn), + "wkv_scale": ((world, d // 32, head), torch.float8_e8m0fnu), + "kv_norm_weight": ((world, head), torch.bfloat16), + "attn_sink": ((world, local_heads), torch.float32), "wo_a": ((world, local_groups, text["o_lora_rank"], heads * head // groups), torch.bfloat16), "wo_b": ((world, local_o, d), torch.float8_e4m3fn), + "wo_b_scale": ((world, local_o // 32, d), torch.float8_e8m0fnu), + "hc_ffn_fn": ((world, mix, text["hc_mult"] * d), torch.float32), + "hc_ffn_scale": ((world, 3), torch.float32), + "hc_ffn_base": ((world, mix), torch.float32), + "ffn_norm_weight": ((world, d), torch.bfloat16), "gate_weight": ((world, text["n_routed_experts"], d), torch.float32), + "correction_bias": ((world, text["n_routed_experts"]), torch.float32), "routed_w1": ((world, local_experts, inter * d // 256, 128), torch.uint8), "routed_w2": ((world, local_experts, inter * d // 256, 128), torch.uint8), "routed_w3": ((world, local_experts, inter * d // 256, 128), torch.uint8), @@ -45,6 +58,11 @@ def expected_geometry(text, topology): "routed_w2_scale": ((world, local_experts * inter // 32, d), torch.float8_e8m0fnu), "routed_w3_scale": ((world, local_experts * d // 32, inter), torch.float8_e8m0fnu), "shared_w1": ((world, d, inter), torch.float8_e4m3fn), + "shared_w1_scale": ((world, d // 32, inter), torch.float8_e8m0fnu), + "shared_w2": ((world, inter, d), torch.float8_e4m3fn), + "shared_w2_scale": ((world, inter // 32, d), torch.float8_e8m0fnu), + "shared_w3": ((world, d, inter), torch.float8_e4m3fn), + "shared_w3_scale": ((world, d // 32, inter), torch.float8_e8m0fnu), "mxfp4_pair_lut": ((world, 2, 256), torch.int16), "c2a_compressor_wkv": ((world, d, head), torch.float32), "c2a_compressor_wgate": ((world, d, head), torch.float32), @@ -81,11 +99,10 @@ def main(): tensor = part.value if not tensor.is_contiguous() or tensor.shape[0] != topology.world: raise ValueError(f"layer {layer_id} {part.name}: invalid EP placement") - if part.name in geometry: - shape, dtype = geometry[part.name] - if tuple(tensor.shape) != shape or tensor.dtype != dtype: - raise ValueError(f"layer {layer_id} {part.name}: got {tensor.shape}/{tensor.dtype}; " - f"expected {shape}/{dtype}") + shape, dtype = geometry[part.name] + if tuple(tensor.shape) != shape or tensor.dtype != dtype: + raise ValueError(f"layer {layer_id} {part.name}: got {tensor.shape}/{tensor.dtype}; " + f"expected {shape}/{dtype}") print(f"layer={layer_id} mode={layer.mode} checked_slices={len(bound)}") del attention, moe, bound gc.collect() From 4c7b6280534b4c180dac9c578e592f3f43246922 Mon Sep 17 00:00:00 2001 From: ChenShenAi Date: Wed, 30 Sep 2026 20:44:24 +0800 Subject: [PATCH 78/78] Validate V4.1 cache families before runner allocation --- docs/developer-guide/deepseek-v41-entry.md | 11 +++ .../model/deepseek_v41/cache_contract.py | 69 ++++++++++++++ .../model/deepseek_v41/npu_runner.py | 6 ++ .../model/deepseek_v41/test_cache_contract.py | 89 +++++++++++++++++++ .../deepseek_v41/test_decode_backbone.py | 8 ++ .../model/deepseek_v41/test_decode_weights.py | 8 ++ .../model/deepseek_v41/test_final_output.py | 8 ++ .../deepseek_v41/test_framework_lifecycle.py | 6 +- tools/check_v41_decode_weight_contract.py | 8 ++ tools/compile_v41_decode_backbone.py | 8 ++ tools/compile_v41_serving_output.py | 8 ++ 11 files changed, 228 insertions(+), 1 deletion(-) create mode 100644 pypto_serving/model/deepseek_v41/cache_contract.py create mode 100644 tests/unit/model/deepseek_v41/test_cache_contract.py diff --git a/docs/developer-guide/deepseek-v41-entry.md b/docs/developer-guide/deepseek-v41-entry.md index 1285644e..c8329748 100644 --- a/docs/developer-guide/deepseek-v41-entry.md +++ b/docs/developer-guide/deepseek-v41-entry.md @@ -190,6 +190,17 @@ contract and is explicitly rejected for now. Cache payloads must not use a generic dense K/V substitute. The adapter must bound physical page IDs against actual allocated pools when a group leaves `num_blocks` unspecified. +For the complete 40-layer model, the runner checks three distinct full-history +scheduler groups before calling `allocate`: a 128-source-token window group +covering every layer, a ratio-2 C2A group for KV producers 2/8/14 with 256 +source tokens per 128-row physical page, and a ratio-1 C1A group for producer +20 with 128 source tokens per physical page. Each group uses two private DP +partitions and must cover `max_seq_len` in both its per-request limit and any +declared physical `num_blocks`. Group names are resolved from producer layer +IDs rather than hardcoded. Physical allocation, index-key alignment and +reset/completion remain adapter responsibilities; this host check does not +prove device cache reuse. + At inspected upstream lib revision `fbe92bfc`, serving dispatches the existing sequence-parallel Attention and packed-FP4 MoE composites through `SwaSegment`/`PrefillSegment`. This does not depend on the older full-layer diff --git a/pypto_serving/model/deepseek_v41/cache_contract.py b/pypto_serving/model/deepseek_v41/cache_contract.py new file mode 100644 index 00000000..c8851e1e --- /dev/null +++ b/pypto_serving/model/deepseek_v41/cache_contract.py @@ -0,0 +1,69 @@ +# Copyright (c) PyPTO Contributors. +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of +# CANN Open Software License Agreement Version 2.0 (the "License"). +# Please refer to the License for details. You may not use this file except in compliance with the License. +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. +# See LICENSE in the root of the software repository for the full text of the License. +# ----------------------------------------------------------------------------------------------------------- +"""Scheduler cache-family contract for the complete V4.1 text backbone. + +The compressed pools have different source-token capacities per physical page: +C2A stores one row per two tokens, while C1A stores one row per token. Their +page IDs may be jointly lowered with index keys inside each family, but cannot +share one scheduler group with a single block-size declaration. +""" + +from dataclasses import dataclass + +from pypto_serving.config.types import KVCacheGroupSpec + +from .execution_plan import LayerPlan + + +@dataclass(frozen=True) +class V41CacheGroups: + window: str + c2a: str + c1a: str + + +def validate_cache_groups( + layers: tuple[LayerPlan, ...], groups: tuple[KVCacheGroupSpec, ...], max_seq_len: int, +) -> V41CacheGroups: + """Resolve the three full-history page families before device allocation.""" + if len(layers) != 40 or tuple(layer.layer_id for layer in layers) != tuple(range(40)): + raise ValueError("V4.1 cache contract requires the complete ordered 40-layer backbone") + if type(max_seq_len) is not int or max_seq_len <= 0: + raise ValueError("V4.1 cache contract requires a positive sequence capacity") + if (len(groups) != 3 or any(not isinstance(group, KVCacheGroupSpec) for group in groups) + or len({group.name for group in groups}) != 3): + raise ValueError("V4.1 requires separate window, C2A and C1A cache groups") + + expected = { + "window": (tuple(range(40)), 128, 1), + "c2a": (tuple(layer.layer_id for layer in layers + if layer.mode == "c2a_full" and layer.kv_source == layer.layer_id), 256, 2), + "c1a": (tuple(layer.layer_id for layer in layers + if layer.mode == "c1a_full" and layer.kv_source == layer.layer_id), 128, 1), + } + if not expected["c2a"][0] or not expected["c1a"][0]: + raise ValueError("V4.1 cache contract requires C2A and C1A KV producers") + resolved = {} + for group in groups: + matches = [family for family, (owners, _, _) in expected.items() + if tuple(group.layer_indices) == owners] + if len(matches) != 1 or matches[0] in resolved: + raise ValueError("V4.1 cache groups must match window and KV producer layers") + family = matches[0] + _, block_size, ratio = expected[family] + if (group.spec.block_size != block_size or group.spec.compress_ratio != ratio + or group.spec.storage_block_size != 128 or group.num_partitions != 2 + or group.sliding_window is not None or group.is_eagle_group): + raise ValueError(f"V4.1 {family} cache page layout disagrees with the lib ABI") + required = (max_seq_len + block_size - 1) // block_size + if (group.max_blocks_per_seq < required + or (group.num_blocks is not None and group.num_blocks < required)): + raise ValueError(f"V4.1 {family} cache cannot cover max_seq_len") + resolved[family] = group.name + return V41CacheGroups(**resolved) diff --git a/pypto_serving/model/deepseek_v41/npu_runner.py b/pypto_serving/model/deepseek_v41/npu_runner.py index cab231c8..eb196009 100644 --- a/pypto_serving/model/deepseek_v41/npu_runner.py +++ b/pypto_serving/model/deepseek_v41/npu_runner.py @@ -8,6 +8,7 @@ # ----------------------------------------------------------------------------------------------------------- """V4-style runner lifecycle for explicit V4.1 composite bindings.""" from .composite import BuildOptions, CompositeBindings, LayerState +from .cache_contract import validate_cache_groups from .input_preparation import lookup_token_embeddings import torch from pypto_serving.config.types import PrefillResult, DecodeResult @@ -32,6 +33,11 @@ def __init__(self, plan: V41ExecutionPlan, bindings: CompositeBindings, *, devic if any(type(i) is not int or i < 0 for i in self.device_ids): raise ValueError("device IDs must be nonnegative integers") bindings.require(plan.layers, plan.placement) + if len(plan.layers) == 40: + self.cache_groups = validate_cache_groups( + plan.layers, bindings.cache_groups, runtime.max_seq_len) + else: + self.cache_groups = None self.runtime = runtime self.build_options = build_options self.resources = None diff --git a/tests/unit/model/deepseek_v41/test_cache_contract.py b/tests/unit/model/deepseek_v41/test_cache_contract.py new file mode 100644 index 00000000..b1f91fd4 --- /dev/null +++ b/tests/unit/model/deepseek_v41/test_cache_contract.py @@ -0,0 +1,89 @@ +# Copyright (c) PyPTO Contributors. +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of +# CANN Open Software License Agreement Version 2.0 (the "License"). +# Please refer to the License for details. You may not use this file except in compliance with the License. +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. +# See LICENSE in the root of the software repository for the full text of the License. +# ----------------------------------------------------------------------------------------------------------- +"""Full-backbone cache layout checks before a worker opens device resources.""" + +from dataclasses import replace +import json +from pathlib import Path +from types import SimpleNamespace + +import pytest + +from pypto_serving.config.types import KVCacheGroupSpec, KVCacheSpec +from pypto_serving.model.deepseek_v41.cache_contract import V41CacheGroups, validate_cache_groups +from pypto_serving.model.deepseek_v41.composite import CompositeBindings +from pypto_serving.model.deepseek_v41.execution_plan import RankPlacement, plan_layers +from pypto_serving.model.deepseek_v41.npu_runner import V41ModelRunner + + +@pytest.fixture +def contract(): + path = Path(__file__).resolve().parents[4] / "tests/fixtures/deepseek_v41/config.json" + layers = plan_layers(json.loads(path.read_text(encoding="utf-8"))) + groups = ( + KVCacheGroupSpec("window", tuple(range(40)), KVCacheSpec(128, 16), + 65, num_blocks=65, num_partitions=2), + KVCacheGroupSpec("c2a", (2, 8, 14), KVCacheSpec(256, 16, 2), + 33, num_blocks=33, num_partitions=2), + KVCacheGroupSpec("c1a", (20,), KVCacheSpec(128, 16), + 65, num_blocks=65, num_partitions=2), + ) + return layers, groups + + +def test_full_backbone_resolves_three_independent_cache_families(contract): + layers, groups = contract + assert validate_cache_groups(layers, groups, 8320) == V41CacheGroups( + window="window", c2a="c2a", c1a="c1a") + + +def test_cache_family_names_follow_producers_not_labels(contract): + layers, groups = contract + renamed = tuple(replace(group, name=f"pool_{index}") for index, group in enumerate(groups)) + assert validate_cache_groups(layers, renamed, 8320) == V41CacheGroups( + window="pool_0", c2a="pool_1", c1a="pool_2") + + +@pytest.mark.parametrize("change,match", [ + (lambda groups: groups[:2], "separate"), + (lambda groups: (groups[0], replace(groups[1], layer_indices=(2, 8, 14, 20)), groups[2]), + "producer"), + (lambda groups: (groups[0], groups[1], replace(groups[2], spec=KVCacheSpec(256, 16, 2))), + "c1a.*layout"), + (lambda groups: (replace(groups[0], max_blocks_per_seq=64), *groups[1:]), + "window.*max_seq_len"), + (lambda groups: (replace(groups[0], num_blocks=64), *groups[1:]), + "window.*max_seq_len"), + (lambda groups: (replace(groups[0], num_partitions=1), *groups[1:]), + "window.*layout"), + (lambda groups: (groups[0], replace(groups[1], max_blocks_per_seq=32), groups[2]), + "c2a.*max_seq_len"), +]) +def test_mismatched_cache_family_rejected(contract, change, match): + layers, groups = contract + with pytest.raises(ValueError, match=match): + validate_cache_groups(layers, change(groups), 8320) + + +def test_runner_rejects_bad_layout_before_device_allocation(contract): + layers, groups = contract + calls = [] + bindings = CompositeBindings( + revision="recording-only", entries={("prefill", layer.mode): lambda *a: None + for layer in layers}, + initialize=lambda *a: None, output=lambda *a: None, + allocate=lambda *a: calls.append("allocate"), prepare_weights=lambda *a: None, + reset_request=lambda *a: None, wait=lambda *a: None, close=lambda *a: None, + cache_groups=groups[:2], decode_backbone=lambda *a: None, + ) + plan = SimpleNamespace(placement=RankPlacement(0), layers=layers) + with pytest.raises(ValueError, match="separate"): + V41ModelRunner(plan, bindings, device_ids=range(8), + runtime=SimpleNamespace(max_seq_len=8320)) + assert calls == [] diff --git a/tests/unit/model/deepseek_v41/test_decode_backbone.py b/tests/unit/model/deepseek_v41/test_decode_backbone.py index a1766ff7..a55e560f 100644 --- a/tests/unit/model/deepseek_v41/test_decode_backbone.py +++ b/tests/unit/model/deepseek_v41/test_decode_backbone.py @@ -1,3 +1,11 @@ +# Copyright (c) PyPTO Contributors. +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of +# CANN Open Software License Agreement Version 2.0 (the "License"). +# Please refer to the License for details. You may not use this file except in compliance with the License. +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. +# See LICENSE in the root of the software repository for the full text of the License. +# ----------------------------------------------------------------------------------------------------------- from types import SimpleNamespace import pytest diff --git a/tests/unit/model/deepseek_v41/test_decode_weights.py b/tests/unit/model/deepseek_v41/test_decode_weights.py index 3af53744..5dd98cc0 100644 --- a/tests/unit/model/deepseek_v41/test_decode_weights.py +++ b/tests/unit/model/deepseek_v41/test_decode_weights.py @@ -1,3 +1,11 @@ +# Copyright (c) PyPTO Contributors. +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of +# CANN Open Software License Agreement Version 2.0 (the "License"). +# Please refer to the License for details. You may not use this file except in compliance with the License. +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. +# See LICENSE in the root of the software repository for the full text of the License. +# ----------------------------------------------------------------------------------------------------------- """Decode weight-slot mapping against the published 40-layer schedule.""" from collections import Counter diff --git a/tests/unit/model/deepseek_v41/test_final_output.py b/tests/unit/model/deepseek_v41/test_final_output.py index 540b30d1..aa91b11c 100644 --- a/tests/unit/model/deepseek_v41/test_final_output.py +++ b/tests/unit/model/deepseek_v41/test_final_output.py @@ -1,3 +1,11 @@ +# Copyright (c) PyPTO Contributors. +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of +# CANN Open Software License Agreement Version 2.0 (the "License"). +# Please refer to the License for details. You may not use this file except in compliance with the License. +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. +# See LICENSE in the root of the software repository for the full text of the License. +# ----------------------------------------------------------------------------------------------------------- from types import SimpleNamespace import pytest diff --git a/tests/unit/model/deepseek_v41/test_framework_lifecycle.py b/tests/unit/model/deepseek_v41/test_framework_lifecycle.py index 1bd3291b..5ef9643e 100644 --- a/tests/unit/model/deepseek_v41/test_framework_lifecycle.py +++ b/tests/unit/model/deepseek_v41/test_framework_lifecycle.py @@ -43,8 +43,10 @@ def test_public_runner_chunked_8k_128_decode_release_and_reuse(): groups = ( KVCacheGroupSpec("window", tuple(range(40)), KVCacheSpec(128, 16), 65, num_blocks=65, num_partitions=2), - KVCacheGroupSpec("compressed", (2, 8, 14, 20), KVCacheSpec(256, 16, 2), + KVCacheGroupSpec("c2a", (2, 8, 14), KVCacheSpec(256, 16, 2), 33, num_blocks=33, num_partitions=2), + KVCacheGroupSpec("c1a", (20,), KVCacheSpec(128, 16), + 65, num_blocks=65, num_partitions=2), ) runtime = RuntimeConfig(max_batch_size=2, max_seq_len=8320, max_num_batched_tokens=2048, max_prefill_tokens_per_request=1024, kv_cache_groups=groups) @@ -124,6 +126,8 @@ def close(allocated): plan = SimpleNamespace(placement=RankPlacement(0), layers=layers, weights=SimpleNamespace(config=config)) plan.for_rank = lambda rank: SimpleNamespace(placement=RankPlacement(rank)) runner = V41ModelRunner(plan, bindings, device_ids=(7, 2, 5, 0, 6, 1, 4, 3), runtime=runtime) + assert (runner.cache_groups.window, runner.cache_groups.c2a, runner.cache_groups.c1a) == ( + "window", "c2a", "c1a") model = SimpleNamespace(config=config) partitions, lengths = {"A": 1, "B": 0, "C": 1}, {"A": 0, "B": 0} prompts = {"A": [(i + 3) % 16 for i in range(8192)], diff --git a/tools/check_v41_decode_weight_contract.py b/tools/check_v41_decode_weight_contract.py index c6047a96..f02136a1 100644 --- a/tools/check_v41_decode_weight_contract.py +++ b/tools/check_v41_decode_weight_contract.py @@ -1,3 +1,11 @@ +# Copyright (c) PyPTO Contributors. +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of +# CANN Open Software License Agreement Version 2.0 (the "License"). +# Please refer to the License for details. You may not use this file except in compliance with the License. +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. +# See LICENSE in the root of the software repository for the full text of the License. +# ----------------------------------------------------------------------------------------------------------- """Check real checkpoint weight slices against the current decode ABI geometry. CPU only. This checks selected layer types without retaining all 40 layers or diff --git a/tools/compile_v41_decode_backbone.py b/tools/compile_v41_decode_backbone.py index a2e60e77..435eac0a 100644 --- a/tools/compile_v41_decode_backbone.py +++ b/tools/compile_v41_decode_backbone.py @@ -1,3 +1,11 @@ +# Copyright (c) PyPTO Contributors. +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of +# CANN Open Software License Agreement Version 2.0 (the "License"). +# Please refer to the License for details. You may not use this file except in compliance with the License. +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. +# See LICENSE in the root of the software repository for the full text of the License. +# ----------------------------------------------------------------------------------------------------------- """Codegen-only check of the 40-layer V4.1 decode serving boundary.""" import argparse diff --git a/tools/compile_v41_serving_output.py b/tools/compile_v41_serving_output.py index 634e45a9..d11866b9 100644 --- a/tools/compile_v41_serving_output.py +++ b/tools/compile_v41_serving_output.py @@ -1,3 +1,11 @@ +# Copyright (c) PyPTO Contributors. +# This program is free software, you can redistribute it and/or modify it under the terms and conditions of +# CANN Open Software License Agreement Version 2.0 (the "License"). +# Please refer to the License for details. You may not use this file except in compliance with the License. +# THIS SOFTWARE IS PROVIDED ON AN "AS IS" BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER EXPRESS OR IMPLIED, +# INCLUDING BUT NOT LIMITED TO NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. +# See LICENSE in the root of the software repository for the full text of the License. +# ----------------------------------------------------------------------------------------------------------- """Codegen-only check of the V4.1 final-state output adapter.""" import argparse