From 5483cc7e31631c5a3a8edaa8963a453e68d54588 Mon Sep 17 00:00:00 2001 From: Hongsheng Liu Date: Wed, 5 Aug 2026 08:46:47 +0800 Subject: [PATCH 1/2] Align vLLM-Omni v0.26.0 knowledge baseline --- adapters/vllm_omni/manifest.yaml | 16 ++++-- adapters/vllm_omni/release_baseline.yaml | 28 +++++----- doc/KNOWLEDGE.md | 4 +- .../ci/guides/inspect-existing-tests-first.md | 28 +++++----- .../benchmark/guides/performance-evidence.md | 2 +- knowledge/repos/vllm-omni/ci/_index.md | 4 +- .../ci/guides/buildkite-structure.md | 35 +++++++------ .../repos/vllm-omni/ci/guides/test-tiers.md | 4 +- .../components/configuration/_index.md | 7 +-- .../components/configuration/architecture.md | 6 +-- .../components/configuration/deploy-yaml.md | 9 ++-- .../configuration/omni-init-args.md | 4 +- .../configuration/pipeline-deploy-schema.md | 13 ++--- .../components/configuration/rules.md | 40 +++++++++++++- .../vllm-omni/components/diffusion/_index.md | 5 +- .../vllm-omni/components/diffusion/rules.md | 30 ++++++++++- .../components/distributed/_index.md | 5 +- .../vllm-omni/components/distributed/rules.md | 36 +++++++++++++ .../components/model-executor/_index.md | 4 +- .../components/model-executor/architecture.md | 6 +-- .../components/model-executor/rules.md | 42 ++++++++++++++- .../vllm-omni/components/scheduler/_index.md | 4 +- .../components/scheduler/architecture.md | 13 +++-- .../vllm-omni/components/scheduler/rules.md | 52 ++++++++++++++++++- .../vllm-omni/components/serving/_index.md | 4 +- .../vllm-omni/components/serving/rules.md | 42 ++++++++++++++- .../repos/vllm-omni/docs/design-doc-map.md | 4 +- .../repos/vllm-omni/models/audex/_index.md | 51 ++++++++++++++++++ knowledge/repos/vllm-omni/models/catalog.md | 42 ++++++++------- .../models/dreamzero/architecture.md | 8 +-- .../vllm-omni/models/gr00t/architecture.md | 2 +- .../models/higgs-audio/architecture.md | 2 +- .../vllm-omni/models/hunyuan-video/_index.md | 2 +- .../repos/vllm-omni/models/ltx2/_index.md | 16 +++--- .../vllm-omni/models/ltx2/architecture.md | 21 ++++---- .../models/mammoth-moda2/architecture.md | 2 +- .../vllm-omni/models/minimax-h3/_index.md | 40 ++++++++++++++ .../vllm-omni/models/soulx-singer/_index.md | 2 +- .../models/soulx-singer/architecture.md | 2 +- knowledge/repos/vllm-omni/rebase/_index.md | 6 +-- .../vllm-omni/rebase/upstream-api-drift.md | 7 +-- knowledge/repos/vllm-omni/rebase/workflow.md | 8 +-- 42 files changed, 502 insertions(+), 156 deletions(-) create mode 100644 knowledge/repos/vllm-omni/components/distributed/rules.md create mode 100644 knowledge/repos/vllm-omni/models/audex/_index.md create mode 100644 knowledge/repos/vllm-omni/models/minimax-h3/_index.md diff --git a/adapters/vllm_omni/manifest.yaml b/adapters/vllm_omni/manifest.yaml index 38d32ad..e45092b 100644 --- a/adapters/vllm_omni/manifest.yaml +++ b/adapters/vllm_omni/manifest.yaml @@ -43,7 +43,7 @@ modules: wave: 1 risk: high model_executor: - local_paths: [vllm_omni/model_executor/] + local_paths: [vllm_omni/model_executor/, vllm_omni/experimental/, vllm_omni/model_extras/] wave: 1 risk: high input_output: @@ -54,17 +54,25 @@ modules: wave: 1 risk: high online_serving: - local_paths: [vllm_omni/entrypoints/] + local_paths: [vllm_omni/entrypoints/, vllm_omni/engine/] wave: 1 model_config: - local_paths: [vllm_omni/config/] + local_paths: [vllm_omni/config/, vllm_omni/deploy/] wave: 1 platform: local_paths: [vllm_omni/platforms/] wave: 1 benchmarks: - local_paths: [benchmarks/] + local_paths: [benchmarks/, vllm_omni/benchmarks/] wave: 2 + diffusion: + local_paths: [vllm_omni/diffusion/] + wave: 1 + risk: high + distributed: + local_paths: [vllm_omni/distributed/] + wave: 1 + risk: high validation: test_manifest_sources: [.buildkite/] diff --git a/adapters/vllm_omni/release_baseline.yaml b/adapters/vllm_omni/release_baseline.yaml index 01267ee..51b41dd 100644 --- a/adapters/vllm_omni/release_baseline.yaml +++ b/adapters/vllm_omni/release_baseline.yaml @@ -2,25 +2,25 @@ schema_version: 1 upstream: repository: vllm-project/vllm-omni - previous_audited_sha: 5d44868e918ecf9d3a6f1158c45acfad5989e1a7 - audited_ref: v0.26.0rc1 - audited_sha: 807db6efd70ff2e9b55a63d6e1b0530e2b74f8f2 + previous_audited_sha: 807db6efd70ff2e9b55a63d6e1b0530e2b74f8f2 + audited_ref: v0.26.0 + audited_sha: a4ea67a21b20054dacc6e83952f9bd407e8ee4e7 # These fingerprints are generated from sorted registry/deploy inventories. # They make the baseline exact without duplicating hundreds of entries here. inventories: autoregressive: - count: 72 - sha256: 79afac742125003b8d65a02e0e6b53398c4fc5f1c6e765fa8e1eb31f74871195 + count: 77 + sha256: 10e7850b0c7804f23ddffeb7cf5f2ad83c616d3977ec3b67eabab5c121564a49 diffusion: - count: 61 - sha256: f5c0b782cb98f20a9c7c77b696fda620a12d0899fd9e8abc9ccb696fe0c457e1 + count: 58 + sha256: 7dc6a5a373ca33a31a44961c1422008327cede3b364577fa095fd37c05ea2085 pipelines: - count: 46 - sha256: ce4f6edb8748e030f04bf7a6fe038231453cd1cd75f64311c898acff8c14b14b + count: 51 + sha256: 8f567368a2a958fee2f89e4bce3ec20ec2e7a420bbdbab2fef9d65fb604e3ae8 deploy_yamls: - count: 71 - sha256: fb4c591580262a2dc8d03dd8c7fd681d98486f31007696d56adf349c24c25147 + count: 79 + sha256: fb75633eeeaadc4f421fd4d2ec62b03b04d9a9f48e959297d88ae51c20d51bc5 # Structural review owners for changed upstream paths. This is deliberately # separate from manifest.modules: changing that runtime map alters rebase fan-out. @@ -44,7 +44,6 @@ path_owners: - docs/ - examples/ - recipes/ - - realtime_video_prompt_interaction_protocol.md model-executor: - vllm_omni/attention/ - vllm_omni/experimental/ @@ -74,6 +73,7 @@ path_owners: - tests/ tooling: - tools/ + - apps/ComfyUI-vLLM-Omni/ runtime-core: - vllm_omni/__init__.py - vllm_omni/data_entry_keys.py @@ -91,7 +91,7 @@ owner_documents: diffusion: - knowledge/repos/vllm-omni/components/diffusion/rules.md distributed: - - knowledge/repos/vllm-omni/components/distributed/_index.md + - knowledge/repos/vllm-omni/components/distributed/rules.md documentation: - knowledge/general/docs/_index.md model-executor: @@ -124,6 +124,8 @@ ignored_paths: reason: legal metadata - pattern: README.md reason: repository overview +- pattern: realtime_video_prompt_interaction_protocol.md + reason: deleted upstream documentation artifact - pattern: SECURITY.md reason: community metadata - pattern: setup.py diff --git a/doc/KNOWLEDGE.md b/doc/KNOWLEDGE.md index 2d66647..603e2f0 100644 --- a/doc/KNOWLEDGE.md +++ b/doc/KNOWLEDGE.md @@ -41,8 +41,8 @@ extensions merely because they are absent from the common source. (per-file upstream created/updated dates, captured before the submodule's git history was removed — used for page frontmatter). - **Code-mirror pin:** the `knowledge/repos/vllm-omni/components/` source maps - are verified against vllm-omni `main @ - 807db6efd70ff2e9b55a63d6e1b0530e2b74f8f2`. The canonical machine baseline + are verified against vllm-omni `v0.26.0 @ + a4ea67a21b20054dacc6e83952f9bd407e8ee4e7`. The canonical machine baseline is `adapters/vllm_omni/release_baseline.yaml`. ## Layout: general vs repo-specific diff --git a/knowledge/general/ci/guides/inspect-existing-tests-first.md b/knowledge/general/ci/guides/inspect-existing-tests-first.md index f77ed4d..b1bf22a 100644 --- a/knowledge/general/ci/guides/inspect-existing-tests-first.md +++ b/knowledge/general/ci/guides/inspect-existing-tests-first.md @@ -4,7 +4,7 @@ created: 2026-07-10 updated: 2026-07-16 type: guide tags: [general, ci] -sources: ["#3297", "tests/e2e/accuracy/test_gebench_h100_smoke.py", "tests/e2e/accuracy/test_hunyuan_image3.py"] +sources: ["#3297", "tests/e2e/accuracy/test_hunyuan_image3.py", "tests/e2e/accuracy/conftest.py"] --- # 规则:先看现有代码,再谈方案 @@ -16,13 +16,11 @@ sources: ["#3297", "tests/e2e/accuracy/test_gebench_h100_smoke.py", "tests/e2e/a - 讨论"用 CLIP-score 替代"、"改 GenBench 砍 it2i"、"采集 9 小时 baseline" - 写了 300 行方案对比文档 -结果用户直接去看仓库: -``` -tests/e2e/accuracy/test_gebench_h100_smoke.py -``` -**已经存在**,跑的就是 type3/type4 T2I,用 `/v1/images/generations`,judge 是 VLM-as-judge(Qwen2.5-VL-7B-Instruct),**零 mmdet/mmcv 依赖**。 - -`--gebench-model` 还是 CLI 参数。加一行 `--gebench-model Tencent/HunyuanImage-3.0-Instruct` 就能跑。 +结果用户直接去看当前 target 的 `tests/e2e/accuracy/`:旧记录中的 +`test_gebench_h100_smoke.py` 和 `test_gedit_bench_h100_smoke.py` 已经删除;当前 +HunyuanImage3 的入口是 `tests/e2e/accuracy/test_hunyuan_image3.py`,并由 +`tests/e2e/accuracy/conftest.py` 提供仍然存在的 accuracy fixtures。旧的 +`--gebench-model` 路径不能再当作现行合同。 我前面所有工作(踩 mmcv 坑、建 Py3.10 venv、写 GenEval 集成代码)全部白做。 @@ -45,23 +43,23 @@ tests/e2e/accuracy/test_gebench_h100_smoke.py | 目录 | 内容 | |---|---| | `tests/e2e/accuracy/` | 所有精度 CI 入口 | -| `tests/e2e/accuracy/conftest.py` | fixture 和 CLI option(`--gebench-model`, `--gedit-model`, `--accuracy-judge-model` 等) | -| `tests/e2e/accuracy/test_gebench_h100_smoke.py` | T2I 精度(type3/type4)用 VLM judge | -| `tests/e2e/accuracy/test_gedit_bench_h100_smoke.py` | IT2I 精度,同款 judge 路径 | +| `tests/e2e/accuracy/conftest.py` | 当前仍存在的 Wan2.2/HunyuanVideo accuracy fixtures 和 CLI options | +| `tests/e2e/accuracy/test_hunyuan_image3.py` | HunyuanImage3 IT2I/COT 对齐入口 | +| `tests/e2e/accuracy/test_hunyuan_image3_pixel_accuracy.py` | HunyuanImage3 像素精度入口 | | `tests/e2e/accuracy/helpers.py` | `reset_artifact_dir` 等 | -| `vllm_omni/benchmarks/accuracy/text_to_image/gbench.py` | GEBench 实现(type3/type4 走 `/v1/images/generations`,type1/2/5 走 `/v1/images/edits`) | ## 下一次做新模型精度 CI -抄 `test_gebench_h100_smoke.py`,改 `--gebench-model` 参数。完成。 -**除非有强理由**(比如新模型 T2I endpoint 不同、判据不同),否则不要考虑 GenEval / CLIP-score / 自研评分器。 +先从当前 `tests/e2e/accuracy/` 选择语义最接近的幸存者,读取它的 fixture、CLI +选项和运行方式,再决定复用还是新增;不要假设历史 GEBench 文件或参数仍存在。 ## 判据模板 用户问"做 XX 精度 CI"时,先回答 3 个问题再提方案: 1. 团队现有类似测试文件?叫什么名字? 2. 它的 fixture 和 CLI option 我的模型能复用吗? -3. 跑一下 `pytest --gebench-model ` 会发生什么?报错才说明需要改,没报错就结束了 +3. 跑一下当前幸存者的最小 pytest 入口会发生什么?只有现行路径的采集或运行错误 + 才说明需要改,不要为已删除的 slug 重建测试文件。 ## HunyuanImage3 现成 IT2I accuracy pytest(2026-06-02) diff --git a/knowledge/repos/vllm-omni/benchmark/guides/performance-evidence.md b/knowledge/repos/vllm-omni/benchmark/guides/performance-evidence.md index 28eccc6..8a2e526 100644 --- a/knowledge/repos/vllm-omni/benchmark/guides/performance-evidence.md +++ b/knowledge/repos/vllm-omni/benchmark/guides/performance-evidence.md @@ -50,7 +50,7 @@ sources: ["PR #5052"] **PR Test Result 规范**: - 表格标题必须带输入口径,例如 `Official IT2I performance comparison`,不要只写 `Performance`。 -- 精度表必须写 reference,例如 `against tests/e2e/accuracy/assets/hunyuan_image_ref.png`。 +- 精度表必须写 reference,例如 `against tests/assets/hunyuan/hunyuan_image_ref.png`。 - 速度表必须写 `model initialization excluded/included`。 - 如果同一 PR 同时有 smoke 和 official e2e,smoke 放在最后,且标题写 `Compatibility smoke`。 diff --git a/knowledge/repos/vllm-omni/ci/_index.md b/knowledge/repos/vllm-omni/ci/_index.md index cee1066..8907931 100644 --- a/knowledge/repos/vllm-omni/ci/_index.md +++ b/knowledge/repos/vllm-omni/ci/_index.md @@ -1,10 +1,10 @@ --- title: "vLLM-Omni CI" created: 2026-07-10 -updated: 2026-07-10 +updated: 2026-08-05 type: index tags: [vllm-omni, ci] -sources: [] +sources: [.buildkite/cuda/pipeline.yml, docs/contributing/ci/test_system_overview.md] --- # vLLM-Omni CI diff --git a/knowledge/repos/vllm-omni/ci/guides/buildkite-structure.md b/knowledge/repos/vllm-omni/ci/guides/buildkite-structure.md index 9ea7ff7..0799436 100644 --- a/knowledge/repos/vllm-omni/ci/guides/buildkite-structure.md +++ b/knowledge/repos/vllm-omni/ci/guides/buildkite-structure.md @@ -1,37 +1,40 @@ --- title: "Buildkite 管线结构" created: 2026-07-16 -updated: 2026-07-16 +updated: 2026-08-05 type: guide tags: [vllm-omni, ci] -sources: [".buildkite/pipeline.yml", "vllm-omni-rebase-agent@122a9468:agent/config.py", "vllm-omni-rebase-agent@122a9468:config.sh"] +sources: [".buildkite/cuda/pipeline.yml", ".buildkite/cuda/rebase-pipeline.yml", "vllm-omni-rebase-agent@122a9468:agent/config.py", "vllm-omni-rebase-agent@122a9468:config.sh"] --- # Buildkite 管线结构 -仓库侧事实在 `main @ 5c390096` 复核;运营侧事实(管线名、队列→硬件映射)来自 +仓库侧事实在 `v0.26.0 @ a4ea67a2` 复核;运营侧事实(管线名、队列→硬件映射)来自 rebase-agent 配置快照(@122a9468),**属运营观测、可能漂移**,用前核对。 ## `.buildkite/` 布局(仓库侧) -- **`pipeline.yml`**(根入口,two-doc 模式):文档 1 经 - `.buildkite/scripts/upload_pipeline.py` 做 skip-ci 判定(docs-only / pytest - skip 标记,diff-aware);`---` 后的文档 2 构建 CI 镜像(`docker/Dockerfile.ci` → - `public.ecr.aws/q9t5s3a7/vllm-ci-test-repo`)并按条件上载子管线。 +- **`cuda/pipeline.yml`**(CUDA hook entry):由 + `.buildkite/common/scripts/upload_pipeline.py` 读取 `cuda/bootstrap-upload-steps.yml` + 并上传实际子管线;skip-ci 的 step key 是 `upload-ci-pipeline`,不再依赖根目录 + `pipeline.yml` 的 two-doc 结构。 - 分级 → 子管线映射:**L2 → `test-ready.yml`**(带 `ready` label 的 PR,diff-aware; 或 main+nightly);**L3 → `test-merge.yml`**(`merge-test` label 或 main+nightly); **L4 → `test-nightly.yml`**(main+nightly 定时,或 PR label + rebuild); **L5 → `test-weekly.yml`**(每周,依赖镜像构建)。 -- 平台管线:`pipeline-intel.yaml`、`pipeline-npu.yaml`、`pipeline-npu-a3.yaml`; - AMD 变体 `test-amd.yaml`/`test-amd-ready.yaml`/`test-amd-merge.yml` + +- 平台管线:`intel/pipeline-intel.yml`、`npu/pipeline-npu.yml`、 + `npu/pipeline-npu-a3.yml`;AMD 变体 `amd/test-amd-ready.yml`/ + `test-amd-merge.yml` + `test-template-amd-omni.j2` + bootstrap 脚本;发布/对齐: - `release-pipeline.yaml`、`rebase-pipeline.yaml`。 -- 硬件 runner 脚本:`scripts/`(`run-amd-test.sh`、`run-xpu-test.sh`、 - `run_npu_test.sh`、nightly-index 与 wheel/镜像发布脚本)。 -- `test-ready.yml` 示例步骤组("Simple Test"): - `pytest tests/diffusion tests/model_executor -m 'core_model and cpu'` + - 互补 "Other Test" + "Custom Pipeline Test" - (`tests/e2e/offline_inference/custom_pipeline/ -m core_model`),跑在 +- `release/release-pipeline.yml` 是发布管线;对齐管线现在是 + `cuda/rebase-pipeline.yml`,分别上传 ready、merge 和 nightly 子管线。 +- 测试树按 owner/feature 归档:原 `tests/ar_diffusion/` 归入 + `tests/diffusion/ar_diffusion/`,full-duplex、custom pipeline、RLHF 和 ComfyUI + 归入 `tests/e2e/features//`;新增 component 测试放在对应 + `tests/{component}/`,不创建平行的旧顶层目录。 +- `cuda/test-ready.yml` 的 CPU fast lanes 仍按 `tests/diffusion`、 + `tests/model_executor`、`tests/entrypoints` 和 `tests/engine` 分组;feature lane + 使用 `tests/e2e/features/custom_pipeline/` 与 `tests/e2e/features/fullduplex/`,跑在 `gpu_1_queue`、CI docker 镜像内、`HF_HOME=/fsx/hf_cache`。 ## 运营事实(rebase-agent 观测,@122a9468) diff --git a/knowledge/repos/vllm-omni/ci/guides/test-tiers.md b/knowledge/repos/vllm-omni/ci/guides/test-tiers.md index 19476da..62b81cc 100644 --- a/knowledge/repos/vllm-omni/ci/guides/test-tiers.md +++ b/knowledge/repos/vllm-omni/ci/guides/test-tiers.md @@ -1,7 +1,7 @@ --- title: "测试分级(L1–L5)与 pytest markers" created: 2026-07-16 -updated: 2026-07-31 +updated: 2026-08-05 type: guide tags: [vllm-omni, ci] sources: [docs/contributing/ci/test_system_overview.md, docs/contributing/ci/test_writing_guide.md] @@ -10,7 +10,7 @@ sources: [docs/contributing/ci/test_system_overview.md, docs/contributing/ci/tes # 测试分级(L1–L5)与 pytest markers 官方 spec:`docs/contributing/ci/test_system_overview.md` + -`test_writing_guide.md`(`main @ 807db6ef` 复核)。测试金字塔五级 + Common 规范 +`test_writing_guide.md`(`v0.26.0 @ a4ea67a2` 复核)。测试金字塔五级 + Common 规范 (PR 模板/checklist 与 CI 失败说明)。 ## 五级定义 diff --git a/knowledge/repos/vllm-omni/components/configuration/_index.md b/knowledge/repos/vllm-omni/components/configuration/_index.md index bb654a9..d7080e8 100644 --- a/knowledge/repos/vllm-omni/components/configuration/_index.md +++ b/knowledge/repos/vllm-omni/components/configuration/_index.md @@ -1,7 +1,7 @@ --- title: "vLLM-Omni Configuration" created: 2026-07-16 -updated: 2026-07-31 +updated: 2026-08-05 type: index tags: [vllm-omni, components, config] sources: ["claude-workflow-starter-private@296ea45", vllm_omni/config/] @@ -11,10 +11,11 @@ sources: ["claude-workflow-starter-private@296ea45", vllm_omni/config/] - 主要源码:`vllm_omni/config/` - 跨边界入口:`vllm_omni/diffusion/data.py`、`vllm_omni/engine/arg_utils.py`、`vllm_omni/engine/stage_init_utils.py`、`vllm_omni/engine/async_omni_engine.py` -- 主要测试:`tests/config/`、`tests/test_config_factory.py`、`tests/test_diffusion_config_propagation.py`,以及各公开入口附近的配置测试 +- 主要测试:`tests/config/`、`tests/config/test_config_factory.py`、 + `tests/diffusion/test_diffusion_config_propagation.py`,以及各公开入口附近的配置测试 - 部署配置:`vllm_omni/deploy/*.yaml`,以及 `pipeline_registry.py`、 `endpoint_policy.py`、`server_settings.py`、`yaml_util.py`、`composable_parallel/` -- 源码校验:以上路径在 `main @ 807db6ef` 验证存在;机器基线见 +- 源码校验:以上路径在 `v0.26.0 @ a4ea67a2` 验证存在;机器基线见 `adapters/vllm_omni/release_baseline.yaml` ## 什么时候查这里 diff --git a/knowledge/repos/vllm-omni/components/configuration/architecture.md b/knowledge/repos/vllm-omni/components/configuration/architecture.md index d72d24a..5ac9eb0 100644 --- a/knowledge/repos/vllm-omni/components/configuration/architecture.md +++ b/knowledge/repos/vllm-omni/components/configuration/architecture.md @@ -1,10 +1,10 @@ --- title: "vLLM-Omni 配置构造架构" created: 2026-07-16 -updated: 2026-07-29 +updated: 2026-08-05 type: architecture tags: [vllm-omni, components, config] -sources: ["claude-workflow-starter-private@296ea45", vllm_omni/config/stage_config.py, vllm_omni/diffusion/data.py] +sources: ["claude-workflow-starter-private@296ea45", vllm_omni/config/stage_config.py, vllm_omni/config/config_factory.py, vllm_omni/config/omni_config.py, vllm_omni/config/composable_parallel/, vllm_omni/diffusion/data.py] --- # vLLM-Omni 配置构造架构 @@ -90,7 +90,7 @@ structured 与 legacy 可以有不同的最终对象,但不能有不同的字 ## PipelineConfig 与 deploy YAML 的具体结构 -以下事实曾在 `main @ 5c390096` 复核;源码会变化,动手前仍须刷新 live 版本。 +以下事实在 `v0.26.0 @ a4ea67a2` 复核;源码会变化,动手前仍须刷新 live 版本。 - **`PipelineConfig`**(模型的冻结 stage 拓扑)由模型的 `pipeline.py` 注册; **deploy YAML**(`vllm_omni/deploy/*.yaml`)只描述“这些 stage 怎么跑”。 diff --git a/knowledge/repos/vllm-omni/components/configuration/deploy-yaml.md b/knowledge/repos/vllm-omni/components/configuration/deploy-yaml.md index 373af52..6905373 100644 --- a/knowledge/repos/vllm-omni/components/configuration/deploy-yaml.md +++ b/knowledge/repos/vllm-omni/components/configuration/deploy-yaml.md @@ -1,7 +1,7 @@ --- title: "vLLM-Omni deploy YAML 实操" created: 2026-07-16 -updated: 2026-07-29 +updated: 2026-08-05 type: guide tags: [vllm-omni, components, config] sources: ["claude-workflow-starter-private@296ea45", vllm_omni/deploy/] @@ -11,7 +11,7 @@ sources: ["claude-workflow-starter-private@296ea45", vllm_omni/deploy/] 面向"要给模型写/改部署配置"的场景;schema 语义 owner 是 [Configuration](architecture.md)(本页不复制字段表)。 -`main @ 5c390096` 复核。 +`v0.26.0 @ a4ea67a2` 复核。 ## 何时需要 YAML,何时 CLI 就够 @@ -34,13 +34,16 @@ sources: ["claude-workflow-starter-private@296ea45", vllm_omni/deploy/] - KV 记账外分配的模型考虑 `kv_cache_memory_bytes` pin([CONF-2a](rules.md))。 - 争议以展开后最终配置为准([CONF-3a](rules.md))。 -## 代表样例(58 份 YAML 中的三类拓扑) +## 代表样例(79 份 YAML 中的三类拓扑) - 单 stage diffusion:不进 `OMNI_PIPELINES`,通常无需 YAML(引擎默认兜底),需要 固化参数时才写。 - AR+DiT 两 stage:`glm_image.yaml`、`hunyuan_image3_{ar,dit,_moe}.yaml`。 - thinker/talker(+code2wav) 多 stage:`qwen2_5_omni.yaml`(1×H100 验证)、 `qwen3_omni_moe.yaml`(2×H100 验证)、`qwen3_tts.yaml`(+ 高并发/对齐器变体)。 +- Audex 四模式各有 2B/30B YAML:`audex_{tts,tta,thinker_only,s2s}{,_30b}.yaml`; + TTS/S2S 使用 streaming codec handoff,TTA 使用同步 full-payload XCodec,30B + 必须显式选择 `_30b` 资源配置。 ## 相关 diff --git a/knowledge/repos/vllm-omni/components/configuration/omni-init-args.md b/knowledge/repos/vllm-omni/components/configuration/omni-init-args.md index e415c68..44946bd 100644 --- a/knowledge/repos/vllm-omni/components/configuration/omni-init-args.md +++ b/knowledge/repos/vllm-omni/components/configuration/omni-init-args.md @@ -31,4 +31,6 @@ sources: ["claude-workflow-starter-private@296ea45"] - 不要再依赖或推荐已不存在的 `nullify_stage_engine_defaults`。 - 增加或移动字段时,先决定它属于 orchestrator、shared 还是 per-stage engine,再修改唯一 owner 集合。 - 至少覆盖:普通 parent field 被丢弃并对非默认值告警、no-warn field 被丢弃但不告警、allowlisted field 被保留、orchestrator-only field 不泄漏、stage-local 显式值覆盖顶层 fallback。 -- 真实行为以当前 checkout 的 `tests/engine/test_arg_utils.py` 和 `tests/test_config_factory.py` 为最低回归入口;如果 example 仍引用旧 helper,只能视为待清理调用点,不能作为现行合同。 +- 真实行为以当前 checkout 的 `tests/engine/test_arg_utils.py` 和 + `tests/config/test_config_factory.py` 为最低回归入口;如果 example 仍引用旧 helper, + 只能视为待清理调用点,不能作为现行合同。 diff --git a/knowledge/repos/vllm-omni/components/configuration/pipeline-deploy-schema.md b/knowledge/repos/vllm-omni/components/configuration/pipeline-deploy-schema.md index e0b007f..dd5c9a4 100644 --- a/knowledge/repos/vllm-omni/components/configuration/pipeline-deploy-schema.md +++ b/knowledge/repos/vllm-omni/components/configuration/pipeline-deploy-schema.md @@ -1,7 +1,7 @@ --- title: "PipelineConfig 与 deploy YAML 详细 schema" created: 2026-07-16 -updated: 2026-07-29 +updated: 2026-08-05 type: guide tags: [vllm-omni, components, config] sources: ["claude-workflow-starter-private@296ea45", vllm_omni/config/stage_config.py, docs/configuration/stage_configs.md] @@ -9,7 +9,7 @@ sources: ["claude-workflow-starter-private@296ea45", vllm_omni/config/stage_conf # PipelineConfig 与 deploy YAML 详细 schema -以下事实在 `main @ 5c390096` 复核;当前稳定职责见 +以下事实在 `v0.26.0 @ a4ea67a2` 复核;当前稳定职责见 [Configuration architecture](architecture.md),官方 spec 见 `docs/configuration/stage_configs.md` (schema 全表)与 `composable_parallel.md`。 @@ -17,7 +17,7 @@ sources: ["claude-workflow-starter-private@296ea45", vllm_omni/config/stage_conf ## 双层 schema:PipelineConfig vs deploy YAML - **`PipelineConfig`**(模型的冻结 stage 拓扑)由模型的 `pipeline.py` 注册; - **deploy YAML**(`vllm_omni/deploy/*.yaml`,58 个)只描述"这些 stage 怎么跑"。 + **deploy YAML**(`vllm_omni/deploy/*.yaml`,79 个)只描述"这些 stage 怎么跑"。 未迁移模型仍走 legacy `--stage-configs-path` + `stage_args` schema (`vllm_omni/model_executor/stage_configs/*.yaml`)。 - 未显式给 `--deploy-config`/`--stage-configs-path` 时,registry 按 `model_type` @@ -50,7 +50,7 @@ sources: ["claude-workflow-starter-private@296ea45", vllm_omni/config/stage_conf ## StageConfigFactory 与 pipeline registry `StageConfigFactory`(config_factory.py:47)按 `model_type` 从 -`pipeline_registry.OMNI_PIPELINES`(~44 个 key)解析出 `PipelineConfig` 或 resolver +`pipeline_registry.OMNI_PIPELINES`(51 个 key)解析出 `PipelineConfig` 或 resolver callable(如 `resolve_qwen3_omni_pipeline`);HF `model_type` 冲突用 `hf_architectures` 消歧(如 MiMo Audio 的 HF model_type 是 qwen2);未注册模型报错 并列出可用 key(:360)。单 stage diffusion 模型**不在**该 registry(走 @@ -66,7 +66,8 @@ callable(如 `resolve_qwen3_omni_pipeline`);HF `model_type` 冲突用 `shutdown_unsupported_routes`(:65)——pipeline 可关闭自己不支持的 serving 路由。 - `composable_parallel/`:`--strategy-config` 把逐 stage 并行轴栈 (tp/dp/pp/ep/stage_replica 已接线;sp/cfg/vae_pp/hsdp 等保留位)以声明式 overlay - 叠加到合并后的 stage 上、先于 CLI override;**不能**与 legacy - `--stage-configs-path` 组合。 + 叠加到合并后的 stage 上、先于 CLI override;显式 deploy 值与 strategy 派生值冲突时 + fail fast,并在 worker 创建前检查 devices 与 `tp × dp × pp` world size;**不能**与 + legacy `--stage-configs-path` 组合。 源码会变化,具体函数与行号在改代码前必须以目标仓库当前版本为准。 diff --git a/knowledge/repos/vllm-omni/components/configuration/rules.md b/knowledge/repos/vllm-omni/components/configuration/rules.md index 5dad1d3..a89d5e8 100644 --- a/knowledge/repos/vllm-omni/components/configuration/rules.md +++ b/knowledge/repos/vllm-omni/components/configuration/rules.md @@ -1,10 +1,10 @@ --- title: "vLLM-Omni 配置开发门禁" created: 2026-07-16 -updated: 2026-07-31 +updated: 2026-08-05 type: rule tags: [vllm-omni, components, config] -sources: ["claude-workflow-starter-private@296ea45", "PR #4281", "PR #5031", "zuiho-kai/claude-workflow-starter@c217fc6", vllm_omni/config/stage_config.py, vllm_omni/config/config_factory.py, vllm_omni/config/omni_config.py] +sources: ["claude-workflow-starter-private@296ea45", "PR #4281", "PR #5031", "zuiho-kai/claude-workflow-starter@c217fc6", vllm_omni/config/stage_config.py, vllm_omni/config/config_factory.py, vllm_omni/config/omni_config.py, vllm_omni/config/composable_parallel/] --- # vLLM-Omni 配置开发门禁 @@ -87,6 +87,42 @@ sources: ["claude-workflow-starter-private@296ea45", "PR #4281", "PR #5031", "zu - 验收:每个公开 axis 都有 spec → translator → final stage config 的正向测试和 unsupported fail-fast 测试;stage 使用稳定名称。 +### CONF-4c — 新并行轴必须进入最终拓扑并由 consumer 校验 + +- 触发:新增 diffusion parallel axis、text-encoder TP、CFG/patch parallel 或把 + `strategy-config` 的声明映射到 stage runtime。 +- 强制:从 strategy spec/CLI 到 `OmniStageDiffusionParallelConfig` 和最终 process + group 展开完整链路;记录 `data × cfg × sequence × pipeline × tensor` 的设备乘积, + 并由实际 consumer 校验 rank/head divisibility、模型不支持的组合和可见设备容量。 +- 禁止:parser 接受字段后在 projection、engine kwargs 或 worker 初始化时丢失;用一个 + generic upper bound 替代模型自己的约束;用 recipe 示例或单纯 dataclass 构造代替真实 + topology/consumer 证据。 +- 验收:正向测试断言新值可从最终 stage config 读回;负向测试覆盖不整除的 head/rank、 + `cfg_parallel_size` 不适用的模型和设备不足,并在 worker 创建前失败。 + +### CONF-4d — strategy、CLI 与 pipeline-wide load balancing 必须保留唯一 owner + +- 触发:`strategy_config`、legacy YAML、`--omni-lb-policy` 或 stage replica/load + balancing 同时提供策略。 +- 强制:registry-backed strategy 才进入 translator;legacy 路径明确 warning 并保持 + 原语义;合并顺序固定为 deploy/pipeline → strategy → CLI,CLI 的显式冲突必须可见, + pipeline-wide load-balancer 只由 orchestrator 拥有并传入各 stage。 +- 禁止:接受 legacy `strategy_config` 后静默声称已生效;把 CLI 的默认值误当显式覆盖; + 每个 stage 各自构造一个相互冲突的全局 load balancer。 +- 验收:覆盖 registry、legacy、CLI 非默认值和冲突值,断言最终拓扑、设备乘积和单一 + load-balancer consumer;策略生效前若设备不足必须 fail fast。 + +### CONF-5b — `trust_remote_code` 的未指定值必须保留 deploy 优先级 + +- 触发:CLI、structured factory、legacy stage factory 或 deploy YAML 同时提供 + `trust_remote_code`。 +- 强制:调用方显式 `True`/`False` 才覆盖 per-stage deploy 值;未指定用 `None` 表示并 + 保留 YAML 值,HF config resolution 才把 `None` 收敛为安全的实际 bool。 +- 禁止:把 `store_true` 的 absent-False 当成显式 False 无条件写入 override;structured + 与 legacy 各自实现 precedence;在末端用默认值静默覆盖用户明确的 False。 +- 验收:覆盖“未指定、显式 True、显式 False”三态,并从 structured/legacy 两条路径 + 断言最终 stage config;`with_trust_remote_code_override` 是唯一合并入口。 + ### CONF-4b — 标准与 headless 启动必须解析出同一拓扑 - 触发:CLI、headless serve、offline entrypoint 或 engine factory 新增/转发配置字段。 diff --git a/knowledge/repos/vllm-omni/components/diffusion/_index.md b/knowledge/repos/vllm-omni/components/diffusion/_index.md index 572eef8..6b02615 100644 --- a/knowledge/repos/vllm-omni/components/diffusion/_index.md +++ b/knowledge/repos/vllm-omni/components/diffusion/_index.md @@ -1,7 +1,7 @@ --- title: "Diffusion" created: 2026-07-10 -updated: 2026-07-31 +updated: 2026-08-05 type: index tags: [vllm-omni, components, diffusion] sources: [] @@ -10,7 +10,8 @@ sources: [] # Diffusion - 源码入口:`vllm_omni/diffusion/` 全树,含 16 个子模块:attention、cache、distributed、executor、hooks、layers、lora、model_loader、models、offloader、postprocess、profiler、quantization、sched、utils、worker -- 源码校验:以上子模块均已在 `main @ 807db6ef` 验证存在 +- 源码校验:以上子模块均已在 `v0.26.0 @ a4ea67a2` 验证存在;本轮新增/扩展的 + async output、distributed layerwise offload 和 MiniMax H3 仍按各自模型/机制规则审查 - 主要职责:多个 diffusion 模型共用的 pipeline、执行循环、scheduler 接入和运行机制 ## 什么时候查这里 diff --git a/knowledge/repos/vllm-omni/components/diffusion/rules.md b/knowledge/repos/vllm-omni/components/diffusion/rules.md index 7a573e7..c9cbfc0 100644 --- a/knowledge/repos/vllm-omni/components/diffusion/rules.md +++ b/knowledge/repos/vllm-omni/components/diffusion/rules.md @@ -1,10 +1,10 @@ --- title: "Diffusion 共享规则" created: 2026-07-20 -updated: 2026-07-31 +updated: 2026-08-05 type: rule tags: [vllm-omni, components, diffusion] -sources: ["PR #4341", "PR #5001", "PR #5087", "PR #5088", "PR #5136", vllm_omni/diffusion/worker/diffusion_model_runner.py, vllm_omni/diffusion/model_loader/diffusers_loader.py, vllm_omni/diffusion/distributed/hsdp.py] +sources: ["PR #4341", "PR #5001", "PR #5087", "PR #5088", "PR #5136", vllm_omni/diffusion/worker/diffusion_model_runner.py, vllm_omni/diffusion/model_loader/diffusers_loader.py, vllm_omni/diffusion/distributed/hsdp.py, vllm_omni/diffusion/executor/multiproc_executor.py, vllm_omni/diffusion/offloader/] confidence: high --- @@ -57,6 +57,20 @@ confidence: high 到达 consumer。Cosmos3 的落地约束见 [Cosmos3 规则](../../models/cosmos3/rules.md)。 ^[PR #5001] +### DIFF-1c — async output 的 compute 与 output-ready 必须分成两个可观察阶段 + +- 触发:把 diffusion D2H/SHM packing 移到 worker 后台线程,或修改 + `ResultPumpThread`、`collective_rpc`、batch output split。 +- 强制:GPU compute 完成只释放下一次 forward;最终输出必须等 side-stream event、D2H + 和 SHM packing 完成后再 resolve。`result_mq` 由唯一 pump reader 分发,batch 结果按 + request 映射拆分;step-mode 保留同步路径。 +- 禁止:在 compute-done 时把未完成的 device tensor 当成最终 artifact;让多个线程同时 + 消费 result queue;用同步 `.cpu()` 掩盖 stream ordering 或让下一 step 重写源 buffer。 +- 验收:分别覆盖 compute-done、output-ready、RPC error、batch split 和 queue cleanup; + 在非 CUDA/step-mode 下证明同步 fallback 仍可构造。首批测试看 + `tests/diffusion/test_async_output_worker.py`、`test_result_pump.py` 和 + `tests/diffusion/test_ipc_async.py`。 + ## Checkpoint 与分布式加载 ### DIFF-2a — checkpoint remap 必须追到已注册且真实消费的目标 @@ -89,6 +103,18 @@ confidence: high 命中/排除集合,并验证 meta-device parameter 不会被提前 move。FLUX.2 的具体边界见 [FLUX.2 规则](../../models/flux2/rules.md)。 ^[PR #5136] +### DIFF-2d — distributed layerwise offload 只能选择一个一致的权重分片合同 + +- 触发:新增 `enable_distributed_layerwise_offload`、`dlo_use_allgather`、模型 + `OffloadPlan` 或 block discovery。 +- 强制:明确 rank 本地 CPU shard、固定双 GPU buffer、H2D/AllGather stream 和模型 + block list 的 owner;`dlo_use_allgather=false` 走每 rank full-weight 路径,开启 + AllGather 时不得再叠加已 DTensor-sharded 的 HSDP 参数。 +- 禁止:用 heuristic 找不到 block 时静默假装 offload 已启用;对 HSDP 参数二次分片; + 把 CPU 内存、AllGather 同步和设备显存开销隐藏在一个泛化的 `enable_cpu_offload` 开关。 +- 验收:CPU/Gloo 单 rank 覆盖 DTensor wrapper、shard padding、双 buffer 和 disable + cleanup;配置测试覆盖 AllGather+HSDP 的明确失败及 no-AllGather 的允许路径。 + ## 质量阈值与资源辅助 ### DIFF-3a — 质量阈值必须由完全相同的测试 case 产生 diff --git a/knowledge/repos/vllm-omni/components/distributed/_index.md b/knowledge/repos/vllm-omni/components/distributed/_index.md index c72febe..864aac9 100644 --- a/knowledge/repos/vllm-omni/components/distributed/_index.md +++ b/knowledge/repos/vllm-omni/components/distributed/_index.md @@ -1,7 +1,7 @@ --- title: "Distributed(跨 stage 通信与数据搬运)" created: 2026-07-16 -updated: 2026-07-31 +updated: 2026-08-05 type: index tags: [vllm-omni, components, distributed] sources: [vllm_omni/distributed/omni_connectors/, vllm_omni/distributed/omni_coordinator/, docs/design/feature/disaggregated_inference.md] @@ -14,7 +14,7 @@ sources: [vllm_omni/distributed/omni_connectors/, vllm_omni/distributed/omni_coo `vllm_omni/distributed/omni_coordinator/`(协调器与 load balancer) - 知识面另覆盖跨 stage ZMQ 路由/端口分配(`vllm_omni/engine/stage_engine_startup.py::OmniMasterServer`) ——组件划分服务知识归属,与 manifest 运行时粒度不同 -- 源码校验:以上路径与下列锚点均已在 `main @ 807db6ef` 验证存在: +- 源码校验:以上路径与下列锚点均已在 `v0.26.0 @ a4ea67a2` 验证存在: `OmniConnectorBase`(connectors/base.py:12)、`OmniKVTransferManager` (kv_transfer_manager.py:341)、`LoadBalancer` 三实现(load_balancer.py:39/64/74/102)、 `OmniMasterServer._allocate_route_locked`(stage_engine_startup.py:254) @@ -52,3 +52,4 @@ sources: [vllm_omni/distributed/omni_connectors/, vllm_omni/distributed/omni_coo | 已修过的 connector/端口产品坑 | [connector pitfalls](connector-pitfalls.md) | | 选择和配置 connector backend | [connector backends](connector-backends.md) | | 跨 stage `async_chunk` 流式语义 | [async chunk](async-chunk.md) | +| TP KV receive consensus、chunk boundary 与 active-window 合同 | [distributed rules](rules.md) | diff --git a/knowledge/repos/vllm-omni/components/distributed/rules.md b/knowledge/repos/vllm-omni/components/distributed/rules.md new file mode 100644 index 0000000..b028b8b --- /dev/null +++ b/knowledge/repos/vllm-omni/components/distributed/rules.md @@ -0,0 +1,36 @@ +--- +title: "Distributed 传输规则" +created: 2026-08-05 +updated: 2026-08-05 +type: rule +tags: [vllm-omni, components, distributed] +sources: [vllm_omni/distributed/omni_connectors/adapter.py, vllm_omni/distributed/omni_connectors/kv_transfer_manager.py, vllm_omni/distributed/omni_connectors/transfer_adapter/chunk_transfer_adapter.py, tests/distributed/omni_connectors/test_kv_recv_tp_consensus.py, tests/distributed/omni_connectors/test_chunk_transfer_adapter.py] +confidence: high +--- + +# Distributed 传输规则 + +只有 `DIST-数字字母` 是本页可审计规则 ID。connector backend 的选择和端口分配见 +[Distributed architecture](architecture.md) 与 [connector pitfalls](connector-pitfalls.md)。 + +## DIST-1a — TP KV receive 必须做全 rank 一致的成功/放弃决定 + +- 触发:纯 TP stage 接收 KV、CFG companion 参与 KV transfer,或 connector receive + 只在部分 rank 得到 state。 +- 强制:只在 TP process group 内交换 receive state;任一 rank 发现 metadata、shape 或 + payload 不一致时,所有 rank 都走同一个 no-KV fallback,不能让部分 rank 继续 collective。 +- 禁止:用本地 `if` 丢掉 KV 后仍让其他 rank进入 receive;使用 global world group 代替 + stage 的 TP group;把 companion 当成普通 stage-0-final request 静默跳过。 +- 验收:模拟单 rank divergence,断言所有 TP rank 都放弃本次 KV 并保持后续 collective + 可继续;覆盖普通和 CFG companion 角色。 + +## DIST-1b — chunk transfer 要区分 upstream exhaustion 与本 stage 完成 + +- 触发:async-chunk/full-payload connector 传递空 segment、最后一个 chunk 或 + `WAITING_FOR_CHUNK`/active-window 状态。 +- 强制:保留空 segment 的边界语义,区分 upstream 已耗尽和当前 stage 已生成完成;收到 + 后续 chunk 时刷新 prefill/request state,并让 active-window admission 继续推进。 +- 禁止:把空 segment 当作完成信号;用本 stage 的 generation completion 终止上游;在 + connector 没有新数据时无限占用 active window 或静默丢掉 terminal update。 +- 验收:覆盖非空→空→非空、upstream exhaustion、stage completion、重复 terminal + update 和窗口恢复;首批测试看 `test_chunk_transfer_adapter.py`。 diff --git a/knowledge/repos/vllm-omni/components/model-executor/_index.md b/knowledge/repos/vllm-omni/components/model-executor/_index.md index 947529b..c57cdbc 100644 --- a/knowledge/repos/vllm-omni/components/model-executor/_index.md +++ b/knowledge/repos/vllm-omni/components/model-executor/_index.md @@ -1,7 +1,7 @@ --- title: "Model Executor" created: 2026-07-10 -updated: 2026-07-31 +updated: 2026-08-05 type: index tags: [vllm-omni, components, model-executor] sources: [] @@ -10,7 +10,7 @@ sources: [] # Model Executor - 源码入口:`vllm_omni/model_executor/`(layers、model_loader、models、stage_input_processors)、`vllm_omni/worker/`(gpu_*_worker、gpu_*_model_runner、mixins)、`vllm_omni/inputs/`(runner 输入预处理:data.py、preprocess.py)和设备平台层 `vllm_omni/platforms//` -- 源码校验:以上路径均已在 `main @ 807db6ef` 验证存在;stage 配置已经迁移到 +- 源码校验:以上路径均已在 `v0.26.0 @ a4ea67a2` 验证存在;stage 配置已经迁移到 `vllm_omni/deploy/`,NPU/XPU 等平台可继续拥有自己的 worker 覆盖 - 测试入口:共享 runner 行为看 `tests/worker/`,具体模型 consumer 看 `tests/model_executor/` - 主要职责:AR/LLM stage、stage 配置、并行与设备启动、runner 到模型的输入预处理合同和跨阶段数据桥接 diff --git a/knowledge/repos/vllm-omni/components/model-executor/architecture.md b/knowledge/repos/vllm-omni/components/model-executor/architecture.md index 6cfac30..aa19679 100644 --- a/knowledge/repos/vllm-omni/components/model-executor/architecture.md +++ b/knowledge/repos/vllm-omni/components/model-executor/architecture.md @@ -1,10 +1,10 @@ --- title: "Model Executor 共享架构" created: 2026-07-10 -updated: 2026-07-16 +updated: 2026-08-05 type: architecture tags: [vllm-omni, components, model-executor] -sources: [vllm_omni/worker/gpu_model_runner.py, vllm_omni/config/stage_config.py, vllm_omni/engine/stage_runtime.py, tests/worker/test_omni_gpu_model_runner.py] +sources: [vllm_omni/worker/gpu_model_runner.py, vllm_omni/config/stage_config.py, vllm_omni/config/omni_config.py, vllm_omni/engine/stage_runtime.py, tests/worker/test_omni_gpu_model_runner.py] --- # Model Executor 共享架构 @@ -23,7 +23,7 @@ sources: [vllm_omni/worker/gpu_model_runner.py, vllm_omni/config/stage_config.py ## 当前源码职责锚点 调查前用 current main 验证这些符号仍存在;路径变化时沿调用方更新本页,不保留失效副本。 -(2026-07-16 在 `main @ 238fc0a6`(此前亦在 `dev/vllm-align @ 4f2b32c` 验证,结果一致) 复核:下列 6 个锚点文件与 +(2026-08-05 在 `v0.26.0 @ a4ea67a2` 复核:下列 6 个锚点文件与 `build_stage_runtime_overrides`、`_preprocess` 符号全部存在。) - 全局 CLI、deploy YAML 和 per-stage override 合并:`vllm_omni/config/stage_config.py` 的 `build_stage_runtime_overrides` 及其 config factory 调用方。 diff --git a/knowledge/repos/vllm-omni/components/model-executor/rules.md b/knowledge/repos/vllm-omni/components/model-executor/rules.md index 3a5eefa..f79250e 100644 --- a/knowledge/repos/vllm-omni/components/model-executor/rules.md +++ b/knowledge/repos/vllm-omni/components/model-executor/rules.md @@ -1,10 +1,10 @@ --- title: "Model Executor 规则" created: 2026-07-10 -updated: 2026-07-31 +updated: 2026-08-05 type: rule tags: [vllm-omni, components, model-executor] -sources: [vllm_omni/worker/gpu_model_runner.py, tests/worker/test_omni_gpu_model_runner.py, vllm_omni/config/stage_config.py, vllm_omni/engine/stage_runtime.py, vllm_omni/engine/stage_engine_startup.py, "PR #3642", "PR #4730", "claude-workflow-starter-private@09dca46"] +sources: [vllm_omni/worker/gpu_model_runner.py, vllm_omni/worker/gpu_ar_model_runner.py, vllm_omni/engine/stage_init_utils.py, tests/worker/test_omni_gpu_model_runner.py, vllm_omni/config/stage_config.py, vllm_omni/config/omni_config.py, vllm_omni/engine/stage_runtime.py, vllm_omni/engine/stage_engine_startup.py, vllm_omni/experimental/fullduplex/, tests/e2e/features/fullduplex/, "PR #3642", "PR #4730", "claude-workflow-starter-private@09dca46"] --- # Model Executor 规则 @@ -39,6 +39,44 @@ sources: [vllm_omni/worker/gpu_model_runner.py, tests/worker/test_omni_gpu_model - 禁止:不能因为严格校验开始报错,就把报错字段加入核心 dataclass、projection、known-fields 或专门白名单;不能接受字段后在分区时丢弃;不能为同一语义新增两个核心字段再补优先级;不能用“构造成功”“字段不在最终 payload”或只有测试 fixture 使用来证明字段必要。 - 验收:每个新增可接受字段都必须有 PR 前已存在或本次明确新增的真实 consumer,并有从公开入口到 consumer 的正向断言;无 consumer 或 owner 不在 stage runtime 的字段必须在入口报错;别名与 canonical 同时出现必须报冲突。最终 diff 中新增的 schema 字段数量应与 producer-consumer 表逐项一致。 +### EXEC-3b — stage loader metadata 与 stateful chunk 能力必须一同到达 consumer + +- 触发:新增 per-stage `model_subdir`/`tokenizer_subdir`、模型 architecture override, + 或模型在 async chunk 间保留执行状态。 +- 强制:从 `PipelineConfig`/deploy 合并结果把 checkpoint/tokenizer 子目录和 + `retains_state_across_chunks` 传入最终 `OmniStageModelConfig`/runner;空的 + `model_arch` 表示使用 checkpoint 自带 architectures,而不是一个空 override。 + Stateful chunk 必须与 scheduler 的容量计数和 connector requeue 合同一起审查。 +- 禁止:只在 legacy YAML 或 model helper 中记录子目录;用空字符串覆盖 checkpoint + architecture;让 scheduler 看不到仍占用 runner slot 的 parked request。 +- 验收:structured 与 legacy stage config 都能读回子目录/状态字段;Audex 等多子目录 + checkpoint 做 loader smoke;stateful async-chunk 测试证明容量上限、requeue 和 cleanup。 + +### EXEC-4a — full-duplex chunk metadata 必须按 span 隔离且可安全序列化 + +- 触发:`experimental/fullduplex` 修改 PCM/audio span、force-listen、rollback、 + resumable append、runtime-control 或 silence continuation。 +- 强制:把 speech/PCM metadata 与 rollback 状态绑定到具体 span;resumable append 重新 + arm 所需 EOS 但不得重复 turn EOS;runtime-control 输出先转换 dataclass/enum 等为 + JSON-safe 值,并在等待后重新检查 session/request 是否仍然有效。 +- 禁止:用上一 chunk 的 force-listen 或 speech marker 污染不规则下一 chunk;把 stale + session 的 silence continuation 继续发送;直接把 Python runtime object 放进控制消息。 +- 验收:覆盖不规则 PCM span、rollback、resumable append、stale continuation 和控制 + redaction;首批测试看 `tests/e2e/features/fullduplex/` 下的 input、runtime adapter + boundary 与 runtime-control redaction。 + +### EXEC-4b — shared runner 必须保留 resolved model architecture 与 model-owned hooks + +- 触发:模型 architecture override 为空、checkpoint 含多个子目录,或模型实现拥有 + native duplex/sampling policy。 +- 强制:空 `model_arch` 回退到 checkpoint architectures;loader 先按 resolved + architecture 解析 connector/checkpoint 子目录;共享 runner 只提供 typed rows 和 + hook seam,真正的 model-owned sampling/turn-boundary policy 由模型 consumer 执行。 +- 禁止:用空 override 覆盖 checkpoint metadata;只按默认架构初始化 connector;用 + generic runner policy 替换 MiniCPM/Audex 等模型自己的 native policy。 +- 验收:覆盖 blank override、多子目录 checkpoint、hook 调用顺序和 mixed-batch + request-local metadata;初始化失败必须在 scheduler/worker 继续运行前暴露。 + ## Runner 到模型的预处理合同 - 触发条件:修改或排查 runner `_preprocess` 的逐请求 metadata 生产、phase 判定、normal/batched preprocess 选择、MTP 路由条件,或多模型共用的输入预处理合同。 diff --git a/knowledge/repos/vllm-omni/components/scheduler/_index.md b/knowledge/repos/vllm-omni/components/scheduler/_index.md index bb192bd..ce5d9b8 100644 --- a/knowledge/repos/vllm-omni/components/scheduler/_index.md +++ b/knowledge/repos/vllm-omni/components/scheduler/_index.md @@ -1,7 +1,7 @@ --- title: "Scheduler(AR/生成请求调度)" created: 2026-07-16 -updated: 2026-07-31 +updated: 2026-08-05 type: index tags: [vllm-omni, components, scheduler] sources: [vllm_omni/core/sched/omni_ar_scheduler.py, vllm_omni/core/prefix_cache.py, docs/design/module/ar_module.md] @@ -11,7 +11,7 @@ sources: [vllm_omni/core/sched/omni_ar_scheduler.py, vllm_omni/core/prefix_cache - 源码入口:`vllm_omni/core/sched/`(`omni_ar_scheduler.py`、`omni_generation_scheduler.py`、 `omni_scheduler_mixin.py`、`omni_scheduling_coordinator.py`)和 `vllm_omni/core/prefix_cache.py` -- 源码校验:以上路径与下列类均已在 `main @ 807db6ef` 验证存在:`OmniARScheduler`(:50)、 +- 源码校验:以上路径与下列类均已在 `v0.26.0 @ a4ea67a2` 验证存在:`OmniARScheduler`(:50)、 `OmniARAsyncScheduler`(:928)、`KVCacheTransferData`(:40)、`OmniGenerationScheduler`(:42)、 `OmniSchedulerMixin`(:40)、`OmniTensorPrefixCache`(prefix_cache.py:33) - 官方设计文档:`docs/design/module/ar_module.md`(继承关系、请求流转图) diff --git a/knowledge/repos/vllm-omni/components/scheduler/architecture.md b/knowledge/repos/vllm-omni/components/scheduler/architecture.md index fd84477..e358dae 100644 --- a/knowledge/repos/vllm-omni/components/scheduler/architecture.md +++ b/knowledge/repos/vllm-omni/components/scheduler/architecture.md @@ -1,15 +1,15 @@ --- title: "Scheduler 共享架构" created: 2026-07-16 -updated: 2026-07-16 +updated: 2026-08-05 type: architecture tags: [vllm-omni, components, scheduler] -sources: [docs/design/module/ar_module.md, vllm_omni/core/sched/omni_ar_scheduler.py, vllm_omni/core/sched/omni_scheduling_coordinator.py, vllm_omni/core/prefix_cache.py, vllm_omni/worker/gpu_ar_model_runner.py] +sources: [docs/design/module/ar_module.md, vllm_omni/core/sched/omni_ar_scheduler.py, vllm_omni/core/sched/omni_generation_scheduler.py, vllm_omni/core/sched/omni_scheduling_coordinator.py, vllm_omni/core/prefix_cache.py, vllm_omni/worker/gpu_ar_model_runner.py] --- # Scheduler 共享架构 -以下事实在 `main @ 5c390096` 复核;官方叙述见 +以下事实在 `v0.26.0 @ a4ea67a2` 复核;官方叙述见 `docs/design/module/ar_module.md`(含继承 classDiagram 与请求流转 flowchart)。 ## 继承链(对 vLLM 的扩展方式) @@ -53,6 +53,13 @@ token,前缀行从缓存重建后拼入跨 stage payload。消费端合同: `higgs_audio_v3_talker.py:236` 显式声明 False)。该机制的失败模式与硬规则见 [rules](rules.md) 的 `SCHED-1a`。 +在目标版本,`OmniGenerationScheduler` 还区分 +`retains_state_across_chunks`:等待 connector chunk 的 request 仍占用 model-runner +capacity,并在 full-payload chunk 到达后重新进入可调度队列。AR scheduler 在消费 +sampled-token logprobs 前完成 request-local 合同校验;stage-0-final 的普通请求可以 +跳过下游 KV,但 companion 等显式带 `omni_force_kv_transfer` 的请求不能走该 shortcut。 +这些跨层语义分别见 [scheduler rules](rules.md) 的 `SCHED-5a`–`SCHED-5c`。 + ## 与 orchestrator 的边界 调度器只负责单 stage 内的请求生命周期与跨 stage 载荷的调度面;逻辑请求在 diff --git a/knowledge/repos/vllm-omni/components/scheduler/rules.md b/knowledge/repos/vllm-omni/components/scheduler/rules.md index 9233a81..81d5f3b 100644 --- a/knowledge/repos/vllm-omni/components/scheduler/rules.md +++ b/knowledge/repos/vllm-omni/components/scheduler/rules.md @@ -1,10 +1,10 @@ --- title: "Scheduler 规则" created: 2026-07-16 -updated: 2026-07-31 +updated: 2026-08-05 type: rule tags: [vllm-omni, components, scheduler] -sources: ["vllm-omni-rebase-agent@122a9468:agent/skills/fix-talker-truncated-prefill-prefix-cache-key-cap/SKILL.md", "vllm-omni-rebase-agent@122a9468:agent/skills/gpu-hang-low-max-num-batched-tokens/SKILL.md", vllm_omni/worker/gpu_ar_model_runner.py, vllm_omni/core/prefix_cache.py, "PR #4106"] +sources: ["vllm-omni-rebase-agent@122a9468:agent/skills/fix-talker-truncated-prefill-prefix-cache-key-cap/SKILL.md", "vllm-omni-rebase-agent@122a9468:agent/skills/gpu-hang-low-max-num-batched-tokens/SKILL.md", vllm_omni/worker/gpu_ar_model_runner.py, vllm_omni/core/prefix_cache.py, vllm_omni/core/sched/omni_ar_scheduler.py, vllm_omni/core/sched/omni_generation_scheduler.py, "PR #4106"] --- # Scheduler 规则 @@ -24,6 +24,8 @@ PR 描述先命中下表,再打开对应规则组和首批源码;changed fil | `max_num_batched_tokens`、prefill throttle、低预算 GPU hang | SCHED-2a | `core/sched/omni_ar_scheduler.py::OmniARScheduler.schedule`;`core/sched/omni_generation_scheduler.py::OmniGenerationScheduler.schedule`;触发它的 deploy/test 配置 | | vLLM bump、scheduler rebase、KV connector stats | SCHED-3a | 两个 scheduler 的 `schedule` / `update_from_output` 与 live upstream `vllm/v1/core/sched/` | | side-stream D2H、pinned host tensor、源 buffer 复用 | SCHED-4a/4b | `worker/gpu_ar_model_runner.py::_copy_tensor_payload_to_cpu`、`_get_or_create_omni_payload_copy_stream`;`core/prefix_cache.py` 的 async copy 路径 | +| sampled-token logprobs、spec decode trim、request-local output error | SCHED-5a | `core/sched/omni_ar_scheduler.py::_slice_sampled_logprobs`、`update_from_output` | +| stateful async chunk、full-payload input、KV cleanup | SCHED-5b/5c | `core/sched/omni_generation_scheduler.py`、`omni_scheduling_coordinator.py`、`omni_ar_scheduler.py::_free_request` | 若描述只写模型症状,先从模型 owner 找到 payload producer/consumer;只有实际断点落在调度、 prefix cache 或 copy lifetime 时才把 Scheduler 加为 owner。 @@ -146,6 +148,52 @@ modules=[online_serving, worker_runner],status=active,run_count=38,2026-06 - 验收:CPU-only 环境能构造并走同步 fallback;CUDA 环境仍使用 pinned + async,两个 分支产生相同内容。 ^[PR #4106] +## SCHED-5a — sampled-token logprobs 先校验再修改 request + +- 触发:AR scheduler 消费 model-runner sampled tokens 和 `num_logprobs`,尤其是 + spec-decode 或 batched output slicing。 +- 强制:按 request 切出 logprob rows,校验二维 shape、行数、第一 token 对齐和有限值; + 失败只将当前 request 置为 `FINISHED_ERROR`,不能先 append token 或污染同批 request。 + stop/EOS 截断后再按最终 emitted token 数重新 slice。 +- 禁止:`logprobs` 缺失时静默跳过;用 truthiness 判断合法空输出;把 runner 的错位 + row 当作用户请求成功。 +- 验收:覆盖 missing、wrong shape/count、token misalignment、NaN、spec-decode trim + 和同批健康 request 继续完成。 + +## SCHED-5b — stateful async-chunk request 必须占用调度容量 + +- 触发:模型在等待下一 chunk 时保留 runner state,或 full-payload connector 负责把下 + 一段重新送入 generation stage。 +- 强制:`retains_state_across_chunks` 为真时,把 connector 中等待 chunk 的 request + 计入 `max_num_seqs`;full-payload consumer 在 chunk 到达时重新进入 waiting queue, + 不要被 base scheduler 提前停放。 +- 禁止:只按 `running` list 计数导致超额 admission;把 connector-fed chunk 当成 API + streaming update;仅在 abort 路径释放 receiver。 +- 验收:mixed batch 覆盖等待 chunk 的容量上限、full-payload requeue、normal finish + 的 receiver cleanup 和 abort/replica-loss cleanup。 + +## SCHED-5c — stage-0 final request 的 KV transfer 例外必须显式标记 + +- 触发:stage-0 的 text/final 输出通常不需要下游 KV,但 CFG companion 或其他终端 + payload 仍需要复用该 cache。 +- 强制:用 `omni_force_kv_transfer` 等 request metadata 明确覆盖“final stage=0”的 + shortcut;scheduler、engine 和 companion tracker 使用同一个标记语义。 +- 禁止:按 final stage id 静默跳过所有 stage-0 KV;只在 companion 的 producer 设置 + 标记而不测试 scheduler 的 transfer decision。 +- 验收:普通 stage-0-final request 不传 KV,标记 request 传 KV,且 metadata 在 + `OmniARScheduler._request_omits_kv_transfer_to_next_stage` 中可读回。 + +## SCHED-5d — CFG companion 必须成对推进,缺失时有限降级并终止等待 + +- 触发:Audex/扩散 CFG companion、双 waiting queue、不同 chunk 进度或 companion + abort/split。 +- 强制:不完整 pair 暂存并按 parent/companion 对齐进度;完成时一起收敛,缺失或拆分 + 的 companion 只能走显式、有界的 fallback,并清理两侧状态。 +- 禁止:只推进 parent 让 companion 永久停在 waiting;把缺 companion 当普通 batch + 请求静默放行;用无限等待掩盖 replica loss 或 abort。 +- 验收:覆盖完整 pair、不同进度、missing/split、parent abort 和 companion abort, + 断言请求不会挂死、错误归属保持 request-local、队列和 connector state 都释放。 + ## 相关 - 机制与边界见 [architecture](architecture.md);跨 stage 数据面见 diff --git a/knowledge/repos/vllm-omni/components/serving/_index.md b/knowledge/repos/vllm-omni/components/serving/_index.md index 072bf42..de66d3f 100644 --- a/knowledge/repos/vllm-omni/components/serving/_index.md +++ b/knowledge/repos/vllm-omni/components/serving/_index.md @@ -1,7 +1,7 @@ --- title: "Serving" created: 2026-07-10 -updated: 2026-07-31 +updated: 2026-08-05 type: index tags: [vllm-omni, components, serving] sources: [] @@ -10,7 +10,7 @@ sources: [] # Serving - 主要源码入口:`vllm_omni/entrypoints/`(cli、openai、openpi 及 omni/async_omni 入口)和 `vllm_omni/engine/`(orchestrator、stage engine core、stage pool/runtime、output processor) -- 源码校验:以上路径均已在 `main @ 807db6ef` 验证存在 +- 源码校验:以上路径均已在 `v0.26.0 @ a4ea67a2` 验证存在 - 主要职责:用户入口、请求解析、在线服务和 engine 边界 ## 什么时候查这里 diff --git a/knowledge/repos/vllm-omni/components/serving/rules.md b/knowledge/repos/vllm-omni/components/serving/rules.md index 8aa99a5..1e6d88b 100644 --- a/knowledge/repos/vllm-omni/components/serving/rules.md +++ b/knowledge/repos/vllm-omni/components/serving/rules.md @@ -1,10 +1,10 @@ --- title: "Serving 规则" created: 2026-07-20 -updated: 2026-07-31 +updated: 2026-08-05 type: rule tags: [vllm-omni, components, serving] -sources: ["PR #3576", "PR #4718", "PR #4834", "PR #4905", "PR #4912", "PR #5157", "claude-workflow-starter-private@09dca46", "zuiho-kai/claude-workflow-starter@c217fc6", vllm_omni/entrypoints/async_omni.py, vllm_omni/entrypoints/openai/diffusion_request_utils.py, vllm_omni/entrypoints/openai/serving_speech.py, vllm_omni/metrics/prometheus.py] +sources: ["PR #3576", "PR #4718", "PR #4834", "PR #4905", "PR #4912", "PR #5157", "claude-workflow-starter-private@09dca46", "zuiho-kai/claude-workflow-starter@c217fc6", vllm_omni/entrypoints/async_omni.py, vllm_omni/entrypoints/openai/diffusion_request_utils.py, vllm_omni/entrypoints/openai/serving_speech.py, vllm_omni/engine/orchestrator.py, vllm_omni/engine/cfg_companion_tracker.py, vllm_omni/metrics/prometheus.py] confidence: high --- @@ -36,6 +36,7 @@ confidence: high | `chat-multimodal-contract` | chat template kwargs、SDK flatten、text/audio response shape | `SERV-4c` + 命中模型规则 | | `endpoint-capability` | endpoint restriction、unsupported route、公开 400 | `SERV-4c`, `SERV-4d` | | `engine-lifecycle` | sleep/wake、partial stage/tag、ACK、generation admission | `SERV-5a`, `SERV-5b` | +| `full-duplex` | duplex opt-in、stage prewarm/fence、async-chunk、CFG companion lifecycle | `SERV-6a`–`SERV-6c` | | `request-contract` | 请求字段、来源、冲突、dispatcher、consumer view | `SERV-4a`, `SERV-4b`, `SERV-4c`, `SERV-4d`, `SERV-4e`, `SERV-4f`, `SERV-4g`, `SERV-4h` | | `author-routing` | 只供 Direct reviewer 导航,不作为 finding 规则 | `SERV-0a`, `SERV-0b` | @@ -192,5 +193,42 @@ confidence: high sleep → wake → generate,不支持的 stage 在调用 worker 前明确拒绝。 ^[PR #4834] ^[PR #4905] ^[PR #4912] +## Full-duplex 与 CFG companion 生命周期 + +### SERV-6a — full-duplex 首次 stage submit 必须预热 async-chunk topology + +- 触发:full-duplex stage port、双工会话、async-chunk 或 stage fence 发生变化。 +- 强制:stage-0 的首次 `submit_initial` 在 async-chunk 开启时预热后续 stage 的 + runtime/pool;每个 stage 保留自己的 fence、submit timestamp 和 request state。 +- 禁止:等到 duplex audio/control 事件真正抵达才首次启动下游 stage;用一个全局 fence + 表示多 stage readiness;prewarm 失败后仍发布 generation-ready。 +- 验收:覆盖首次 stage-0 submit、后续 stage submit、重复 update 和 prewarm failure; + CPU/mock 测试必须证明 request state 与 stage pool 生命周期一致。 + +### SERV-6b — CFG companion 输出要非破坏性聚合并在所有 teardown 路径清理 + +- 触发:CFG companion、deferred parent、streaming terminal update、abort、stage error + 或 replica loss。 +- 强制:parent re-submit 前 companion outputs 可重复读取;parent 只在全部 companion + 完成后释放;companion error/abort 必须给 deferred parent 发送结构化错误,并由统一 + cleanup 路径清理 parent、companion、tracker 和 stage-pool binding。 +- 禁止:用 destructive `pop` 让第二次 terminal update 丢 companion;只清 parent 不清 + companion;companion 永久等待时静默挂起 parent。 +- 验收:覆盖正常全量完成、重复读取、companion error、parent abort、companion abort + 和 replica loss,断言没有遗留 tracker state 或未完成的 deferred parent。 + +### SERV-6c — duplex capability gate 必须区分未启用与配置失败 + +- 触发:`session_mode=duplex`、`/v1/realtime?duplex=true`、runtime extension 或 + duplex deploy configuration 发生变化。 +- 强制:入口只在明确的 duplex opt-in 下建立 handler;能力检查必须保留结构化配置 + 错误与“不支持该模式”的区别,session、request、stage fence、lease 和 ordered + mailbox 的生命周期必须在 disconnect、abort、expiry 与 replica loss 时一起终止。 +- 禁止:把所有 config-load exception 折叠成 `duplex unavailable`;用普通 streaming + handler 冒充持久 session;用全局 fence 或无界 pending input 替代 per-session bound。 +- 验收:覆盖未 opt-in、有效 opt-in、malformed config、stale fence、lease expiry、 + disconnect 和 terminal response ordering;控制消息中的 dataclass/enum/tensor 必须 + 在发送前变成 JSON-safe 值。 + 请求到 engine 的边界见 [Serving architecture](architecture.md);公开协议通用检查见 [review contracts](../../../../general/review/guides/reviewer-lens-contracts.md)。 diff --git a/knowledge/repos/vllm-omni/docs/design-doc-map.md b/knowledge/repos/vllm-omni/docs/design-doc-map.md index 20ae4ef..f02a2fd 100644 --- a/knowledge/repos/vllm-omni/docs/design-doc-map.md +++ b/knowledge/repos/vllm-omni/docs/design-doc-map.md @@ -1,7 +1,7 @@ --- title: "官方设计文档地图(docs/design/**)" created: 2026-07-16 -updated: 2026-07-16 +updated: 2026-08-05 type: guide tags: [vllm-omni, docs] sources: [docs/design/index.md, docs/design/architecture_overview.md] @@ -9,7 +9,7 @@ sources: [docs/design/index.md, docs/design/architecture_overview.md] # 官方设计文档地图(docs/design/**) -`main @ 5c390096` 复核。注意:官方 `docs/design/index.md` 只列了子集(Architecture +`v0.26.0 @ a4ea67a2` 复核。注意:官方 `docs/design/index.md` 只列了子集(Architecture Overview、5 篇 feature、metrics、3 篇 module);**完整树比索引大得多**—— `feature/` 下还有 async_chunk、cache_dit、teacache、prefix_caching、7 篇并行策略、 `omni_connectors/` 逐后端 spec 等未入索引的文档,找 spec 时直接 `ls docs/design/feature/`。 diff --git a/knowledge/repos/vllm-omni/models/audex/_index.md b/knowledge/repos/vllm-omni/models/audex/_index.md new file mode 100644 index 0000000..99c7546 --- /dev/null +++ b/knowledge/repos/vllm-omni/models/audex/_index.md @@ -0,0 +1,51 @@ +--- +title: "Nemotron-Labs Audex" +created: 2026-08-05 +updated: 2026-08-05 +type: index +tags: [vllm-omni, models, serving] +sources: [vllm_omni/model_executor/models/audex/, vllm_omni/config/pipeline_registry.py, vllm_omni/deploy/audex_tts.yaml, vllm_omni/deploy/audex_tts_30b.yaml, recipes/NVIDIA/Nemotron-Labs-Audex.md, examples/offline_inference/audex/, examples/online_serving/audex/] +confidence: high +--- + +# Nemotron-Labs Audex + +## 名称、源码与注册 + +- checkpoints:`nvidia/Nemotron-Labs-Audex-2B` 与 + `nvidia/Nemotron-Labs-Audex-30B-A3B`;共享 `audex` model-executor 目录,包含 + thinker、codec、decoder、pipeline 和 stage input processor。 +- AR registry 新增五个 architecture:`NemotronDenseForCausalLM`、 + `AudexCode2Wav`、`AudexXCodec1`、`NemotronDenseAudexForConditionalGeneration`、 + `NemotronHAudexForConditionalGeneration`。 +- pipeline registry 提供 `audex_tts`、`audex_tta`、`audex_thinker_only`、 + `audex_s2s`;repo-root 的 `model_type: nemotron_labs_audex` 是默认 TTS + pipeline 的显式 alias,不应误路由到 thinker-only。 + +## 四种 pipeline 的 stage 合同 + +- TTS:thinker 逐 chunk 产生 codec frames,经 streaming `AudexCode2Wav` 解码; + terminal chunk 带 `stream_finished`,不能把空 codec token 当作 engine failure。 +- S2S:text 走 stage 0 的最终文本输出,audio payload 继续送到 `Code2Wav`;两类输出 + 必须按同一 request 对齐,不能把 audio bridge 当成普通文本 completion。 +- thinker-only:单 stage audio understanding,不自动追加 speech decoder;TTA 则是 + synchronous full-payload 路径,使用完整 RVQ payload 交给外部 `AudexXCodec1` 解码。 +- 这些 stage/shape 语义由 `stage_input_processors/audex.py` 和各模式 deploy 共同定义; + 只通过 registry import 或单一 TTS smoke 不能证明其他三种模式的 handoff。 + +## 部署边界 + +- 四种模式各有 2B/30B deploy YAML。TTS、TTA、thinker-only 和 S2S 的入口与输入/输出 + 组合以官方 recipe 为准;共享 16 kHz 流式 decoder 不意味着所有模式都提供 speech output。 +- 30B-A3B 必须显式选择对应的 `*_30b.yaml`;不能让 repo-root 自动检测复用 2B 的默认 + stage topology。改 deploy 时同时核对 stage 显存预算、Mamba prefix-cache 约束和 + decoder 的跨 stage payload。 +- Audex 的模型专属事实留在本页和 recipe;共享 stage/config/serving 合同分别归 + [Model Executor](../../components/model-executor/_index.md)、[Configuration](../../components/configuration/_index.md) + 和 [Serving](../../components/serving/_index.md)。 + +## 验证入口 + +离线脚本位于 `examples/offline_inference/audex/`,在线客户端和 server launcher 位于 +`examples/online_serving/audex/`。审查新模式时至少按目标 mode 绑定对应 deploy、endpoint +和音频输出类型;单独通过 registry import 不能证明 stage handoff 或服务合同。 diff --git a/knowledge/repos/vllm-omni/models/catalog.md b/knowledge/repos/vllm-omni/models/catalog.md index e5c6a5a..77e9a70 100644 --- a/knowledge/repos/vllm-omni/models/catalog.md +++ b/knowledge/repos/vllm-omni/models/catalog.md @@ -1,7 +1,7 @@ --- title: "模型代码入口与 registry 快照" created: 2026-07-16 -updated: 2026-07-31 +updated: 2026-08-05 type: guide tags: [vllm-omni, models] sources: [vllm_omni/model_executor/models/registry.py, vllm_omni/diffusion/registry.py, vllm_omni/config/pipeline_registry.py, vllm_omni/deploy/] @@ -9,8 +9,8 @@ sources: [vllm_omni/model_executor/models/registry.py, vllm_omni/diffusion/regis # 模型代码入口与 registry 快照 -本页提供模型描述到代码目录的自动定位入口,不维护逐模型 class 映射。下方计数仍是 -`v0.26.0rc1 @ 807db6ef`(2026-07-28)快照,数字会漂移,不能凭它断言“不支持”。 +本页提供模型描述到代码目录的自动定位入口,不维护逐模型 class 映射。下方计数是 +`v0.26.0 @ a4ea67a2`(2026-08-03)快照,数字会漂移,不能凭它断言“不支持”。 ## Direct 模型代码入口 @@ -35,36 +35,38 @@ adapter。已有专属知识 owner 可从 [models index](_index.md) 按名称进 | 注册点 | 位置 | 计数 | |---|---|---| -| AR/omni 架构 | `model_executor/models/registry.py` `_OMNI_MODELS` | 72 个架构名 / 26 个模型族目录 | -| Diffusion pipeline | `diffusion/registry.py` `_DIFFUSION_MODELS` | 61 条 pipeline / 37 个模型族目录 | -| Pipeline(model_type) | `config/pipeline_registry.py` `OMNI_PIPELINES` | 46 个 key | -| Deploy YAML | `vllm_omni/deploy/*.yaml` | 71 份 | +| AR/omni 架构 | `model_executor/models/registry.py` `_OMNI_MODELS` | 77 个架构名 / 27 个模型族目录 | +| Diffusion pipeline | `diffusion/registry.py` `_DIFFUSION_MODELS` | 58 条 pipeline / 38 个模型族目录 | +| Pipeline(model_type) | `config/pipeline_registry.py` `OMNI_PIPELINES` | 51 个 key | +| Deploy YAML | `vllm_omni/deploy/*.yaml` | 79 份 | -对比上一审计快照(`5d44868e`,2026-07-21):AR 架构 69→72,diffusion -pipeline 59→61,OMNI_PIPELINES 保持 46,deploy 65→71。新增 diffusion -家族是 `boogu_image` 和 `lingbot_video`;AR 新增项属于已有的 -`mammoth_moda2` 与 `minicpmo_4_5` 家族。 +对比上一审计快照(`807db6ef`):AR 架构 72→77,diffusion pipeline 61→58, +OMNI_PIPELINES 46→51,deploy 71→79。新增 AR 家族是 `audex`;diffusion 新增 +`minimax_h3`,同时 LTX-2/LTX-2.3 的五个旧 registry names 合并为两个入口。新增 +pipeline keys 是四个 Audex 模式和 `nemotron_labs_audex` alias;deploy 新增八个 +Audex files 与 `qwen3_omni_moe_thinking.yaml`,删除 `minicpmo_4_5_batching.yaml`。 -## AR/omni 模型族(26) +## AR/omni 模型族(27) -aura_omni、bagel、cosyvoice3、covo_audio、dynin_omni、fish_speech、glm_image、 +aura_omni、audex、bagel、cosyvoice3、covo_audio、dynin_omni、fish_speech、glm_image、 glm_tts、higgs_audio_v2、higgs_audio_v3、hunyuan_image3、indextts2、 mammoth_moda2、mimo_audio、ming_flash_omni、ming_tts、minicpmo_4_5、moss_tts、 moss_tts_nano、omnivoice、qwen2_5_omni、qwen3_omni、qwen3_tts、step_audio2、 voxcpm2、voxtral_tts -## Diffusion 模型族(37) +## Diffusion 模型族(38) audiox、bagel、boogu_image、cosmos3、diffusers_adapter(通用 diffusers 桥)、dreamid_omni、 dreamzero、ernie_image、flux、flux2、flux2_klein、glm_image、gr00t、helios、 hidream_image、hunyuan_image3、hunyuan_video、internvla_a1、krea2、lance、 lingbot_video、longcat_image、ltx2、magi_human、ming_flash_omni、nextstep_1_1、omnigen2、 omnivoice、ovis_image、qwen_image、sd3、sdxl、sensenova_u1、soulx_singer、 -stable_audio、wan2_2、z_image +stable_audio、wan2_2、z_image、minimax_h3 -## OMNI_PIPELINES key(46) +## OMNI_PIPELINES key(51) -Gr00tN1d7(注意:唯一 CamelCase key)、aura_omni、bagel、bagel_single_stage、 +Gr00tN1d7(注意:唯一 CamelCase key)、aura_omni、audex_s2s、audex_thinker_only、 +audex_tta、audex_tts、bagel、bagel_single_stage、 bagel_think、cosyvoice3、covo_audio、dreamzero、dynin_omni、fish_qwen3_omni、 glm_image、glm_tts、higgs_audio_v2、higgs_multimodal_qwen3、hunyuan_image3_ar、 hunyuan_image3_dit、hunyuan_image_3_moe、hunyuan_video_15、indextts2、lance、 @@ -74,7 +76,7 @@ ming_tts、ming_tts_moe、minicpmo_4_5、moss_tts_delay、moss_tts_local、 moss_tts_nano、moss_tts_realtime、omnivoice、qwen2_5_omni、 qwen2_5_omni_thinker_only、qwen3_omni_moe(resolver)、qwen3_tts、 soulxsinger_svc、soulxsinger_svs、step_audio_2、step_audio_2_asr、voxcpm2、 -voxtral_tts、wan2_2_ti2v +voxtral_tts、wan2_2_ti2v、nemotron_labs_audex 注意:单 stage diffusion 模型**多数不在** `OMNI_PIPELINES`(引擎为它们生成 默认 diffusion stage 配置,见 [Config 组件](../components/configuration/architecture.md)); @@ -85,8 +87,8 @@ voxtral_tts、wan2_2_ti2v ```bash python tools/audit_vllm_omni_release.py \ - --from 5d44868e \ - --to v0.26.0rc1 \ + --from 807db6ef \ + --to a4ea67a2 \ --repo \ --mode report-only ``` diff --git a/knowledge/repos/vllm-omni/models/dreamzero/architecture.md b/knowledge/repos/vllm-omni/models/dreamzero/architecture.md index 738b4b9..0f0763f 100644 --- a/knowledge/repos/vllm-omni/models/dreamzero/architecture.md +++ b/knowledge/repos/vllm-omni/models/dreamzero/architecture.md @@ -66,14 +66,14 @@ sources: [vllm_omni/diffusion/models/dreamzero/pipeline_dreamzero.py, vllm_omni/ ## 怎样验证功能、精度和性能 -pin 上有**上游一致性测试**(`tests/dreamzero/upstream/` 对上游 socket -server 的 e2e 源一致性)与较全单测;本次调查未发现性能 gate——但有 +pin 上有 `tests/e2e/accuracy/test_dreamzero.py` 与较全的 +`tests/diffusion/models/dreamzero/` 单测;本次调查未发现性能 gate——但有 `DZ_PHASE_TIMING=1` 分相计时插桩(逐 mark 同步 CUDA,**benchmark 时应 关闭**,源注释明示)。共享设施见 [Diffusion 组件](../../components/diffusion/_index.md)。 -- 单测:`tests/dreamzero/`(crossattn cache、fused_qk_rms_norm、QKV 融合、 - pipeline state、utils、OpenPI helper);e2e +- 单测:`tests/diffusion/models/dreamzero/`(crossattn cache、fused_qk_rms_norm、 + QKV 融合、pipeline state、utils、OpenPI helper);e2e `tests/e2e/online_serving/test_dreamzero_expansion.py`;配置解析 `tests/entrypoints/test_resolve_dreamzero_config.py`;示例 `examples/{offline_inference,online_serving}/dreamzero/` diff --git a/knowledge/repos/vllm-omni/models/gr00t/architecture.md b/knowledge/repos/vllm-omni/models/gr00t/architecture.md index 6edd8cd..165e677 100644 --- a/knowledge/repos/vllm-omni/models/gr00t/architecture.md +++ b/knowledge/repos/vllm-omni/models/gr00t/architecture.md @@ -61,7 +61,7 @@ sources: [vllm_omni/diffusion/models/gr00t/pipeline_gr00t.py, vllm_omni/diffusio pin 上有**位级回归**(对 Isaac-GR00T ZMQ 参考值 max_diff=0.0)——这是行为 正确性 gate;无性能 gate;examples 缺失。 -- e2e:`tests/e2e/online_serving/test_gr00t_openpi.py` +- e2e:`tests/e2e/online_serving/test_gr00t_openpi_expansion.py` (`GR00T_NOISE_SEED=42`,init 1200 s/stage 900 s,缺 `websockets`+ `openpi_client` 即跳过);测试客户端 `tests/gr00t/openpi_client_helper.py`; 单测 `tests/diffusion/models/gr00t/test_pipeline.py`(stub policy,观测 diff --git a/knowledge/repos/vllm-omni/models/higgs-audio/architecture.md b/knowledge/repos/vllm-omni/models/higgs-audio/architecture.md index 341ef44..5035cb7 100644 --- a/knowledge/repos/vllm-omni/models/higgs-audio/architecture.md +++ b/knowledge/repos/vllm-omni/models/higgs-audio/architecture.md @@ -61,7 +61,7 @@ sources: [vllm_omni/model_executor/models/higgs_audio_v3/higgs_audio_v3_talker.p ## 怎样验证功能、精度和性能 -- 单元:`tests/unit/higgs_audio_v3/test_higgs_audio_v3.py`(无 GPU,AC-1..10: +- 单元:`tests/model_executor/models/higgs_audio_v3/test_higgs_audio_v3.py`(无 GPU,AC-1..10: config/prompt/融合模块/delay/stage processor/registry)。 - e2e:v2 `tests/e2e/{offline_inference,online_serving}/test_higgs_audio_v2_expansion.py`; v3 `tests/e2e/online_serving/test_higgs_audio_v3.py`;perf diff --git a/knowledge/repos/vllm-omni/models/hunyuan-video/_index.md b/knowledge/repos/vllm-omni/models/hunyuan-video/_index.md index c57f887..b9b7416 100644 --- a/knowledge/repos/vllm-omni/models/hunyuan-video/_index.md +++ b/knowledge/repos/vllm-omni/models/hunyuan-video/_index.md @@ -29,7 +29,7 @@ diffusion 视频,AR registry 无入口;image3 的结构见该页)。 **I2V 不在 OMNI_PIPELINES**(走单 stage diffusion 兜底)。pipeline config **无 `default_deploy_config_name`**——`hunyuan_video_15.yaml` 不会 自动加载;显式传裸文件名时按 `_DEPLOY_DIR` 解析(pin 上仅 - `tests/test_config_factory.py` 按名加载它)。 + `tests/config/test_config_factory.py` 按名加载它)。 - T2V vs I2V 差异速览:I2V 加 SigLIP 图像编码器 + 图像预处理(从 `max_area` 推 H/W)+ 首帧 VAE 条件 latent/掩码;T2V 供零 image_embeds、 零 cond_latents、零 mask。 diff --git a/knowledge/repos/vllm-omni/models/ltx2/_index.md b/knowledge/repos/vllm-omni/models/ltx2/_index.md index 65eb727..36b15c0 100644 --- a/knowledge/repos/vllm-omni/models/ltx2/_index.md +++ b/knowledge/repos/vllm-omni/models/ltx2/_index.md @@ -1,10 +1,10 @@ --- title: "LTX-2 家族(含 LTX-2.3)" created: 2026-07-16 -updated: 2026-07-16 +updated: 2026-08-05 type: index tags: [vllm-omni, models, ltx2] -sources: [vllm_omni/diffusion/models/ltx2/, vllm_omni/diffusion/registry.py, recipes/LTX/LTX-2.3.md] +sources: [vllm_omni/diffusion/models/ltx2/, vllm_omni/diffusion/registry.py, recipes/LTX/LTX-2.md] --- # LTX-2 家族(含 LTX-2.3) @@ -14,12 +14,12 @@ sources: [vllm_omni/diffusion/models/ltx2/, vllm_omni/diffusion/registry.py, rec - 厂商/模型:Lightricks;22B 参数文本→视频+音频生成(T2V/I2V,48kHz 同步音频, 768x512 可达 20+ 秒);Diffusers 格式 checkpoint `dg845/LTX-2.3-Diffusers` - 源码:`vllm_omni/diffusion/models/ltx2/`(纯 diffusion,无 AR stage) -- registry(`vllm_omni/diffusion/registry.py`,`main @ 5c390096` 验证): - LTX-2 条目 `LTX2Pipeline`(:69)/`LTX2ImageToVideoPipeline`(:74)/ - `LTX2TwoStagesPipeline`(:79)+ DMD2 蒸馏变体;LTX-2.3 条目 `LTX23Pipeline` - (:99)/`LTX23ImageToVideoPipeline`(:104),后处理共用 - `get_ltx2_post_process_func`(:504) -- 官方 recipe:`recipes/LTX/LTX-2.md`、`recipes/LTX/LTX-2.3.md` +- registry:`LTX2Pipeline` 统一 LTX-2/LTX-2.3 的 one-stage T2V/I2V; + `LTX2DistilledPipeline` 统一 distilled two-stage T2V/I2V;DMD2 仍由 + `LTX2T2VDMD2Pipeline`/`LTX2I2VDMD2Pipeline` 提供。旧的 LTX23、ImageToVideo 和 + TwoStages registry names 已删除,没有兼容 alias。 +- 官方 recipe 已合并为 `recipes/LTX/LTX-2.md`;已删除的 + `recipes/LTX/LTX-2.3.md` 不能继续作为 source 或事实依据。 - 依赖共享 [Diffusion 组件](../../components/diffusion/_index.md) ## 什么时候查这里 diff --git a/knowledge/repos/vllm-omni/models/ltx2/architecture.md b/knowledge/repos/vllm-omni/models/ltx2/architecture.md index 0d957e1..b2464d4 100644 --- a/knowledge/repos/vllm-omni/models/ltx2/architecture.md +++ b/knowledge/repos/vllm-omni/models/ltx2/architecture.md @@ -1,15 +1,15 @@ --- title: "LTX-2/2.3 模型架构与证据索引" created: 2026-07-16 -updated: 2026-07-16 +updated: 2026-08-05 type: architecture tags: [vllm-omni, models, ltx2] -sources: [recipes/LTX/LTX-2.3.md, vllm_omni/diffusion/registry.py, "#4381", "#4464"] +sources: [recipes/LTX/LTX-2.md, vllm_omni/diffusion/registry.py, "#4381", "#4464"] --- # LTX-2/2.3 模型架构与证据索引 -以下事实在 `main @ 5c390096` 复核(recipe 与 registry)。 +以下事实在 `v0.26.0 @ a4ea67a2` 复核(合并后的 recipe 与 live registry)。 ## 结构与 serving @@ -17,12 +17,15 @@ sources: [recipes/LTX/LTX-2.3.md, vllm_omni/diffusion/registry.py, "#4381", "#44 T2V 与 I2V 皆可,输出带 48kHz 同步音频。验证建议从 96GB 级 GPU 起步 (recipe 原文)。 - serving 入口(LTX-2.3): - `vllm serve dg845/LTX-2.3-Diffusers --omni --model-class-name LTX23Pipeline - --stage-init-timeout 600`;需要 `diffusers >= 0.38.0`(git 安装)。 -- pipeline 变体:常规单段(`LTX2Pipeline`/`LTX23Pipeline`)、I2V - (`*ImageToVideoPipeline`)、两段式(`LTX2TwoStagesPipeline`)与 DMD2 蒸馏 - (`LTX2T2VDMD2Pipeline`/`LTX2I2VDMD2Pipeline`);LTX-2.3 后处理复用 ltx2 的 - `get_ltx2_post_process_func`。 + `vllm serve diffusers/LTX-2.3-Diffusers --omni --stage-init-timeout 600`;统一 + `LTX2Pipeline` 由 checkpoint metadata 选择 LTX-2 或 LTX-2.3 profile。 +- pipeline 变体:one-stage 使用 `LTX2Pipeline`,distilled two-stage 使用 + `LTX2DistilledPipeline`,DMD2 使用 `LTX2T2VDMD2Pipeline`/ + `LTX2I2VDMD2Pipeline`;T2V/I2V 通过是否提供 `image=` 选择,不再使用单独的 + `*ImageToVideoPipeline` registry names。 +- Python `forward` 只允许 `req` 作为 positional argument,其他参数必须 keyword-only; + 这对直接调用者和显式 `--model-class-name` 覆盖是 breaking change,CLI/HTTP recipe + 已按 named fields 调用。 - 单段 diffusion 模型不在 `OMNI_PIPELINES` registry(走 `async_omni_engine.py` 的默认 diffusion stage 兜底),deploy 语义见 [Config 组件](../../components/configuration/architecture.md)。 diff --git a/knowledge/repos/vllm-omni/models/mammoth-moda2/architecture.md b/knowledge/repos/vllm-omni/models/mammoth-moda2/architecture.md index 9185381..5de8447 100644 --- a/knowledge/repos/vllm-omni/models/mammoth-moda2/architecture.md +++ b/knowledge/repos/vllm-omni/models/mammoth-moda2/architecture.md @@ -75,7 +75,7 @@ pin 上只有**功能/config 面**的验证入口,没有专门的精度基线或 - e2e:`tests/e2e/offline_inference/test_mammoth_moda2_expansion.py` (t2i + AR);config 单测 - `tests/unit/mammoth_moda2/test_mammoth_moda2_config.py`;AR 图像理解示例 + `tests/model_executor/models/mammoth_moda2/test_mammoth_moda2_config.py`;AR 图像理解示例 `examples/offline_inference/mammothmodal2_preview/`。 - 已知未决:runner 侧把 hidden 打进 `multimodal_output["latent"]` 的确切 代码位、`Mammoth2DecoderLayer` 每层 moe 布线、DiT 去噪逐步数学——分析时 diff --git a/knowledge/repos/vllm-omni/models/minimax-h3/_index.md b/knowledge/repos/vllm-omni/models/minimax-h3/_index.md new file mode 100644 index 0000000..7bdf663 --- /dev/null +++ b/knowledge/repos/vllm-omni/models/minimax-h3/_index.md @@ -0,0 +1,40 @@ +--- +title: "MiniMax H3" +created: 2026-08-05 +updated: 2026-08-05 +type: index +tags: [vllm-omni, models, diffusion] +sources: [vllm_omni/diffusion/models/minimax_h3/, vllm_omni/diffusion/registry.py, recipes/MiniMaxAI/MiniMax-H3.md, recipes/MiniMaxAI/MiniMax-H3-NPU.md, tests/diffusion/models/minimax_h3/, vllm_omni/entrypoints/openai/video_api_utils.py] +confidence: high +--- + +# MiniMax H3 + +## 名称、源码与任务 + +- checkpoint:`MiniMaxAI/MiniMax-H3`;纯 diffusion registry architecture 是 + `MiniMaxH3Pipeline`,实现位于 `diffusion/models/minimax_h3/`。 +- 支持 `t2va`、`fl2va`、`ref2va` 三种 joint video+audio 条件模式;一个 server 进程 + 一次只加载 `FL2VA` 或 `Ref2VA` 分区,不能把两个分区当成同一份权重同时服务。 +- 输出是带同步音频的 MP4;参考视频、图像和音频的预处理合同由 pipeline 与 video API + 共同决定,不能仅凭 endpoint 名称推断输入组合。 +- 生成合同固定为 24 FPS 视频与 32 kHz 音频;空间尺寸必须按 32 对齐,宽高比限制在 + `1:4` 到 `4:1`,这些约束在 request validation 阶段执行而不是由 VAE 静默修正。 + +## 并行与加载约束 + +- H3 是 CFG-distilled pipeline,`cfg_parallel_size` 必须保持 `1`。VAE patch parallel + 使用 H3 原生 `tile` 模式;`text_encoder_tp_size` 作用在 DiT group 的前 N 个 rank, + 并且必须同时整除 Qwen3-VL 的 64 个 attention heads 与 8 个 KV heads。 +- 单卡 accuracy 路径使用 CPU offload;多卡部署可使用 Ulysses、text-encoder TP、VAE + tile/patch parallel 或 layerwise offload,但每个组合都必须按最终并行拓扑和设备数验证。 +- 音频加载优先使用 torchaudio;当 TorchCodec/torchaudio 在 CPU-only aarch64 环境不可用 + 时,`reference_video.load_audio_file` 回退到 soundfile,再对 libsndfile 不支持的格式 + 通过 ffmpeg 转 WAV。该回退保持 `(channels, samples)` float32 与原始 sample rate 合同。 + +## 验证入口 + +模型专属 contract、packing、parallel 和 e2e 测试在 `tests/diffusion/models/minimax_h3/`。 +硬件 recipe 只记录已验证的 GPU/NPU 形状;性能数字不能从 recipe 的配置示例泛化为全硬件 +保证。共享 offloader、并行和请求合同分别归 [Diffusion](../../components/diffusion/_index.md)、 +[Configuration](../../components/configuration/_index.md) 和 [Serving](../../components/serving/_index.md)。 diff --git a/knowledge/repos/vllm-omni/models/soulx-singer/_index.md b/knowledge/repos/vllm-omni/models/soulx-singer/_index.md index e8986fd..b21d9a2 100644 --- a/knowledge/repos/vllm-omni/models/soulx-singer/_index.md +++ b/knowledge/repos/vllm-omni/models/soulx-singer/_index.md @@ -59,7 +59,7 @@ sources: [vllm_omni/diffusion/models/soulx_singer/, vllm_omni/model_executor/mod - extra-body 面在 pipeline 类的 ClassVar(无 model_extras 文件):SVS 收 metadata/语言/control/max_merge_duration…出 `f0_shift`;SVC 收 wav/F0 路径 出 `pitch_shift`(`auto_shift` 支持,SVC 移调 = `pitch_shift×5` 粗 F0 bin)。 -- 自动化测试仅见离线 e2e(`test_soulxsinger.py`);本次调查未发现在线示例 +- 自动化测试仅见离线 e2e(`test_soulxsinger_expansion.py`);本次调查未发现在线示例 的 CI 覆盖。 ## 什么时候查这里 diff --git a/knowledge/repos/vllm-omni/models/soulx-singer/architecture.md b/knowledge/repos/vllm-omni/models/soulx-singer/architecture.md index 5a79f7e..8f46d66 100644 --- a/knowledge/repos/vllm-omni/models/soulx-singer/architecture.md +++ b/knowledge/repos/vllm-omni/models/soulx-singer/architecture.md @@ -59,7 +59,7 @@ sources: [vllm_omni/diffusion/models/soulx_singer/pipeline_soulx_singer_base.py, 自动化测试仅覆盖离线 e2e,另有一个可复跑基准;本次调查未发现在线链路的 CI 覆盖——精度/性能结论需另行实测。 -- e2e:`tests/e2e/offline_inference/test_soulxsinger.py`(快照双仓、建 +- e2e:`tests/e2e/offline_inference/test_soulxsinger_expansion.py`(快照双仓、建 SVS/SVC view 目录、校验 phone_set.json;资产 `tests/assets/soulxsinger/`);示例 `examples/offline_inference/text_to_speech/soulxsinger/end2end.py` 与 diff --git a/knowledge/repos/vllm-omni/rebase/_index.md b/knowledge/repos/vllm-omni/rebase/_index.md index 6cfef22..16e1bc8 100644 --- a/knowledge/repos/vllm-omni/rebase/_index.md +++ b/knowledge/repos/vllm-omni/rebase/_index.md @@ -1,16 +1,16 @@ --- title: "Rebase(对齐 upstream vLLM)" created: 2026-07-16 -updated: 2026-07-16 +updated: 2026-08-05 type: index tags: [vllm-omni, rebase] -sources: [".buildkite/rebase-pipeline.yaml", "vllm-omni-rebase-agent@122a9468:agent/config.py"] +sources: [".buildkite/cuda/rebase-pipeline.yml", "vllm-omni-rebase-agent@122a9468:agent/config.py"] --- # Rebase(对齐 upstream vLLM) 把 vllm-omni 对齐到新版 upstream vLLM 的周期性工作:专用分支 `dev/vllm-align`、 -专用管线 `.buildkite/rebase-pipeline.yaml`(`main @ 5c390096` 验证存在)、专职 +专用管线 `.buildkite/cuda/rebase-pipeline.yml`(`v0.26.0 @ a4ea67a2` 验证存在)、专职 自动化(rebase-agent 仓库——运营系统在那里,本主题只沉淀领域知识)。 ## 什么时候查这里 diff --git a/knowledge/repos/vllm-omni/rebase/upstream-api-drift.md b/knowledge/repos/vllm-omni/rebase/upstream-api-drift.md index eb8b0da..0194c52 100644 --- a/knowledge/repos/vllm-omni/rebase/upstream-api-drift.md +++ b/knowledge/repos/vllm-omni/rebase/upstream-api-drift.md @@ -33,7 +33,7 @@ modules=[scheduler],status=active,run_count=20,2026-06-15 创建 / 07-11 `use_harmony: bool = False` 作类属性;被删方法在 `OmniOpenAIServingChat` 恢复为本地副本;核对 `vllm.entrypoints.openai.parser.harmony_utils` 的 import ——`parse_chat_output` 若从该模块移除则内联。回归:`tests/entrypoints/test_stream_finish_reason.py` - + `tests/comfyui/test_comfyui_integration.py::test_understanding_node`; + + `tests/e2e/features/comfyui/test_comfyui_integration.py::test_understanding_node`; pre-commit(ruff format 可能内联被删 import)。 ^[SK-fix-omni-serving-chat-upstream-harmony-refactor] - 细化实例(debug-memory #184,key=`missing_should_check_unstreamed_tool_arg_tokens`, @@ -192,8 +192,9 @@ modules=[online_serving, model_executor],status=active,run_count=6, - 验证:`cd /rebase/vllm-omni && /rebase/.venv/bin/python -m pytest --collect-only -q ` → rc=0 且 `N tests collected`;重跑 pipeline 测试进入执行。 -- 禁止:为满足陈旧配置**重造上游已删除的测试文件**(曾手写 - `tests/e2e/online_serving/test_hunyuan_video_15.py`——恢复上游有意删除的覆盖、与幸存者重复、以后每次 +- 禁止:为满足陈旧配置**重造上游已删除的测试文件**(旧的 + `tests/e2e/online_serving/test_hunyuan_video_15.py` 已由 + `tests/e2e/accuracy/hunyuanvideo15_{i2v,t2v}/` 幸存者取代;恢复旧路径会与幸存者重复、以后每次 rebase 都冲突);把 rc=4/5 当 OOM 在 GPU 上重试——**结果永远不会改变**,只白烧 3 次重试并硬停;盲目放宽 marker 把 `slow`/`full_model` 幸存者塞进 merge job(改变上游的 CI 成本/分层意图)。 ^[SK-fix-stale-test-path-collection-error-after-rename] diff --git a/knowledge/repos/vllm-omni/rebase/workflow.md b/knowledge/repos/vllm-omni/rebase/workflow.md index e7d9130..44360bb 100644 --- a/knowledge/repos/vllm-omni/rebase/workflow.md +++ b/knowledge/repos/vllm-omni/rebase/workflow.md @@ -1,22 +1,22 @@ --- title: "Rebase 工作流:分支、波次与失败路由" created: 2026-07-16 -updated: 2026-07-16 +updated: 2026-08-05 type: guide tags: [vllm-omni, rebase] -sources: ["vllm-omni-rebase-agent@122a9468:agent/config.py", "vllm-omni-rebase-agent@122a9468:config.sh", ".buildkite/rebase-pipeline.yaml"] +sources: ["vllm-omni-rebase-agent@122a9468:agent/config.py", "vllm-omni-rebase-agent@122a9468:config.sh", ".buildkite/cuda/rebase-pipeline.yml"] --- # Rebase 工作流:分支、波次与失败路由 运营事实来自 rebase-agent 配置快照(@122a9468,**可能漂移**);仓库侧事实在 -`main @ 5c390096` 复核。运营系统以 rebase-agent 仓库为准,本页是知识树快照 +`v0.26.0 @ a4ea67a2` 复核。运营系统以 rebase-agent 仓库为准,本页是知识树快照 (2026-07-16)。 ## 分支与管线 - 对齐分支 `dev/vllm-align`(目标合回 `main`);专用 Buildkite 管线 - `.buildkite/rebase-pipeline.yaml`(仓库侧)+ 运营管线 `vllm-omni-rebase` + `.buildkite/cuda/rebase-pipeline.yml`(仓库侧)+ 运营管线 `vllm-omni-rebase` (nightly 与 main CI)、`vllm-omni-release`(CI),org `vllm`;wheel 变体 `cu130`;上次 rebase 的 vLLM 提交 pin `1acd67a795ebccdf9b9db7697ae9082058301657`。 From 6e1e140c73ae916b95dd37a11d2160a8ca01ea06 Mon Sep 17 00:00:00 2001 From: hsliu_ustc Date: Thu, 6 Aug 2026 12:18:52 +0800 Subject: [PATCH 2/2] Add daily vLLM-Omni knowledge rules --- adapters/vllm_omni/release_baseline.yaml | 7 +++- .../repos/vllm-omni/components/_index.md | 3 +- .../vllm-omni/components/comfyui/_index.md | 32 ++++++++++++++++ .../vllm-omni/components/comfyui/rules.md | 35 +++++++++++++++++ .../components/model-executor/_index.md | 10 +++-- .../components/model-executor/rules.md | 35 ++++++++++++++++- .../vllm-omni/components/scheduler/_index.md | 9 +++-- .../vllm-omni/components/scheduler/rules.md | 19 +++++++++- knowledge/repos/vllm-omni/docs/_index.md | 3 +- knowledge/repos/vllm-omni/docs/rules.md | 35 +++++++++++++++++ .../repos/vllm-omni/models/bagel/_index.md | 3 +- .../repos/vllm-omni/models/bagel/rules.md | 33 ++++++++++++++++ .../vllm-omni/models/minimax-h3/_index.md | 8 +++- .../vllm-omni/models/minimax-h3/rules.md | 38 +++++++++++++++++++ 14 files changed, 253 insertions(+), 17 deletions(-) create mode 100644 knowledge/repos/vllm-omni/components/comfyui/_index.md create mode 100644 knowledge/repos/vllm-omni/components/comfyui/rules.md create mode 100644 knowledge/repos/vllm-omni/docs/rules.md create mode 100644 knowledge/repos/vllm-omni/models/bagel/rules.md create mode 100644 knowledge/repos/vllm-omni/models/minimax-h3/rules.md diff --git a/adapters/vllm_omni/release_baseline.yaml b/adapters/vllm_omni/release_baseline.yaml index 51b41dd..f5e492c 100644 --- a/adapters/vllm_omni/release_baseline.yaml +++ b/adapters/vllm_omni/release_baseline.yaml @@ -32,6 +32,8 @@ path_owners: - .buildkite/ - .github/ - docker/ + comfyui: + - apps/ComfyUI-vLLM-Omni/ configuration: - vllm_omni/config/ - vllm_omni/deploy/ @@ -73,7 +75,6 @@ path_owners: - tests/ tooling: - tools/ - - apps/ComfyUI-vLLM-Omni/ runtime-core: - vllm_omni/__init__.py - vllm_omni/data_entry_keys.py @@ -86,6 +87,8 @@ owner_documents: - knowledge/repos/vllm-omni/benchmark/_index.md ci: - knowledge/repos/vllm-omni/ci/_index.md + comfyui: + - knowledge/repos/vllm-omni/components/comfyui/rules.md configuration: - knowledge/repos/vllm-omni/components/configuration/rules.md diffusion: @@ -93,6 +96,7 @@ owner_documents: distributed: - knowledge/repos/vllm-omni/components/distributed/rules.md documentation: + - knowledge/repos/vllm-omni/docs/rules.md - knowledge/general/docs/_index.md model-executor: - knowledge/repos/vllm-omni/components/model-executor/rules.md @@ -139,6 +143,7 @@ pin_documents: - doc/KNOWLEDGE.md - knowledge/repos/vllm-omni/models/catalog.md - knowledge/repos/vllm-omni/components/configuration/_index.md +- knowledge/repos/vllm-omni/components/comfyui/_index.md - knowledge/repos/vllm-omni/components/diffusion/_index.md - knowledge/repos/vllm-omni/components/distributed/_index.md - knowledge/repos/vllm-omni/components/model-executor/_index.md diff --git a/knowledge/repos/vllm-omni/components/_index.md b/knowledge/repos/vllm-omni/components/_index.md index 590a849..89e022b 100644 --- a/knowledge/repos/vllm-omni/components/_index.md +++ b/knowledge/repos/vllm-omni/components/_index.md @@ -1,7 +1,7 @@ --- title: "vLLM-Omni 组件 owner" created: 2026-07-10 -updated: 2026-07-31 +updated: 2026-08-06 type: index tags: [vllm-omni, components] sources: [] @@ -15,6 +15,7 @@ sources: [] | Owner | 负责范围 | 直接入口 | |---|---|---| | [Configuration](configuration/_index.md) | deploy YAML、PipelineConfig、registry、字段归属、default 和 endpoint policy | [rules](configuration/rules.md) | +| [ComfyUI](comfyui/_index.md) | ComfyUI video node/client、T2VA/FL2VA/Ref2VA 路由和 multipart 字段 | [rules](comfyui/rules.md) | | [Serving](serving/_index.md) | 用户请求、OpenAI API、响应、AsyncOmni engine 生命周期 | [rules](serving/rules.md) | | [Model Executor](model-executor/_index.md) | stage config/input、模型加载、worker、跨 stage 数据桥 | [rules](model-executor/rules.md) | | [Diffusion](diffusion/_index.md) | diffusion pipeline、denoise、VAE/DiT、并行和 cache | [rules](diffusion/rules.md) | diff --git a/knowledge/repos/vllm-omni/components/comfyui/_index.md b/knowledge/repos/vllm-omni/components/comfyui/_index.md new file mode 100644 index 0000000..efda082 --- /dev/null +++ b/knowledge/repos/vllm-omni/components/comfyui/_index.md @@ -0,0 +1,32 @@ +--- +title: "ComfyUI vLLM-Omni" +created: 2026-08-06 +updated: 2026-08-06 +type: index +tags: [vllm-omni, components, serving] +sources: [apps/ComfyUI-vLLM-Omni/, tests/e2e/features/comfyui/, "PR #5756"] +confidence: high +--- + +# ComfyUI vLLM-Omni + +- 源码入口:`apps/ComfyUI-vLLM-Omni/` 的 nodes、API client、types 和 example workflows。 +- 源码校验:app 与 `tests/e2e/features/comfyui/` 已在 `v0.26.0 @ a4ea67a2` 验证存在。 +- 主要职责:把 ComfyUI video-generation 节点输入校验并编译为 vLLM-Omni 的 T2VA、 + FL2VA 或 Ref2VA multipart 请求;服务端 endpoint 本身归 [Serving](../serving/_index.md)。 + +## 什么时候查这里 + +- 修改 ComfyUI 节点输入、reference 组合、mode 选择、multipart 字段或 client 请求。 +- 用 [MiniMax H3](../../models/minimax-h3/_index.md) 验证 T2VA/FL2VA/Ref2VA workflow。 + +## 目录内容 + +| 遇到什么 | 查看哪里 | +|---|---| +| frame/reference 互斥、mode 路由、multipart 字段 | [rules](rules.md) | + +## 不放什么 + +- 通用 OpenAI-compatible request normalization 或 endpoint policy;这些属于 Serving。 +- MiniMax H3 pipeline 内部的 packing、量化和 denoise 行为;这些属于模型 owner。 diff --git a/knowledge/repos/vllm-omni/components/comfyui/rules.md b/knowledge/repos/vllm-omni/components/comfyui/rules.md new file mode 100644 index 0000000..b036db6 --- /dev/null +++ b/knowledge/repos/vllm-omni/components/comfyui/rules.md @@ -0,0 +1,35 @@ +--- +title: "ComfyUI vLLM-Omni 规则" +created: 2026-08-06 +updated: 2026-08-06 +type: rule +tags: [vllm-omni, components, serving] +sources: [apps/ComfyUI-vLLM-Omni/comfyui_vllm_omni/nodes.py, apps/ComfyUI-vLLM-Omni/comfyui_vllm_omni/utils/api_client.py, tests/e2e/features/comfyui/test_comfyui_integration.py, "PR #5756"] +confidence: high +--- + +# ComfyUI vLLM-Omni 规则 + +只有 `COMFY-数字字母` 是可审计规则 ID。 + +## Direct 代码快速入口 + +| PR 描述信号 | 规则组 | 第一批源码 | +|---|---|---| +| ComfyUI、T2VA/FL2VA/Ref2VA、frame/reference、multipart | COMFY-1a | `apps/ComfyUI-vLLM-Omni/comfyui_vllm_omni/nodes.py` → `utils/api_client.py` → integration tests | + +## COMFY-1a — video mode 必须由互斥的 canonical 输入组合唯一决定 + +- 触发:ComfyUI video-generation node/client 修改 frame、image/audio/video references、 + mode 路由或 multipart 字段。 +- 强制:先校验 frame 与所有 reference 互斥,再确定唯一模式:有 frame 走 FL2VA;frame + 和 references 都没有走 T2VA;reference-only 只接受“恰好一张图 + 一段音频”或“仅一个 + 或多个视频”,并走 Ref2VA。图像写入 `input_reference`,音频写入 + `audio_reference`,视频按顺序重复写入 `input_references`。 +- 禁止:视频与 image/audio 混用;frame 与 reference 同时发送;按字段遍历顺序覆盖已经 + 选定的 mode;让 node 校验与 API client 使用不同的组合矩阵或字段名。 +- 验收:node 和 client 都有非法组合负例;E2E 分别覆盖 T2VA、FL2VA、image+audio + Ref2VA 和 multi-video Ref2VA,并断言最终 mode 与 multipart key/count。 ^[PR #5756] + +owner 范围见 [ComfyUI index](_index.md);服务端请求合同见 +[Serving 规则](../serving/rules.md),模型能力见 [MiniMax H3](../../models/minimax-h3/_index.md)。 diff --git a/knowledge/repos/vllm-omni/components/model-executor/_index.md b/knowledge/repos/vllm-omni/components/model-executor/_index.md index c57cdbc..1344615 100644 --- a/knowledge/repos/vllm-omni/components/model-executor/_index.md +++ b/knowledge/repos/vllm-omni/components/model-executor/_index.md @@ -1,7 +1,7 @@ --- title: "Model Executor" created: 2026-07-10 -updated: 2026-08-05 +updated: 2026-08-06 type: index tags: [vllm-omni, components, model-executor] sources: [] @@ -13,12 +13,14 @@ sources: [] - 源码校验:以上路径均已在 `v0.26.0 @ a4ea67a2` 验证存在;stage 配置已经迁移到 `vllm_omni/deploy/`,NPU/XPU 等平台可继续拥有自己的 worker 覆盖 - 测试入口:共享 runner 行为看 `tests/worker/`,具体模型 consumer 看 `tests/model_executor/` -- 主要职责:AR/LLM stage、stage 配置、并行与设备启动、runner 到模型的输入预处理合同和跨阶段数据桥接 +- 主要职责:AR/LLM stage、stage 配置、并行与设备启动、runner 到模型的输入预处理合同、 + payload connector ownership、异步输出物化和跨阶段数据桥接 ## 什么时候查这里 - 调查模型执行 stage、stage config、并行度、设备映射或 worker 启动。 -- 修改 runner `preprocess`、`_omni_*` 逐行 metadata、`talker_mtp`、chunked-prefill phase 或共享输入处理合同。 +- 修改 runner `preprocess`、`_omni_*` 逐行 metadata、`talker_mtp`、chunked-prefill phase、 + payload connector ownership 或异步输出物化合同。 - 调查 AR 到 diffusion 的共享桥接。 ## 不放什么 @@ -31,4 +33,4 @@ sources: [] | 遇到什么 | 查看哪里 | |---|---| | 理解共享职责和阶段边界 | [architecture](architecture.md) | -| 根据 PR 描述直达 stage config、runner preprocess、stage runtime、bridge/batch 或 loader 的规则组与第一批源码 | [rules 与代码地图](rules.md) | +| 根据 PR 描述直达 stage config、runner preprocess、stage runtime、bridge/batch、payload connector、async output 或 loader 的规则组与第一批源码 | [rules 与代码地图](rules.md) | diff --git a/knowledge/repos/vllm-omni/components/model-executor/rules.md b/knowledge/repos/vllm-omni/components/model-executor/rules.md index f79250e..207b6ea 100644 --- a/knowledge/repos/vllm-omni/components/model-executor/rules.md +++ b/knowledge/repos/vllm-omni/components/model-executor/rules.md @@ -1,10 +1,10 @@ --- title: "Model Executor 规则" created: 2026-07-10 -updated: 2026-08-05 +updated: 2026-08-06 type: rule tags: [vllm-omni, components, model-executor] -sources: [vllm_omni/worker/gpu_model_runner.py, vllm_omni/worker/gpu_ar_model_runner.py, vllm_omni/engine/stage_init_utils.py, tests/worker/test_omni_gpu_model_runner.py, vllm_omni/config/stage_config.py, vllm_omni/config/omni_config.py, vllm_omni/engine/stage_runtime.py, vllm_omni/engine/stage_engine_startup.py, vllm_omni/experimental/fullduplex/, tests/e2e/features/fullduplex/, "PR #3642", "PR #4730", "claude-workflow-starter-private@09dca46"] +sources: [vllm_omni/worker/gpu_model_runner.py, vllm_omni/worker/gpu_ar_model_runner.py, vllm_omni/worker/omni_connector_model_runner_mixin.py, vllm_omni/engine/stage_init_utils.py, tests/worker/test_omni_gpu_model_runner.py, tests/worker/test_omni_connector_mixin.py, vllm_omni/config/stage_config.py, vllm_omni/config/omni_config.py, vllm_omni/engine/stage_runtime.py, vllm_omni/engine/stage_engine_startup.py, vllm_omni/experimental/fullduplex/, tests/e2e/features/fullduplex/, docs/design/feature/omni_async_output_materialization.md, "PR #3642", "PR #4730", "PR #5610", "PR #5744", "claude-workflow-starter-private@09dca46"] --- # Model Executor 规则 @@ -21,6 +21,8 @@ sources: [vllm_omni/worker/gpu_model_runner.py, vllm_omni/worker/gpu_ar_model_ru | stage TP/PP/DP、devices、replica、visible devices、worker 启动、容量 fail-fast | `stage-runtime`:本页“Stage 并行度和设备容量必须一起验收” | `vllm_omni/config/stage_config.py::build_stage_runtime_overrides` → `vllm_omni/engine/stage_runtime.py::{StageRuntime.initialize,StageRuntime._resolve_replica_physical_devices}` → `stage_engine_startup.py::{launch_stage_replica,get_headless_replica_devices}` | | `runtime_info`、`OmniOutput`、multimodal payload、跨 stage bridge、batch 串线 | `bridge-batch`:`EXEC-1a`, `EXEC-1b` | `vllm_omni/worker/gpu_model_runner.py::{extract_multimodal_outputs,_gather_runtime_additional_information,_build_model_kwargs_extra}` → `vllm_omni/model_executor/stage_input_processors/<命中模型>` | | loader dtype、只取 checkpoint config、避免整仓权重下载 | `loader-contract`:`EXEC-2a` | `vllm_omni/model_executor/model_loader/weight_utils.py::download_weights_from_hf_specific` → `vllm_omni/model_executor/models/<命中模型>` loader | +| payload connector、KV-only sender、`custom_process_next_stage_input_func` | `payload-connector-ownership`:`EXEC-5a` | `vllm_omni/worker/omni_connector_model_runner_mixin.py::{_should_create_payload_connector,initialize_omni_connector}` → payload I/O callers | +| async output、output builder、step snapshot、connector signal drain | `async-output-materialization`:`EXEC-5b` | `vllm_omni/worker/gpu_ar_model_runner.py::{_should_use_async_omni_output,execute_model,_build_omni_output}` | | 审查组 | 什么时候触发 | 规则 ID | |---|---|---| @@ -28,6 +30,8 @@ sources: [vllm_omni/worker/gpu_model_runner.py, vllm_omni/worker/gpu_ar_model_ru | `strict-stage-config` | stage schema、projection、known fields | `EXEC-3a` | | `bridge-batch` | runtime info、跨 stage payload、batch | `EXEC-1a`, `EXEC-1b` | | `loader-contract` | dtype、checkpoint config 获取、loader | `EXEC-2a` | +| `payload-connector-ownership` | payload connector、KV-only sender、next-stage hook | `EXEC-5a` | +| `async-output-materialization` | async output、snapshot、connector signal drain | `EXEC-5b` | | `author-routing` | 只供 Direct reviewer 导航,不作为 finding 规则 | `EXEC-0a`, `EXEC-0b` | ## 严格配置校验 @@ -77,6 +81,33 @@ sources: [vllm_omni/worker/gpu_model_runner.py, vllm_omni/worker/gpu_ar_model_ru - 验收:覆盖 blank override、多子目录 checkpoint、hook 调用顺序和 mixed-batch request-local metadata;初始化失败必须在 scheduler/worker 继续运行前暴露。 +### EXEC-5a — payload connector 的 ownership 与 KV transfer 角色分开判定 + +- 触发:runner 初始化 payload connector,修改 sender/receiver 角色、KV transfer 配置, + 或设置 `custom_process_next_stage_input_func`。 +- 强制:receiver 始终拥有 payload connector;sender 只有在配置了非空的 + `custom_process_next_stage_input_func` 时才拥有它。connector 的构造、启动、发送、接收 + 和清理必须复用同一 ownership 判定,缺失 connector 的路径显式保持 no-op。 +- 禁止:因为 stage 是 KV sender 就顺带构造 payload connector;让 KV-only sender 启动 + payload I/O;只在初始化处加条件、下游调用仍无条件解引用 connector。 +- 验收:参数化覆盖 receiver/无 hook、sender/无 hook、sender/有 hook 三种组合,分别断言 + ownership 和 payload I/O 调用次数;KV-only sender 不创建 connector,receiver 和带 hook + 的 sender 保持原行为。 ^[PR #5744] + +### EXEC-5b — deferred Omni output 必须先冻结 step 状态并唯一消费 connector 信号 + +- 触发:AR runner 把 Omni output materialization 推迟到后台 builder,或修改 async-output + feature gate、输出累积与 connector signal 的交接时序。 +- 强制:提交后台任务前复制本 step 拥有的 scheduler output、request IDs/映射、token 和 + logprob 等可变容器;builder 是该 cycle 唯一 drain connector signal 的 owner,并在输出 + 累积完成后消费。只在 async scheduling、`async_chunk` 和模型 opt-in 同时成立,且未启用 + prefix cache、spec decode、routed experts 或不兼容 postprocess 时启用异步路径。 +- 禁止:后台任务读取下一 step 会复用或原地修改的 runner 容器;同步路径和 builder 对同一 + signal 各 drain 一次;把 GPU runner 的 feature gate 泛化成 NPU 或其他未走该调用链的 + 平台支持声明。 +- 验收:提交后原地修改 runner 状态不改变已排队输出;connector signal 在累积后恰好消费 + 一次;逐项翻转每个 guard 都回退到同步路径,NPU 覆盖仍保持同步。 ^[PR #5610] + ## Runner 到模型的预处理合同 - 触发条件:修改或排查 runner `_preprocess` 的逐请求 metadata 生产、phase 判定、normal/batched preprocess 选择、MTP 路由条件,或多模型共用的输入预处理合同。 diff --git a/knowledge/repos/vllm-omni/components/scheduler/_index.md b/knowledge/repos/vllm-omni/components/scheduler/_index.md index ce5d9b8..61a1377 100644 --- a/knowledge/repos/vllm-omni/components/scheduler/_index.md +++ b/knowledge/repos/vllm-omni/components/scheduler/_index.md @@ -1,7 +1,7 @@ --- title: "Scheduler(AR/生成请求调度)" created: 2026-07-16 -updated: 2026-08-05 +updated: 2026-08-06 type: index tags: [vllm-omni, components, scheduler] sources: [vllm_omni/core/sched/omni_ar_scheduler.py, vllm_omni/core/prefix_cache.py, docs/design/module/ar_module.md] @@ -16,8 +16,9 @@ sources: [vllm_omni/core/sched/omni_ar_scheduler.py, vllm_omni/core/prefix_cache `OmniSchedulerMixin`(:40)、`OmniTensorPrefixCache`(prefix_cache.py:33) - 官方设计文档:`docs/design/module/ar_module.md`(继承关系、请求流转图) - 测试入口:`tests/core/` -- 主要职责:AR/生成 stage 的请求调度(继承 vLLM Scheduler)、跨 stage KV transfer 的调度面、 - chunk/full-payload 输入等待状态机、omni tensor prefix cache +- 主要职责:AR/生成 stage 的请求调度(继承 vLLM Scheduler)、两类 scheduler 的共享 + lifecycle contract、跨 stage KV transfer 的调度面、chunk/full-payload 输入等待状态机、 + omni tensor prefix cache ## 什么时候查这里 @@ -35,4 +36,4 @@ sources: [vllm_omni/core/sched/omni_ar_scheduler.py, vllm_omni/core/prefix_cache | 遇到什么 | 查看哪里 | |---|---| | 理解调度器继承链、KV transfer 与 prefix cache 语义 | [architecture](architecture.md) | -| 按 PR 描述直达 prefix cache、token budget、upstream 接口或 side-stream 首批源码 | [rules / Direct 代码快速入口](rules.md#direct-代码快速入口) | +| 按 PR 描述直达 prefix cache、token budget、upstream 接口、shared lifecycle 或 side-stream 首批源码 | [rules / Direct 代码快速入口](rules.md#direct-代码快速入口) | diff --git a/knowledge/repos/vllm-omni/components/scheduler/rules.md b/knowledge/repos/vllm-omni/components/scheduler/rules.md index 81d5f3b..9228d34 100644 --- a/knowledge/repos/vllm-omni/components/scheduler/rules.md +++ b/knowledge/repos/vllm-omni/components/scheduler/rules.md @@ -1,10 +1,10 @@ --- title: "Scheduler 规则" created: 2026-07-16 -updated: 2026-08-05 +updated: 2026-08-06 type: rule tags: [vllm-omni, components, scheduler] -sources: ["vllm-omni-rebase-agent@122a9468:agent/skills/fix-talker-truncated-prefill-prefix-cache-key-cap/SKILL.md", "vllm-omni-rebase-agent@122a9468:agent/skills/gpu-hang-low-max-num-batched-tokens/SKILL.md", vllm_omni/worker/gpu_ar_model_runner.py, vllm_omni/core/prefix_cache.py, vllm_omni/core/sched/omni_ar_scheduler.py, vllm_omni/core/sched/omni_generation_scheduler.py, "PR #4106"] +sources: ["vllm-omni-rebase-agent@122a9468:agent/skills/fix-talker-truncated-prefill-prefix-cache-key-cap/SKILL.md", "vllm-omni-rebase-agent@122a9468:agent/skills/gpu-hang-low-max-num-batched-tokens/SKILL.md", vllm_omni/worker/gpu_ar_model_runner.py, vllm_omni/core/prefix_cache.py, vllm_omni/core/sched/omni_ar_scheduler.py, vllm_omni/core/sched/omni_generation_scheduler.py, vllm_omni/core/sched/omni_scheduler_mixin.py, tests/core/sched/test_omni_scheduler_mixin_shared.py, tests/entrypoints/test_omni_new_request_data.py, "PR #4106", "PR #5461"] --- # Scheduler 规则 @@ -26,6 +26,7 @@ PR 描述先命中下表,再打开对应规则组和首批源码;changed fil | side-stream D2H、pinned host tensor、源 buffer 复用 | SCHED-4a/4b | `worker/gpu_ar_model_runner.py::_copy_tensor_payload_to_cpu`、`_get_or_create_omni_payload_copy_stream`;`core/prefix_cache.py` 的 async copy 路径 | | sampled-token logprobs、spec decode trim、request-local output error | SCHED-5a | `core/sched/omni_ar_scheduler.py::_slice_sampled_logprobs`、`update_from_output` | | stateful async chunk、full-payload input、KV cleanup | SCHED-5b/5c | `core/sched/omni_generation_scheduler.py`、`omni_scheduling_coordinator.py`、`omni_ar_scheduler.py::_free_request` | +| AR/generation shared lifecycle、mixin、output envelope、finish cleanup | SCHED-6a | `core/sched/omni_scheduler_mixin.py::OmniSchedulerMixin` → AR 与 generation scheduler 的 `schedule` / `update_from_output` | 若描述只写模型症状,先从模型 owner 找到 payload producer/consumer;只有实际断点落在调度、 prefix cache 或 copy lifetime 时才把 Scheduler 加为 owner。 @@ -194,6 +195,20 @@ modules=[online_serving, worker_runner],status=active,run_count=38,2026-06 - 验收:覆盖完整 pair、不同进度、missing/split、parent abort 和 companion abort, 断言请求不会挂死、错误归属保持 request-local、队列和 connector state 都释放。 +## SCHED-6a — AR 与 generation 的共享生命周期只能有一个 canonical 实现 + +- 触发:把 AR/generation scheduler 的 I/O、输出包装、finish 或统计逻辑移入 + `OmniSchedulerMixin`,或修改该 mixin 及任一 scheduler 的 shared hook。 +- 强制:两类 scheduler 共用 pending connector/chunk 输入处理、wait queue restore、 + `OmniSchedulerOutput` 包装、失败 KV load 的 terminal output、finished IDs、队列清理、 + stats/events 和 finish cleanup;`NewRequestData` 转成 Omni 结构时逐字段无损,并保留已经 + 是 Omni 类型的 fast path。AR 的 synthetic-abort 等差异策略必须作为显式参数留在调用点。 +- 禁止:把共享生命周期复制回两个 subclass;重建 `NewRequestData` 时只挑当前已知字段; + 在 mixin 内按 scheduler 类型隐式猜差异策略;失败 KV load 只改状态而不发 terminal output。 +- 验收:同一组 contract 测试分别驱动 AR 与 generation 输入路径;断言 base dataclass 的 + 每个字段均被转交、现有 Omni entry 保持 identity、失败 KV load 终止输出、finished ID/ + stats/events/cleanup 一致,并单独验证显式 abort policy。 ^[PR #5461] + ## 相关 - 机制与边界见 [architecture](architecture.md);跨 stage 数据面见 diff --git a/knowledge/repos/vllm-omni/docs/_index.md b/knowledge/repos/vllm-omni/docs/_index.md index 1dc85a5..837af91 100644 --- a/knowledge/repos/vllm-omni/docs/_index.md +++ b/knowledge/repos/vllm-omni/docs/_index.md @@ -1,7 +1,7 @@ --- title: "vLLM-Omni 文档" created: 2026-07-10 -updated: 2026-07-16 +updated: 2026-08-06 type: index tags: [vllm-omni, docs] sources: [] @@ -21,5 +21,6 @@ sources: [] | 遇到什么 | 查看哪里 | |---|---| +| 发布版本、compatibility、CUDA/NPU pin 与支持声明同步 | [rules](rules.md) | | 判断 RFC 是否仍在进行 | [RFC status](rfcs/_index.md) | | 找上游官方设计文档与其知识树 owner | [design-doc map](design-doc-map.md) | diff --git a/knowledge/repos/vllm-omni/docs/rules.md b/knowledge/repos/vllm-omni/docs/rules.md new file mode 100644 index 0000000..37845e8 --- /dev/null +++ b/knowledge/repos/vllm-omni/docs/rules.md @@ -0,0 +1,35 @@ +--- +title: "vLLM-Omni 文档规则" +created: 2026-08-06 +updated: 2026-08-06 +type: rule +tags: [vllm-omni, docs] +sources: [README.md, docs/README.md, docs/configuration/README.md, docs/getting_started/installation/README.md, docs/getting_started/installation/gpu/cuda.inc.md, docs/getting_started/installation/npu/npu.inc.md, "PR #5715"] +confidence: high +--- + +# vLLM-Omni 文档规则 + +只有 `DOCS-数字字母` 是可审计规则 ID。 + +## Direct 代码快速入口 + +| PR 描述信号 | 规则组 | 第一批文档 | +|---|---|---| +| release/version、compatibility、CUDA/NPU image、supported models | DOCS-1a | root/docs README → configuration upstream link → installation index 与 GPU/NPU includes | + +## DOCS-1a — release version 与支持声明必须作为一个文档集合更新 + +- 触发:发布 vLLM-Omni 版本,更新 vLLM compatibility、安装命令、CUDA/NPU 镜像或支持 + 模型列表。 +- 强制:同一变更同步核对 root README 的发布公告/节奏、configuration 的 upstream 链接、 + installation compatibility、CUDA package/image pin、NPU branch/image pin 和 supported-model + 声明;每处版本都来自本次发布矩阵,而不是从相邻示例复制。 +- 禁止:只更新首页版本而保留旧安装 pin;CUDA 与 NPU 文档指向不同发布代际;链接文字 + 已更新但目标仍是旧 upstream;用一次发布的具体版本号写成永久规则。 +- 验收:对文档树做 repository-wide 的旧版本/旧分支扫描并逐个归类预期保留项;严格模式 + 构建完整文档,检查内部链接、includes 与导航;从每个公开安装入口回读到同一兼容矩阵。 + ^[PR #5715] + +owner 边界见 [vLLM-Omni 文档索引](_index.md);跨仓库写作方法见 +[通用文档知识](../../../general/docs/_index.md),仓库级规则见 [vLLM-Omni rules](../rules.md)。 diff --git a/knowledge/repos/vllm-omni/models/bagel/_index.md b/knowledge/repos/vllm-omni/models/bagel/_index.md index 52cf19b..fa40399 100644 --- a/knowledge/repos/vllm-omni/models/bagel/_index.md +++ b/knowledge/repos/vllm-omni/models/bagel/_index.md @@ -1,7 +1,7 @@ --- title: "BAGEL(统一模型多形态部署参照)" created: 2026-07-21 -updated: 2026-07-21 +updated: 2026-08-06 type: index tags: [vllm-omni, models, diffusion] sources: [vllm_omni/model_executor/models/bagel/, vllm_omni/diffusion/models/bagel/, vllm_omni/deploy/bagel.yaml] @@ -44,6 +44,7 @@ sources: [vllm_omni/model_executor/models/bagel/, vllm_omni/diffusion/models/bag | 遇到什么 | 查看哪里 | 说明 | |---|---|---| | KV 桥接、3 路 CFG、MoT、变体拓扑 | [architecture](architecture.md) | AR→DiT 数据流与 reviewer 陷阱 | +| CFG position IDs、1-D/2-D multimodal RoPE、Lance 继承回归 | [rules](rules.md) | 可审计规则与验收条件 | ## 配置与 checkpoint 差异 diff --git a/knowledge/repos/vllm-omni/models/bagel/rules.md b/knowledge/repos/vllm-omni/models/bagel/rules.md new file mode 100644 index 0000000..dc118f4 --- /dev/null +++ b/knowledge/repos/vllm-omni/models/bagel/rules.md @@ -0,0 +1,33 @@ +--- +title: "BAGEL 规则" +created: 2026-08-06 +updated: 2026-08-06 +type: rule +tags: [vllm-omni, models, diffusion] +sources: [vllm_omni/diffusion/models/bagel/bagel_transformer.py, "PR #5775"] +confidence: high +--- + +# BAGEL 规则 + +只有 `BAGEL-数字字母` 是可审计规则 ID。 + +## Direct 代码快速入口 + +| PR 描述信号 | 规则组 | 第一批源码 | +|---|---|---| +| CFG branch、position IDs、multimodal RoPE、Lance | BAGEL-1a | `diffusion/models/bagel/bagel_transformer.py` 的 CFG input preparation → [Lance](../lance/_index.md) 继承调用链 | + +## BAGEL-1a — CFG position IDs 按 RoPE rank 选择序列维 + +- 触发:BAGEL/Lance 组装 CFG branch position IDs,或改变 1-D 与 multimodal RoPE 的 + position-id rank。 +- 强制:legacy 1-D position IDs 沿 `dim=0` 拼接;形状为 `(3, S)` 的 2-D multimodal + RoPE position IDs 沿序列维 `dim=1` 拼接,并保留三个坐标行的顺序和值。 +- 禁止:对两种 rank 固定使用同一个 concat dim;通过 flatten/squeeze 把 2-D 输入伪装成 + 1-D;只验证 token 总数而不验证坐标行。 +- 验收:BAGEL 1-D 与 Lance 2-D 各有 shape 和逐值断言;2-D 用不同坐标行和不同 branch + 长度,确保错误的 `dim=0` 会失败。 ^[PR #5775] + +模型拓扑与 CFG 数据流见 [BAGEL architecture](architecture.md);共享 diffusion 合同见 +[Diffusion 规则](../../components/diffusion/rules.md)。 diff --git a/knowledge/repos/vllm-omni/models/minimax-h3/_index.md b/knowledge/repos/vllm-omni/models/minimax-h3/_index.md index 7bdf663..c3471c2 100644 --- a/knowledge/repos/vllm-omni/models/minimax-h3/_index.md +++ b/knowledge/repos/vllm-omni/models/minimax-h3/_index.md @@ -1,7 +1,7 @@ --- title: "MiniMax H3" created: 2026-08-05 -updated: 2026-08-05 +updated: 2026-08-06 type: index tags: [vllm-omni, models, diffusion] sources: [vllm_omni/diffusion/models/minimax_h3/, vllm_omni/diffusion/registry.py, recipes/MiniMaxAI/MiniMax-H3.md, recipes/MiniMaxAI/MiniMax-H3-NPU.md, tests/diffusion/models/minimax_h3/, vllm_omni/entrypoints/openai/video_api_utils.py] @@ -38,3 +38,9 @@ confidence: high 硬件 recipe 只记录已验证的 GPU/NPU 形状;性能数字不能从 recipe 的配置示例泛化为全硬件 保证。共享 offloader、并行和请求合同分别归 [Diffusion](../../components/diffusion/_index.md)、 [Configuration](../../components/configuration/_index.md) 和 [Serving](../../components/serving/_index.md)。 + +## 目录内容 + +| 遇到什么 | 查看哪里 | +|---|---| +| online FP8 scope、ignored layers、QKV/gate-up loader 顺序 | [rules](rules.md) | diff --git a/knowledge/repos/vllm-omni/models/minimax-h3/rules.md b/knowledge/repos/vllm-omni/models/minimax-h3/rules.md new file mode 100644 index 0000000..8937c79 --- /dev/null +++ b/knowledge/repos/vllm-omni/models/minimax-h3/rules.md @@ -0,0 +1,38 @@ +--- +title: "MiniMax H3 规则" +created: 2026-08-06 +updated: 2026-08-06 +type: rule +tags: [vllm-omni, models, diffusion] +sources: [vllm_omni/diffusion/models/minimax_h3/minimax_h3_transformer.py, tests/diffusion/models/minimax_h3/test_minimax_h3_quantization.py, tests/diffusion/models/minimax_h3/test_minimax_h3_quantization_quality.py, docs/user_guide/quantization/fp8.md, "PR #5737"] +confidence: high +--- + +# MiniMax H3 规则 + +只有 `H3-数字字母` 是可审计规则 ID。 + +## Direct 代码快速入口 + +| PR 描述信号 | 规则组 | 第一批源码 | +|---|---|---| +| online FP8、quant config、ignored layers、QKV/gate-up loader | H3-1a | `diffusion/models/minimax_h3/minimax_h3_transformer.py::{MiniMaxH3Transformer.__init__,load_weights}` → component quant-config resolution | + +## H3-1a — online FP8 的量化边界与 checkpoint 变换顺序必须一起保持 + +- 触发:MiniMax H3 online FP8、`ignored_layers`、vLLM linear replacement,或 QKV/ + gate-up checkpoint loader 发生变化。 +- 强制:condition projection、token refiner、DiT blocks 和 AdaLN linears 接收 resolved + quant config;video/audio patch projection、timestep projection 与最终 video/audio output + projection 保持全精度。grouped QKV reorder 和 fused gate/up split 必须先完成,再调用参数上 + 当前生效的 vLLM `weight_loader`,以保留 online FP8 wrapper;`ignored_layers` 使用带完整 + runtime prefix 的模块名。 +- 禁止:先绕过 wrapper 用 default loader 写入再做 layout 变换;把 FP32 输入/输出边界纳入 + online FP8;用 checkpoint 短名匹配 runtime `ignored_layers`;把 online FP8 与 layerwise + offload 组合宣称为已支持。 +- 验收:测试 quant-config prefix 与 ignored-layer 命中、QKV reorder、gate/up 分片和实际 + loader 调用顺序;从 component config 入口证明量化配置到达 transformer;BF16 对 FP8 的 + 质量回归与峰值显存分别设门槛,且 layerwise offload 组合明确拒绝。 ^[PR #5737] + +共享量化/loader 边界见 [Diffusion 规则](../../components/diffusion/rules.md);配置解析见 +[Configuration 规则](../../components/configuration/rules.md)。